From 4dc08e599f0c7ff42371af482aec2a4af51cb80d Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 19:57:13 +0000 Subject: [PATCH 01/59] feat: add shared summary kernels and typed physical values Start `asap-physical-operators` with thin summary kernels over `asap_sketchlib`, the kernel capability checks and the typed value model. The boundary is: - `asap_sketchlib` owns sketch algorithms and their state encodings. - Kernels hold one population's in-memory state. They expose `merge`, a typed sketch readout (`estimate(&SketchQuery)`) and memory accounting. Exact states answer a typed `ExactReadout`; empty MIN/MAX read as `None`. - Group-by belongs to physical operators. - Deployments own wire decoding, delta frames, edge sampling and storage statistics. So wire decoding, `SerializableToSink`, `AggregationType`, `aux_stats`, `reset_to_empty` and the keyed/sum/min/max kernels are not carried over from ASAPQuery-backend. The `asap_sketch_codec` crate is not carried over either; it moves to `asap_sketchlib`. Hydra KLL remains as the Hydra shared-grouping kernel. HLL uses sketchlib's classic estimator. Co-Authored-By: Claude Opus 5.5 Rebased onto main: the Cargo.lock conflict with #478 (new asap-planner crate) was resolved by regenerating workspace entries with `cargo update -w --offline`. --- Cargo.lock | 35 +- Cargo.toml | 1 + crates/asap-physical-operators/Cargo.toml | 16 + .../asap-physical-operators/src/capability.rs | 241 ++++++ crates/asap-physical-operators/src/error.rs | 17 + .../src/key_by_label_values.rs | 126 +++ crates/asap-physical-operators/src/lib.rs | 22 + .../src/measurement.rs | 48 ++ .../asap-physical-operators/src/statistic.rs | 67 ++ .../src/summary_kernels/count_min_sketch.rs | 76 ++ .../count_min_sketch_with_heap.rs | 234 +++++ .../src/summary_kernels/count_sketch.rs | 96 +++ .../summary_kernels/count_sketch_with_heap.rs | 143 ++++ .../src/summary_kernels/datasketches_kll.rs | 106 +++ .../src/summary_kernels/dd_sketch.rs | 88 ++ .../src/summary_kernels/exact.rs | 215 +++++ .../src/summary_kernels/factory.rs | 799 ++++++++++++++++++ .../src/summary_kernels/hll_sketch.rs | 79 ++ .../src/summary_kernels/hydra_kll.rs | 55 ++ .../src/summary_kernels/increase.rs | 213 +++++ .../src/summary_kernels/mod.rs | 26 + .../src/summary_kernels/traits.rs | 35 + .../src/summary_kernels/univmon.rs | 109 +++ .../src/summary_kernels/weighted_frequency.rs | 152 ++++ crates/asap-physical-operators/src/values.rs | 381 +++++++++ .../tests/deployment.rs | 90 ++ 26 files changed, 3468 insertions(+), 2 deletions(-) create mode 100644 crates/asap-physical-operators/Cargo.toml create mode 100644 crates/asap-physical-operators/src/capability.rs create mode 100644 crates/asap-physical-operators/src/error.rs create mode 100644 crates/asap-physical-operators/src/key_by_label_values.rs create mode 100644 crates/asap-physical-operators/src/lib.rs create mode 100644 crates/asap-physical-operators/src/measurement.rs create mode 100644 crates/asap-physical-operators/src/statistic.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/count_sketch.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/exact.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/factory.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/increase.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/mod.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/traits.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/univmon.rs create mode 100644 crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs create mode 100644 crates/asap-physical-operators/src/values.rs create mode 100644 crates/asap-physical-operators/tests/deployment.rs diff --git a/Cargo.lock b/Cargo.lock index 319fda30..96a1f63a 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -309,7 +309,7 @@ version = "0.1.0" dependencies = [ "asap-frontend-promql", "asap-types", - "asap_sketchlib", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", "serde", "serde_json", "thiserror 2.0.18", @@ -369,11 +369,24 @@ dependencies = [ "asap-frontend-promql", "asap-frontend-sql", "asap-types", - "asap_sketchlib", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", "serde_json", "tokio", ] +[[package]] +name = "asap-physical-operators" +version = "0.1.0" +dependencies = [ + "asap-aware-mapping", + "asap-frontend-promql", + "asap-types", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369)", + "serde", + "thiserror 2.0.18", + "tracing", +] + [[package]] name = "asap-planner" version = "0.1.0" @@ -400,6 +413,24 @@ dependencies = [ "thiserror 2.0.18", ] +[[package]] +name = "asap_sketchlib" +version = "0.3.0" +source = "git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369#5f03ccbd798ed5fec62bdd839bcb331123cab369" +dependencies = [ + "bincode", + "bytes", + "prost", + "rand 0.9.5", + "rmp-serde", + "serde", + "serde-big-array", + "serde_bytes", + "smallvec", + "twox-hash 2.1.2", + "xxhash-rust", +] + [[package]] name = "asap_sketchlib" version = "0.3.0" diff --git a/Cargo.toml b/Cargo.toml index a1daf49d..a2b019af 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,5 +1,6 @@ [workspace] members = [ + "crates/asap-physical-operators", "crates/types", "crates/sql-function-catalog", "crates/asap-aware-mapping", diff --git a/crates/asap-physical-operators/Cargo.toml b/crates/asap-physical-operators/Cargo.toml new file mode 100644 index 00000000..51dd6a04 --- /dev/null +++ b/crates/asap-physical-operators/Cargo.toml @@ -0,0 +1,16 @@ +[package] +name = "asap-physical-operators" +version = "0.1.0" +edition = "2021" + +[dependencies] +planner-types = { package = "asap-types", path = "../types" } +asap_sketchlib = { git = "https://github.com/ProjectASAP/asap_sketchlib", rev = "5f03ccbd798ed5fec62bdd839bcb331123cab369" } +serde = { version = "1", features = ["derive", "rc"] } +tracing = "0.1" +thiserror = "2" + + +[dev-dependencies] +asap-aware-mapping = { path = "../asap-aware-mapping" } +asap-frontend-promql = { path = "../frontend-promql" } diff --git a/crates/asap-physical-operators/src/capability.rs b/crates/asap-physical-operators/src/capability.rs new file mode 100644 index 00000000..2a1cc095 --- /dev/null +++ b/crates/asap-physical-operators/src/capability.rs @@ -0,0 +1,241 @@ +//! Capability boundaries, checked without constructing accumulator state. +//! +//! `validate_summary_kernel` checks update kernels, including families without a +//! native batch representation. `validate_native_family` and +//! `validate_sketch_readout` / `validate_exact_readout` check native state and readout support. +//! Keyed weighted-frequency readouts are checked by `Operator::keyed_readout`. +//! A successful kernel check alone does not mean a physical DAG will bind. +//! +//! Stored-state encodings belong to deployments. Full plan acceptance is +//! owned by `binding`, which also validates schemas, expressions and inputs. +use crate::Error; +use planner_types::post_asap::{ + ExactKind, ExactParams, GroupingStrategy, SketchAlgorithm, SketchParams, SketchQuery, + SummaryFamilyType, SummaryUpdate, +}; + +/// Check the same contract used by `create_planner_accumulator` before a plan +/// is accepted. Execution timing is deliberately not a kernel property. +pub fn validate_summary_kernel( + family: &SummaryFamilyType, + input: &SummaryUpdate, + grouping: &GroupingStrategy, +) -> Result<(), String> { + if grouping != &GroupingStrategy::PerSubpopulationInstance { + return Err("shared summary grouping has no registered kernel".into()); + } + let keyed = match family { + SummaryFamilyType::ExactAggregate(kind, params) => { + use ExactKind as K; + use ExactParams as P; + if !matches!( + (kind, params), + (K::Sum, P::Sum) + | (K::Count, P::Count) + | (K::Min, P::Min) + | (K::Max, P::Max) + | (K::Rate, P::Rate) + | (K::Increase, P::Increase) + ) { + return Err(format!("unsupported exact kernel {family:?}")); + } + input.item.is_some() + } + SummaryFamilyType::Sketch(kind, layout) => { + if layout != grouping { + return Err("Planner family and operator grouping disagree".into()); + } + use SketchAlgorithm as A; + use SketchParams as P; + match (kind.algorithm(), kind.params()) { + (A::Kll, P::Kll { k }) if (8..=u16::MAX as u32).contains(k) => false, + (A::DDSketch, P::DDSketch { alpha }) + if alpha.is_finite() && *alpha > 0.0 && *alpha < 1.0 => + { + false + } + (A::Hll, P::Hll { precision }) if (4..=18).contains(precision) => false, + (A::Cms, P::Cms { width, depth }) + | (A::CountSketch, P::CountSketch { width, depth }) + if valid_matrix(*width, *depth) => + { + true + } + ( + A::CmsWithHeap, + P::CmsWithHeap { + width, + depth, + heap_size, + }, + ) + | ( + A::CountSketchWithHeap, + P::CountSketchWithHeap { + width, + depth, + heap_size, + }, + ) if valid_matrix(*width, *depth) && *heap_size > 0 => true, + ( + A::UnivMon, + P::UnivMon { + heap_size, + sketch_rows, + sketch_cols, + layers, + }, + ) if *heap_size > 0 + && *sketch_cols > 0 + && (1..=20).contains(sketch_rows) + && (1..=64).contains(layers) + && (*sketch_rows as usize) + .checked_mul(*sketch_cols as usize) + .and_then(|n| n.checked_mul(*layers as usize)) + .is_some() => + { + false + } + _ => { + return Err(format!( + "unsupported kernel or invalid parameters: {kind:?}" + )) + } + } + } + _ => return Err(format!("unsupported summary kernel {family:?}")), + }; + if keyed != input.item.is_some() && !is_unit_sample_frequency(input) { + return Err("Planner item expression does not match kernel layout".into()); + } + Ok(()) +} + +fn valid_matrix(width: u32, depth: u32) -> bool { + // Construction uses the kernel's native row hashing, so no encoded-size + // limit applies here. + width > 0 + && depth > 0 + && (width as usize) + .checked_mul(depth as usize) + .and_then(|n| n.checked_mul(std::mem::size_of::())) + .is_some() +} + +pub(crate) fn is_unit_sample_frequency(update: &planner_types::post_asap::SummaryUpdate) -> bool { + use planner_types::post_asap::{NonNegativeWeightProof, SummaryInputExpr, WeightDomain}; + matches!( + update.item, + Some(SummaryInputExpr::Column( + planner_types::pre_asap::ColumnRef::SampleValue + )) + ) && matches!(update.weight, SummaryInputExpr::Constant(1.0)) + && matches!( + update.weight_domain, + WeightDomain::NonNegative { + proof: NonNegativeWeightProof::UnitCount + } + ) +} + +pub fn validate_native_family(family: &SummaryFamilyType) -> Result<(), Error> { + use planner_types::post_asap::SketchAlgorithm as A; + if let SummaryFamilyType::Sketch(kind, grouping) = family { + if matches!(kind.algorithm(), A::CmsWithHeap | A::CountSketchWithHeap) { + let (_, width, depth, _) = + crate::summary_kernels::weighted_frequency::WeightedFrequency::configuration(kind)?; + return if valid_matrix(width as u32, depth as u32) && grouping == &Default::default() { + Ok(()) + } else { + Err(Error::Invalid( + "invalid weighted frequency dimensions or grouping strategy".into(), + )) + }; + } + } + match family { + SummaryFamilyType::ExactAggregate(..) => {} + SummaryFamilyType::Sketch(kind, _) + if matches!(kind.algorithm(), A::Kll | A::DDSketch | A::Hll) => {} + _ => { + return Err(Error::Invalid( + "summary family has no native DAG state implementation".into(), + )) + } + } + crate::capability::validate_summary_kernel( + family, + &planner_types::post_asap::SummaryUpdate::column( + planner_types::pre_asap::ColumnRef::SampleValue, + ), + &Default::default(), + ) + .map_err(Error::Invalid) +} + +/// A sketch readout is native only for the families Planner can read directly. +pub fn validate_sketch_readout( + family: &SummaryFamilyType, + query: &SketchQuery, +) -> Result<(), Error> { + validate_native_family(family)?; + use planner_types::post_asap::SketchAlgorithm as A; + // A point count without an item value reads the total count. + let bare_count = matches!(query, SketchQuery::PointCount { value: None, .. }); + let supported = match family { + SummaryFamilyType::Sketch(kind, _) => match (kind.algorithm(), query) { + (A::Kll, SketchQuery::Quantile { q }) | (A::DDSketch, SketchQuery::Quantile { q }) => { + if !(0.0..=1.0).contains(q) { + return Err(Error::Invalid( + "quantile readout requires quantile in [0,1]".into(), + )); + } + true + } + (A::DDSketch, _) => bare_count, + (A::Hll, SketchQuery::Cardinality) => true, + (A::Hll, _) => bare_count, + _ => false, + }, + _ => false, + }; + if !supported { + return Err(Error::Invalid( + "readout is not implemented for this summary family".into(), + )); + } + Ok(()) +} + +/// An exact readout must match the exact family it reads. +pub fn validate_exact_readout( + family: &SummaryFamilyType, + readout: &crate::summary_kernels::exact::ExactReadout, +) -> Result<(), Error> { + validate_native_family(family)?; + use crate::Statistic as S; + use planner_types::post_asap::ExactKind as E; + let supported = matches!( + (family, readout.statistic), + (SummaryFamilyType::ExactAggregate(E::Sum, _), S::Sum) + | (SummaryFamilyType::ExactAggregate(E::Count, _), S::Count) + | (SummaryFamilyType::ExactAggregate(E::Min, _), S::Min) + | (SummaryFamilyType::ExactAggregate(E::Max, _), S::Max) + | (SummaryFamilyType::ExactAggregate(E::Rate, _), S::Rate) + | ( + SummaryFamilyType::ExactAggregate(E::Increase, _), + S::Increase + ) + ); + if !supported { + return Err(Error::Invalid( + "readout is not implemented for this summary family".into(), + )); + } + if readout.lookback_ms.is_some_and(|lookback| { + lookback <= 0 || !matches!(readout.statistic, S::Rate | S::Increase) + }) { + return Err(Error::Invalid("invalid exact counter lookback".into())); + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/error.rs b/crates/asap-physical-operators/src/error.rs new file mode 100644 index 00000000..06ce16e9 --- /dev/null +++ b/crates/asap-physical-operators/src/error.rs @@ -0,0 +1,17 @@ +#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] +pub enum Error { + #[error("invalid DAG: {0}")] + Invalid(String), + #[error("operator failed: {0}")] + Operator(String), + #[error("node {node} ({operation}) failed: {source}")] + AtNode { + node: u64, + operation: String, + source: Box, + }, + #[error("execution memory limit exceeded")] + MemoryLimit, + #[error("execution cancelled")] + Cancelled, +} diff --git a/crates/asap-physical-operators/src/key_by_label_values.rs b/crates/asap-physical-operators/src/key_by_label_values.rs new file mode 100644 index 00000000..e574da23 --- /dev/null +++ b/crates/asap-physical-operators/src/key_by_label_values.rs @@ -0,0 +1,126 @@ +use serde::{Deserialize, Serialize}; +// use std::collections::HashMap; +use std::hash::{Hash, Hasher}; + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct KeyByLabelValues { + // pub labels: HashMap, + pub labels: Vec, +} + +impl KeyByLabelValues { + pub fn new() -> Self { + Self { labels: Vec::new() } + } + + pub fn new_with_labels(labels: Vec) -> Self { + Self { labels } + } + + pub fn insert(&mut self, value: String) { + self.labels.push(value); + } + + pub fn get(&self, index: usize) -> Option<&String> { + self.labels.get(index) + } + + /// Encode labels as a semicolon-joined string — the canonical key format used + /// for sketch item hashing (CountMinSketch, CountSketch, HydraKLL). + pub fn to_semicolon_str(&self) -> String { + self.labels.join(";") + } + + #[cfg(test)] + /// Decode a semicolon-joined string back into a KeyByLabelValues. + pub fn from_semicolon_str(s: &str) -> Self { + Self { + labels: s.split(';').map(|s| s.to_string()).collect(), + } + } + + pub fn is_empty(&self) -> bool { + self.labels.is_empty() + } + + pub fn len(&self) -> usize { + self.labels.len() + } +} + +impl Hash for KeyByLabelValues { + fn hash(&self, state: &mut H) { + // Create a sorted vector of key-value pairs for consistent hashing + let mut sorted_pairs: Vec<_> = self.labels.iter().collect(); + sorted_pairs.sort(); + + for value in sorted_pairs { + value.hash(state); + } + } +} + +impl Default for KeyByLabelValues { + fn default() -> Self { + Self::new() + } +} + +impl std::fmt::Display for KeyByLabelValues { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{{")?; + let mut first = true; + for value in &self.labels { + if !first { + write!(f, ", ")?; + } + write!(f, "{value}")?; + first = false; + } + write!(f, "}}") + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_key_by_label_values() { + let mut key = KeyByLabelValues::new(); + key.insert("localhost:8080".to_string()); + key.insert("prometheus".to_string()); + + assert_eq!(key.len(), 2); + assert_eq!(key.get(0), Some(&"localhost:8080".to_string())); + assert_eq!(key.get(1), Some(&"prometheus".to_string())); + } + + #[test] + fn test_semicolon_roundtrip() { + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string(), "prod".to_string()]); + assert_eq!(key.to_semicolon_str(), "web;prod"); + let roundtripped = KeyByLabelValues::from_semicolon_str("web;prod"); + assert_eq!(roundtripped, key); + } + + #[test] + fn test_hash_consistency() { + let mut key1 = KeyByLabelValues::new(); + key1.insert("a".to_string()); + key1.insert("b".to_string()); + + let mut key2 = KeyByLabelValues::new(); + key2.insert("b".to_string()); + key2.insert("a".to_string()); + + // Should hash to the same value regardless of insertion order + let mut hasher1 = std::collections::hash_map::DefaultHasher::new(); + let mut hasher2 = std::collections::hash_map::DefaultHasher::new(); + + key1.hash(&mut hasher1); + key2.hash(&mut hasher2); + + assert_eq!(hasher1.finish(), hasher2.finish()); + } +} diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs new file mode 100644 index 00000000..5026cb6d --- /dev/null +++ b/crates/asap-physical-operators/src/lib.rs @@ -0,0 +1,22 @@ +//! Shared physical operators. This layer holds summary kernels and typed values. + +pub mod key_by_label_values; +pub mod measurement; +pub mod summary_kernels; +pub use summary_kernels::traits; + +mod statistic; +pub use key_by_label_values::KeyByLabelValues; +pub use measurement::Measurement; +pub use statistic::Statistic; +pub use traits::*; + +pub mod capability; +pub use summary_kernels::factory; + +/// The exact Planner contract used by these kernels. +pub use planner_types as planner; + +mod error; +pub use error::Error; +pub mod values; diff --git a/crates/asap-physical-operators/src/measurement.rs b/crates/asap-physical-operators/src/measurement.rs new file mode 100644 index 00000000..57234f01 --- /dev/null +++ b/crates/asap-physical-operators/src/measurement.rs @@ -0,0 +1,48 @@ +use serde::{Deserialize, Serialize}; +use std::ops::Add; + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct Measurement { + pub value: f64, +} + +impl Measurement { + pub fn new(value: f64) -> Self { + Self { value } + } +} + +impl Add for Measurement { + type Output = Measurement; + + fn add(self, other: Measurement) -> Measurement { + Measurement::new(self.value + other.value) + } +} + +impl Add for &Measurement { + type Output = Measurement; + + fn add(self, other: &Measurement) -> Measurement { + Measurement::new(self.value + other.value) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_measurement_creation() { + let measurement = Measurement::new(42.5); + assert_eq!(measurement.value, 42.5); + } + + #[test] + fn test_measurement_addition() { + let m1 = Measurement::new(10.0); + let m2 = Measurement::new(20.0); + let result = m1 + m2; + assert_eq!(result.value, 30.0); + } +} diff --git a/crates/asap-physical-operators/src/statistic.rs b/crates/asap-physical-operators/src/statistic.rs new file mode 100644 index 00000000..7053308d --- /dev/null +++ b/crates/asap-physical-operators/src/statistic.rs @@ -0,0 +1,67 @@ +use std::{fmt, str::FromStr}; +use tracing::debug; +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] +pub enum Statistic { + Count, + Sum, + Cardinality, + FrequencyL2, + FrequencyEntropy, + Increase, + Rate, + Min, + Max, + Quantile, + Topk, +} + +impl fmt::Display for Statistic { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + debug!("Formatting Statistic: {:?}", self); + match self { + Statistic::Count => write!(f, "count"), + Statistic::Sum => write!(f, "sum"), + Statistic::Cardinality => write!(f, "cardinality"), + Statistic::FrequencyL2 => write!(f, "frequency_l2"), + Statistic::FrequencyEntropy => write!(f, "frequency_entropy"), + Statistic::Increase => write!(f, "increase"), + Statistic::Rate => write!(f, "rate"), + Statistic::Min => write!(f, "min"), + Statistic::Max => write!(f, "max"), + Statistic::Quantile => write!(f, "quantile"), + Statistic::Topk => write!(f, "topk"), + } + } +} + +#[allow(clippy::should_implement_trait)] +impl Statistic { + pub fn from_str(s: &str) -> Option { + debug!("Parsing Statistic from string: {}", s); + match s.to_lowercase().as_str() { + "count" => Some(Statistic::Count), + "sum" => Some(Statistic::Sum), + "cardinality" => Some(Statistic::Cardinality), + "frequency_l2" => Some(Statistic::FrequencyL2), + "frequency_entropy" => Some(Statistic::FrequencyEntropy), + "increase" => Some(Statistic::Increase), + "rate" => Some(Statistic::Rate), + "min" => Some(Statistic::Min), + "max" => Some(Statistic::Max), + "quantile" => Some(Statistic::Quantile), + "topk" => Some(Statistic::Topk), + _ => None, + } + } +} + +impl FromStr for Statistic { + type Err = (); + + /// Parse a statistic from a string (case-insensitive). + /// Use `s.parse::()` or `Statistic::from_str(s)`. + fn from_str(s: &str) -> Result { + debug!("FromStr trait parsing Statistic: {}", s); + Statistic::from_str(s).ok_or(()) + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs new file mode 100644 index 00000000..5a78fc48 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs @@ -0,0 +1,76 @@ +//! Count-Min Sketch frequency summary over `asap_sketchlib::CountMinSketch`. +use crate::{AggregateCore, KernelError, KeyByLabelValues}; +use asap_sketchlib::CountMinSketch; + +#[derive(Debug, Clone)] +pub struct CountMinSketchAccumulator { + pub inner: CountMinSketch, +} + +impl CountMinSketchAccumulator { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + inner: CountMinSketch::new(row_num, col_num), + } + } + + /// Estimated frequency of one item. + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + self.inner.estimate(&key.to_semicolon_str()) + } +} + +impl AggregateCore for CountMinSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("Count-Min Sketch merges only with Count-Min Sketch")?; + Ok(Box::new(Self { + inner: CountMinSketch::merge_refs(&[&self.inner, &other.inner])?, + })) + } + + fn approx_memory_bytes(&self) -> usize { + 16 * 1024 + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Merged point counts add item frequencies and never underestimate. + #[test] + fn merged_point_counts_add() { + let (mut a, mut b) = ( + CountMinSketchAccumulator::new(3, 128), + CountMinSketchAccumulator::new(3, 128), + ); + let key = KeyByLabelValues::new_with_labels(vec!["checkout".into()]); + a.inner.update(&key.to_semicolon_str(), 2.0); + b.inner.update(&key.to_semicolon_str(), 3.0); + let merged = a.merge_with(&b).unwrap(); + let merged = merged + .as_any() + .downcast_ref::() + .unwrap(); + assert!(merged.query_key(&key) >= 5.0); + } + + // Merge rejects a different summary family. + #[test] + fn rejects_foreign_merge() { + let cms = CountMinSketchAccumulator::new(3, 128); + let kll = crate::summary_kernels::DatasketchesKLLAccumulator::new(200); + assert!(cms.merge_with(&kll).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs new file mode 100644 index 00000000..da07aeee --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs @@ -0,0 +1,234 @@ +use crate::{AggregateCore, KeyByLabelValues}; +use asap_sketchlib::CountMinSketchWithHeap; + +/// Count-Min Sketch with a top-k heap over `asap_sketchlib::CountMinSketchWithHeap`. +#[derive(Debug, Clone)] +pub struct CountMinSketchWithHeapAccumulator { + pub inner: CountMinSketchWithHeap, +} + +impl CountMinSketchWithHeapAccumulator { + pub fn new(row_num: usize, col_num: usize, heap_size: usize) -> Self { + Self { + inner: CountMinSketchWithHeap::new(row_num, col_num, heap_size), + } + } + + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + let key_string = key.labels.join(";"); + self.inner.estimate(&key_string) + } + + /// VALUE-WEIGHTED heavy-hitter update (FIX: CountSketch/CMS topk + /// recall-0). The default ingest path inserts `+1` per occurrence keyed + /// by the raw `item`, so the heap ranks groups by OCCURRENCE COUNT — the + /// wrong answer for `topk(k, sum by (label) (metric))`, which asks for + /// the top groups by SUM OF VALUE. This update adds the sample `value` + /// (not `+1`) into both the CMS matrix and the top-k heap, keyed by the + /// GROUP LABEL (e.g. the `host` / `zone` value), so the heap's ranking is + /// by summed value. Repeated calls for the same `group_label` accumulate, + /// so after folding a window the heap holds Σvalue per group. + /// + /// Delegates to the library's value-weighted `CountMinSketchWithHeap:: + /// update(key, value)` (`sketchlib_cms_heap_update` → `insert_many(key, + /// round(value))`), which is the "separate update path" the evaluation + /// plan (Fig 3c) called for. + pub fn insert_value(&mut self, group_label: &str, value: f64) { + self.inner.update(group_label, value); + } + + /// Read the top-`k` GROUPS ranked by summed VALUE (descending), keyed by + /// the group label. Pairs with [`Self::insert_value`]: the heap built by + /// value-weighted updates ranks by Σvalue, so this returns the + /// value-weighted top-k (not the occurrence-count top-k the raw `item` + /// heap would give). Sorted descending by value; ties broken by key for + /// determinism; truncated to `k`. + pub fn topk_by_value(&self, k: usize) -> Vec<(String, f64)> { + let mut items: Vec<(String, f64)> = self + .inner + .topk_heap_items() + .into_iter() + .map(|it| (it.key, it.value)) + .collect(); + items.sort_by(|a, b| { + b.1.partial_cmp(&a.1) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.0.cmp(&b.0)) + }); + items.truncate(k); + items + } + + /// Get all keys from the top-k heap. + pub fn get_topk_keys(&self) -> Vec { + self.inner + .topk_heap_items() + .iter() + .map(|item| { + let labels: Vec = item.key.split(';').map(|s| s.to_string()).collect(); + KeyByLabelValues { labels } + }) + .collect() + } +} + +impl AggregateCore for CountMinSketchWithHeapAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other_cms = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountMinSketchWithHeapAccumulator")?; + + let mut merged = self.clone(); + merged.inner.merge(&other_cms.inner)?; + Ok(Box::new(merged)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_count_min_sketch_with_heap_creation() { + let cms = CountMinSketchWithHeapAccumulator::new(4, 1000, 20); + assert_eq!(cms.inner.rows(), 4); + assert_eq!(cms.inner.cols(), 1000); + assert_eq!(cms.inner.heap_size, 20); + assert_eq!(cms.inner.topk_heap_items().len(), 0); + } + + #[test] + fn test_get_topk_keys() { + let mut cms = CountMinSketchWithHeapAccumulator::new(2, 3, 5); + cms.inner.update("label1;label2", 100.0); + cms.inner.update("label3;label4", 50.0); + + let keys = cms.get_topk_keys(); + assert_eq!(keys.len(), 2); + // Heap order is not part of the contract; compare as a set. + let label_sets: std::collections::HashSet<_> = + keys.iter().map(|k| k.labels.clone()).collect(); + assert!(label_sets.contains(&vec!["label1".to_string(), "label2".to_string()])); + assert!(label_sets.contains(&vec!["label3".to_string(), "label4".to_string()])); + } + + // ---------------------------------------------------------------- + // FIX 1 — VALUE-WEIGHTED top-k (recall 0 → correct). + // + // `topk(k, sum by (host) (cpu_load))` asks for the top-k hosts by + // SUM OF VALUE. The heavy-hitter heap built by the default `+1`-per- + // occurrence update ranks by COUNT keyed by `item`, so its recall + // against the value-weighted ground truth is 0 when the busiest host + // (most samples) is NOT the heaviest host (largest Σvalue). + // `insert_value(group_label, value)` adds the sample VALUE keyed by the + // GROUP LABEL, so `topk_by_value` ranks by Σvalue — correct recall. + // ---------------------------------------------------------------- + + /// Crafted adversarial dataset: the host with the MOST samples + /// (`h_chatty`, 100 tiny samples) is NOT the host with the largest + /// value-sum (`h_heavy`, a handful of huge samples). A COUNT-ranked + /// heap would surface `h_chatty`; the value-weighted top-k must surface + /// the true heavy hitters by Σvalue, giving recall 1.0 against the + /// ground-truth top-k-by-value-sum. + #[test] + fn value_weighted_topk_has_full_recall_vs_count_topk() { + // (host, per-sample value, sample count) → true Σvalue: + // h_heavy : 1000 × 3 = 3000 (few samples, huge value) + // h_mid : 200 × 5 = 1000 + // h_small : 50 × 6 = 300 + // h_chatty: 1 × 100 = 100 (MOST samples, tiny value) + let data: &[(&str, f64, usize)] = &[ + ("h_heavy", 1000.0, 3), + ("h_mid", 200.0, 5), + ("h_small", 50.0, 6), + ("h_chatty", 1.0, 100), + ]; + + // Wide CMS + heap large enough to hold every group exactly (4 groups) + // so the estimate equals the true Σvalue with no hash collisions. + let mut acc = CountMinSketchWithHeapAccumulator::new(5, 4096, 16); + let mut truth: std::collections::HashMap<&str, f64> = std::collections::HashMap::new(); + for (host, value, count) in data { + for _ in 0..*count { + acc.insert_value(host, *value); + } + *truth.entry(*host).or_insert(0.0) += value * (*count as f64); + } + + // Ground-truth top-2 by value-sum: h_heavy (3000), h_mid (1000). + let mut truth_ranked: Vec<(&str, f64)> = truth.into_iter().collect(); + truth_ranked.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap()); + let truth_top2: std::collections::HashSet<&str> = + truth_ranked.iter().take(2).map(|(k, _)| *k).collect(); + assert!( + truth_top2.contains("h_heavy") && truth_top2.contains("h_mid"), + "ground-truth top-2 by value-sum should be h_heavy + h_mid" + ); + + // Value-weighted top-2 from the heap. + let got = acc.topk_by_value(2); + assert_eq!(got.len(), 2, "k=2 → two groups: {got:?}"); + let got_keys: std::collections::HashSet<&str> = + got.iter().map(|(k, _)| k.as_str()).collect(); + + // RECALL = |got ∩ truth| / |truth| must be 1.0. + let hits = got_keys.intersection(&truth_top2).count(); + let recall = hits as f64 / truth_top2.len() as f64; + assert_eq!( + recall, 1.0, + "value-weighted top-k recall must be 1.0 (count-ranked heap would \ + surface h_chatty and miss h_heavy → recall < 1): got={got:?}" + ); + + // The busiest-by-count host (h_chatty) must NOT be in the top-2, + // proving we rank by value-sum, not occurrence count. + assert!( + !got_keys.contains("h_chatty"), + "h_chatty (most samples, smallest value-sum) must be excluded: {got:?}" + ); + + // Estimates are exact here (no collisions, heap holds all groups): + // top-1 must be h_heavy with Σvalue 3000. + assert_eq!(got[0].0, "h_heavy"); + assert!( + (got[0].1 - 3000.0).abs() < 1e-6, + "h_heavy value-sum estimate ≈ 3000, got {}", + got[0].1 + ); + assert_eq!(got[1].0, "h_mid"); + assert!( + (got[1].1 - 1000.0).abs() < 1e-6, + "h_mid value-sum estimate ≈ 1000, got {}", + got[1].1 + ); + } + + /// A single value-weighted insert must put the full value (not +1) into + /// the heap, and repeated inserts for the same group must accumulate. + #[test] + fn insert_value_accumulates_summed_value_in_heap() { + let mut acc = CountMinSketchWithHeapAccumulator::new(4, 1024, 8); + acc.insert_value("g", 10.0); + acc.insert_value("g", 25.0); + let top = acc.topk_by_value(1); + assert_eq!(top.len(), 1); + assert_eq!(top[0].0, "g"); + assert!( + (top[0].1 - 35.0).abs() < 1e-6, + "summed value should be 35 (10+25), got {}", + top[0].1 + ); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs new file mode 100644 index 00000000..2728b5f2 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs @@ -0,0 +1,96 @@ +//! CountSketch accumulator backed by `asap_sketchlib::CountSketch`. +//! +//! Per-key queries delegate to sketchlib's median-of-signed-rows estimator. +//! Top-k requires the separate heap-bearing accumulator. + +use crate::{AggregateCore, KeyByLabelValues}; +use asap_sketchlib::CountSketch; + +/// Count Sketch accumulator — inner matrix of signed counts. +#[derive(Debug, Clone)] +pub struct CountSketchAccumulator { + pub inner: CountSketch, +} + +impl CountSketchAccumulator { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + inner: CountSketch::new(row_num, col_num), + } + } + + /// Median-of-signed-rows point estimate for `key`, via + /// `asap_sketchlib::CountSketch::estimate`. + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + self.inner.estimate(&key.to_semicolon_str()) + } +} + +impl AggregateCore for CountSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other_cs = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountSketchAccumulator")?; + + let merged_inner = CountSketch::merge_refs(&[&self.inner, &other_cs.inner])?; + Ok(Box::new(Self { + inner: merged_inner, + })) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_query_key_uses_real_sketchlib_estimator() { + // `query_key` must match sketchlib's estimator and hash specification. + let mut cs = CountSketchAccumulator::new(4, 1000); + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + cs.inner.update(&key.to_semicolon_str(), 10.0); + assert_eq!( + cs.query_key(&key), + cs.inner.estimate(&key.to_semicolon_str()) + ); + } + + #[test] + fn test_aggregate_core_merge_matches_matrix_add() { + let a = CountSketchAccumulator { + inner: CountSketch::from_legacy_matrix(vec![vec![1.0, -2.0], vec![3.0, -4.0]], 2, 2), + }; + let b = CountSketchAccumulator { + inner: CountSketch::from_legacy_matrix(vec![vec![-1.0, 2.0], vec![-3.0, 4.0]], 2, 2), + }; + let merged_box = a.merge_with(&b).expect("merge ok"); + let merged = merged_box + .as_any() + .downcast_ref::() + .expect("downcast ok"); + let m = merged.inner.sketch(); + assert_eq!(m[0], vec![0.0, 0.0]); + assert_eq!(m[1], vec![0.0, 0.0]); + } + + #[test] + fn test_aggregate_core_merge_wrong_type_rejects() { + use crate::summary_kernels::count_min_sketch::CountMinSketchAccumulator; + let cs = CountSketchAccumulator::new(2, 3); + let cms = CountMinSketchAccumulator::new(2, 3); + let result = cs.merge_with(&cms); + assert!(result.is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs b/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs new file mode 100644 index 00000000..a735e4bd --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs @@ -0,0 +1,143 @@ +//! CountSketch with a top-k heap over `asap_sketchlib::CountSketchWithHeap` +//! (median-of-signed-rows), distinct from the Count-Min heap variant. + +use crate::{AggregateCore, KeyByLabelValues}; +use asap_sketchlib::CountSketchWithHeap; + +#[derive(Debug, Clone)] +pub struct CountSketchWithHeapAccumulator { + pub inner: CountSketchWithHeap, +} + +impl CountSketchWithHeapAccumulator { + pub fn new(row_num: usize, col_num: usize, heap_size: usize) -> Self { + Self { + inner: CountSketchWithHeap::new(row_num, col_num, heap_size), + } + } + + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + let key_string = key.labels.join(";"); + self.inner.estimate(&key_string) + } + + /// Value-weighted heavy-hitter update -- see + /// `CountMinSketchWithHeapAccumulator::insert_value`'s doc for why + /// this (not a `+1`-per-occurrence update) is the correct semantics + /// for `topk(k, sum by (label) (metric))`-shaped queries. + pub fn insert_value(&mut self, group_label: &str, value: f64) { + self.inner.update(group_label, value); + } + + /// Read the top-`k` groups ranked by summed value (descending, tie-broken + /// by key for determinism). Mirrors `CountMinSketchWithHeapAccumulator::topk_by_value`. + pub fn topk_by_value(&self, k: usize) -> Vec<(String, f64)> { + let mut items: Vec<(String, f64)> = self + .inner + .topk_heap_items() + .into_iter() + .map(|it| (it.key, it.value)) + .collect(); + items.sort_by(|a, b| { + b.1.partial_cmp(&a.1) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.0.cmp(&b.0)) + }); + items.truncate(k); + items + } + + /// Get all keys from the top-k heap. + pub fn get_topk_keys(&self) -> Vec { + self.inner + .topk_heap_items() + .iter() + .map(|item| { + let labels: Vec = item.key.split(';').map(|s| s.to_string()).collect(); + KeyByLabelValues { labels } + }) + .collect() + } +} + +impl AggregateCore for CountSketchWithHeapAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other_cs = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountSketchWithHeapAccumulator")?; + + let mut merged = self.clone(); + merged.inner.merge(&other_cs.inner)?; + Ok(Box::new(merged)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_count_sketch_with_heap_creation() { + let cs = CountSketchWithHeapAccumulator::new(4, 1000, 20); + assert_eq!(cs.inner.rows(), 4); + assert_eq!(cs.inner.cols(), 1000); + assert_eq!(cs.inner.heap_size, 20); + assert_eq!(cs.inner.topk_heap_items().len(), 0); + } + + #[test] + fn test_get_topk_keys() { + let mut cs = CountSketchWithHeapAccumulator::new(2, 3, 5); + cs.inner.update("label1;label2", 100.0); + cs.inner.update("label3;label4", 50.0); + + let keys = cs.get_topk_keys(); + assert_eq!(keys.len(), 2); + let label_sets: std::collections::HashSet<_> = + keys.iter().map(|k| k.labels.clone()).collect(); + assert!(label_sets.contains(&vec!["label1".to_string(), "label2".to_string()])); + assert!(label_sets.contains(&vec!["label3".to_string(), "label4".to_string()])); + } + + #[test] + fn insert_value_accumulates_summed_value_in_heap() { + let mut acc = CountSketchWithHeapAccumulator::new(4, 1024, 8); + acc.insert_value("g", 10.0); + acc.insert_value("g", 25.0); + let top = acc.topk_by_value(1); + assert_eq!(top.len(), 1); + assert_eq!(top[0].0, "g"); + assert!( + (top[0].1 - 35.0).abs() < 1e-6, + "summed value should be 35 (10+25), got {}", + top[0].1 + ); + } + + /// CountSketch and Count-Min heap states are distinct families and never merge. + #[test] + fn test_rejects_merge_with_cms_family_accumulator() { + use crate::summary_kernels::count_min_sketch_with_heap::CountMinSketchWithHeapAccumulator; + + let cs = CountSketchWithHeapAccumulator::new(4, 64, 10); + let cms = CountMinSketchWithHeapAccumulator::new(4, 64, 10); + let result = cs.merge_with(&cms); + assert!( + result.is_err(), + "CountSketchWithHeapAccumulator must not merge with CountMinSketchWithHeapAccumulator \ + -- different algorithms sharing only a storage shape" + ); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs b/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs new file mode 100644 index 00000000..0f740a9d --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs @@ -0,0 +1,106 @@ +//! KLL quantile summary over `asap_sketchlib::KllSketch`. +use crate::{AggregateCore, KernelError}; +use asap_sketchlib::KllSketch; +use planner_types::post_asap::SketchQuery; + +#[derive(Clone)] +pub struct DatasketchesKLLAccumulator { + pub inner: KllSketch, +} + +impl DatasketchesKLLAccumulator { + pub fn new(k: u16) -> Self { + Self { + inner: KllSketch::new(k), + } + } + + pub fn update(&mut self, value: f64) { + self.inner.update(value); + } + + pub fn get_quantile(&self, quantile: f64) -> f64 { + self.inner.quantile(quantile) + } +} + +impl std::fmt::Debug for DatasketchesKLLAccumulator { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("DatasketchesKLLAccumulator") + .field("k", &self.inner.k) + .field("sketch_n", &self.inner.count()) + .finish() + } +} + +// SAFETY: `KllSketch` owns its buffers and has no interior mutability; the +// accumulator is only mutated through `&mut self`. +unsafe impl Send for DatasketchesKLLAccumulator {} +unsafe impl Sync for DatasketchesKLLAccumulator {} + +impl AggregateCore for DatasketchesKLLAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("KLL merges only with KLL")?; + Ok(Box::new(Self { + inner: KllSketch::merge_refs(&[&self.inner, &other.inner])?, + })) + } + + fn estimate(&self, query: &SketchQuery) -> Result { + match query { + SketchQuery::Quantile { q } if (0.0..=1.0).contains(q) => Ok(self.get_quantile(*q)), + SketchQuery::Quantile { .. } => Err("quantile must be in [0, 1]".into()), + other => Err(format!("KLL does not answer {other:?}").into()), + } + } + + fn approx_memory_bytes(&self) -> usize { + // KLL with default k=200 holds ~2*k items (~3 KiB); round up for overhead. + 4 * 1024 + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Merging two KLL states reads like one state built over both inputs. + #[test] + fn merged_quantile_matches_single_build() { + let (mut a, mut b, mut all) = ( + DatasketchesKLLAccumulator::new(200), + DatasketchesKLLAccumulator::new(200), + DatasketchesKLLAccumulator::new(200), + ); + for v in 0..100 { + a.update(f64::from(v)); + all.update(f64::from(v)); + } + for v in 100..200 { + b.update(f64::from(v)); + all.update(f64::from(v)); + } + let merged = a.merge_with(&b).unwrap(); + let q = SketchQuery::Quantile { q: 0.5 }; + assert_eq!(merged.estimate(&q).unwrap(), all.estimate(&q).unwrap()); + } + + // KLL answers only quantiles in [0, 1]. + #[test] + fn rejects_unsupported_or_out_of_range_queries() { + let kll = DatasketchesKLLAccumulator::new(200); + assert!(kll.estimate(&SketchQuery::Quantile { q: 1.5 }).is_err()); + assert!(kll.estimate(&SketchQuery::Cardinality).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs new file mode 100644 index 00000000..1d6be98c --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs @@ -0,0 +1,88 @@ +//! DDSketch quantile summary over `asap_sketchlib::DdSketch`. +use crate::{AggregateCore, KernelError}; +use asap_sketchlib::DdSketch; +use planner_types::post_asap::SketchQuery; + +#[derive(Debug, Clone)] +pub struct DDSketchAccumulator { + pub inner: DdSketch, +} + +impl DDSketchAccumulator { + pub fn new(alpha: f64) -> Self { + Self { + inner: DdSketch::new(alpha), + } + } +} + +impl AggregateCore for DDSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("DDSketch merges only with DDSketch")?; + Ok(Box::new(Self { + inner: DdSketch::merge_refs(&[&self.inner, &other.inner])?, + })) + } + + /// Quantiles, and the total sample count as a bare `PointCount`. + fn estimate(&self, query: &SketchQuery) -> Result { + match query { + SketchQuery::Quantile { q } if (0.0..=1.0).contains(q) => self + .inner + .quantile(*q) + .ok_or_else(|| "DDSketch quantile of an empty population".into()), + SketchQuery::Quantile { .. } => Err("quantile must be in [0, 1]".into()), + SketchQuery::PointCount { value: None, .. } => Ok(self.inner.total_count() as f64), + other => Err(format!("DDSketch does not answer {other:?}").into()), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use planner_types::pre_asap::ColumnRef; + + fn bare_count() -> SketchQuery { + SketchQuery::PointCount { + key: ColumnRef::SampleValue, + value: None, + } + } + + // A bare point count reads the total sample count, and merge adds counts. + #[test] + fn count_and_quantile_survive_merge() { + let (mut a, mut b) = ( + DDSketchAccumulator::new(0.01), + DDSketchAccumulator::new(0.01), + ); + for v in 1..=50 { + a.inner.update(f64::from(v)); + b.inner.update(f64::from(v + 50)); + } + let merged = a.merge_with(&b).unwrap(); + assert_eq!(merged.estimate(&bare_count()).unwrap(), 100.0); + let median = merged.estimate(&SketchQuery::Quantile { q: 0.5 }).unwrap(); + assert!((median - 50.0).abs() <= 1.0, "{median}"); + } + + // An empty DDSketch has no quantile, and unsupported queries are errors. + #[test] + fn empty_quantile_and_unsupported_queries_fail() { + let dd = DDSketchAccumulator::new(0.01); + assert!(dd.estimate(&SketchQuery::Quantile { q: 0.5 }).is_err()); + assert!(dd.estimate(&SketchQuery::Cardinality).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/exact.rs b/crates/asap-physical-operators/src/summary_kernels/exact.rs new file mode 100644 index 00000000..3343e875 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/exact.rs @@ -0,0 +1,215 @@ +//! Exact summary state identified by Planner family, independent of keyed layout. +use super::increase::IncreaseAccumulator; +use crate::Statistic; +use crate::{AggregateCore, KeyByLabelValues, Measurement}; +use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; +use serde::{Deserialize, Serialize}; +use std::collections::HashMap; + +type Error = Box; + +#[derive(Debug, Clone, Serialize, Deserialize)] +enum ScalarState { + Sum(f64), + Count(u64), + Min(Option), + Max(Option), + Counter(Option), +} + +/// Both the family and population layout survive persistence. Sharing counter +/// arithmetic never authorizes a Rate state to answer an Increase readout. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ExactAccumulator { + family: SummaryFamilyType, + scalar: ScalarState, + keyed: Option>, +} + +/// Planned readout of an exact summary. `lookback_ms` is the logical PromQL +/// counter window; the evaluation range is resolved from it at run time. +#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] +pub struct ExactReadout { + pub statistic: Statistic, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub lookback_ms: Option, +} + +impl ExactAccumulator { + /// Read one population. An empty MIN/MAX population reads as `None`. + /// `range_ms` extrapolates a counter Rate/Increase to that evaluation range. + pub fn readout( + &self, + statistic: Statistic, + range_ms: Option<(i64, i64)>, + key: Option<&KeyByLabelValues>, + ) -> Result, Error> { + if statistic != self.statistic() { + return Err("readout differs from Planner exact family".into()); + } + let state = match (&self.keyed, key) { + (Some(states), Some(key)) => states.get(key).ok_or("unknown exact population")?, + (None, None) => &self.scalar, + _ => return Err("readout population differs from installed layout".into()), + }; + match state { + ScalarState::Sum(sum) => Ok(Some(*sum)), + ScalarState::Count(count) => Ok(Some(*count as f64)), + ScalarState::Min(value) | ScalarState::Max(value) => Ok(*value), + ScalarState::Counter(Some(counter)) => counter + .extrapolated_value(range_ms, statistic == Statistic::Rate) + .map(Some), + ScalarState::Counter(None) => Err("empty counter population".into()), + } + } + + /// Exact integer count of an unkeyed Count state. + pub fn count(&self) -> Option { + match (&self.keyed, &self.scalar) { + (None, ScalarState::Count(count)) => Some(*count), + _ => None, + } + } + + /// Accumulate into run-local scratch state. Persistent input states remain + /// immutable; a failed merge discards this scratch state. + pub(crate) fn merge_from(&mut self, other: &Self) -> Result<(), Error> { + if self.family != other.family || self.is_keyed() != other.is_keyed() { + return Err("cannot merge different Planner families or layouts".into()); + } + if let (Some(target), Some(source)) = (&mut self.keyed, &other.keyed) { + for (key, state) in source { + let combined = match target.get(key) { + Some(old) => merge_scalar(old, state)?, + None => state.clone(), + }; + target.insert(key.clone(), combined); + } + } else { + self.scalar = merge_scalar(&self.scalar, &other.scalar)?; + } + Ok(()) + } + + pub fn new(family: SummaryFamilyType, keyed: bool) -> Result { + use ExactKind as K; + use ExactParams as P; + let scalar = match &family { + SummaryFamilyType::ExactAggregate(K::Sum, P::Sum) => ScalarState::Sum(0.0), + SummaryFamilyType::ExactAggregate(K::Count, P::Count) => ScalarState::Count(0), + SummaryFamilyType::ExactAggregate(K::Min, P::Min) => ScalarState::Min(None), + SummaryFamilyType::ExactAggregate(K::Max, P::Max) => ScalarState::Max(None), + SummaryFamilyType::ExactAggregate(K::Rate, P::Rate) + | SummaryFamilyType::ExactAggregate(K::Increase, P::Increase) => { + ScalarState::Counter(None) + } + _ => return Err(format!("unsupported exact Planner family: {family:?}")), + }; + Ok(Self { + family, + scalar, + keyed: keyed.then(HashMap::new), + }) + } + + pub fn family(&self) -> &SummaryFamilyType { + &self.family + } + pub fn is_keyed(&self) -> bool { + self.keyed.is_some() + } + + pub fn update(&mut self, key: Option<&KeyByLabelValues>, value: f64, timestamp: i64) { + let state = match (&mut self.keyed, key) { + (Some(states), Some(key)) => states + .entry(key.clone()) + .or_insert_with(|| self.scalar.clone()), + (None, None) => &mut self.scalar, + _ => panic!("exact update population layout differs from installed DAG"), + }; + match state { + ScalarState::Sum(sum) => *sum += value, + ScalarState::Count(count) => { + *count = count.checked_add(1).expect("exact count overflow") + } + ScalarState::Min(current) => { + *current = Some(current.map_or(value, |old| old.min(value))) + } + ScalarState::Max(current) => { + *current = Some(current.map_or(value, |old| old.max(value))) + } + ScalarState::Counter(current) => match current { + Some(counter) => counter.update(Measurement::new(value), timestamp), + None => { + *current = Some(IncreaseAccumulator::new( + Measurement::new(value), + timestamp, + Measurement::new(value), + timestamp, + )) + } + }, + } + } + + fn statistic(&self) -> Statistic { + match self.family { + SummaryFamilyType::ExactAggregate(ExactKind::Sum, _) => Statistic::Sum, + SummaryFamilyType::ExactAggregate(ExactKind::Count, _) => Statistic::Count, + SummaryFamilyType::ExactAggregate(ExactKind::Min, _) => Statistic::Min, + SummaryFamilyType::ExactAggregate(ExactKind::Max, _) => Statistic::Max, + SummaryFamilyType::ExactAggregate(ExactKind::Rate, _) => Statistic::Rate, + SummaryFamilyType::ExactAggregate(ExactKind::Increase, _) => Statistic::Increase, + _ => unreachable!("validated exact family"), + } + } +} + +fn merge_scalar(left: &ScalarState, right: &ScalarState) -> Result { + Ok(match (left, right) { + (ScalarState::Sum(a), ScalarState::Sum(b)) => ScalarState::Sum(a + b), + (ScalarState::Count(a), ScalarState::Count(b)) => { + ScalarState::Count(a.checked_add(*b).ok_or("exact count overflow")?) + } + (ScalarState::Min(a), ScalarState::Min(b)) => { + ScalarState::Min(a.iter().chain(b).copied().reduce(f64::min)) + } + (ScalarState::Max(a), ScalarState::Max(b)) => { + ScalarState::Max(a.iter().chain(b).copied().reduce(f64::max)) + } + (ScalarState::Counter(a), ScalarState::Counter(b)) => ScalarState::Counter(match (a, b) { + (Some(a), Some(b)) => Some(IncreaseAccumulator::merge_pair(a, b)), + (a, b) => a.clone().or_else(|| b.clone()), + }), + _ => return Err("exact scalar state families differ".into()), + }) +} + +impl AggregateCore for ExactAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + fn as_any(&self) -> &dyn std::any::Any { + self + } + fn merge_with(&self, other: &dyn AggregateCore) -> Result, Error> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("merge requires Planner exact state")?; + let mut merged = self.clone(); + merged.merge_from(other)?; + Ok(Box::new(merged)) + } + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::() + + self.keyed.as_ref().map_or(0, |m| { + m.keys() + .map(|k| { + std::mem::size_of::() + + k.labels.iter().map(String::len).sum::() + }) + .sum::() + }) + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/factory.rs b/crates/asap-physical-operators/src/summary_kernels/factory.rs new file mode 100644 index 00000000..d44e64ab --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/factory.rs @@ -0,0 +1,799 @@ +use crate::summary_kernels::hll_sketch::HllSketchAccumulator; +use crate::summary_kernels::univmon::UnivMonAccumulator; +use crate::summary_kernels::{ + CountMinSketchAccumulator, CountMinSketchWithHeapAccumulator, CountSketchAccumulator, + CountSketchWithHeapAccumulator, DDSketchAccumulator, DatasketchesKLLAccumulator, + HydraKllSketchAccumulator, +}; +use crate::{AggregateCore, KeyByLabelValues}; +use planner_types::post_asap::{SketchAlgorithm, SketchParams, SummaryFamilyType}; + +/// Generate the clone-based `AccumulatorUpdater` methods for updaters whose +/// inner `acc` field implements `Clone + AggregateCore`. +macro_rules! impl_clone_accumulator_methods { + ($acc_field:ident) => { + fn take_accumulator(&mut self) -> Box { + let result = Box::new(self.$acc_field.clone()); + self.reset(); + result + } + + fn snapshot_accumulator(&self) -> Box { + Box::new(self.$acc_field.clone()) + } + + fn into_accumulator(self: Box) -> Box { + // Consume the updater and MOVE the accumulator out — no clone. + // Avoids a clone when a pane is evicted at window close. + let this = *self; + Box::new(this.$acc_field) + } + }; +} + +/// Shared update interface for query-time and precompute-time accumulation. +/// +/// This provides a uniform interface over all accumulator types so that the +/// operators don't need to know which concrete type they're dealing with. +pub trait AccumulatorUpdater: Send { + /// Validate an immutable precompute input before an updater can silently + /// discard a value outside its representable domain. + fn validate_single_input(&self, value: f64) -> Result<(), String> { + if value.is_finite() { + Ok(()) + } else { + Err("accumulator input must be finite".into()) + } + } + + /// Feed a single (value, timestamp_ms) pair — for SingleSubpopulation types. + fn update_single(&mut self, value: f64, timestamp_ms: i64); + + /// Feed a keyed (key, value, timestamp_ms) triple, e.g. a frequency item or an exact keyed state. + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, timestamp_ms: i64); + + /// Extract the final accumulator as a boxed `AggregateCore`. + fn take_accumulator(&mut self) -> Box; + + /// Non-destructive read of the current accumulator state (clone without reset). + /// Used by pane-based sliding windows to read shared panes. + fn snapshot_accumulator(&self) -> Box; + + /// Consume the updater and return its accumulator by move, avoiding the + /// clone that `take_accumulator`/`snapshot_accumulator` pay. Default falls + /// back to a clone for updaters that can't move their inner state out. + fn into_accumulator(self: Box) -> Box { + self.snapshot_accumulator() + } + + /// Reset internal state for reuse (avoids re-allocation). + fn reset(&mut self); + + /// Whether this updater consumes keyed updates. + fn is_keyed(&self) -> bool; + + /// Estimated memory usage in bytes. + fn memory_usage_bytes(&self) -> usize; +} + +// --------------------------------------------------------------------------- +// KllAccumulatorUpdater +// --------------------------------------------------------------------------- + +pub struct KllAccumulatorUpdater { + acc: DatasketchesKLLAccumulator, + k: u16, +} + +impl KllAccumulatorUpdater { + pub fn new(k: u16) -> Self { + Self { + acc: DatasketchesKLLAccumulator::new(k), + k, + } + } +} + +impl AccumulatorUpdater for KllAccumulatorUpdater { + fn update_single(&mut self, value: f64, _timestamp_ms: i64) { + self.acc.update(value); + } + + fn update_keyed(&mut self, _key: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = DatasketchesKLLAccumulator::new(self.k); + } + + fn is_keyed(&self) -> bool { + false + } + + fn memory_usage_bytes(&self) -> usize { + // KLL sketch size is hard to estimate precisely; use a rough estimate + std::mem::size_of::() + 4096 + } +} + +// --------------------------------------------------------------------------- +// DDSketchAccumulatorUpdater +// --------------------------------------------------------------------------- +pub struct DDSketchAccumulatorUpdater { + acc: DDSketchAccumulator, + alpha: f64, +} + +impl DDSketchAccumulatorUpdater { + pub fn new(alpha: f64) -> Self { + Self { + acc: DDSketchAccumulator::new(alpha), + alpha, + } + } +} + +impl AccumulatorUpdater for DDSketchAccumulatorUpdater { + fn validate_single_input(&self, value: f64) -> Result<(), String> { + let (minimum, maximum) = + asap_sketchlib::sketches::ddsketch::ddsketch_indexable_bounds(self.alpha); + if value.is_finite() && value > 0.0 && value >= minimum && value <= maximum { + Ok(()) + } else { + Err("DDS maintenance input is outside its positive representable domain".into()) + } + } + + fn update_single(&mut self, value: f64, _timestamp_ms: i64) { + self.acc.inner.update(value); + } + + fn update_keyed(&mut self, _key: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = DDSketchAccumulator::new(self.alpha); + } + + fn is_keyed(&self) -> bool { + false + } + + fn memory_usage_bytes(&self) -> usize { + // Bucket store is variable; rough estimate matches KLL. + std::mem::size_of::() + 4096 + } +} + +// --------------------------------------------------------------------------- +// CmsAccumulatorUpdater (CountMinSketch) +// --------------------------------------------------------------------------- + +/// Keyed weighted-frequency updater. +/// +/// A raw Prometheus sample represents the observed metric value, so a bare CMS +/// adds `value` for its key. Counting each received sample as one is a distinct +/// event-count operation and requires an explicit typed plan contract; it must +/// not be inferred from the sketch algorithm alone. +pub struct CmsAccumulatorUpdater { + acc: CountMinSketchAccumulator, + row_num: usize, + col_num: usize, +} + +impl CmsAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + acc: CountMinSketchAccumulator::new(row_num, col_num), + row_num, + col_num, + } + } +} + +impl AccumulatorUpdater for CmsAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.inner.update(&key.to_semicolon_str(), value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = CountMinSketchAccumulator::new(self.row_num, self.col_num); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + } +} + +// --------------------------------------------------------------------------- +// CmsHeapAccumulatorUpdater — value-weighted / count-weighted top-k +// --------------------------------------------------------------------------- + +/// What quantity the top-k heap ranks keys by. +/// +/// These are DIFFERENT query semantics and must be chosen explicitly: +/// +/// * [`TopkWeight::Value`] — accumulate **Σ of the datapoint value** per key. +/// This answers "top-k by total " (e.g. "top-k hosts by +/// total CPU"). The heap value is the summed metric value, so the read-side +/// reducer's "sort heap descending by value" yields the correct ranking. +/// +/// * [`TopkWeight::Count`] — accumulate **+1 per event** per key (occurrence +/// frequency), the textbook heavy-hitter / frequency-top-k semantics +/// ("which keys appear most often"). +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum TopkWeight { + /// Σ datapoint value per key (value-weighted top-k). + Value, + /// +1 per event per key (count-weighted / frequency top-k). + Count, +} + +/// Keyed top-k updater backed by a real `CountMinSketchWithHeap` (a CMS +/// matrix PLUS a size-`heap_size` top-k heap). Unlike the heap-LESS +/// `CmsAccumulatorUpdater`, this enumerates top-k keys at read time +/// (`get_topk_keys` / `topk_heap_items`), which is what `topk(...)` queries +/// need. +/// +/// The key is the frequency item supplied by the operator (e.g. a `host` +/// value), not a group-by population. The accumulated quantity is selected +/// by [`TopkWeight`]: +/// * `Value` → `inner.update(key, value)` adds the datapoint value (Σ value). +/// * `Count` → `inner.update(key, 1.0)` adds one per event (Σ count). +/// +/// Both `CountMinSketchWithHeap` and `CountSketchWithHeap` raw-input policies +/// route here; the heap is the shared distinguishing payload. +pub struct CmsHeapAccumulatorUpdater { + acc: CountMinSketchWithHeapAccumulator, + row_num: usize, + col_num: usize, + heap_size: usize, + weight: TopkWeight, +} + +impl CmsHeapAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize, heap_size: usize, weight: TopkWeight) -> Self { + Self { + acc: CountMinSketchWithHeapAccumulator::new(row_num, col_num, heap_size), + row_num, + col_num, + heap_size, + weight, + } + } +} + +impl AccumulatorUpdater for CmsHeapAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + // Heap key = the group-by label-value vector (e.g. `host`), joined the + // same way the read-side `get_topk_keys` splits it back apart (`;`). + let weighted = match self.weight { + // Σ value: feed the datapoint value. sketchlib's CMS-heap + // `update(key, w)` adds `w.round()` occurrences of `key`, so the + // heap value accumulates the (rounded) summed metric value. + TopkWeight::Value => value, + // Σ count: one occurrence per event, regardless of value. + TopkWeight::Count => 1.0, + }; + self.acc.inner.update(&key.to_semicolon_str(), weighted); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = + CountMinSketchWithHeapAccumulator::new(self.row_num, self.col_num, self.heap_size); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + + self.heap_size * (std::mem::size_of::() + 32) + } +} + +// --------------------------------------------------------------------------- +// CountSketchAccumulatorUpdater (real median-of-signed-rows CountSketch) +// --------------------------------------------------------------------------- + +/// Keyed point-frequency updater backed by a real `asap_sketchlib::CountSketch` +/// (signed rows, median-of-rows estimator) — distinct math from +/// `CmsAccumulatorUpdater`'s CMS (min-of-rows). +/// +/// As with bare CMS, each raw Prometheus sample contributes its `value`. +/// Unit event counting must be selected explicitly by a future typed plan +/// contract rather than being implied by `SketchAlgorithm::CountSketch`. +pub struct CountSketchAccumulatorUpdater { + acc: CountSketchAccumulator, + row_num: usize, + col_num: usize, +} + +impl CountSketchAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + acc: CountSketchAccumulator::new(row_num, col_num), + row_num, + col_num, + } + } +} + +impl AccumulatorUpdater for CountSketchAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.inner.update(&key.to_semicolon_str(), value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = CountSketchAccumulator::new(self.row_num, self.col_num); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + } +} + +// --------------------------------------------------------------------------- +// CountSketchWithHeapAccumulatorUpdater (real CountSketch + top-k heap) +// --------------------------------------------------------------------------- + +/// Keyed top-k updater backed by a real `CountSketchWithHeap` (signed-row +/// CountSketch matrix PLUS a size-`heap_size` top-k heap). Distinct math from +/// `CmsHeapAccumulatorUpdater`'s CMS-with-heap (min-of-rows); shares the same +/// [`TopkWeight`] semantics and heap payload shape. +pub struct CountSketchWithHeapAccumulatorUpdater { + acc: CountSketchWithHeapAccumulator, + row_num: usize, + col_num: usize, + heap_size: usize, + weight: TopkWeight, +} + +impl CountSketchWithHeapAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize, heap_size: usize, weight: TopkWeight) -> Self { + Self { + acc: CountSketchWithHeapAccumulator::new(row_num, col_num, heap_size), + row_num, + col_num, + heap_size, + weight, + } + } +} + +impl AccumulatorUpdater for CountSketchWithHeapAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + let weighted = match self.weight { + TopkWeight::Value => value, + TopkWeight::Count => 1.0, + }; + self.acc.inner.update(&key.to_semicolon_str(), weighted); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = CountSketchWithHeapAccumulator::new(self.row_num, self.col_num, self.heap_size); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + + self.heap_size * (std::mem::size_of::() + 32) + } +} + +// --------------------------------------------------------------------------- +// HydraKllAccumulatorUpdater +// --------------------------------------------------------------------------- + +pub struct HydraKllAccumulatorUpdater { + acc: HydraKllSketchAccumulator, + row_num: usize, + col_num: usize, + k: u16, +} + +impl HydraKllAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize, k: u16) -> Self { + Self { + acc: HydraKllSketchAccumulator::new(row_num, col_num, k), + row_num, + col_num, + k, + } + } +} + +impl AccumulatorUpdater for HydraKllAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.update(key, value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = HydraKllSketchAccumulator::new(self.row_num, self.col_num, self.k); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + // Rough estimate: each cell is a KLL sketch + std::mem::size_of::() + self.row_num * self.col_num * 4096 + } +} + +// --------------------------------------------------------------------------- +// Config helpers +// --------------------------------------------------------------------------- + +fn cms_dims(params: &SketchParams) -> (usize, usize) { + match params { + SketchParams::Cms { width, depth } | SketchParams::CountSketch { width, depth } => { + (*depth as usize, *width as usize) + } + other => unreachable!( + "accumulator_spec() paired SketchAlgorithm::Cms/CountSketch with unexpected params: {other:?}" + ), + } +} + +/// Read `(rows = depth, columns = width, heap_size)` out of `SketchParams::CmsWithHeap` +/// or `::CountSketchWithHeap`. +fn cms_heap_dims(params: &SketchParams) -> (usize, usize, usize) { + match params { + SketchParams::CmsWithHeap { + width, + depth, + heap_size, + } + | SketchParams::CountSketchWithHeap { + width, + depth, + heap_size, + } => (*depth as usize, *width as usize, *heap_size as usize), + other => unreachable!( + "accumulator_spec() paired a WithHeap SketchAlgorithm with unexpected params: {other:?}" + ), + } +} + +/// Construct the kernel declared by a Planner SummaryAgg. No deployment config +/// tags participate in this dispatch and unsupported payloads are errors. +pub fn create_planner_accumulator( + family: &SummaryFamilyType, + input: &planner_types::post_asap::SummaryUpdate, + grouping: &planner_types::post_asap::GroupingStrategy, +) -> Result, String> { + if input.item.is_some() + && matches!( + input.weight_domain, + planner_types::post_asap::WeightDomain::NonNegative { + proof: + planner_types::post_asap::NonNegativeWeightProof::ResetAwareCounterDerivative + } + ) + { + return Err("window-weighted summaries require typed DAG binding; integer heap updaters cannot consume rates".into()); + } + + crate::capability::validate_summary_kernel(family, input, grouping)?; + use planner_types::post_asap::GroupingStrategy; + if grouping != &GroupingStrategy::PerSubpopulationInstance { + return Err("shared summary grouping requires a supported Planner Hydra kernel".into()); + } + if matches!(family, SummaryFamilyType::ExactAggregate(..)) { + return Ok(Box::new(PlannerExactUpdater { + acc: crate::summary_kernels::exact::ExactAccumulator::new( + family.clone(), + input.item.is_some(), + )?, + })); + } + let SummaryFamilyType::Sketch(kind, family_grouping) = family else { + return Err(format!("unsupported Planner summary family {family:?}")); + }; + if family_grouping != grouping { + return Err("Planner family and operator grouping disagree".into()); + } + let updater: Box = match (kind.algorithm(), kind.params()) { + (SketchAlgorithm::Kll, SketchParams::Kll { k }) => Box::new(KllAccumulatorUpdater::new( + u16::try_from(*k).map_err(|_| "KLL k exceeds runtime bound")?, + )), + (SketchAlgorithm::DDSketch, SketchParams::DDSketch { alpha }) => { + Box::new(DDSketchAccumulatorUpdater::new(*alpha)) + } + (SketchAlgorithm::Cms, params @ SketchParams::Cms { .. }) => { + let (r, c) = cms_dims(params); + Box::new(CmsAccumulatorUpdater::new(r, c)) + } + (SketchAlgorithm::CountSketch, params @ SketchParams::CountSketch { .. }) => { + let (r, c) = cms_dims(params); + Box::new(CountSketchAccumulatorUpdater::new(r, c)) + } + (SketchAlgorithm::CmsWithHeap, params @ SketchParams::CmsWithHeap { .. }) => { + let (r, c, h) = cms_heap_dims(params); + Box::new(CmsHeapAccumulatorUpdater::new(r, c, h, TopkWeight::Value)) + } + ( + SketchAlgorithm::CountSketchWithHeap, + params @ SketchParams::CountSketchWithHeap { .. }, + ) => { + let (r, c, h) = cms_heap_dims(params); + Box::new(CountSketchWithHeapAccumulatorUpdater::new( + r, + c, + h, + TopkWeight::Value, + )) + } + (SketchAlgorithm::Hll, SketchParams::Hll { precision }) => Box::new(HllUpdater { + acc: HllSketchAccumulator::new( + asap_sketchlib::HllVariant::Regular, + u32::from(*precision), + ), + }), + ( + SketchAlgorithm::UnivMon, + SketchParams::UnivMon { + heap_size, + sketch_rows, + sketch_cols, + layers, + }, + ) => Box::new(UnivMonUpdater { + acc: UnivMonAccumulator::new( + *heap_size as usize, + *sketch_rows as usize, + *sketch_cols as usize, + *layers as usize, + ) + .map_err(|e| e.to_string())?, + }), + _ => { + return Err(format!( + "unsupported Planner algorithm/parameters: {kind:?}" + )) + } + }; + if updater.is_keyed() != input.item.is_some() + && !crate::capability::is_unit_sample_frequency(input) + { + return Err("Planner item expression does not match the selected kernel layout".into()); + } + Ok(updater) +} + +struct PlannerExactUpdater { + acc: crate::summary_kernels::exact::ExactAccumulator, +} +impl AccumulatorUpdater for PlannerExactUpdater { + fn update_single(&mut self, value: f64, timestamp: i64) { + self.acc.update(None, value, timestamp); + } + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, timestamp: i64) { + self.acc.update(Some(key), value, timestamp); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc = crate::summary_kernels::exact::ExactAccumulator::new( + self.acc.family().clone(), + self.acc.is_keyed(), + ) + .expect("installed exact family"); + } + fn is_keyed(&self) -> bool { + self.acc.is_keyed() + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } +} + +struct UnivMonUpdater { + acc: UnivMonAccumulator, +} + +struct HllUpdater { + acc: HllSketchAccumulator, +} + +impl AccumulatorUpdater for HllUpdater { + fn is_keyed(&self) -> bool { + false + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } + fn update_single(&mut self, value: f64, _: i64) { + if !value.is_nan() { + let bits = if value == 0.0 { 0 } else { value.to_bits() }; + self.acc.inner.update(&bits.to_le_bytes()); + } + } + fn update_keyed(&mut self, _: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc.inner = + asap_sketchlib::HllSketch::new(self.acc.inner.variant, self.acc.inner.precision); + } +} + +impl AccumulatorUpdater for UnivMonUpdater { + fn is_keyed(&self) -> bool { + false + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } + fn update_single(&mut self, value: f64, _: i64) { + self.acc + .insert_sample(value) + .expect("UnivMon sample counter overflow"); + } + fn update_keyed(&mut self, _: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc.clear(); + } +} + +#[cfg(test)] +mod planner_parameter_regression { + use super::*; + use planner_types::post_asap::{SketchKind, SummaryInputExpr, SummaryUpdate}; + + // Planner width is the bucket count; depth is the independent hash-row count. + #[test] + fn planner_sketch_dimensions_are_not_transposed() { + for (algorithm, params) in [ + ( + SketchAlgorithm::Cms, + SketchParams::Cms { + width: 128, + depth: 3, + }, + ), + ( + SketchAlgorithm::CountSketch, + SketchParams::CountSketch { + width: 128, + depth: 3, + }, + ), + ( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 128, + depth: 3, + heap_size: 8, + }, + ), + ( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 128, + depth: 3, + heap_size: 8, + }, + ), + ] { + let family = SummaryFamilyType::Sketch( + SketchKind::new(algorithm.clone(), params), + Default::default(), + ); + let update = SummaryUpdate { + item: Some(SummaryInputExpr::Column( + planner_types::pre_asap::ColumnRef::Named("host".into()), + )), + weight: SummaryInputExpr::Constant(1.0), + weight_domain: Default::default(), + }; + let state = create_planner_accumulator(&family, &update, &Default::default()) + .unwrap() + .snapshot_accumulator(); + let dims = match algorithm { + SketchAlgorithm::Cms => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows(), s.inner.cols()) + } + SketchAlgorithm::CountSketch => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows, s.inner.cols) + } + SketchAlgorithm::CmsWithHeap => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows(), s.inner.cols()) + } + SketchAlgorithm::CountSketchWithHeap => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows(), s.inner.cols()) + } + _ => unreachable!(), + }; + assert_eq!(dims, (3, 128), "{algorithm:?}"); + } + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs new file mode 100644 index 00000000..089261c5 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs @@ -0,0 +1,79 @@ +//! HyperLogLog distinct-count summary over `asap_sketchlib::HllSketch`. +use crate::{AggregateCore, KernelError}; +use asap_sketchlib::{HllSketch, HllVariant}; +use planner_types::post_asap::SketchQuery; + +#[derive(Debug, Clone)] +pub struct HllSketchAccumulator { + pub inner: HllSketch, +} + +impl HllSketchAccumulator { + pub fn new(variant: HllVariant, precision: u32) -> Self { + Self { + inner: HllSketch::new(variant, precision), + } + } +} + +impl AggregateCore for HllSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("HLL merges only with HLL")?; + Ok(Box::new(Self { + inner: HllSketch::merge_refs(&[&self.inner, &other.inner])?, + })) + } + + /// Distinct count. A bare `PointCount` over an HLL also reads the distinct count. + fn estimate(&self, query: &SketchQuery) -> Result { + match query { + SketchQuery::Cardinality | SketchQuery::PointCount { value: None, .. } => { + Ok(self.inner.estimate()) + } + other => Err(format!("HLL does not answer {other:?}").into()), + } + } + + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::().saturating_add(self.inner.registers.capacity()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Distinct count after merge counts overlapping items once. + #[test] + fn merged_cardinality_deduplicates_overlap() { + let (mut a, mut b) = ( + HllSketchAccumulator::new(HllVariant::Regular, 12), + HllSketchAccumulator::new(HllVariant::Regular, 12), + ); + for v in 0..1000u32 { + a.inner.update(&v.to_le_bytes()); + b.inner.update(&(v + 500).to_le_bytes()); + } + let merged = a.merge_with(&b).unwrap(); + let estimate = merged.estimate(&SketchQuery::Cardinality).unwrap(); + assert!((estimate - 1500.0).abs() / 1500.0 < 0.05, "{estimate}"); + } + + // HLL does not answer quantiles. + #[test] + fn rejects_quantile() { + let hll = HllSketchAccumulator::new(HllVariant::Regular, 12); + assert!(hll.estimate(&SketchQuery::Quantile { q: 0.5 }).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs b/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs new file mode 100644 index 00000000..c0347e69 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs @@ -0,0 +1,55 @@ +use crate::{AggregateCore, KeyByLabelValues}; +use asap_sketchlib::HydraKllSketch; + +/// HydraKLL (shared-grouping quantiles) over `asap_sketchlib::HydraKllSketch`. +#[derive(Debug, Clone)] +pub struct HydraKllSketchAccumulator { + pub inner: HydraKllSketch, +} + +impl HydraKllSketchAccumulator { + pub fn new(row_num: usize, col_num: usize, k: u16) -> Self { + Self { + inner: HydraKllSketch::new(row_num, col_num, k), + } + } + + pub fn update(&mut self, key: &KeyByLabelValues, value: f64) { + self.inner.update(&key.to_semicolon_str(), value); + } + + pub fn query_key(&self, key: &KeyByLabelValues, quantile: f64) -> f64 { + self.inner.quantile(&key.to_semicolon_str(), quantile) + } +} + +impl AggregateCore for HydraKllSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let hk = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to HydraKllSketchAccumulator")?; + + let mut merged = self.clone(); + merged.inner.merge(&hk.inner)?; + Ok(Box::new(merged)) + } + + fn approx_memory_bytes(&self) -> usize { + // HydraKLL is a row*col grid of KLL sketches; typical instances + // are on the order of tens of KiB. 32 KiB is a conservative + // per-instance default. + 32 * 1024 + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/increase.rs b/crates/asap-physical-operators/src/summary_kernels/increase.rs new file mode 100644 index 00000000..f92ee582 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/increase.rs @@ -0,0 +1,213 @@ +use crate::{AggregateCore, Measurement}; +use serde::{Deserialize, Serialize}; + +/// Accumulator for tracking increases in counter metrics +/// Stores the starting and last seen measurements with timestamps +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct IncreaseAccumulator { + pub starting_measurement: Measurement, + pub starting_timestamp: i64, + pub last_seen_measurement: Measurement, + pub last_seen_timestamp: i64, + /// Sum of monotonic deltas, adding the post-reset value whenever the + /// counter decreases. This is the reset correction Prometheus applies. + #[serde(default)] + pub total_increase: f64, + #[serde(default)] + pub sample_count: u64, +} + +impl IncreaseAccumulator { + /// Merge two counter intervals without a temporary collection. Ties retain + /// the left input, matching the stable ordering of multi-pane merges. + pub(crate) fn merge_pair(left: &Self, right: &Self) -> Self { + let (first, second) = if left.starting_timestamp <= right.starting_timestamp { + (left, right) + } else { + (right, left) + }; + let mut merged = first.clone(); + if second.starting_timestamp > merged.last_seen_timestamp { + merged.total_increase += + if second.starting_measurement.value >= merged.last_seen_measurement.value { + second.starting_measurement.value - merged.last_seen_measurement.value + } else { + second.starting_measurement.value + }; + } + merged.total_increase += second.total_increase; + merged.sample_count = merged.sample_count.saturating_add(second.sample_count); + if second.last_seen_timestamp > merged.last_seen_timestamp { + merged.last_seen_measurement = second.last_seen_measurement.clone(); + merged.last_seen_timestamp = second.last_seen_timestamp; + } + + merged + } + + pub fn new( + starting_measurement: Measurement, + starting_timestamp: i64, + last_seen_measurement: Measurement, + last_seen_timestamp: i64, + ) -> Self { + let total_increase = if last_seen_timestamp <= starting_timestamp { + 0.0 + } else if last_seen_measurement.value >= starting_measurement.value { + last_seen_measurement.value - starting_measurement.value + } else { + last_seen_measurement.value + }; + let sample_count = if last_seen_timestamp > starting_timestamp { + 2 + } else { + 1 + }; + Self { + starting_measurement, + starting_timestamp, + last_seen_measurement, + last_seen_timestamp, + total_increase, + sample_count, + } + } + + pub fn update(&mut self, measurement: Measurement, timestamp: i64) { + if timestamp < self.last_seen_timestamp { + return; + } + if timestamp == self.last_seen_timestamp { + return; + } + if measurement.value >= self.last_seen_measurement.value { + self.total_increase += measurement.value - self.last_seen_measurement.value; + } else { + self.total_increase += measurement.value; + } + self.last_seen_measurement = measurement; + self.last_seen_timestamp = timestamp; + self.sample_count = self.sample_count.saturating_add(1); + } +} + +impl AggregateCore for IncreaseAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + // Downcast to IncreaseAccumulator + let other_increase = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to IncreaseAccumulator")?; + + let merged = Self::merge_pair(self, other_increase); + Ok(Box::new(merged)) + } + + fn approx_memory_bytes(&self) -> usize { + // Two Measurements + two i64s. Measurements are a few f64 fields. + std::mem::size_of::() + } +} + +impl IncreaseAccumulator { + /// PromQL-style increase or rate, extrapolated to `range_ms` when given. + pub(crate) fn extrapolated_value( + &self, + range_ms: Option<(i64, i64)>, + is_rate: bool, + ) -> Result> { + if self.sample_count < 2 || self.last_seen_timestamp <= self.starting_timestamp { + return Err("at least two ordered counter samples are required".into()); + } + let sampled_interval = (self.last_seen_timestamp - self.starting_timestamp) as f64 / 1000.0; + let Some((range_start, range_end)) = range_ms else { + return Ok(if is_rate { + self.total_increase / sampled_interval + } else { + self.total_increase + }); + }; + if range_end <= range_start { + return Err("invalid counter evaluation range".into()); + } + + let mut duration_to_start = + (self.starting_timestamp.saturating_sub(range_start)) as f64 / 1000.0; + let duration_to_end = (range_end.saturating_sub(self.last_seen_timestamp)) as f64 / 1000.0; + let average_sample_interval = sampled_interval / (self.sample_count - 1) as f64; + let extrapolation_threshold = average_sample_interval * 1.1; + + if self.total_increase > 0.0 && self.starting_measurement.value >= 0.0 { + let duration_to_zero = + sampled_interval * (self.starting_measurement.value / self.total_increase); + duration_to_start = duration_to_start.min(duration_to_zero); + } + let mut extrapolate_to = sampled_interval; + extrapolate_to += if duration_to_start < extrapolation_threshold { + duration_to_start.max(0.0) + } else { + average_sample_interval / 2.0 + }; + extrapolate_to += if duration_to_end < extrapolation_threshold { + duration_to_end.max(0.0) + } else { + average_sample_interval / 2.0 + }; + let mut factor = extrapolate_to / sampled_interval; + if is_rate { + factor /= (range_end - range_start) as f64 / 1000.0; + } + Ok(self.total_increase * factor) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_increase_accumulator_creation() { + let starting_measurement = Measurement::new(10.0); + let last_seen_measurement = Measurement::new(25.0); + let acc = IncreaseAccumulator::new( + starting_measurement.clone(), + 1000, + last_seen_measurement.clone(), + 2000, + ); + + assert_eq!(acc.starting_measurement.value, 10.0); + assert_eq!(acc.starting_timestamp, 1000); + assert_eq!(acc.last_seen_measurement.value, 25.0); + assert_eq!(acc.last_seen_timestamp, 2000); + } + + #[test] + fn test_increase_accumulator_update() { + let starting_measurement = Measurement::new(10.0); + let mut acc = IncreaseAccumulator::new( + starting_measurement.clone(), + 1000, + starting_measurement.clone(), + 1000, + ); + + let new_measurement = Measurement::new(25.0); + acc.update(new_measurement.clone(), 2000); + + assert_eq!(acc.last_seen_measurement.value, 25.0); + assert_eq!(acc.last_seen_timestamp, 2000); + assert_eq!(acc.starting_measurement.value, 10.0); // Should remain unchanged + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/mod.rs b/crates/asap-physical-operators/src/summary_kernels/mod.rs new file mode 100644 index 00000000..478e75d0 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/mod.rs @@ -0,0 +1,26 @@ +//! In-memory summary state: thin adapters over `asap_sketchlib` and exact Planner state. +pub mod count_min_sketch; +pub mod count_min_sketch_with_heap; +pub mod count_sketch; +pub mod count_sketch_with_heap; +pub mod datasketches_kll; +pub mod dd_sketch; +pub mod exact; +pub mod hll_sketch; +pub mod hydra_kll; +pub mod increase; +pub mod univmon; + +pub use count_min_sketch::*; +pub use count_min_sketch_with_heap::*; +pub use count_sketch::*; +pub use count_sketch_with_heap::*; +pub use datasketches_kll::*; +pub use dd_sketch::*; +pub use hll_sketch::*; +pub use hydra_kll::*; +pub use increase::*; + +pub mod factory; +pub mod traits; +pub mod weighted_frequency; diff --git a/crates/asap-physical-operators/src/summary_kernels/traits.rs b/crates/asap-physical-operators/src/summary_kernels/traits.rs new file mode 100644 index 00000000..7d866077 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/traits.rs @@ -0,0 +1,35 @@ +use planner_types::post_asap::SketchQuery; + +pub type KernelError = Box; + +/// In-memory state of one population's summary. +/// +/// Kernels adapt `asap_sketchlib` structures (or exact Planner state) to the +/// operations physical operators need: merge, typed readout and memory +/// accounting. Grouping belongs to operators; byte encodings belong to +/// `asap_sketchlib` and deployments. +pub trait AggregateCore: Send + Sync { + fn clone_boxed_core(&self) -> Box; + + fn as_any(&self) -> &dyn std::any::Any; + + /// Merge with a state of the same family and shape, leaving both inputs unchanged. + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError>; + + /// Answer a sketch readout. Exact states are read through + /// [`ExactAccumulator::readout`](super::exact::ExactAccumulator::readout). + fn estimate(&self, query: &SketchQuery) -> Result { + Err(format!("{query:?} is not supported by this summary").into()) + } + + /// Approximate in-memory footprint, used for execution memory reservations. + fn approx_memory_bytes(&self) -> usize { + 4096 + } +} + +impl Clone for Box { + fn clone(&self) -> Self { + self.clone_boxed_core() + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/univmon.rs b/crates/asap-physical-operators/src/summary_kernels/univmon.rs new file mode 100644 index 00000000..bffd8afa --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/univmon.rs @@ -0,0 +1,109 @@ +//! One frequency state shared by count, distinct, L2 and entropy readouts. + +use crate::AggregateCore; +use asap_sketchlib::{DataInput, UnivMon}; + +type Error = Box; + +#[derive(Debug, Clone)] +pub struct UnivMonAccumulator { + inner: UnivMon, +} + +impl UnivMonAccumulator { + /// Empty the sketch in place, keeping its shape. + pub(crate) fn clear(&mut self) { + self.inner.free(); + } + + pub fn new(heap_size: usize, rows: usize, cols: usize, layers: usize) -> Result { + if heap_size == 0 || cols == 0 || !(1..=20).contains(&rows) || !(1..=64).contains(&layers) { + return Err("invalid UnivMon dimensions".into()); + } + rows.checked_mul(cols) + .and_then(|n| n.checked_mul(layers)) + .ok_or("UnivMon dimensions overflow")?; + Ok(Self { + inner: UnivMon::init_univmon(heap_size, rows, cols, layers), + }) + } + + /// Each non-NaN sample is one occurrence. Signed zero has one identity. + pub fn insert_sample(&mut self, value: f64) -> Result<(), Error> { + if value.is_nan() { + return Ok(()); + } + self.inner + .bucket_size + .checked_add(1) + .ok_or("UnivMon count overflow")?; + let bits = if value == 0.0 { 0 } else { value.to_bits() }; + self.inner.insert(&DataInput::U64(bits), 1); + Ok(()) + } + + fn compatible(&self, other: &Self) -> bool { + ( + self.inner.heap_size, + self.inner.sketch_row, + self.inner.sketch_col, + self.inner.layer_size, + ) == ( + other.inner.heap_size, + other.inner.sketch_row, + other.inner.sketch_col, + other.inner.layer_size, + ) + } + + pub fn dimensions(&self) -> (usize, usize, usize, usize) { + ( + self.inner.heap_size, + self.inner.sketch_row, + self.inner.sketch_col, + self.inner.layer_size, + ) + } + + pub fn merge_in_place(&mut self, other: &Self) -> Result<(), Error> { + if !self.compatible(other) { + return Err("incompatible UnivMon dimensions".into()); + } + self.inner + .bucket_size + .checked_add(other.inner.bucket_size) + .ok_or("UnivMon count overflow")?; + self.inner.merge(&other.inner); + Ok(()) + } +} + +impl AggregateCore for UnivMonAccumulator { + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::().saturating_add( + self.inner.layer_size.saturating_mul( + self.inner + .sketch_row + .saturating_mul(self.inner.sketch_col) + .saturating_mul(16) + .saturating_add(self.inner.heap_size.saturating_mul(256)), + ), + ) + } + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, Error> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("expected UnivMon state")?; + let mut merged = self.clone(); + merged.merge_in_place(other)?; + Ok(Box::new(merged)) + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs new file mode 100644 index 00000000..6f008532 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs @@ -0,0 +1,152 @@ +//! ASAP type and trait adapter for sketchlib's Float64 weighted frequency kernel. +use crate::AggregateCore; +use crate::{values::Value, Error}; +pub use asap_sketchlib::FrequencyAlgorithm; +use asap_sketchlib::{FrequencyIdentity, WeightedFrequency as Kernel, WeightedFrequencyError}; +use serde::{Deserialize, Serialize}; + +fn adapt_error(error: WeightedFrequencyError) -> Error { + match error { + WeightedFrequencyError::Invalid(message) => Error::Invalid(message), + WeightedFrequencyError::Update(message) => Error::Operator(message), + } +} +fn identity(value: &Value) -> Result { + Ok(match value { + Value::Null => FrequencyIdentity::Null, + Value::Bool(v) => FrequencyIdentity::Bool(*v), + Value::Int64(v) => FrequencyIdentity::Int64(*v), + Value::Float64(v) => FrequencyIdentity::Float64(*v), + Value::Utf8(v) => FrequencyIdentity::Utf8(v.to_string()), + _ => { + return Err(Error::Invalid( + "unsupported weighted frequency identity".into(), + )) + } + }) +} +fn value(identity: FrequencyIdentity) -> Value { + match identity { + FrequencyIdentity::Null => Value::Null, + FrequencyIdentity::Bool(v) => Value::Bool(v), + FrequencyIdentity::Int64(v) => Value::Int64(v), + FrequencyIdentity::Float64(v) => Value::Float64(v), + FrequencyIdentity::Utf8(v) => Value::Utf8(v.into()), + } +} +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(transparent)] +pub struct WeightedFrequency { + inner: Kernel, +} +impl WeightedFrequency { + pub(crate) fn configuration( + kind: &planner_types::post_asap::SketchKind, + ) -> Result<(FrequencyAlgorithm, usize, usize, usize), Error> { + use planner_types::post_asap::{SketchAlgorithm as A, SketchParams as P}; + let (algorithm, width, depth, capacity) = match (kind.algorithm(), kind.params()) { + ( + A::CmsWithHeap, + P::CmsWithHeap { + width, + depth, + heap_size, + }, + ) => (FrequencyAlgorithm::Cms, *width, *depth, *heap_size), + ( + A::CountSketchWithHeap, + P::CountSketchWithHeap { + width, + depth, + heap_size, + }, + ) if depth % 2 == 1 => (FrequencyAlgorithm::CountSketch, *width, *depth, *heap_size), + _ => { + return Err(Error::Invalid( + "unsupported weighted frequency family or depth".into(), + )) + } + }; + if width == 0 || depth == 0 || capacity == 0 { + return Err(Error::Invalid( + "invalid weighted frequency dimensions".into(), + )); + } + Ok((algorithm, width as usize, depth as usize, capacity as usize)) + } + + pub(crate) fn algorithm(&self) -> FrequencyAlgorithm { + self.inner.algorithm() + } + pub(crate) fn shape(&self) -> (usize, usize, usize) { + self.inner.shape() + } + pub fn new( + algorithm: FrequencyAlgorithm, + width: usize, + depth: usize, + capacity: usize, + ) -> Result { + Kernel::new(algorithm, width, depth, capacity) + .map(|inner| Self { inner }) + .map_err(adapt_error) + } + pub fn update(&mut self, values: &[Value], weight: f64) -> Result<(), Error> { + let values = values.iter().map(identity).collect::, _>>()?; + self.inner.update(&values, weight).map_err(adapt_error) + } + pub fn rows(&self, n: usize) -> Vec> { + self.inner + .topk(n) + .into_iter() + .map(|(items, score)| { + let mut row = items.into_iter().map(value).collect::>(); + row.push(Value::Float64(score)); + row + }) + .collect() + } +} +impl AggregateCore for WeightedFrequency { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + fn as_any(&self) -> &dyn std::any::Any { + self + } + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("weighted frequency state type mismatch")?; + Ok(Box::new(Self { + inner: self.inner.merge(&other.inner)?, + })) + } + fn approx_memory_bytes(&self) -> usize { + self.inner.approx_memory_bytes() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Merge uses the same Float64 state representation and rejects other shapes. + #[test] + fn compatible_merge_preserves_fractional_weights() { + let mut left = WeightedFrequency::new(FrequencyAlgorithm::Cms, 4096, 5, 8).unwrap(); + let mut right = left.clone(); + left.update(&[Value::Int64(7)], 0.125).unwrap(); + right.update(&[Value::Int64(7)], 0.25).unwrap(); + let merged = left.merge_with(&right).unwrap(); + let merged = merged.as_any().downcast_ref::().unwrap(); + assert!(matches!(merged.rows(1)[0][1], Value::Float64(0.375))); + assert!(left + .merge_with(&WeightedFrequency::new(FrequencyAlgorithm::Cms, 32, 5, 8).unwrap()) + .is_err()); + } +} diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs new file mode 100644 index 00000000..6e761c11 --- /dev/null +++ b/crates/asap-physical-operators/src/values.rs @@ -0,0 +1,381 @@ +//! Runtime values preserve Planner schemas; summary states are typed values too. +use crate::AggregateCore; +use crate::Error; +use planner_types::{ + post_asap::{SummaryFamilyType, SummarySchema}, + pre_asap::DataType, +}; +use std::{cmp::Ordering, sync::Arc}; +pub type Schema = Arc; +#[derive(Clone, serde::Serialize, serde::Deserialize)] +pub enum Value { + Null, + Bool(bool), + Int64(i64), + Float64(f64), + Utf8(Arc), + Timestamp(i64), + Date(i32), + Interval { + months: i32, + days: i32, + nanos: i64, + }, + List(Arc<[Value]>), + Struct(Arc<[Value]>), + Map(Arc<[(Value, Value)]>), + #[serde(skip)] + Summary { + family: SummaryFamilyType, + state: Arc, + }, +} +impl std::fmt::Debug for Value { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Summary { family, .. } => f.debug_tuple("Summary").field(family).finish(), + _ => write!(f, "{:?}", self.key()), + } + } +} +impl Value { + pub fn bytes(&self) -> usize { + std::mem::size_of::() + + match self { + Self::Utf8(s) => s.len(), + Self::List(v) | Self::Struct(v) => v.iter().map(Self::bytes).sum(), + Self::Map(v) => v.iter().map(|(k, v)| k.bytes() + v.bytes()).sum(), + Self::Summary { state, .. } => state.approx_memory_bytes(), + _ => 0, + } + } + pub fn matches(&self, dtype: &DataType, nullable: bool) -> bool { + if matches!(self, Self::Null) { + return nullable || matches!(dtype, DataType::Null); + } + match (self, dtype) { + (Self::Bool(_), DataType::Bool) + | (Self::Int64(_), DataType::Int64) + | (Self::Float64(_), DataType::Float64) + | (Self::Utf8(_), DataType::Utf8) + | (Self::Timestamp(_), DataType::Timestamp) + | (Self::Date(_), DataType::Date) + | (Self::Interval { .. }, DataType::Interval) => true, + (Self::List(v), DataType::List { element }) => v + .iter() + .all(|v| v.matches(&element.dtype, element.nullable)), + (Self::Struct(v), DataType::Struct { fields }) => { + v.len() == fields.len() + && v.iter() + .zip(fields) + .all(|(v, f)| v.matches(&f.dtype, f.nullable)) + } + ( + Self::Map(v), + DataType::Map { + key, + value, + value_nullable, + }, + ) => v + .iter() + .all(|(k, v)| k.matches(key, false) && v.matches(value, *value_nullable)), + _ => false, + } + } + /// Stable typed equality key. Zero signs and NaN payloads form one group. + pub fn key(&self) -> Result, Error> { + let mut out = Vec::new(); + macro_rules! number { + ($tag:expr,$v:expr) => {{ + out.push($tag); + out.extend_from_slice(&$v.to_le_bytes()); + }}; + } + match self { + Self::Null => out.push(0), + Self::Bool(v) => out.extend([1, *v as u8]), + Self::Int64(v) => number!(2, v), + Self::Float64(v) => { + let bits = if *v == 0. { + 0 + } else if v.is_nan() { + f64::NAN.to_bits() + } else { + v.to_bits() + }; + number!(3, bits); + } + Self::Utf8(v) => { + out.push(4); + out.extend(v.as_bytes()); + } + Self::Timestamp(v) => number!(5, v), + Self::Date(v) => number!(6, v), + Self::Interval { + months, + days, + nanos, + } => { + number!(7, months); + number!(8, days); + number!(9, nanos); + } + Self::List(v) | Self::Struct(v) => { + out.push(if matches!(self, Self::List(_)) { + 10 + } else { + 11 + }); + for v in v.iter() { + let key = v.key()?; + out.extend((key.len() as u64).to_le_bytes()); + out.extend(key); + } + } + Self::Map(v) => { + out.push(12); + for (k, v) in v.iter() { + for value in [k, v] { + let key = value.key()?; + out.extend((key.len() as u64).to_le_bytes()); + out.extend(key); + } + } + } + Self::Summary { .. } => { + return Err(Error::Invalid( + "summary states cannot be grouping keys".into(), + )) + } + } + Ok(out) + } + pub fn compare(&self, other: &Self) -> Result { + Ok(match (self, other) { + (Self::Null, Self::Null) => Ordering::Equal, + (Self::Int64(a), Self::Int64(b)) | (Self::Timestamp(a), Self::Timestamp(b)) => a.cmp(b), + (Self::Float64(a), Self::Float64(b)) => { + if a == b { + Ordering::Equal + } else { + a.total_cmp(b) + } + } + (Self::Utf8(a), Self::Utf8(b)) => a.cmp(b), + (Self::Bool(a), Self::Bool(b)) => a.cmp(b), + (Self::Date(a), Self::Date(b)) => a.cmp(b), + (Self::Map(left), Self::Map(right)) => { + let mut result = Ordering::Equal; + for ((lk, lv), (rk, rv)) in left.iter().zip(right.iter()) { + result = lk.compare(rk)?; + if result != Ordering::Equal { + break; + } + result = match (lv, rv) { + (Self::Null, Self::Null) => Ordering::Equal, + (Self::Null, _) => Ordering::Greater, + (_, Self::Null) => Ordering::Less, + _ => lv.compare(rv)?, + }; + if result != Ordering::Equal { + break; + } + } + if result == Ordering::Equal { + left.len().cmp(&right.len()) + } else { + result + } + } + _ => { + return Err(Error::Operator( + "values do not have a supported common ordering".into(), + )) + } + }) + } +} +#[derive(Clone, Debug)] +pub struct Batch { + schema: Schema, + rows: Vec>, +} +impl Batch { + pub fn try_new(schema: Schema, rows: Vec>) -> Result { + validate_schema(&schema)?; + for row in &rows { + if row.len() != schema.fields.len() { + return Err(Error::Invalid( + "row width differs from Planner schema".into(), + )); + } + for (value, field) in row.iter().zip(&schema.fields) { + let matches = match (&field.dtype, value) { + (SummaryFamilyType::Plain(dtype), value) => { + value.matches(dtype, field.nullable) + } + (expected, Value::Summary { family, state }) => { + expected == family && validate_state(family, state.as_ref()).is_ok() + } + _ => false, + }; + if !matches { + return Err(Error::Invalid(format!( + "value differs from type of {}", + field.name + ))); + } + } + } + Ok(Self { schema, rows }) + } + pub fn schema(&self) -> &Schema { + &self.schema + } + pub fn rows(&self) -> &[Vec] { + &self.rows + } + pub fn bytes(&self) -> usize { + std::mem::size_of::() + + self.rows.capacity() * std::mem::size_of::>() + + self + .rows + .iter() + .flat_map(|r| r.iter()) + .map(Value::bytes) + .sum::() + } +} + +pub(crate) use crate::capability::validate_native_family as validate_family; + +fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Result<(), Error> { + use crate::summary_kernels::{ + datasketches_kll::DatasketchesKLLAccumulator, dd_sketch::DDSketchAccumulator, + exact::ExactAccumulator, hll_sketch::HllSketchAccumulator, + }; + use planner_types::post_asap::SketchParams; + validate_family(family)?; + let valid = match family { + SummaryFamilyType::Sketch(kind, _) + if matches!( + kind.params(), + SketchParams::CmsWithHeap { .. } | SketchParams::CountSketchWithHeap { .. } + ) => + { + use crate::summary_kernels::weighted_frequency::WeightedFrequency; + let (algorithm, width, depth, capacity) = WeightedFrequency::configuration(kind)?; + state + .as_any() + .downcast_ref::() + .is_some_and(|state| { + state.algorithm() == algorithm && state.shape() == (width, depth, capacity) + }) + } + + SummaryFamilyType::ExactAggregate(..) => state + .as_any() + .downcast_ref::() + .is_some_and(|s| s.family() == family && !s.is_keyed()), + SummaryFamilyType::Sketch(kind, _) => match kind.params() { + SketchParams::Kll { k } => state + .as_any() + .downcast_ref::() + .is_some_and(|s| u32::from(s.inner.k()) == *k), + SketchParams::DDSketch { alpha } => state + .as_any() + .downcast_ref::() + .is_some_and(|s| s.inner.alpha == *alpha), + SketchParams::Hll { precision } => state + .as_any() + .downcast_ref::() + .is_some_and(|s| s.inner.precision == u32::from(*precision)), + _ => false, + }, + _ => false, + }; + if valid { + Ok(()) + } else { + Err(Error::Invalid( + "state payload differs from declared family, parameters or population layout".into(), + )) + } +} + +pub(crate) fn validate_schema(schema: &Schema) -> Result<(), Error> { + if schema.time_index.is_some_and(|index| { + schema + .fields + .get(index) + .is_none_or(|field| field.dtype != SummaryFamilyType::Plain(DataType::Timestamp)) + }) { + return Err(Error::Invalid( + "time index must name a Timestamp column".into(), + )); + } + for field in &schema.fields { + if !matches!(field.dtype, SummaryFamilyType::Plain(_)) { + validate_family(&field.dtype)?; + if field.nullable { + return Err(Error::Invalid( + "nullable summary states are not supported".into(), + )); + } + } + } + Ok(()) +} + +#[cfg(test)] +mod weighted_state_tests { + use super::*; + use crate::summary_kernels::weighted_frequency::{FrequencyAlgorithm, WeightedFrequency}; + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + + // A state cannot acquire a different family or shape merely by relabeling its batch. + #[test] + fn weighted_state_family_and_shape_must_match() { + let cms = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 32, + depth: 5, + heap_size: 8, + }, + ), + Default::default(), + ); + let cs = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 32, + depth: 5, + heap_size: 8, + }, + ), + Default::default(), + ); + let state = WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 32, 5, 8).unwrap(); + assert!(validate_state(&cs, &state).is_ok()); + assert!(validate_state(&cms, &state).is_err()); + let wrong_shape = + WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 64, 5, 8).unwrap(); + assert!(validate_state(&cs, &wrong_shape).is_err()); + let even_depth = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 32, + depth: 4, + heap_size: 8, + }, + ), + Default::default(), + ); + assert!(validate_family(&even_depth).is_err()); + } +} diff --git a/crates/asap-physical-operators/tests/deployment.rs b/crates/asap-physical-operators/tests/deployment.rs new file mode 100644 index 00000000..0c261a03 --- /dev/null +++ b/crates/asap-physical-operators/tests/deployment.rs @@ -0,0 +1,90 @@ +//! Exercise the public library without a backend server, store, or scheduler. +use asap_physical_operators::planner::{ + post_asap::{ + GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryFamilyType, + SummaryUpdate, + }, + pre_asap::ColumnRef, +}; +use asap_physical_operators::{factory::create_planner_accumulator, AggregateCore}; + +fn family(k: u32) -> SummaryFamilyType { + SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), + GroupingStrategy::PerSubpopulationInstance, + ) +} +fn build(values: &[f64]) -> Box { + let mut operator = create_planner_accumulator( + &family(512), + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .unwrap(); + for (at, value) in values.iter().enumerate() { + operator.validate_single_input(*value).unwrap(); + operator.update_single(*value, at as i64); + } + operator.into_accumulator() +} +fn read(state: &dyn AggregateCore) -> f64 { + state + .estimate(&asap_physical_operators::planner::post_asap::SketchQuery::Quantile { q: 0.5 }) + .unwrap() +} + +// The same kernels work when every build is query-time, when only a prefix +// was precomputed, and when all state was precomputed before the readout. +#[test] +fn raw_partial_and_fully_precomputed_use_the_same_kernels() { + let raw: Vec = (0..128).map(f64::from).collect(); + let raw_only = build(&raw); + let stored_prefix = build(&raw[..64]); + let query_time_suffix = build(&raw[64..]); + let partial = stored_prefix.merge_with(&*query_time_suffix).unwrap(); + let stored_complete = build(&raw); + assert_eq!(read(&*raw_only), read(&*partial)); + assert_eq!(read(&*partial), read(&*stored_complete)); + assert!((read(&*raw_only) - 64.0).abs() <= 1.0); +} + +// A compiler must reject invalid physical parameters before starting execution. +#[test] +fn invalid_kll_parameters_are_rejected_at_binding() { + let result = create_planner_accumulator( + &family(0), + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ); + assert!(result.is_err()); +} + +// Native CountSketch supports the confidence-sized depth used by the backend; +// a packed-wire column-bit budget must not be imposed on this constructor. +#[test] +fn native_count_sketch_dimensions_are_not_packed_wire_dimensions() { + use asap_physical_operators::planner::post_asap::SummaryInputExpr; + use asap_physical_operators::KeyByLabelValues; + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 1200, + depth: 55, + heap_size: 3, + }, + ), + Default::default(), + ); + let mut update = SummaryUpdate::column(ColumnRef::SampleValue); + update.item = Some(SummaryInputExpr::Column(ColumnRef::Named("host".into()))); + let mut operator = create_planner_accumulator(&family, &update, &Default::default()).unwrap(); + let key = KeyByLabelValues::new_with_labels(vec!["a".into()]); + operator.update_keyed(&key, 7.0, 1000); + let state = operator.into_accumulator(); + let state = state + .as_any() + .downcast_ref::() + .unwrap(); + assert_eq!(state.query_key(&key), 7.0); +} From fe80d7fd47cd75a6e4568c80f6f0ae26d1fdbcf6 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 20:00:34 +0000 Subject: [PATCH 02/59] feat: add native physical operators and shared DAG runtime Add the execution layer of `asap-physical-operators`: - `plan`, `runtime` and `sources`: physical DAGs, run context, bounded backpressure, cancellation, memory reservations and raw-source scans. - `expressions` and `operators`: relational/scalar operators, windows, temporal panes, current-series snapshots, and summary build, merge and readout. A readout is typed: a `SketchQuery`, or an `ExactReadout` whose counter lookback resolves to the run's evaluation range. Deserialized operators are validated before use. - `readout`: readouts over merged exact summary states. There is no physical planner yet; operators are built directly. Co-Authored-By: Claude Opus 5.5 --- Cargo.lock | 2 + crates/asap-physical-operators/Cargo.toml | 2 + crates/asap-physical-operators/src/error.rs | 3 +- .../src/expressions/arithmetic.rs | 63 ++ .../src/expressions/mod.rs | 348 +++++++++++ .../src/expressions/planner.rs | 549 +++++++++++++++++ crates/asap-physical-operators/src/lib.rs | 11 +- .../src/operators/aggregate/mod.rs | 293 +++++++++ .../src/operators/aggregate/temporal.rs | 300 +++++++++ .../src/operators/aligned_binary.rs | 144 +++++ .../src/operators/common.rs | 75 +++ .../src/operators/current_series.rs | 135 +++++ .../src/operators/filter.rs | 39 ++ .../src/operators/joins/mod.rs | 219 +++++++ .../src/operators/limit.rs | 69 +++ .../src/operators/mod.rs | 316 ++++++++++ .../src/operators/panes.rs | 228 +++++++ .../src/operators/projection.rs | 56 ++ .../src/operators/sort.rs | 172 ++++++ .../src/operators/source.rs | 96 +++ .../src/operators/summary/mod.rs | 567 ++++++++++++++++++ .../src/operators/unchecked.rs | 140 +++++ .../src/operators/vector_binary.rs | 237 ++++++++ .../src/operators/vector_window.rs | 147 +++++ .../asap-physical-operators/src/plan/mod.rs | 160 +++++ .../src/plan/properties.rs | 32 + crates/asap-physical-operators/src/readout.rs | 111 ++++ .../src/runtime/batch_execution.rs | 210 +++++++ .../src/runtime/context.rs | 133 ++++ .../src/runtime/cooperative.rs | 40 ++ .../src/runtime/mod.rs | 272 +++++++++ .../src/runtime/tests.rs | 262 ++++++++ .../src/sources/memory.rs | 44 ++ .../src/sources/mod.rs | 187 ++++++ .../src/summary_kernels/exact.rs | 22 + crates/asap-physical-operators/src/values.rs | 26 +- .../tests/blocking_resources.rs | 242 ++++++++ .../tests/plan_properties.rs | 151 +++++ 38 files changed, 6100 insertions(+), 3 deletions(-) create mode 100644 crates/asap-physical-operators/src/expressions/arithmetic.rs create mode 100644 crates/asap-physical-operators/src/expressions/mod.rs create mode 100644 crates/asap-physical-operators/src/expressions/planner.rs create mode 100644 crates/asap-physical-operators/src/operators/aggregate/mod.rs create mode 100644 crates/asap-physical-operators/src/operators/aggregate/temporal.rs create mode 100644 crates/asap-physical-operators/src/operators/aligned_binary.rs create mode 100644 crates/asap-physical-operators/src/operators/common.rs create mode 100644 crates/asap-physical-operators/src/operators/current_series.rs create mode 100644 crates/asap-physical-operators/src/operators/filter.rs create mode 100644 crates/asap-physical-operators/src/operators/joins/mod.rs create mode 100644 crates/asap-physical-operators/src/operators/limit.rs create mode 100644 crates/asap-physical-operators/src/operators/mod.rs create mode 100644 crates/asap-physical-operators/src/operators/panes.rs create mode 100644 crates/asap-physical-operators/src/operators/projection.rs create mode 100644 crates/asap-physical-operators/src/operators/sort.rs create mode 100644 crates/asap-physical-operators/src/operators/source.rs create mode 100644 crates/asap-physical-operators/src/operators/summary/mod.rs create mode 100644 crates/asap-physical-operators/src/operators/unchecked.rs create mode 100644 crates/asap-physical-operators/src/operators/vector_binary.rs create mode 100644 crates/asap-physical-operators/src/operators/vector_window.rs create mode 100644 crates/asap-physical-operators/src/plan/mod.rs create mode 100644 crates/asap-physical-operators/src/plan/properties.rs create mode 100644 crates/asap-physical-operators/src/readout.rs create mode 100644 crates/asap-physical-operators/src/runtime/batch_execution.rs create mode 100644 crates/asap-physical-operators/src/runtime/context.rs create mode 100644 crates/asap-physical-operators/src/runtime/cooperative.rs create mode 100644 crates/asap-physical-operators/src/runtime/mod.rs create mode 100644 crates/asap-physical-operators/src/runtime/tests.rs create mode 100644 crates/asap-physical-operators/src/sources/memory.rs create mode 100644 crates/asap-physical-operators/src/sources/mod.rs create mode 100644 crates/asap-physical-operators/tests/blocking_resources.rs create mode 100644 crates/asap-physical-operators/tests/plan_properties.rs diff --git a/Cargo.lock b/Cargo.lock index 96a1f63a..a063dd78 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -382,7 +382,9 @@ dependencies = [ "asap-frontend-promql", "asap-types", "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369)", + "futures", "serde", + "serde_json", "thiserror 2.0.18", "tracing", ] diff --git a/crates/asap-physical-operators/Cargo.toml b/crates/asap-physical-operators/Cargo.toml index 51dd6a04..74c172e3 100644 --- a/crates/asap-physical-operators/Cargo.toml +++ b/crates/asap-physical-operators/Cargo.toml @@ -4,9 +4,11 @@ version = "0.1.0" edition = "2021" [dependencies] +futures = "0.3" planner-types = { package = "asap-types", path = "../types" } asap_sketchlib = { git = "https://github.com/ProjectASAP/asap_sketchlib", rev = "5f03ccbd798ed5fec62bdd839bcb331123cab369" } serde = { version = "1", features = ["derive", "rc"] } +serde_json = "1" tracing = "0.1" thiserror = "2" diff --git a/crates/asap-physical-operators/src/error.rs b/crates/asap-physical-operators/src/error.rs index 06ce16e9..afee33d8 100644 --- a/crates/asap-physical-operators/src/error.rs +++ b/crates/asap-physical-operators/src/error.rs @@ -1,3 +1,4 @@ +use crate::plan::NodeId; #[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] pub enum Error { #[error("invalid DAG: {0}")] @@ -6,7 +7,7 @@ pub enum Error { Operator(String), #[error("node {node} ({operation}) failed: {source}")] AtNode { - node: u64, + node: NodeId, operation: String, source: Box, }, diff --git a/crates/asap-physical-operators/src/expressions/arithmetic.rs b/crates/asap-physical-operators/src/expressions/arithmetic.rs new file mode 100644 index 00000000..30e277d4 --- /dev/null +++ b/crates/asap-physical-operators/src/expressions/arithmetic.rs @@ -0,0 +1,63 @@ +//! Float64 arithmetic shared by ASAP execution engines. +//! Preserve IEEE non-finite results; callers own their output policies. + +pub fn evaluate_float64_arithmetic( + operator: &planner_types::pre_asap::ArithmeticOpKind, + left: f64, + right: f64, +) -> f64 { + use planner_types::pre_asap::ArithmeticOpKind::*; + match operator { + Add => left + right, + Sub => left - right, + Mul => left * right, + Div => left / right, + Mod => left % right, + Pow => left.powf(right), + Atan2 => left.atan2(right), + } +} + +/// Execute the Planner binary contract after a deployment has resolved matching rows. +pub fn evaluate_binary( + operator: &planner_types::post_asap::BinaryOperator, + left: f64, + right: f64, +) -> Result { + use crate::{values::Value, Error}; + use planner_types::pre_asap::{ArithmeticOpKind, BinaryOpKind, CompareOpKind}; + let invalid = + || Error::Invalid("unsupported binary operation or invalid checked-division domain".into()); + if operator.vector_match.is_some() { + return Err(invalid()); + } + if operator.checked_relative_division || operator.checked_finite_division { + if operator.kind != BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + || !left.is_finite() + || !right.is_finite() + || right == 0. + { + return Err(invalid()); + } + let value = left / right; + if !value.is_finite() || (operator.checked_relative_division && !value.is_normal()) { + return Err(invalid()); + } + return Ok(Value::Float64(value)); + } + Ok(match operator.kind { + BinaryOpKind::Arithmetic(ref op) => { + Value::Float64(evaluate_float64_arithmetic(op, left, right)) + } + BinaryOpKind::Compare(ref op) => Value::Bool(match op { + CompareOpKind::Eq => left == right, + CompareOpKind::Ne => left != right, + CompareOpKind::Lt => left < right, + CompareOpKind::Le => left <= right, + CompareOpKind::Gt => left > right, + CompareOpKind::Ge => left >= right, + _ => return Err(invalid()), + }), + _ => return Err(invalid()), + }) +} diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs new file mode 100644 index 00000000..2916cca7 --- /dev/null +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -0,0 +1,348 @@ +//! Scalar semantics and typed expression binding. Planner expressions enter through CompiledExpression. +use crate::{ + values::{plain, Schema, Value}, + Error, +}; +use planner_types::pre_asap::{ArithmeticOpKind, DataType}; +pub mod arithmetic; +mod planner; +pub use planner::CompiledExpression; +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] +pub enum Expression { + Binary { + operator: planner_types::post_asap::BinaryOperator, + left: Box, + right: Box, + }, + Planner(Box), + Column(usize), + ExactFloat64(usize), + FiniteFloat64(Box), + LabelSet { + column: usize, + labels: Vec, + without: bool, + }, + Literal { + value: Value, + dtype: DataType, + }, + Negate(Box), + Arithmetic { + op: ArithmeticOpKind, + left: Box, + right: Box, + }, + Equal(Box, Box), + Less(Box, Box), + And(Box, Box), + Or(Box, Box), + Not(Box), + IsNull(Box), +} +impl Expression { + pub fn planner(expression: crate::expressions::CompiledExpression) -> Self { + Self::Planner(Box::new(expression)) + } + pub(crate) fn dtype(&self, input: &Schema) -> Result<(DataType, bool), Error> { + use Expression::*; + match self { + Binary { + operator, + left, + right, + } => { + use planner_types::pre_asap::{BinaryOpKind, CompareOpKind}; + let (a, n) = left.dtype(input)?; + let (b, m) = right.dtype(input)?; + if a != DataType::Float64 || b != a || operator.vector_match.is_some() { + return Err(invalid( + "binary expression requires resolved Float64 operands", + )); + } + if (operator.checked_relative_division || operator.checked_finite_division) + && operator.kind != BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + { + return Err(invalid("checked division contract on non-division")); + } + let dtype = match operator.kind { + BinaryOpKind::Arithmetic(_) => DataType::Float64, + BinaryOpKind::Compare( + CompareOpKind::Eq + | CompareOpKind::Ne + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge, + ) => DataType::Bool, + _ => return Err(invalid("unsupported binary operation")), + }; + Ok((dtype, n || m)) + } + Planner(expression) => { + expression.validate_input(input)?; + Ok(expression.dtype()) + } + FiniteFloat64(expression) => { + if expression.dtype(input)? != (DataType::Float64, false) { + return Err(invalid("finite update requires non-null Float64")); + } + Ok((DataType::Float64, false)) + } + ExactFloat64(column) => { + let (dtype, nullable) = plain(input, *column)?; + if nullable || !matches!(dtype, DataType::Int64 | DataType::Float64) { + return Err(invalid( + "exact Float64 conversion requires non-null numeric input", + )); + } + Ok((DataType::Float64, false)) + } + LabelSet { column, labels, .. } => { + let (dtype, nullable) = plain(input, *column)?; + let expected = DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }; + if dtype != &expected + || nullable + || labels + .iter() + .collect::>() + .len() + != labels.len() + { + return Err(invalid( + "label projection requires a non-null Utf8 map and unique label names", + )); + } + Ok((expected, false)) + } + Column(i) => { + let (t, n) = plain(input, *i)?; + Ok((t.clone(), n)) + } + Literal { value, dtype } => { + if value.matches(dtype, true) { + Ok((dtype.clone(), matches!(value, Value::Null))) + } else { + Err(invalid("literal type mismatch")) + } + } + Negate(v) => { + let (t, n) = v.dtype(input)?; + if matches!(t, DataType::Int64 | DataType::Float64) { + Ok((t, n)) + } else { + Err(invalid("numeric negation required")) + } + } + Arithmetic { op, left, right } => { + let (a, n) = left.dtype(input)?; + let (b, m) = right.dtype(input)?; + if a == b + && matches!(a, DataType::Int64 | DataType::Float64) + && !(a == DataType::Int64 && *op == ArithmeticOpKind::Atan2) + { + Ok((a, n || m)) + } else { + Err(invalid("arithmetic requires matching numeric types")) + } + } + Equal(a, b) | Less(a, b) => { + let (a, n) = a.dtype(input)?; + let (b, m) = b.dtype(input)?; + if a == b && ordered(&a) { + Ok((DataType::Bool, n || m)) + } else { + Err(invalid("comparison requires matching ordered types")) + } + } + And(a, b) | Or(a, b) => { + let (a, n) = a.dtype(input)?; + let (b, m) = b.dtype(input)?; + if a == DataType::Bool && b == DataType::Bool { + Ok((DataType::Bool, n || m)) + } else { + Err(invalid("boolean operands required")) + } + } + Not(v) => { + let (t, n) = v.dtype(input)?; + if t == DataType::Bool { + Ok((t, n)) + } else { + Err(invalid("boolean operand required")) + } + } + IsNull(v) => { + v.dtype(input)?; + Ok((DataType::Bool, false)) + } + } + } + pub(crate) fn evaluate(&self, row: &[Value]) -> Result { + use Expression::*; + Ok(match self { + FiniteFloat64(expression) => match expression.evaluate(row)? { + Value::Float64(value) if value.is_finite() => Value::Float64(value), + _ => return Err(invalid("summary update must be finite")), + }, + ExactFloat64(column) => match row[*column] { + Value::Float64(value) => Value::Float64(value), + Value::Int64(value) if value.unsigned_abs() <= (1u64 << 53) => { + Value::Float64(value as f64) + } + _ => { + return Err(invalid( + "numeric result cannot be represented exactly as Float64", + )) + } + }, + LabelSet { + column, + labels, + without, + } => { + let Value::Map(entries) = &row[*column] else { + return Err(invalid("label projection requires a map")); + }; + let mut selected = std::collections::BTreeMap::new(); + let mut seen = std::collections::BTreeSet::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("label projection requires Utf8 entries")); + }; + if !seen.insert(key.clone()) { + return Err(invalid("duplicate label name")); + } + let keep = if *without { + key.as_ref() != "__name__" + && !labels.iter().any(|label| label.as_str() == key.as_ref()) + } else { + labels.iter().any(|label| label.as_str() == key.as_ref()) + }; + if keep && !value.is_empty() { + selected.insert(key.clone(), value.clone()); + } + } + Value::Map( + selected + .into_iter() + .map(|(k, v)| (Value::Utf8(k), Value::Utf8(v))) + .collect::>() + .into(), + ) + } + Binary { + operator, + left, + right, + } => { + let (a, b) = (left.evaluate(row)?, right.evaluate(row)?); + if matches!(a, Value::Null) || matches!(b, Value::Null) { + Value::Null + } else { + let (Value::Float64(a), Value::Float64(b)) = (a, b) else { + return Err(invalid("binary value schema mismatch")); + }; + arithmetic::evaluate_binary(operator, a, b)? + } + } + Planner(expression) => expression.evaluate(row)?, + Column(i) => row[*i].clone(), + Literal { value, .. } => value.clone(), + Negate(v) => match v.evaluate(row)? { + Value::Int64(v) => Value::Int64( + v.checked_neg() + .ok_or_else(|| invalid("integer negation overflow"))?, + ), + Value::Float64(v) => Value::Float64(-v), + Value::Null => Value::Null, + _ => return Err(invalid("numeric negation required")), + }, + Arithmetic { op, left, right } => { + numeric(op, left.evaluate(row)?, right.evaluate(row)?)? + } + Equal(a, b) | Less(a, b) => { + let (a, b) = (a.evaluate(row)?, b.evaluate(row)?); + if matches!(a, Value::Null) || matches!(b, Value::Null) { + Value::Null + } else if matches!((&a,&b),(Value::Float64(a),Value::Float64(b)) if a.is_nan() || b.is_nan()) + { + Value::Bool(false) + } else { + let c = a.compare(&b)?; + Value::Bool(if matches!(self, Equal(..)) { + c.is_eq() + } else { + c.is_lt() + }) + } + } + And(a, b) | Or(a, b) => { + let (a, b) = (a.evaluate(row)?, b.evaluate(row)?); + match (a, b, matches!(self, And(..))) { + (Value::Bool(false), _, true) | (_, Value::Bool(false), true) => { + Value::Bool(false) + } + (Value::Bool(true), _, false) | (_, Value::Bool(true), false) => { + Value::Bool(true) + } + (Value::Null, _, _) | (_, Value::Null, _) => Value::Null, + (Value::Bool(a), Value::Bool(b), true) => Value::Bool(a && b), + (Value::Bool(a), Value::Bool(b), false) => Value::Bool(a || b), + _ => return Err(invalid("boolean operands required")), + } + } + Not(v) => match v.evaluate(row)? { + Value::Bool(v) => Value::Bool(!v), + Value::Null => Value::Null, + _ => return Err(invalid("boolean operand required")), + }, + IsNull(v) => Value::Bool(matches!(v.evaluate(row)?, Value::Null)), + }) + } +} +pub(crate) fn ordered(dtype: &DataType) -> bool { + if let DataType::Map { key, value, .. } = dtype { + return ordered(key) && ordered(value); + } + matches!( + dtype, + DataType::Null + | DataType::Int64 + | DataType::Float64 + | DataType::Utf8 + | DataType::Bool + | DataType::Timestamp + | DataType::Date + ) +} +pub(crate) fn numeric(op: &ArithmeticOpKind, a: Value, b: Value) -> Result { + use ArithmeticOpKind::*; + Ok(match (a, b) { + (Value::Null, _) | (_, Value::Null) => Value::Null, + (Value::Float64(a), Value::Float64(b)) => { + Value::Float64(arithmetic::evaluate_float64_arithmetic(op, a, b)) + } + (Value::Int64(a), Value::Int64(b)) => Value::Int64( + match op { + Add => a.checked_add(b), + Sub => a.checked_sub(b), + Mul => a.checked_mul(b), + Div => a.checked_div(b), + Mod => a.checked_rem(b), + Pow => u32::try_from(b).ok().and_then(|b| a.checked_pow(b)), + Atan2 => None, + } + .ok_or_else(|| invalid("invalid integer arithmetic or overflow"))?, + ), + _ => return Err(invalid("arithmetic type mismatch")), + }) +} + +fn invalid(message: &str) -> Error { + Error::Invalid(message.into()) +} diff --git a/crates/asap-physical-operators/src/expressions/planner.rs b/crates/asap-physical-operators/src/expressions/planner.rs new file mode 100644 index 00000000..2130a2f7 --- /dev/null +++ b/crates/asap-physical-operators/src/expressions/planner.rs @@ -0,0 +1,549 @@ +//! Planner scalar expressions evaluated over native typed rows. +use crate::{ + values::{Schema, Value}, + Error, +}; +use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind, DataType, QueryExpr, ScalarValue}; +use std::{cmp::Ordering, sync::Arc}; + +pub(super) fn evaluate( + expr: &QueryExpr, + row: &[Value], + schema: &planner_types::pre_asap::Schema, +) -> Result { + match expr { + QueryExpr::Column(index) => row.get(*index).cloned().ok_or(Error::Invalid(format!( + "column {index} outside row width {}", + row.len() + ))), + QueryExpr::Literal(value) => Ok(match value { + ScalarValue::Interval { + months, + days, + nanos, + } => Value::Interval { + months: *months, + days: *days, + nanos: *nanos, + }, + ScalarValue::Int64(value) => Value::Int64(*value), + ScalarValue::Float64(value) => Value::Float64(*value), + ScalarValue::Utf8(value) => Value::Utf8(value.clone().into()), + ScalarValue::Boolean(value) => Value::Bool(*value), + ScalarValue::Null => Value::Null, + }), + QueryExpr::Compare { left, op, right } => { + let left = evaluate(left, row, schema)?; + let right = evaluate(right, row, schema)?; + compare(op, left, right) + } + QueryExpr::Arithmetic { op, left, right } => arithmetic( + op, + evaluate(left, row, schema)?, + evaluate(right, row, schema)?, + ), + QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + let and = matches!(expr, QueryExpr::BoolAnd(_)); + let mut null = false; + for part in parts { + match evaluate(part, row, schema)? { + Value::Bool(value) if value != and => return Ok(Value::Bool(value)), + Value::Bool(_) => {} + Value::Null => null = true, + _ => return Err(Error::Invalid("boolean predicate required".into())), + } + } + Ok(if null { Value::Null } else { Value::Bool(and) }) + } + QueryExpr::Not(value) => match evaluate(value, row, schema)? { + Value::Bool(value) => Ok(Value::Bool(!value)), + Value::Null => Ok(Value::Null), + _ => Err(Error::Invalid("boolean predicate required".into())), + }, + QueryExpr::IsNull(value) => Ok(Value::Bool(matches!( + evaluate(value, row, schema)?, + Value::Null + ))), + QueryExpr::IsNotNull(value) => Ok(Value::Bool(!matches!( + evaluate(value, row, schema)?, + Value::Null + ))), + QueryExpr::FunctionCall { name, args } => { + use planner_types::pre_asap::scalar_signature::MapScalarFunction; + if name.eq_ignore_ascii_case("asap_struct_field") { + expr.scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))?; + let DataType::Struct { fields } = args[0] + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))? + .0 + else { + unreachable!() + }; + let offset = match &args[1] { + QueryExpr::Literal(ScalarValue::Int64(index)) => { + usize::try_from(index - 1).ok() + } + QueryExpr::Literal(ScalarValue::Utf8(name)) => { + fields.iter().position(|field| &field.name == name) + } + _ => None, + } + .ok_or_else(|| Error::Invalid("struct field selector".into()))?; + let Value::Struct(values) = evaluate(&args[0], row, schema)? else { + return Err(Error::Invalid("struct field input".into())); + }; + return values + .get(offset) + .cloned() + .ok_or_else(|| Error::Invalid("struct field value".into())); + } + if name.eq_ignore_ascii_case("asap_element_access") { + let (output_type, _) = expr + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))?; + if let DataType::List { element } = args[0] + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))? + .0 + { + let Value::List(values) = evaluate(&args[0], row, schema)? else { + return Err(Error::Invalid("array access input".into())); + }; + let index = match evaluate(&args[1], row, schema)? { + Value::Null => return Ok(Value::Null), + Value::Int64(index) => index, + _ => return Err(Error::Invalid("array access index".into())), + }; + let offset = if index > 0 { + usize::try_from(index - 1).ok() + } else if index < 0 { + usize::try_from(index.unsigned_abs()) + .ok() + .and_then(|distance| values.len().checked_sub(distance)) + } else { + None + }; + return match offset.and_then(|offset| values.get(offset)) { + Some(value) => Ok(value.clone()), + None => default_collection_element(&output_type, element.nullable), + }; + } + } + let function = (if name.eq_ignore_ascii_case("asap_element_access") { + Some(MapScalarFunction::Access) + } else { + MapScalarFunction::from_name(name) + }) + .ok_or_else(|| Error::Invalid(format!("scalar function {name}")))?; + expr.scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))?; + let values = args + .iter() + .map(|arg| evaluate(arg, row, schema)) + .collect::, _>>()?; + match function { + MapScalarFunction::Construct => { + let mut values = values.into_iter(); + let mut entries = Vec::new(); + while let Some(key) = values.next() { + if !matches!(key, Value::Int64(_) | Value::Utf8(_) | Value::Bool(_)) { + return Err(Error::Invalid("map key value type".into())); + } + entries.push(( + key, + values + .next() + .ok_or_else(|| Error::Invalid("odd map argument count".into()))?, + )); + } + Ok(Value::Map(entries.into())) + } + MapScalarFunction::Concat => { + let mut entries = Vec::new(); + for value in values { + let Value::Map(next) = value else { + return Err(Error::Invalid("map concat argument".into())); + }; + entries.extend(next.iter().cloned()); + } + Ok(Value::Map(entries.into())) + } + MapScalarFunction::Access => { + let [Value::Map(entries), key] = values.as_slice() else { + return Err(Error::Invalid("map access arguments".into())); + }; + if matches!(key, Value::Null) { + return Ok(Value::Null); + } + if !matches!(key, Value::Int64(_) | Value::Utf8(_) | Value::Bool(_)) { + return Err(Error::Invalid("map lookup key type".into())); + } + if let Some((_, value)) = entries + .iter() + .find(|(candidate, _)| cell_cmp(candidate, key) == Some(Ordering::Equal)) + { + return Ok(value.clone()); + } + let ( + DataType::Map { + value, + value_nullable, + .. + }, + _, + ) = args[0] + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))? + else { + unreachable!() + }; + default_collection_element(&value, value_nullable) + } + } + } + other => Err(Error::Invalid(format!("scalar expression {other:?}"))), + } +} + +fn default_collection_element(dtype: &DataType, nullable: bool) -> Result { + if nullable { + return Ok(Value::Null); + } + Ok(match dtype { + DataType::Interval | DataType::Date => { + return Err(Error::Invalid("temporal value transport".into())) + } + DataType::Null => Value::Null, + DataType::Int64 => Value::Int64(0), + DataType::Float64 => Value::Float64(0.0), + DataType::Utf8 => Value::Utf8("".into()), + DataType::Bool => Value::Bool(false), + DataType::Map { .. } => Value::Map(Arc::from([])), + DataType::List { .. } => Value::List(Arc::from([])), + DataType::Struct { fields } => Value::Struct( + fields + .iter() + .map(|field| default_collection_element(&field.dtype, field.nullable)) + .collect::, _>>()? + .into(), + ), + _ => { + return Err(Error::Invalid( + "collection missing-element default type".into(), + )) + } + }) +} + +fn compare(op: &CompareOpKind, left: Value, right: Value) -> Result { + if matches!(left, Value::Null) || matches!(right, Value::Null) { + return Ok(Value::Null); + } + // NaN is unordered, not a type mismatch. Match the native scalar path. + if matches!(&left, Value::Float64(v) if v.is_nan()) + || matches!(&right, Value::Float64(v) if v.is_nan()) + { + return match op { + CompareOpKind::Ne => Ok(Value::Bool(true)), + CompareOpKind::Eq + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge => Ok(Value::Bool(false)), + _ => Err(Error::Invalid(format!("comparison {op:?}"))), + }; + } + let ordering = cell_cmp(&left, &right) + .ok_or_else(|| Error::Invalid("comparison of incompatible values".into()))?; + let value = match op { + CompareOpKind::Eq => ordering == Ordering::Equal, + CompareOpKind::Ne => ordering != Ordering::Equal, + CompareOpKind::Lt => ordering == Ordering::Less, + CompareOpKind::Le => ordering != Ordering::Greater, + CompareOpKind::Gt => ordering == Ordering::Greater, + CompareOpKind::Ge => ordering != Ordering::Less, + _ => return Err(Error::Invalid(format!("comparison {op:?}"))), + }; + Ok(Value::Bool(value)) +} + +fn arithmetic(op: &ArithmeticOpKind, left: Value, right: Value) -> Result { + let (left, right) = match (left, right) { + (Value::Int64(a), Value::Float64(b)) => (Value::Float64(a as f64), Value::Float64(b)), + (Value::Float64(a), Value::Int64(b)) => (Value::Float64(a), Value::Float64(b as f64)), + pair => pair, + }; + super::numeric(op, left, right) +} + +fn integer_float_cmp(integer: i64, float: f64) -> Option { + if float.is_nan() { + return None; + } + // These bounds are powers of two, exactly representable as Float64. + if float >= 9_223_372_036_854_775_808.0 { + return Some(Ordering::Less); + } + if float < -9_223_372_036_854_775_808.0 { + return Some(Ordering::Greater); + } + let integral = float as i64; + match integer.cmp(&integral) { + Ordering::Equal => 0.0_f64.partial_cmp(&float.fract()), + other => Some(other), + } +} + +fn cell_cmp(left: &Value, right: &Value) -> Option { + match (left, right) { + (Value::Int64(left), Value::Int64(right)) => Some(left.cmp(right)), + (Value::Float64(left), Value::Float64(right)) => left.partial_cmp(right), + (Value::Int64(left), Value::Float64(right)) => integer_float_cmp(*left, *right), + (Value::Float64(left), Value::Int64(right)) => { + integer_float_cmp(*right, *left).map(Ordering::reverse) + } + (Value::Utf8(left), Value::Utf8(right)) => Some(left.cmp(right)), + (Value::Bool(left), Value::Bool(right)) => Some(left.cmp(right)), + (Value::Timestamp(left), Value::Timestamp(right)) => Some(left.cmp(right)), + (Value::Map(left), Value::Map(right)) => { + for ((left_key, left_value), (right_key, right_value)) in left.iter().zip(right.iter()) + { + let order = cell_cmp(left_key, right_key)?; + if order != Ordering::Equal { + return Some(order); + } + let order = match (left_value, right_value) { + (Value::Null, Value::Null) => Ordering::Equal, + (Value::Null, _) => Ordering::Greater, + (_, Value::Null) => Ordering::Less, + _ => cell_cmp(left_value, right_value)?, + }; + if order != Ordering::Equal { + return Some(order); + } + } + Some(left.len().cmp(&right.len())) + } + _ => None, + } +} + +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] +pub struct CompiledExpression { + expression: QueryExpr, + schema: planner_types::pre_asap::Schema, + output: (DataType, bool), +} +impl CompiledExpression { + pub(crate) fn expression(&self) -> &QueryExpr { + &self.expression + } + + pub fn compile(expression: &QueryExpr, input: &Schema) -> Result { + let schema = input + .fields + .iter() + .map(|field| { + let planner_types::post_asap::SummaryFamilyType::Plain(dtype) = &field.dtype else { + return Err(Error::Invalid( + "scalar expression cannot consume opaque summary state".into(), + )); + }; + Ok(planner_types::pre_asap::Column::new( + field.name.clone(), + dtype.clone(), + field.nullable, + )) + }) + .collect::, Error>>()?; + let schema = planner_types::pre_asap::Schema::new(schema); + validate(expression, &schema)?; + let output = expression + .scalar_type(&schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + Ok(Self { + expression: expression.clone(), + schema, + output, + }) + } + pub(crate) fn dtype(&self) -> (DataType, bool) { + self.output.clone() + } + pub(crate) fn validate_input(&self, input: &Schema) -> Result<(), Error> { + let checked = Self::compile(&self.expression, input)?; + if checked.output != self.output { + return Err(Error::Invalid( + "persisted expression type differs from its semantics".into(), + )); + } + if input.fields.len() != self.schema.columns.len() + || input + .fields + .iter() + .zip(&self.schema.columns) + .any(|(field, column)| { + field.dtype + != planner_types::post_asap::SummaryFamilyType::Plain(column.dtype.clone()) + || field.nullable != column.nullable + }) + { + return Err(Error::Invalid( + "expression input differs from its bound schema".into(), + )); + } + Ok(()) + } + /// Evaluate a row under the same typed schema used when binding the expression. + pub fn evaluate(&self, row: &[Value]) -> Result { + if row.len() != self.schema.columns.len() + || row + .iter() + .zip(&self.schema.columns) + .any(|(value, column)| !value.matches(&column.dtype, column.nullable)) + { + return Err(Error::Invalid( + "expression input differs from its bound schema".into(), + )); + } + evaluate(&self.expression, row, &self.schema) + } +} +fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Result<(), Error> { + let invalid = || Error::Invalid(format!("unsupported scalar expression: {expr:?}")); + expr.scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + match expr { + QueryExpr::Column(_) | QueryExpr::Literal(_) => Ok(()), + QueryExpr::Arithmetic { left, right, .. } => { + for value in [left, right] { + validate(value, schema)?; + if !matches!( + value + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0, + DataType::Int64 | DataType::Float64 | DataType::Null + ) { + return Err(invalid()); + } + } + Ok(()) + } + QueryExpr::Compare { left, right, op } => { + if !matches!( + op, + CompareOpKind::Eq + | CompareOpKind::Ne + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge + ) { + return Err(invalid()); + } + validate(left, schema)?; + validate(right, schema)?; + let (a, _) = left + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + let (b, _) = right + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + fn comparable(dtype: &DataType) -> bool { + match dtype { + DataType::Null + | DataType::Int64 + | DataType::Float64 + | DataType::Utf8 + | DataType::Bool + | DataType::Timestamp => true, + DataType::Map { key, value, .. } => comparable(key) && comparable(value), + _ => false, + } + } + let numeric = |dtype: &DataType| matches!(dtype, DataType::Int64 | DataType::Float64); + if !comparable(&a) + || !comparable(&b) + || (a != b + && !matches!(a, DataType::Null) + && !matches!(b, DataType::Null) + && !(numeric(&a) && numeric(&b))) + { + return Err(invalid()); + } + Ok(()) + } + QueryExpr::FunctionCall { name, args } => { + if name != "asap_struct_field" + && name != "asap_element_access" + && planner_types::pre_asap::scalar_signature::MapScalarFunction::from_name(name) + .is_none() + { + return Err(invalid()); + } + for arg in args { + validate(arg, schema)?; + } + Ok(()) + } + QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + for part in parts { + validate(part, schema)?; + if !matches!( + part.scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0, + DataType::Bool | DataType::Null + ) { + return Err(invalid()); + } + } + Ok(()) + } + QueryExpr::Not(value) => { + validate(value, schema)?; + if !matches!( + value + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0, + DataType::Bool | DataType::Null + ) { + return Err(invalid()); + } + Ok(()) + } + QueryExpr::IsNull(value) | QueryExpr::IsNotNull(value) => validate(value, schema), + _ => Err(invalid()), + } +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn mixed_comparison_preserves_integer_precision_and_boundaries() { + assert_eq!( + integer_float_cmp(9_007_199_254_740_993, 9_007_199_254_740_992.0), + Some(Ordering::Greater) + ); + assert_eq!( + integer_float_cmp(i64::MAX, 9_223_372_036_854_775_808.0), + Some(Ordering::Less) + ); + assert_eq!( + integer_float_cmp(i64::MIN, -9_223_372_036_854_775_808.0), + Some(Ordering::Equal) + ); + assert_eq!(integer_float_cmp(-1, -1.5), Some(Ordering::Greater)); + assert_eq!(integer_float_cmp(1, 1.5), Some(Ordering::Less)); + assert_eq!(integer_float_cmp(0, f64::INFINITY), Some(Ordering::Less)); + assert_eq!( + integer_float_cmp(0, f64::NEG_INFINITY), + Some(Ordering::Greater) + ); + assert_eq!(integer_float_cmp(0, f64::NAN), None); + } +} diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs index 5026cb6d..0835159c 100644 --- a/crates/asap-physical-operators/src/lib.rs +++ b/crates/asap-physical-operators/src/lib.rs @@ -1,4 +1,5 @@ -//! Shared physical operators. This layer holds summary kernels and typed values. +//! Shared physical operators: summary kernels, typed values, native operators +//! and the DAG runtime. pub mod key_by_label_values; pub mod measurement; @@ -20,3 +21,11 @@ pub use planner_types as planner; mod error; pub use error::Error; pub mod values; + +pub use expressions::arithmetic; +pub mod expressions; +pub mod operators; +pub mod plan; +pub mod readout; +pub mod runtime; +pub mod sources; diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs new file mode 100644 index 00000000..71e7134c --- /dev/null +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -0,0 +1,293 @@ +use super::*; +impl Operator { + pub fn aggregate( + input: Schema, + groups: Vec, + measures: Vec<(String, Reduction)>, + ) -> Result { + validate_groups(&input, &groups)?; + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + for (name, reduction) in &measures { + let (t, n) = match reduction { + Reduction::Count => (DataType::Int64, false), + Reduction::Sum(i) | Reduction::Avg(i) => { + let (t, _) = plain(&input, *i)?; + if !matches!(t, DataType::Int64 | DataType::Float64) { + return Err(invalid("numeric aggregate input required")); + } + ( + if matches!(reduction, Reduction::Avg(_)) { + DataType::Float64 + } else { + t.clone() + }, + false, + ) + } + Reduction::Min(i) | Reduction::Max(i) => { + let (t, nullable) = plain(&input, *i)?; + if !ordered(t) { + return Err(invalid("ordered aggregate input required")); + } + (t.clone(), nullable || groups.is_empty()) + } + }; + fields.push(result_field(name, t, n)); + } + Ok(Self { + kind: Kind::Aggregate { + groups, + measures: measures.into_iter().map(|(_, r)| r).collect(), + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn window( + input: Schema, + intent: planner_types::pre_asap::AggIntent, + coordinate: usize, + value: usize, + groups: Vec, + window: Option<(i64, i64)>, + ) -> Result { + use planner_types::pre_asap::AggIntent; + validate_groups(&input, &groups)?; + let histogram = matches!(intent, AggIntent::HistogramQuantile { .. }); + if !matches!( + intent, + AggIntent::Rate + | AggIntent::Increase + | AggIntent::Count { .. } + | AggIntent::Sum { col: None } + | AggIntent::Avg { col: None } + | AggIntent::Min { col: None } + | AggIntent::Max { col: None } + | AggIntent::HistogramQuantile { .. } + ) { + return Err(invalid( + "unsupported temporal intent or unresolved value column", + )); + } + let coordinate_type = if histogram { + DataType::Float64 + } else { + DataType::Timestamp + }; + if plain(&input, coordinate)? != (&coordinate_type, false) + || plain(&input, value)? != (&DataType::Float64, false) + { + return Err(invalid("window coordinate/value schema mismatch")); + } + if (!histogram && !matches!(window, Some((start, end)) if start < end)) + || (histogram && window.is_some()) + { + return Err(invalid("invalid temporal window")); + } + let mut fields = groups + .iter() + .map(|i| input.fields[*i].clone()) + .collect::>(); + fields.push(result_field( + "value", + if matches!(intent, AggIntent::Count { .. }) { + DataType::Int64 + } else { + DataType::Float64 + }, + false, + )); + Ok(Self { + kind: Kind::Window { + intent: Box::new(intent), + coordinate, + value, + groups, + window, + }, + inputs: vec![input], + output: schema(fields), + }) + } +} +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] +pub enum Reduction { + Count, + Sum(usize), + Avg(usize), + Min(usize), + Max(usize), +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &operator.kind { + Kind::Window { + intent, + coordinate, + value, + groups, + window, + } => { + crate::operators::aggregate::temporal::reduce( + rows, + intent, + groups, + *coordinate, + *value, + *window, + &context, + ) + .await? + } + Kind::Aggregate { groups, measures } => { + reduce(rows, groups, measures, &operator.inputs[0], &context).await? + } + _ => unreachable!(), + }; + Batch::try_new(output, result) + }) + .boxed_local()) +} + +pub(super) mod temporal; +async fn reduce( + rows: Vec>, + groups: &[usize], + measures: &[Reduction], + input: &Schema, + context: &RunContext, +) -> Result>, Error> { + let mut work = Cooperative::new(context); + let mut workspace = Workspace::new(context)?; + let mut grouped = BTreeMap::>, Vec>>::new(); + if rows.is_empty() && groups.is_empty() { + grouped.insert(vec![], vec![]); + } + for row in rows { + work.checkpoint().await?; + let key = group_key(&row, groups)?; + workspace.grow(std::mem::size_of::>())?; + if !grouped.contains_key(&key) { + workspace.grow(key_bytes(&key))?; + } + grouped.entry(key).or_default().push(row); + } + let mut output = Vec::new(); + for rows in grouped.into_values() { + work.checkpoint().await?; + let mut result = groups + .iter() + .map(|&i| rows[0][i].clone()) + .collect::>(); + for measure in measures { + result.push(reduce_one(&rows, measure, input, &mut work).await?); + } + workspace.grow(row_bytes(&result))?; + output.push(result); + } + Ok(output) +} + +async fn reduce_one( + rows: &[Vec], + measure: &Reduction, + input: &Schema, + work: &mut Cooperative, +) -> Result { + let column = match measure { + Reduction::Count => { + return Ok(Value::Int64( + i64::try_from(rows.len()).map_err(|_| invalid("count overflow"))?, + )) + } + Reduction::Sum(i) | Reduction::Avg(i) | Reduction::Min(i) | Reduction::Max(i) => *i, + }; + let values = rows + .iter() + .map(|r| &r[column]) + .filter(|v| !matches!(v, Value::Null)); + if matches!(measure, Reduction::Min(_) | Reduction::Max(_)) { + if plain(input, column)?.0 == &DataType::Float64 { + // Match exact-state kernels: ignore NaN when a numeric value exists. + let mut best: Option = None; + for value in values { + work.checkpoint().await?; + let Value::Float64(value) = value else { + return Err(invalid("floating aggregate value required")); + }; + best = Some(best.map_or(*value, |old| { + if matches!(measure, Reduction::Min(_)) { + old.min(*value) + } else { + old.max(*value) + } + })); + } + return Ok(best.map(Value::Float64).unwrap_or(Value::Null)); + } + let mut best: Option<&Value> = None; + for value in values { + work.checkpoint().await?; + if best + .map(|b| value.compare(b)) + .transpose()? + .is_none_or(|order| { + if matches!(measure, Reduction::Min(_)) { + order.is_lt() + } else { + order.is_gt() + } + }) + { + best = Some(value); + } + } + return Ok(best.cloned().unwrap_or(Value::Null)); + } + let mut count = 0usize; + let dtype = plain(input, column)?.0; + if dtype == &DataType::Int64 { + let mut sum = 0i128; + for v in values { + work.checkpoint().await?; + let Value::Int64(v) = v else { + return Err(invalid("integer aggregate value required")); + }; + sum = sum + .checked_add(i128::from(*v)) + .ok_or_else(|| invalid("integer aggregate overflow"))?; + count += 1; + } + return if matches!(measure, Reduction::Avg(_)) { + Ok(Value::Float64(sum as f64 / count as f64)) + } else { + Ok(Value::Int64( + i64::try_from(sum).map_err(|_| invalid("integer sum overflow"))?, + )) + }; + } + let mut sum = -0.0; + for v in values { + work.checkpoint().await?; + let Value::Float64(v) = v else { + return Err(invalid("floating aggregate value required")); + }; + sum += v; + count += 1; + } + Ok(Value::Float64(if matches!(measure, Reduction::Avg(_)) { + sum / count as f64 + } else { + sum + })) +} diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs new file mode 100644 index 00000000..dbf66108 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -0,0 +1,300 @@ +//! Windowed computations use Planner intents; deployments supply the input window. +use crate::{ + operators::{ + common::{key_bytes, row_bytes, Workspace}, + sort::cooperative_sort, + }, + runtime::{Cooperative, RunContext}, +}; +use crate::{ + values::{group_key, Value}, + Error, +}; +use planner_types::pre_asap::{AggIntent, ColumnRef}; +use std::collections::BTreeMap; + +pub(in crate::operators) async fn reduce( + rows: Vec>, + intent: &AggIntent, + groups: &[usize], + coordinate: usize, + value: usize, + window: Option<(i64, i64)>, + context: &RunContext, +) -> Result>, Error> { + let mut work = Cooperative::new(context); + let mut workspace = Workspace::new(context)?; + let mut grouped = + BTreeMap::>, (Vec, Vec<(f64, f64)>, Vec<(i64, f64)>)>::new(); + for row in rows { + work.checkpoint().await?; + let key = group_key(&row, groups)?; + workspace.grow(32)?; + if !grouped.contains_key(&key) { + workspace.grow(key_bytes(&key) + row_bytes(&row))?; + } + let entry = grouped.entry(key).or_insert_with(|| { + ( + groups.iter().map(|i| row[*i].clone()).collect(), + vec![], + vec![], + ) + }); + let Value::Float64(v) = row[value] else { + return Err(Error::Invalid("window value must be Float64".into())); + }; + match row[coordinate] { + Value::Timestamp(t) => entry.2.push((t, v)), + Value::Float64(bound) => entry.1.push((bound, v)), + _ => return Err(Error::Invalid("invalid window coordinate".into())), + } + } + let mut output = Vec::new(); + for (_, (mut keys, buckets, points)) in grouped { + work.checkpoint().await?; + let result = if let AggIntent::HistogramQuantile { q } = intent { + Some(Value::Float64(bucket_quantile(*q, buckets, context).await?)) + } else { + let points = cooperative_sort(points, |a, b| a.0.cmp(&b.0), context).await?; + let (start, end) = + window.ok_or_else(|| Error::Invalid("missing temporal window".into()))?; + if points.iter().any(|p| p.0 < start || p.0 > end) + || points.windows(2).any(|p| p[0].0 == p[1].0) + { + return Err(Error::Invalid( + "duplicate or out-of-window timestamp".into(), + )); + } + match intent { + AggIntent::Rate => rate(&points, start, end).map(Value::Float64), + AggIntent::Increase => rate(&points, start, end) + .map(|v| Value::Float64(v * (end as f64 - start as f64) / 1000.)), + AggIntent::Count { .. } => Some(Value::Int64( + i64::try_from(points.len()) + .map_err(|_| Error::Invalid("count overflow".into()))?, + )), + AggIntent::Sum { .. } => Some(Value::Float64(points.iter().map(|p| p.1).sum())), + AggIntent::Avg { .. } => Some(Value::Float64( + points.iter().map(|p| p.1).sum::() / points.len() as f64, + )), + AggIntent::Min { .. } => { + Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { + if a.is_nan() || p.1 < a { + p.1 + } else { + a + } + }))) + } + AggIntent::Max { .. } => { + Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { + if a.is_nan() || p.1 > a { + p.1 + } else { + a + } + }))) + } + _ => return Err(Error::Invalid("unsupported temporal intent".into())), + } + }; + if let Some(result) = result { + keys.push(result); + output.push(keys); + } + } + Ok(output) +} + +fn rate(points: &[(i64, f64)], start: i64, end: i64) -> Option { + if points.len() < 2 { + return None; + } + let (first_t, first) = points[0]; + let (last_t, last) = *points.last()?; + let span = (last_t as f64 - first_t as f64) / 1000.; + if span <= 0. { + return None; + } + let mut delta = last - first; + for pair in points.windows(2) { + if pair[1].1 < pair[0].1 { + delta += pair[0].1; + } + } + let average = span / (points.len() - 1) as f64; + let mut to_start = (first_t as f64 - start as f64) / 1000.; + let mut to_end = (end as f64 - last_t as f64) / 1000.; + if to_start >= average * 1.1 { + to_start = average / 2.; + } + // Apply the zero bound after the sparse-window half-interval cap. + if delta > 0. && first >= 0. { + to_start = to_start.min(span * first / delta); + } + if to_end >= average * 1.1 { + to_end = average / 2.; + } + Some(delta * (span + to_start + to_end) / span / ((end as f64 - start as f64) / 1000.)) +} + +async fn bucket_quantile( + q: f64, + mut b: Vec<(f64, f64)>, + context: &RunContext, +) -> Result { + let mut work = Cooperative::new(context); + let _scratch = context.reserve(b.len().checked_mul(16).ok_or(Error::MemoryLimit)?)?; + if q.is_nan() { + return Ok(f64::NAN); + } + if q < 0. { + return Ok(f64::NEG_INFINITY); + } + if q > 1. { + return Ok(f64::INFINITY); + } + b.retain(|p| !p.0.is_nan()); + b = cooperative_sort(b, |a, b| a.0.total_cmp(&b.0), context).await?; + let mut buckets: Vec<(f64, f64)> = Vec::new(); + for p in b { + work.checkpoint().await?; + if let Some(last) = buckets.last_mut() { + if last.0 == p.0 { + last.1 += p.1; + continue; + } + } + buckets.push(p); + } + if buckets.len() < 2 || buckets.last().unwrap().0 != f64::INFINITY { + return Ok(f64::NAN); + } + let mut prev = buckets[0].1; + for p in buckets.iter_mut().skip(1) { + work.checkpoint().await?; + if p.1 < prev || (p.1 - prev).abs() <= 1e-12 * (p.1.abs() + prev.abs()) { + p.1 = prev; + } + prev = p.1; + } + let count = buckets.last().unwrap().1; + if count == 0. { + return Ok(f64::NAN); + } + let rank = q * count; + let idx = buckets[..buckets.len() - 1].partition_point(|p| p.1 < rank); + if idx == buckets.len() - 1 { + return Ok(buckets[idx - 1].0); + } + if idx == 0 && buckets[0].0 <= 0. { + return Ok(buckets[0].0); + } + let (start, base) = if idx == 0 { (0., 0.) } else { buckets[idx - 1] }; + let (end, upper) = buckets[idx]; + Ok(start + (end - start) * (rank - base) / (upper - base)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::{ + operators::Operator, + runtime::{batch_execution::evaluate_batch, Limits, RunContext, Scope}, + values::Batch, + }; + use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, + types::AccuracyTarget, + }; + use std::sync::Arc; + + // The same window operator must give the same answer in either engine phase. + #[test] + fn temporal_windows_execute_in_both_phases_and_count_is_integer() { + let schema = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "time".into(), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), + nullable: false, + }, + SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }, + ], + time_index: Some(0), + }); + for scope in [ + Scope::Query { + evaluation_time_ms: 2000, + revision: 1, + }, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }, + ] { + for (intent, expected) in [ + (AggIntent::Rate, Value::Float64(2.)), + (AggIntent::Increase, Value::Float64(4.)), + ( + AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }, + Value::Int64(3), + ), + ] { + let batch = Batch::try_new( + schema.clone(), + vec![ + vec![Value::Timestamp(0), Value::Float64(2.)], + vec![Value::Timestamp(1000), Value::Float64(4.)], + vec![Value::Timestamp(2000), Value::Float64(2.)], + ], + ) + .unwrap(); + let operator = + Operator::window(schema.clone(), intent, 0, 1, vec![], Some((0, 2000))) + .unwrap(); + let result = evaluate_batch( + batch, + vec![operator], + RunContext::new(scope.clone(), Limits::default()).unwrap(), + ) + .unwrap(); + assert_eq!( + format!("{:?}", result[0].rows()[0][0]), + format!("{expected:?}") + ); + } + } + assert!(Operator::window(schema, AggIntent::Rate, 0, 1, vec![], Some((1, 1))).is_err()); + } + + // Histogram interpolation requires an infinite terminal bucket and coalesces duplicates. + #[test] + fn histogram_boundaries_and_duplicate_buckets() { + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + let bucket_quantile = |q, buckets| { + futures::executor::block_on(super::bucket_quantile(q, buckets, &context)).unwrap() + }; + assert_eq!( + bucket_quantile(0.5, vec![(1., 1.), (1., 1.), (2., 4.), (f64::INFINITY, 4.)]), + 1. + ); + assert!(bucket_quantile(0.5, vec![(1., 2.), (2., 4.)]).is_nan()); + assert_eq!(bucket_quantile(-0.1, vec![]), f64::NEG_INFINITY); + } +} diff --git a/crates/asap-physical-operators/src/operators/aligned_binary.rs b/crates/asap-physical-operators/src/operators/aligned_binary.rs new file mode 100644 index 00000000..941a9f61 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/aligned_binary.rs @@ -0,0 +1,144 @@ +//! Arithmetic on complete, aligned population/window rows used by precomputation. +use super::*; +use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; +use std::collections::BTreeSet; + +impl Operator { + /// Match every row by the declared identity columns. Unlike an inner join, + /// incomplete or duplicate keys are errors: dropping an update changes state. + pub fn aligned_binary( + left: Schema, + right: Schema, + keys: Vec<(usize, usize)>, + values: (usize, usize), + operator: BinaryOperator, + ) -> Result { + if keys.is_empty() + || !matches!(operator.kind, BinaryOpKind::Arithmetic(_)) + || operator.vector_match.is_some() + { + return Err(invalid( + "aligned arithmetic requires explicit keys and arithmetic semantics", + )); + } + for (input, value) in [(&left, values.0), (&right, values.1)] { + if input.fields.get(value).is_none_or(|f| { + f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Float64) + }) { + return Err(invalid( + "aligned arithmetic requires non-null Float64 values", + )); + } + } + let mut left_keys = BTreeSet::new(); + let mut right_keys = BTreeSet::new(); + for &(l, r) in &keys { + if l == values.0 + || r == values.1 + || !left_keys.insert(l) + || !right_keys.insert(r) + || left + .fields + .get(l) + .zip(right.fields.get(r)) + .is_none_or(|(l, r)| l.nullable || r.nullable || l.dtype != r.dtype) + { + return Err(invalid("invalid aligned arithmetic keys")); + } + } + if left_keys.len() + 1 != left.fields.len() || right_keys.len() + 1 != right.fields.len() { + return Err(invalid( + "aligned arithmetic must account for every input column", + )); + } + Ok(Self { + output: left.clone(), + inputs: vec![left, right], + kind: Kind::AlignedBinary { + keys, + values, + operator, + }, + }) + } +} + +pub(super) fn execute<'a>( + op: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::AlignedBinary { + keys, + values, + operator, + } = &op.kind + else { + unreachable!() + }; + let right = inputs + .pop() + .ok_or_else(|| invalid("missing aligned right input"))?; + let left = inputs + .pop() + .ok_or_else(|| invalid("missing aligned left input"))?; + Ok(futures::stream::once(async move { + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + if left.is_empty() || left.len() != right.len() { + return Err(invalid( + "aligned arithmetic requires matching nonempty key sets", + )); + } + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + let columns = |side: bool| { + keys.iter() + .map(|&(l, r)| if side { r } else { l }) + .collect::>() + }; + let left_columns = columns(false); + let right_columns = columns(true); + let mut indexed = BTreeMap::new(); + for row in right { + work.checkpoint().await?; + let key = group_key(&row, &right_columns)?; + let Value::Float64(value) = row[values.1] else { + return Err(invalid("invalid aligned value")); + }; + if !value.is_finite() { + return Err(invalid("aligned arithmetic input is non-finite")); + } + workspace.grow(key_bytes(&key) + 64)?; + if indexed.insert(key, value).is_some() { + return Err(invalid("aligned arithmetic input has duplicate keys")); + } + } + let mut rows = Vec::new(); + for mut row in left { + work.checkpoint().await?; + let key = group_key(&row, &left_columns)?; + let right = indexed + .remove(&key) + .ok_or_else(|| invalid("aligned arithmetic input has missing or duplicate keys"))?; + let Value::Float64(left) = row[values.0] else { + return Err(invalid("invalid aligned value")); + }; + if !left.is_finite() { + return Err(invalid("aligned arithmetic input is non-finite")); + } + let result = crate::expressions::arithmetic::evaluate_binary(operator, left, right)?; + if !matches!(result, Value::Float64(value) if value.is_finite()) { + return Err(invalid("aligned arithmetic produced a non-finite update")); + } + row[values.0] = result; + workspace.grow(std::mem::size_of::>())?; + rows.push(row); + } + if !indexed.is_empty() { + return Err(invalid("aligned arithmetic has unmatched input keys")); + } + Batch::try_new(op.output.clone(), rows) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/common.rs b/crates/asap-physical-operators/src/operators/common.rs new file mode 100644 index 00000000..9f1709da --- /dev/null +++ b/crates/asap-physical-operators/src/operators/common.rs @@ -0,0 +1,75 @@ +use super::*; +pub(super) fn invalid(message: &str) -> Error { + Error::Invalid(message.into()) +} +pub(super) fn schema(fields: Vec) -> Schema { + Arc::new(SummarySchema { + fields, + time_index: None, + }) +} +pub(super) fn result_field(name: &str, dtype: DataType, nullable: bool) -> SummaryField { + SummaryField { + name: name.into(), + dtype: SummaryFamilyType::Plain(dtype), + nullable, + } +} + +pub(super) fn validate_groups(input: &Schema, groups: &[usize]) -> Result<(), Error> { + for &i in groups { + plain(input, i)?; + } + if groups + .iter() + .collect::>() + .len() + != groups.len() + { + return Err(invalid("duplicate group columns")); + } + Ok(()) +} +pub(super) async fn collect_rows( + mut input: Input<'_, Batch>, + context: &RunContext, +) -> Result<(Vec>, Vec), Error> { + let mut rows = Vec::new(); + let mut work = Cooperative::new(context); + let mut reservations = Vec::new(); + while let Some(batch) = input.next().await { + let batch = batch?; + reservations.push(context.reserve(batch.bytes())?); + for row in batch.rows() { + work.checkpoint().await?; + rows.push(row.clone()); + } + } + Ok((rows, reservations)) +} +/// Estimates retained workspace before growing collections. It is not an RSS limit. +pub(super) struct Workspace { + reservation: Reservation, + bytes: usize, +} +impl Workspace { + pub(super) fn new(context: &RunContext) -> Result { + Ok(Self { + reservation: context.reserve(0)?, + bytes: 0, + }) + } + pub(super) fn grow(&mut self, bytes: usize) -> Result<(), Error> { + self.bytes = self.bytes.checked_add(bytes).ok_or(Error::MemoryLimit)?; + self.reservation.resize(self.bytes) + } +} +pub(super) fn row_bytes(row: &[Value]) -> usize { + std::mem::size_of::>() + row.iter().map(Value::bytes).sum::() +} +pub(super) fn key_bytes(key: &[Vec]) -> usize { + 64 + key + .iter() + .map(|part| std::mem::size_of::>() + part.len()) + .sum::() +} diff --git a/crates/asap-physical-operators/src/operators/current_series.rs b/crates/asap-physical-operators/src/operators/current_series.rs new file mode 100644 index 00000000..937036c5 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/current_series.rs @@ -0,0 +1,135 @@ +//! A bounded instant-vector snapshot: select latest before removing stale markers. +use super::*; + +impl Operator { + pub fn current_series( + input: Schema, + identity: usize, + coordinate: usize, + value: usize, + lookback_ms: i64, + ) -> Result { + if lookback_ms <= 0 + || plain(&input, identity)? != (&DataType::Utf8, false) + || plain(&input, coordinate)? != (&DataType::Timestamp, false) + || plain(&input, value)? != (&DataType::Float64, false) + { + return Err(invalid("invalid current-series input contract")); + } + Ok(Self { + kind: Kind::CurrentSeries { + identity, + coordinate, + value, + lookback_ms, + }, + inputs: vec![input.clone()], + output: input, + }) + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::CurrentSeries { + identity, + coordinate, + value, + lookback_ms, + } = operator.kind + else { + unreachable!() + }; + let input = inputs + .pop() + .ok_or_else(|| invalid("current-series input missing"))?; + let output = operator.output.clone(); + let (start, end) = window(lookback_ms, &context)?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let mut latest = BTreeMap::, usize>::new(); + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + for (index, row) in rows.iter().enumerate() { + work.checkpoint().await?; + let Value::Timestamp(timestamp) = row[coordinate] else { + unreachable!() + }; + if timestamp <= start || timestamp > end { + continue; + } + let key = row[identity].key()?; + if let Some(&previous) = latest.get(&key) { + let Value::Timestamp(previous_time) = rows[previous][coordinate] else { + unreachable!() + }; + if timestamp < previous_time { + continue; + } + if timestamp == previous_time { + let (Value::Float64(a), Value::Float64(b)) = + (&row[value], &rows[previous][value]) + else { + unreachable!() + }; + if a.to_bits() != b.to_bits() { + return Err(invalid("conflicting samples for one series timestamp")); + } + continue; + } + } else { + workspace.grow(64 + key.len())?; + } + latest.insert(key, index); + } + let mut result = Vec::new(); + for index in latest.into_values() { + work.checkpoint().await?; + let Value::Float64(sample) = rows[index][value] else { + unreachable!() + }; + if sample.to_bits() == 0x7ff0_0000_0000_0002 { + continue; + } + workspace.grow(row_bytes(&rows[index]))?; + let mut row = rows[index].clone(); + row[coordinate] = Value::Timestamp(end); + result.push(row); + } + Batch::try_new(output, result) + }) + .boxed_local()) +} + +fn window(lookback_ms: i64, context: &RunContext) -> Result<(i64, i64), Error> { + let end = match context.scope { + crate::runtime::Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + crate::runtime::Scope::Ingestion { window_end_ms, .. } => window_end_ms, + }; + let start = end + .checked_sub(lookback_ms) + .ok_or_else(|| invalid("current-series window overflows"))?; + if let crate::runtime::Scope::Ingestion { + window_start_ms, .. + } = context.scope + { + if window_start_ms != start { + return Err(invalid( + "current-series maintenance window differs from lookback", + )); + } + } + Ok((start, end)) +} + +pub(super) fn validate_context(operator: &Operator, context: &RunContext) -> Result<(), Error> { + if let Kind::CurrentSeries { lookback_ms, .. } = operator.kind { + window(lookback_ms, context)?; + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/operators/filter.rs b/crates/asap-physical-operators/src/operators/filter.rs new file mode 100644 index 00000000..8862aacc --- /dev/null +++ b/crates/asap-physical-operators/src/operators/filter.rs @@ -0,0 +1,39 @@ +use super::*; +impl Operator { + pub fn filter(input: Schema, predicate: Expression) -> Result { + if predicate.dtype(&input)?.0 != DataType::Bool { + return Err(invalid("filter predicate must be boolean")); + } + Ok(Self { + kind: Kind::Filter(predicate), + inputs: vec![input.clone()], + output: input, + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::Filter(predicate) => Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + if matches!(predicate.evaluate(row)?, Value::Bool(true)) { + rows.push(row.clone()); + } + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/joins/mod.rs b/crates/asap-physical-operators/src/operators/joins/mod.rs new file mode 100644 index 00000000..608fd8d7 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/joins/mod.rs @@ -0,0 +1,219 @@ +use super::*; +impl Operator { + pub fn semi_join( + left: Schema, + right: Schema, + keys: Vec<(usize, usize)>, + ) -> Result { + if keys.is_empty() { + return Err(invalid("semi-join needs matching keys")); + } + for &(l, r) in &keys { + if plain(&left, l)?.0 != plain(&right, r)?.0 { + return Err(invalid("join key types differ")); + } + } + Ok(Self { + kind: Kind::SemiJoin { + keys, + require_complete_right: false, + }, + inputs: vec![left.clone(), right], + output: left, + }) + } + pub(crate) fn require_complete_right(mut self) -> Self { + if let Kind::SemiJoin { + require_complete_right, + .. + } = &mut self.kind + { + *require_complete_right = true; + } + self + } + pub fn relational_join( + left: Schema, + right: Schema, + kind: planner_types::pre_asap::JoinKind, + predicate: &planner_types::pre_asap::Predicate, + output: Schema, + ) -> Result { + use planner_types::pre_asap::JoinKind; + let mut joined = left.fields.clone(); + joined.extend(right.fields.clone()); + let predicate = + crate::expressions::CompiledExpression::compile(&predicate.0, &schema(joined.clone()))?; + if predicate.dtype().0 != DataType::Bool { + return Err(invalid("join predicate must be boolean")); + } + let fields = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + left.fields.clone() + } else { + for field in &mut joined[..left.fields.len()] { + if matches!(kind, JoinKind::Right | JoinKind::Full) { + field.nullable = true; + } + } + for field in &mut joined[left.fields.len()..] { + if matches!(kind, JoinKind::Left | JoinKind::Full) { + field.nullable = true; + } + } + joined + }; + Self { + kind: Kind::Join { + kind, + predicate: Box::new(predicate), + }, + inputs: vec![left, right], + output: schema(fields), + } + .with_output_schema(output) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + if let Kind::Join { kind, predicate } = &operator.kind { + let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; + let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; + return Ok(futures::stream::once(async move { + use planner_types::pre_asap::JoinKind; + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + let mut workspace = Workspace::new(&context)?; + let mut work = Cooperative::new(&context); + workspace.grow(right.len())?; + let mut result = Vec::new(); + let mut right_matched = vec![false; right.len()]; + for left_row in &left { + work.checkpoint().await?; + let mut matched = false; + for (i, right_row) in right.iter().enumerate() { + work.checkpoint().await?; + let mut joined = left_row.clone(); + joined.extend(right_row.iter().cloned()); + if *kind == JoinKind::Cross + || matches!(predicate.evaluate(&joined)?, Value::Bool(true)) + { + matched = true; + right_matched[i] = true; + match kind { + JoinKind::Semi => { + workspace.grow(row_bytes(left_row))?; + result.push(left_row.clone()); + break; + } + JoinKind::Anti => break, + _ => { + workspace.grow(row_bytes(&joined))?; + result.push(joined); + } + } + } + } + if !matched { + match kind { + JoinKind::Left | JoinKind::Full => { + let mut joined = left_row.clone(); + joined.resize( + joined.len() + operator.inputs[1].fields.len(), + Value::Null, + ); + workspace.grow(row_bytes(&joined))?; + result.push(joined); + } + JoinKind::Anti => { + workspace.grow(row_bytes(left_row))?; + result.push(left_row.clone()); + } + _ => {} + } + } + } + if matches!(kind, JoinKind::Right | JoinKind::Full) { + for (matched, row) in right_matched.into_iter().zip(right) { + work.checkpoint().await?; + if !matched { + let mut joined = vec![Value::Null; operator.inputs[0].fields.len()]; + joined.extend(row); + workspace.grow(row_bytes(&joined))?; + result.push(joined); + } + } + } + Batch::try_new(output, result) + }) + .boxed_local()); + } + if let Kind::SemiJoin { + keys, + require_complete_right, + } = &operator.kind + { + let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; + let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; + return Ok(futures::stream::once(async move { + // Poll both branches together: either may depend on a common producer. + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + let right_cols = keys.iter().map(|(_, r)| *r).collect::>(); + let left_cols = keys.iter().map(|(l, _)| *l).collect::>(); + let mut members = std::collections::BTreeSet::new(); + let mut workspace = Workspace::new(&context)?; + let mut work = Cooperative::new(&context); + for row in &right { + work.checkpoint().await?; + if right_cols.iter().all(|&i| matchable_key(&row[i])) { + let key = group_key(row, &right_cols)?; + if !members.contains(&key) { + workspace.grow(key_bytes(&key))?; + members.insert(key); + } + } else if *require_complete_right { + return Err(invalid( + "certified pruning candidate has an unmatchable key", + )); + } + } + let mut rows = Vec::new(); + let mut covered = std::collections::BTreeSet::new(); + for row in left { + work.checkpoint().await?; + if left_cols.iter().all(|&i| matchable_key(&row[i])) + && members.contains(&group_key(&row, &left_cols)?) + { + if *require_complete_right { + let key = group_key(&row, &left_cols)?; + if !covered.contains(&key) { + workspace.grow(key_bytes(&key))?; + covered.insert(key); + } + } + workspace.grow(std::mem::size_of::>())?; + rows.push(row); + } + } + if *require_complete_right && members != covered { + return Err(invalid("certified pruning key has no authoritative value")); + } + Batch::try_new(output, rows) + }) + .boxed_local()); + } + unreachable!() +} + +// Group keys canonicalize NaNs, but equality joins must not match them. +fn matchable_key(value: &Value) -> bool { + match value { + Value::Null => false, + Value::Float64(v) => !v.is_nan(), + _ => true, + } +} diff --git a/crates/asap-physical-operators/src/operators/limit.rs b/crates/asap-physical-operators/src/operators/limit.rs new file mode 100644 index 00000000..5f5e601f --- /dev/null +++ b/crates/asap-physical-operators/src/operators/limit.rs @@ -0,0 +1,69 @@ +use super::*; +impl Operator { + pub fn limit(input: Schema, n: u64, offset: u64, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + Ok(Self { + kind: Kind::Limit { n, offset, groups }, + inputs: vec![input.clone()], + output: input, + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::Limit { n, offset, groups } => { + let counts = BTreeMap::>, u64>::new(); + Ok(futures::stream::try_unfold( + (input, counts, Vec::::new(), false), + move |(mut input, mut counts, mut memory, done)| { + let output = output.clone(); + let context = context.clone(); + async move { + if done || *n == 0 { + return Ok(None); + } + let Some(batch) = input.next().await else { + return Ok(None); + }; + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + let key = group_key(row, groups)?; + if !counts.contains_key(&key) { + memory.push( + context.reserve( + key.iter() + .map(|part| part.len() + std::mem::size_of::>()) + .sum::() + + 64, + )?, + ); + } + let count = counts.entry(key).or_default(); + if *count >= *offset && count.saturating_sub(*offset) < *n { + rows.push(row.clone()); + } + *count = count.saturating_add(1); + } + let done = groups.is_empty() + && counts + .get(&vec![]) + .is_some_and(|count| count.saturating_sub(*offset) >= *n); + Ok(Some(( + Batch::try_new(output, rows)?, + (input, counts, memory, done), + ))) + } + }, + ) + .boxed_local()) + } + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs new file mode 100644 index 00000000..e4c6944f --- /dev/null +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -0,0 +1,316 @@ +//! Native physical operators. Each module owns its constructors and execution. +mod aligned_binary; +use crate::plan::{Boundedness, Emission, PhysicalOperator, PlanProperties}; +use crate::{ + runtime::{Cooperative, Input, OutputStream, Reservation, RunContext}, + values::{field, group_key, plain, Batch, Schema, Value}, + Error, +}; +use futures::StreamExt; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema, SummaryUpdate}, + pre_asap::{ColumnRef, DataType}, +}; +use std::{collections::BTreeMap, sync::Arc}; +pub(crate) mod common; +use crate::expressions::ordered; +pub use crate::expressions::Expression; +use common::*; +mod aggregate; +mod current_series; +mod filter; +mod joins; +mod limit; +mod panes; +mod projection; +mod sort; +mod source; +mod summary; +mod unchecked; +pub(crate) mod vector_binary; +pub(crate) mod vector_window; +pub use aggregate::Reduction; +pub use sort::SortKey; +pub use summary::ReadoutQuery; +#[derive(Clone, serde::Serialize, serde::Deserialize)] +enum Kind { + #[serde(skip)] + Source(Vec), + Constant { + value: Value, + dtype: DataType, + }, + PaneInput { + coordinate: usize, + layout: planner_types::post_asap::PaneLayout, + offset_ms: Option, + }, + ScopeTimestamp { + columns: Vec>, + }, + CurrentSeries { + identity: usize, + coordinate: usize, + value: usize, + lookback_ms: i64, + }, + Union, + VectorToScalar { + column: usize, + }, + VectorBinary { + operator: planner_types::post_asap::BinaryOperator, + return_bool: bool, + }, + AlignedBinary { + keys: Vec<(usize, usize)>, + values: (usize, usize), + operator: planner_types::post_asap::BinaryOperator, + }, + RangeWindow { + intent: Box>, + }, + HistogramQuantile, + Project(Vec), + Filter(Expression), + Limit { + n: u64, + offset: u64, + groups: Vec, + }, + Sort { + keys: Vec, + groups: Vec, + }, + Window { + intent: Box>, + coordinate: usize, + value: usize, + groups: Vec, + window: Option<(i64, i64)>, + }, + Aggregate { + groups: Vec, + measures: Vec, + }, + SemiJoin { + keys: Vec<(usize, usize)>, + require_complete_right: bool, + }, + Join { + kind: planner_types::pre_asap::JoinKind, + predicate: Box, + }, + SummaryBuild { + family: SummaryFamilyType, + value: usize, + time: Option, + groups: Vec, + }, + KeyedSummaryBuild { + family: SummaryFamilyType, + value: usize, + items: Vec, + groups: Vec, + }, + KeyedReadout { + state: usize, + k: usize, + }, + SummaryMerge { + state: usize, + groups: Vec, + }, + Readout { + state: usize, + query: ReadoutQuery, + }, +} +/// A bound operation has a fully checked input/output contract before execution. +#[derive(Clone, serde::Serialize, serde::Deserialize)] +#[serde(try_from = "unchecked::UncheckedOperator")] +pub struct Operator { + kind: Kind, + inputs: Vec, + output: Schema, +} +impl Operator { + /// Resolve a counter readout's logical lookback to this run's evaluation range. + pub(super) fn readout_range(&self, context: &RunContext) -> Result, Error> { + let Kind::Readout { + query: + ReadoutQuery::Exact(crate::summary_kernels::exact::ExactReadout { + lookback_ms: Some(lookback), + .. + }), + .. + } = &self.kind + else { + return Ok(None); + }; + let end = match context.scope { + crate::runtime::Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + crate::runtime::Scope::Ingestion { window_end_ms, .. } => window_end_ms, + }; + let start = end + .checked_sub(*lookback) + .ok_or_else(|| invalid("counter window overflows Int64"))?; + if let crate::runtime::Scope::Ingestion { + window_start_ms, .. + } = context.scope + { + if window_start_ms != start { + return Err(invalid( + "maintenance window differs from logical counter window", + )); + } + } + Ok(Some((start, end))) + } + + pub(crate) fn with_output_schema(mut self, output: Schema) -> Result { + if self.output.fields.len() != output.fields.len() + || self + .output + .fields + .iter() + .zip(&output.fields) + .any(|(actual, declared)| { + actual.dtype != declared.dtype || (actual.nullable && !declared.nullable) + }) + { + return Err(invalid("native output type differs from Planner output")); + } + if output.time_index.is_some_and(|i| { + i >= output.fields.len() + || output.fields[i].dtype != SummaryFamilyType::Plain(DataType::Timestamp) + }) { + return Err(invalid("invalid output time column")); + } + self.output = output; + Ok(self) + } + pub fn schema(&self) -> Schema { + self.output.clone() + } +} +impl PhysicalOperator for Operator { + fn requires_bounded_input(&self) -> bool { + matches!( + self.kind, + Kind::Sort { .. } + | Kind::AlignedBinary { .. } + | Kind::VectorBinary { .. } + | Kind::RangeWindow { .. } + | Kind::HistogramQuantile + | Kind::CurrentSeries { .. } + | Kind::Aggregate { .. } + | Kind::Window { .. } + | Kind::Join { .. } + | Kind::SemiJoin { .. } + | Kind::SummaryBuild { .. } + | Kind::KeyedSummaryBuild { .. } + | Kind::SummaryMerge { .. } + | Kind::VectorToScalar { .. } + ) + } + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + let boundedness = match &self.kind { + Kind::Source(_) | Kind::Constant { .. } => Boundedness::Bounded, + Kind::Limit { groups, .. } if groups.is_empty() => Boundedness::Bounded, + _ => Boundedness::from_inputs(inputs), + }; + PlanProperties { + boundedness, + emission: if matches!( + self.kind, + Kind::PaneInput { .. } | Kind::ScopeTimestamp { .. } + ) { + inputs + .first() + .map_or(Emission::Unknown, |input| input.emission) + } else if self.requires_bounded_input() { + Emission::AfterInput + } else { + Emission::Incremental + }, + } + } + + fn name(&self) -> &str { + match self.kind { + Kind::Source(_) => "Source", + Kind::Constant { .. } => "Constant", + Kind::PaneInput { .. } => "PaneInput", + Kind::ScopeTimestamp { .. } => "ScopeTimestamp", + Kind::Union => "Union", + Kind::CurrentSeries { .. } => "CurrentSeries", + Kind::VectorToScalar { .. } => "VectorToScalar", + Kind::VectorBinary { .. } => "VectorBinary", + Kind::AlignedBinary { .. } => "AlignedBinary", + Kind::RangeWindow { .. } => "RangeWindow", + Kind::HistogramQuantile => "HistogramQuantile", + Kind::Project(_) => "Project", + Kind::Filter(_) => "Filter", + Kind::Limit { .. } => "Limit", + Kind::Sort { .. } => "Sort", + Kind::Aggregate { .. } => "Aggregate", + Kind::Window { .. } => "WindowAggregate", + Kind::SemiJoin { .. } => "SemiJoin", + Kind::Join { .. } => "RelationalJoin", + Kind::SummaryBuild { .. } | Kind::KeyedSummaryBuild { .. } => "SummaryAgg", + Kind::KeyedReadout { .. } => "SummaryEstimate", + Kind::SummaryMerge { .. } => "SummaryMerge", + Kind::Readout { .. } => "SummaryReadout", + } + } + fn validate_context(&self, context: &RunContext) -> Result<(), Error> { + panes::validate_context(self, context)?; + current_series::validate_context(self, context)?; + self.readout_range(context).map(|_| ()) + } + fn input_schemas(&self) -> Vec { + self.inputs.clone() + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, value: &Batch) -> usize { + value.bytes() + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + match self.kind { + Kind::Source(_) | Kind::Constant { .. } | Kind::Union | Kind::VectorToScalar { .. } => { + source::execute(self, inputs, context) + } + Kind::VectorBinary { .. } => vector_binary::execute(self, inputs, context), + Kind::AlignedBinary { .. } => aligned_binary::execute(self, inputs, context), + Kind::RangeWindow { .. } | Kind::HistogramQuantile => { + vector_window::execute(self, inputs, context) + } + Kind::Project(_) => projection::execute(self, inputs, context), + Kind::CurrentSeries { .. } => current_series::execute(self, inputs, context), + Kind::PaneInput { .. } | Kind::ScopeTimestamp { .. } => { + panes::execute(self, inputs, context) + } + Kind::Filter(_) => filter::execute(self, inputs, context), + Kind::Limit { .. } => limit::execute(self, inputs, context), + Kind::Sort { .. } => sort::execute(self, inputs, context), + Kind::Window { .. } | Kind::Aggregate { .. } => { + aggregate::execute(self, inputs, context) + } + Kind::Join { .. } | Kind::SemiJoin { .. } => joins::execute(self, inputs, context), + Kind::SummaryMerge { .. } => summary::execute_merge(self, inputs, context), + Kind::SummaryBuild { .. } + | Kind::Readout { .. } + | Kind::KeyedSummaryBuild { .. } + | Kind::KeyedReadout { .. } => summary::execute(self, inputs, context), + } + } +} diff --git a/crates/asap-physical-operators/src/operators/panes.rs b/crates/asap-physical-operators/src/operators/panes.rs new file mode 100644 index 00000000..4748db09 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/panes.rs @@ -0,0 +1,228 @@ +//! Run-scoped pane population checks and timestamp restoration after reduction. +use super::*; +use crate::runtime::Scope; +use planner_types::post_asap::{validate_pane_coverage, PaneLayout, WindowEdgeCoverage}; + +impl Operator { + pub(crate) fn pane_input( + input: Schema, + coordinate: usize, + layout: PaneLayout, + offset_ms: Option, + ) -> Result { + if plain(&input, coordinate)? != (&DataType::Timestamp, false) { + return Err(invalid("pane input requires a non-null timestamp")); + } + if layout.pane_width_ms > i64::MAX as u64 { + return Err(invalid("pane width exceeds timestamp range")); + } + validate_pane_coverage( + &layout, + layout.pane_origin_ms, + &WindowEdgeCoverage::PaneAligned, + ) + .map_err(|error| Error::Invalid(format!("invalid pane layout: {error:?}")))?; + if offset_ms.is_some_and(|offset| offset < 0) { + return Err(invalid("negative pane offset")); + } + Ok(Self { + kind: Kind::PaneInput { + coordinate, + layout, + offset_ms, + }, + inputs: vec![input.clone()], + output: input, + }) + } + + pub(crate) fn scope_timestamp(input: Schema, output: Schema) -> Result { + crate::values::validate_schema(&output)?; + let coordinate = output + .time_index + .ok_or_else(|| invalid("temporal output requires a time index"))?; + if plain(&output, coordinate)? != (&DataType::Timestamp, false) { + return Err(invalid("temporal output requires a non-null timestamp")); + } + let mut columns = Vec::new(); + let mut used = std::collections::BTreeSet::new(); + for (index, field) in output.fields.iter().enumerate() { + if index == coordinate { + columns.push(None); + continue; + } + let matches: Vec<_> = input + .fields + .iter() + .enumerate() + .filter(|(_, candidate)| { + candidate.dtype == field.dtype + && candidate.nullable == field.nullable + && (candidate.name == field.name + || !matches!(field.dtype, SummaryFamilyType::Plain(_))) + }) + .map(|(index, _)| index) + .collect(); + let [column] = matches.as_slice() else { + return Err(invalid("temporal output column missing or ambiguous")); + }; + if !used.insert(*column) { + return Err(invalid("temporal output repeats an input column")); + } + columns.push(Some(*column)); + } + if used.len() != input.fields.len() { + return Err(invalid("temporal output drops an input column")); + } + Ok(Self { + kind: Kind::ScopeTimestamp { columns }, + inputs: vec![input], + output, + }) + } +} + +pub(super) fn validate_context(operator: &Operator, context: &RunContext) -> Result<(), Error> { + let Kind::PaneInput { + layout, offset_ms, .. + } = &operator.kind + else { + return Ok(()); + }; + let end = match (&context.scope, offset_ms) { + ( + Scope::Ingestion { + window_start_ms, + window_end_ms, + .. + }, + None, + ) => { + if window_end_ms.checked_sub(*window_start_ms) != Some(layout.pane_width_ms as i64) { + return Err(invalid("maintenance run must cover exactly one pane")); + } + *window_end_ms + } + ( + Scope::Query { + evaluation_time_ms, .. + }, + Some(offset), + ) => { + let end = evaluation_time_ms + .checked_sub(*offset) + .ok_or_else(|| invalid("query pane timestamp overflows"))?; + end.checked_sub(layout.pane_width_ms as i64) + .ok_or_else(|| invalid("query pane start overflows"))?; + end + } + _ => return Err(invalid("pane operator received the wrong execution scope")), + }; + validate_pane_coverage(layout, Some(end), &WindowEdgeCoverage::PaneAligned).map_err(|error| { + Error::Invalid(format!( + "query requires aligned panes or boundary residuals: {error:?}" + )) + }) +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + validate_context(operator, &context)?; + let input = inputs.pop().ok_or_else(|| invalid("pane input missing"))?; + let output = operator.output.clone(); + let mut seen = std::collections::BTreeSet::new(); + let mut memory = context.reserve(0)?; + let mut key_bytes = 0; + Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + match &operator.kind { + Kind::PaneInput { + coordinate, + offset_ms, + .. + } => { + let groups: Vec<_> = output + .fields + .iter() + .enumerate() + .filter(|(index, field)| { + *index != *coordinate + && matches!(field.dtype, SummaryFamilyType::Plain(_)) + }) + .map(|(index, _)| index) + .collect(); + for row in batch.rows() { + let Value::Timestamp(timestamp) = row[*coordinate] else { + return Err(invalid("pane timestamp type mismatch")); + }; + match (&context.scope, offset_ms) { + ( + Scope::Ingestion { + window_start_ms, + window_end_ms, + .. + }, + None, + ) if timestamp > *window_start_ms && timestamp <= *window_end_ms => {} + ( + Scope::Query { + evaluation_time_ms, .. + }, + Some(offset), + ) if timestamp + == evaluation_time_ms + .checked_sub(*offset) + .ok_or_else(|| invalid("pane timestamp overflows"))? => + { + let key = group_key(row, &groups)?; + if seen.contains(&key) { + return Err(invalid("duplicate entity state within a pane")); + } + key_bytes += key.iter().map(Vec::len).sum::() + + key.len() * std::mem::size_of::>() + + 64; + memory.resize(key_bytes)?; + seen.insert(key); + } + _ => { + return Err(invalid("input population differs from required pane")) + } + } + } + Ok(batch.value().clone()) + } + Kind::ScopeTimestamp { columns } => { + let timestamp = match context.scope { + Scope::Ingestion { window_end_ms, .. } => window_end_ms, + Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + }; + let rows = batch + .rows() + .iter() + .map(|row| { + columns + .iter() + .map(|column| { + column.map_or(Value::Timestamp(timestamp), |column| { + row[column].clone() + }) + }) + .collect() + }) + .collect(); + Batch::try_new(output.clone(), rows) + } + _ => unreachable!(), + } + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/projection.rs b/crates/asap-physical-operators/src/operators/projection.rs new file mode 100644 index 00000000..4a174ed9 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/projection.rs @@ -0,0 +1,56 @@ +use super::*; +impl Operator { + pub fn project(input: Schema, columns: Vec<(String, Expression)>) -> Result { + let fields = columns + .iter() + .map(|(name, e)| { + if let Expression::Column(index) = e { + let mut field = input + .fields + .get(*index) + .ok_or_else(|| invalid("projection column out of range"))? + .clone(); + field.name = name.clone(); + return Ok(field); + } + let (t, n) = e.dtype(&input)?; + Ok(result_field(name, t, n)) + }) + .collect::>()?; + Ok(Self { + kind: Kind::Project(columns.into_iter().map(|(_, e)| e).collect()), + inputs: vec![input], + output: schema(fields), + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::Project(expressions) => Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + let rows = batch + .rows() + .iter() + .map(|r| { + expressions + .iter() + .map(|e| e.evaluate(r)) + .collect::, _>>() + }) + .collect::, _>>()?; + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/sort.rs b/crates/asap-physical-operators/src/operators/sort.rs new file mode 100644 index 00000000..71f81de0 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/sort.rs @@ -0,0 +1,172 @@ +use super::*; +impl Operator { + pub fn sort(input: Schema, keys: Vec, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + for key in &keys { + if !ordered(plain(&input, key.column)?.0) { + return Err(invalid("unsupported sort type")); + } + } + Ok(Self { + kind: Kind::Sort { keys, groups }, + inputs: vec![input.clone()], + output: input, + }) + } +} +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] +pub struct SortKey { + pub column: usize, + pub descending: bool, + pub nulls_first: bool, +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &operator.kind { + Kind::Sort { keys, groups } => { + let mut grouped = BTreeMap::>, Vec>>::new(); + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + for row in rows { + work.checkpoint().await?; + for key in keys { + if matches!(row[key.column], Value::Map(_)) && nested_nan(&row[key.column]) + { + return Err(invalid("NaN in collection sort key")); + } + } + let key = group_key(&row, groups)?; + workspace.grow(std::mem::size_of::>())?; + if !grouped.contains_key(&key) { + workspace.grow(key_bytes(&key))?; + } + grouped.entry(key).or_default().push(row); + } + let mut result = Vec::new(); + for rows in grouped.into_values() { + let rows = + cooperative_sort(rows, |a, b| compare_rows(a, b, keys), &context).await?; + result.extend(rows); + } + result + } + _ => unreachable!(), + }; + Batch::try_new(output, result) + }) + .boxed_local()) +} + +fn compare_rows(a: &[Value], b: &[Value], keys: &[SortKey]) -> std::cmp::Ordering { + use std::cmp::Ordering::*; + for key in keys { + let (a, b) = (&a[key.column], &b[key.column]); + let order = match (a, b) { + (Value::Null, Value::Null) => Equal, + (Value::Null, _) => { + if key.nulls_first { + Less + } else { + Greater + } + } + (_, Value::Null) => { + if key.nulls_first { + Greater + } else { + Less + } + } + (Value::Float64(a), Value::Float64(b)) if a.is_nan() || b.is_nan() => { + match (a.is_nan(), b.is_nan()) { + (true, true) => Equal, + (true, false) => Greater, + _ => Less, + } + } + _ => { + let order = a.compare(b).expect("bound ordered types"); + if key.descending { + order.reverse() + } else { + order + } + } + }; + if order != Equal { + return order; + } + } + Equal +} +fn nested_nan(value: &Value) -> bool { + match value { + Value::Float64(value) => value.is_nan(), + Value::Map(values) => values + .iter() + .any(|(key, value)| nested_nan(key) || nested_nan(value)), + Value::List(values) | Value::Struct(values) => values.iter().any(nested_nan), + _ => false, + } +} +/// Stable in-memory merge sort with bounded synchronous chunks. Scratch storage +/// is reserved before allocation; comparisons yield between merge steps. +pub(super) async fn cooperative_sort( + rows: Vec, + compare: impl Fn(&T, &T) -> std::cmp::Ordering, + context: &RunContext, +) -> Result, Error> { + use std::collections::VecDeque; + let bytes = rows + .len() + .checked_mul(std::mem::size_of::() + std::mem::size_of::>()) + .and_then(|n| n.checked_mul(3)) + .ok_or(Error::MemoryLimit)?; + let _scratch = context.reserve(bytes)?; + let mut work = Cooperative::new(context); + let mut rows = rows.into_iter(); + let mut runs = VecDeque::new(); + loop { + work.checkpoint().await?; + let mut chunk = rows.by_ref().take(256).collect::>(); + if chunk.is_empty() { + break; + } + chunk.sort_by(&compare); + runs.push_back(VecDeque::from(chunk)); + } + // Merge adjacent runs in rounds to preserve ties in original input order. + while runs.len() > 1 { + let mut next = VecDeque::new(); + while let Some(mut left) = runs.pop_front() { + let Some(mut right) = runs.pop_front() else { + next.push_back(left); + break; + }; + let mut merged = VecDeque::with_capacity(left.len() + right.len()); + while !left.is_empty() || !right.is_empty() { + work.checkpoint().await?; + let take_left = match (left.front(), right.front()) { + (Some(a), Some(b)) => !compare(a, b).is_gt(), + (Some(_), None) => true, + _ => false, + }; + merged.push_back(if take_left { + left.pop_front().unwrap() + } else { + right.pop_front().unwrap() + }); + } + next.push_back(merged); + } + runs = next; + } + Ok(runs.pop_front().unwrap_or_default().into()) +} diff --git a/crates/asap-physical-operators/src/operators/source.rs b/crates/asap-physical-operators/src/operators/source.rs new file mode 100644 index 00000000..42701e17 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/source.rs @@ -0,0 +1,96 @@ +use super::*; +impl Operator { + pub fn source(output: Schema, batches: Vec) -> Result { + crate::values::validate_schema(&output)?; + if batches.iter().any(|b| b.schema() != &output) { + return Err(invalid("source schema mismatch")); + } + Ok(Self { + kind: Kind::Source(batches), + inputs: vec![], + output, + }) + } + pub fn scalar(value: Value, dtype: DataType) -> Result { + let output = schema(vec![result_field( + "value", + dtype.clone(), + matches!(value, Value::Null), + )]); + Batch::try_new(output.clone(), vec![vec![value.clone()]])?; + Ok(Self { + kind: Kind::Constant { value, dtype }, + inputs: vec![], + output, + }) + } + pub fn vector_to_scalar(input: Schema, column: usize) -> Result { + if plain(&input, column)? != (&DataType::Float64, false) { + return Err(invalid("scalar conversion requires non-null Float64")); + } + Ok(Self { + kind: Kind::VectorToScalar { column }, + inputs: vec![input], + output: schema(vec![result_field("value", DataType::Float64, false)]), + }) + } + pub fn union(input: Schema, arity: usize) -> Result { + if arity == 0 { + return Err(invalid("union needs at least one input")); + } + Ok(Self { + kind: Kind::Union, + inputs: vec![input.clone(); arity], + output: input, + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + if let Kind::Source(batches) = &operator.kind { + return Ok(futures::stream::iter(batches.iter().cloned().map(Ok)).boxed_local()); + } + if let Kind::Constant { value, .. } = &operator.kind { + return Ok(futures::stream::once(async move { + Batch::try_new(output, vec![vec![value.clone()]]) + }) + .boxed_local()); + } + if matches!(operator.kind, Kind::Union) { + return Ok(futures::stream::select_all(inputs) + .map(|batch| batch.map(|batch| batch.value().clone())) + .boxed_local()); + } + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::VectorToScalar { column } => Ok(futures::stream::once(async move { + let mut input = input; + let mut work = Cooperative::new(&context); + let mut value = f64::NAN; + let mut count = 0usize; + while let Some(batch) = input.next().await { + for row in batch?.rows() { + work.checkpoint().await?; + count = count.saturating_add(1); + if let Value::Float64(v) = row[*column] { + value = v; + } + } + } + Batch::try_new( + output, + vec![vec![Value::Float64(if count == 1 { + value + } else { + f64::NAN + })]], + ) + }) + .boxed_local()), + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs new file mode 100644 index 00000000..cc87a75a --- /dev/null +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -0,0 +1,567 @@ +use super::*; +/// A summary readout: a sketch query, or an exact readout with typed parameters. +#[derive(Clone, Debug, PartialEq, serde::Serialize, serde::Deserialize)] +pub enum ReadoutQuery { + Sketch(planner_types::post_asap::SketchQuery), + Exact(crate::summary_kernels::exact::ExactReadout), +} + +impl Operator { + pub fn keyed_summary_build( + input: Schema, + family: SummaryFamilyType, + value: usize, + items: Vec, + groups: Vec, + ) -> Result { + use crate::summary_kernels::weighted_frequency::WeightedFrequency; + crate::values::validate_family(&family)?; + let SummaryFamilyType::Sketch(kind, _) = &family else { + return Err(invalid("keyed sketch required")); + }; + WeightedFrequency::configuration(kind)?; + validate_groups(&input, &groups)?; + if items.is_empty() || plain(&input, value)? != (&DataType::Float64, false) { + return Err(invalid( + "keyed summary requires identities and non-null Float64 weights", + )); + } + for &item in &items { + if !matches!( + plain(&input, item)?.0, + DataType::Utf8 + | DataType::Timestamp + | DataType::Int64 + | DataType::Float64 + | DataType::Bool + | DataType::Null + ) { + return Err(invalid("unsupported keyed summary identity type")); + } + } + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }); + Ok(Self { + kind: Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn keyed_readout( + input: Schema, + state: usize, + k: usize, + output: Schema, + ) -> Result { + use crate::summary_kernels::weighted_frequency::WeightedFrequency; + crate::values::validate_family(&field(&input, state)?.dtype)?; + let SummaryFamilyType::Sketch(kind, _) = &field(&input, state)?.dtype else { + return Err(invalid("keyed readout requires summary state")); + }; + let (_, _, _, capacity) = WeightedFrequency::configuration(kind)?; + if k > capacity || output.fields.len() <= input.fields.len() { + return Err(invalid("invalid keyed readout shape or capacity")); + } + if state + 1 != input.fields.len() + || output.fields[..state] != input.fields[..state] + || output.fields.last().unwrap().dtype != SummaryFamilyType::Plain(DataType::Float64) + { + return Err(invalid( + "keyed readout must preserve partitions and return a Float64 score", + )); + } + crate::values::validate_schema(&output)?; + Ok(Self { + kind: Kind::KeyedReadout { state, k }, + inputs: vec![input], + output, + }) + } + pub fn summary_build( + input: Schema, + family: SummaryFamilyType, + value: usize, + time: Option, + groups: Vec, + ) -> Result { + crate::values::validate_family(&family)?; + validate_groups(&input, &groups)?; + if plain(&input, value)?.0 != &DataType::Float64 { + return Err(invalid("summary numeric update requires Float64")); + } + if let Some(time) = time { + if plain(&input, time)? != (&DataType::Timestamp, false) { + return Err(invalid("summary time column must be a timestamp")); + } + } + if time.is_none() + && matches!( + family, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Rate + | planner_types::post_asap::ExactKind::Increase, + _ + ) + ) + { + return Err(invalid("counter summary requires a timestamp column")); + } + crate::capability::validate_summary_kernel( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .map_err(Error::Invalid)?; + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }); + Ok(Self { + kind: Kind::SummaryBuild { + family, + value, + time, + groups, + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn summary_merge(input: Schema, state: usize, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + crate::values::validate_family(&field(&input, state)?.dtype)?; + if matches!(field(&input, state)?.dtype, SummaryFamilyType::Plain(_)) { + return Err(invalid("summary state required")); + } + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(input.fields[state].clone()); + Ok(Self { + kind: Kind::SummaryMerge { state, groups }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn readout(input: Schema, state: usize, query: ReadoutQuery) -> Result { + let family = &field(&input, state)?.dtype; + crate::values::validate_family(family)?; + match &query { + ReadoutQuery::Sketch(query) => { + crate::capability::validate_sketch_readout(family, query)? + } + ReadoutQuery::Exact(readout) => { + crate::capability::validate_exact_readout(family, readout)? + } + } + let mut fields = input.fields.clone(); + let result_type = if matches!( + fields[state].dtype, + SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) + ) { + DataType::Int64 + } else { + DataType::Float64 + }; + // A state-only row represents the global population. Its extrema may + // be empty, just like an ordinary ungrouped MIN/MAX aggregate. + let nullable = fields.len() == 1 + && matches!( + fields[state].dtype, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Min + | planner_types::post_asap::ExactKind::Max, + _ + ) + ); + fields[state] = result_field("value", result_type, nullable); + Ok(Self { + kind: Kind::Readout { state, query }, + inputs: vec![input], + output: schema(fields), + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let range_ms = operator.readout_range(&context)?; + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::SummaryBuild { + family, + value, + time, + groups, + } => Ok(futures::stream::once(async move { + Batch::try_new( + output, + build_summary(input, family, *value, *time, groups, &context).await?, + ) + }) + .boxed_local()), + Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + } => Ok(futures::stream::once(async move { + Batch::try_new( + output, + build_keyed_summary(input, family, *value, items, groups, &context).await?, + ) + }) + .boxed_local()), + Kind::KeyedReadout { state, k } => Ok(input + .map(move |batch| { + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + let Value::Summary { state: summary, .. } = &row[*state] else { + return Err(invalid("summary value required")); + }; + let summary = summary + .as_any() + .downcast_ref::() + .ok_or_else(|| invalid("weighted frequency typed state required"))?; + for items in summary.rows(*k) { + let mut values = row[..*state].to_vec(); + values.extend(items); + // The typed output schema restores epoch-millisecond + // timestamp keys from the kernel's Int64 representation. + for (value, field) in values.iter_mut().zip(&output.fields) { + if field.dtype == SummaryFamilyType::Plain(DataType::Timestamp) { + if let Value::Int64(time) = value { + *value = Value::Timestamp(*time); + } + } + } + rows.push(values); + } + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + Kind::Readout { state, query } => Ok(input + .map(move |batch| { + let batch = batch?; + let mut rows = batch.rows().to_vec(); + if let ReadoutQuery::Exact(readout) = query { + rows.retain(|row| !matches!(&row[*state], Value::Summary { state: summary, .. } + if crate::readout::insufficient_counter_samples(summary.as_ref(), readout.statistic))); + } + for row in &mut rows { + let Value::Summary { state: summary, .. } = &row[*state] else { + return Err(invalid("summary value required")); + }; + row[*state] = match query { + ReadoutQuery::Sketch(query) => Value::Float64( + summary + .estimate(query) + .map_err(|e| Error::Operator(e.to_string()))?, + ), + ReadoutQuery::Exact(readout) => { + let exact = summary + .as_any() + .downcast_ref::() + .ok_or_else(|| invalid("exact readout requires exact state"))?; + if output.fields[*state].dtype == SummaryFamilyType::Plain(DataType::Int64) { + let count = exact.count().ok_or_else(|| { + Error::Operator("exact count state lacks an integer count".into()) + })?; + Value::Int64(i64::try_from(count).map_err(|_| { + Error::Operator("exact count exceeds Int64".into()) + })?) + } else { + match exact + .readout(readout.statistic, range_ms, None) + .map_err(|e| Error::Operator(e.to_string()))? + { + Some(value) => Value::Float64(value), + None if output.fields[*state].nullable => Value::Null, + None => { + return Err(Error::Operator( + "empty exact population".into(), + )) + } + } + } + } + }; + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => unreachable!(), + } +} +pub(super) fn execute_merge<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &operator.kind { + Kind::SummaryMerge { state, groups } => { + merge_summary(rows, *state, groups, &context).await? + } + _ => unreachable!(), + }; + Batch::try_new(output, result) + }) + .boxed_local()) +} + +async fn build_summary( + mut input: Input<'_, Batch>, + family: &SummaryFamilyType, + value: usize, + time: Option, + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + type State = ( + Vec, + Box, + Reservation, + usize, + Option, + ); + let create = |labels: Vec, key_bytes: usize| -> Result { + let updater = crate::factory::create_planner_accumulator( + family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .map_err(Error::Operator)?; + let overhead = labels.iter().map(Value::bytes).sum::() + key_bytes + 64; + let memory = context.reserve(updater.memory_usage_bytes() + overhead)?; + Ok((labels, updater, memory, overhead, None)) + }; + let mut work = Cooperative::new(context); + let mut states = BTreeMap::>, State>::new(); + if groups.is_empty() { + states.insert(vec![], create(vec![], 0)?); + } + let ordered_time = matches!( + family, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Rate + | planner_types::post_asap::ExactKind::Increase, + _ + ) + ); + while let Some(batch) = input.next().await { + let batch = batch?; + for row in batch.rows() { + work.checkpoint().await?; + let key = group_key(row, groups)?; + if !states.contains_key(&key) { + let labels = groups.iter().map(|&i| row[i].clone()).collect(); + let state = create( + labels, + key.iter() + .map(|v| v.len() + std::mem::size_of::>()) + .sum(), + )?; + states.insert(key.clone(), state); + } + let (_, updater, memory, overhead, previous) = + states.get_mut(&key).expect("inserted group"); + // SQL aggregates ignore NULL samples while retaining the group. + // A missing counter sample also contributes no observation. + let value = match row[value] { + Value::Float64(value) => value, + Value::Null => continue, + _ => return Err(invalid("summary update type")), + }; + let timestamp = if let Some(time) = time { + let Value::Timestamp(time) = row[time] else { + return Err(invalid("summary time type")); + }; + time + } else { + 0 + }; + if ordered_time && previous.is_some_and(|prior| timestamp <= prior) { + return Err(Error::Operator( + "counter samples must have strictly increasing timestamps within each group" + .into(), + )); + } + updater + .validate_single_input(value) + .map_err(Error::Operator)?; + updater.update_single(value, timestamp); + *previous = Some(timestamp); + memory.resize(updater.memory_usage_bytes() + *overhead)?; + } + } + Ok(states + .into_values() + .map(|(mut labels, updater, _memory, _, _)| { + labels.push(Value::Summary { + family: family.clone(), + state: Arc::from(updater.into_accumulator()), + }); + labels + }) + .collect()) +} +async fn merge_summary( + rows: Vec>, + state_column: usize, + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + type GroupState = (Vec, SummaryFamilyType, Arc); + let mut states: BTreeMap>, GroupState> = BTreeMap::new(); + let mut work = Cooperative::new(context); + let mut memory = context.reserve(0)?; + let mut retained = 0usize; + for row in rows { + work.checkpoint().await?; + let Value::Summary { family, state } = &row[state_column] else { + return Err(invalid("summary state required")); + }; + let key = group_key(&row, groups)?; + if let Some((_, expected, existing)) = states.get_mut(&key) { + if expected != family { + return Err(invalid("incompatible summary family")); + } + let old_bytes = existing.approx_memory_bytes(); + // Reserve an estimate for the replacement while both input states remain live. + memory.resize( + retained + .checked_add(old_bytes) + .and_then(|n| n.checked_add(state.approx_memory_bytes())) + .ok_or(Error::MemoryLimit)?, + )?; + *existing = Arc::from( + existing + .merge_with(state.as_ref()) + .map_err(|e| Error::Operator(e.to_string()))?, + ); + retained = retained + .checked_sub(old_bytes) + .and_then(|n| n.checked_add(existing.approx_memory_bytes())) + .ok_or(Error::MemoryLimit)?; + memory.resize(retained)?; + } else { + retained = retained + .checked_add(key_bytes(&key) + row_bytes(&row)) + .ok_or(Error::MemoryLimit)?; + memory.resize(retained)?; + states.insert( + key, + ( + groups.iter().map(|&i| row[i].clone()).collect(), + family.clone(), + state.clone(), + ), + ); + } + } + Ok(states + .into_values() + .map(|(mut keys, family, state)| { + keys.push(Value::Summary { family, state }); + keys + }) + .collect()) +} + +async fn build_keyed_summary( + mut input: Input<'_, Batch>, + family: &SummaryFamilyType, + value: usize, + items: &[usize], + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + use crate::{summary_kernels::weighted_frequency::WeightedFrequency, AggregateCore}; + let SummaryFamilyType::Sketch(kind, _) = family else { + unreachable!() + }; + let (algorithm, width, depth, capacity) = WeightedFrequency::configuration(kind)?; + let mut work = Cooperative::new(context); + let mut states = + BTreeMap::>, (Vec, WeightedFrequency, Reservation, usize)>::new(); + while let Some(batch) = input.next().await { + let batch = batch?; + for row in batch.rows() { + work.checkpoint().await?; + let key = group_key(row, groups)?; + if !states.contains_key(&key) { + let labels = groups.iter().map(|&i| row[i].clone()).collect::>(); + let overhead = labels.iter().map(Value::bytes).sum::() + + key.iter().map(|v| v.len() + 24).sum::() + + 128; + let bytes = width + .checked_mul(depth) + .and_then(|n| n.checked_mul(8)) + .and_then(|n| n.checked_add(overhead)) + .ok_or_else(|| invalid("weighted frequency memory size overflow"))?; + let reservation = context.reserve(bytes)?; + states.insert( + key.clone(), + ( + labels, + WeightedFrequency::new(algorithm, width, depth, capacity)?, + reservation, + overhead, + ), + ); + } + let (_, summary, reservation, overhead) = states.get_mut(&key).unwrap(); + let Value::Float64(weight) = row[value] else { + return Err(invalid("weighted frequency weight type")); + }; + summary.update( + &items + .iter() + .map(|&i| match &row[i] { + Value::Timestamp(time) => Value::Int64(*time), + value => value.clone(), + }) + .collect::>(), + weight, + )?; + reservation.resize(summary.approx_memory_bytes() + *overhead)?; + } + } + Ok(states + .into_values() + .map(|(mut labels, summary, _, _)| { + labels.push(Value::Summary { + family: family.clone(), + state: Arc::new(summary), + }); + labels + }) + .collect()) +} diff --git a/crates/asap-physical-operators/src/operators/unchecked.rs b/crates/asap-physical-operators/src/operators/unchecked.rs new file mode 100644 index 00000000..2ecbaabe --- /dev/null +++ b/crates/asap-physical-operators/src/operators/unchecked.rs @@ -0,0 +1,140 @@ +//! Deserialized operators are validated before use, whatever the encoding. +use super::*; + +#[derive(serde::Deserialize)] +#[serde(deny_unknown_fields)] +pub(super) struct UncheckedOperator { + kind: Kind, + inputs: Vec, + output: Schema, +} +impl TryFrom for Operator { + type Error = Error; + fn try_from(unchecked: UncheckedOperator) -> Result { + let expected_kind = + serde_json::to_value(&unchecked.kind).map_err(|error| invalid(&error.to_string()))?; + let UncheckedOperator { + kind, + inputs, + output, + } = unchecked; + for schema in inputs.iter().chain(std::iter::once(&output)) { + crate::values::validate_schema(schema)?; + } + let input = |index| { + inputs + .get(index) + .cloned() + .ok_or_else(|| invalid("missing operator input")) + }; + let op = match kind { + Kind::Source(_) => return Err(invalid("physical plans cannot serialize live sources")), + Kind::Constant { value, dtype } => Operator::scalar(value, dtype)?, + Kind::PaneInput { + coordinate, + layout, + offset_ms, + } => Operator::pane_input(input(0)?, coordinate, layout, offset_ms)?, + Kind::ScopeTimestamp { .. } => Operator::scope_timestamp(input(0)?, output.clone())?, + Kind::Union => Operator::union(input(0)?, inputs.len())?, + Kind::CurrentSeries { + identity, + coordinate, + value, + lookback_ms, + } => Operator::current_series(input(0)?, identity, coordinate, value, lookback_ms)?, + Kind::VectorToScalar { column } => Operator::vector_to_scalar(input(0)?, column)?, + Kind::VectorBinary { + operator, + return_bool, + } => Operator::vector_binary(input(0)?, input(1)?, operator, return_bool)?, + Kind::AlignedBinary { + keys, + values, + operator, + } => Operator::aligned_binary(input(0)?, input(1)?, keys, values, operator)?, + Kind::RangeWindow { intent } => Operator::range_window(*intent)?, + Kind::HistogramQuantile => Operator::histogram_quantile(), + Kind::Project(expressions) => { + if expressions.len() != output.fields.len() { + return Err(invalid("projection width mismatch")); + } + Operator::project( + input(0)?, + output + .fields + .iter() + .zip(expressions) + .map(|(f, e)| (f.name.clone(), e)) + .collect(), + )? + } + Kind::Filter(expression) => Operator::filter(input(0)?, expression)?, + Kind::Limit { n, offset, groups } => Operator::limit(input(0)?, n, offset, groups)?, + Kind::Sort { keys, groups } => Operator::sort(input(0)?, keys, groups)?, + Kind::Window { + intent, + coordinate, + value, + groups, + window, + } => Operator::window(input(0)?, *intent, coordinate, value, groups, window)?, + Kind::Aggregate { groups, measures } => { + if groups.len() + measures.len() != output.fields.len() { + return Err(invalid("aggregate width mismatch")); + } + let names = output.fields[groups.len()..].iter().map(|f| f.name.clone()); + Operator::aggregate(input(0)?, groups, names.zip(measures).collect())? + } + Kind::SemiJoin { + keys, + require_complete_right, + } => { + let operator = Operator::semi_join(input(0)?, input(1)?, keys)?; + if require_complete_right { + operator.require_complete_right() + } else { + operator + } + } + Kind::Join { kind, predicate } => Operator::relational_join( + input(0)?, + input(1)?, + kind, + &planner_types::pre_asap::Predicate(std::rc::Rc::new( + predicate.expression().clone(), + )), + output.clone(), + )?, + Kind::SummaryBuild { + family, + value, + time, + groups, + } => Operator::summary_build(input(0)?, family, value, time, groups)?, + Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + } => Operator::keyed_summary_build(input(0)?, family, value, items, groups)?, + Kind::KeyedReadout { state, k } => { + Operator::keyed_readout(input(0)?, state, k, output.clone())? + } + Kind::SummaryMerge { state, groups } => { + Operator::summary_merge(input(0)?, state, groups)? + } + Kind::Readout { state, query } => Operator::readout(input(0)?, state, query)?, + } + .with_output_schema(output)?; + if serde_json::to_value(&op.kind).map_err(|error| invalid(&error.to_string()))? + != expected_kind + { + return Err(invalid("operator contains inconsistent compiled fields")); + } + if op.inputs != inputs { + return Err(invalid("operator input contracts differ")); + } + Ok(op) + } +} diff --git a/crates/asap-physical-operators/src/operators/vector_binary.rs b/crates/asap-physical-operators/src/operators/vector_binary.rs new file mode 100644 index 00000000..cc60ff4c --- /dev/null +++ b/crates/asap-physical-operators/src/operators/vector_binary.rs @@ -0,0 +1,237 @@ +//! Label matching and scalar broadcasting are physical computation, not source binding. +use super::*; +use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; + +pub(crate) fn value_schema(scalar: bool) -> Schema { + let mut fields = Vec::new(); + if !scalar { + fields.push(result_field( + "labels", + DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }, + false, + )); + } + fields.push(result_field( + if scalar { "$promql_scalar" } else { "value" }, + DataType::Float64, + false, + )); + schema(fields) +} + +fn is_scalar(input: &Schema) -> Result { + for scalar in [true, false] { + let expected = value_schema(scalar); + if input.fields.len() == expected.fields.len() + && input + .fields + .iter() + .zip(&expected.fields) + .all(|(a, b)| a.dtype == b.dtype && !a.nullable) + { + return Ok(scalar); + } + } + Err(invalid( + "vector binary requires Float64 scalars or complete label-map vectors", + )) +} + +impl Operator { + pub fn vector_binary( + left: Schema, + right: Schema, + operator: BinaryOperator, + return_bool: bool, + ) -> Result { + let scalar = is_scalar(&left)? && is_scalar(&right)?; + is_scalar(&right)?; + let expression = Expression::Binary { + operator: operator.clone(), + left: Box::new(Expression::Column(0)), + right: Box::new(Expression::Column(1)), + }; + expression.dtype(&schema(vec![ + result_field("left", DataType::Float64, false), + result_field("right", DataType::Float64, false), + ]))?; + let comparison = matches!(operator.kind, BinaryOpKind::Compare(_)); + if (return_bool && !comparison) || (scalar && comparison && !return_bool) { + return Err(invalid("invalid scalar/vector comparison bool mode")); + } + Ok(Self { + inputs: vec![left, right], + output: value_schema(scalar), + kind: Kind::VectorBinary { + operator, + return_bool, + }, + }) + } +} + +type Labels = BTreeMap, Arc>; +fn labels(row: &[Value]) -> Result { + let Some(Value::Map(entries)) = row.first() else { + return Err(invalid("vector requires label map")); + }; + let mut result = BTreeMap::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("labels must be Utf8")); + }; + if result.insert(key.clone(), value.clone()).is_some() { + return Err(invalid("duplicate label name")); + } + } + Ok(result) +} +fn identity(mut labels: Labels) -> Labels { + labels.remove("__name__"); + labels.retain(|_, value| !value.is_empty()); + labels +} +fn value(row: &[Value]) -> Result { + match row.last() { + Some(Value::Float64(value)) => Ok(*value), + _ => Err(invalid("binary value must be Float64")), + } +} +fn label_bytes(labels: &Labels) -> usize { + labels.iter().map(|(k, v)| 64 + k.len() + v.len()).sum() +} + +pub(super) fn execute<'a>( + op: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::VectorBinary { + operator, + return_bool, + } = &op.kind + else { + unreachable!() + }; + let left_scalar = is_scalar(&op.inputs[0])?; + let right_scalar = is_scalar(&op.inputs[1])?; + let right = inputs.pop().ok_or_else(|| invalid("missing right input"))?; + let left = inputs.pop().ok_or_else(|| invalid("missing left input"))?; + Ok(futures::stream::once(async move { + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + if (left_scalar && left.len() != 1) || (right_scalar && right.len() != 1) { + return Err(invalid("scalar input must contain exactly one value")); + } + let mut workspace = Workspace::new(&context)?; + let mut work = Cooperative::new(&context); + let mut rows = Vec::new(); + let mut result_identities = std::collections::BTreeSet::new(); + let mut emit = |labels: Labels, a: f64, b: f64| -> Result<(), Error> { + let arithmetic = matches!(operator.kind, BinaryOpKind::Arithmetic(_)); + let result = match crate::expressions::arithmetic::evaluate_binary(operator, a, b)? { + Value::Float64(value) => value, + Value::Bool(value) if *return_bool => { + if value { + 1. + } else { + 0. + } + } + Value::Bool(true) => { + if left_scalar { + b + } else { + a + } + } + Value::Bool(false) => return Ok(()), + _ => return Err(invalid("invalid binary result")), + }; + let mut row = Vec::new(); + if !left_scalar || !right_scalar { + let labels = if arithmetic || *return_bool { + identity(labels) + } else { + labels + }; + workspace.grow(label_bytes(&labels) + 64)?; + if !result_identities.insert(labels.clone()) { + return Err(invalid("duplicate vector result labels")); + } + workspace.grow( + label_bytes(&labels) + + std::mem::size_of::>() + + 2 * std::mem::size_of::(), + )?; + row.push(Value::Map( + labels + .into_iter() + .map(|(k, v)| (Value::Utf8(k), Value::Utf8(v))) + .collect::>() + .into(), + )); + } else { + workspace.grow(std::mem::size_of::>() + std::mem::size_of::())?; + } + row.push(Value::Float64(result)); + rows.push(row); + Ok(()) + }; + if left_scalar || right_scalar { + let vectors = if left_scalar { &right } else { &left }; + for row in vectors { + work.checkpoint().await?; + let labels = if left_scalar && right_scalar { + Labels::new() + } else { + labels(row)? + }; + emit( + labels, + if left_scalar { + value(&left[0])? + } else { + value(row)? + }, + if right_scalar { + value(&right[0])? + } else { + value(row)? + }, + )?; + } + } else { + let mut rhs = BTreeMap::new(); + // Keep matching workspace separate from the output reservation captured by emit. + let mut matching = Workspace::new(&context)?; + for row in &right { + work.checkpoint().await?; + let key = identity(labels(row)?); + matching.grow(label_bytes(&key) + 64)?; + if rhs.insert(key, value(row)?).is_some() { + return Err(invalid("duplicate vector matching labels")); + } + } + let mut seen = std::collections::BTreeSet::new(); + for row in &left { + work.checkpoint().await?; + let labels = labels(row)?; + let key = identity(labels.clone()); + matching.grow(label_bytes(&key) + 64)?; + if !seen.insert(key.clone()) { + return Err(invalid("duplicate vector matching labels")); + } + if let Some(b) = rhs.get(&key) { + emit(labels, value(row)?, *b)?; + } + } + } + Batch::try_new(op.output.clone(), rows) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/vector_window.rs b/crates/asap-physical-operators/src/operators/vector_window.rs new file mode 100644 index 00000000..407f04ea --- /dev/null +++ b/crates/asap-physical-operators/src/operators/vector_window.rs @@ -0,0 +1,147 @@ +//! Window bounds are typed input data; aggregation and histogram semantics stay native. +use super::*; +use planner_types::pre_asap::AggIntent; + +pub(crate) fn matrix_schema() -> Schema { + let mut fields = vector_binary::value_schema(false).fields.clone(); + fields.insert(1, result_field("timestamp", DataType::Timestamp, false)); + fields.push(result_field("window_start", DataType::Timestamp, false)); + fields.push(result_field("window_end", DataType::Timestamp, false)); + Arc::new(SummarySchema { + fields, + time_index: Some(1), + }) +} + +impl Operator { + pub fn range_window(intent: AggIntent) -> Result { + // Reuse the window constructor's semantic admission, without fixing request time. + Self::window(matrix_schema(), intent.clone(), 1, 2, vec![0], Some((0, 1)))?; + Ok(Self { + kind: Kind::RangeWindow { + intent: Box::new(intent), + }, + inputs: vec![matrix_schema()], + output: vector_binary::value_schema(false), + }) + } + pub fn histogram_quantile() -> Self { + Self { + kind: Kind::HistogramQuantile, + inputs: vec![ + vector_binary::value_schema(true), + vector_binary::value_schema(false), + ], + output: vector_binary::value_schema(false), + } + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + match &operator.kind { + Kind::RangeWindow { intent } => { + let input = inputs + .pop() + .ok_or_else(|| invalid("missing matrix input"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let mut window = None; + let mut work = Cooperative::new(&context); + for row in &rows { + work.checkpoint().await?; + let (Value::Timestamp(start), Value::Timestamp(end)) = (&row[3], &row[4]) + else { + return Err(invalid("missing matrix window bounds")); + }; + if start >= end || window.is_some_and(|bounds| bounds != (*start, *end)) { + return Err(invalid("matrix rows must share one nonempty window")); + } + window = Some((*start, *end)); + } + let mut rows = + aggregate::temporal::reduce(rows, intent, &[0], 1, 2, window, &context).await?; + for row in &mut rows { + work.checkpoint().await?; + row[1] = Expression::ExactFloat64(1).evaluate(row)?; + } + Batch::try_new(operator.output.clone(), rows) + }) + .boxed_local()) + } + Kind::HistogramQuantile => { + let buckets = inputs + .pop() + .ok_or_else(|| invalid("missing histogram buckets"))?; + let quantile = inputs.pop().ok_or_else(|| invalid("missing quantile"))?; + Ok(futures::stream::once(async move { + let ((quantile, _q_memory), (buckets, _bucket_memory)) = futures::try_join!( + collect_rows(quantile, &context), + collect_rows(buckets, &context) + )?; + let [row] = quantile.as_slice() else { + return Err(invalid("histogram quantile requires one scalar")); + }; + let [Value::Float64(q)] = row.as_slice() else { + return Err(invalid("invalid quantile scalar")); + }; + let mut rows = Vec::new(); + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + for row in buckets { + work.checkpoint().await?; + let Value::Map(entries) = &row[0] else { + return Err(invalid("histogram buckets require labels")); + }; + let mut bound = None; + let mut labels = BTreeMap::new(); + let mut seen = std::collections::BTreeSet::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("histogram labels must be Utf8")); + }; + if !seen.insert(key) { + return Err(invalid("duplicate histogram label")); + } + if key.as_ref() == "le" { + bound = value.parse::().ok(); + } else if key.as_ref() != "__name__" && !value.is_empty() { + labels.insert(key.clone(), value.clone()); + } + } + if let Some(bound) = bound { + let projected = vec![ + Value::Map( + labels + .into_iter() + .map(|(k, v)| (Value::Utf8(k), Value::Utf8(v))) + .collect::>() + .into(), + ), + Value::Float64(bound), + row[1].clone(), + ]; + workspace.grow(row_bytes(&projected))?; + rows.push(projected); + } + } + let result = aggregate::temporal::reduce( + rows, + &AggIntent::HistogramQuantile { q: *q }, + &[0], + 1, + 2, + None, + &context, + ) + .await?; + Batch::try_new(operator.output.clone(), result) + }) + .boxed_local()) + } + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/plan/mod.rs b/crates/asap-physical-operators/src/plan/mod.rs new file mode 100644 index 00000000..0dce4dd8 --- /dev/null +++ b/crates/asap-physical-operators/src/plan/mod.rs @@ -0,0 +1,160 @@ +//! Immutable physical graph, operator contracts and pre-execution validation. +use crate::{ + runtime::{Input, OutputStream, RunContext}, + Error, +}; +use std::{ + collections::{BTreeMap, BTreeSet}, + fmt::Debug, +}; +pub type NodeId = u64; +mod properties; +pub use properties::{Boundedness, Emission, PlanProperties}; +/// Operators own computation. The runtime provides already-connected inputs; +/// an operator must not recursively execute another plan node itself. +pub trait PhysicalOperator { + fn name(&self) -> &str; + /// Source implementations must explicitly declare finite input before feeding blocking operators. + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + PlanProperties { + boundedness: Boundedness::from_inputs(inputs), + emission: Emission::Unknown, + } + } + fn requires_bounded_input(&self) -> bool { + false + } + + /// Validate run-specific contracts before any source is opened. + fn validate_context(&self, _context: &RunContext) -> Result<(), Error> { + Ok(()) + } + fn input_schemas(&self) -> Vec; + fn output_schema(&self) -> S; + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error>; + fn output_bytes(&self, value: &V) -> usize; +} +pub(crate) struct Node<'a, V, S> { + pub(crate) inputs: Vec, + pub(crate) operator: Box + 'a>, +} +pub struct PhysicalDag<'a, V, S> { + pub(crate) nodes: BTreeMap>, +} +impl Default for PhysicalDag<'_, V, S> { + fn default() -> Self { + Self { + nodes: BTreeMap::new(), + } + } +} +impl<'a, V: 'a, S: Clone + PartialEq + Debug + 'a> PhysicalDag<'a, V, S> { + pub fn add( + &mut self, + id: NodeId, + inputs: Vec, + operator: impl PhysicalOperator + 'a, + ) -> Result<(), Error> { + self.add_boxed(id, inputs, Box::new(operator)) + } + pub fn add_boxed( + &mut self, + id: NodeId, + inputs: Vec, + operator: Box + 'a>, + ) -> Result<(), Error> { + if self.nodes.contains_key(&id) { + return Err(Error::Invalid(format!("duplicate node {id}"))); + } + self.nodes.insert(id, Node { inputs, operator }); + Ok(()) + } + pub fn validate(&self, roots: &[NodeId]) -> Result<(), Error> { + self.properties(roots).map(|_| ()) + } + /// Derive properties while checking topology and schemas, before starting sources. + pub fn properties(&self, roots: &[NodeId]) -> Result, Error> { + fn visit( + dag: &PhysicalDag<'_, V, S>, + id: NodeId, + active: &mut BTreeSet, + done: &mut BTreeMap, + ) -> Result { + if let Some((depth, _)) = done.get(&id) { + return Ok(*depth); + } + if active.len() >= 128 { + return Err(Error::Invalid( + "DAG exceeds the supported execution depth of 128".into(), + )); + } + if !active.insert(id) { + return Err(Error::Invalid(format!("cycle at node {id}"))); + } + let node = dag + .nodes + .get(&id) + .ok_or_else(|| Error::Invalid(format!("missing node {id}")))?; + let expected = node.operator.input_schemas(); + if expected.len() != node.inputs.len() { + return Err(Error::Invalid(format!("node {id} input arity mismatch"))); + } + let mut depth = 1; + let mut input_properties = Vec::new(); + for (input, schema) in node.inputs.iter().zip(expected) { + depth = depth.max(1 + visit(dag, *input, active, done)?); + input_properties.push(done[input].1); + let actual = dag.nodes[input].operator.output_schema(); + if actual != schema { + return Err(Error::Invalid(format!( + "node {id} input {input} schema mismatch: {actual:?} vs {schema:?}" + ))); + } + } + if depth > 128 { + return Err(Error::Invalid( + "DAG exceeds the supported execution depth of 128".into(), + )); + } + if node.operator.requires_bounded_input() + && input_properties + .iter() + .any(|p| p.boundedness != Boundedness::Bounded) + { + return Err(Error::Invalid(format!( + "node {id} ({}) requires bounded inputs", + node.operator.name() + ))); + } + let properties = node.operator.properties(&input_properties); + active.remove(&id); + done.insert(id, (depth, properties)); + Ok(depth) + } + if roots.is_empty() { + return Err(Error::Invalid("execution needs a root".into())); + } + let mut done = BTreeMap::new(); + for &root in roots { + visit(self, root, &mut BTreeSet::new(), &mut done)?; + } + Ok(done + .into_iter() + .map(|(id, (_, properties))| (id, properties)) + .collect()) + } + pub fn execute<'r>( + &'r self, + roots: &[NodeId], + context: RunContext, + ) -> Result>, Error> + where + 'a: 'r, + { + crate::runtime::execute(self, roots, context) + } +} diff --git a/crates/asap-physical-operators/src/plan/properties.rs b/crates/asap-physical-operators/src/plan/properties.rs new file mode 100644 index 00000000..7ec770ea --- /dev/null +++ b/crates/asap-physical-operators/src/plan/properties.rs @@ -0,0 +1,32 @@ +//! Execution facts used to reject operators that cannot finish on their inputs. +#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug, PartialEq, Eq)] +pub enum Boundedness { + /// The source or operator promises a finite result for this run. + Bounded, + Unbounded, + /// No finite-input guarantee has been supplied. + Unknown, +} +impl Boundedness { + pub fn from_inputs(inputs: &[PlanProperties]) -> Self { + if inputs.iter().any(|p| p.boundedness == Self::Unbounded) { + Self::Unbounded + } else if inputs.is_empty() || inputs.iter().any(|p| p.boundedness == Self::Unknown) { + Self::Unknown + } else { + Self::Bounded + } + } +} +#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug, PartialEq, Eq)] +pub enum Emission { + Incremental, + /// Produces its result only after all inputs end, even if accumulation is incremental. + AfterInput, + Unknown, +} +#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug, PartialEq, Eq)] +pub struct PlanProperties { + pub boundedness: Boundedness, + pub emission: Emission, +} diff --git a/crates/asap-physical-operators/src/readout.rs b/crates/asap-physical-operators/src/readout.rs new file mode 100644 index 00000000..e884ecc3 --- /dev/null +++ b/crates/asap-physical-operators/src/readout.rs @@ -0,0 +1,111 @@ +//! Readouts over merged exact summary states. +use crate::summary_kernels::exact::ExactAccumulator; +use crate::{AggregateCore, KeyByLabelValues, Statistic}; +use std::sync::Arc; + +fn merge_exact_states( + states: impl IntoIterator>, +) -> Result { + let mut states = states.into_iter(); + let exact = |state: &Arc| { + state + .as_any() + .downcast_ref::() + .cloned() + .ok_or_else(|| "readout requires Planner exact state".to_string()) + }; + let mut merged = exact(&states.next().ok_or("empty exact state input")?)?; + for state in states { + merged + .merge_from(&exact(&state)?) + .map_err(|error| error.to_string())?; + } + Ok(merged) +} + +/// PromQL counter readouts omit a series with fewer than two samples. Other +/// state/type/range failures remain errors rather than empty results. +pub fn insufficient_counter_samples(state: &dyn AggregateCore, statistic: Statistic) -> bool { + matches!(statistic, Statistic::Rate | Statistic::Increase) + && state + .as_any() + .downcast_ref::() + .is_some_and(|state| state.insufficient_counter_samples(statistic, &None)) +} + +/// Merge already selected exact panes and read one population. `None` means +/// the population is absent from the result: a counter with too few samples, +/// or an empty MIN/MAX. +pub fn exact_readout( + states: impl IntoIterator>, + statistic: Statistic, + range_ms: Option<(i64, i64)>, + key: Option<&KeyByLabelValues>, +) -> Result, String> { + let merged = merge_exact_states(states)?; + if merged.insufficient_counter_samples(statistic, &key.cloned()) { + return Ok(None); + } + merged + .readout(statistic, range_ms, key) + .map_err(|error| error.to_string()) +} + +#[cfg(test)] +mod counter_tests { + use super::*; + use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; + + fn counter(kind: ExactKind, params: ExactParams, keyed: bool) -> ExactAccumulator { + ExactAccumulator::new(SummaryFamilyType::ExactAggregate(kind, params), keyed).unwrap() + } + + // A counter population with a single sample is absent, keyed or not. + #[test] + fn planner_counter_population_omits_insufficient_samples() { + for (kind, params, statistic) in [ + (ExactKind::Rate, ExactParams::Rate, Statistic::Rate), + ( + ExactKind::Increase, + ExactParams::Increase, + Statistic::Increase, + ), + ] { + for keyed in [false, true] { + let mut state = counter(kind.clone(), params.clone(), keyed); + let key = keyed.then(|| KeyByLabelValues::new_with_labels(vec!["checkout".into()])); + state.update(key.as_ref(), 10., 10_000); + assert_eq!( + exact_readout( + [Arc::new(state) as Arc], + statistic, + None, + key.as_ref() + ) + .unwrap(), + None + ); + } + } + } + + // Two ordered samples read a rate; an inverted range and empty input fail. + #[test] + fn sparse_counter_is_absent_but_invalid_ranges_still_fail() { + let mut state = counter(ExactKind::Rate, ExactParams::Rate, false); + state.update(None, 10., 10_000); + let rate = Statistic::Rate; + let one = [Arc::new(state.clone()) as Arc]; + assert_eq!( + exact_readout(one, rate, Some((0, 60_000)), None).unwrap(), + None + ); + state.update(None, 20., 20_000); + let two = || [Arc::new(state.clone()) as Arc]; + assert!(exact_readout(two(), rate, Some((0, 60_000)), None) + .unwrap() + .is_some()); + assert!(exact_readout(two(), rate, Some((60_000, 0)), None).is_err()); + assert!(exact_readout([], rate, Some((0, 60_000)), None).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/runtime/batch_execution.rs b/crates/asap-physical-operators/src/runtime/batch_execution.rs new file mode 100644 index 00000000..f6300440 --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/batch_execution.rs @@ -0,0 +1,210 @@ +//! Execute a bounded in-memory batch through native operators. This is also the +//! bridge for deployments whose boundary values are not yet streaming batches. +use crate::{ + operators::Operator, + plan::PhysicalDag, + runtime::{RunContext, SharedValue}, + values::Batch, + Error, +}; +use futures::{FutureExt, StreamExt}; + +/// Every input is already in memory; the chain contains native operators only. +/// This deliberately does not enter a nested executor when called from a DAG +/// adapter. I/O belongs to source operators in the surrounding execution. +pub fn evaluate_batch( + input: Batch, + operators: Vec, + context: RunContext, +) -> Result>, Error> { + let mut graph = PhysicalDag::default(); + graph.add( + 0, + vec![], + Operator::source(input.schema().clone(), vec![input])?, + )?; + let mut root = 0; + for operator in operators { + graph.add(root + 1, vec![root], operator)?; + root += 1; + } + evaluate_graph(graph, root, context) +} + +/// Bind the ordered in-memory inputs of a native multi-input operator. +pub fn evaluate_inputs( + inputs: Vec, + operator: Operator, + context: RunContext, +) -> Result>, Error> { + let mut graph = PhysicalDag::default(); + let root = inputs.len() as u64; + for (id, input) in inputs.into_iter().enumerate() { + graph.add( + id as u64, + vec![], + Operator::source(input.schema().clone(), vec![input])?, + )?; + } + graph.add(root, (0..root).collect(), operator)?; + evaluate_graph(graph, root, context) +} + +/// Evaluate a native in-memory source, including scalar sources, in the caller's scope. +pub fn evaluate_source( + source: Operator, + context: RunContext, +) -> Result>, Error> { + let mut graph = PhysicalDag::default(); + graph.add(0, vec![], source)?; + evaluate_graph(graph, 0, context) +} + +fn evaluate_graph( + graph: PhysicalDag<'_, Batch, crate::values::Schema>, + root: crate::plan::NodeId, + context: RunContext, +) -> Result>, Error> { + let mut output = graph.execute(&[root], context)?.remove(0); + let mut batches = Vec::new(); + loop { + match output.next().now_or_never() { + Some(Some(Ok(batch))) => batches.push(batch), + Some(Some(Err(error))) => return Err(error), + Some(None) => return Ok(batches), + // Native operators have no I/O sources here. Pending is the + // shared runtime's cooperative yield after a batch quantum. + None => continue, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::{ + operators::Expression, + runtime::{Limits, Scope}, + values::Value, + }; + use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, + }; + use std::sync::Arc; + + // Engine adapters can run the identical native chain from an outer executor. + #[test] + fn same_native_chain_inside_query_and_ingestion_execution() { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }); + for scope in [ + Scope::Query { + evaluation_time_ms: 20, + revision: 1, + }, + Scope::Ingestion { + window_start_ms: 10, + window_end_ms: 20, + revision: 1, + }, + ] { + let batch = Batch::try_new(schema.clone(), vec![vec![Value::Float64(7.)]]).unwrap(); + let negate = Operator::project( + schema.clone(), + vec![( + "value".into(), + Expression::Negate(Box::new(Expression::Column(0))), + )], + ) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let result = futures::executor::block_on(async { + evaluate_batch(batch, vec![negate], context.clone()) + }) + .unwrap(); + assert!(matches!(result[0].rows()[0][0], Value::Float64(-7.))); + let source = Operator::scalar(Value::Float64(9.), DataType::Float64).unwrap(); + let scalar = evaluate_source(source, context).unwrap(); + assert!(matches!(scalar[0].rows()[0][0], Value::Float64(9.))); + } + } + + // Native sources may cross the runtime's cooperative batch quantum. + #[test] + fn in_memory_source_drives_cooperative_yields() { + let schema = Arc::new(SummarySchema { + fields: vec![], + time_index: None, + }); + let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); + let source = Operator::source(schema, vec![batch; 65]).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + assert_eq!(evaluate_source(source, context).unwrap().len(), 65); + } + + // An adapter-held output must retain its parent's reservation after execution. + #[test] + fn returned_batches_keep_their_resource_reservation() { + let schema = Arc::new(SummarySchema { + fields: vec![], + time_index: None, + }); + let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); + let bytes = batch.bytes(); + let source = Operator::source(schema, vec![batch]).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_bytes: bytes, + max_buffered_batches: 1, + }, + ) + .unwrap(); + let held = evaluate_source(source.clone(), context.clone()).unwrap(); + assert_eq!(context.retained_bytes(), bytes); + assert!(evaluate_source(source.clone(), context.clone()).is_err()); + drop(held); + assert_eq!(context.retained_bytes(), 0); + assert!(evaluate_source(source, context).is_ok()); + } + + // A cancelled surrounding execution also prevents its native computation. + #[test] + fn cancellation_is_not_bypassed_by_in_memory_execution() { + let schema = Arc::new(SummarySchema { + fields: vec![], + time_index: None, + }); + let batch = Batch::try_new(schema, vec![vec![]]).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + context.cancel(); + assert!(matches!( + evaluate_batch(batch, vec![], context), + Err(Error::Cancelled) + )); + } +} diff --git a/crates/asap-physical-operators/src/runtime/context.rs b/crates/asap-physical-operators/src/runtime/context.rs new file mode 100644 index 00000000..c145a457 --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/context.rs @@ -0,0 +1,133 @@ +use crate::Error; +use std::{ + cell::{Cell, RefCell}, + rc::Rc, + task::Waker, +}; +/// Scope is part of an execution instance, never mutable state in a reusable plan. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Scope { + Ingestion { + window_start_ms: i64, + window_end_ms: i64, + revision: u64, + }, + Query { + evaluation_time_ms: i64, + revision: u64, + }, +} +#[derive(Clone, Debug)] +pub struct Limits { + pub max_buffered_batches: usize, + pub max_bytes: usize, +} +impl Default for Limits { + fn default() -> Self { + Self { + max_buffered_batches: 8, + max_bytes: 64 * 1024 * 1024, + } + } +} +pub(super) struct Control { + cancelled: Cell, + bytes: Cell, + peak: Cell, + pub(super) limits: Limits, + waiters: RefCell>, +} +#[derive(Clone)] +pub struct RunContext { + pub scope: Scope, + pub(super) control: Rc, +} +impl RunContext { + pub fn new(scope: Scope, limits: Limits) -> Result { + if limits.max_buffered_batches == 0 || limits.max_bytes == 0 { + return Err(Error::Invalid("execution limits must be positive".into())); + } + if matches!(&scope, Scope::Ingestion { window_start_ms, window_end_ms, .. } if window_start_ms > window_end_ms) + { + return Err(Error::Invalid("inverted ingestion window".into())); + } + Ok(Self { + scope, + control: Rc::new(Control { + cancelled: Cell::new(false), + bytes: Cell::new(0), + peak: Cell::new(0), + limits, + waiters: RefCell::new(Vec::new()), + }), + }) + } + pub fn cancel(&self) { + self.control.cancelled.set(true); + for waiter in self.control.waiters.borrow_mut().drain(..) { + waiter.wake(); + } + } + pub fn is_cancelled(&self) -> bool { + self.control.cancelled.get() + } + pub fn retained_bytes(&self) -> usize { + self.control.bytes.get() + } + pub fn peak_bytes(&self) -> usize { + self.control.peak.get() + } + pub fn reserve(&self, bytes: usize) -> Result { + let total = self + .control + .bytes + .get() + .checked_add(bytes) + .ok_or(Error::MemoryLimit)?; + if total > self.control.limits.max_bytes { + return Err(Error::MemoryLimit); + } + self.control.bytes.set(total); + self.control.peak.set(self.control.peak.get().max(total)); + Ok(Reservation { + bytes, + control: Rc::clone(&self.control), + }) + } + pub(super) fn register(&self, waker: &Waker) { + let mut waiters = self.control.waiters.borrow_mut(); + if !waiters.iter().any(|old| old.will_wake(waker)) { + waiters.push(waker.clone()); + } + } +} +pub struct Reservation { + bytes: usize, + pub(super) control: Rc, +} +impl Reservation { + /// Adjust an operator-owned allocation without accumulating bookkeeping entries. + pub fn resize(&mut self, bytes: usize) -> Result<(), Error> { + let total = self + .control + .bytes + .get() + .checked_sub(self.bytes) + .and_then(|total| total.checked_add(bytes)) + .ok_or(Error::MemoryLimit)?; + if total > self.control.limits.max_bytes { + return Err(Error::MemoryLimit); + } + self.control.bytes.set(total); + self.control.peak.set(self.control.peak.get().max(total)); + self.bytes = bytes; + Ok(()) + } +} +impl Drop for Reservation { + fn drop(&mut self) { + self.control + .bytes + .set(self.control.bytes.get().saturating_sub(self.bytes)); + } +} diff --git a/crates/asap-physical-operators/src/runtime/cooperative.rs b/crates/asap-physical-operators/src/runtime/cooperative.rs new file mode 100644 index 00000000..fae869d6 --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/cooperative.rs @@ -0,0 +1,40 @@ +//! Worker-local CPU loops yield so other consumers and cancellation can progress. +use super::RunContext; +use crate::Error; +use std::task::Poll; + +pub(crate) struct Cooperative { + context: RunContext, + remaining: usize, +} +impl Cooperative { + pub(crate) fn new(context: &RunContext) -> Self { + Self { + context: context.clone(), + remaining: 1024, + } + } + pub(crate) async fn checkpoint(&mut self) -> Result<(), Error> { + if self.context.is_cancelled() { + return Err(Error::Cancelled); + } + self.remaining -= 1; + if self.remaining == 0 { + self.remaining = 1024; + let mut yielded = false; + futures::future::poll_fn(|cx| { + if self.context.is_cancelled() { + return Poll::Ready(Err(Error::Cancelled)); + } + if yielded { + return Poll::Ready(Ok(())); + } + yielded = true; + cx.waker().wake_by_ref(); + Poll::Pending + }) + .await?; + } + Ok(()) + } +} diff --git a/crates/asap-physical-operators/src/runtime/mod.rs b/crates/asap-physical-operators/src/runtime/mod.rs new file mode 100644 index 00000000..f72a57ef --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/mod.rs @@ -0,0 +1,272 @@ +//! Per-run producer sharing, streams, backpressure and resource ownership. +use crate::{ + plan::{NodeId, PhysicalDag}, + Error, +}; +use futures::{stream::LocalBoxStream, Stream}; +use std::{ + cell::RefCell, + collections::{BTreeMap, VecDeque}, + fmt::Debug, + pin::Pin, + rc::Rc, + sync::Arc, + task::{Context, Poll, Waker}, +}; +mod context; +pub use context::{Limits, Reservation, RunContext, Scope}; +pub type OutputStream<'a, V> = LocalBoxStream<'a, Result>; +/// An output owns its memory reservation even after it leaves the DAG's queue. +pub struct SharedValue { + value: Arc, + _reservation: Rc, +} +impl Clone for SharedValue { + fn clone(&self) -> Self { + Self { + value: Arc::clone(&self.value), + _reservation: Rc::clone(&self._reservation), + } + } +} +impl std::ops::Deref for SharedValue { + type Target = V; + fn deref(&self) -> &V { + &self.value + } +} +impl SharedValue { + pub fn value(&self) -> &V { + &self.value + } +} + +pub(crate) fn execute<'r, V: 'r, S: Clone + PartialEq + Debug + 'r>( + dag: &'r PhysicalDag<'_, V, S>, + roots: &[NodeId], + context: RunContext, +) -> Result>, Error> { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + dag.validate(roots)?; + let mut pending = roots.to_vec(); + let mut visited = std::collections::BTreeSet::new(); + while let Some(id) = pending.pop() { + if visited.insert(id) { + let node = &dag.nodes[&id]; + node.operator.validate_context(&context)?; + pending.extend(node.inputs.iter().copied()); + } + } + fn build<'r, V: 'r, S: 'r>( + dag: &'r PhysicalDag<'_, V, S>, + id: NodeId, + context: &RunContext, + states: &mut BTreeMap>>>, + ) -> Result>>, Error> { + if let Some(state) = states.get(&id) { + return Ok(Rc::clone(state)); + } + let node = &dag.nodes[&id]; + let mut inputs = Vec::new(); + for &child in &node.inputs { + inputs.push(Input::subscribe(build(dag, child, context, states)?)); + } + let stream = node + .operator + .start(inputs, context.clone()) + .map_err(|source| Error::AtNode { + node: id, + operation: node.operator.name().into(), + source: Box::new(source), + })?; + let op = node.operator.as_ref(); + let state = Rc::new(RefCell::new(Producer { + stream: Some(stream), + node: id, + operation: node.operator.name().into(), + size: Box::new(move |value| op.output_bytes(value)), + context: context.clone(), + queue: VecDeque::new(), + base: 0, + next_reader: 0, + batches_polled: 0, + readers: BTreeMap::new(), + waiters: BTreeMap::new(), + finished: false, + failure: None, + })); + states.insert(id, Rc::clone(&state)); + Ok(state) + } + let mut states = BTreeMap::new(); + roots + .iter() + .map(|&id| build(dag, id, &context, &mut states).map(Input::subscribe)) + .collect() +} +struct Producer<'a, V> { + node: NodeId, + operation: String, + stream: Option>, + size: Box usize + 'a>, + context: RunContext, + queue: VecDeque>, + base: u64, + next_reader: u64, + batches_polled: usize, + readers: BTreeMap, + waiters: BTreeMap, + finished: bool, + failure: Option, +} +impl Producer<'_, V> { + fn trim(&mut self) { + let minimum = self + .readers + .values() + .copied() + .min() + .unwrap_or(self.base + self.queue.len() as u64); + while self.base < minimum { + self.queue.pop_front(); + self.base += 1; + } + for (_, waker) in std::mem::take(&mut self.waiters) { + waker.wake(); + } + if self.readers.is_empty() { + self.stream = None; + self.queue.clear(); + } + } +} +pub struct Input<'a, V> { + producer: Rc>>, + reader: u64, + done: bool, +} +impl<'a, V> Input<'a, V> { + fn subscribe(producer: Rc>>) -> Self { + let reader = { + let mut state = producer.borrow_mut(); + let id = state.next_reader; + state.next_reader += 1; + let base = state.base; + state.readers.insert(id, base); + id + }; + Self { + producer, + reader, + done: false, + } + } +} +impl Drop for Input<'_, V> { + fn drop(&mut self) { + let mut state = self.producer.borrow_mut(); + state.readers.remove(&self.reader); + state.waiters.remove(&self.reader); + state.trim(); + } +} +impl Stream for Input<'_, V> { + type Item = Result, Error>; + fn poll_next(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + let this = self.get_mut(); + if this.done { + return Poll::Ready(None); + } + let mut state = this.producer.borrow_mut(); + state.context.register(cx.waker()); + if state.context.is_cancelled() { + state.failure = Some(Error::Cancelled); + state.finished = true; + state.stream = None; + state.queue.clear(); + } + let position = state.readers[&this.reader]; + let index = (position - state.base) as usize; + if let Some(value) = state.queue.get(index).cloned() { + state.readers.insert(this.reader, position + 1); + state.trim(); + return Poll::Ready(Some(Ok(value))); + } + if state.finished { + this.done = true; + state.readers.remove(&this.reader); + let failure = state.failure.clone(); + state.trim(); + return Poll::Ready(failure.map(Err)); + } + state.waiters.insert(this.reader, cx.waker().clone()); + if state.queue.len() >= state.context.control.limits.max_buffered_batches { + return Poll::Pending; + } + // Always-ready sources must still give cancellation and other roots a turn. + if state.batches_polled >= 32 { + state.batches_polled = 0; + cx.waker().wake_by_ref(); + return Poll::Pending; + } + let polled = state + .stream + .as_mut() + .expect("unfinished producer") + .as_mut() + .poll_next(cx); + if matches!(&polled, Poll::Ready(Some(Ok(_)))) { + state.batches_polled += 1; + } + match polled { + Poll::Pending => Poll::Pending, + Poll::Ready(Some(Ok(value))) => match state.context.reserve((state.size)(&value)) { + Ok(reservation) => { + let value = SharedValue { + value: Arc::new(value), + _reservation: Rc::new(reservation), + }; + state.queue.push_back(value.clone()); + state.readers.insert(this.reader, position + 1); + state.trim(); + Poll::Ready(Some(Ok(value))) + } + Err(error) => { + state.failure = Some(error.clone()); + state.finished = true; + state.stream = None; + this.done = true; + state.readers.remove(&this.reader); + state.trim(); + Poll::Ready(Some(Err(error))) + } + }, + Poll::Ready(result) => { + let error = result.and_then(Result::err).map(|source| match source { + Error::AtNode { .. } | Error::Cancelled | Error::MemoryLimit => source, + source => Error::AtNode { + node: state.node, + operation: state.operation.clone(), + source: Box::new(source), + }, + }); + state.failure = error.clone(); + state.finished = true; + state.stream = None; + this.done = true; + state.readers.remove(&this.reader); + state.trim(); + Poll::Ready(error.map(Err)) + } + } + } +} + +pub mod batch_execution; +#[cfg(test)] +mod tests; + +mod cooperative; +pub(crate) use cooperative::Cooperative; diff --git a/crates/asap-physical-operators/src/runtime/tests.rs b/crates/asap-physical-operators/src/runtime/tests.rs new file mode 100644 index 00000000..f683041c --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/tests.rs @@ -0,0 +1,262 @@ +use super::*; +use crate::plan::PhysicalOperator; +use futures::{executor::block_on, stream, StreamExt}; +use std::cell::Cell; + +struct Source { + starts: Rc>, + polls: Rc>, + fail: bool, + end: u64, +} +impl PhysicalOperator for Source { + fn name(&self) -> &str { + "CountingSource" + } + fn input_schemas(&self) -> Vec<()> { + vec![] + } + fn output_schema(&self) {} + fn output_bytes(&self, _: &u64) -> usize { + 8 + } + fn start<'a>( + &'a self, + _: Vec>, + _: RunContext, + ) -> Result, Error> { + self.starts.set(self.starts.get() + 1); + Ok(stream::iter(0..self.end) + .map(move |n| { + self.polls.set(self.polls.get() + 1); + if self.fail && n == 1 { + Err(Error::Operator("source failure".into())) + } else { + Ok(n) + } + }) + .boxed_local()) + } +} +struct Identity; +impl PhysicalOperator for Identity { + fn name(&self) -> &str { + "Identity" + } + fn input_schemas(&self) -> Vec<()> { + vec![()] + } + fn output_schema(&self) {} + fn output_bytes(&self, _: &u64) -> usize { + 8 + } + fn start<'a>( + &'a self, + mut inputs: Vec>, + _: RunContext, + ) -> Result, Error> { + Ok(inputs + .remove(0) + .map(|value| value.map(|v| *v)) + .boxed_local()) + } +} +fn context() -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 100, + revision: 1, + }, + Limits { + max_buffered_batches: 1, + max_bytes: 1024, + }, + ) + .unwrap() +} +fn source(fail: bool) -> (Source, Rc>, Rc>) { + let starts = Rc::new(Cell::new(0)); + let polls = Rc::new(Cell::new(0)); + ( + Source { + starts: starts.clone(), + polls: polls.clone(), + fail, + end: 4, + }, + starts, + polls, + ) +} + +// A shared producer runs once, and the slow reader bounds producer progress. +#[test] +fn shared_source_backpressure_and_reader_drop() { + let (source, starts, polls) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let context = context(); + let mut readers = dag.execute(&[0, 0], context.clone()).unwrap(); + let mut slow = readers.pop().unwrap(); + let mut fast = readers.pop().unwrap(); + assert_eq!(starts.get(), 1); + let first = block_on(fast.next()).unwrap().unwrap(); + assert_eq!(*first, 0); + let mut cx = Context::from_waker(futures::task::noop_waker_ref()); + assert!(Pin::new(&mut fast).poll_next(&mut cx).is_pending()); + assert_eq!(polls.get(), 1); + let same = block_on(slow.next()).unwrap().unwrap(); + assert!(Arc::ptr_eq(&first.value, &same.value)); + drop(same); + drop(first); + assert_eq!(context.retained_bytes(), 0); + assert_eq!(*block_on(fast.next()).unwrap().unwrap(), 1); + drop(slow); + assert_eq!(*block_on(fast.next()).unwrap().unwrap(), 2); + assert_eq!(*block_on(fast.next()).unwrap().unwrap(), 3); + assert!(block_on(fast.next()).is_none()); + assert_eq!(polls.get(), 4); + drop(fast); + assert_eq!(context.retained_bytes(), 0); +} + +// Independent branches consume a common node concurrently without duplicate work. +#[test] +fn diamond_and_run_isolation() { + let (source, starts, polls) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + dag.add(1, vec![0], Identity).unwrap(); + dag.add(2, vec![0], Identity).unwrap(); + for _ in 0..2 { + let mut outputs = dag.execute(&[1, 2], context()).unwrap(); + let a = outputs.pop().unwrap(); + let b = outputs.pop().unwrap(); + let (a, b) = + block_on(async { futures::join!(a.collect::>(), b.collect::>()) }); + assert_eq!( + a.iter().map(|v| **v.as_ref().unwrap()).collect::>(), + vec![0, 1, 2, 3] + ); + assert_eq!( + b.iter().map(|v| **v.as_ref().unwrap()).collect::>(), + vec![0, 1, 2, 3] + ); + } + assert_eq!(starts.get(), 2); + assert_eq!(polls.get(), 8); +} + +// Failure reaches every subscriber; cancellation stops further producer work. +#[test] +fn broadcast_error_and_cancel() { + let (source, _, polls) = source(true); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let mut outputs = dag.execute(&[0, 0], context()).unwrap(); + let a = outputs.pop().unwrap(); + let b = outputs.pop().unwrap(); + let (a, b) = block_on(async { futures::join!(a.collect::>(), b.collect::>()) }); + for values in [a, b] { + assert_eq!(values.len(), 2); + assert!(matches!(values[1], Err(Error::AtNode { node: 0, .. }))); + } + assert_eq!(polls.get(), 2); + let run = context(); + let mut output = dag.execute(&[0], run.clone()).unwrap().remove(0); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + assert!(block_on(output.next()).is_none()); + assert_eq!(polls.get(), 2); +} + +// Retaining a consumer output retains its budget lease after queue eviction. +#[test] +fn retained_outputs_count_against_budget() { + let (source, _, _) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let run = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_buffered_batches: 1, + max_bytes: 8, + }, + ) + .unwrap(); + let mut input = dag.execute(&[0], run.clone()).unwrap().remove(0); + let held = block_on(input.next()).unwrap().unwrap(); + assert_eq!(run.retained_bytes(), 8); + assert!(matches!( + block_on(input.next()), + Some(Err(Error::MemoryLimit)) + )); + drop(input); + assert_eq!(run.retained_bytes(), 8); + drop(held); + assert_eq!(run.retained_bytes(), 0); +} + +// Invalid graphs fail before even starting a source. +#[test] +fn invalid_graphs_do_not_start_sources() { + let (source, starts, _) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + dag.add(1, vec![2], Identity).unwrap(); + dag.add(2, vec![1], Identity).unwrap(); + assert!(dag.execute(&[0, 1], context()).is_err()); + assert_eq!(starts.get(), 0); + let mut missing = PhysicalDag::default(); + missing.add(1, vec![9], Identity).unwrap(); + assert!(missing.validate(&[1]).is_err()); + let mut arity = PhysicalDag::default(); + arity.add(1, vec![], Identity).unwrap(); + assert!(arity.validate(&[1]).is_err()); +} + +// An always-ready source must yield so cancellation can be polled on this worker. +#[test] +fn ready_sources_cooperate_with_cancellation() { + let (mut source, _, polls) = source(false); + source.end = 10_000; + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let context = context(); + let mut input = dag.execute(&[0], context.clone()).unwrap().remove(0); + block_on(async { + let drain = async { + while let Some(result) = input.next().await { + if let Err(error) = result { + assert_eq!(error, Error::Cancelled); + return; + } + } + panic!("source completed without yielding"); + }; + let cancel = async { + context.cancel(); + }; + futures::join!(drain, cancel); + }); + assert_eq!(polls.get(), 32); + assert_eq!(context.retained_bytes(), 0); +} + +// Cached shorter paths must not hide an over-deep path through shared nodes. +#[test] +fn depth_limit_covers_shared_paths() { + let (source, _, _) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + for id in 1..129 { + dag.add(id, vec![id - 1], Identity).unwrap(); + } + assert!(dag.validate(&(0..129).collect::>()).is_err()); +} diff --git a/crates/asap-physical-operators/src/sources/memory.rs b/crates/asap-physical-operators/src/sources/memory.rs new file mode 100644 index 00000000..7f77c54f --- /dev/null +++ b/crates/asap-physical-operators/src/sources/memory.rs @@ -0,0 +1,44 @@ +use super::*; +/// Immutable in-memory raw data. The connector owns the resident input; each +/// cursor clones only the next requested batch, not the entire data set. +pub struct MemorySource { + schema: Schema, + batches: Vec, +} +impl MemorySource { + pub fn new(schema: Schema, batches: Vec) -> Result { + crate::values::validate_schema(&schema)?; + if schema + .fields + .iter() + .any(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + { + return Err(Error::Invalid( + "raw source cannot contain summary states".into(), + )); + } + if batches.iter().any(|batch| batch.schema() != &schema) { + return Err(Error::Invalid("memory source batch schema mismatch".into())); + } + Ok(Self { schema, batches }) + } +} +impl RawSource for MemorySource { + fn boundedness(&self) -> crate::plan::Boundedness { + crate::plan::Boundedness::Bounded + } + fn schema(&self) -> Schema { + self.schema.clone() + } + fn scan(&self, context: RunContext) -> Result, Error> { + Ok(stream::iter(self.batches.iter()) + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let _allocation = context.reserve(batch.bytes())?; + Ok(batch.clone()) + }) + .boxed_local()) + } +} diff --git a/crates/asap-physical-operators/src/sources/mod.rs b/crates/asap-physical-operators/src/sources/mod.rs new file mode 100644 index 00000000..4da778b0 --- /dev/null +++ b/crates/asap-physical-operators/src/sources/mod.rs @@ -0,0 +1,187 @@ +//! Raw data access. Connectors provide rows; Scan owns Planner predicate semantics. +use crate::{ + expressions::CompiledExpression, + plan::PhysicalOperator, + runtime::{Input, OutputStream, RunContext}, + values::{Batch, Schema, Value}, + Error, +}; +use futures::{stream, StreamExt}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{DataType, QueryExpr, Source}, +}; +use std::sync::Arc; + +/// A bound data source. Metadata must be stable for the lifetime of the binding. +/// Each scan opens an independent cursor. Connectors return raw, unfiltered rows +/// and must honor cancellation and bound their own I/O buffers. Dropping a cursor +/// must release its resources. A connector error is never an empty successful scan. +pub trait RawSource { + fn schema(&self) -> Schema; + /// Declare a finite snapshot/window explicitly; execution scope alone does not bound a cursor. + fn boundedness(&self) -> crate::plan::Boundedness { + crate::plan::Boundedness::Unknown + } + fn scan(&self, context: RunContext) -> Result, Error>; +} + +/// Explicit source identities; no implicit network discovery or fallback. +#[derive(Default)] +pub struct DataSources { + sources: Vec<(Source, Arc)>, +} +impl DataSources { + pub fn register(&mut self, identity: Source, source: Arc) -> Result<(), Error> { + if self.sources.iter().any(|(key, _)| key == &identity) { + return Err(Error::Invalid("duplicate data source".into())); + } + crate::values::validate_schema(&source.schema())?; + self.sources.push((identity, source)); + Ok(()) + } + pub fn bind(&self, expression: &QueryExpr) -> Result { + let QueryExpr::Scan { + source, + predicates, + schema, + } = expression + else { + return Err(Error::Invalid( + "raw Scan requires a Planner Scan leaf".into(), + )); + }; + let output = Arc::new(SummarySchema { + fields: schema + .columns + .iter() + .map(|column| SummaryField { + name: column.name.clone(), + dtype: SummaryFamilyType::Plain(column.dtype.clone()), + nullable: column.nullable, + }) + .collect(), + time_index: schema.time_index, + }); + crate::values::validate_schema(&output)?; + let reader = self + .sources + .iter() + .find(|(key, _)| key == source) + .map(|(_, reader)| reader.clone()) + .ok_or_else(|| Error::Invalid(format!("unbound raw source: {source:?}")))?; + if reader.schema() != output { + return Err(Error::Invalid( + "raw source differs from Planner Scan schema".into(), + )); + } + let predicates = predicates + .iter() + .map(|predicate| { + let predicate = CompiledExpression::compile(&predicate.0, &output)?; + if predicate.dtype().0 != DataType::Bool { + return Err(Error::Invalid("Scan predicate must be boolean".into())); + } + Ok(predicate) + }) + .collect::, Error>>()?; + Ok(Scan { + reader, + output, + predicates, + }) + } +} + +pub struct Scan { + reader: Arc, + output: Schema, + predicates: Vec, +} +impl PhysicalOperator for Scan { + fn properties(&self, _: &[crate::plan::PlanProperties]) -> crate::plan::PlanProperties { + crate::plan::PlanProperties { + boundedness: self.reader.boundedness(), + emission: crate::plan::Emission::Incremental, + } + } + + fn name(&self) -> &str { + "Scan" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, batch: &Batch) -> usize { + batch.bytes() + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + if !inputs.is_empty() { + return Err(Error::Invalid("Scan cannot have inputs".into())); + } + if context.is_cancelled() { + return Err(Error::Cancelled); + } + // Opening is lazy: validation and construction of a run perform no I/O. + let opening = context.clone(); + let stream = stream::once(async move { + if opening.is_cancelled() { + return Err(Error::Cancelled); + } + self.reader.scan(opening) + }); + use futures::TryStreamExt; + Ok(stream + .try_flatten() + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + if batch.schema() != &self.output { + return Err(Error::Invalid( + "connector returned a different Scan schema".into(), + )); + } + if self.predicates.is_empty() { + return Ok(batch); + } + let _workspace = + context.reserve(batch.bytes().checked_mul(2).ok_or(Error::MemoryLimit)?)?; + let mut rows = Vec::new(); + for row in batch.rows() { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let mut keep = true; + for predicate in &self.predicates { + match predicate.evaluate(row)? { + Value::Bool(true) => {} + Value::Bool(false) | Value::Null => { + keep = false; + break; + } + _ => { + return Err(Error::Invalid("Scan predicate is not boolean".into())) + } + } + } + if keep { + rows.push(row.clone()); + } + } + Batch::try_new(self.output.clone(), rows) + }) + .boxed_local()) + } +} + +mod memory; +pub use memory::MemorySource; diff --git a/crates/asap-physical-operators/src/summary_kernels/exact.rs b/crates/asap-physical-operators/src/summary_kernels/exact.rs index 3343e875..cb9d6099 100644 --- a/crates/asap-physical-operators/src/summary_kernels/exact.rs +++ b/crates/asap-physical-operators/src/summary_kernels/exact.rs @@ -115,6 +115,28 @@ impl ExactAccumulator { pub fn family(&self) -> &SummaryFamilyType { &self.family } + pub(crate) fn insufficient_counter_samples( + &self, + statistic: Statistic, + key: &Option, + ) -> bool { + if statistic != self.statistic() { + return false; + } + let state = match (&self.keyed, key) { + (Some(states), Some(key)) => states.get(key), + (None, None) => Some(&self.scalar), + _ => None, + }; + match state { + Some(ScalarState::Counter(None)) => true, + Some(ScalarState::Counter(Some(counter))) => { + counter.sample_count < 2 + || counter.last_seen_timestamp == counter.starting_timestamp + } + _ => false, + } + } pub fn is_keyed(&self) -> bool { self.keyed.is_some() } diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs index 6e761c11..56d43633 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -2,7 +2,7 @@ use crate::AggregateCore; use crate::Error; use planner_types::{ - post_asap::{SummaryFamilyType, SummarySchema}, + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, pre_asap::DataType, }; use std::{cmp::Ordering, sync::Arc}; @@ -247,6 +247,16 @@ impl Batch { .sum::() } } +pub(crate) fn group_key(row: &[Value], columns: &[usize]) -> Result>, Error> { + columns + .iter() + .map(|&i| { + row.get(i) + .ok_or_else(|| Error::Invalid("group column out of range".into()))? + .key() + }) + .collect() +} pub(crate) use crate::capability::validate_native_family as validate_family; @@ -328,6 +338,20 @@ pub(crate) fn validate_schema(schema: &Schema) -> Result<(), Error> { Ok(()) } +pub(crate) fn field(schema: &Schema, column: usize) -> Result<&SummaryField, Error> { + schema + .fields + .get(column) + .ok_or_else(|| Error::Invalid("column out of range".into())) +} +pub(crate) fn plain(schema: &Schema, column: usize) -> Result<(&DataType, bool), Error> { + let f = field(schema, column)?; + let SummaryFamilyType::Plain(dtype) = &f.dtype else { + return Err(Error::Invalid("plain value required".into())); + }; + Ok((dtype, f.nullable)) +} + #[cfg(test)] mod weighted_state_tests { use super::*; diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/asap-physical-operators/tests/blocking_resources.rs new file mode 100644 index 00000000..7a312892 --- /dev/null +++ b/crates/asap-physical-operators/tests/blocking_resources.rs @@ -0,0 +1,242 @@ +//! Blocking operators enforce resources before returning their first batch. +use asap_physical_operators::{ + operators::Operator, + plan::{PhysicalDag, PhysicalOperator}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Schema, Value}, + Error, +}; +use futures::{executor::block_on, FutureExt, StreamExt}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{DataType, JoinKind, Predicate, QueryExpr, ScalarValue}, +}; +use std::sync::Arc; + +fn schema(width: usize) -> Schema { + Arc::new(SummarySchema { + fields: (0..width) + .map(|i| SummaryField { + name: format!("v{i}"), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: false, + }) + .collect(), + time_index: None, + }) +} +fn context(max_bytes: usize) -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_bytes, + ..Limits::default() + }, + ) + .unwrap() +} +fn source(n: usize) -> PhysicalDag<'static, Batch, Schema> { + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + schema(1), + vec![Batch::try_new(schema(1), vec![vec![Value::Int64(1)]; n]).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag +} +fn cross_join() -> Operator { + Operator::relational_join( + schema(1), + schema(1), + JoinKind::Cross, + &Predicate(std::rc::Rc::new(QueryExpr::Literal(ScalarValue::Boolean( + true, + )))), + schema(2), + ) + .unwrap() +} + +// Even callers starting an operator directly cannot bypass its workspace budget. +#[test] +fn join_reserves_result_growth_before_returning_output() { + let sources = source(64); + let run = context(32 * 1024); + let inputs = sources.execute(&[0, 0], run.clone()).unwrap(); + let join = cross_join(); + let mut output = join.start(inputs, run.clone()).unwrap(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::MemoryLimit)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); +} + +// A single large input batch must not monopolize the worker during a join. +#[test] +fn join_yields_during_computation_and_observes_cancellation() { + let sources = source(64); + let run = context(16 * 1024 * 1024); + let inputs = sources.execute(&[0, 0], run.clone()).unwrap(); + let join = cross_join(); + let mut output = join.start(inputs, run.clone()).unwrap(); + assert!( + output.next().now_or_never().is_none(), + "join should yield before producing all 4096 rows" + ); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); +} + +// Sorting and grouping yield even for one large batch. +#[test] +fn blocking_reductions_yield_and_release_memory_on_cancellation() { + use asap_physical_operators::{ + operators::{Reduction, SortKey}, + plan::PhysicalOperator, + }; + let operators = vec![ + Operator::sort( + schema(1), + vec![SortKey { + column: 0, + descending: false, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + Operator::aggregate(schema(1), vec![], vec![("sum".into(), Reduction::Sum(0))]).unwrap(), + ]; + for operator in operators { + let sources = source(768); + let run = context(16 * 1024 * 1024); + let inputs = sources.execute(&[0], run.clone()).unwrap(); + let mut output = operator.start(inputs, run.clone()).unwrap(); + assert!(output.next().now_or_never().is_none()); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); + } +} + +// Merge-sort rounds preserve input order for tied keys across chunk boundaries. +#[test] +fn cooperative_sort_preserves_ties_across_chunks() { + use asap_physical_operators::operators::SortKey; + let batch = Batch::try_new( + schema(2), + (0..1025) + .rev() + .map(|i| vec![Value::Int64(i % 3), Value::Int64(i)]) + .collect(), + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(schema(2), vec![batch]).unwrap()) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::sort( + schema(2), + vec![SortKey { + column: 0, + descending: false, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + ) + .unwrap(); + let mut output = dag + .execute(&[1], context(16 * 1024 * 1024)) + .unwrap() + .remove(0); + let batch = block_on(output.next()).unwrap().unwrap(); + let expected = (0..3) + .flat_map(|key| (0..1025).rev().filter(move |i| i % 3 == key)) + .collect::>(); + for (row, expected) in batch.rows().iter().zip(expected) { + assert!(matches!(row[1], Value::Int64(i) if i == expected)); + } + assert_eq!(batch.rows().len(), 1025); +} + +// The integrated weighted-summary path obeys the same cooperative cancellation contract. +#[test] +fn weighted_summary_build_yields_within_a_batch() { + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let input = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "item".into(), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: false, + }, + SummaryField { + name: "weight".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }, + ], + time_index: None, + }); + let mut sources = PhysicalDag::default(); + let batch = Batch::try_new( + input.clone(), + (0..1500) + .map(|i| vec![Value::Int64(i % 8), Value::Float64(0.25)]) + .collect(), + ) + .unwrap(); + sources + .add( + 0, + vec![], + Operator::source(input.clone(), vec![batch]).unwrap(), + ) + .unwrap(); + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 64, + depth: 3, + heap_size: 8, + }, + ), + Default::default(), + ); + let operator = Operator::keyed_summary_build(input, family, 1, vec![0], vec![]).unwrap(); + let run = context(16 * 1024 * 1024); + let inputs = sources.execute(&[0], run.clone()).unwrap(); + let mut output = operator.start(inputs, run.clone()).unwrap(); + assert!(output.next().now_or_never().is_none()); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); +} diff --git a/crates/asap-physical-operators/tests/plan_properties.rs b/crates/asap-physical-operators/tests/plan_properties.rs new file mode 100644 index 00000000..f9a509c2 --- /dev/null +++ b/crates/asap-physical-operators/tests/plan_properties.rs @@ -0,0 +1,151 @@ +//! Finite-input contracts are validated before source execution. +use asap_physical_operators::{ + operators::{Operator, SortKey}, + plan::{Boundedness, Emission, PhysicalDag}, + runtime::{Limits, OutputStream, RunContext, Scope}, + sources::{DataSources, RawSource}, + values::{Batch, Schema}, + Error, +}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{Column, DataType, QueryExpr, Schema as LogicalSchema, Source}, +}; +use std::sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, +}; +struct DeclaredSource { + schema: Schema, + boundedness: Boundedness, + opens: Arc, +} +impl RawSource for DeclaredSource { + fn schema(&self) -> Schema { + self.schema.clone() + } + fn boundedness(&self) -> Boundedness { + self.boundedness + } + fn scan(&self, _: RunContext) -> Result, Error> { + self.opens.fetch_add(1, Ordering::SeqCst); + Ok(Box::pin(futures::stream::empty())) + } +} +// A blocking parent must reject unknown and unbounded Scan inputs without opening a reader. +#[test] +fn blocking_inputs_require_an_explicit_finite_source() { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "v".into(), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: false, + }], + time_index: None, + }); + for boundedness in [ + Boundedness::Unknown, + Boundedness::Unbounded, + Boundedness::Bounded, + ] { + let opens = Arc::new(AtomicUsize::new(0)); + let mut registry = DataSources::default(); + let identity = Source::Table { + table_ref: "t".into(), + }; + registry + .register( + identity.clone(), + Arc::new(DeclaredSource { + schema: schema.clone(), + boundedness, + opens: opens.clone(), + }), + ) + .unwrap(); + let scan = registry + .bind(&QueryExpr::Scan { + source: identity, + schema: LogicalSchema::new(vec![Column::new("v", DataType::Int64, false)]), + predicates: vec![], + }) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], scan).unwrap(); + dag.add( + 1, + vec![0], + Operator::sort( + schema.clone(), + vec![SortKey { + column: 0, + descending: false, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + ) + .unwrap(); + let run = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + if boundedness == Boundedness::Bounded { + let properties = dag.properties(&[1]).unwrap(); + assert_eq!(properties[&1].emission, Emission::AfterInput); + assert_eq!(properties[&1].boundedness, Boundedness::Bounded); + assert!(dag.execute(&[1], run).is_ok()); + } else { + assert!( + matches!(dag.execute(&[1], run), Err(Error::Invalid(message)) if message.contains("requires bounded inputs")) + ); + } + assert_eq!(opens.load(Ordering::SeqCst), 0); + } +} + +// Kernel support must not be mistaken for executable native state/readout support. +#[test] +fn summary_capability_levels_are_distinct() { + use asap_physical_operators::{ + capability::{validate_native_family, validate_sketch_readout, validate_summary_kernel}, + planner::post_asap::SketchQuery, + }; + use planner_types::{ + post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate}, + pre_asap::ColumnRef, + }; + let grouping = GroupingStrategy::default(); + let cms = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::Cms, + SketchParams::Cms { + width: 64, + depth: 4, + }, + ), + grouping.clone(), + ); + let update = SummaryUpdate { + item: Some(planner_types::post_asap::SummaryInputExpr::Column( + ColumnRef::Named("host".into()), + )), + weight: planner_types::post_asap::SummaryInputExpr::Constant(1.0), + weight_domain: Default::default(), + }; + assert!(validate_summary_kernel(&cms, &update, &grouping).is_ok()); + assert!(validate_native_family(&cms).is_err()); + let kll = SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 128 }), + grouping, + ); + assert!(validate_native_family(&kll).is_ok()); + assert!(validate_sketch_readout(&kll, &SketchQuery::Quantile { q: 1.5 }).is_err()); + assert!(validate_sketch_readout(&kll, &SketchQuery::Cardinality).is_err()); + assert!(validate_sketch_readout(&kll, &SketchQuery::Quantile { q: 0.5 }).is_ok()); +} From 0ced125a2d054011fe4ea3ca1068c6894514c31c Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 20:07:17 +0000 Subject: [PATCH 03/59] feat: compile Planner selections into physical DAG candidates Add `physical_planner`: reader-independent compilation of selected logical candidates into precompute/query physical DAGs with typed materialization frontiers, bounded frontier enumeration, workload cost selection, temporal KLL pane compilation and PromQL row/value lowering. `dag` remains a compatibility re-export. Tests cover physical DAGs, plan recovery, precompute candidates and populations, PromQL values and binaries, weighted TopK binding and current-series heaps, plus Planner-to-execution integration tests. The design doc describes the ownership boundary with deployments. Co-Authored-By: Claude Opus 5.5 --- Cargo.lock | 2 + crates/asap-physical-operators/README.md | 130 ++ crates/asap-physical-operators/src/dag/mod.rs | 8 + crates/asap-physical-operators/src/lib.rs | 14 +- .../src/operators/joins/mod.rs | 9 + .../src/operators/mod.rs | 36 + .../src/physical_planner/candidates.rs | 275 ++++ .../src/physical_planner/compiled.rs | 310 ++++ .../src/physical_planner/mod.rs | 900 ++++++++++ .../src/physical_planner/precompute.rs | 384 +++++ .../src/physical_planner/promql_rows.rs | 389 +++++ .../src/physical_planner/promql_values.rs | 278 ++++ .../src/physical_planner/temporal_panes.rs | 331 ++++ .../tests/current_series_heap.rs | 401 +++++ .../tests/physical_dag.rs | 1443 +++++++++++++++++ .../tests/physical_plan_recovery.rs | 105 ++ .../tests/physical_semantics.rs | 695 ++++++++ .../tests/precompute_candidates.rs | 548 +++++++ .../tests/precompute_population.rs | 425 +++++ .../tests/promql_binary.rs | 270 +++ .../tests/promql_values.rs | 492 ++++++ .../asap-physical-operators/tests/raw_scan.rs | 386 +++++ .../tests/summary_projection.rs | 161 ++ .../tests/weighted_topk_binding.rs | 1018 ++++++++++++ crates/integration-tests/Cargo.toml | 3 + .../tests/kll_pane_execution.rs | 286 ++++ .../tests/physical_common/mod.rs | 42 + .../tests/sql_to_physical.rs | 165 ++ .../summary_maintenance_lifecycle_e2e.rs | 730 ++++++++- .../physical-planning-and-deployment.md | 512 ++++++ docs/develop_docs/native-promql-inputs.md | 55 + 31 files changed, 10751 insertions(+), 52 deletions(-) create mode 100644 crates/asap-physical-operators/README.md create mode 100644 crates/asap-physical-operators/src/dag/mod.rs create mode 100644 crates/asap-physical-operators/src/physical_planner/candidates.rs create mode 100644 crates/asap-physical-operators/src/physical_planner/compiled.rs create mode 100644 crates/asap-physical-operators/src/physical_planner/mod.rs create mode 100644 crates/asap-physical-operators/src/physical_planner/precompute.rs create mode 100644 crates/asap-physical-operators/src/physical_planner/promql_rows.rs create mode 100644 crates/asap-physical-operators/src/physical_planner/promql_values.rs create mode 100644 crates/asap-physical-operators/src/physical_planner/temporal_panes.rs create mode 100644 crates/asap-physical-operators/tests/current_series_heap.rs create mode 100644 crates/asap-physical-operators/tests/physical_dag.rs create mode 100644 crates/asap-physical-operators/tests/physical_plan_recovery.rs create mode 100644 crates/asap-physical-operators/tests/physical_semantics.rs create mode 100644 crates/asap-physical-operators/tests/precompute_candidates.rs create mode 100644 crates/asap-physical-operators/tests/precompute_population.rs create mode 100644 crates/asap-physical-operators/tests/promql_binary.rs create mode 100644 crates/asap-physical-operators/tests/promql_values.rs create mode 100644 crates/asap-physical-operators/tests/raw_scan.rs create mode 100644 crates/asap-physical-operators/tests/summary_projection.rs create mode 100644 crates/asap-physical-operators/tests/weighted_topk_binding.rs create mode 100644 crates/integration-tests/tests/kll_pane_execution.rs create mode 100644 crates/integration-tests/tests/physical_common/mod.rs create mode 100644 crates/integration-tests/tests/sql_to_physical.rs create mode 100644 docs/design_docs/physical-planning-and-deployment.md create mode 100644 docs/develop_docs/native-promql-inputs.md diff --git a/Cargo.lock b/Cargo.lock index a063dd78..7f0a3755 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -368,8 +368,10 @@ dependencies = [ "asap-aware-mapping", "asap-frontend-promql", "asap-frontend-sql", + "asap-physical-operators", "asap-types", "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", + "futures", "serde_json", "tokio", ] diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md new file mode 100644 index 00000000..5111d316 --- /dev/null +++ b/crates/asap-physical-operators/README.md @@ -0,0 +1,130 @@ +# ASAP physical operators + +An independent Rust physical operator DAG runtime shared by ingestion time and +query time execution. The library requires neither backend engine, a server, +a storage implementation, Arrow nor DataFusion. DataFusion informed the design; +it is not the execution framework. + +`plan::PhysicalDag` binds typed operator inputs to node IDs. Each execution starts +one producer per reachable node, shares output batches among its consumers, and +bounds buffering. Dropping one consumer does not cancel other consumers. A +`RunContext` carries query or ingestion scope, cancellation and byte accounting. +Executions use the caller's worker and worker-local streams, with no internal +thread pool. Poll multiple root streams concurrently when they share inputs. + +`operators::Operator` implements native batch sources, scalar values, +projection, filtering, grouped exact aggregation, semi-join, grouped Sort and +Limit, vector-to-scalar conversion, Union, and summary construction/merge/readout. +Sort followed by Limit implements grouped ranking; no dedicated TopK physical +operator is needed. Summary construction updates state batch by batch. End of +input means the supplied query range or ingestion window is complete. + +```rust +use asap_physical_operators::{ + expressions::Expression, + operators::Operator, + values::Value, + plan::PhysicalDag, + runtime::{Limits, RunContext, Scope}, +}; +use asap_physical_operators::planner::pre_asap::DataType; +use futures::{executor::block_on, StreamExt}; + +let source = Operator::scalar(Value::Int64(7), DataType::Int64)?; +let negate = Operator::project(source.schema(), vec![ + ("value".into(), Expression::Negate(Box::new(Expression::Column(0)))), +])?; +let mut plan = PhysicalDag::default(); +plan.add(0, vec![], source)?; +plan.add(1, vec![0], negate)?; +let run = RunContext::new( + Scope::Query { evaluation_time_ms: 1000, revision: 1 }, + Limits::default(), +)?; +let mut output = plan.execute(&[1], run)?.remove(0); +let batch = block_on(output.next()).unwrap()?; +assert!(matches!(batch.rows()[0][0], Value::Int64(-7))); +# Ok::<(), asap_physical_operators::dag::Error>(()) +``` + +`physical_planner::compile` accepts a logical Post-ASAP DAG (`PostAsapDag`) and typed input contracts. +The resulting candidate is instantiated with deployment readers after selection. It rejects unsupported operations and +schema mismatches before starting a source. Implement `PhysicalOperator` for a +deployment source, including asynchronous I/O; computation operators remain in +the library. The public `planner` export identifies the exact Planner types used +by the crate. The physical compiler currently supports a subset of those types and +operations; it does not interpret an unknown node as external fallback. + +Plain values preserve Planner scalar/collection types and nullability. Numeric +arithmetic uses matching Int64 or Float64 inputs; integer overflow is an error. +Boolean predicates use three-valued logic. Native summary states currently cover +exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS and CountSketch with candidate heaps. Binding checks family, +parameters and readout compatibility; source batches also validate state payloads. +Existing accumulator algorithms are reused as kernels behind these operators. + +This crate is owned by ASAPPlanner. Its `planner-types` dependency is the local +IR crate, so a contract change and its execution tests belong in the same PR. +Deployments supply storage/ingestion sources and adapt output protocols. The +library has no ASAPQuery-backend dependency. Backend raw Scan remains a separate +deployment capability. + +See [the design](../../docs/design_docs/physical-planning-and-deployment.md). + +## Module boundaries + +- `plan`: immutable graph, operator interface, schemas and execution properties. +- `runtime`: per-run streams, shared producers, memory reservations and cancellation. +- `expressions`: scalar evaluation; typed builders and the Planner expression adapter. +- `operators`: projection, filter, joins, aggregate/window, sort, limit and summary implementations. +- `sources`: raw-source interface, Scan and the memory connector. +- `physical_planner`: native operator lowering, typed input contracts and checked instantiation. +- `summary_kernels`: in-memory summary state over `asap_sketchlib` and exact Planner state: merge, typed readout and update adapters. +- `readout`: readouts over merged exact summary states. +- `capability`: explicit kernel, native-batch and typed readout validation. + +The `dag`, `factory`, `traits` and `arithmetic` paths are re-exports. They contain no alternative +execution implementations. + +Sketch algorithms and their state encodings belong to `asap_sketchlib`. Wire decoding, delta +frames, edge sampling and storage statistics belong to deployments. Kernels hold one population's +state; operators own grouping. + +A source must declare `Boundedness::Bounded` to feed a blocking operator. +The default for a custom raw source is `Unknown`; query or ingestion scope alone +does not promise that its cursor ends. `PhysicalDag::properties` validates these +requirements before any source starts and returns boundedness and emission mode +for every reachable node. The memory connector declares finite input. Custom +physical sources expose the same facts through `PhysicalOperator::properties`. + +Blocking operators reserve estimated workspace and yield cooperatively during +row processing and sort merges. Cancellation releases reservations when the +stream is polled or dropped. Individual scalar evaluations, bounded sort chunks +and sketch kernel calls are synchronous; this is not preemptive execution. +There is no spill or partitioned parallel execution in this implementation. + +## Physical compilation and deployment inputs + +`physical_planner::compile` accepts a Planner `PostAsapDag`, typed +`InputContract`s and output roots. It returns a reusable `CompiledPhysicalDag` +containing selected native operators and no live readers. Compilation validates +schemas, input ordering, sharing and boundedness before deployment source access. + +A deployment calls `CompiledPhysicalDag::instantiate` with exactly the declared +inputs. This checks source schemas and execution properties and constructs the +runnable graph without repeating logical lowering. The graph executes through +the shared runtime with independent per-run state. Window coverage, revision and +maintenance-policy admission remain deployment/planning contracts; this compiler +does not discover storage or silently change a selected maintenance strategy. + +`physical_planner::compile_temporal_pane_candidate` lowers a selected continuous +KLL lifecycle and Sliding/Tumbling framework into maintenance and query DAGs. +`TemporalPaneMaintenance` supplies pane geometry and a resolved complete entity +identity contract. The compiler inserts population guards, scan predicates, +pane construction, ordered state slots, a shared merge and quantile readouts. +Pane outputs have distinct physical identities from the logical whole-window +summary, and the returned candidate retains the maintenance contract for binding. +Each run checks phase, pane timestamps and duplicate entity states. The initial +realization uses complete bounded snapshots; partial edges, exponential +histograms and cross-run delta accumulation are unsupported. Storage identities, +revision selection, completeness/readiness evidence and scheduling stay with +deployment. diff --git a/crates/asap-physical-operators/src/dag/mod.rs b/crates/asap-physical-operators/src/dag/mod.rs new file mode 100644 index 00000000..c7837672 --- /dev/null +++ b/crates/asap-physical-operators/src/dag/mod.rs @@ -0,0 +1,8 @@ +//! Compatibility imports. New code should use plan, runtime, operators, physical_planner and sources directly. +pub use crate::plan::{NodeId, PhysicalDag, PhysicalOperator}; +pub use crate::runtime::batch_execution; +pub use crate::runtime::{ + Input, Limits, OutputStream, Reservation, RunContext, Scope, SharedValue, +}; +pub use crate::Error; +pub use crate::{expressions, operators, physical_planner as planner, sources as scan, values}; diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs index 0835159c..345ee762 100644 --- a/crates/asap-physical-operators/src/lib.rs +++ b/crates/asap-physical-operators/src/lib.rs @@ -1,5 +1,4 @@ -//! Shared physical operators: summary kernels, typed values, native operators -//! and the DAG runtime. +#![doc = include_str!("../README.md")] pub mod key_by_label_values; pub mod measurement; @@ -12,20 +11,23 @@ pub use measurement::Measurement; pub use statistic::Statistic; pub use traits::*; +pub use expressions::arithmetic; pub mod capability; pub use summary_kernels::factory; /// The exact Planner contract used by these kernels. pub use planner_types as planner; +pub mod dag; + +pub mod readout; + mod error; pub use error::Error; -pub mod values; - -pub use expressions::arithmetic; pub mod expressions; pub mod operators; +pub mod physical_planner; pub mod plan; -pub mod readout; pub mod runtime; pub mod sources; +pub mod values; diff --git a/crates/asap-physical-operators/src/operators/joins/mod.rs b/crates/asap-physical-operators/src/operators/joins/mod.rs index 608fd8d7..a4397033 100644 --- a/crates/asap-physical-operators/src/operators/joins/mod.rs +++ b/crates/asap-physical-operators/src/operators/joins/mod.rs @@ -32,6 +32,15 @@ impl Operator { } self } + pub(crate) fn certified_pruning_keys(&self) -> Option<&[(usize, usize)]> { + match &self.kind { + Kind::SemiJoin { + keys, + require_complete_right: true, + } => Some(keys), + _ => None, + } + } pub fn relational_join( left: Schema, right: Schema, diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index e4c6944f..0f20fa3d 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -135,6 +135,42 @@ pub struct Operator { output: Schema, } impl Operator { + pub(crate) fn row_preserving_input(&self) -> Option { + match self.kind { + Kind::Filter(_) | Kind::Sort { .. } | Kind::Limit { .. } | Kind::SemiJoin { .. } => { + Some(0) + } + _ => None, + } + } + + pub(crate) fn is_counter_readout(&self) -> bool { + matches!( + self.kind, + Kind::Readout { + query: ReadoutQuery::Exact(crate::summary_kernels::exact::ExactReadout { + statistic: crate::Statistic::Rate | crate::Statistic::Increase, + .. + }), + .. + } + ) + } + + pub(crate) fn with_counter_lookback(mut self, lookback: i64) -> Result { + if lookback <= 0 { + return Err(invalid("counter lookback must be positive")); + } + if let Kind::Readout { + query: ReadoutQuery::Exact(readout), + .. + } = &mut self.kind + { + readout.lookback_ms = Some(lookback); + } + Ok(self) + } + /// Resolve a counter readout's logical lookback to this run's evaluation range. pub(super) fn readout_range(&self, context: &RunContext) -> Result, Error> { let Kind::Readout { diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs new file mode 100644 index 00000000..a8f476f3 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -0,0 +1,275 @@ +//! Compile maintenance-selected frontiers without deployment-specific graph rewrites. +use super::*; + +/// One computation realization; lifecycle/window/revision requirements accompany +/// it during optimization and deployment. Stored outputs have no storage identity. +/// Deserialization validates the producer/reader boundary. +#[derive(Clone, serde::Serialize, serde::Deserialize)] +#[serde(try_from = "UncheckedCandidate")] +pub struct PhysicalCandidate { + pub precompute: Option, + pub query: CompiledPhysicalDag, + pub materialized_outputs: BTreeMap, +} + +/// Compile an explicit materialization frontier selected by Planner maintenance +/// search. Operators upstream of that frontier run in precompute, including +/// readouts/reductions; query execution receives their typed output values. +/// Empty frontiers retain the full computation in the query DAG. +/// +/// Repeated windows must be instantiated with the same evaluation/population +/// contract used to build each output. This API never treats a result from a +/// different window or revision as interchangeable merely because types match. +pub fn compile_candidate( + dag: &PostAsapDag, + inputs: BTreeMap, + roots: &[NodeId], + frontier: &[NodeId], +) -> Result { + if frontier.is_empty() { + return Ok(PhysicalCandidate { + precompute: None, + query: compile(dag, inputs, roots)?, + materialized_outputs: BTreeMap::new(), + }); + } + let frontier_set: BTreeSet<_> = frontier.iter().copied().collect(); + if frontier_set.len() != frontier.len() || frontier.iter().any(|id| inputs.contains_key(id)) { + return Err(invalid("frontier must contain distinct computed outputs")); + } + let full = compile(dag, inputs.clone(), roots)?; + let precompute = compile(dag, inputs.clone(), frontier)?; + let mut materialized_outputs = BTreeMap::new(); + for &id in frontier { + // Also proves that the frontier is reachable from the requested roots. + full.output_contract(id)?; + let mut output = precompute.output_contract(id)?; + if output.properties.boundedness != Boundedness::Bounded { + return Err(invalid("materialized output requires bounded execution")); + } + // A stored reader may stream batches even when the producer blocked. + // Its timing is independent; the retained result still must be finite. + output.properties.emission = Emission::Unknown; + materialized_outputs.insert(id, output); + } + let mut query_inputs = inputs; + query_inputs.extend(materialized_outputs.clone()); + let query = compile(dag, query_inputs, roots)?; + let used: BTreeSet<_> = query.input_contracts().map(|(id, _)| id).collect(); + if !frontier.iter().all(|id| used.contains(id)) { + return Err(invalid( + "frontier contains an output shadowed by another boundary", + )); + } + Ok(PhysicalCandidate { + precompute: Some(precompute), + query, + materialized_outputs, + }) +} + +/// Enumerate bounded, reachable materialization frontiers above explicit inputs. +/// Each frontier is an antichain: storing an output and its ancestor together +/// would leave the ancestor unused by query execution. Lifecycle eligibility +/// and deployment feasibility are evaluated separately before cost selection. +/// Exceeding the search budget returns an error, never a partial inventory. +pub fn enumerate_frontiers( + dag: &PostAsapDag, + inputs: &BTreeMap, + roots: &[NodeId], + max_candidates: usize, +) -> Result>, Error> { + if max_candidates == 0 { + return Err(invalid( + "frontier search requires a positive candidate budget", + )); + } + let compiled = compile(dag, inputs.clone(), roots)?; + let mut ancestors = BTreeMap::>::new(); + let mut eligible = Vec::new(); + for node in &dag.nodes { + let id = u64::from(node.id.0); + if inputs.contains_key(&id) { + continue; + } + let Ok(contract) = compiled.output_contract(id) else { + continue; + }; + if contract.properties.boundedness != Boundedness::Bounded { + continue; + } + let mut seen = BTreeSet::new(); + let mut pending = vec![id]; + while let Some(current) = pending.pop() { + if !seen.insert(current) || inputs.contains_key(¤t) { + continue; + } + pending.extend( + dag.edges + .iter() + .filter(|edge| u64::from(edge.consumer.0) == current) + .map(|edge| u64::from(edge.producer.0)), + ); + } + ancestors.insert(id, seen); + eligible.push(id); + } + eligible.sort_unstable(); + let mut frontiers = vec![vec![]]; + for id in eligible { + let additions = frontiers + .iter() + .filter(|frontier| { + frontier.iter().all(|previous| { + !ancestors[&id].contains(previous) && !ancestors[previous].contains(&id) + }) + }) + .map(|frontier| { + let mut next = frontier.clone(); + next.push(id); + next + }) + .collect::>(); + if additions.len() > max_candidates.saturating_sub(frontiers.len()) { + return Err(invalid( + "materialization frontier search exceeds candidate budget", + )); + } + frontiers.extend(additions); + } + Ok(frontiers) +} + +/// Lower every maintenance candidate before feasibility/cost evaluation. Keep +/// individual failures visible; do not substitute another computation on error. +pub fn compile_candidates( + dag: &PostAsapDag, + inputs: BTreeMap, + roots: &[NodeId], + frontiers: &[Vec], +) -> Vec> { + frontiers + .iter() + .map(|frontier| compile_candidate(dag, inputs.clone(), roots, frontier)) + .collect() +} + +/// Complete workload cost supplied by scoped optimizer/deployment evidence. +/// The evaluator includes build/update work, retained state, shared producers +/// and recurrent reads over the same horizon; these are not per-query timings. +#[derive(Clone, Debug)] +pub struct CandidateCost { + pub workload_scope: String, + pub horizon_seconds: f64, + pub total_cost: f64, +} + +pub struct CandidateSelection { + pub candidate: T, + pub candidate_index: usize, + pub cost: CandidateCost, +} + +/// Select only compiled and deployment-feasible physical candidates. `None` +/// rejects an unbindable candidate before pricing. Comparable scoped costs are +/// required; deployment never rewrites the selected frontier after this step. +/// The payload is generic so deployments can retain binding/diagnostic metadata +/// alongside each compiled computation without duplicating winner selection. +pub fn select_candidate( + candidates: Vec>, + mut evaluate: impl FnMut(&T) -> Result, Error>, +) -> Result, Error> { + let mut scope: Option<(String, f64)> = None; + let mut selected: Option> = None; + for (candidate_index, candidate) in candidates.into_iter().enumerate() { + let Ok(candidate) = candidate else { continue }; + let Some(cost) = evaluate(&candidate)? else { + continue; + }; + if cost.workload_scope.is_empty() + || !cost.horizon_seconds.is_finite() + || cost.horizon_seconds <= 0. + || !cost.total_cost.is_finite() + || cost.total_cost < 0. + { + return Err(invalid( + "candidate cost lacks a valid workload scope/horizon", + )); + } + let current_scope = (cost.workload_scope.clone(), cost.horizon_seconds); + if scope.as_ref().is_some_and(|scope| scope != ¤t_scope) { + return Err(invalid( + "candidate costs describe different workloads or horizons", + )); + } + scope = Some(current_scope); + if selected + .as_ref() + .is_none_or(|selected| cost.total_cost < selected.cost.total_cost) + { + selected = Some(CandidateSelection { + candidate, + candidate_index, + cost, + }); + } + } + selected.ok_or_else(|| invalid("no feasible priced physical candidate")) +} + +#[derive(serde::Deserialize)] +#[serde(deny_unknown_fields)] +struct UncheckedCandidate { + precompute: Option, + query: CompiledPhysicalDag, + materialized_outputs: BTreeMap, +} +impl TryFrom for PhysicalCandidate { + type Error = Error; + fn try_from(candidate: UncheckedCandidate) -> Result { + let result = Self { + precompute: candidate.precompute, + query: candidate.query, + materialized_outputs: candidate.materialized_outputs, + }; + result.validate()?; + Ok(result) + } +} + +impl PhysicalCandidate { + /// Validate the physical handoff, including the producer/reader boundary. + pub fn validate(&self) -> Result<(), Error> { + self.query.validate()?; + let Some(precompute) = &self.precompute else { + return if self.materialized_outputs.is_empty() { + Ok(()) + } else { + Err(invalid("materialized outputs have no producer DAG")) + }; + }; + precompute.validate()?; + let outputs: BTreeSet<_> = self.materialized_outputs.keys().copied().collect(); + if outputs.is_empty() || outputs != precompute.roots().iter().copied().collect() { + return Err(invalid("physical frontier differs from precompute outputs")); + } + let readers: BTreeMap<_, _> = self.query.input_contracts().collect(); + for (&id, contract) in &self.materialized_outputs { + let produced = precompute.output_contract(id)?; + // Direct frontiers retain their node IDs. Temporal candidates can + // read several window instances through distinct input slots; + // their deployment bindings must validate those slots separately. + let reader = readers.get(&id); + if contract.schema != produced.schema + || reader.is_some_and(|reader| contract.schema != reader.schema) + || produced.properties.boundedness != Boundedness::Bounded + || contract.properties.boundedness != Boundedness::Bounded + || reader + .is_some_and(|reader| reader.properties.boundedness != Boundedness::Bounded) + { + return Err(invalid("physical frontier schema or boundedness mismatch")); + } + } + Ok(()) + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs new file mode 100644 index 00000000..70d6a9ff --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -0,0 +1,310 @@ +//! Reader-independent physical computation and checked deployment instantiation. +use super::*; + +/// A typed execution boundary, without storage identity or a live reader. +#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)] +pub struct InputContract { + pub schema: Schema, + pub properties: PlanProperties, +} +impl InputContract { + pub fn bounded(schema: Schema) -> Self { + Self { + schema, + properties: PlanProperties { + boundedness: Boundedness::Bounded, + emission: Emission::Unknown, + }, + } + } + pub fn from_source(source: &dyn PhysicalOperator) -> Self { + Self { + schema: source.output_schema(), + properties: source.properties(&[]), + } + } +} +#[derive(Clone, serde::Serialize, serde::Deserialize)] +enum Node { + Input(InputContract), + Operator { + inputs: Vec, + operator: Operator, + }, +} + +/// Selected native operators and input slots. Rebinding never repeats lowering. +/// Serde is format-agnostic; deployments choose the encoding and its versioning. +/// Deserialization validates the graph before it is usable. +#[derive(Clone, serde::Serialize, serde::Deserialize)] +#[serde(try_from = "UncheckedDag")] +pub struct CompiledPhysicalDag { + nodes: BTreeMap, + roots: Vec, +} +#[derive(serde::Deserialize)] +#[serde(deny_unknown_fields)] +struct UncheckedDag { + nodes: BTreeMap, + roots: Vec, +} +impl TryFrom for CompiledPhysicalDag { + type Error = Error; + fn try_from(dag: UncheckedDag) -> Result { + let result = Self { + nodes: dag.nodes, + roots: dag.roots, + }; + result.validate()?; + Ok(result) + } +} + +impl CompiledPhysicalDag { + /// Link already-selected physical fragments without lowering operators again. + /// Fragment keys and source keys share a namespace; repeated dependency IDs + /// therefore remain one producer in the composed graph. + pub fn compose( + sources: BTreeMap, + fragments: BTreeMap, Self)>, + roots: Vec, + ) -> Result { + if sources.keys().any(|id| fragments.contains_key(id)) { + return Err(invalid("physical source and fragment IDs overlap")); + } + let mut contracts = sources.clone(); + for (&id, (_, fragment)) in &fragments { + fragment.validate()?; + let [root] = fragment.roots() else { + return Err(invalid("composed fragment requires one root")); + }; + if fragment.input_contracts().any(|(id, _)| id == *root) { + return Err(invalid("fragment root must be a computed output")); + } + contracts.insert(id, fragment.output_contract(*root)?); + } + let mut next = contracts + .keys() + .next_back() + .copied() + .unwrap_or(0) + .checked_add(1) + .ok_or_else(|| invalid("physical node ID overflow"))?; + let mut result = Self::new(roots); + for (id, contract) in sources { + result.add_input(id, contract)?; + } + for (id, (inputs, fragment)) in fragments { + if inputs.len() != fragment.input_contracts().count() { + return Err(invalid("physical fragment input arity mismatch")); + } + let mut mapping = BTreeMap::new(); + for ((local, expected), global) in fragment.input_contracts().zip(inputs) { + let actual = contracts + .get(&global) + .ok_or_else(|| invalid("missing physical fragment dependency"))?; + if expected.schema != actual.schema + || (expected.properties.boundedness == Boundedness::Bounded + && actual.properties.boundedness != Boundedness::Bounded) + { + return Err(invalid("physical fragment dependency contract mismatch")); + } + mapping.insert(local, global); + } + mapping.insert(fragment.roots[0], id); + for local in fragment.nodes.keys() { + if !mapping.contains_key(local) { + mapping.insert(*local, next); + next = next + .checked_add(1) + .ok_or_else(|| invalid("physical node ID overflow"))?; + } + } + for (local, node) in fragment.nodes { + if let Node::Operator { inputs, operator } = node { + result.add( + mapping[&local], + inputs.into_iter().map(|input| mapping[&input]).collect(), + operator, + )?; + } + } + } + result.validate()?; + Ok(result) + } + + /// Assemble already-lowered operators and typed external inputs. This is + /// useful for engines that compose multiple compiled computation fragments. + pub fn from_operators( + inputs: BTreeMap, + operators: BTreeMap, Operator)>, + roots: Vec, + ) -> Result { + let mut result = Self::new(roots); + for (id, contract) in inputs { + result.add_input(id, contract)?; + } + for (id, (inputs, operator)) in operators { + result.add(id, inputs, operator)?; + } + result.validate()?; + Ok(result) + } + pub(super) fn new(roots: Vec) -> Self { + Self { + nodes: BTreeMap::new(), + roots, + } + } + pub(super) fn add_input(&mut self, id: NodeId, contract: InputContract) -> Result<(), Error> { + self.insert(id, Node::Input(contract)) + } + pub(super) fn add( + &mut self, + id: NodeId, + inputs: Vec, + operator: Operator, + ) -> Result<(), Error> { + self.insert(id, Node::Operator { inputs, operator }) + } + fn insert(&mut self, id: NodeId, node: Node) -> Result<(), Error> { + if self.nodes.insert(id, node).is_some() { + return Err(invalid(format!("duplicate physical node {id}"))); + } + Ok(()) + } + /// Identify the external input whose rows survive unchanged at this output. + /// Protocol adapters can retain labels that are outside a closed physical schema. + pub fn row_source(&self, id: NodeId) -> Option { + match self.nodes.get(&id)? { + Node::Input(_) => Some(id), + Node::Operator { inputs, operator } => { + let index = operator.row_preserving_input()?; + self.row_source(*inputs.get(index)?) + } + } + } + + /// Selected operator name, for plan inspection without decoding its wire format. + /// Certified candidate pruning checks authoritative-key coverage inside this operator. + pub fn certified_pruning_keys(&self, id: NodeId) -> Option<&[(usize, usize)]> { + match self.nodes.get(&id)? { + Node::Operator { operator, .. } => operator.certified_pruning_keys(), + Node::Input(_) => None, + } + } + pub fn operator_name(&self, id: NodeId) -> Option<&str> { + match self.nodes.get(&id)? { + Node::Input(_) => Some("Input"), + Node::Operator { operator, .. } => Some(operator.name()), + } + } + + pub fn roots(&self) -> &[NodeId] { + &self.roots + } + pub fn input_contracts(&self) -> impl Iterator { + self.nodes.iter().filter_map(|(&id, node)| match node { + Node::Input(contract) => Some((id, contract)), + Node::Operator { .. } => None, + }) + } + /// Derive a reachable output contract without opening deployment readers. + pub fn output_contract(&self, id: NodeId) -> Result { + let sources = self + .input_contracts() + .map(|(id, contract)| (id, Box::new(contract.clone()) as Source<'_>)) + .collect(); + let graph = self.instantiate(sources)?; + let properties = graph.properties(&self.roots)?; + let properties = *properties + .get(&id) + .ok_or_else(|| invalid("output is not reachable"))?; + let schema = match self + .nodes + .get(&id) + .ok_or_else(|| invalid("missing output"))? + { + Node::Input(contract) => contract.schema.clone(), + Node::Operator { operator, .. } => operator.output_schema(), + }; + Ok(InputContract { schema, properties }) + } + /// Validate using contract-only sources. No deployment reader is available. + pub fn validate(&self) -> Result<(), Error> { + let sources = self + .input_contracts() + .map(|(id, c)| (id, Box::new(c.clone()) as Source<'_>)) + .collect(); + self.instantiate(sources).map(|_| ()) + } + /// Resolve exactly the declared inputs and validate before any source starts. + pub fn instantiate<'a>( + &self, + mut sources: BTreeMap>, + ) -> Result, Error> { + let mut graph = PhysicalDag::default(); + for (&id, node) in &self.nodes { + match node { + Node::Input(contract) => { + let source = sources + .remove(&id) + .ok_or_else(|| invalid(format!("missing physical input {id}")))?; + let actual = source.properties(&[]); + if !source.input_schemas().is_empty() + || source.output_schema() != contract.schema + || (contract.properties.boundedness != Boundedness::Unknown + && actual.boundedness != contract.properties.boundedness) + || (contract.properties.emission != Emission::Unknown + && actual.emission != contract.properties.emission) + { + return Err(invalid(format!( + "physical input {id} violates its compiled contract" + ))); + } + graph.add_boxed( + id, + vec![], + Box::new(CheckedSource { + source, + output: contract.schema.clone(), + }), + )?; + } + Node::Operator { inputs, operator } => { + graph.add(id, inputs.clone(), operator.clone())?; + } + } + } + if !sources.is_empty() { + return Err(invalid("unexpected physical input binding")); + } + graph.validate(&self.roots)?; + Ok(graph) + } +} +impl PhysicalOperator for InputContract { + fn name(&self) -> &str { + "UnresolvedInput" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.schema.clone() + } + fn properties(&self, _: &[PlanProperties]) -> PlanProperties { + self.properties + } + fn output_bytes(&self, batch: &Batch) -> usize { + batch.bytes() + } + fn start<'a>( + &'a self, + _: Vec>, + _: crate::runtime::RunContext, + ) -> Result, Error> { + Err(invalid("physical input must be resolved before execution")) + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs new file mode 100644 index 00000000..142a2c25 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -0,0 +1,900 @@ +//! Compile logical computation to native operators with typed external inputs. +//! Compilation needs no readers; deployment resolves inputs after selection. +use crate::operators::ReadoutQuery; +use crate::summary_kernels::exact::ExactReadout; +use crate::{ + operators::{Expression, Operator, Reduction, SortKey}, + plan::{Boundedness, Emission, NodeId, PhysicalDag, PhysicalOperator, PlanProperties}, + values::{Batch, Schema}, + Error, +}; +use planner_types::{ + post_asap::{ + ExactOperation, PostAsapDag, PostAsapDagNode, PostAsapOperatorPayload as Payload, + SketchQuery, SummaryFamilyType, SummaryInputExpr, ValueOperation, + }, + pre_asap::{ + AggIntent, ColumnRef, CompareOpKind, GroupKeys, QueryExpr, Reduction as PlannerReduction, + }, +}; +use std::{ + collections::{BTreeMap, BTreeSet}, + sync::Arc, +}; +fn invalid(message: impl Into) -> Error { + Error::Invalid(message.into()) +} + +/// Source nodes cut the DAG at an installed storage/ingestion frontier. The +/// binding must have exactly the declared schema and no upstream dependencies. +/// A deployment must authorize these frontiers before calling this function. +pub type Source<'a> = Box + 'a>; + +pub mod precompute; +pub mod promql_rows; +pub mod promql_values; + +mod candidates; +pub use candidates::{ + compile_candidate, compile_candidates, enumerate_frontiers, select_candidate, CandidateCost, + CandidateSelection, PhysicalCandidate, +}; + +mod temporal_panes; +pub use temporal_panes::{ + compile_temporal_pane_candidate, TemporalEntityIdentity, TemporalPaneCandidate, + TemporalPaneMaintenance, +}; + +mod compiled; +pub use compiled::{CompiledPhysicalDag, InputContract}; + +/// Compile computation without opening or retaining deployment readers. +/// Input contracts identify explicit boundaries selected by maintenance planning. +pub fn compile( + dag: &PostAsapDag, + inputs: BTreeMap, + roots: &[NodeId], +) -> Result { + compile_internal(dag, inputs, roots) +} + +/// Convenience for callers that already resolved inputs. Lowering still uses +/// only their contracts, and instantiation checks those contracts again. +pub fn bind<'a>( + dag: &PostAsapDag, + sources: BTreeMap>, + roots: &[NodeId], +) -> Result, Error> { + let inputs = sources + .iter() + .map(|(&id, source)| (id, InputContract::from_source(source.as_ref()))) + .collect(); + compile(dag, inputs, roots)?.instantiate(sources) +} + +/// Resolve raw scan connectors before invoking the reader-independent compiler. +pub fn bind_with_data_sources<'a>( + dag: &PostAsapDag, + mut sources: BTreeMap>, + roots: &[NodeId], + data_sources: &crate::sources::DataSources, +) -> Result, Error> { + // Only resolve scans reachable below the selected input boundaries. + let mut pending = roots.to_vec(); + let mut seen = BTreeSet::new(); + while let Some(id) = pending.pop() { + if !seen.insert(id) || sources.contains_key(&id) { + continue; + } + let node = dag + .nodes + .iter() + .find(|n| u64::from(n.id.0) == id) + .ok_or_else(|| invalid(format!("missing node {id}")))?; + if let Payload::Fallback { + expression: expression @ QueryExpr::Scan { .. }, + } = &node.payload + { + sources.insert(id, Box::new(data_sources.bind(expression)?)); + } else { + pending.extend( + dag.edges + .iter() + .filter(|e| u64::from(e.consumer.0) == id) + .map(|e| u64::from(e.producer.0)), + ); + } + } + bind(dag, sources, roots) +} + +fn compile_internal( + dag: &PostAsapDag, + mut sources: BTreeMap, + roots: &[NodeId], +) -> Result { + preflight_depth(dag)?; + dag.validate().map_err(|e| invalid(e.to_string()))?; + let nodes = dag + .nodes + .iter() + .map(|node| (u64::from(node.id.0), node)) + .collect::>(); + let mut dependencies = BTreeMap::>::new(); + // Binary input order is semantic; serialized edge order is not. + let mut edges = dag.edges.iter().collect::>(); + edges.sort_by_key(|edge| { + ( + edge.consumer.0, + match edge.role { + planner_types::post_asap::EdgeRole::Left => 0, + planner_types::post_asap::EdgeRole::Input => 1, + planner_types::post_asap::EdgeRole::Right => 2, + }, + ) + }); + for edge in edges { + dependencies + .entry(u64::from(edge.consumer.0)) + .or_default() + .push(u64::from(edge.producer.0)); + } + if sources.keys().any(|id| !nodes.contains_key(id)) { + return Err(invalid("source binding names an unknown node")); + } + let mut ordered = Vec::new(); + let mut seen = BTreeSet::new(); + let mut pending = roots.iter().map(|&id| (id, false)).collect::>(); + while let Some((id, expanded)) = pending.pop() { + if expanded { + ordered.push(id); + continue; + } + if !seen.insert(id) { + continue; + } + if !nodes.contains_key(&id) { + return Err(invalid(format!("missing root {id}"))); + } + pending.push((id, true)); + if !sources.contains_key(&id) { + for &input in dependencies.get(&id).into_iter().flatten() { + pending.push((input, false)); + } + } + } + let mut graph = CompiledPhysicalDag::new(roots.to_vec()); + let mut auxiliary = u64::MAX; + for id in ordered { + let node = nodes[&id]; + let output = Arc::new(node.output_schema.clone()); + crate::values::validate_schema(&output)?; + if let Some(source) = sources.remove(&id) { + if source.schema != output { + return Err(invalid("frontier does not have the declared schema")); + } + graph.add_input(id, source)?; + } else { + let mut inputs = dependencies.get(&id).cloned().unwrap_or_default(); + let mut schemas = inputs + .iter() + .map(|id| Arc::new(nodes[id].output_schema.clone())) + .collect::>(); + if matches!(node.payload, Payload::SummaryMerge) && inputs.len() > 1 { + if schemas.iter().any(|s| s != &schemas[0]) { + return Err(invalid("summary merge inputs have different schemas")); + } + graph.add( + auxiliary, + inputs, + Operator::union(schemas[0].clone(), schemas.len())?, + )?; + inputs = vec![auxiliary]; + auxiliary -= 1; + schemas.truncate(1); + } + if let Payload::Value { + operation: ValueOperation::MaintainPopulation { population }, + } = &node.payload + { + use planner_types::post_asap::maintained_population::PopulationInput; + let PopulationInput::CurrentSeries(spec) = &population.input else { + return Err(invalid( + "native maintained population requires a current-series input", + )); + }; + let [input] = schemas.as_slice() else { + return Err(invalid("current-series population requires one input")); + }; + if spec.without { + return Err(invalid( + "dynamic without grouping requires label-set projection", + )); + } + let identity = named_column( + input, + &ColumnRef::Named(promql_rows::SERIES_IDENTITY_COLUMN.into()), + )?; + let coordinate = input + .time_index + .ok_or_else(|| invalid("current-series input lacks timestamp"))?; + let value = named_column(input, &ColumnRef::SampleValue)?; + let lookback = i64::try_from(spec.lookback_ms) + .map_err(|_| invalid("current-series lookback overflows"))?; + graph.add( + id, + inputs, + Operator::current_series(input.clone(), identity, coordinate, value, lookback)? + .with_output_schema(output)?, + )?; + continue; + } + if let Payload::Value { + operation: ValueOperation::ReadPopulation { readout }, + } = &node.payload + { + use planner_types::post_asap::maintained_population::{ + PopulationInput, PopulationReadout, + }; + let PopulationReadout::TopK { k } = readout else { + return Err(invalid( + "native population readout does not support this operation", + )); + }; + let [producer] = inputs.as_slice() else { + return Err(invalid("population readout requires one input")); + }; + let Payload::Value { + operation: ValueOperation::MaintainPopulation { population }, + } = &nodes[producer].payload + else { + return Err(invalid( + "population readout requires its declared population", + )); + }; + let PopulationInput::CurrentSeries(spec) = &population.input else { + return Err(invalid("current-series population required")); + }; + if spec.without { + return Err(invalid( + "dynamic without ranking requires label-set projection", + )); + } + let input = schemas[0].clone(); + let groups = spec + .grouping + .iter() + .map(|name| named_column(&input, &ColumnRef::Named(name.clone()))) + .collect::, _>>()?; + let value = named_column(&input, &ColumnRef::SampleValue)?; + graph.add( + auxiliary, + inputs, + Operator::sort( + input.clone(), + vec![SortKey { + column: value, + descending: true, + nulls_first: false, + }], + groups.clone(), + )?, + )?; + graph.add( + id, + vec![auxiliary], + Operator::limit(input, *k as u64, 0, groups)?.with_output_schema(output)?, + )?; + auxiliary -= 1; + continue; + } + // A closed row must include either all source labels or the explicit + // complete-label identity. Projected labels alone are insufficient. + if let Payload::SummaryAgg { + family, + input: update, + reduction: PlannerReduction::PerEntity, + grouping, + } = &node.payload + { + let [input_id] = inputs.as_slice() else { + return Err(invalid("per-entity summary requires one input")); + }; + let Payload::Fallback { + expression: QueryExpr::TimeRange { child, .. }, + } = &nodes[input_id].payload + else { + return Err(invalid( + "per-entity summary requires a resolved raw time range", + )); + }; + let QueryExpr::Scan { schema, .. } = child.as_ref() else { + return Err(invalid("per-entity summary requires a resolved source")); + }; + if !schema.closed || update.item.is_some() { + return Err(invalid( + "per-entity summary requires complete source identity", + )); + } + crate::capability::validate_summary_kernel(family, update, grouping) + .map_err(Error::Invalid)?; + let SummaryInputExpr::Column(value) = &update.weight else { + return Err(invalid( + "per-entity update requires a projected value column", + )); + }; + let input = schemas[0].clone(); + let value = named_column(&input, value)?; + let coordinate = input + .time_index + .ok_or_else(|| invalid("temporal input lacks time"))?; + let groups = (0..input.fields.len()) + .filter(|&column| column != value && column != coordinate) + .collect(); + let build = Operator::summary_build( + input, + family.clone(), + value, + Some(coordinate), + groups, + )?; + let compact = build.schema(); + graph.add(auxiliary, inputs, build)?; + graph.add( + id, + vec![auxiliary], + Operator::scope_timestamp(compact, output)?, + )?; + auxiliary -= 1; + continue; + } + let mut operator = compile_node(node, &schemas) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + if operator.is_counter_readout() { + let mut pending = vec![id]; + let mut visited = BTreeSet::new(); + let mut ranges = BTreeSet::new(); + while let Some(ancestor) = pending.pop() { + if !visited.insert(ancestor) { + continue; + } + if let Payload::Fallback { + expression: QueryExpr::TimeRange { range, .. }, + } = &nodes[&ancestor].payload + { + ranges.insert( + i64::try_from(range.as_millis()) + .map_err(|_| invalid("counter lookback exceeds Int64"))?, + ); + continue; + } + pending.extend(dependencies.get(&ancestor).into_iter().flatten().copied()); + } + if ranges.len() > 1 { + return Err(invalid("counter readout has ambiguous logical windows")); + } + if let Some(lookback) = ranges.into_iter().next() { + operator = operator.with_counter_lookback(lookback)?; + } + } + graph.add(id, inputs, operator)?; + } + } + graph.validate()?; + Ok(graph) +} + +/// Bind a Planner node against the schemas supplied by its deployment edges. +/// This is the same checked path used by complete DAG binding. +pub fn compile_node(node: &PostAsapDagNode, inputs: &[Schema]) -> Result { + for schema in inputs { + crate::values::validate_schema(schema)?; + } + bind_operation(node, inputs)?.with_output_schema(Arc::new(node.output_schema.clone())) +} + +fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result { + if let Payload::Binary { operator } = &node.payload { + let [left, right] = inputs else { + return Err(invalid("binary requires two inputs")); + }; + if node.output_state.timing == planner_types::post_asap::ExecutionTiming::IngestionTime { + let value = |schema: &Schema| -> Result { + let columns = schema + .fields + .iter() + .enumerate() + .filter(|(_, field)| { + field.dtype + == SummaryFamilyType::Plain(planner_types::pre_asap::DataType::Float64) + }) + .map(|(i, _)| i) + .collect::>(); + match columns.as_slice() { + [value] => Ok(*value), + _ => Err(invalid("aligned binary requires one value column")), + } + }; + let (l, r) = (value(left)?, value(right)?); + let keys = left + .fields + .iter() + .enumerate() + .filter(|(i, _)| *i != l) + .map(|(i, field)| { + right + .fields + .iter() + .position(|other| other.name == field.name && other.dtype == field.dtype) + .map(|j| (i, j)) + .ok_or_else(|| invalid("aligned input identities differ")) + }) + .collect::, _>>()?; + return Operator::aligned_binary( + left.clone(), + right.clone(), + keys, + (l, r), + operator.clone(), + ); + } + return Operator::vector_binary(left.clone(), right.clone(), operator.clone(), false); + } + if let Payload::RelationalJoin { + join_kind, + pred, + pruning, + } = &node.payload + { + use planner_types::{post_asap::CandidateCompleteness, pre_asap::JoinKind}; + if pruning.is_some() && *join_kind != JoinKind::Semi { + return Err(invalid("pruning certificate requires a semi-join")); + } + if matches!(pruning,Some(CandidateCompleteness::Certified { guarantee }) if guarantee.has_unknown() || guarantee.metric != planner_types::post_asap::ErrorMetric::TopKMembership) + { + return Err(invalid("invalid pruning certificate")); + } + let [left, right] = inputs else { + return Err(invalid("join requires two inputs")); + }; + if *join_kind == JoinKind::Semi { + if let Ok(keys) = equijoin_keys(pred, left, right) { + let operator = Operator::semi_join(left.clone(), right.clone(), keys)?; + return Ok( + if matches!(pruning, Some(CandidateCompleteness::Certified { .. })) { + operator.require_complete_right() + } else { + operator + }, + ); + } + } + if matches!(pruning, Some(CandidateCompleteness::Certified { .. })) { + return Err(invalid("certified pruning requires explicit equijoin keys")); + } + return Operator::relational_join( + left.clone(), + right.clone(), + join_kind.clone(), + pred, + Arc::new(node.output_schema.clone()), + ); + } + let [input] = inputs else { + return Err(invalid( + "native Planner binding currently requires a unary operation or an explicit source", + )); + }; + match &node.payload { + Payload::Value { operation, .. } => match operation { + ValueOperation::Project { cols, .. } => Operator::project( + input.clone(), + cols.iter() + .enumerate() + .map(|(i, col)| { + Ok(( + node.output_schema + .fields + .get(i) + .ok_or_else(|| invalid("projection width mismatch"))? + .name + .clone(), + match &col.expr { + QueryExpr::Column(index) => Expression::Column(*index), + expr => expression(expr, input)?, + }, + )) + }) + .collect::>()?, + ), + ValueOperation::Filter { pred } => { + Operator::filter(input.clone(), expression(&pred.0, input)?) + } + ValueOperation::Sort { keys, partition_by } => Operator::sort( + input.clone(), + keys.iter() + .map(|key| { + let QueryExpr::Column(column) = key.expr else { + return Err(invalid( + "sort expression must be projected before sorting", + )); + }; + Ok(SortKey { + column, + descending: !key.ascending, + nulls_first: key.nulls_first, + }) + }) + .collect::>()?, + groups(input, partition_by)?, + ), + ValueOperation::Limit { + n, + offset, + partition_by, + } => Operator::limit( + input.clone(), + *n as u64, + *offset as u64, + groups(input, partition_by)?, + ), + ValueOperation::Exact(ExactOperation::Aggregate { + reduction, + measures, + output_names, + having: None, + }) => { + if measures.len() != output_names.len() { + return Err(invalid("aggregate output names differ from measures")); + } + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid( + "per-entity aggregate requires an explicit entity binding", + )); + }; + let measures = measures + .iter() + .zip(output_names) + .map(|(m, name)| { + let column = |col: Option| { + col.map(Ok) + .unwrap_or_else(|| named_column(input, &ColumnRef::SampleValue)) + }; + let m = match m { + AggIntent::Count { .. } => Reduction::Count, + AggIntent::Sum { col } => Reduction::Sum(column(*col)?), + AggIntent::Avg { col } => Reduction::Avg(column(*col)?), + AggIntent::Min { col } => Reduction::Min(column(*col)?), + AggIntent::Max { col } => Reduction::Max(column(*col)?), + _ => { + return Err(invalid( + "aggregate intent has no native implementation", + )) + } + }; + Ok((name.clone(), m)) + }) + .collect::>()?; + Operator::aggregate(input.clone(), groups(input, keys)?, measures) + } + ValueOperation::FinalizeExactAccumulator => { + let state = summary_column(input)?; + use crate::Statistic as S; + use planner_types::post_asap::ExactKind as E; + let statistic = match &input.fields[state].dtype { + SummaryFamilyType::ExactAggregate(kind, _) => match kind { + E::Sum => S::Sum, + E::Count => S::Count, + E::Min => S::Min, + E::Max => S::Max, + E::Rate => S::Rate, + E::Increase => S::Increase, + _ => return Err(invalid("exact family readout is unsupported")), + }, + _ => return Err(invalid("exact finalization requires exact state")), + }; + Operator::readout( + input.clone(), + state, + ReadoutQuery::Exact(ExactReadout { + statistic, + lookback_ms: None, + }), + ) + } + _ => Err(invalid("value operation has no native implementation")), + }, + Payload::SummaryAgg { + family, + input: update, + reduction, + grouping, + } => { + if let Some(item) = &update.item { + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid("keyed summary requires explicit partitions")); + }; + let SummaryInputExpr::Column(weight) = &update.weight else { + return Err(invalid( + "keyed summary weight must be a finalized value column", + )); + }; + if matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::CmsWithHeap) + && !matches!( + update.weight_domain, + planner_types::post_asap::WeightDomain::NonNegative { .. } + ) + { + return Err(invalid("CMS requires a nonnegative weight contract")); + } + fn columns( + expr: &SummaryInputExpr, + input: &Schema, + result: &mut Vec, + ) -> Result<(), Error> { + match expr { + SummaryInputExpr::Column(column) => { + result.push(named_column(input, column)?) + } + SummaryInputExpr::Tuple(items) => { + for item in items { + columns(item, input, result)?; + } + } + _ => return Err(invalid("keyed summary needs explicit item columns")), + } + Ok(()) + } + let mut items = Vec::new(); + columns(item, input, &mut items)?; + return Operator::keyed_summary_build( + input.clone(), + family.clone(), + named_column(input, weight)?, + items, + groups(input, keys)?, + ); + } + crate::capability::validate_summary_kernel(family, update, grouping) + .map_err(Error::Invalid)?; + let SummaryInputExpr::Column(column) = &update.weight else { + return Err(invalid( + "summary update expression must be projected to a column", + )); + }; + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid( + "summary construction requires explicit grouping columns", + )); + }; + Operator::summary_build( + input.clone(), + family.clone(), + named_column(input, column)?, + input.time_index, + groups(input, keys)?, + ) + } + Payload::SummaryMerge => { + let state = summary_column(input)?; + Operator::summary_merge( + input.clone(), + state, + (0..input.fields.len()) + .filter(|&i| i != state && Some(i) != input.time_index) + .collect(), + ) + } + Payload::SummaryEstimate { query } => { + if let SketchQuery::TopK { k } = query { + return Operator::keyed_readout( + input.clone(), + summary_column(input)?, + *k, + Arc::new(node.output_schema.clone()), + ); + } + Operator::readout( + input.clone(), + summary_column(input)?, + ReadoutQuery::Sketch(query.clone()), + ) + } + _ => Err(invalid( + "physical operation has no native binding; no fallback is installed", + )), + } +} +fn summary_column(input: &Schema) -> Result { + let columns = input + .fields + .iter() + .enumerate() + .filter(|(_, f)| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + .map(|(i, _)| i) + .collect::>(); + match columns.as_slice() { + [column] => Ok(*column), + _ => Err(invalid("one summary state column required")), + } +} +fn named_column(input: &Schema, column: &ColumnRef) -> Result { + let name = match column { + // Executable SummarySchema retains column names, not table qualifiers. + // Frontend binding has resolved the qualifier; still reject ambiguous + // names here rather than guessing a join side. + ColumnRef::Named(name) | ColumnRef::Qualified { name, .. } => name.as_str(), + ColumnRef::SampleValue => "value", + _ => { + return Err(invalid( + "summary update requires an unambiguous bound column", + )) + } + }; + let matches = input + .fields + .iter() + .enumerate() + .filter(|(_, field)| field.name == name) + .map(|(i, _)| i) + .collect::>(); + match matches.as_slice() { + [column] => Ok(*column), + _ => Err(invalid("summary update column missing or ambiguous")), + } +} +fn groups(input: &Schema, groups: &GroupKeys) -> Result, Error> { + if groups.is_without() { + return Err(invalid("grouping without requires resolved label columns")); + } + if groups.keys().iter().any(|&i| i >= input.fields.len()) { + return Err(invalid("grouping column out of range")); + } + Ok(groups.keys().to_vec()) +} +fn expression(expr: &QueryExpr, input: &Schema) -> Result { + Ok(Expression::planner( + crate::expressions::CompiledExpression::compile(expr, input)?, + )) +} + +struct CheckedSource<'a> { + source: Source<'a>, + output: Schema, +} +impl PhysicalOperator for CheckedSource<'_> { + fn properties(&self, inputs: &[crate::plan::PlanProperties]) -> crate::plan::PlanProperties { + self.source.properties(inputs) + } + + fn name(&self) -> &str { + self.source.name() + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, batch: &Batch) -> usize { + self.source.output_bytes(batch) + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: crate::runtime::RunContext, + ) -> Result, Error> { + use futures::StreamExt; + Ok(self + .source + .start(inputs, context)? + .map(|batch| { + let batch = batch?; + if batch.schema() != &self.output { + return Err(invalid("source batch differs from its bound schema")); + } + Ok(batch) + }) + .boxed_local()) + } +} + +// Bound recursion before invoking the upstream recursive provenance validator. +fn preflight_depth(dag: &PostAsapDag) -> Result<(), Error> { + let mut remaining = dag + .nodes + .iter() + .map(|node| (node.id, 0usize)) + .collect::>(); + if remaining.len() != dag.nodes.len() { + return Err(invalid("duplicate Planner node")); + } + let mut consumers = BTreeMap::<_, Vec<_>>::new(); + for edge in &dag.edges { + if !remaining.contains_key(&edge.producer) { + return Err(invalid("missing Planner edge producer")); + } + *remaining + .get_mut(&edge.consumer) + .ok_or_else(|| invalid("missing Planner edge consumer"))? += 1; + consumers + .entry(edge.producer) + .or_default() + .push(edge.consumer); + } + let mut ready = remaining + .iter() + .filter(|(_, n)| **n == 0) + .map(|(id, _)| *id) + .collect::>(); + let mut depths = BTreeMap::new(); + let mut visited = 0; + while let Some(id) = ready.pop_front() { + visited += 1; + let depth = *depths.get(&id).unwrap_or(&1usize); + if depth > 128 { + return Err(invalid("DAG exceeds the supported execution depth of 128")); + } + for &consumer in consumers.get(&id).into_iter().flatten() { + let next = depths.entry(consumer).or_insert(1); + *next = (*next).max(depth + 1); + let count = remaining.get_mut(&consumer).expect("validated endpoint"); + *count -= 1; + if *count == 0 { + ready.push_back(consumer); + } + } + } + if visited != dag.nodes.len() { + return Err(invalid("Planner DAG contains a cycle")); + } + Ok(()) +} + +/// Join predicates address the concatenated left/right schema. +fn semi_join_keys( + expr: &QueryExpr, + left: usize, + right: usize, + keys: &mut Vec<(usize, usize)>, +) -> Result<(), Error> { + match expr { + QueryExpr::BoolAnd(parts) => { + for part in parts { + semi_join_keys(part, left, right, keys)?; + } + } + QueryExpr::Compare { + left: a, + op: CompareOpKind::Eq, + right: b, + } => { + let (QueryExpr::Column(a), QueryExpr::Column(b)) = (a.as_ref(), b.as_ref()) else { + return Err(invalid("semi-join requires column equality keys")); + }; + let (a, b) = if a < b { (*a, *b) } else { (*b, *a) }; + if a >= left || b < left || b >= left + right { + return Err(invalid("semi-join key must match left to right")); + } + keys.push((a, b - left)); + } + _ => return Err(invalid("unsupported semi-join predicate")), + } + Ok(()) +} + +/// Resolve equality keys against the Planner join's concatenated input schema. +/// Deployments may use these positions to bind their source columns. +pub fn equijoin_keys( + pred: &planner_types::pre_asap::Predicate, + left: &planner_types::post_asap::SummarySchema, + right: &planner_types::post_asap::SummarySchema, +) -> Result, Error> { + let mut keys = Vec::new(); + semi_join_keys(&pred.0, left.fields.len(), right.fields.len(), &mut keys)?; + if keys.is_empty() { + return Err(invalid("semi-join requires explicit matching keys")); + } + Ok(keys) +} diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs new file mode 100644 index 00000000..063bd0b8 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -0,0 +1,384 @@ +//! Compile immutable summary-input computation with explicit population and pane identity. +use super::*; +use planner_types::{ + post_asap::{ExecutionTiming, GroupingStrategy, SummarySchema}, + pre_asap::DataType, +}; + +/// Physical rows carry the population and pane coordinate alongside the logical value. +/// These fields preserve identities which are implicit in a stored summary instance. +pub fn population_schema(family: SummaryFamilyType) -> Schema { + Arc::new(SummarySchema { + fields: vec![ + planner_types::post_asap::SummaryField { + name: "$population".into(), + dtype: SummaryFamilyType::Plain(DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }), + nullable: false, + }, + planner_types::post_asap::SummaryField { + name: "$window_end".into(), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), + nullable: false, + }, + planner_types::post_asap::SummaryField { + name: "value".into(), + dtype: family, + nullable: false, + }, + ], + time_index: Some(1), + }) +} + +/// Validate the adapter layout during installed-plan recovery without lowering operators. +pub fn source_schema(logical: &SummarySchema) -> Result { + let states = logical + .fields + .iter() + .filter(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + .collect::>(); + let [state] = states.as_slice() else { + return Err(invalid( + "stored population requires one typed summary state", + )); + }; + if logical.fields.iter().enumerate().any(|(i, field)| matches!(&field.dtype, SummaryFamilyType::Plain(dtype) + if field.nullable || !matches!(dtype, DataType::Utf8) && !(Some(i) == logical.time_index && *dtype == DataType::Timestamp))) { + return Err(invalid("stored population metadata cannot reconstruct extra value columns")); + } + if state.nullable { + return Err(invalid("stored population state cannot be null")); + } + Ok(population_schema(state.dtype.clone())) +} + +pub fn is_population_schema(schema: &Schema) -> bool { + schema + .fields + .get(2) + .is_some_and(|field| *schema == population_schema(field.dtype.clone())) +} + +/// Compile a complete selected precompute sub-DAG. Inputs are already-computed +/// state boundaries; the deployment supplies groups, panes and states, never operations. +pub fn compile( + dag: &PostAsapDag, + frontiers: &[NodeId], + roots: &[NodeId], +) -> Result { + preflight_depth(dag)?; + dag.validate().map_err(|e| invalid(e.to_string()))?; + let nodes = dag + .nodes + .iter() + .map(|n| (u64::from(n.id.0), n)) + .collect::>(); + let frontier = frontiers.iter().copied().collect::>(); + if frontier.len() != frontiers.len() || roots.iter().any(|r| frontier.contains(r)) { + return Err(invalid( + "precompute boundaries must be distinct from outputs", + )); + } + let mut dependencies = BTreeMap::>::new(); + let mut edges = dag.edges.iter().collect::>(); + edges.sort_by_key(|edge| { + ( + edge.consumer.0, + match edge.role { + planner_types::post_asap::EdgeRole::Left => 0, + planner_types::post_asap::EdgeRole::Input => 1, + planner_types::post_asap::EdgeRole::Right => 2, + }, + ) + }); + for edge in edges { + dependencies + .entry(u64::from(edge.consumer.0)) + .or_default() + .push(u64::from(edge.producer.0)); + } + let mut ordered = Vec::new(); + let mut seen = BTreeSet::new(); + let mut pending = roots.iter().map(|&id| (id, false)).collect::>(); + while let Some((id, expanded)) = pending.pop() { + if expanded { + ordered.push(id); + continue; + } + if !seen.insert(id) { + continue; + } + if !nodes.contains_key(&id) { + return Err(invalid("missing precompute node")); + } + pending.push((id, true)); + if !frontier.contains(&id) { + pending.extend( + dependencies + .get(&id) + .into_iter() + .flatten() + .map(|id| (*id, false)), + ); + } + } + let mut sources = BTreeMap::new(); + let mut fragments = BTreeMap::new(); + let mut outputs = BTreeMap::::new(); + for id in ordered { + let node = nodes[&id]; + if frontier.contains(&id) { + let schema = source_schema(&node.output_schema)?; + sources.insert(id, InputContract::bounded(schema.clone())); + outputs.insert(id, schema); + continue; + } + if node.output_state.timing != ExecutionTiming::IngestionTime { + return Err(invalid("precompute graph contains a query-time operation")); + } + let inputs = dependencies.get(&id).cloned().unwrap_or_default(); + let schemas = inputs + .iter() + .map(|id| { + outputs + .get(id) + .cloned() + .ok_or_else(|| invalid("missing precompute input")) + }) + .collect::, _>>()?; + let graph = fragment( + node, + &schemas, + &inputs.iter().map(|id| nodes[id]).collect::>(), + )?; + outputs.insert(id, graph.output_contract(graph.roots()[0])?.schema); + fragments.insert(id, (inputs, graph)); + } + CompiledPhysicalDag::compose(sources, fragments, roots.to_vec()) +} + +fn validate_value_output(node: &PostAsapDagNode) -> Result<(), Error> { + let schema = &node.output_schema; + let values = schema + .fields + .iter() + .enumerate() + .filter(|(i, _)| Some(*i) != schema.time_index) + .collect::>(); + if !matches!(values.as_slice(), [(_, field)] if !field.nullable && field.dtype == SummaryFamilyType::Plain(DataType::Float64)) + || schema.time_index.is_some_and(|i| { + schema.fields.get(i).is_none_or(|f| { + f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Timestamp) + }) + }) + { + return Err(invalid( + "precompute value schema requires Float64 and an optional declared timestamp", + )); + } + Ok(()) +} + +fn fragment( + node: &PostAsapDagNode, + schemas: &[Schema], + parents: &[&PostAsapDagNode], +) -> Result { + let sources = schemas + .iter() + .enumerate() + .map(|(id, schema)| (id as u64, InputContract::bounded(schema.clone()))) + .collect(); + let mut operators = BTreeMap::new(); + let mut next = schemas.len() as u64; + let mut add = |inputs: Vec, op: Operator| -> Result { + let id = next; + next += 1; + operators.insert(id, (inputs, op)); + Ok(id) + }; + let root = match &node.payload { + Payload::Binary { operator } => { + validate_value_output(node)?; + if node.output_schema.time_index.is_none() + || parents.iter().any(|p| p.output_schema.time_index.is_none()) + { + return Err(invalid( + "precompute binary requires declared window timestamps", + )); + } + let [left, right] = schemas else { + return Err(invalid("precompute binary requires two inputs")); + }; + add( + vec![0, 1], + Operator::aligned_binary( + left.clone(), + right.clone(), + vec![(0, 0), (1, 1)], + (2, 2), + operator.clone(), + )?, + )? + } + Payload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + } => { + let [input] = schemas else { + return Err(invalid("finalize requires one state input")); + }; + validate_value_output(node)?; + let statistic = match &input.fields[2].dtype { + SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Sum, _) => { + crate::Statistic::Sum + } + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Count, + _, + ) => crate::Statistic::Count, + _ => { + return Err(invalid( + "precompute finalization requires explicit Sum or Count semantics", + )) + } + }; + let read = Operator::readout( + input.clone(), + 2, + ReadoutQuery::Exact(ExactReadout { + statistic, + lookback_ms: None, + }), + )?; + let output = read.schema(); + let read = add(vec![0], read)?; + let project = Operator::project( + output, + vec![ + ("$population".into(), Expression::Column(0)), + ("$window_end".into(), Expression::Column(1)), + ( + "value".into(), + Expression::FiniteFloat64(Box::new(Expression::ExactFloat64(2))), + ), + ], + )? + .with_output_schema(population_schema(SummaryFamilyType::Plain( + DataType::Float64, + )))?; + add(vec![read], project)? + } + Payload::SummaryAgg { + family, + input: update, + reduction, + grouping, + } => { + let [input] = schemas else { + return Err(invalid("summary update requires one input")); + }; + if update.item.is_some() + || !matches!(grouping, GroupingStrategy::PerSubpopulationInstance) + { + return Err(invalid( + "precompute keyed/shared update needs its dedicated physical candidate", + )); + } + crate::capability::validate_summary_kernel(family, update, grouping) + .map_err(Error::Invalid)?; + let labels = match reduction { + PlannerReduction::PerEntity => Expression::Column(0), + PlannerReduction::Reduce(keys) => Expression::LabelSet { + column: 0, + labels: keys + .keys() + .iter() + .map(|key| { + parents[0] + .output_schema + .fields + .get(*key) + .filter(|field| { + !field.nullable + && field.dtype == SummaryFamilyType::Plain(DataType::Utf8) + }) + .map(|f| f.name.clone()) + .ok_or_else(|| { + invalid("summary grouping must identify population labels") + }) + }) + .collect::, _>>()?, + without: keys.is_without(), + }, + }; + let weight = match &update.weight { + SummaryInputExpr::Constant(value) => Expression::Literal { + value: crate::values::Value::Float64(*value), + dtype: DataType::Float64, + }, + SummaryInputExpr::Column(ColumnRef::SampleValue) => Expression::Column(2), + SummaryInputExpr::Column(ColumnRef::Named(name)) + if parents[0].output_schema.fields.iter().any(|f| { + f.name == *name && f.dtype == SummaryFamilyType::Plain(DataType::Float64) + }) => + { + Expression::Column(2) + } + _ => { + return Err(invalid( + "summary weight does not resolve to the input value", + )) + } + }; + let project = Operator::project( + input.clone(), + vec![ + ("$population".into(), labels), + ("$window_end".into(), Expression::Column(1)), + ("value".into(), Expression::FiniteFloat64(Box::new(weight))), + ], + )? + .with_output_schema(population_schema(SummaryFamilyType::Plain( + DataType::Float64, + )))?; + let projected = project.schema(); + let project = add(vec![0], project)?; + let build = Operator::summary_build(projected, family.clone(), 2, Some(1), vec![0])?; + let built = build.schema(); + let build = add(vec![project], build)?; + add( + vec![build], + Operator::scope_timestamp(built, population_schema(family.clone()))?, + )? + } + Payload::SummaryMerge => { + let Some(input) = schemas.first() else { + return Err(invalid("summary merge requires inputs")); + }; + if schemas.iter().any(|s| s != input) { + return Err(invalid("summary merge inputs differ")); + } + let union = add( + (0..schemas.len() as u64).collect(), + Operator::union(input.clone(), schemas.len())?, + )?; + let merge = Operator::summary_merge(input.clone(), 2, vec![0])?; + let merged = merge.schema(); + let merge = add(vec![union], merge)?; + add( + vec![merge], + Operator::scope_timestamp(merged, input.clone())?, + )? + } + _ => { + return Err(invalid( + "precompute operation has no native population implementation", + )) + } + }; + CompiledPhysicalDag::from_operators(sources, operators, vec![root]) +} diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs new file mode 100644 index 00000000..cb680096 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -0,0 +1,389 @@ +//! A bounded PromQL source row carries the entire label set, not just labels +//! mentioned by the query. The source adapter owns this lossless encoding. +use super::*; +use planner_types::pre_asap::{Column, DataType, Source as LogicalSource}; +use std::rc::Rc; + +/// Not a legal PromQL label name, so it cannot shadow a user label. +pub use planner_types::pre_asap::schema::PROMQL_SERIES_IDENTITY as SERIES_IDENTITY_COLUMN; + +/// Canonical, reversible identity. JSON object encoding preserves label names, +/// empty values and escaping; sorting makes ingestion order irrelevant. +pub fn encode_series_identity(labels: &BTreeMap) -> Result { + serde_json::to_string(labels).map_err(|error| invalid(error.to_string())) +} + +pub fn decode_series_identity(encoded: &str) -> Result, Error> { + let labels: BTreeMap = + serde_json::from_str(encoded).map_err(|error| invalid(error.to_string()))?; + if encode_series_identity(&labels)? != encoded { + return Err(invalid("series identity is not canonically encoded")); + } + Ok(labels) +} + +/// Resolve the row representation before candidate search. `closed` describes +/// physical columns here: the final column contains every dynamic source label. +/// It does not assert that the query's projected labels are the full label set. +/// +/// This realization supports explicit `by` grouping and per-series computation. +/// Operators that rewrite or implicitly match dynamic label sets require their +/// own realization; they must not accidentally treat the opaque identity as a +/// user label or silently discard it. +pub fn with_series_identity(root: &QueryExpr) -> Result { + let mut root = root.clone(); + fn visit(node: &mut QueryExpr) -> Result<(), Error> { + use planner_types::pre_asap::Reduction; + match node { + QueryExpr::Scan { + source: LogicalSource::TimeSeries { .. }, + schema, + .. + } => { + if schema + .columns + .iter() + .any(|column| column.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "source already contains a physical series identity", + )); + } + if schema.closed { + return Err(invalid( + "dynamic series identity requires an open PromQL source", + )); + } + schema + .columns + .push(Column::new(SERIES_IDENTITY_COLUMN, DataType::Utf8, false)); + schema.closed = true; + Ok(()) + } + QueryExpr::TimeRange { child, .. } | QueryExpr::Limit { child, .. } => { + visit(Rc::make_mut(child)) + } + QueryExpr::Aggregate { + child, reduction, .. + } => { + if matches!(reduction, Reduction::Reduce(keys) if keys.is_without()) { + return Err(invalid( + "dynamic without grouping requires label-set projection", + )); + } + visit(Rc::make_mut(child)) + } + QueryExpr::Sort { + child, + partition_by, + .. + } => { + if partition_by.is_without() { + return Err(invalid( + "dynamic without ranking requires label-set projection", + )); + } + visit(Rc::make_mut(child)) + } + _ => Err(invalid( + "operator has no dynamic series-identity realization", + )), + } + } + visit(&mut root)?; + root.output_schema() + .map_err(|error| invalid(error.to_string()))?; + Ok(root) +} + +/// Construct source rows only from full identities. The named label columns +/// are projections of that same identity and cannot independently redefine it. +pub fn series_row( + schema: &Schema, + labels: &BTreeMap, + timestamp: i64, + value: f64, +) -> Result, Error> { + use crate::values::Value; + let identity = encode_series_identity(labels)?; + let mut found = false; + let row = schema + .fields + .iter() + .enumerate() + .map(|(index, field)| { + if field.name == SERIES_IDENTITY_COLUMN { + if field.dtype != SummaryFamilyType::Plain(DataType::Utf8) + || field.nullable + || found + { + return Err(invalid("invalid series identity column")); + } + found = true; + Ok(Value::Utf8(identity.clone().into())) + } else if Some(index) == schema.time_index { + Ok(Value::Timestamp(timestamp)) + } else if field.name == "value" + && field.dtype == SummaryFamilyType::Plain(DataType::Float64) + { + Ok(Value::Float64(value)) + } else if field.dtype == SummaryFamilyType::Plain(DataType::Utf8) { + Ok(labels.get(&field.name).map_or_else( + || Value::Utf8("".into()), + |value| Value::Utf8(value.clone().into()), + )) + } else { + Err(invalid("unsupported PromQL source column")) + } + }) + .collect::, _>>()?; + if !found { + return Err(invalid("source lacks its full series identity")); + } + Ok(row) +} + +/// Compile the selected TopK computation above an existing maintained-population +/// source. The boundary supplies the complete eligible vector, not a truncated +/// TopK result; ranking remains a native physical operator. +pub fn compile_current_series_readout( + selected: &Rc, +) -> Result { + use planner_types::post_asap::{ + compile_post_asap_dag, maintained_population::PopulationReadout, SummaryField, + }; + let mut dag = compile_post_asap_dag(selected).map_err(|error| invalid(error.to_string()))?; + // Typed snapshot candidates already carry full identity throughout the DAG. + // Cut at the population output, preserving all selected heap/readout nodes. + let populations = dag.nodes.iter().filter(|node| matches!(&node.payload, + Payload::Value { operation: ValueOperation::MaintainPopulation { population } } + if matches!(population.input, planner_types::post_asap::maintained_population::PopulationInput::CurrentSeries(_)) + )).collect::>(); + if let [population] = populations.as_slice() { + if population + .output_schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + { + return compile( + &dag, + BTreeMap::from([( + u64::from(population.id.0), + InputContract::bounded(Arc::new(population.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + ); + } + } + if dag.nodes.len() != 3 + || !dag.nodes.iter().any(|node| { + node.id == dag.root + && matches!( + node.payload, + Payload::Value { + operation: ValueOperation::ReadPopulation { + readout: PopulationReadout::TopK { .. } + } + } + ) + }) + { + return Err(invalid( + "expected one selected current-series TopK computation", + )); + } + let mut frontier = None; + for node in &mut dag.nodes { + match &mut node.payload { + Payload::Fallback { expression } => { + *expression = with_series_identity(expression)?; + } + Payload::Value { + operation: ValueOperation::MaintainPopulation { .. }, + } => { + frontier = Some(u64::from(node.id.0)); + } + Payload::Value { + operation: + ValueOperation::ReadPopulation { + readout: PopulationReadout::TopK { .. }, + }, + } => {} + _ => return Err(invalid("unsupported current-series readout dependency")), + } + if node + .output_schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "current-series input already has a physical identity column", + )); + } + node.output_schema.fields.push(SummaryField { + name: SERIES_IDENTITY_COLUMN.into(), + dtype: SummaryFamilyType::Plain(DataType::Utf8), + nullable: false, + }); + } + for edge in &mut dag.edges { + edge.intermediate_schema = dag + .nodes + .iter() + .find(|node| node.id == edge.producer) + .unwrap() + .output_schema + .clone(); + } + let frontier = frontier.ok_or_else(|| invalid("missing current-series population"))?; + let schema = Arc::new( + dag.nodes + .iter() + .find(|node| u64::from(node.id.0) == frontier) + .unwrap() + .output_schema + .clone(), + ); + compile( + &dag, + BTreeMap::from([(frontier, InputContract::bounded(schema))]), + &[u64::from(dag.root.0)], + ) +} + +/// Compile selected ranking or aggregation above an exact per-series Rate +/// readout. Deployments bind complete window readouts at this boundary; +/// the heap is rebuilt independently for each evaluation. This does not move +/// that frontier to ingestion time or authorize combining finalized rates. +pub fn compile_rate_ranking( + selected: &Rc, +) -> Result< + ( + Rc, + CompiledPhysicalDag, + ), + Error, +> { + use planner_types::post_asap::{ + compile_post_asap_dag_with_node_ids, ExactKind, SummaryExpr, SummaryNode, + }; + fn frontier(node: &Rc) -> Option> { + match &node.expr { + SummaryExpr::ValueOperation { + child, + operation: ValueOperation::FinalizeExactAccumulator, + timing: planner_types::post_asap::ExecutionTiming::QueryTime, + } if matches!(&child.expr, SummaryExpr::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + reduction: planner_types::pre_asap::Reduction::PerEntity, + child: raw, .. + } if matches!(&raw.expr, SummaryExpr::KeepPreAsap(expr) if matches!(expr.as_ref(), QueryExpr::TimeRange { .. }))) => + { + Some(Rc::clone(node)) + } + SummaryExpr::ValueOperation { child, .. } | SummaryExpr::SummaryAgg { child, .. } => { + frontier(child) + } + SummaryExpr::SummaryEstimate { summary_input, .. } => frontier(summary_input), + _ => None, + } + } + let source = frontier(selected) + .ok_or_else(|| invalid("ranking requires one exact per-series Rate frontier"))?; + if !source + .schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid("Rate ranking requires complete series identity")); + } + let compiled = compile_post_asap_dag_with_node_ids(selected) + .map_err(|error| invalid(error.to_string()))?; + let id = u64::from( + compiled + .node_ids + .node_id(&source) + .ok_or_else(|| invalid("missing Rate frontier"))? + .0, + ); + let program = compile( + &compiled.dag, + BTreeMap::from([(id, InputContract::bounded(Arc::new(source.schema.clone())))]), + &[u64::from(compiled.dag.root.0)], + )?; + Ok((source, program)) +} + +/// The selected logical placement requires fresh aggregate state per closed window. +/// Compile both physical graphs before deployment chooses storage or scheduling. +/// The input is the complete collection of per-series exact counter states. +pub fn compile_fixed_window_rate_aggregation( + selected: &Rc, +) -> Result { + use planner_types::post_asap::{ + compile_post_asap_dag, ExactKind, ExecutionTiming, SketchAlgorithm, + }; + let dag = compile_post_asap_dag(selected).map_err(|e| invalid(e.to_string()))?; + let sources = dag + .nodes + .iter() + .filter(|n| { + matches!( + &n.payload, + Payload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + reduction: planner_types::pre_asap::Reduction::PerEntity, + .. + } + ) + }) + .collect::>(); + let heaps = dag + .nodes + .iter() + .filter(|n| { + n.output_state.timing == ExecutionTiming::IngestionTime + && match &n.payload { + Payload::SummaryAgg { + family: SummaryFamilyType::Sketch(kind, _), + .. + } => matches!( + kind.algorithm(), + SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap + ), + Payload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + .. + } => true, + _ => false, + } + }) + .collect::>(); + let ([source], [heap]) = (sources.as_slice(), heaps.as_slice()) else { + return Err(invalid( + "expected one selected fixed-window Rate aggregation", + )); + }; + if !source + .output_schema + .fields + .iter() + .any(|f| f.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "fixed-window Rate aggregation requires complete series identity", + )); + } + compile_candidate( + &dag, + BTreeMap::from([( + u64::from(source.id.0), + InputContract::bounded(Arc::new(source.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &[u64::from(heap.id.0)], + ) +} diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/asap-physical-operators/src/physical_planner/promql_values.rs new file mode 100644 index 00000000..98032505 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/promql_values.rs @@ -0,0 +1,278 @@ +//! Physical scalar/vector contracts preserve complete label sets across native computation. +use super::*; + +pub fn scalar_schema() -> Schema { + crate::operators::vector_binary::value_schema(true) +} +pub fn vector_schema() -> Schema { + crate::operators::vector_binary::value_schema(false) +} + +pub fn matrix_schema() -> Schema { + crate::operators::vector_window::matrix_schema() +} + +pub fn compile_scalar(value: f64) -> Result { + let operator = Operator::scalar( + crate::values::Value::Float64(value), + planner_types::pre_asap::DataType::Float64, + )? + .with_output_schema(scalar_schema())?; + CompiledPhysicalDag::from_operators( + BTreeMap::new(), + BTreeMap::from([(0, (vec![], operator))]), + vec![0], + ) +} + +pub fn compile_temporal( + intent: &AggIntent, + preserve_metric_name: bool, +) -> Result { + let operator = Operator::range_window(intent.clone())?; + let mut operators = vec![operator]; + if !preserve_metric_name { + operators.push(Operator::project( + vector_schema(), + vec![ + ( + "labels".into(), + Expression::LabelSet { + column: 0, + labels: vec![], + without: true, + }, + ), + ("value".into(), Expression::Column(1)), + ], + )?); + } + unary(operators, matrix_schema()) +} + +pub fn compile_histogram_quantile() -> Result { + CompiledPhysicalDag::from_operators( + BTreeMap::from([ + (0, InputContract::bounded(scalar_schema())), + (1, InputContract::bounded(vector_schema())), + ]), + BTreeMap::from([(2, (vec![0, 1], Operator::histogram_quantile()))]), + vec![2], + ) +} + +/// Compile before deployment chooses readers. Input slots 0 and 1 retain operand order. +pub fn compile_binary( + operator: &planner_types::post_asap::BinaryOperator, + return_bool: bool, + left_scalar: bool, + right_scalar: bool, +) -> Result { + let left = crate::operators::vector_binary::value_schema(left_scalar); + let right = crate::operators::vector_binary::value_schema(right_scalar); + let op = Operator::vector_binary(left.clone(), right.clone(), operator.clone(), return_bool)?; + CompiledPhysicalDag::from_operators( + BTreeMap::from([ + (0, InputContract::bounded(left)), + (1, InputContract::bounded(right)), + ]), + BTreeMap::from([(2, (vec![0, 1], op))]), + vec![2], + ) +} + +fn unary(operators: Vec, input: Schema) -> Result { + let root = operators.len() as u64; + CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(input))]), + operators + .into_iter() + .enumerate() + .map(|(i, op)| ((i + 1) as u64, (vec![i as u64], op))) + .collect(), + vec![root], + ) +} + +fn grouped(grouping: &GroupKeys) -> Result { + let labels = grouping + .keys() + .iter() + .map(|key| match key { + ColumnRef::Named(label) => Ok(label.clone()), + _ => Err(invalid("vector grouping requires label names")), + }) + .collect::, _>>()?; + Operator::project( + vector_schema(), + vec![ + ("labels".into(), Expression::Column(0)), + ("value".into(), Expression::Column(1)), + ( + "group".into(), + Expression::LabelSet { + column: 0, + labels, + without: grouping.is_without(), + }, + ), + ], + ) +} + +fn vector_output(input: Schema, labels: usize, value: usize) -> Result { + let value = Expression::ExactFloat64(value); + Operator::project( + input, + vec![ + ("labels".into(), Expression::Column(labels)), + ("value".into(), value), + ], + ) +} + +pub fn compile_aggregate( + intent: &AggIntent, + grouping: &GroupKeys, +) -> Result { + let project = grouped(grouping)?; + let reduction = match intent { + AggIntent::Sum { .. } => Reduction::Sum(1), + AggIntent::Avg { .. } => Reduction::Avg(1), + AggIntent::Count { .. } => Reduction::Count, + AggIntent::Min { .. } => Reduction::Min(1), + AggIntent::Max { .. } => Reduction::Max(1), + _ => return Err(invalid("unsupported vector aggregate")), + }; + let aggregate = + Operator::aggregate(project.schema(), vec![2], vec![("value".into(), reduction)])?; + let output = vector_output(aggregate.schema(), 0, 1)?; + unary(vec![project, aggregate, output], vector_schema()) +} + +pub fn compile_sort( + descending: bool, + grouping: &GroupKeys, +) -> Result { + let project = grouped(grouping)?; + let sort = Operator::sort( + project.schema(), + vec![SortKey { + column: 1, + descending, + nulls_first: false, + }], + vec![2], + )?; + let output = vector_output(sort.schema(), 0, 1)?; + unary(vec![project, sort, output], vector_schema()) +} + +pub fn compile_limit( + n: u64, + offset: u64, + grouping: &GroupKeys, +) -> Result { + let project = grouped(grouping)?; + let limit = Operator::limit(project.schema(), n, offset, vec![2])?; + let output = vector_output(limit.schema(), 0, 1)?; + unary(vec![project, limit, output], vector_schema()) +} + +pub fn compile_negate(scalar: bool) -> Result { + let input = if scalar { + scalar_schema() + } else { + vector_schema() + }; + let mut columns = Vec::new(); + if !scalar { + columns.push(("labels".into(), Expression::Column(0))); + } + columns.push(( + if scalar { + "$promql_scalar".into() + } else { + "value".into() + }, + Expression::Negate(Box::new(Expression::Column(if scalar { 0 } else { 1 }))), + )); + unary(vec![Operator::project(input.clone(), columns)?], input) +} + +pub fn compile_vector_to_scalar() -> Result { + unary( + vec![Operator::vector_to_scalar(vector_schema(), 1)?.with_output_schema(scalar_schema())?], + vector_schema(), + ) +} + +/// A stored exact-state input retains the complete population identity. The +/// deployment supplies eligible panes; merging and finalization are computation. +pub fn exact_state_schema(family: SummaryFamilyType) -> Result { + if !matches!(family, SummaryFamilyType::ExactAggregate(..)) { + return Err(invalid("exact-state input requires an exact family")); + } + crate::values::validate_family(&family)?; + let mut schema = (*vector_schema()).clone(); + schema.fields[1].dtype = family; + Ok(Arc::new(schema)) +} + +/// Retain exact readout semantics before any deployment state is opened. +pub fn compile_exact_readout( + family: SummaryFamilyType, + lookback_ms: u64, + preserve_metric_name: bool, +) -> Result { + use planner_types::post_asap::ExactKind; + let statistic = match &family { + SummaryFamilyType::ExactAggregate(kind, _) => match kind { + ExactKind::Sum => crate::Statistic::Sum, + ExactKind::Count => crate::Statistic::Count, + ExactKind::Min => crate::Statistic::Min, + ExactKind::Max => crate::Statistic::Max, + ExactKind::Rate => crate::Statistic::Rate, + ExactKind::Increase => crate::Statistic::Increase, + ExactKind::IRate => return Err(invalid("instant-rate state readout is not supported")), + }, + _ => return Err(invalid("exact readout requires an exact family")), + }; + let input = exact_state_schema(family)?; + let merge = Operator::summary_merge(input.clone(), 1, vec![0])?; + let mut readout = Operator::readout( + merge.schema(), + 1, + ReadoutQuery::Exact(ExactReadout { + statistic, + lookback_ms: None, + }), + )?; + if matches!( + statistic, + crate::Statistic::Rate | crate::Statistic::Increase + ) { + readout = readout.with_counter_lookback( + i64::try_from(lookback_ms).map_err(|_| invalid("counter lookback exceeds Int64"))?, + )?; + } + let project = Operator::project( + readout.schema(), + vec![ + ( + "labels".into(), + if preserve_metric_name { + Expression::Column(0) + } else { + Expression::LabelSet { + column: 0, + labels: vec![], + without: true, + } + }, + ), + ("value".into(), Expression::ExactFloat64(1)), + ], + )?; + unary(vec![merge, readout, project], input) +} diff --git a/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs b/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs new file mode 100644 index 00000000..d535c777 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs @@ -0,0 +1,331 @@ +//! Lower a selected temporal maintenance contract; deployment supplies readers. +use super::*; +use planner_types::post_asap::{ + EvaluationSchedule, OutputRepresentation, PaneLayout, SketchAlgorithm, + SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, + SummaryWindowFramework, +}; + +/// Resolved source identity, supplied with physical capability evidence. +/// A schemaless PromQL projection cannot establish the complete label set. +#[derive(Clone, Debug)] +pub enum TemporalEntityIdentity { + /// The input resolver guarantees that the slot contains one entity. + SingleEntity, + /// All entity keys are represented by these columns; there are no hidden + /// labels distinguishing two rows with the same key. + Columns(Vec), +} + +/// Planner-selected lifecycle/window requirements for one temporal producer. +/// Pane geometry is semantic input, not a storage identity or scheduling policy. +#[derive(Clone, Debug)] +pub struct TemporalPaneMaintenance { + pub summary_node: NodeId, + pub lifecycle: SummaryMaintenanceLifecycleGuarantee, + pub framework: SummaryWindowFramework, + pub layout: PaneLayout, + pub entity_identity: TemporalEntityIdentity, +} + +/// Generated precompute and query computation. `pane_inputs` is ordered from +/// the oldest complete pane to the newest; each run checks actual timestamps. +#[derive(Clone)] +pub struct TemporalPaneCandidate { + pub physical: PhysicalCandidate, + pub maintenance: TemporalPaneMaintenance, + pub pane_inputs: Vec, + pub merged_state: NodeId, + pub window_width_ms: u64, +} + +/// Compile bounded pane construction and a shared pane merge for temporal KLL +/// quantile roots. The selected contract remains authoritative; unsupported +/// lifecycle/framework/operator shapes fail rather than being substituted. +/// This initial realization consumes complete pane populations and emits full +/// state snapshots. Cross-run delta accumulation belongs to other candidates. +pub fn compile_temporal_pane_candidate( + dag: &PostAsapDag, + inputs: BTreeMap, + roots: &[NodeId], + maintenance: &TemporalPaneMaintenance, +) -> Result { + dag.validate().map_err(|error| invalid(error.to_string()))?; + if maintenance.lifecycle.summary_maintenance_lifecycle + != SummaryMaintenanceLifecycle::ContinuouslyMaintained + || maintenance.lifecycle.summary_maintenance_mode != SummaryMaintenanceMode::Incremental + || maintenance.lifecycle.evaluation_schedule != EvaluationSchedule::PerUpdate + || maintenance.lifecycle.output_representation != OutputRepresentation::SummaryState + { + return Err(invalid( + "pane candidate requires continuous incremental summary maintenance", + )); + } + let build = dag + .nodes + .iter() + .find(|node| u64::from(node.id.0) == maintenance.summary_node) + .ok_or_else(|| invalid("unknown maintained producer"))?; + let Payload::SummaryAgg { + family, + input: update, + reduction: PlannerReduction::PerEntity, + grouping, + } = &build.payload + else { + return Err(invalid( + "pane candidate requires a temporal per-entity summary", + )); + }; + if !matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll) + || update.item.is_some() + { + return Err(invalid("pane candidate supports unkeyed temporal KLL only")); + } + crate::capability::validate_summary_kernel(family, update, grouping).map_err(Error::Invalid)?; + let dependencies: Vec<_> = dag + .edges + .iter() + .filter(|edge| edge.consumer == build.id) + .map(|edge| edge.producer) + .collect(); + let [raw_id] = dependencies.as_slice() else { + return Err(invalid("temporal producer requires one raw input")); + }; + let raw = dag + .nodes + .iter() + .find(|node| node.id == *raw_id) + .ok_or_else(|| invalid("missing raw input"))?; + let Payload::Fallback { + expression: QueryExpr::TimeRange { range, child }, + } = &raw.payload + else { + return Err(invalid( + "temporal producer requires an explicit logical time range", + )); + }; + let QueryExpr::Scan { predicates, .. } = child.as_ref() else { + return Err(invalid("temporal pane source requires a raw scan")); + }; + let window_width_ms: u64 = range + .as_millis() + .try_into() + .map_err(|_| invalid("temporal window overflows"))?; + if window_width_ms == 0 + || window_width_ms > i64::MAX as u64 + || range.subsec_nanos() % 1_000_000 != 0 + { + return Err(invalid( + "temporal window requires positive integral milliseconds", + )); + } + let width = maintenance.layout.pane_width_ms; + if width == 0 || width > window_width_ms || !window_width_ms.is_multiple_of(width) { + return Err(invalid("temporal window must contain whole panes")); + } + match maintenance.framework { + SummaryWindowFramework::Sliding => {} + SummaryWindowFramework::Tumbling if width == window_width_ms => {} + _ => return Err(invalid("unsupported temporal window realization")), + } + let count = window_width_ms / width; + if count > 4096 { + return Err(invalid("temporal pane candidate exceeds input budget")); + } + let raw_id = u64::from(raw_id.0); + if inputs.len() != 1 { + return Err(invalid( + "pane candidate requires exactly its raw input contract", + )); + } + let contract = inputs + .get(&raw_id) + .ok_or_else(|| invalid("missing raw input contract"))?; + let raw_schema = Arc::new(raw.output_schema.clone()); + if contract.schema != raw_schema || contract.properties.boundedness != Boundedness::Bounded { + return Err(invalid("pane source requires its declared bounded schema")); + } + let coordinate = raw_schema + .time_index + .ok_or_else(|| invalid("temporal source requires a time index"))?; + let SummaryInputExpr::Column(value) = &update.weight else { + return Err(invalid("pane builder requires a value column")); + }; + let value = named_column(&raw_schema, value)?; + let groups: Vec<_> = (0..raw_schema.fields.len()) + .filter(|&index| index != coordinate && index != value) + .collect(); + match &maintenance.entity_identity { + TemporalEntityIdentity::SingleEntity if groups.is_empty() => {} + TemporalEntityIdentity::Columns(columns) + if !columns.is_empty() + && columns.len() == columns.iter().collect::>().len() + && columns.iter().copied().collect::>() + == groups.iter().copied().collect() => {} + _ => { + return Err(invalid( + "pane input requires its complete resolved entity identity", + )) + } + } + let mut next = dag + .nodes + .iter() + .map(|node| u64::from(node.id.0)) + .max() + .unwrap_or(0) + + 1; + let mut allocate = || { + let id = next; + next += 1; + id + }; + let mut operators = BTreeMap::new(); + let guard = allocate(); + operators.insert( + guard, + ( + vec![raw_id], + Operator::pane_input( + raw_schema.clone(), + coordinate, + maintenance.layout.clone(), + None, + )?, + ), + ); + let mut previous = guard; + for predicate in predicates { + let id = allocate(); + operators.insert( + id, + ( + vec![previous], + Operator::filter(raw_schema.clone(), expression(&predicate.0, &raw_schema)?)?, + ), + ); + previous = id; + } + let native = + Operator::summary_build(raw_schema, family.clone(), value, Some(coordinate), groups)?; + let compact_state = native.schema(); + let native_id = allocate(); + operators.insert(native_id, (vec![previous], native)); + let state_schema = Arc::new(build.output_schema.clone()); + let pane_output = allocate(); + operators.insert( + pane_output, + ( + vec![native_id], + Operator::scope_timestamp(compact_state, state_schema.clone())?, + ), + ); + let precompute = CompiledPhysicalDag::from_operators(inputs, operators, vec![pane_output])?; + let state_coordinate = state_schema + .time_index + .ok_or_else(|| invalid("pane state requires a time index"))?; + let state_column = summary_column(&state_schema)?; + let mut query_inputs = BTreeMap::new(); + let mut operators = BTreeMap::new(); + let mut pane_inputs = Vec::new(); + let mut guarded_inputs = Vec::new(); + for pane in 0..count { + let input = allocate(); + let guard = allocate(); + query_inputs.insert(input, InputContract::bounded(state_schema.clone())); + let offset = ((count - 1 - pane) * width) as i64; + operators.insert( + guard, + ( + vec![input], + Operator::pane_input( + state_schema.clone(), + state_coordinate, + maintenance.layout.clone(), + Some(offset), + )?, + ), + ); + pane_inputs.push(input); + guarded_inputs.push(guard); + } + let union = allocate(); + operators.insert( + union, + ( + guarded_inputs, + Operator::union(state_schema.clone(), count as usize)?, + ), + ); + let merge = Operator::summary_merge( + state_schema.clone(), + state_column, + (0..state_schema.fields.len()) + .filter(|&index| index != state_coordinate && index != state_column) + .collect(), + )?; + let merged_schema = merge.schema(); + let merged_state = allocate(); + operators.insert(merged_state, (vec![union], merge)); + if roots.is_empty() || roots.iter().copied().collect::>().len() != roots.len() { + return Err(invalid("temporal query requires distinct output roots")); + } + for &root in roots { + let node = dag + .nodes + .iter() + .find(|node| u64::from(node.id.0) == root) + .ok_or_else(|| invalid("unknown temporal output root"))?; + let Payload::SummaryEstimate { + query: SketchQuery::Quantile { q }, + } = node.payload + else { + return Err(invalid("temporal pane root must be a KLL quantile")); + }; + if !q.is_finite() || !(0. ..=1.).contains(&q) { + return Err(invalid("invalid temporal quantile")); + } + let dependencies: Vec<_> = dag + .edges + .iter() + .filter(|edge| edge.consumer == node.id) + .map(|edge| u64::from(edge.producer.0)) + .collect(); + if dependencies != [maintenance.summary_node] { + return Err(invalid( + "temporal readout must consume the maintained producer", + )); + } + let readout = Operator::readout( + merged_schema.clone(), + summary_column(&merged_schema)?, + ReadoutQuery::Sketch(SketchQuery::Quantile { q }), + )?; + let readout_schema = readout.schema(); + let readout_id = allocate(); + operators.insert(readout_id, (vec![merged_state], readout)); + operators.insert( + root, + ( + vec![readout_id], + Operator::scope_timestamp(readout_schema, Arc::new(node.output_schema.clone()))?, + ), + ); + } + let query = CompiledPhysicalDag::from_operators(query_inputs, operators, roots.to_vec())?; + let mut output = precompute.output_contract(pane_output)?; + // Persisted readers have independent timing from the blocking builder. + output.properties.emission = Emission::Unknown; + Ok(TemporalPaneCandidate { + physical: PhysicalCandidate { + precompute: Some(precompute), + query, + materialized_outputs: BTreeMap::from([(pane_output, output)]), + }, + maintenance: maintenance.clone(), + pane_inputs, + merged_state, + window_width_ms, + }) +} diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs new file mode 100644 index 00000000..322bd6fe --- /dev/null +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -0,0 +1,401 @@ +//! Spatial heap weights come from a fresh instant vector, never sample history. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{ + promql_rows::{decode_series_identity, series_row, SERIES_IDENTITY_COLUMN}, + CompiledPhysicalDag, InputContract, Source, + }, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{post_asap::*, pre_asap::DataType}; +use std::{collections::BTreeMap, sync::Arc}; + +fn schema() -> Arc { + Arc::new(SummarySchema { + fields: [ + ("ts", DataType::Timestamp), + ("value", DataType::Float64), + ("job", DataType::Utf8), + (SERIES_IDENTITY_COLUMN, DataType::Utf8), + ] + .into_iter() + .map(|(name, dtype)| SummaryField { + name: name.into(), + dtype: SummaryFamilyType::Plain(dtype), + nullable: false, + }) + .collect(), + time_index: Some(0), + }) +} +fn run(program: &CompiledPhysicalDag, data: Batch, end: i64) -> Result, String> { + let recovered = serde_json::from_slice::( + &serde_json::to_vec(&program).map_err(|e| e.to_string())?, + ) + .map_err(|e| e.to_string())?; + let input_id = recovered.input_contracts().next().unwrap().0; + let graph = recovered + .instantiate(BTreeMap::from([( + input_id, + Box::new(Operator::source(data.schema().clone(), vec![data]).unwrap()) as Source<'_>, + )])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: end, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = graph.execute(recovered.roots(), context).unwrap().remove(0); + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push((*batch.map_err(|e| e.to_string())?).clone()); + } + Ok(batches) + }) +} +fn input(samples: &[(&str, i64, f64)]) -> Batch { + let schema = schema(); + let rows = samples + .iter() + .map(|(instance, time, value)| { + series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("hidden_instance".into(), (*instance).into()), + ]), + *time, + *value, + ) + .unwrap() + }) + .collect(); + Batch::try_new(schema, rows).unwrap() +} +fn snapshot_plan() -> CompiledPhysicalDag { + CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema()))]), + BTreeMap::from([( + 1, + ( + vec![0], + Operator::current_series(schema(), 3, 0, 1, 60_000).unwrap(), + ), + )]), + vec![1], + ) + .unwrap() +} + +// Replacement, expiry and stale markers act before sketch updates. Hidden labels +// survive even when every series has the same projected `job` value. +#[test] +fn latest_snapshot_replaces_decreases_expires_and_retains_full_identity() { + let plan = snapshot_plan(); + let batches = run( + &plan, + input(&[ + ("decrease", 10_000, 100.), + ("decrease", 50_000, 1.), + ("steady", 40_000, 20.), + ("expired", 0, 1_000.), + ("stale", 20_000, 500.), + ("stale", 55_000, f64::from_bits(0x7ff0_0000_0000_0002)), + ("future", 60_001, 2_000.), + ]), + 60_000, + ) + .unwrap(); + let values = batches + .iter() + .flat_map(|batch| batch.rows()) + .map(|row| { + let Value::Utf8(identity) = &row[3] else { + panic!() + }; + let Value::Float64(value) = row[1] else { + panic!() + }; + assert!(matches!(row[0], Value::Timestamp(60_000))); + ( + decode_series_identity(identity).unwrap()["hidden_instance"].clone(), + value, + ) + }) + .collect::>(); + assert_eq!( + values, + BTreeMap::from([("decrease".into(), 1.), ("steady".into(), 20.)]) + ); + assert!(run(&plan, input(&[("steady", 40_000, 20.)]), 100_000) + .unwrap() + .iter() + .all(|batch| batch.rows().is_empty())); + assert!(run( + &plan, + input(&[("conflict", 50_000, 1.), ("conflict", 50_000, 2.)]), + 60_000 + ) + .is_err()); +} + +#[test] +fn spatial_heap_ranks_latest_values_in_independent_runs() { + for algorithm in [ + SketchAlgorithm::CmsWithHeap, + SketchAlgorithm::CountSketchWithHeap, + ] { + let params = match algorithm { + SketchAlgorithm::CmsWithHeap => SketchParams::CmsWithHeap { + width: 2048, + depth: 5, + heap_size: 100, + }, + _ => SketchParams::CountSketchWithHeap { + width: 2048, + depth: 5, + heap_size: 100, + }, + }; + let family = + SummaryFamilyType::Sketch(SketchKind::new(algorithm, params), Default::default()); + let build = Operator::keyed_summary_build(schema(), family, 1, vec![3], vec![2]).unwrap(); + let output = Arc::new(SummarySchema { + fields: vec![ + schema().fields[2].clone(), + schema().fields[3].clone(), + schema().fields[1].clone(), + ], + time_index: None, + }); + let read = Operator::keyed_readout(build.schema(), 1, 1, output).unwrap(); + let plan = CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema()))]), + BTreeMap::from([ + ( + 1, + ( + vec![0], + Operator::current_series(schema(), 3, 0, 1, 60_000).unwrap(), + ), + ), + (2, (vec![1], build)), + (3, (vec![2], read)), + ]), + vec![3], + ) + .unwrap(); + for (samples, end, winner, score) in [ + ( + vec![("a", 10_000, 100.), ("a", 50_000, 1.), ("b", 50_000, 20.)], + 60_000, + "b", + 20., + ), + ( + vec![("a", 110_000, 3.), ("b", 50_000, 20.)], + 120_000, + "a", + 3., + ), + ] { + let batches = run(&plan, input(&samples), end).unwrap(); + let rows = batches + .iter() + .flat_map(|batch| batch.rows()) + .collect::>(); + assert_eq!(rows.len(), 1); + let Value::Utf8(encoded) = &rows[0][1] else { + panic!() + }; + assert_eq!( + decode_series_identity(encoded).unwrap()["hidden_instance"], + winner + ); + assert!(matches!(rows[0][2], Value::Float64(actual) if actual == score)); + } + } +} + +// Blocking membership selection shares the run's cancellation and byte budget. +#[test] +fn current_series_observes_resource_limits() { + use asap_physical_operators::Error; + let plan = snapshot_plan(); + for cancelled in [false, true] { + let data = input(&[("one", 50_000, 1.)]); + let graph = plan + .instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(data.schema().clone(), vec![data]).unwrap()) + as Source<'_>, + )])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 60_000, + revision: 0, + }, + Limits { + max_bytes: if cancelled { 1 << 20 } else { 1 }, + ..Limits::default() + }, + ) + .unwrap(); + if cancelled { + context.cancel(); + } + let result = match graph.execute(&[1], context.clone()) { + Err(error) => Err(error), + Ok(mut streams) => block_on(streams.remove(0).next()).unwrap().map(|_| ()), + }; + assert!(matches!( + (cancelled, result), + (true, Err(Error::Cancelled)) | (false, Err(Error::MemoryLimit)) + )); + assert_eq!(context.retained_bytes(), 0); + } +} + +#[test] +fn identity_encoding_is_lossless_and_rejects_noncanonical_inputs() { + use asap_physical_operators::physical_planner::promql_rows::encode_series_identity; + let labels = BTreeMap::from([ + ("a".into(), "quote\"slash\\".into()), + ("other".into(), "".into()), + ]); + assert_eq!( + decode_series_identity(&encode_series_identity(&labels).unwrap()).unwrap(), + labels + ); + for invalid in [ + "[]", + "{\"a\":1}", + "{\"a\":\"x\",\"a\":\"x\"}", + "{ \"a\":\"x\"}", + ] { + assert!(decode_series_identity(invalid).is_err(), "{invalid}"); + } +} + +// The actual Planner population candidate lowers to native operators; this +// test does not manually assemble the computation or its dependency edges. +#[test] +fn planner_current_series_candidate_compiles_with_dynamic_identity() { + use asap_physical_operators::physical_planner::{compile, promql_rows::with_series_identity}; + use planner_types::{types::AccuracyTarget, workload::*}; + use std::rc::Rc; + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("topk by(job)(1, m)".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(60_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let original = asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0); + let open_root = Rc::new(original.clone()); + let open_selected = + asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&open_root), + ) + .candidate(&open_root) + .unwrap(); + let snapshot_program = + asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + &open_selected, + ) + .unwrap(); + let encoded = String::from_utf8(serde_json::to_vec(&snapshot_program).unwrap()).unwrap(); + assert!( + !encoded.contains("CurrentSeries"), + "maintained input must not be rebuilt" + ); + assert!(encoded.contains("Sort") && encoded.contains("Limit")); + assert_eq!(snapshot_program.input_contracts().count(), 1); + let root = Rc::new(with_series_identity(&original).unwrap()); + let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&root), + ) + .candidate(&root) + .unwrap(); + let logical = compile_post_asap_dag(&selected).unwrap(); + let raw = logical + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .unwrap(); + let raw_schema = Arc::new(raw.output_schema.clone()); + let physical = compile( + &logical, + BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(raw_schema.clone()), + )]), + &[u64::from(logical.root.0)], + ) + .unwrap(); + let bytes = String::from_utf8(serde_json::to_vec(&physical).unwrap()).unwrap(); + assert!(bytes.contains("CurrentSeries")); + assert!(bytes.contains("Sort")); + assert!(bytes.contains("Limit")); + let rows = [("a", 10_000, 100.), ("a", 50_000, 1.), ("b", 50_000, 20.)] + .into_iter() + .map(|(member, at, value)| { + series_row( + &raw_schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("unreferenced".into(), member.into()), + ]), + at, + value, + ) + .unwrap() + }) + .collect(); + let batches = run(&physical, Batch::try_new(raw_schema, rows).unwrap(), 60_000).unwrap(); + let rows = batches + .iter() + .flat_map(|batch| batch.rows()) + .collect::>(); + assert_eq!(rows.len(), 1); + assert!(matches!(rows[0][1], Value::Float64(20.))); + let id = batches[0] + .schema() + .fields + .iter() + .position(|field| field.name == SERIES_IDENTITY_COLUMN) + .unwrap(); + let Value::Utf8(encoded) = &rows[0][id] else { + panic!() + }; + assert_eq!( + decode_series_identity(encoded).unwrap()["unreferenced"], + "b" + ); +} diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs new file mode 100644 index 00000000..96e74e54 --- /dev/null +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -0,0 +1,1443 @@ +//! Acceptance tests use the library directly, without either backend engine. +use asap_physical_operators::{ + dag::{ + operators::{Expression, Operator, Reduction, SortKey}, + values::{Batch, Schema, Value}, + Limits, PhysicalDag, RunContext, Scope, + }, + Statistic, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{ExactKind, ExactParams, SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, +}; +use std::sync::Arc; +fn schema(fields: &[(&str, DataType, bool)]) -> Schema { + Arc::new(SummarySchema { + fields: fields + .iter() + .map(|(name, dtype, nullable)| SummaryField { + name: (*name).into(), + dtype: SummaryFamilyType::Plain(dtype.clone()), + nullable: *nullable, + }) + .collect(), + time_index: None, + }) +} +fn run(dag: &PhysicalDag<'_, Batch, Schema>, root: u64, scope: Scope) -> Vec> { + let context = RunContext::new( + scope, + Limits { + max_buffered_batches: 1, + ..Limits::default() + }, + ) + .unwrap(); + block_on(async { + let mut stream = dag.execute(&[root], context.clone()).unwrap().remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend(batch.unwrap().rows().iter().cloned()); + } + assert_eq!(context.retained_bytes(), 0); + rows + }) +} +fn query() -> Scope { + Scope::Query { + evaluation_time_ms: 1000, + revision: 2, + } +} +fn floats(rows: &[Vec], column: usize) -> Vec { + rows.iter() + .map(|r| { + if let Value::Float64(v) = r[column] { + v + } else { + panic!("not Float64") + } + }) + .collect() +} + +// Sort followed by partitioned Limit implements ranking independently per group. +#[test] +fn grouped_sort_limit_across_batches() { + let schema = schema(&[ + ("group", DataType::Int64, false), + ("score", DataType::Float64, false), + ]); + let batches = [ + vec![(1, 1.), (2, 4.), (1, 9.)], + vec![(2, 8.), (1, 5.), (2, 2.)], + ] + .into_iter() + .map(|rows| { + Batch::try_new( + schema.clone(), + rows.into_iter() + .map(|(g, v)| vec![Value::Int64(g), Value::Float64(v)]) + .collect(), + ) + .unwrap() + }) + .collect(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(schema.clone(), batches).unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::sort( + schema.clone(), + vec![SortKey { + column: 1, + descending: true, + nulls_first: false, + }], + vec![0], + ) + .unwrap(), + ) + .unwrap(); + dag.add(2, vec![1], Operator::limit(schema, 1, 1, vec![0]).unwrap()) + .unwrap(); + assert_eq!(floats(&run(&dag, 2, query()), 1), vec![5., 4.]); +} + +// The same computation runs in either engine scope with fresh per-run state. +#[test] +fn summary_construction_merge_and_readout_at_both_phases() { + let schema = schema(&[("v", DataType::Float64, false)]); + let batches = (1..=20) + .map(|v| Batch::try_new(schema.clone(), vec![vec![Value::Float64(v as f64)]]).unwrap()) + .collect(); + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let build = Operator::summary_build(schema.clone(), family, 0, None, vec![]).unwrap(); + let state = build.schema(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(schema, batches).unwrap()) + .unwrap(); + dag.add(1, vec![0], build).unwrap(); + dag.add(2, vec![1, 1], Operator::union(state.clone(), 2).unwrap()) + .unwrap(); + dag.add( + 3, + vec![2], + Operator::summary_merge(state.clone(), 0, vec![]).unwrap(), + ) + .unwrap(); + dag.add( + 4, + vec![3], + Operator::readout( + state, + 0, + asap_physical_operators::operators::ReadoutQuery::Exact( + asap_physical_operators::summary_kernels::exact::ExactReadout { + statistic: Statistic::Sum, + lookback_ms: None, + }, + ), + ) + .unwrap(), + ) + .unwrap(); + for scope in [ + query(), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1000, + revision: 2, + }, + ] { + assert_eq!(floats(&run(&dag, 4, scope), 0), vec![420.]); + } +} + +// A semi-join can consume two branches of one producer with a one-batch buffer. +#[test] +fn diamond_semijoin_preserves_left_values_and_multiplicity() { + let schema = schema(&[("key", DataType::Int64, false)]); + let batches = [1, 2, 2, 3] + .into_iter() + .map(|v| Batch::try_new(schema.clone(), vec![vec![Value::Int64(v)]]).unwrap()) + .collect(); + let filter = Operator::filter( + schema.clone(), + Expression::Equal( + Box::new(Expression::Column(0)), + Box::new(Expression::Literal { + value: Value::Int64(2), + dtype: DataType::Int64, + }), + ), + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(schema.clone(), batches).unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], filter).unwrap(); + dag.add( + 2, + vec![0, 1], + Operator::semi_join(schema.clone(), schema, vec![(0, 0)]).unwrap(), + ) + .unwrap(); + let rows = run(&dag, 2, query()); + assert_eq!(rows.len(), 2); + assert!(rows.iter().all(|r| matches!(r[0], Value::Int64(2)))); +} + +// Integer aggregation must not silently lose precision through Float64. +#[test] +fn exact_integer_and_empty_extrema() { + let schema = schema(&[("v", DataType::Int64, false)]); + let aggregate = Operator::aggregate( + schema.clone(), + vec![], + vec![("sum".into(), Reduction::Sum(0))], + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + let value = 9_007_199_254_740_993; + dag.add( + 0, + vec![], + Operator::source( + schema.clone(), + vec![Batch::try_new( + schema.clone(), + vec![vec![Value::Int64(value)], vec![Value::Int64(2)]], + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], aggregate).unwrap(); + assert!(matches!(run(&dag,1,query())[0][0],Value::Int64(v) if v==value+2)); + let mut empty = PhysicalDag::default(); + empty + .add(0, vec![], Operator::source(schema.clone(), vec![]).unwrap()) + .unwrap(); + empty + .add( + 1, + vec![0], + Operator::aggregate(schema, vec![], vec![("min".into(), Reduction::Min(0))]).unwrap(), + ) + .unwrap(); + assert!(matches!(run(&empty, 1, query())[0][0], Value::Null)); +} + +// Plain value operators are library implementations, including NaN comparison. +#[test] +fn scalar_negation_and_vector_conversion() { + let scalar = Operator::scalar(Value::Float64(7.), DataType::Float64).unwrap(); + let project = Operator::project( + scalar.schema(), + vec![( + "v".into(), + Expression::Negate(Box::new(Expression::Column(0))), + )], + ) + .unwrap(); + let convert = Operator::vector_to_scalar(project.schema(), 0).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], scalar).unwrap(); + dag.add(1, vec![0], project).unwrap(); + dag.add(2, vec![1], convert).unwrap(); + assert_eq!(floats(&run(&dag, 2, query()), 0), vec![-7.]); + let scalar = Operator::scalar(Value::Float64(f64::NAN), DataType::Float64).unwrap(); + let predicate = Expression::Equal( + Box::new(Expression::Column(0)), + Box::new(Expression::Column(0)), + ); + let filter = Operator::filter(scalar.schema(), predicate).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], scalar).unwrap(); + dag.add(1, vec![0], filter).unwrap(); + assert!(run(&dag, 1, query()).is_empty()); +} + +// Invalid operations fail at binding rather than becoming external fallbacks. +#[test] +fn binding_rejects_unsupported_operations() { + let schema = schema(&[("v", DataType::Float64, false)]); + assert!(Operator::summary_build( + schema.clone(), + SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), + 0, + None, + vec![] + ) + .is_err()); + let sum = Operator::summary_build( + schema.clone(), + SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + 0, + None, + vec![], + ) + .unwrap(); + assert!(Operator::readout( + sum.schema(), + 0, + asap_physical_operators::operators::ReadoutQuery::Sketch( + planner_types::post_asap::SketchQuery::Quantile { q: 0.5 } + ) + ) + .is_err()); + assert!(Operator::filter(schema, Expression::Column(0)).is_err()); +} + +// KLL is one family example: precomputation changes input sources, not operators. +#[test] +fn kll_raw_partial_and_precomputed_are_native_dags() { + use planner_types::post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("value", DataType::Float64, false)]); + let family = SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), + GroupingStrategy::PerSubpopulationInstance, + ); + let build = Operator::summary_build(input.clone(), family, 0, None, vec![]).unwrap(); + let state = build.schema(); + let build_range = |start: u32, end: u32| { + let mut dag = PhysicalDag::default(); + let batch = Batch::try_new( + input.clone(), + (start..end) + .map(|v| vec![Value::Float64(f64::from(v))]) + .collect(), + ) + .unwrap(); + dag.add( + 0, + vec![], + Operator::source(input.clone(), vec![batch]).unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], build.clone()).unwrap(); + run( + &dag, + 1, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1000, + revision: 1, + }, + ) + }; + let prefix = build_range(0, 64); + let complete = build_range(0, 128); + let query_plan = |stored: Option>>, raw_start: Option| { + let mut dag = PhysicalDag::default(); + let mut states = vec![]; + if let Some(rows) = stored { + dag.add( + 0, + vec![], + Operator::source( + state.clone(), + vec![Batch::try_new(state.clone(), rows).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + states.push(0); + } + if let Some(start) = raw_start { + dag.add( + 1, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new( + input.clone(), + (start..128) + .map(|v| vec![Value::Float64(f64::from(v))]) + .collect(), + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add(2, vec![1], build.clone()).unwrap(); + states.push(2); + } + dag.add( + 3, + states.clone(), + Operator::union(state.clone(), states.len()).unwrap(), + ) + .unwrap(); + dag.add( + 4, + vec![3], + Operator::summary_merge(state.clone(), 0, vec![]).unwrap(), + ) + .unwrap(); + dag.add( + 5, + vec![4], + Operator::readout( + state.clone(), + 0, + asap_physical_operators::operators::ReadoutQuery::Sketch( + planner_types::post_asap::SketchQuery::Quantile { q: 0.5 }, + ), + ) + .unwrap(), + ) + .unwrap(); + floats(&run(&dag, 5, query()), 0)[0] + }; + let raw = query_plan(None, Some(0)); + let partial = query_plan(Some(prefix), Some(64)); + let full = query_plan(Some(complete), None); + assert_eq!(raw, partial); + assert_eq!(partial, full); + assert!((raw - 64.).abs() <= 1.); +} + +// Exact state must match its declared family; a mislabeled state is rejected. +#[test] +fn exact_state_and_family_validation() { + use asap_physical_operators::summary_kernels::exact::ExactAccumulator; + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); + acc.update(None, 7., 0); + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }], + time_index: None, + }); + let value = Value::Summary { + family: family.clone(), + state: Arc::new(acc), + }; + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + schema.clone(), + vec![Batch::try_new(schema.clone(), vec![vec![value]]).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::readout( + schema.clone(), + 0, + asap_physical_operators::operators::ReadoutQuery::Exact( + asap_physical_operators::summary_kernels::exact::ExactReadout { + statistic: Statistic::Sum, + lookback_ms: None, + }, + ), + ) + .unwrap(), + ) + .unwrap(); + assert_eq!(floats(&run(&dag, 1, query()), 0), vec![7.]); + let wrong = ExactAccumulator::new( + SummaryFamilyType::ExactAggregate(ExactKind::Max, ExactParams::Max), + false, + ) + .unwrap(); + assert!(Batch::try_new( + schema, + vec![vec![Value::Summary { + family, + state: Arc::new(wrong) + }]] + ) + .is_err()); +} + +// Planner binding rejects unknown computation instead of accepting a fallback. +#[test] +fn bind_post_asap_before_execution() { + use asap_physical_operators::dag::planner::bind; + use planner_types::{ + post_asap::{ + EdgeRole, ExecutionDataState, GroupingEdgeCompatibility, PostAsapDag, PostAsapDagEdge, + PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, ValueOperation, + WindowEdgeCompatibility, + }, + pre_asap::{ArithmeticOpKind, ProjectItem, QueryExpr, ScalarValue}, + }; + use std::{collections::BTreeMap, rc::Rc}; + let schema = schema(&[("value", DataType::Float64, false)]); + let node = |id, payload| PostAsapDagNode { + id: PostAsapNodeId(id), + payload, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*schema).clone(), + guarantee: None, + }; + let mut dag = PostAsapDag { + nodes: vec![ + node( + 0, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(1.), + }, + ), + node( + 1, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Project { + cols: vec![ProjectItem { + alias: None, + expr: QueryExpr::Arithmetic { + op: ArithmeticOpKind::Add, + left: Rc::new(QueryExpr::Column(0)), + right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.))), + }, + }], + qualifier: None, + }, + }, + ), + ], + edges: vec![PostAsapDagEdge { + producer: PostAsapNodeId(0), + consumer: PostAsapNodeId(1), + role: EdgeRole::Input, + intermediate_schema: (*schema).clone(), + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }], + root: PostAsapNodeId(1), + }; + let sources = || -> BTreeMap> { + BTreeMap::from([( + 0, + Box::new( + Operator::source( + schema.clone(), + vec![Batch::try_new(schema.clone(), vec![vec![Value::Float64(1.)]]).unwrap()], + ) + .unwrap(), + ) as asap_physical_operators::dag::planner::Source<'static>, + )]) + }; + let native = bind(&dag, sources(), &[1]).unwrap(); + assert_eq!(floats(&run(&native, 1, query()), 0), vec![3.]); + assert!(bind(&dag, BTreeMap::new(), &[1]).is_err()); + dag.nodes[1].payload = PostAsapOperatorPayload::Value { + operation: ValueOperation::Extension { + name: "unknown".into(), + }, + }; + assert!(bind(&dag, sources(), &[1]).is_err()); +} + +// A completed empty population has an exact zero count, with integer output. +#[test] +fn empty_exact_count_is_an_integer_state_readout() { + let input = schema(&[("value", DataType::Float64, false)]); + let build = Operator::summary_build( + input.clone(), + SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count), + 0, + None, + vec![], + ) + .unwrap(); + let read = Operator::readout( + build.schema(), + 0, + asap_physical_operators::operators::ReadoutQuery::Exact( + asap_physical_operators::summary_kernels::exact::ExactReadout { + statistic: Statistic::Count, + lookback_ms: None, + }, + ), + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(input, vec![]).unwrap()) + .unwrap(); + dag.add(1, vec![0], build).unwrap(); + dag.add(2, vec![1], read).unwrap(); + assert!(matches!(run(&dag, 2, query())[0][0], Value::Int64(0))); +} + +// A deployment source cannot pass a different row shape to bound expressions. +#[test] +fn source_batches_must_match_the_bound_schema() { + use asap_physical_operators::dag::{self, PhysicalOperator}; + use planner_types::{ + post_asap::{ + ExecutionDataState, PostAsapDag, PostAsapDagNode, PostAsapNodeId, + PostAsapOperatorPayload, + }, + pre_asap::QueryExpr, + }; + use std::{cell::Cell, collections::BTreeMap, rc::Rc}; + struct WrongSource { + schema: Schema, + starts: Rc>, + } + impl PhysicalOperator for WrongSource { + fn name(&self) -> &str { + "ExternalSource" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.schema.clone() + } + fn output_bytes(&self, value: &Batch) -> usize { + value.bytes() + } + fn start<'a>( + &'a self, + _: Vec>, + _: RunContext, + ) -> Result, dag::Error> { + self.starts.set(self.starts.get() + 1); + Ok( + futures::stream::once(async { Batch::try_new(schema(&[]), vec![vec![]]) }) + .boxed_local(), + ) + } + } + let expected = schema(&[("value", DataType::Float64, false)]); + let starts = Rc::new(Cell::new(0)); + let plan = PostAsapDag { + nodes: vec![PostAsapDagNode { + id: PostAsapNodeId(0), + payload: PostAsapOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(1.), + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*expected).clone(), + guarantee: None, + }], + edges: vec![], + root: PostAsapNodeId(0), + }; + let source = Box::new(WrongSource { + schema: expected, + starts: starts.clone(), + }) as dag::planner::Source<'static>; + let native = dag::planner::bind(&plan, BTreeMap::from([(0, source)]), &[0]).unwrap(); + assert_eq!(starts.get(), 0); + let context = RunContext::new(query(), Limits::default()).unwrap(); + let mut output = native.execute(&[0], context).unwrap().remove(0); + assert!(matches!( + block_on(output.next()), + Some(Err(dag::Error::AtNode { node: 0, .. })) + )); + assert_eq!(starts.get(), 1); +} + +// Float extrema have the same NaN behavior as the exact summary kernels. +#[test] +fn extrema_preserve_numeric_values_in_the_presence_of_nan() { + let input = schema(&[("v", DataType::Float64, false)]); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new( + input.clone(), + vec![vec![Value::Float64(-f64::NAN)], vec![Value::Float64(5.)]], + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::aggregate( + input, + vec![], + vec![ + ("min".into(), Reduction::Min(0)), + ("max".into(), Reduction::Max(0)), + ], + ) + .unwrap(), + ) + .unwrap(); + let rows = run(&dag, 1, query()); + assert_eq!(floats(&rows, 0), vec![5.]); + assert_eq!(floats(&rows, 1), vec![5.]); +} + +// Planner wire nodes, including grouping and edge roles, are executable at either phase. +#[test] +fn planner_semijoin_sort_limit_contract_at_both_phases() { + use asap_physical_operators::dag::planner::{bind, Source}; + use planner_types::{ + post_asap::*, + pre_asap::{CompareOpKind, GroupKeys, JoinKind, Predicate, QueryExpr, SortKey}, + }; + use std::{collections::BTreeMap, rc::Rc}; + let rows_schema = schema(&[ + ("group", DataType::Utf8, false), + ("key", DataType::Utf8, false), + ("score", DataType::Float64, false), + ]); + let keys_schema = schema(&[("key", DataType::Utf8, false)]); + let node = |id, payload, schema: &Schema| PostAsapDagNode { + id: PostAsapNodeId(id), + payload, + output_schema: (**schema).clone(), + output_state: ExecutionDataState::QUERY_ROWS, + guarantee: None, + }; + let edge = |producer, consumer, role, schema: &Schema| PostAsapDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role, + intermediate_schema: (**schema).clone(), + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }; + let groups = GroupKeys::by(vec![0]); + let dag = PostAsapDag { + nodes: vec![ + node( + 0, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(0.), + }, + &rows_schema, + ), + node( + 1, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(0.), + }, + &keys_schema, + ), + node( + 2, + PostAsapOperatorPayload::RelationalJoin { + join_kind: JoinKind::Semi, + pruning: None, + pred: Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(1)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(3)), + })), + }, + &rows_schema, + ), + node( + 3, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Sort { + keys: vec![SortKey { + expr: QueryExpr::Column(2), + ascending: false, + nulls_first: false, + }], + partition_by: groups.clone(), + }, + }, + &rows_schema, + ), + node( + 4, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Limit { + n: 1, + offset: 0, + partition_by: groups, + }, + }, + &rows_schema, + ), + ], + // Deliberately put Right before Left: list order must not swap inputs. + edges: vec![ + edge(1, 2, EdgeRole::Right, &keys_schema), + edge(0, 2, EdgeRole::Left, &rows_schema), + edge(2, 3, EdgeRole::Input, &rows_schema), + edge(3, 4, EdgeRole::Input, &rows_schema), + ], + root: PostAsapNodeId(4), + }; + let text = |v: &str| Value::Utf8(v.into()); + for (phase, scope) in [ + (ExecutionTiming::QueryTime, query()), + ( + ExecutionTiming::IngestionTime, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1000, + revision: 2, + }, + ), + ] { + let dag = dag + .with_execution_phases(&dag.nodes.iter().map(|node| (node.id, phase)).collect()) + .unwrap(); + let sources: BTreeMap> = BTreeMap::from([ + ( + 0, + Box::new( + Operator::source( + rows_schema.clone(), + vec![Batch::try_new( + rows_schema.clone(), + vec![ + vec![text("a"), text("x"), Value::Float64(8.)], + vec![text("a"), text("y"), Value::Float64(9.)], + vec![text("b"), text("x"), Value::Float64(2.)], + vec![text("b"), text("z"), Value::Float64(99.)], + ], + ) + .unwrap()], + ) + .unwrap(), + ) as Source<'static>, + ), + ( + 1, + Box::new( + Operator::source( + keys_schema.clone(), + vec![Batch::try_new( + keys_schema.clone(), + vec![vec![text("x")], vec![text("y")]], + ) + .unwrap()], + ) + .unwrap(), + ) as Source<'static>, + ), + ]); + let native = bind(&dag, sources, &[4]).unwrap(); + let mut scores = floats(&run(&native, 4, scope), 2); + scores.sort_by(f64::total_cmp); + assert_eq!(scores, vec![2., 9.]); + } +} + +// Planner scalar signatures, collection access and null predicates share native execution. +#[test] +fn planner_expressions_preserve_collection_and_nullable_types() { + use asap_physical_operators::dag::expressions::CompiledExpression; + use planner_types::pre_asap::{CompareOpKind, QueryExpr, ScalarValue}; + use std::rc::Rc; + let input_schema = schema(&[( + "items", + DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Int64), + value_nullable: false, + }, + false, + )]); + let access = QueryExpr::FunctionCall { + name: "asap_element_access".into(), + args: vec![ + QueryExpr::Column(0), + QueryExpr::Literal(ScalarValue::Utf8("count".into())), + ], + }; + let project = Operator::project( + input_schema.clone(), + vec![( + "count".into(), + Expression::planner(CompiledExpression::compile(&access, &input_schema).unwrap()), + )], + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + input_schema.clone(), + vec![Batch::try_new( + input_schema.clone(), + vec![ + vec![Value::Map( + vec![(Value::Utf8("count".into()), Value::Int64(7))].into(), + )], + vec![Value::Map(Arc::from([]))], + ], + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + let projected = project.schema(); + dag.add(1, vec![0], project).unwrap(); + let predicate = QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Ge, + right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + }; + dag.add( + 2, + vec![1], + Operator::filter( + projected.clone(), + Expression::planner(CompiledExpression::compile(&predicate, &projected).unwrap()), + ) + .unwrap(), + ) + .unwrap(); + let rows = run(&dag, 2, query()); + assert!(matches!(rows.as_slice(),[row] if matches!(row.as_slice(),[Value::Int64(7)]))); + let unknown = QueryExpr::FunctionCall { + name: "unregistered_function".into(), + args: vec![QueryExpr::Column(0)], + }; + assert!(CompiledExpression::compile(&unknown, &input_schema).is_err()); +} + +// Outer, semi and anti joins share Planner predicates and preserve SQL null behavior. +#[test] +fn native_relational_join_kinds_preserve_unmatched_rows() { + use planner_types::pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}; + use std::rc::Rc; + let input = schema(&[("key", DataType::Int64, true)]); + let predicate = Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(1)), + })); + for (kind, count) in [ + (JoinKind::Inner, 1), + (JoinKind::Left, 3), + (JoinKind::Right, 3), + (JoinKind::Full, 5), + (JoinKind::Semi, 1), + (JoinKind::Anti, 2), + (JoinKind::Cross, 9), + ] { + let output = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + input.clone() + } else { + schema(&[ + ("left", DataType::Int64, true), + ("right", DataType::Int64, true), + ]) + }; + let mut dag = PhysicalDag::default(); + for (id, rows) in [ + ( + 0, + vec![ + vec![Value::Int64(1)], + vec![Value::Int64(2)], + vec![Value::Null], + ], + ), + ( + 1, + vec![ + vec![Value::Int64(2)], + vec![Value::Int64(3)], + vec![Value::Null], + ], + ), + ] { + dag.add( + id, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new(input.clone(), rows).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + } + dag.add( + 2, + vec![0, 1], + Operator::relational_join( + input.clone(), + input.clone(), + kind.clone(), + &predicate, + output, + ) + .unwrap(), + ) + .unwrap(); + assert_eq!(run(&dag, 2, query()).len(), count, "{kind:?}"); + } +} + +// Per-series fractional rates feed either weighted frequency family per job, in either scope. +#[test] +fn weighted_rate_topk_preserves_partitions_fractional_scores_and_evaluation_scope() { + for count_sketch in [false, true] { + assert_weighted_rate_topk(count_sketch); + } +} +fn assert_weighted_rate_topk(count_sketch: bool) { + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let raw = schema(&[ + ("service", DataType::Utf8, false), + ("job", DataType::Utf8, false), + ("instance", DataType::Int64, false), + ("t", DataType::Timestamp, false), + ("value", DataType::Float64, false), + ]); + let mut rows = Vec::new(); + // Multiple instances of auth accumulate. Batch has a very different scale. + for (service, job, instance, rate) in [ + ("auth", "api", 1, 0.125), + ("auth", "api", 2, 0.25), + ("checkout", "api", 1, 0.3125), + ("search", "api", 1, 0.0625), + ("ingest", "batch", 1, 100.0), + ("export", "batch", 1, 80.0), + ("cleanup", "batch", 1, 20.0), + ] { + for (t, value) in [(0, 0.0), (30_000, rate * 30.0), (60_000, rate * 60.0)] { + rows.push(vec![ + Value::Utf8(service.into()), + Value::Utf8(job.into()), + Value::Int64(instance), + Value::Timestamp(t), + Value::Float64(value), + ]); + } + } + let rates = Operator::window( + raw.clone(), + planner_types::pre_asap::AggIntent::Rate, + 3, + 4, + vec![0, 1, 2], + Some((0, 60_000)), + ) + .unwrap(); + let family = SummaryFamilyType::Sketch( + SketchKind::new( + if count_sketch { + SketchAlgorithm::CountSketchWithHeap + } else { + SketchAlgorithm::CmsWithHeap + }, + if count_sketch { + SketchParams::CountSketchWithHeap { + width: 4096, + depth: 5, + heap_size: 8, + } + } else { + SketchParams::CmsWithHeap { + width: 4096, + depth: 5, + heap_size: 8, + } + }, + ), + Default::default(), + ); + let build = Operator::keyed_summary_build(rates.schema(), family, 3, vec![0], vec![1]).unwrap(); + let output = schema(&[ + ("job", DataType::Utf8, false), + ("service", DataType::Utf8, false), + ("score", DataType::Float64, false), + ]); + let readout = Operator::keyed_readout(build.schema(), 1, 8, output.clone()).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(raw.clone(), vec![Batch::try_new(raw, rows).unwrap()]).unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], rates).unwrap(); + dag.add(2, vec![1], build).unwrap(); + dag.add(3, vec![2], readout).unwrap(); + dag.add( + 4, + vec![3], + Operator::sort( + output.clone(), + vec![SortKey { + column: 2, + descending: true, + nulls_first: false, + }], + vec![0], + ) + .unwrap(), + ) + .unwrap(); + dag.add(5, vec![4], Operator::limit(output, 2, 0, vec![0]).unwrap()) + .unwrap(); + for scope in [ + query(), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 60_000, + revision: 2, + }, + query(), + ] { + let result = run(&dag, 5, scope); + assert_eq!(result.len(), 4); + assert_eq!(floats(&result, 2), vec![0.375, 0.3125, 100.0, 80.0]); + let services = result + .iter() + .map(|row| match &row[1] { + Value::Utf8(v) => v.as_ref(), + _ => panic!("service"), + }) + .collect::>(); + assert_eq!(services, vec!["auth", "checkout", "ingest", "export"]); + } +} + +// The grouped temporal reducer's sample schema must survive physical Sort/Limit binding. +#[test] +fn grouped_temporal_schema_compiles_and_executes_topk() { + use asap_physical_operators::physical_planner::{ + compile_node, CompiledPhysicalDag, InputContract, Source, + }; + use planner_types::post_asap::{ + ExecutionDataState, PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, + ValueOperation, + }; + use planner_types::pre_asap::{ + aggregate_output_schema, AggIntent, Column, GroupKeys, QueryExpr, Reduction as IrReduction, + Schema as IrSchema, + }; + let grouped = IrSchema::new(vec![ + Column::new("job", DataType::Utf8, false), + Column::new("sum", DataType::Float64, false), + ]); + let output = aggregate_output_schema( + &grouped, + &IrReduction::PerEntity, + &[AggIntent::Avg { col: None }], + &[], + ) + .unwrap(); + let input = schema( + &output + .columns + .iter() + .map(|c| (c.name.as_str(), c.dtype.clone(), c.nullable)) + .collect::>(), + ); + let node = |id, operation| PostAsapDagNode { + id: PostAsapNodeId(id), + payload: PostAsapOperatorPayload::Value { operation }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*input).clone(), + guarantee: None, + }; + let sort = compile_node( + &node( + 1, + ValueOperation::Sort { + keys: vec![planner_types::pre_asap::SortKey { + expr: QueryExpr::Column(1), + ascending: false, + nulls_first: false, + }], + partition_by: GroupKeys::none(), + }, + ), + std::slice::from_ref(&input), + ) + .unwrap(); + let limit = compile_node( + &node( + 2, + ValueOperation::Limit { + n: 1, + offset: 0, + partition_by: GroupKeys::none(), + }, + ), + std::slice::from_ref(&input), + ) + .unwrap(); + let compiled = CompiledPhysicalDag::from_operators( + [(0, InputContract::bounded(input.clone()))].into(), + [(1, (vec![0], sort)), (2, (vec![1], limit))].into(), + vec![2], + ) + .unwrap(); + let recovered = + serde_json::from_slice::(&serde_json::to_vec(&compiled).unwrap()) + .unwrap(); + assert_eq!(recovered.row_source(2), Some(0)); + assert_eq!(recovered.operator_name(2), Some("Limit")); + let expected = vec![Value::Utf8("api".into()), Value::Float64(9.)]; + let batch = Batch::try_new( + input.clone(), + vec![ + vec![Value::Utf8("worker".into()), Value::Float64(2.)], + expected.clone(), + ], + ) + .unwrap(); + let source = Box::new(Operator::source(input, vec![batch]).unwrap()) as Source<'_>; + let physical = recovered.instantiate([(0, source)].into()).unwrap(); + let mut stream = physical + .execute(&[2], RunContext::new(query(), Limits::default()).unwrap()) + .unwrap() + .remove(0); + let rows = block_on(async { + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend_from_slice(batch.unwrap().rows()); + } + rows + }); + assert_eq!(rows.len(), 1); + assert!(matches!(&rows[0][0], Value::Utf8(label) if label.as_ref() == "api")); + assert!(matches!(rows[0][1], Value::Float64(9.))); +} + +// A certified candidate set must have authoritative values for every key, including after recovery. +#[test] +fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { + use asap_physical_operators::physical_planner::{ + compile_node, CompiledPhysicalDag, InputContract, Source, + }; + use planner_types::{ + post_asap::*, + pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}, + }; + use std::{collections::BTreeMap, rc::Rc}; + let schema = schema(&[("key", DataType::Utf8, false)]); + for certified in [false, true] { + let node = PostAsapDagNode { + id: PostAsapNodeId(2), + output_schema: (*schema).clone(), + output_state: ExecutionDataState::QUERY_ROWS, + guarantee: None, + payload: PostAsapOperatorPayload::RelationalJoin { + join_kind: JoinKind::Semi, + pred: Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(1)), + })), + pruning: certified.then_some(CandidateCompleteness::Certified { + guarantee: ResultGuarantee { + metric: ErrorMetric::TopKMembership, + bound: BoundExpr::Zero, + failure_probability: ProbabilityExpr::Constant { value: 0.01 }, + provenance: vec![], + }, + }), + }, + }; + let graph = CompiledPhysicalDag::from_operators( + [ + (0, InputContract::bounded(schema.clone())), + (1, InputContract::bounded(schema.clone())), + ] + .into(), + [( + 2, + ( + vec![0, 1], + compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(), + ), + )] + .into(), + vec![2], + ) + .unwrap(); + let graph = + serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + .unwrap(); + assert_eq!( + graph.certified_pruning_keys(2), + certified.then_some(&[(0, 0)][..]) + ); + for complete in [false, true] { + let sources = [ + vec!["a"], + if complete { + vec!["a"] + } else { + vec!["a", "missing"] + }, + ] + .into_iter() + .enumerate() + .map(|(i, keys)| { + let batch = Batch::try_new( + schema.clone(), + keys.into_iter() + .map(|k| vec![Value::Utf8(k.into())]) + .collect(), + ) + .unwrap(); + ( + i as u64, + Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect::>(); + let bound = graph.instantiate(sources).unwrap(); + let result = block_on(async { + let mut stream = bound + .execute( + graph.roots(), + RunContext::new(query(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok::<_, asap_physical_operators::Error>(rows) + }); + if certified && !complete { + assert!(result + .unwrap_err() + .to_string() + .contains("no authoritative value")); + } else { + let rows = result.unwrap(); + assert_eq!(rows.len(), 1); + assert!(matches!(&rows[0][0], Value::Utf8(key) if key.as_ref() == "a")); + } + } + } +} + +// Precompute arithmetic must match population/window identities, never zip arrival order. +#[test] +fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { + use asap_physical_operators::physical_planner::{ + compile_node, CompiledPhysicalDag, InputContract, Source, + }; + use planner_types::{ + post_asap::*, + pre_asap::{ArithmeticOpKind, BinaryOpKind}, + }; + use std::collections::BTreeMap; + let input = schema(&[ + ("population", DataType::Utf8, false), + ("time", DataType::Timestamp, false), + ("value", DataType::Float64, false), + ]); + let node = PostAsapDagNode { + id: PostAsapNodeId(2), + output_schema: (*input).clone(), + output_state: ExecutionDataState::INGESTION_ROWS, + guarantee: None, + payload: PostAsapOperatorPayload::Binary { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + }, + }; + let program = CompiledPhysicalDag::from_operators( + [ + (0, InputContract::bounded(input.clone())), + (1, InputContract::bounded(input.clone())), + ] + .into(), + [( + 2, + ( + vec![0, 1], + compile_node(&node, &[input.clone(), input.clone()]).unwrap(), + ), + )] + .into(), + vec![2], + ) + .unwrap(); + let program = + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + .unwrap(); + for (right, expected) in [ + (vec![("b", 2, 3.), ("a", 1, 2.)], Some(vec![8., 17.])), + (vec![("b", 2, 3.)], None), + (vec![("a", 1, 2.), ("a", 1, 2.)], None), + (vec![("a", 2, 2.), ("b", 1, 3.)], None), + (vec![("a", 1, f64::NAN), ("b", 2, 3.)], None), + ] { + let sources = [vec![("a", 1, 10.), ("b", 2, 20.)], right] + .into_iter() + .enumerate() + .map(|(i, rows)| { + let rows = rows + .into_iter() + .map(|(group, time, value)| { + vec![ + Value::Utf8(group.into()), + Value::Timestamp(time), + Value::Float64(value), + ] + }) + .collect(); + let batch = Batch::try_new(input.clone(), rows).unwrap(); + ( + i as u64, + Box::new(Operator::source(input.clone(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect::>(); + let graph = program.instantiate(sources).unwrap(); + let result = block_on(async { + let mut stream = graph + .execute( + program.roots(), + RunContext::new(query(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok::<_, asap_physical_operators::Error>(rows) + }); + match expected { + Some(values) => assert_eq!(floats(&result.unwrap(), 2), values), + None => assert!(result.is_err()), + } + } +} diff --git a/crates/asap-physical-operators/tests/physical_plan_recovery.rs b/crates/asap-physical-operators/tests/physical_plan_recovery.rs new file mode 100644 index 00000000..5a920a8d --- /dev/null +++ b/crates/asap-physical-operators/tests/physical_plan_recovery.rs @@ -0,0 +1,105 @@ +//! Deserialized physical plans recover selected operators without logical lowering. +//! Deployments choose the encoding; JSON is used here only as a test format. +use asap_physical_operators::{ + operators::{Operator, SortKey}, + physical_planner::{CompiledPhysicalDag, InputContract}, +}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, +}; +use std::{collections::BTreeMap, sync::Arc}; + +fn sorted() -> CompiledPhysicalDag { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }); + CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), + BTreeMap::from([( + 1, + ( + vec![0], + Operator::sort( + schema, + vec![SortKey { + column: 0, + descending: true, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + ), + )]), + vec![1], + ) + .unwrap() +} + +#[test] +fn recovery_retains_selected_operator_and_rejects_invalid_contracts() { + let bytes = serde_json::to_vec(&sorted()).unwrap(); + let recovered = serde_json::from_slice::(&bytes).unwrap(); + assert_eq!(serde_json::to_vec(&recovered).unwrap(), bytes); + for mutation in ["column", "edge", "output"] { + let mut wire: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + match mutation { + "column" => { + wire["nodes"]["1"]["Operator"]["operator"]["kind"]["Sort"]["keys"][0]["column"] = + 7.into() + } + "edge" => wire["nodes"]["1"]["Operator"]["inputs"][0] = 999.into(), + "output" => { + wire["nodes"]["1"]["Operator"]["operator"]["output"]["fields"][0]["dtype"] = + serde_json::json!({"Plain":"utf8"}) + } + _ => unreachable!(), + } + assert!( + serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()) + .is_err(), + "accepted {mutation}" + ); + } +} + +#[test] +fn candidate_recovery_preserves_materialization_boundary() { + use asap_physical_operators::physical_planner::PhysicalCandidate; + let precompute = sorted(); + let output = InputContract::bounded(precompute.output_contract(1).unwrap().schema); + let query = CompiledPhysicalDag::from_operators( + BTreeMap::from([(1, output.clone())]), + BTreeMap::from([( + 2, + ( + vec![1], + Operator::limit(output.schema.clone(), 3, 0, vec![]).unwrap(), + ), + )]), + vec![2], + ) + .unwrap(); + let candidate = PhysicalCandidate { + precompute: Some(precompute), + query, + materialized_outputs: BTreeMap::from([(1, output)]), + }; + let bytes = serde_json::to_vec(&candidate).unwrap(); + let restored = serde_json::from_slice::(&bytes).unwrap(); + assert_eq!(restored.precompute.as_ref().unwrap().roots(), &[1]); + assert_eq!(restored.query.roots(), &[2]); + assert_eq!(serde_json::to_vec(&restored).unwrap(), bytes); + let mut wire: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + wire["materialized_outputs"]["1"]["schema"]["fields"][0]["dtype"] = + serde_json::json!({"Plain":"utf8"}); + assert!( + serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()).is_err() + ); +} diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs new file mode 100644 index 00000000..5770d7d6 --- /dev/null +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -0,0 +1,695 @@ +//! Contract tests inspired by DataFusion's limit, sort and join test matrices. +//! Expectations follow ASAP's IR (notably row-count and IEEE NaN equality). +//! Reference: apache/datafusion e2ca7f3, physical-plan/src/{limit.rs,sorts/sort.rs}. +use asap_physical_operators::{ + expressions::CompiledExpression, + operators::{Expression, Operator, Reduction, SortKey}, + plan::PhysicalDag, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Schema, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{CompareOpKind, DataType, JoinKind, Predicate, QueryExpr}, +}; +use std::{rc::Rc, sync::Arc}; + +fn schema(fields: &[(&str, DataType, bool)]) -> Schema { + Arc::new(SummarySchema { + fields: fields + .iter() + .map(|(name, dtype, nullable)| SummaryField { + name: (*name).into(), + dtype: SummaryFamilyType::Plain(dtype.clone()), + nullable: *nullable, + }) + .collect(), + time_index: None, + }) +} +fn context() -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits { + max_buffered_batches: 1, + ..Limits::default() + }, + ) + .unwrap() +} +fn collect(dag: &PhysicalDag<'_, Batch, Schema>, root: u64) -> Vec> { + let run = context(); + let rows = block_on(async { + let mut stream = dag.execute(&[root], run.clone()).unwrap().remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend_from_slice(batch.unwrap().rows()); + } + rows + }); + assert_eq!(run.retained_bytes(), 0); + rows +} +fn unary(input: Schema, batches: Vec>>, op: Operator) -> Vec> { + let mut dag = PhysicalDag::default(); + let batches = batches + .into_iter() + .map(|rows| Batch::try_new(input.clone(), rows).unwrap()) + .collect(); + dag.add(0, vec![], Operator::source(input, batches).unwrap()) + .unwrap(); + dag.add(1, vec![0], op).unwrap(); + collect(&dag, 1) +} +fn keys(rows: &[Vec]) -> Vec>> { + rows.iter() + .map(|r| r.iter().map(|v| v.key().unwrap()).collect()) + .collect() +} +fn eq_predicate() -> Predicate { + Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(1)), + })) +} +fn join(left: Vec, right: Vec, kind: JoinKind, keyed: bool) -> Vec> { + let input = schema(&[("key", DataType::Float64, true)]); + let output = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + input.clone() + } else { + schema(&[ + ("left", DataType::Float64, true), + ("right", DataType::Float64, true), + ]) + }; + let op = if keyed { + Operator::semi_join(input.clone(), input.clone(), vec![(0, 0)]).unwrap() + } else { + Operator::relational_join(input.clone(), input.clone(), kind, &eq_predicate(), output) + .unwrap() + }; + let mut dag = PhysicalDag::default(); + for (id, values) in [(0, left), (1, right)] { + let batches = values + .into_iter() + .map(|v| Batch::try_new(input.clone(), vec![vec![v]]).unwrap()) + .collect(); + dag.add( + id, + vec![], + Operator::source(input.clone(), batches).unwrap(), + ) + .unwrap(); + } + dag.add(2, vec![0, 1], op).unwrap(); + collect(&dag, 2) +} + +// OFFSET/FETCH must be invariant to empty batches and input batch boundaries. +#[test] +fn limit_offset_fetch_matrix() { + let input = schema(&[("v", DataType::Int64, false)]); + for chunk in [1, 2, 5, 12] { + let values = (0..9).map(|n| vec![Value::Int64(n)]).collect::>(); + let mut batches = vec![vec![]]; + for rows in values.chunks(chunk) { + batches.push(rows.to_vec()); + batches.push(vec![]); + } + for offset in [0, 1, 8, 9, 10, u64::MAX] { + for n in [0, 1, 3, 12, u64::MAX] { + let rows = unary( + input.clone(), + batches.clone(), + Operator::limit(input.clone(), n, offset, vec![]).unwrap(), + ); + let expected = values + .iter() + .skip(offset.min(9) as usize) + .take(n.min(9) as usize) + .cloned() + .collect::>(); + assert_eq!( + keys(&rows), + keys(&expected), + "chunk={chunk}, offset={offset}, n={n}" + ); + } + } + } +} + +// Zero-column batches still have rows: LIMIT must not infer cardinality from columns. +#[test] +fn limit_preserves_zero_column_row_count() { + let input = schema(&[]); + let rows = unary( + input.clone(), + vec![vec![vec![]; 5], vec![vec![]; 5]], + Operator::limit(input, 4, 3, vec![]).unwrap(), + ); + assert_eq!(rows.len(), 4); +} + +// NULL placement is independent of sort direction; ties retain original row order. +#[test] +fn sort_direction_null_placement_and_ties() { + let input = schema(&[("v", DataType::Int64, true), ("id", DataType::Int64, false)]); + let values = [Some(2), None, Some(1), Some(2), None]; + let rows = values + .iter() + .enumerate() + .map(|(i, v)| { + vec![ + v.map(Value::Int64).unwrap_or(Value::Null), + Value::Int64(i as i64), + ] + }) + .collect::>(); + for (descending, nulls_first, expected) in [ + (false, false, vec![2, 0, 3, 1, 4]), + (false, true, vec![1, 4, 2, 0, 3]), + (true, false, vec![0, 3, 2, 1, 4]), + (true, true, vec![1, 4, 0, 3, 2]), + ] { + let op = Operator::sort( + input.clone(), + vec![SortKey { + column: 0, + descending, + nulls_first, + }], + vec![], + ) + .unwrap(); + let result = unary( + input.clone(), + vec![rows[..2].to_vec(), vec![], rows[2..].to_vec()], + op, + ); + let ids = result + .iter() + .map(|r| match r[1] { + Value::Int64(n) => n, + _ => unreachable!(), + }) + .collect::>(); + assert_eq!(ids, expected); + } +} + +// Outer joins preserve unmatched NULLs, while semi/anti joins preserve left multiplicity. +#[test] +fn joins_nulls_duplicates_and_empty_sides() { + for (kind, expected_len) in [ + (JoinKind::Inner, 4), + (JoinKind::Left, 6), + (JoinKind::Right, 6), + (JoinKind::Full, 8), + (JoinKind::Semi, 2), + (JoinKind::Anti, 2), + ] { + let left = vec![ + Value::Float64(1.), + Value::Float64(1.), + Value::Float64(2.), + Value::Null, + ]; + let right = vec![ + Value::Float64(1.), + Value::Float64(1.), + Value::Float64(3.), + Value::Null, + ]; + let result = join(left, right, kind.clone(), false); + assert_eq!(result.len(), expected_len, "{kind:?}"); + } + for (kind, expected_len) in [ + (JoinKind::Inner, 0), + (JoinKind::Left, 1), + (JoinKind::Right, 0), + (JoinKind::Full, 1), + (JoinKind::Semi, 0), + (JoinKind::Anti, 1), + ] { + assert_eq!( + join(vec![Value::Float64(7.)], vec![], kind.clone(), false).len(), + expected_len, + "{kind:?}" + ); + } + let result = join(vec![Value::Float64(7.)], vec![], JoinKind::Left, false); + assert!(matches!( + result[0].as_slice(), + [Value::Float64(7.), Value::Null] + )); +} + +// Changing the semi-join algorithm must not turn IEEE NaN != NaN into a match. +#[test] +fn keyed_semijoin_obeys_ieee_equality_for_nan_and_zero() { + let left = vec![ + Value::Float64(f64::NAN), + Value::Float64(-0.), + Value::Float64(0.), + Value::Null, + ]; + let right = vec![Value::Float64(f64::NAN), Value::Float64(0.), Value::Null]; + let keyed = join(left, right, JoinKind::Semi, true); + let expected = vec![vec![Value::Float64(-0.)], vec![Value::Float64(0.)]]; + assert_eq!(keys(&keyed), keys(&expected)); +} + +// Group equality intentionally differs from predicate equality: NULL and NaNs group together. +#[test] +fn grouping_canonicalizes_null_nan_and_signed_zero() { + let input = schema(&[("v", DataType::Float64, true)]); + let op = Operator::aggregate( + input.clone(), + vec![0], + vec![("count".into(), Reduction::Count)], + ) + .unwrap(); + let values = vec![ + Value::Null, + Value::Null, + Value::Float64(0.), + Value::Float64(-0.), + Value::Float64(f64::NAN), + Value::Float64(f64::from_bits(0x7ff8000000000001)), + ]; + let result = unary( + input, + values.into_iter().map(|v| vec![vec![v]]).collect(), + op, + ); + assert_eq!(result.len(), 3); + assert!(result.iter().all(|r| matches!(r[1], Value::Int64(2)))); +} + +// Global empty input yields one aggregate row; grouped empty input yields none. +#[test] +fn aggregate_empty_and_all_null_follow_asap_contract() { + let input = schema(&[("v", DataType::Int64, true)]); + for batches in [ + vec![], + vec![vec![]], + vec![vec![vec![Value::Null], vec![Value::Null]]], + ] { + let n = batches.iter().map(Vec::len).sum::(); + let op = Operator::aggregate( + input.clone(), + vec![], + vec![ + ("count".into(), Reduction::Count), + ("min".into(), Reduction::Min(0)), + ("max".into(), Reduction::Max(0)), + ], + ) + .unwrap(); + let result = unary(input.clone(), batches, op); + assert_eq!(result.len(), 1); + assert!(matches!(result[0][0], Value::Int64(v) if v == n as i64)); + assert!(matches!(result[0][1], Value::Null)); + assert!(matches!(result[0][2], Value::Null)); + } + let op = Operator::aggregate( + input.clone(), + vec![0], + vec![("count".into(), Reduction::Count)], + ) + .unwrap(); + assert!(unary(input, vec![], op).is_empty()); +} + +// A precompiled expression with a different input contract must fail during binding. +#[test] +fn projection_rejects_expression_bound_to_another_schema() { + let original = schema(&[("a", DataType::Int64, false), ("b", DataType::Int64, false)]); + let current = schema(&[("a", DataType::Int64, false)]); + let expr = CompiledExpression::compile(&QueryExpr::Column(1), &original).unwrap(); + assert!(Operator::project(current, vec![("b".into(), Expression::planner(expr))]).is_err()); +} + +// A valid Planner MIN/MAX schema must bind even for a non-null input column. +#[test] +fn global_extrema_bind_with_planner_derived_schema() { + use asap_physical_operators::physical_planner::compile_node; + use planner_types::{ + post_asap::*, + pre_asap::{AggIntent, Column, GroupKeys, Reduction as PlanReduction}, + }; + let input = schema(&[("v", DataType::Int64, false)]); + for measure in [ + AggIntent::Min { col: Some(0) }, + AggIntent::Max { col: Some(0) }, + ] { + let planner_input = + planner_types::pre_asap::Schema::new(vec![Column::new("v", DataType::Int64, false)]); + let derived = planner_types::pre_asap::query_expr::aggregate_output_schema( + &planner_input, + &PlanReduction::Reduce(GroupKeys::by(vec![])), + std::slice::from_ref(&measure), + &[], + ) + .unwrap(); + let result = derived.columns[0].clone(); + let output = schema(&[(&result.name, result.dtype, result.nullable)]); + let node = PostAsapDagNode { + id: PostAsapNodeId(1), + payload: PostAsapOperatorPayload::Value { + operation: ValueOperation::Exact(ExactOperation::Aggregate { + reduction: PlanReduction::Reduce(GroupKeys::by(vec![])), + measures: vec![measure], + output_names: vec![result.name], + having: None, + }), + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*output).clone(), + guarantee: None, + }; + let operator = compile_node(&node, std::slice::from_ref(&input)) + .expect("global extremum should bind to its Planner schema"); + assert!(operator.schema().fields[0].nullable); + let empty = unary(input.clone(), vec![], operator.clone()); + assert!(matches!(empty[0][0], Value::Null)); + let nonempty = unary(input.clone(), vec![vec![vec![Value::Int64(7)]]], operator); + assert!(matches!(nonempty[0][0], Value::Int64(7))); + } +} + +// NaN is a valid numeric input, not a schema error; all six comparisons obey IEEE rules. +#[test] +fn planner_comparisons_handle_nan_without_execution_errors() { + let input = schema(&[ + ("a", DataType::Float64, false), + ("b", DataType::Float64, false), + ]); + for op in [ + CompareOpKind::Eq, + CompareOpKind::Ne, + CompareOpKind::Lt, + CompareOpKind::Le, + CompareOpKind::Gt, + CompareOpKind::Ge, + ] { + let expression = QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: op.clone(), + right: Rc::new(QueryExpr::Column(1)), + }; + let compiled = CompiledExpression::compile(&expression, &input).unwrap(); + for row in [ + [Value::Float64(f64::NAN), Value::Float64(1.)], + [Value::Float64(1.), Value::Float64(f64::NAN)], + [Value::Float64(f64::NAN), Value::Float64(f64::NAN)], + ] { + let actual = compiled.evaluate(&row).unwrap(); + assert!(matches!(actual,Value::Bool(value) if value == (op == CompareOpKind::Ne))); + } + } +} + +// A bounded LIMIT branch must unsubscribe so another branch can drain the producer. +#[test] +fn limit_branch_finishes_without_blocking_shared_sibling() { + let input = schema(&[("v", DataType::Int64, false)]); + let mut dag = PhysicalDag::default(); + let batches = (0..100) + .map(|v| Batch::try_new(input.clone(), vec![vec![Value::Int64(v)]]).unwrap()) + .collect(); + dag.add(0, vec![], Operator::source(input.clone(), batches).unwrap()) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::limit(input.clone(), 1, 0, vec![]).unwrap(), + ) + .unwrap(); + dag.add(2, vec![0, 1], Operator::union(input, 2).unwrap()) + .unwrap(); + // Bound polls as well as rows so a backpressure regression cannot hang the suite. + use futures::{task::noop_waker_ref, Stream}; + use std::{ + pin::Pin, + task::{Context, Poll}, + }; + let run = context(); + let mut stream = dag.execute(&[2], run.clone()).unwrap().remove(0); + let mut cx = Context::from_waker(noop_waker_ref()); + let mut count = 0; + for _ in 0..2000 { + match Pin::new(&mut stream).poll_next(&mut cx) { + Poll::Ready(Some(batch)) => count += batch.unwrap().rows().len(), + Poll::Ready(None) => { + assert_eq!(count, 101); + drop(stream); + assert_eq!(run.retained_bytes(), 0); + return; + } + Poll::Pending => {} + } + } + panic!("shared LIMIT/Union failed to make progress"); +} + +// Mixed numeric comparisons must not round Int64 values through f64 before comparing. +#[test] +fn mixed_numeric_comparisons_preserve_large_integer_precision() { + let input = schema(&[ + ("a", DataType::Int64, false), + ("b", DataType::Float64, false), + ]); + let expr = QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Gt, + right: Rc::new(QueryExpr::Column(1)), + }; + let compiled = CompiledExpression::compile(&expr, &input).unwrap(); + for (a, b, expected) in [ + (9_007_199_254_740_993, 9_007_199_254_740_992.0, true), + (i64::MAX, 9_223_372_036_854_775_808.0, false), + (i64::MIN, f64::NEG_INFINITY, true), + ] { + assert!( + matches!(compiled.evaluate(&[Value::Int64(a),Value::Float64(b)]).unwrap(), Value::Bool(v) if v == expected) + ); + } +} + +// Both expression paths must implement all nine combinations of three-valued booleans. +#[test] +fn boolean_truth_tables_agree_between_expression_paths() { + let input = schema(&[("a", DataType::Bool, true), ("b", DataType::Bool, true)]); + for and in [true, false] { + for a in [None, Some(false), Some(true)] { + for b in [None, Some(false), Some(true)] { + let parts = vec![QueryExpr::Column(0), QueryExpr::Column(1)]; + let planner = if and { + QueryExpr::BoolAnd(parts) + } else { + QueryExpr::BoolOr(parts) + }; + let native = if and { + Expression::And( + Box::new(Expression::Column(0)), + Box::new(Expression::Column(1)), + ) + } else { + Expression::Or( + Box::new(Expression::Column(0)), + Box::new(Expression::Column(1)), + ) + }; + let expected = match (a, b, and) { + (Some(false), _, true) | (_, Some(false), true) => Some(false), + (Some(true), _, false) | (_, Some(true), false) => Some(true), + (None, _, _) | (_, None, _) => None, + (Some(a), Some(b), true) => Some(a && b), + (Some(a), Some(b), false) => Some(a || b), + } + .map(Value::Bool) + .unwrap_or(Value::Null); + let row = vec![ + a.map(Value::Bool).unwrap_or(Value::Null), + b.map(Value::Bool).unwrap_or(Value::Null), + ]; + let compiled = CompiledExpression::compile(&planner, &input).unwrap(); + assert_eq!( + compiled.evaluate(&row).unwrap().key().unwrap(), + expected.key().unwrap() + ); + let op = Operator::project(input.clone(), vec![("result".into(), native)]).unwrap(); + let result = unary(input.clone(), vec![vec![row]], op); + assert_eq!(result[0][0].key().unwrap(), expected.key().unwrap()); + } + } + } +} + +// Partial/final execution must agree with one build for an uncompacted KLL population. +#[test] +fn kll_partial_merge_and_multiple_readouts_preserve_population() { + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("v", DataType::Float64, false)]); + let family = SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), + Default::default(), + ); + let mut dag = PhysicalDag::default(); + for (id, range) in [(0, 0..64), (1, 64..128), (2, 0..128)] { + let rows = range.map(|n| vec![Value::Float64(n as f64)]).collect(); + dag.add( + id, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new(input.clone(), rows).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add( + id + 3, + vec![id], + Operator::summary_build(input.clone(), family.clone(), 0, None, vec![]).unwrap(), + ) + .unwrap(); + } + let state = Operator::summary_build(input, family, 0, None, vec![]) + .unwrap() + .schema(); + dag.add(6, vec![3, 4], Operator::union(state.clone(), 2).unwrap()) + .unwrap(); + dag.add( + 7, + vec![6], + Operator::summary_merge(state.clone(), 0, vec![]).unwrap(), + ) + .unwrap(); + let mut roots = vec![]; + for (i, q) in [0.0, 0.5, 1.0].into_iter().enumerate() { + for (j, build) in [5, 7].into_iter().enumerate() { + let id = 8 + (i * 2 + j) as u64; + dag.add( + id, + vec![build], + Operator::readout( + state.clone(), + 0, + asap_physical_operators::operators::ReadoutQuery::Sketch( + planner_types::post_asap::SketchQuery::Quantile { q }, + ), + ) + .unwrap(), + ) + .unwrap(); + roots.push(id); + } + } + for _ in 0..2 { + let run = context(); + let outputs = block_on(futures::future::join_all( + dag.execute(&roots, run.clone()) + .unwrap() + .into_iter() + .map(|s| s.collect::>()), + )); + for (pair, expected) in outputs.chunks(2).zip([0., 64., 127.]) { + let value = |batches: &[Result< + asap_physical_operators::runtime::SharedValue, + asap_physical_operators::Error, + >]| { + assert_eq!(batches.len(), 1); + match batches[0].as_ref().unwrap().rows()[0][0] { + Value::Float64(v) => v, + _ => panic!("quantile must be Float64"), + } + }; + assert_eq!(value(&pair[0]), value(&pair[1])); + assert!((value(&pair[0]) - expected).abs() <= 1.); + } + drop(outputs); + assert_eq!(run.retained_bytes(), 0); + } +} + +// Retained zero-column rows still own Vec headers and must consume the output budget. +#[test] +fn zero_column_output_obeys_memory_limit() { + use asap_physical_operators::Error; + let input = schema(&[]); + let batch = Batch::try_new(input.clone(), vec![vec![]; 200]).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(input, vec![batch]).unwrap()) + .unwrap(); + let run = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_bytes: 1024, + ..Limits::default() + }, + ) + .unwrap(); + let mut stream = dag.execute(&[0], run.clone()).unwrap().remove(0); + assert!(matches!( + block_on(stream.next()), + Some(Err(Error::MemoryLimit)) + )); + drop(stream); + assert_eq!(run.retained_bytes(), 0); +} + +// Empty exact-state finalization must preserve ordinary global MIN/MAX null semantics. +#[test] +fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { + use asap_physical_operators::Statistic; + use planner_types::post_asap::{ExactKind, ExactParams}; + let input = schema(&[("v", DataType::Float64, false)]); + for (kind, params, statistic) in [ + (ExactKind::Min, ExactParams::Min, Statistic::Min), + (ExactKind::Max, ExactParams::Max, Statistic::Max), + ] { + let build = Operator::summary_build( + input.clone(), + SummaryFamilyType::ExactAggregate(kind, params), + 0, + None, + vec![], + ) + .unwrap(); + let state = build.schema(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(input.clone(), vec![]).unwrap()) + .unwrap(); + dag.add(1, vec![0], build).unwrap(); + dag.add( + 2, + vec![1], + Operator::readout( + state, + 0, + asap_physical_operators::operators::ReadoutQuery::Exact( + asap_physical_operators::summary_kernels::exact::ExactReadout { + statistic, + lookback_ms: None, + }, + ), + ) + .unwrap(), + ) + .unwrap(); + let rows = collect(&dag, 2); + assert_eq!(rows.len(), 1); + assert!(matches!(rows[0][0], Value::Null)); + } +} diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs new file mode 100644 index 00000000..c00b388f --- /dev/null +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -0,0 +1,548 @@ +//! Materialized frontiers are compiled by Planner, never rewritten by deployment. +use asap_aware_mapping::{cost_model::DefaultCostModel, search_workload}; +use asap_physical_operators::{ + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{ + compile_candidates, select_candidate, CandidateCost, CompiledPhysicalDag, InputContract, + Source, + }, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{post_asap::*, pre_asap::DataType, types::AccuracyTarget, workload::*}; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; + +fn grouped_rate_space() -> asap_aware_mapping::PlanSpace<&'static str> { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("sum by(job)(rate(m[1m]))".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let root = Rc::new( + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0), + ); + let root = Rc::new( + asap_physical_operators::physical_planner::promql_rows::with_series_identity(&root) + .unwrap(), + ); + search_workload(vec![("grouped-rate", root)]) +} + +fn grouped_rate() -> PostAsapDag { + let space = grouped_rate_space(); + let selected = space + .global_selection(&DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + compile_post_asap_dag(&selected).unwrap() +} +fn run(plan: &CompiledPhysicalDag, inputs: BTreeMap, scope: Scope) -> Vec { + let sources = inputs + .into_iter() + .map(|(id, batch)| { + let source = Operator::source(batch.schema().clone(), vec![batch]).unwrap(); + (id, Box::new(source) as Source<'_>) + }) + .collect(); + let dag = plan.instantiate(sources).unwrap(); + block_on(async { + let context = RunContext::new(scope, Limits::default()).unwrap(); + let mut output = dag.execute(plan.roots(), context).unwrap().remove(0); + let mut batches = vec![]; + while let Some(batch) = output.next().await { + batches.push((*batch.unwrap()).clone()); + } + batches + }) +} + +/// Rate readouts and grouped Sum can run together during bounded precompute; +/// storing per-series rates instead leaves the same Sum in the query DAG. +#[test] +fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { + let dag = grouped_rate(); + let state = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) + .unwrap(); + let readout = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator + } + ) + }) + .unwrap(); + let input_schema = Arc::new(state.output_schema.clone()); + let (family, update, grouping) = match &state.payload { + PostAsapOperatorPayload::SummaryAgg { + family, + input, + grouping, + .. + } => (family, input, grouping), + _ => unreachable!(), + }; + let range_ms = Some((-58_000, 2_000)); + let mut expected_rate_sum = 0.; + let rows = [[100., 0., 100.], [100., 200., 0.]] + .into_iter() + .enumerate() + .map(|(index, values)| { + let mut accumulator = create_planner_accumulator(family, update, grouping).unwrap(); + for (i, value) in values.into_iter().enumerate() { + accumulator.update_single(value, i as i64 * 1000); + } + let state = accumulator.into_accumulator(); + expected_rate_sum += state + .as_any() + .downcast_ref::() + .unwrap() + .readout(asap_physical_operators::Statistic::Rate, range_ms, None) + .unwrap() + .unwrap(); + let summary = Value::Summary { + family: family.clone(), + state: Arc::from(state), + }; + input_schema + .fields + .iter() + .map(|field| match &field.dtype { + SummaryFamilyType::ExactAggregate(..) => summary.clone(), + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(2000), + SummaryFamilyType::Plain(DataType::Utf8) => { + Value::Utf8(if field.name == "job" { + "api".into() + } else { + format!("series-{index}").into() + }) + } + _ => panic!("unexpected input field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(input_schema.clone(), rows).unwrap(); + let root = u64::from(dag.root.0); + let state_id = u64::from(state.id.0); + let rate_id = u64::from(readout.id.0); + let frontiers = asap_physical_operators::physical_planner::enumerate_frontiers( + &dag, + &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), + &[root], + 128, + ) + .unwrap(); + assert!(frontiers.contains(&vec![])); + assert!(frontiers.contains(&vec![rate_id])); + assert!(frontiers.contains(&vec![root])); + assert!(!frontiers.contains(&vec![root, rate_id])); + assert!( + asap_physical_operators::physical_planner::enumerate_frontiers( + &dag, + &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), + &[root], + 1, + ) + .is_err() + ); + let candidates = compile_candidates( + &dag, + BTreeMap::from([(state_id, InputContract::bounded(input_schema))]), + &[root], + &[vec![], vec![rate_id], vec![root]], + ); + // Scoped cost fixtures select either precompute boundary. No readers are + // opened during candidate construction or selection. + for prefer_grouped in [false, true] { + let inventory = compile_candidates( + &dag, + BTreeMap::from([( + state_id, + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]), + &[root], + &[vec![999], vec![rate_id], vec![root]], + ); + assert!(inventory[0].is_err()); + let mut evaluated = 0; + let selected = select_candidate(inventory, |candidate| { + evaluated += 1; + let grouped = candidate.materialized_outputs.contains_key(&root); + Ok(Some(CandidateCost { + workload_scope: "reset-counter-workload".into(), + horizon_seconds: 300., + total_cost: if grouped == prefer_grouped { 1. } else { 100. }, + })) + }) + .unwrap(); + assert_eq!( + selected.candidate.materialized_outputs.contains_key(&root), + prefer_grouped + ); + assert_eq!(selected.cost.total_cost, 1.); + assert_eq!(evaluated, 2, "uncompilable candidates must never be priced"); + let candidate = selected.candidate; + let precompute = candidate.precompute.as_ref().unwrap(); + let stored = run( + precompute, + BTreeMap::from([(state_id, batch.clone())]), + Scope::Ingestion { + window_start_ms: -58_000, + window_end_ms: 2000, + revision: 1, + }, + ); + let output = run( + &candidate.query, + BTreeMap::from([(precompute.roots()[0], stored[0].clone())]), + Scope::Query { + evaluation_time_ms: 2000, + revision: 1, + }, + ); + assert!( + matches!(output[0].rows()[0][1], Value::Float64(value) if value == expected_rate_sum) + ); + } + let contracts = BTreeMap::from([( + state_id, + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + for frontier in [vec![rate_id, rate_id], vec![root, rate_id], vec![999]] { + assert!( + asap_physical_operators::physical_planner::compile_candidate( + &dag, + contracts.clone(), + &[root], + &frontier + ) + .is_err() + ); + } + let inventory = compile_candidates( + &dag, + contracts.clone(), + &[root], + &[vec![rate_id], vec![root]], + ); + let selected = select_candidate(inventory, |candidate| { + if candidate.materialized_outputs.contains_key(&root) { + return Ok(None); + } + Ok(Some(CandidateCost { + workload_scope: "same-workload".into(), + horizon_seconds: 300., + total_cost: 100., + })) + }) + .unwrap(); + assert!(selected + .candidate + .materialized_outputs + .contains_key(&rate_id)); + let inventory = compile_candidates(&dag, contracts, &[root], &[vec![rate_id], vec![root]]); + assert!( + select_candidate(inventory, |candidate| Ok(Some(CandidateCost { + workload_scope: "same-workload".into(), + horizon_seconds: if candidate.materialized_outputs.contains_key(&root) { + 60. + } else { + 300. + }, + total_cost: 1., + }))) + .is_err() + ); + let query_scope = Scope::Query { + evaluation_time_ms: 2000, + revision: 1, + }; + let maintenance_scope = Scope::Ingestion { + window_start_ms: -58_000, + window_end_ms: 2000, + revision: 1, + }; + let mut results = vec![]; + for candidate in candidates { + let candidate = candidate.unwrap(); + let inputs = if let Some(precompute) = &candidate.precompute { + let source = Operator::source(batch.schema().clone(), vec![batch.clone()]).unwrap(); + let invalid = precompute + .instantiate(BTreeMap::from([(state_id, Box::new(source) as Source<'_>)])) + .unwrap(); + let context = RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + assert!(invalid.execute(precompute.roots(), context).is_err()); + let stored = run( + precompute, + BTreeMap::from([(state_id, batch.clone())]), + maintenance_scope.clone(), + ); + assert_eq!(stored.len(), 1); + let boundary = precompute.roots()[0]; + assert_eq!( + candidate.materialized_outputs[&boundary].schema, + *stored[0].schema() + ); + BTreeMap::from([(boundary, stored[0].clone())]) + } else { + BTreeMap::from([(state_id, batch.clone())]) + }; + let output = run(&candidate.query, inputs, query_scope.clone()); + assert_eq!(output.len(), 1); + assert_eq!(output[0].rows().len(), 1); + assert!(matches!(&output[0].rows()[0][0], Value::Utf8(job) if job.as_ref() == "api")); + assert!( + matches!(output[0].rows()[0][1], Value::Float64(value) if value == expected_rate_sum) + ); + results.push( + output[0].rows()[0] + .iter() + .map(|value| value.key().unwrap()) + .collect::>(), + ); + } + assert_eq!(results[0], results[1]); + assert_eq!(results[1], results[2]); + let mut wrong_order = create_planner_accumulator(family, update, grouping).unwrap(); + for (i, value) in [200., 200., 100.].into_iter().enumerate() { + wrong_order.update_single(value, i as i64 * 1000); + } + let rate_of_sum = wrong_order + .into_accumulator() + .as_any() + .downcast_ref::() + .unwrap() + .readout(asap_physical_operators::Statistic::Rate, range_ms, None) + .unwrap() + .unwrap(); + assert_ne!( + expected_rate_sum, rate_of_sum, + "counter resets prohibit moving Sum before Rate" + ); +} + +/// Enumerated frontiers include both grouped-result and per-series readout +/// persistence; an explicit Rate-state input retains its original semantics. +#[test] +fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { + use asap_physical_operators::physical_planner::enumerate_frontiers; + let dag = grouped_rate(); + let state = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) + .unwrap(); + let inputs = BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + let roots = [u64::from(dag.root.0)]; + let frontiers = enumerate_frontiers(&dag, &inputs, &roots, 4096).unwrap(); + let candidates = compile_candidates(&dag, inputs.clone(), &roots, &frontiers) + .into_iter() + .collect::, _>>() + .unwrap(); + assert!(candidates.iter().any(|c| c.precompute.is_none())); + assert!(candidates + .iter() + .any(|c| c.materialized_outputs.contains_key(&roots[0]))); + assert!(candidates + .iter() + .any(|c| !c.materialized_outputs.is_empty() + && !c.materialized_outputs.contains_key(&roots[0]))); + assert!(enumerate_frontiers(&dag, &inputs, &roots, 1).is_err()); +} + +#[test] +fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { + let inventory = grouped_rate_space().enumerate_candidate_dags(4096).unwrap(); + let mut executed = 0; + for forest in inventory.candidates { + let root = &forest[0].1; + let dag = compile_post_asap_dag(root).unwrap(); + let Some(state) = dag.nodes.iter().find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) else { + continue; + }; + let boundary = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + .. + } + ) + }) + .map(|node| u64::from(node.id.0)) + .unwrap_or(u64::from(dag.root.0)); + let physical_candidates = compile_candidates( + &dag, + BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &[vec![], vec![boundary]], + ); + let (family, input, grouping) = match &state.payload { + PostAsapOperatorPayload::SummaryAgg { + family, + input, + grouping, + .. + } => (family, input, grouping), + _ => unreachable!(), + }; + let schema = Arc::new(state.output_schema.clone()); + let rows = ["a", "b"] + .into_iter() + .map(|instance| { + let mut accumulator = create_planner_accumulator(family, input, grouping).unwrap(); + for (timestamp, value) in [(1_000, 1.), (31_000, 31.), (59_000, 59.)] { + accumulator.update_single(value, timestamp); + } + let summary = Value::Summary { + family: family.clone(), + state: Arc::from(accumulator.into_accumulator()), + }; + schema + .fields + .iter() + .map(|field| match &field.dtype { + SummaryFamilyType::ExactAggregate(..) => summary.clone(), + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), + SummaryFamilyType::Plain(DataType::Utf8) + if field.name == "$promql_series_identity" => + { + Value::Utf8( + serde_json::to_string(&BTreeMap::from([ + ("job", "api"), + ("instance", instance), + ])) + .unwrap() + .into(), + ) + } + SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8("api".into()), + _ => panic!("unexpected input field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(schema, rows).unwrap(); + for physical in physical_candidates { + let physical = physical.unwrap(); + let inputs = if let Some(precompute) = &physical.precompute { + let source_id = precompute.input_contracts().next().unwrap().0; + let stored = run( + precompute, + BTreeMap::from([(source_id, batch.clone())]), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 60_000, + revision: 1, + }, + ); + assert_eq!(stored.len(), 1); + BTreeMap::from([(precompute.roots()[0], stored[0].clone())]) + } else { + BTreeMap::from([( + physical.query.input_contracts().next().unwrap().0, + batch.clone(), + )]) + }; + let output = run( + &physical.query, + inputs, + Scope::Query { + evaluation_time_ms: 60_000, + revision: 1, + }, + ); + assert_eq!(output.len(), 1); + assert_eq!(output[0].rows().len(), 1); + assert!(output[0] + .schema() + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + assert!( + output[0].rows()[0] + .iter() + .any(|value| matches!(value, Value::Float64(x) if (*x - 2.).abs() < 1e-12)), + "{:?}", + output[0].rows() + ); + executed += 1; + } + } + assert!( + executed >= 2, + "must execute both stored and query-time grouped Rate candidates: {executed}" + ); +} diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs new file mode 100644 index 00000000..3bb3af6c --- /dev/null +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -0,0 +1,425 @@ +//! Persisted precompute graphs preserve group/window identity and execute state-to-state computation. +use asap_physical_operators::{ + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{precompute, CompiledPhysicalDag, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, + Statistic, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{ArithmeticOpKind, BinaryOpKind, ColumnRef, DataType, GroupKeys, Reduction}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +#[test] +fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let schema = |dtype| SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype, + nullable: false, + }], + time_index: None, + }; + let state_schema = schema(family.clone()); + let mut value_schema = schema(SummaryFamilyType::Plain(DataType::Float64)); + value_schema.fields.push(SummaryField { + name: "time".into(), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), + nullable: false, + }); + value_schema.time_index = Some(1); + for (weight, expected) in [ + (SummaryInputExpr::Column(ColumnRef::SampleValue), 60.), + (SummaryInputExpr::Constant(1.), 4.), + ] { + let nodes = vec![ + PostAsapDagNode { + id: PostAsapNodeId(0), + payload: PostAsapOperatorPayload::SummaryMerge, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: state_schema.clone(), + guarantee: None, + }, + PostAsapDagNode { + id: PostAsapNodeId(1), + payload: PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + }, + output_state: ExecutionDataState::INGESTION_ROWS, + output_schema: value_schema.clone(), + guarantee: None, + }, + PostAsapDagNode { + id: PostAsapNodeId(2), + payload: PostAsapOperatorPayload::Binary { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + }, + output_state: ExecutionDataState::INGESTION_ROWS, + output_schema: value_schema.clone(), + guarantee: None, + }, + PostAsapDagNode { + id: PostAsapNodeId(3), + payload: PostAsapOperatorPayload::SummaryAgg { + family: family.clone(), + input: SummaryUpdate { + weight, + ..SummaryUpdate::column(ColumnRef::SampleValue) + }, + reduction: Reduction::Reduce(GroupKeys::by(vec![])), + grouping: GroupingStrategy::PerSubpopulationInstance, + }, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: state_schema.clone(), + guarantee: None, + }, + ]; + let edges = [ + (0, 1, EdgeRole::Input), + (1, 2, EdgeRole::Left), + (1, 2, EdgeRole::Right), + (2, 3, EdgeRole::Input), + ] + .into_iter() + .map(|(producer, consumer, role)| PostAsapDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role, + intermediate_schema: nodes[producer as usize].output_schema.clone(), + data_state: nodes[producer as usize].output_state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }) + .collect(); + let dag = PostAsapDag { + nodes, + edges, + root: PostAsapNodeId(3), + }; + let mut invalid_grouping = dag.clone(); + let PostAsapOperatorPayload::SummaryAgg { reduction, .. } = + &mut invalid_grouping.nodes[3].payload + else { + unreachable!() + }; + *reduction = Reduction::Reduce(GroupKeys::by(vec![0])); + assert!( + precompute::compile(&invalid_grouping, &[0], &[3]).is_err(), + "numeric values cannot be reinterpreted as population labels" + ); + let program = precompute::compile(&dag, &[0], &[3]).unwrap(); + let program = + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + .unwrap(); + assert_eq!(program.input_contracts().count(), 1); + for revision in [1, 2] { + let rows = [ + ("a", 1000, 2.), + ("a", 2000, 4.), + ("b", 1000, 8.), + ("b", 2000, 16.), + ] + .into_iter() + .map(|(group, time, value)| { + let mut state = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + state.update_single(value, time); + vec![ + Value::Map( + vec![(Value::Utf8("instance".into()), Value::Utf8(group.into()))].into(), + ), + Value::Timestamp(time), + Value::Summary { + family: family.clone(), + state: Arc::from(state.into_accumulator()), + }, + ] + }) + .collect(); + let input = + Batch::try_new(precompute::population_schema(family.clone()), rows).unwrap(); + let sources = BTreeMap::from([( + 0, + Box::new(Operator::source(input.schema().clone(), vec![input]).unwrap()) + as Source<'_>, + )]); + let graph = program.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision, + }, + Limits::default(), + ) + .unwrap(); + let output = block_on(async { + let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let batch = stream.next().await.unwrap().unwrap(); + assert!(stream.next().await.is_none()); + batch + }); + assert_eq!(output.rows().len(), 1); + assert!(matches!(&output.rows()[0][0], Value::Map(labels) if labels.is_empty())); + assert!(matches!(&output.rows()[0][1], Value::Timestamp(2000))); + let Value::Summary { state, .. } = &output.rows()[0][2] else { + panic!("state expected") + }; + assert_eq!( + state + .as_any() + .downcast_ref::() + .unwrap() + .readout(Statistic::Sum, None, None) + .unwrap() + .unwrap(), + expected + ); + } + } +} + +fn logical_schema(family: SummaryFamilyType) -> SummarySchema { + SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: family, + nullable: false, + }], + time_index: None, + } +} +fn state_graph( + family: SummaryFamilyType, + target: Option, + merge: bool, +) -> CompiledPhysicalDag { + let mut nodes = vec![PostAsapDagNode { + id: PostAsapNodeId(0), + payload: PostAsapOperatorPayload::SummaryMerge, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: logical_schema(family.clone()), + guarantee: None, + }]; + if merge { + nodes.push(PostAsapDagNode { + id: PostAsapNodeId(1), + payload: PostAsapOperatorPayload::SummaryMerge, + ..nodes[0].clone() + }); + } + let read_id = nodes.len() as u32; + nodes.push(PostAsapDagNode { + id: PostAsapNodeId(read_id), + payload: PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + }, + output_state: ExecutionDataState::INGESTION_ROWS, + output_schema: logical_schema(SummaryFamilyType::Plain(DataType::Float64)), + guarantee: None, + }); + if let Some(target) = target { + nodes.push(PostAsapDagNode { + id: PostAsapNodeId(nodes.len() as u32), + payload: PostAsapOperatorPayload::SummaryAgg { + family: target.clone(), + input: SummaryUpdate::column(ColumnRef::SampleValue), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + }, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: logical_schema(target), + guarantee: None, + }); + } + let edges = (1..nodes.len()) + .map(|i| PostAsapDagEdge { + producer: nodes[i - 1].id, + consumer: nodes[i].id, + role: EdgeRole::Input, + intermediate_schema: nodes[i - 1].output_schema.clone(), + data_state: nodes[i - 1].output_state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }) + .collect(); + let root = nodes.last().unwrap().id; + precompute::compile( + &PostAsapDag { nodes, edges, root }, + &[0], + &[u64::from(root.0)], + ) + .unwrap() +} +fn native_run( + program: &CompiledPhysicalDag, + family: SummaryFamilyType, + states: Vec>, + context: RunContext, +) -> Result>, asap_physical_operators::Error> { + let program = + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + .unwrap(); + let rows = states + .into_iter() + .enumerate() + .map(|(i, state)| { + vec![ + Value::Map(vec![].into()), + Value::Timestamp((i as i64 + 1) * 1000), + Value::Summary { + family: family.clone(), + state, + }, + ] + }) + .collect(); + let input = Batch::try_new(precompute::population_schema(family), rows)?; + let graph = program.instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(input.schema().clone(), vec![input])?) as Source<'_>, + )]))?; + block_on(async { + let mut rows = Vec::new(); + let mut stream = graph.execute(program.roots(), context)?.remove(0); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok(rows) + }) +} +fn ingestion_context(limits: Limits) -> RunContext { + RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }, + limits, + ) + .unwrap() +} +fn sum_state(value: f64) -> Arc { + let mut state = asap_physical_operators::summary_kernels::exact::ExactAccumulator::new( + planner_types::post_asap::SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Sum, + planner_types::post_asap::ExactParams::Sum, + ), + false, + ) + .unwrap(); + state.update(None, value, 0); + Arc::new(state) +} + +// Only an explicit merge may collapse distinct pane updates before finalization. +#[test] +fn explicit_merge_changes_pane_cardinality() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + for (merge, expected) in [(false, vec![2., 7.]), (true, vec![9.])] { + let program = state_graph(family.clone(), None, merge); + let rows = native_run( + &program, + family.clone(), + vec![sum_state(2.), sum_state(7.)], + ingestion_context(Limits::default()), + ) + .unwrap(); + let values = rows + .iter() + .map(|row| match row[2] { + Value::Float64(v) => v, + _ => panic!("numeric readout expected"), + }) + .collect::>(); + assert_eq!(values, expected); + assert!(matches!(rows.last().unwrap()[1], Value::Timestamp(2000))); + } +} + +// Typed updates reject invalid domains before publishing any target state. +#[test] +fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { + let source = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let target = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::DDSketch, + SketchParams::DDSketch { alpha: 0.01 }, + ), + GroupingStrategy::default(), + ); + let program = state_graph(source.clone(), Some(target), false); + assert!(native_run( + &program, + source.clone(), + vec![sum_state(20.)], + ingestion_context(Limits::default()) + ) + .is_ok()); + for value in [-20., 0., f64::MAX, f64::NAN, f64::INFINITY] { + assert!(native_run( + &program, + source.clone(), + vec![sum_state(value)], + ingestion_context(Limits::default()) + ) + .is_err()); + } +} + +// An exact count must not silently lose units when exposed through Float64 rows. +#[test] +fn precompute_count_conversion_checks_precision() { + use asap_physical_operators::summary_kernels::exact::ExactAccumulator; + let family = SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count); + let program = state_graph(family.clone(), None, false); + for (count, valid) in [(3u64, true), ((1u64 << 53) + 1, false)] { + let mut state = + serde_json::to_value(ExactAccumulator::new(family.clone(), false).unwrap()).unwrap(); + state["scalar"]["Count"] = count.into(); + let state: ExactAccumulator = serde_json::from_value(state).unwrap(); + let result = native_run( + &program, + family.clone(), + vec![Arc::new(state)], + ingestion_context(Limits::default()), + ); + assert_eq!(result.is_ok(), valid); + } +} + +// Graph execution retains terminal cancellation and shared workspace limits. +#[test] +fn precompute_graph_enforces_cancellation_and_budget() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let program = state_graph(family.clone(), None, true); + let context = ingestion_context(Limits::default()); + context.cancel(); + let error = native_run(&program, family.clone(), vec![sum_state(1.)], context).unwrap_err(); + assert!(format!("{error:?}").contains("Cancelled")); + let error = native_run( + &program, + family, + vec![sum_state(1.)], + ingestion_context(Limits { + max_bytes: 1, + ..Limits::default() + }), + ) + .unwrap_err(); + assert!(format!("{error:?}").contains("MemoryLimit")); +} diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs new file mode 100644 index 00000000..24844b5e --- /dev/null +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -0,0 +1,270 @@ +//! Binary computation must be fully compiled before deployment binds values. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{compile_node, CompiledPhysicalDag, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Schema, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{ + BinaryOperator, ExecutionDataState, PostAsapDagNode, PostAsapNodeId, + PostAsapOperatorPayload, SummaryFamilyType, SummaryField, SummarySchema, + }, + pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +fn schema() -> Schema { + Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "labels".into(), + dtype: SummaryFamilyType::Plain(DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }), + nullable: false, + }, + SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }, + ], + time_index: None, + }) +} +fn row(name: &str, job: &str, value: f64) -> Vec { + vec![ + Value::Map( + vec![ + (Value::Utf8("__name__".into()), Value::Utf8(name.into())), + (Value::Utf8("job".into()), Value::Utf8(job.into())), + ] + .into(), + ), + Value::Float64(value), + ] +} +fn program() -> CompiledPhysicalDag { + let schema = schema(); + let node = PostAsapDagNode { + id: PostAsapNodeId(2), + payload: PostAsapOperatorPayload::Binary { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + vector_match: None, + checked_relative_division: true, + checked_finite_division: false, + }, + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*schema).clone(), + guarantee: None, + }; + let operator = compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(); + let graph = CompiledPhysicalDag::from_operators( + BTreeMap::from([ + (0, InputContract::bounded(schema.clone())), + (1, InputContract::bounded(schema)), + ]), + BTreeMap::from([(2, (vec![0, 1], operator))]), + vec![2], + ) + .unwrap(); + serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()).unwrap() +} +fn evaluate( + left: Vec>, + right: Vec>, +) -> Result>, asap_physical_operators::Error> { + let graph = program(); + let sources = [left, right] + .into_iter() + .enumerate() + .map(|(id, rows)| { + let batch = Batch::try_new(schema(), rows).unwrap(); + ( + id as u64, + Box::new(Operator::source(schema(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect(); + let bound = graph.instantiate(sources)?; + let ctx = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 0, + }, + Limits::default(), + )?; + block_on(async { + let mut stream = bound.execute(&[2], ctx)?.remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok(rows) + }) +} + +#[test] +fn compiled_binary_matches_series_and_preserves_checked_division() { + let rows = evaluate( + vec![row("left", "api", 6.), row("left", "unmatched", 8.)], + vec![row("right", "api", 2.)], + ) + .unwrap(); + assert_eq!( + serde_json::to_value(&rows).unwrap(), + serde_json::to_value(vec![vec![ + Value::Map(vec![(Value::Utf8("job".into()), Value::Utf8("api".into()))].into()), + Value::Float64(3.) + ]]) + .unwrap() + ); + assert!(evaluate(vec![row("a", "api", 1.)], vec![row("b", "api", 0.)]).is_err()); +} + +#[test] +fn duplicate_matching_identity_is_rejected() { + assert!(evaluate( + vec![row("a", "api", 1.)], + vec![row("b", "api", 2.), row("c", "api", 3.)] + ) + .is_err()); +} + +// Scalar broadcasting and comparison filtering keep the vector operand's value. +#[test] +fn scalar_broadcast_and_bool_comparison_are_distinct() { + use asap_physical_operators::physical_planner::promql_values; + use planner_types::pre_asap::CompareOpKind; + for return_bool in [false, true] { + let graph = promql_values::compile_binary( + &BinaryOperator { + kind: BinaryOpKind::Compare(CompareOpKind::Lt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + true, + false, + ) + .unwrap(); + let graph = + serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + .unwrap(); + let scalar = promql_values::scalar_schema(); + let vector = promql_values::vector_schema(); + let sources = BTreeMap::from([ + ( + 0, + Box::new( + Operator::source( + scalar.clone(), + vec![Batch::try_new(scalar, vec![vec![Value::Float64(2.)]]).unwrap()], + ) + .unwrap(), + ) as Source<'_>, + ), + ( + 1, + Box::new( + Operator::source( + vector.clone(), + vec![Batch::try_new( + vector, + vec![row("requests", "api", 4.), row("requests", "worker", 1.)], + ) + .unwrap()], + ) + .unwrap(), + ) as Source<'_>, + ), + ]); + let bound = graph.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + let result = block_on(async { + bound + .execute(&[2], context) + .unwrap() + .remove(0) + .next() + .await + .unwrap() + .unwrap() + }); + assert_eq!(result.rows().len(), if return_bool { 2 } else { 1 }); + assert!( + matches!(result.rows()[0].last(), Some(Value::Float64(v)) if *v == if return_bool { 1. } else { 4. }) + ); + let Value::Map(labels) = &result.rows()[0][0] else { + panic!("missing labels") + }; + assert_eq!( + labels + .iter() + .any(|(key, _)| matches!(key, Value::Utf8(s) if s.as_ref() == "__name__")), + !return_bool + ); + } +} + +// Terminal request controls retain their native error classification. +#[test] +fn binary_obeys_memory_and_cancellation() { + for cancel in [false, true] { + let graph = program(); + let sources = (0..2) + .map(|id| { + ( + id, + Box::new( + Operator::source( + schema(), + vec![Batch::try_new(schema(), vec![row("x", "api", 1.)]).unwrap()], + ) + .unwrap(), + ) as Source<'_>, + ) + }) + .collect(); + let bound = graph.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 0, + }, + Limits { + max_bytes: if cancel { 10000 } else { 1 }, + ..Limits::default() + }, + ) + .unwrap(); + if cancel { + context.cancel(); + } + let result = block_on(async { + match bound.execute(&[2], context) { + Err(error) => Err(error), + Ok(mut streams) => streams.remove(0).next().await.unwrap().map(|_| ()), + } + }); + assert!(matches!( + (cancel, result), + (true, Err(asap_physical_operators::Error::Cancelled)) + | (false, Err(asap_physical_operators::Error::MemoryLimit)) + )); + } +} diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs new file mode 100644 index 00000000..886b7d00 --- /dev/null +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -0,0 +1,492 @@ +//! Compile, persist and rebind dynamic-label computation without deployment lowering. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{promql_values::*, CompiledPhysicalDag, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::pre_asap::{AggIntent, ColumnRef, GroupKeys}; +use std::collections::BTreeMap; + +fn row(labels: &[(&str, &str)], value: f64) -> Vec { + vec![ + Value::Map( + labels + .iter() + .map(|(k, v)| (Value::Utf8((*k).into()), Value::Utf8((*v).into()))) + .collect::>() + .into(), + ), + Value::Float64(value), + ] +} +fn run(graph: CompiledPhysicalDag, rows: Vec>) -> Vec> { + run_inputs(graph, vec![Batch::try_new(vector_schema(), rows).unwrap()]).unwrap() +} +fn run_inputs( + graph: CompiledPhysicalDag, + batches: Vec, +) -> Result>, asap_physical_operators::Error> { + let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + .unwrap(); + let sources = batches + .into_iter() + .enumerate() + .map(|(id, batch)| { + ( + id as u64, + Box::new(Operator::source(batch.schema().clone(), vec![batch]).unwrap()) + as Source<'_>, + ) + }) + .collect::>(); + let bound = graph.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = bound.execute(graph.roots(), context).unwrap().remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok(rows) + }) +} +fn equal_rows(actual: Vec>, expected: Vec>) { + let mut actual = actual + .into_iter() + .map(|r| serde_json::to_string(&r).unwrap()) + .collect::>(); + let mut expected = expected + .into_iter() + .map(|r| serde_json::to_string(&r).unwrap()) + .collect::>(); + actual.sort(); + expected.sort(); + assert_eq!(actual, expected); +} + +#[test] +fn grouping_preserves_unenumerated_labels_and_empty_label_semantics() { + let rows = vec![ + row(&[("__name__", "m"), ("instance", "a"), ("job", "api")], 1.), + row(&[("__name__", "m"), ("instance", "b"), ("job", "api")], 2.), + row(&[("instance", "c"), ("job", "")], 4.), + row(&[("instance", "d")], 8.), + ]; + equal_rows( + run( + compile_aggregate( + &AggIntent::Sum { col: None }, + &GroupKeys::without(vec![ColumnRef::Named("instance".into())]), + ) + .unwrap(), + rows.clone(), + ), + vec![row(&[("job", "api")], 3.), row(&[], 12.)], + ); + equal_rows( + run( + compile_aggregate( + &AggIntent::Count { + accuracy: planner_types::types::AccuracyTarget::Exact, + }, + &GroupKeys::by(vec![ColumnRef::Named("job".into())]), + ) + .unwrap(), + rows, + ), + vec![row(&[("job", "api")], 2.), row(&[], 2.)], + ); +} + +#[test] +fn ranking_and_grouped_limit_preserve_full_selected_series() { + let grouping = GroupKeys::by(vec![ColumnRef::Named("job".into())]); + let rows = vec![ + row(&[("instance", "a"), ("job", "api")], 1.), + row(&[("instance", "b"), ("job", "api")], 3.), + row(&[("instance", "c"), ("job", "worker")], 2.), + ]; + let sorted = run(compile_sort(true, &grouping).unwrap(), rows); + let selected = run(compile_limit(1, 0, &grouping).unwrap(), sorted); + equal_rows( + selected, + vec![ + row(&[("instance", "b"), ("job", "api")], 3.), + row(&[("instance", "c"), ("job", "worker")], 2.), + ], + ); + assert!(run(compile_limit(0, 0, &grouping).unwrap(), vec![row(&[], 1.)]).is_empty()); +} + +#[test] +fn empty_vector_aggregation_stays_empty() { + assert!(run( + compile_aggregate(&AggIntent::Sum { col: None }, &GroupKeys::default()).unwrap(), + vec![] + ) + .is_empty()); + let scalar = run(compile_vector_to_scalar().unwrap(), vec![]); + assert!(matches!(scalar[0][0],Value::Float64(v) if v.is_nan())); +} + +// One persisted temporal graph accepts different request windows and detects resets. +#[test] +fn temporal_graph_uses_bound_window_without_recompilation() { + let graph = compile_temporal(&AggIntent::Rate, false).unwrap(); + for start in [0, 60_000] { + let labels = row(&[("__name__", "counter"), ("job", "api")], 0.)[0].clone(); + let samples = [(0, 5.), (30_000, 1.), (60_000, 7.)]; + let rows = samples + .into_iter() + .map(|(time, value)| { + vec![ + labels.clone(), + Value::Timestamp(start + time), + Value::Float64(value), + Value::Timestamp(start), + Value::Timestamp(start + 60_000), + ] + }) + .collect(); + let output = run_inputs( + graph.clone(), + vec![Batch::try_new(matrix_schema(), rows).unwrap()], + ) + .unwrap(); + equal_rows(output, vec![row(&[("job", "api")], 7. / 60.)]); + } + let labels = row(&[("job", "api")], 0.)[0].clone(); + let rows = vec![ + vec![ + labels.clone(), + Value::Timestamp(0), + Value::Float64(1.), + Value::Timestamp(0), + Value::Timestamp(1000), + ], + vec![ + labels, + Value::Timestamp(1000), + Value::Float64(2.), + Value::Timestamp(0), + Value::Timestamp(2000), + ], + ]; + assert!(run_inputs(graph, vec![Batch::try_new(matrix_schema(), rows).unwrap()]).is_err()); +} + +// The quantile is an ordinary scalar input, and bucket labels are native computation. +#[test] +fn histogram_quantile_keeps_each_label_group() { + let graph = compile_histogram_quantile().unwrap(); + let buckets = vec![ + row(&[("job", "api"), ("le", "1")], 2.), + row(&[("job", "api"), ("le", "2")], 4.), + row(&[("job", "api"), ("le", "+Inf")], 4.), + ]; + let output = run_inputs( + graph, + vec![ + Batch::try_new(scalar_schema(), vec![vec![Value::Float64(0.75)]]).unwrap(), + Batch::try_new(vector_schema(), buckets).unwrap(), + ], + ) + .unwrap(); + equal_rows(output, vec![row(&[("job", "api")], 1.5)]); +} + +// Linking an ensemble preserves its shared producer and every selected operator. +#[test] +fn composed_ensemble_shares_a_producer_across_roots() { + use asap_physical_operators::{ + physical_planner::InputContract, + plan::{PhysicalOperator, PlanProperties}, + runtime::{Input, OutputStream}, + values::Schema, + }; + use planner_types::{ + post_asap::BinaryOperator, + pre_asap::{ArithmeticOpKind, BinaryOpKind}, + }; + struct Counted { + source: Operator, + starts: std::rc::Rc>, + } + impl PhysicalOperator for Counted { + fn name(&self) -> &str { + "CountedInput" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.source.schema() + } + fn output_bytes(&self, batch: &Batch) -> usize { + batch.bytes() + } + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + self.source.properties(inputs) + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, asap_physical_operators::Error> { + self.starts.set(self.starts.get() + 1); + self.source.start(inputs, context) + } + } + let aggregate = compile_aggregate( + &AggIntent::Sum { col: None }, + &GroupKeys::by(vec![ColumnRef::Named("job".into())]), + ) + .unwrap(); + let binary = compile_binary( + &BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_finite_division: false, + checked_relative_division: false, + }, + false, + false, + false, + ) + .unwrap(); + let graph = CompiledPhysicalDag::compose( + BTreeMap::from([(0, InputContract::bounded(vector_schema()))]), + BTreeMap::from([ + (10, (vec![0], aggregate)), + (20, (vec![10, 10], binary)), + (30, (vec![10], compile_negate(false).unwrap())), + ]), + vec![20, 30], + ) + .unwrap(); + let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + .unwrap(); + assert_eq!(graph.input_contracts().count(), 1); + let starts = std::rc::Rc::new(std::cell::Cell::new(0)); + for _ in 0..2 { + let input = Batch::try_new(vector_schema(), vec![row(&[("job", "api")], 3.)]).unwrap(); + let source = Counted { + source: Operator::source(vector_schema(), vec![input]).unwrap(), + starts: starts.clone(), + }; + let bound = graph + .instantiate(BTreeMap::from([(0, Box::new(source) as Source<'_>)])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_buffered_batches: 1, + ..Limits::default() + }, + ) + .unwrap(); + let results = + block_on(futures::future::join_all( + bound + .execute(graph.roots(), context) + .unwrap() + .into_iter() + .map(|mut stream| async move { + stream.next().await.unwrap().unwrap().rows().to_vec() + }), + )); + equal_rows(results[0].clone(), vec![row(&[("job", "api")], 6.)]); + equal_rows(results[1].clone(), vec![row(&[("job", "api")], -3.)]); + } + assert_eq!(starts.get(), 2); +} + +#[test] +fn compiled_constant_needs_no_deployment_source() { + let graph = compile_scalar(3.).unwrap(); + assert_eq!(graph.input_contracts().count(), 0); + let result = run_inputs(graph, vec![]).unwrap(); + assert!(matches!(result[0][0], Value::Float64(3.))); +} + +// Scalar broadcasting cannot silently create duplicate result identities when +// arithmetic or bool comparisons remove the metric name. +#[test] +fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { + use planner_types::{ + post_asap::BinaryOperator, + pre_asap::{ArithmeticOpKind, BinaryOpKind, CompareOpKind}, + }; + for left_scalar in [false, true] { + for names in [["a", "a"], ["a", "b"]] { + for (kind, return_bool) in [ + (BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), false), + (BinaryOpKind::Compare(CompareOpKind::Gt), true), + ] { + let graph = compile_binary( + &BinaryOperator { + kind, + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + left_scalar, + !left_scalar, + ) + .unwrap(); + let vector = Batch::try_new( + vector_schema(), + vec![ + row(&[("__name__", names[0]), ("job", "api")], 2.), + row(&[("__name__", names[1]), ("job", "api")], 3.), + ], + ) + .unwrap(); + let scalar = + Batch::try_new(scalar_schema(), vec![vec![Value::Float64(1.)]]).unwrap(); + let result = run_inputs( + graph, + if left_scalar { + vec![scalar, vector] + } else { + vec![vector, scalar] + }, + ); + assert!(result.is_err(), "duplicate output label sets were accepted"); + } + } + } + let graph = compile_binary( + &BinaryOperator { + kind: BinaryOpKind::Compare(CompareOpKind::Gt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + false, + false, + true, + ) + .unwrap(); + let rows = vec![ + row(&[("__name__", "a"), ("job", "api")], 2.), + row(&[("__name__", "b"), ("job", "api")], 3.), + ]; + equal_rows( + run_inputs( + graph, + vec![ + Batch::try_new(vector_schema(), rows.clone()).unwrap(), + Batch::try_new(scalar_schema(), vec![vec![Value::Float64(1.)]]).unwrap(), + ], + ) + .unwrap(), + rows, + ); +} + +// Persisted exact readout graphs, rather than the storage adapter, merge panes, +// finalize each population, and preserve the requested metric-name semantics. +#[test] +fn exact_state_readouts_recover_and_finalize_panes() { + use asap_physical_operators::factory::create_planner_accumulator; + use planner_types::post_asap::*; + use std::sync::Arc; + for (kind, params, expected) in [ + (ExactKind::Sum, ExactParams::Sum, 12.), + (ExactKind::Count, ExactParams::Count, 4.), + (ExactKind::Min, ExactParams::Min, 1.), + (ExactKind::Max, ExactParams::Max, 5.), + ] { + let family = SummaryFamilyType::ExactAggregate(kind, params); + for preserve in [false, true] { + let rows = [[1., 2.], [4., 5.]] + .into_iter() + .map(|samples| { + let mut state = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + for sample in samples { + state.update_single(sample, 0); + } + let labels = row(&[("__name__", "m"), ("instance", "a")], 0.).remove(0); + vec![ + labels, + Value::Summary { + family: family.clone(), + state: Arc::from(state.into_accumulator()), + }, + ] + }) + .collect(); + let output = run_inputs( + compile_exact_readout(family.clone(), 60_000, preserve).unwrap(), + vec![Batch::try_new(exact_state_schema(family.clone()).unwrap(), rows).unwrap()], + ) + .unwrap(); + let labels = if preserve { + vec![("__name__", "m"), ("instance", "a")] + } else { + vec![("instance", "a")] + }; + equal_rows(output, vec![row(&labels, expected)]); + } + } +} + +#[test] +fn recovered_exact_counter_uses_window_and_omits_insufficient_samples() { + use asap_physical_operators::factory::create_planner_accumulator; + use planner_types::post_asap::*; + use std::sync::Arc; + for (kind, params, expected) in [ + (ExactKind::Rate, ExactParams::Rate, 1.), + (ExactKind::Increase, ExactParams::Increase, 60.), + ] { + let family = SummaryFamilyType::ExactAggregate(kind, params); + let rows = [1, 2] + .into_iter() + .map(|count| { + let mut state = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + state.update_single(100., -50_000); + if count == 2 { + state.update_single(140., -10_000); + } + vec![ + row(&[("instance", if count == 1 { "one" } else { "two" })], 0.).remove(0), + Value::Summary { + family: family.clone(), + state: Arc::from(state.into_accumulator()), + }, + ] + }) + .collect(); + let output = run_inputs( + compile_exact_readout(family.clone(), 60_000, false).unwrap(), + vec![Batch::try_new(exact_state_schema(family).unwrap(), rows).unwrap()], + ) + .unwrap(); + equal_rows(output, vec![row(&[("instance", "two")], expected)]); + } +} diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs new file mode 100644 index 00000000..78e4c8c0 --- /dev/null +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -0,0 +1,386 @@ +//! Scan acceptance uses the public connector contract and Planner physical DAGs. +use asap_physical_operators::dag::{ + planner::bind_with_data_sources, + scan::{DataSources, MemorySource, RawSource}, + values::{Batch, Schema, Value}, + Error, Limits, OutputStream, RunContext, Scope, +}; +use futures::{executor::block_on, stream, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{Column, DataType, GroupKeys, Predicate, QueryExpr, Source}, +}; +use std::{ + collections::BTreeMap, + rc::Rc, + sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, + }, +}; + +fn fixture() -> (QueryExpr, Schema, Vec) { + let schema = + planner_types::pre_asap::Schema::new(vec![Column::new("value", DataType::Int64, true)]); + let output = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: true, + }], + time_index: None, + }); + let scan = QueryExpr::Scan { + source: Source::Table { + table_ref: "numbers".into(), + }, + predicates: vec![Predicate(Rc::new(QueryExpr::IsNotNull(Rc::new( + QueryExpr::Column(0), + ))))], + schema, + }; + let batches = vec![ + Batch::try_new( + output.clone(), + vec![vec![Value::Int64(3)], vec![Value::Null]], + ) + .unwrap(), + Batch::try_new( + output.clone(), + vec![vec![Value::Int64(9)], vec![Value::Int64(2)]], + ) + .unwrap(), + ]; + (scan, output, batches) +} +fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsapDag { + let node = |id, payload| PostAsapDagNode { + id: PostAsapNodeId(id), + payload, + output_state: state, + output_schema: (**schema).clone(), + guarantee: None, + }; + let edge = |producer, consumer| PostAsapDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role: EdgeRole::Input, + intermediate_schema: (**schema).clone(), + data_state: state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }; + PostAsapDag { + nodes: vec![ + node(0, PostAsapOperatorPayload::Fallback { expression: scan }), + node( + 1, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Sort { + keys: vec![planner_types::pre_asap::SortKey { + expr: QueryExpr::Column(0), + ascending: false, + nulls_first: false, + }], + partition_by: GroupKeys::by(vec![]), + }, + }, + ), + node( + 2, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Limit { + n: 2, + offset: 0, + partition_by: GroupKeys::by(vec![]), + }, + }, + ), + ], + edges: vec![edge(0, 1), edge(1, 2)], + root: PostAsapNodeId(2), + } +} +fn registry(source: Arc) -> DataSources { + let mut r = DataSources::default(); + r.register( + Source::Table { + table_ref: "numbers".into(), + }, + source, + ) + .unwrap(); + r +} +fn context() -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits::default(), + ) + .unwrap() +} + +// Raw-only execution filters nulls and ranks across batches at either phase. +#[test] +fn raw_scan_to_sort_limit_at_both_phases() { + let (scan, schema, batches) = fixture(); + let sources = registry(Arc::new( + MemorySource::new(schema.clone(), batches).unwrap(), + )); + for state in [ + ExecutionDataState::QUERY_ROWS, + ExecutionDataState::INGESTION_ROWS, + ] { + let dag = plan(scan.clone(), &schema, state); + let bound = bind_with_data_sources(&dag, BTreeMap::new(), &[2], &sources).unwrap(); + let ctx = if state == ExecutionDataState::QUERY_ROWS { + context() + } else { + RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1, + revision: 1, + }, + Limits::default(), + ) + .unwrap() + }; + let rows = block_on(async { + let mut output = bound.execute(&[2], ctx.clone()).unwrap().remove(0); + let mut rows = vec![]; + while let Some(batch) = output.next().await { + rows.extend(batch.unwrap().rows().iter().cloned()); + } + rows + }); + assert!( + matches!(rows.as_slice(), [a,b] if matches!(a.as_slice(), [Value::Int64(9)]) && matches!(b.as_slice(), [Value::Int64(3)])) + ); + assert_eq!(ctx.retained_bytes(), 0); + } +} +struct CountingSource { + schema: Schema, + opened: Arc, + fail: bool, +} +impl RawSource for CountingSource { + fn boundedness(&self) -> asap_physical_operators::plan::Boundedness { + asap_physical_operators::plan::Boundedness::Bounded + } + fn schema(&self) -> Schema { + self.schema.clone() + } + fn scan(&self, _: RunContext) -> Result, Error> { + self.opened.fetch_add(1, Ordering::SeqCst); + if self.fail { + return Err(Error::Operator("reader failed".into())); + } + Ok(stream::iter(vec![Batch::try_new( + self.schema.clone(), + vec![vec![Value::Int64(7)]], + )]) + .boxed_local()) + } +} +// Binding and cancellation do not perform I/O; fan-out opens one cursor per run. +#[test] +fn lazy_open_shared_producer_and_cancellation() { + let (scan, schema, _) = fixture(); + let opened = Arc::new(AtomicUsize::new(0)); + let sources = registry(Arc::new(CountingSource { + schema: schema.clone(), + opened: opened.clone(), + fail: false, + })); + let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let bound = bind_with_data_sources(&plan, BTreeMap::new(), &[0, 2], &sources).unwrap(); + let ctx = context(); + let streams = bound.execute(&[0, 2], ctx.clone()).unwrap(); + assert_eq!(opened.load(Ordering::SeqCst), 0); + ctx.cancel(); + drop(streams); + assert_eq!(opened.load(Ordering::SeqCst), 0); + for _ in 0..2 { + block_on(async { + let streams = bound.execute(&[0, 2], context()).unwrap(); + let all = + futures::future::join_all(streams.into_iter().map(|s| s.collect::>())).await; + assert!(all.iter().flatten().all(Result::is_ok)); + }); + } + assert_eq!(opened.load(Ordering::SeqCst), 2); +} +// Unavailable sources and unsupported predicates fail before opening any cursor. +#[test] +fn binding_errors_and_reader_errors_are_not_empty_results() { + let (mut scan, schema, _) = fixture(); + assert!(DataSources::default().bind(&scan).is_err()); + let opened = Arc::new(AtomicUsize::new(0)); + let sources = registry(Arc::new(CountingSource { + schema: schema.clone(), + opened: opened.clone(), + fail: true, + })); + if let QueryExpr::Scan { predicates, .. } = &mut scan { + predicates.push(Predicate(Rc::new(QueryExpr::Column(0)))); + } + assert!(sources.bind(&scan).is_err()); + assert_eq!(opened.load(Ordering::SeqCst), 0); + let (scan, _, _) = fixture(); + let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let bound = bind_with_data_sources(&plan, BTreeMap::new(), &[2], &sources).unwrap(); + block_on(async { + let mut stream = bound.execute(&[2], context()).unwrap().remove(0); + assert!(stream.next().await.unwrap().is_err()); + }); +} + +// Schema drift cannot enter the DAG, and connector batches obey execution limits. +#[test] +fn schema_drift_and_memory_limits_fail_the_scan() { + struct Drift { + expected: Schema, + batch: Batch, + } + impl RawSource for Drift { + fn schema(&self) -> Schema { + self.expected.clone() + } + fn scan(&self, _: RunContext) -> Result, Error> { + Ok(stream::once(async { Ok(self.batch.clone()) }).boxed_local()) + } + } + let (scan, schema, batches) = fixture(); + let mut different = (*schema).clone(); + different.fields[0].name = "wrong".into(); + let bad = Batch::try_new(Arc::new(different), vec![vec![Value::Int64(1)]]).unwrap(); + let sources = registry(Arc::new(Drift { + expected: schema.clone(), + batch: bad, + })); + let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + block_on(async { + let mut s = graph.execute(&[0], context()).unwrap().remove(0); + assert!(s.next().await.unwrap().is_err()); + }); + let sources = registry(Arc::new(MemorySource::new(schema, batches).unwrap())); + let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + let ctx = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits { + max_bytes: 1, + max_buffered_batches: 1, + }, + ) + .unwrap(); + block_on(async { + let mut s = graph.execute(&[0], ctx.clone()).unwrap().remove(0); + assert!(s.next().await.unwrap().is_err()); + }); + assert_eq!(ctx.retained_bytes(), 0); +} + +// An empty table is a valid empty scan; nullable comparisons retain only TRUE. +#[test] +fn empty_sources_and_three_valued_predicates() { + use planner_types::pre_asap::{CompareOpKind, ScalarValue}; + let (mut scan, schema, batches) = fixture(); + if let QueryExpr::Scan { + predicates, source, .. + } = &mut scan + { + *source = Source::TimeSeries { + metric: "samples".into(), + }; + *predicates = vec![Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Gt, + right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(2))), + }))]; + } + for (batches, expected) in [(vec![], 0), (batches, 2)] { + let mut sources = DataSources::default(); + sources + .register( + Source::TimeSeries { + metric: "samples".into(), + }, + Arc::new(MemorySource::new(schema.clone(), batches).unwrap()), + ) + .unwrap(); + let plan = plan(scan.clone(), &schema, ExecutionDataState::QUERY_ROWS); + let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + block_on(async { + let mut s = graph.execute(&[0], context()).unwrap().remove(0); + let mut count = 0; + while let Some(b) = s.next().await { + count += b.unwrap().rows().len(); + } + assert_eq!(count, expected); + }); + } +} + +// A physical candidate can be compiled once without readers and rebound per run. +#[test] +fn compile_without_readers_and_rebind_inputs() { + use asap_physical_operators::{ + operators::Operator, + physical_planner::{compile, InputContract, Source}, + }; + let (scan, schema, batches) = fixture(); + let dag = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let compiled = compile( + &dag, + BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), + &[2], + ) + .unwrap(); + assert_eq!(compiled.input_contracts().count(), 1); + for _ in 0..2 { + let sources = BTreeMap::from([( + 0, + Box::new(Operator::source(schema.clone(), batches.clone()).unwrap()) as Source<'_>, + )]); + let graph = compiled.instantiate(sources).unwrap(); + let mut outputs = graph.execute(compiled.roots(), context()).unwrap(); + let result = block_on(outputs.remove(0).collect::>()); + assert!(result.iter().all(Result::is_ok)); + assert_eq!( + result + .iter() + .map(|b| b.as_ref().unwrap().rows().len()) + .sum::(), + 2 + ); + } + assert!(compiled.instantiate(BTreeMap::new()).is_err()); +} + +// Input boundedness must be proved during compilation, before readers exist. +#[test] +fn compilation_rejects_unknown_boundedness_for_sort() { + use asap_physical_operators::{ + physical_planner::{compile, InputContract}, + plan::{Boundedness, Emission, PlanProperties}, + }; + let (scan, schema, _) = fixture(); + let dag = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let input = InputContract { + schema, + properties: PlanProperties { + boundedness: Boundedness::Unknown, + emission: Emission::Unknown, + }, + }; + assert!(compile(&dag, BTreeMap::from([(0, input)]), &[2]).is_err()); +} diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs new file mode 100644 index 00000000..a61a596c --- /dev/null +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -0,0 +1,161 @@ +//! Opaque state travels through a retained physical projection without scalar decoding. +use asap_physical_operators::{ + expressions::Expression, + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{compile, CompiledPhysicalDag, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{ColumnRef, DataType, ProjectItem, QueryExpr}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +// A Post-ASAP projection may reorder/rename summary columns; recovery must retain +// the family and pass through the same immutable state, without decoding the payload. +#[test] +fn post_asap_summary_projection_survives_recovery() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let schema = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }, + SummaryField { + name: "service".into(), + dtype: SummaryFamilyType::Plain(DataType::Utf8), + nullable: false, + }, + ], + time_index: None, + }); + let output = SummarySchema { + fields: vec![ + schema.fields[1].clone(), + SummaryField { + name: "renamed".into(), + ..schema.fields[0].clone() + }, + ], + time_index: None, + }; + let dag = PostAsapDag { + nodes: vec![ + PostAsapDagNode { + id: PostAsapNodeId(0), + payload: PostAsapOperatorPayload::SummaryMerge, + output_schema: (*schema).clone(), + output_state: ExecutionDataState::INGESTION_SUMMARY, + guarantee: None, + }, + PostAsapDagNode { + id: PostAsapNodeId(1), + payload: PostAsapOperatorPayload::Value { + operation: ValueOperation::Project { + cols: vec![1, 0] + .into_iter() + .map(|index| ProjectItem { + alias: None, + expr: QueryExpr::Column(index), + }) + .collect(), + qualifier: None, + }, + }, + output_schema: output.clone(), + output_state: ExecutionDataState::INGESTION_SUMMARY, + guarantee: None, + }, + ], + edges: vec![PostAsapDagEdge { + producer: PostAsapNodeId(0), + consumer: PostAsapNodeId(1), + role: EdgeRole::Input, + intermediate_schema: (*schema).clone(), + data_state: ExecutionDataState::INGESTION_SUMMARY, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }], + root: PostAsapNodeId(1), + }; + let program = compile( + &dag, + BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), + &[1], + ) + .unwrap(); + let encoded = serde_json::to_vec(&program).unwrap(); + let program = serde_json::from_slice::(&encoded).unwrap(); + let mut forged: serde_json::Value = serde_json::from_slice(&encoded).unwrap(); + forged["nodes"]["1"]["Operator"]["operator"]["output"]["fields"][1]["dtype"] = + serde_json::json!({"Plain": "float64"}); + assert!( + serde_json::from_slice::(&serde_json::to_vec(&forged).unwrap()) + .is_err() + ); + assert!(Operator::project( + schema.clone(), + vec![("invalid".into(), Expression::Column(2))] + ) + .is_err()); + assert!(Operator::project( + schema.clone(), + vec![( + "invalid".into(), + Expression::Negate(Box::new(Expression::Column(0))) + )] + ) + .is_err()); + assert_eq!(*program.output_contract(1).unwrap().schema, output); + let mut accumulator = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + accumulator.update_single(7., 1); + let state = Arc::from(accumulator.into_accumulator()); + let batch = Batch::try_new( + schema.clone(), + vec![vec![ + Value::Summary { + family, + state: Arc::clone(&state), + }, + Value::Utf8("api".into()), + ]], + ) + .unwrap(); + let graph = program + .instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, + )])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut output = graph.execute(&[1], context).unwrap().remove(0); + let batch = output.next().await.unwrap().unwrap(); + assert!(matches!(&batch.rows()[0][0], Value::Utf8(label) if label.as_ref() == "api")); + let Value::Summary { + state: projected, .. + } = &batch.rows()[0][1] + else { + panic!("missing summary") + }; + assert!(Arc::ptr_eq(&state, projected)); + assert!(output.next().await.is_none()); + }); +} diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs new file mode 100644 index 00000000..664ae799 --- /dev/null +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -0,0 +1,1018 @@ +//! Planner output binds directly to the shared runtime at a declared rate-value frontier. +use asap_aware_mapping::{ + accuracy::{ + AccuracyEvidenceProvider, DefaultAccuracyModel, EqualSplitAllocator, PropagationStats, + }, + cost_model::DefaultCostModel, + Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG, +}; +use asap_physical_operators::dag::{ + operators::Operator, + planner::{compile, InputContract, Source}, + values::{Batch, Value}, + Limits, RunContext, Scope, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{DataType, QueryExpr}, + types::AccuracyTarget, +}; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; +struct Evidence; +impl AccuracyEvidenceProvider for Evidence { + fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + Some(1000) + } + fn propagation_stats( + &self, + op: &CompositionOperator, + _: &SummaryFamilyType, + _: Option<&SketchQuery>, + ) -> PropagationStats { + if matches!(op, CompositionOperator::TopKSelection) { + PropagationStats { + topk_selected_lower_bound: Some(101.), + topk_excluded_upper_bound: Some(100.), + topk_interval_failure_probability: Some(0.001), + ..Default::default() + } + } else { + Default::default() + } + } +} +// The evidence here exercises binding; it is not inferred from the sample data. +#[test] +fn planner_weighted_topk_binds_at_either_deployment_phase() { + assert_weighted_binding(&Evidence, SketchAlgorithm::CmsWithHeap); + assert_weighted_binding(&Evidence, SketchAlgorithm::CountSketchWithHeap); +} + +// Binding validates representation, while deployment owns evidence acceptance. +#[test] +fn physical_binding_does_not_impose_an_accuracy_acceptance_policy() { + assert_weighted_binding( + &asap_aware_mapping::accuracy::NoAccuracyEvidence, + SketchAlgorithm::CmsWithHeap, + ); + assert_weighted_binding( + &asap_aware_mapping::accuracy::NoAccuracyEvidence, + SketchAlgorithm::CountSketchWithHeap, + ); +} + +fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: SketchAlgorithm) { + let root = Rc::new( + lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.1), + ) + .unwrap(), + ); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + evidence, + ); + let plan = strategy + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .find_map(|candidate| match candidate.replacement { + Replacement::Summary(node) + if candidate.rationale.contains(&format!("{algorithm:?}")) => + { + Some(node) + } + _ => None, + }) + .unwrap(); + let dag = compile_post_asap_dag(&plan).unwrap(); + let build=dag.nodes.iter().find(|node|matches!(&node.payload,PostAsapOperatorPayload::SummaryAgg{family:SummaryFamilyType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); + let rate_id = dag + .edges + .iter() + .find(|edge| edge.consumer == build.id) + .unwrap() + .producer; + let rates = Arc::new( + dag.nodes + .iter() + .find(|node| node.id == rate_id) + .unwrap() + .output_schema + .clone(), + ); + let rows = [ + ("auth", "api", 0.125), + ("auth", "api", 0.25), + ("checkout", "api", 0.3125), + ("search", "api", 0.0625), + ("ingest", "batch", 100.), + ("export", "batch", 80.), + ("cleanup", "batch", 20.), + ] + .into_iter() + .map(|(service, job, value)| { + rates + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "job" => Value::Utf8(job.into()), + "value" => Value::Float64(value), + _ => match field.dtype { + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), + _ => panic!("unexpected rate column {field:?}"), + }, + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(rates.clone(), rows).unwrap(); + for (phase, scope) in [ + ( + ExecutionTiming::IngestionTime, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 60_000, + revision: 1, + }, + ), + ( + ExecutionTiming::QueryTime, + Scope::Query { + evaluation_time_ms: 60_000, + revision: 1, + }, + ), + ] { + let placed = dag + .with_execution_phases(&dag.nodes.iter().map(|node| (node.id, phase)).collect()) + .unwrap(); + let source = Box::new(Operator::source(rates.clone(), vec![batch.clone()]).unwrap()) + as Source<'static>; + let compiled = compile( + &placed, + BTreeMap::from([(rate_id.0 as u64, InputContract::bounded(rates.clone()))]), + &[dag.root.0 as u64], + ) + .unwrap(); + let graph = compiled + .instantiate(BTreeMap::from([(rate_id.0 as u64, source)])) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let output = block_on(async { + let mut output = Vec::new(); + let mut stream = graph + .execute(&[dag.root.0 as u64], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + output.extend(batch.unwrap().rows().iter().cloned()); + } + output + }); + assert_eq!(output.len(), 4); + let mut scores = output + .iter() + .map(|row| { + row.iter() + .find_map(|v| { + if let Value::Float64(v) = v { + Some(*v) + } else { + None + } + }) + .unwrap() + }) + .collect::>(); + scores.sort_by(f64::total_cmp); + assert_eq!(scores, vec![0.3125, 0.375, 80., 100.]); + } +} + +use asap_frontend_promql::lower_promql_workload; +use planner_types::workload::{ + AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence as WorkloadEvidence, + PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, + TimeSelection, +}; +pub fn lower_promql( + query: &str, + accuracy: AccuracyTarget, +) -> Result { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: WorkloadEvidence { + value: Some(DurationMs(1_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let mut lowered = lower_promql_workload(&workload, 0)?; + Ok(lowered.remove(0)) +} + +// The old untyped heap updater must not silently round a Planner rate update. +#[test] +fn rate_updates_cannot_enter_integer_heap_factory() { + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 272, + depth: 5, + heap_size: 100, + }, + ), + Default::default(), + ); + let input = SummaryUpdate { + item: Some(SummaryInputExpr::Column( + planner_types::pre_asap::ColumnRef::Named("service".into()), + )), + weight: SummaryInputExpr::Column(planner_types::pre_asap::ColumnRef::SampleValue), + weight_domain: WeightDomain::NonNegative { + proof: NonNegativeWeightProof::ResetAwareCounterDerivative, + }, + }; + assert!( + asap_physical_operators::factory::create_planner_accumulator( + &family, + &input, + &Default::default() + ) + .is_err() + ); +} + +/// A catalog-resolved per-series rate can feed a heap sketch directly, without +/// requiring an otherwise unnecessary grouped Sum between Rate and TopK. +#[test] +fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { + check_direct_rate_topk(false); +} + +// Unreferenced labels still distinguish series throughout Rate and heap readout. +#[test] +fn direct_rate_topk_preserves_dynamic_unreferenced_labels() { + check_direct_rate_topk(true); +} + +fn check_direct_rate_topk(dynamic: bool) { + use asap_physical_operators::physical_planner::promql_rows::{ + decode_series_identity, series_row, with_series_identity, SERIES_IDENTITY_COLUMN, + }; + let mut logical = + lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(); + fn resolve_catalog(node: &mut QueryExpr) { + match node { + QueryExpr::Aggregate { child, .. } | QueryExpr::TimeRange { child, .. } => { + resolve_catalog(Rc::make_mut(child)) + } + QueryExpr::Scan { schema, .. } => { + schema.closed = true; + schema + .columns + .push(planner_types::pre_asap::schema::Column::new( + "service", + DataType::Utf8, + false, + )); + } + _ => panic!("unexpected input shape: {node:?}"), + } + } + if dynamic { + logical = with_series_identity(&logical).unwrap(); + } else { + resolve_catalog(&mut logical); + } + let root = Rc::new(logical); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let candidates = strategy.replacements(&TargetSubDAG::new(&root)); + for algorithm in [ + SketchAlgorithm::CmsWithHeap, + SketchAlgorithm::CountSketchWithHeap, + ] { + let candidate = candidates + .iter() + .find_map(|candidate| match &candidate.replacement { + Replacement::Summary(node) + if candidate.rationale.contains(&format!("{algorithm:?}")) => + { + Some(node) + } + _ => None, + }) + .unwrap_or_else(|| panic!("missing {algorithm:?} over direct Rate")); + if dynamic { + let (source, ranked) = + asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking( + candidate, + ) + .unwrap(); + assert!(matches!( + source.expr, + SummaryExpr::ValueOperation { + operation: ValueOperation::FinalizeExactAccumulator, + .. + } + )); + assert_eq!(ranked.input_contracts().count(), 1); + let encoded = String::from_utf8(serde_json::to_vec(&ranked).unwrap()).unwrap(); + assert!(encoded.contains("KeyedSummaryBuild")); + assert!(encoded.contains("KeyedReadout")); + assert!( + !encoded.contains("\"Rate\""), + "Rate must be supplied by its exact stored-state readout" + ); + } + let dag = compile_post_asap_dag(candidate).unwrap(); + assert!(dag.nodes.iter().any(|node| matches!(&node.payload, + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); + let build = dag.nodes.iter().find(|node| matches!(&node.payload, + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); + let input_id = dag + .edges + .iter() + .find(|edge| edge.consumer == build.id) + .unwrap() + .producer; + let schema = Arc::new( + dag.nodes + .iter() + .find(|node| node.id == input_id) + .unwrap() + .output_schema + .clone(), + ); + let raw = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::TimeRange { .. } + } + ) + }) + .unwrap_or_else(|| panic!("no raw counter source: {dag:?}")); + let raw_schema = Arc::new(raw.output_schema.clone()); + let raw_compiled = compile( + &dag, + BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(raw_schema.clone()), + )]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + let bytes = serde_json::to_vec(&raw_compiled).unwrap(); + let raw_compiled = serde_json::from_slice::< + asap_physical_operators::physical_planner::CompiledPhysicalDag, + >(&bytes) + .unwrap(); + // Each evaluation receives a complete raw window. A reset, a stopped + // series and an expired leader must not retain last run's heap weights. + for (end, series, expected) in [ + ( + 60_000, + vec![ + ("auth", vec![10., 30., 50.]), + ("checkout", vec![10., 50., 90.]), + ("search", vec![10., 70., 130.]), + ], + vec![11. / 6., 8. / 3.], + ), + ( + 120_000, + vec![ + ("auth", vec![100., 10., 50.]), + ("checkout", vec![100., 100., 100.]), + ], + vec![0., 1.25], + ), + ] { + let mut raw_rows = Vec::new(); + for (service, samples) in series { + for (offset, value) in [10_000, 30_000, 50_000].into_iter().zip(samples) { + if dynamic { + raw_rows.push( + series_row( + &raw_schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("service".into(), service.into()), + ("unreferenced".into(), format!("{service}-extra")), + ]), + end - 60_000 + offset, + value, + ) + .unwrap(), + ); + continue; + } + raw_rows.push( + raw_schema + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "job" => Value::Utf8("api".into()), + "value" => Value::Float64(value), + "ts" => Value::Timestamp(end - 60_000 + offset), + _ => panic!("unexpected raw field"), + }) + .collect(), + ); + } + } + let raw_batch = Batch::try_new(raw_schema.clone(), raw_rows).unwrap(); + for scope in [ + Scope::Ingestion { + window_start_ms: end - 60_000, + window_end_ms: end, + revision: 1, + }, + Scope::Query { + evaluation_time_ms: end, + revision: 1, + }, + ] { + let source = Box::new( + Operator::source(raw_schema.clone(), vec![raw_batch.clone()]).unwrap(), + ) as Source<'static>; + let graph = raw_compiled + .instantiate(BTreeMap::from([(u64::from(raw.id.0), source)])) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let mut raw_scores = block_on(async { + let mut scores = Vec::new(); + let mut stream = graph + .execute(&[u64::from(dag.root.0)], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + let batch = batch.unwrap(); + for row in batch.rows() { + if dynamic { + let column = batch + .schema() + .fields + .iter() + .position(|field| field.name == SERIES_IDENTITY_COLUMN) + .unwrap(); + let Value::Utf8(encoded) = &row[column] else { + panic!("identity lost"); + }; + let labels = decode_series_identity(encoded).unwrap(); + assert_eq!(labels["job"], "api"); + assert_eq!( + labels["unreferenced"], + format!("{}-extra", labels["service"]) + ); + } + assert!(row.iter().any( + |value| matches!(value, Value::Timestamp(time) if *time == end) + )); + scores.extend(row.iter().filter_map(|value| match value { + Value::Float64(value) => Some(*value), + _ => None, + })); + } + } + scores + }); + raw_scores.sort_by(f64::total_cmp); + assert_eq!(raw_scores.len(), expected.len()); + for (actual, expected) in raw_scores.iter().zip(&expected) { + assert!( + (actual - expected).abs() < 1e-12, + "raw counter semantics must precede heap ranking: {raw_scores:?}" + ); + } + } + } + let compiled = compile( + &dag, + BTreeMap::from([( + u64::from(input_id.0), + InputContract::bounded(schema.clone()), + )]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + for (time, values, expected) in [ + ( + 60_000, + vec![("auth", 3.), ("checkout", 2.), ("search", 1.)], + vec![2., 3.], + ), + ( + 61_000, + vec![("auth", 0.), ("checkout", 2.), ("search", 4.)], + vec![2., 4.], + ), + (62_000, vec![("auth", 0.), ("checkout", 2.)], vec![0., 2.]), + ] { + let rows = values + .into_iter() + .map(|(service, value)| { + if dynamic { + return series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("service".into(), service.into()), + ]), + time, + value, + ) + .unwrap(); + } + schema + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "job" => Value::Utf8("api".into()), + "value" => Value::Float64(value), + "ts" => Value::Timestamp(time), + _ => panic!("unexpected rate field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(schema.clone(), rows).unwrap(); + for scope in [ + Scope::Query { + evaluation_time_ms: time, + revision: 1, + }, + Scope::Ingestion { + window_start_ms: time - 60_000, + window_end_ms: time, + revision: 1, + }, + ] { + let source = + Box::new(Operator::source(schema.clone(), vec![batch.clone()]).unwrap()) + as Source<'static>; + let graph = compiled + .instantiate(BTreeMap::from([(u64::from(input_id.0), source)])) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let mut scores = block_on(async { + let mut scores = vec![]; + let mut stream = graph + .execute(&[u64::from(dag.root.0)], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + let batch = batch.unwrap(); + for row in batch.rows() { + assert!(row.iter().any( + |value| matches!(value, Value::Timestamp(actual) if *actual == time) + )); + scores.push( + row.iter() + .find_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .unwrap(), + ); + } + } + scores + }); + scores.sort_by(f64::total_cmp); + assert_eq!( + scores, expected, + "heap snapshots must not accumulate across evaluations" + ); + } + } + } +} + +// Spatial ranking consumes one eligible instant vector. Signed values require +// CountSketch; a raw metric does not establish the non-negative CMS contract. +#[test] +fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { + use asap_physical_operators::physical_planner::promql_rows::{ + decode_series_identity, series_row, with_series_identity, SERIES_IDENTITY_COLUMN, + }; + let logical = lower_promql("topk by(job)(1, m)", AccuracyTarget::Epsilon(0.1)).unwrap(); + let root = Rc::new(with_series_identity(&logical).unwrap()); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let candidates = strategy + .current_series_topk_candidates(&root, &AccuracyTarget::Epsilon(0.1)) + .candidates; + assert!(!candidates + .iter() + .any(|c| c.rationale.contains("CmsWithHeap"))); + let selected = candidates + .iter() + .find_map(|candidate| match &candidate.replacement { + Replacement::Summary(node) if candidate.rationale.contains("CountSketchWithHeap") => { + Some(node) + } + _ => None, + }) + .expect("signed spatial TopK must expose CountSketch with heap"); + let dag = compile_post_asap_dag(selected).unwrap(); + let raw = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::TimeRange { .. } + } + ) + }) + .unwrap(); + let schema = Arc::new(raw.output_schema.clone()); + let program = compile( + &dag, + BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + let snapshot_program = + asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + selected, + ) + .unwrap(); + let encoded: serde_json::Value = + serde_json::from_slice(&serde_json::to_vec(&snapshot_program).unwrap()).unwrap(); + assert!(!encoded.to_string().contains("CurrentSeries")); + assert!(encoded.to_string().contains("KeyedSummaryBuild")); + assert!(encoded.to_string().contains("KeyedReadout")); + for (values, expected, score) in [ + ([100., 20.], "a", 100.), + ([1., 20.], "b", 20.), + ([-10., -2.], "b", -2.), + ] { + let rows = ["a", "b"] + .into_iter() + .zip(values) + .map(|(instance, value)| { + series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("unreferenced".into(), instance.into()), + ]), + 60_000, + value, + ) + .unwrap() + }) + .collect(); + let batch = Batch::try_new(schema.clone(), rows).unwrap(); + let graph = program + .instantiate(BTreeMap::from([( + u64::from(raw.id.0), + Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'_>, + )])) + .unwrap(); + block_on(async { + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 60_000, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut result = Vec::new(); + while let Some(batch) = stream.next().await { + let batch = batch.unwrap(); + let identity = batch + .schema() + .fields + .iter() + .position(|f| f.name == SERIES_IDENTITY_COLUMN) + .unwrap(); + let value = batch + .schema() + .fields + .iter() + .position(|f| f.name == "value") + .unwrap(); + for row in batch.rows() { + let Value::Utf8(labels) = &row[identity] else { + panic!() + }; + let Value::Float64(v) = row[value] else { + panic!() + }; + result.push(( + decode_series_identity(labels).unwrap()["unreferenced"].clone(), + v, + )); + } + } + assert_eq!(result, vec![(expected.into(), score)]); + }); + } +} + +// Placement changes execution ownership only. Every fixed-window candidate +// contains Rate finalization before a fresh heap, with query readout downstream. +#[test] +fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { + use asap_physical_operators::physical_planner::{ + compile_candidate, promql_rows::with_series_identity, + }; + let root = Rc::new( + with_series_identity( + &lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(), + ) + .unwrap(), + ); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let candidates = strategy.fixed_window_rate_candidates(&root).candidates; + assert_eq!(candidates.len(), 2); + for candidate in candidates { + let Replacement::Summary(root) = candidate.replacement else { + panic!() + }; + let dag = compile_post_asap_dag(&root).unwrap(); + let state = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) + .unwrap(); + let heap = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::Sketch(..), + .. + } + ) + }) + .unwrap(); + assert_eq!(heap.output_state.timing, ExecutionTiming::IngestionTime); + let physical = compile_candidate( + &dag, + BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &[u64::from(heap.id.0)], + ) + .unwrap(); + let exported = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_aggregation(&root).unwrap(); + assert_eq!( + serde_json::to_vec(&exported).unwrap(), + serde_json::to_vec(&physical).unwrap() + ); + assert!( + asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking(&root) + .is_err(), + "query binding must not move the selected precompute frontier" + ); + // Execute the selected split across a state serialization boundary. + // Each run builds fresh weights from that window's counters. + let execute = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDag, + input: Batch, + scope: Scope| { + let id = plan.input_contracts().next().unwrap().0; + let source = Box::new(Operator::source(input.schema().clone(), vec![input]).unwrap()) + as Source<'static>; + let graph = plan.instantiate(BTreeMap::from([(id, source)])).unwrap(); + block_on(async { + let mut stream = graph + .execute( + plan.roots(), + RunContext::new(scope, Limits::default()).unwrap(), + ) + .unwrap() + .remove(0); + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push((*batch.unwrap()).clone()); + } + assert_eq!(batches.len(), 1); + batches.remove(0) + }) + }; + let (family, input, grouping) = match &state.payload { + PostAsapOperatorPayload::SummaryAgg { + family, + input, + grouping, + .. + } => (family, input, grouping), + _ => unreachable!(), + }; + for (end, samples, leader) in [ + ( + 60_000, + [[0., 100., 200.], [0., 10., 20.], [0., 1., 2.]], + "a", + ), + ( + 120_000, + [[200., 200., 200.], [100., 0., 300.], [2., 3., 4.]], + "b", + ), + ] { + let schema = Arc::new(state.output_schema.clone()); + let rows = samples + .into_iter() + .zip(["a", "b", "c"]) + .map(|(samples, label)| { + let mut accumulator = + asap_physical_operators::factory::create_planner_accumulator( + family, input, grouping, + ) + .unwrap(); + for (offset, value) in [10_000, 30_000, 50_000].into_iter().zip(samples) { + accumulator.update_single(value, end - 60_000 + offset); + } + let summary = Value::Summary { + family: family.clone(), + state: Arc::from(accumulator.into_accumulator()), + }; + schema + .fields + .iter() + .map(|field| match &field.dtype { + SummaryFamilyType::ExactAggregate(..) => summary.clone(), + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(end), + SummaryFamilyType::Plain(DataType::Utf8) + if field.name == "$promql_series_identity" => + { + Value::Utf8( + serde_json::to_string(&BTreeMap::from([ + ("job", "api"), + ("instance", label), + ])) + .unwrap() + .into(), + ) + } + SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8("api".into()), + _ => panic!("unexpected state field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(schema, rows).unwrap(); + let precompute = physical.precompute.as_ref().unwrap(); + let heap = execute( + precompute, + batch, + Scope::Ingestion { + window_start_ms: end - 60_000, + window_end_ms: end, + revision: 1, + }, + ); + let result = execute( + &physical.query, + heap, + Scope::Query { + evaluation_time_ms: end, + revision: 1, + }, + ); + let identity = result + .schema() + .fields + .iter() + .position(|f| f.name == "$promql_series_identity") + .unwrap(); + let Value::Utf8(encoded) = &result.rows()[0][identity] else { + panic!() + }; + let labels: BTreeMap = serde_json::from_str(encoded).unwrap(); + assert_eq!(labels["instance"], leader); + assert_eq!(result.rows().len(), 2); + } + let precompute = + String::from_utf8(serde_json::to_vec(&physical.precompute.unwrap()).unwrap()).unwrap(); + assert!(precompute.contains("KeyedSummaryBuild")); + assert!(precompute.contains("Rate")); + assert!( + !String::from_utf8(serde_json::to_vec(&physical.query).unwrap()) + .unwrap() + .contains("KeyedSummaryBuild") + ); + } +} + +// Grouped Rate has a legal stored Sum candidate as well as query-time reduction. +#[test] +fn grouped_rate_exposes_precomputed_sum_with_query_readout() { + let root = Rc::new( + asap_physical_operators::physical_planner::promql_rows::with_series_identity( + &lower_promql("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact).unwrap(), + ) + .unwrap(), + ); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let direct = strategy.query_time_rate_aggregation_candidates(&root); + assert!( + direct.candidates.iter().any(|candidate| { + let Replacement::Summary(root) = &candidate.replacement else { + return false; + }; + let Ok((_, program)) = + asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking(root) + else { + return false; + }; + let output = program.output_contract(program.roots()[0]).unwrap(); + output + .schema + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_))) + }), + "query-time grouped Rate must finalize Sum inside the physical graph" + ); + let candidates = strategy.fixed_window_rate_candidates(&root).candidates; + assert!( + !candidates.is_empty(), + "Planner must expose Rate -> grouped Sum at ingestion" + ); + for candidate in candidates { + let Replacement::Summary(root) = candidate.replacement else { + panic!() + }; + let physical = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_aggregation(&root).unwrap(); + let precompute = + String::from_utf8(serde_json::to_vec(&physical.precompute.unwrap()).unwrap()).unwrap(); + assert!( + precompute.contains("SummaryBuild") + && precompute.contains("Rate") + && precompute.contains("Sum") + ); + let query = String::from_utf8(serde_json::to_vec(&physical.query).unwrap()).unwrap(); + assert!(query.contains("Readout") && !query.contains("SummaryBuild")); + } +} diff --git a/crates/integration-tests/Cargo.toml b/crates/integration-tests/Cargo.toml index 7b0c0d3e..5a5de990 100644 --- a/crates/integration-tests/Cargo.toml +++ b/crates/integration-tests/Cargo.toml @@ -13,3 +13,6 @@ asap-aware-mapping = { path = "../asap-aware-mapping" } asap_sketchlib = { workspace = true } serde_json = "1" tokio = { version = "1", features = ["rt", "macros", "rt-multi-thread"] } + +asap-physical-operators = { path = "../asap-physical-operators" } +futures = "0.3" diff --git a/crates/integration-tests/tests/kll_pane_execution.rs b/crates/integration-tests/tests/kll_pane_execution.rs new file mode 100644 index 00000000..e818796f --- /dev/null +++ b/crates/integration-tests/tests/kll_pane_execution.rs @@ -0,0 +1,286 @@ +//! Maintenance -> stored pane state -> independently bound query execution. +mod physical_common; +use asap_physical_operators::{ + operators::{Operator, ReadoutQuery}, + physical_planner::{CompiledPhysicalDag, InputContract, Source}, + plan::{PhysicalDag, PhysicalOperator, PlanProperties}, + runtime::{Input, Limits, OutputStream, RunContext, Scope}, + summary_kernels::datasketches_kll::DatasketchesKLLAccumulator, + values::{Batch, Schema, Value}, + AggregateCore, Error, +}; +use asap_types::{ + post_asap::{ + SketchAlgorithm, SketchKind, SketchParams, SketchQuery, SummaryFamilyType, SummaryField, + SummarySchema, + }, + pre_asap::DataType, +}; +use futures::{executor::block_on, StreamExt}; +use std::{ + collections::BTreeMap, + sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, + }, +}; + +fn family(k: u32) -> SummaryFamilyType { + SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), + Default::default(), + ) +} +fn raw_schema() -> Schema { + Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }) +} +fn query_scope() -> Scope { + Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + } +} +fn fixture() -> (CompiledPhysicalDag, Operator, Schema) { + let raw = raw_schema(); + let build = Operator::summary_build(raw.clone(), family(200), 0, None, vec![]).unwrap(); + let state = build.schema(); + let maintenance = CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(raw))]), + BTreeMap::from([(1, (vec![0], build))]), + vec![1], + ) + .unwrap(); + let merge = Operator::summary_merge(state.clone(), 0, vec![]).unwrap(); + (maintenance, merge, state) +} +fn pane_state(maintenance: &CompiledPhysicalDag, pane: i64) -> Arc { + // Twenty samples in each (start,end] one-minute pane; k=200 avoids + // compaction so quantiles and sample counts have deterministic oracles. + let raw = raw_schema(); + let rows = (0..20) + .map(|i| vec![Value::Float64((pane * 20 + i) as f64)]) + .collect(); + let state = physical_common::execute( + maintenance, + BTreeMap::from([(0, Batch::try_new(raw, rows).unwrap())]), + Scope::Ingestion { + window_start_ms: pane * 60_000, + window_end_ms: (pane + 1) * 60_000, + revision: 1, + }, + ); + let Value::Summary { state, .. } = &state[0][0].rows()[0][0] else { + panic!("missing KLL") + }; + state.clone() +} +fn restore(schema: Schema, states: &[Arc]) -> Batch { + Batch::try_new( + schema, + states + .iter() + .map(|state| { + vec![Value::Summary { + family: family(200), + state: state.clone(), + }] + }) + .collect(), + ) + .unwrap() +} +fn readout(schema: Schema, q: f64) -> Operator { + Operator::readout(schema, 0, ReadoutQuery::Sketch(SketchQuery::Quantile { q })).unwrap() +} +struct CountStarts { + operator: Operator, + starts: Arc, +} +impl PhysicalOperator for CountStarts { + fn name(&self) -> &str { + self.operator.name() + } + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + self.operator.properties(inputs) + } + fn requires_bounded_input(&self) -> bool { + self.operator.requires_bounded_input() + } + fn input_schemas(&self) -> Vec { + self.operator.input_schemas() + } + fn output_schema(&self) -> Schema { + self.operator.output_schema() + } + fn output_bytes(&self, batch: &Batch) -> usize { + self.operator.output_bytes(batch) + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + self.starts.fetch_add(1, Ordering::SeqCst); + self.operator.start(inputs, context) + } +} + +/// Actual codec bytes survive destruction of maintenance state; one shared +/// native merge supplies p50, p99 and the population-count oracle per run. +#[test] +fn five_panes_roundtrip_and_shared_merge_runs_once() { + let (maintenance, merge, schema) = fixture(); + let panes: Vec<_> = (0..6).map(|pane| pane_state(&maintenance, pane)).collect(); + drop(maintenance); + let compiled = CompiledPhysicalDag::from_operators( + (0..5) + .map(|id| (id, InputContract::bounded(schema.clone()))) + .collect(), + BTreeMap::from([ + ( + 5, + ( + vec![0, 1, 2, 3, 4], + Operator::union(schema.clone(), 5).unwrap(), + ), + ), + (6, (vec![5], merge.clone())), + (7, (vec![6], readout(schema.clone(), 0.5))), + (8, (vec![6], readout(schema.clone(), 0.99))), + ]), + vec![6, 7, 8], + ) + .unwrap(); + let layout = asap_types::post_asap::PaneLayout { + pane_width_ms: 60_000, + pane_origin_ms: Some(0), + }; + assert!( + asap_types::post_asap::validate_pane_coverage( + &layout, + Some(330_000), + &asap_types::post_asap::WindowEdgeCoverage::PaneAligned + ) + .is_err(), + "moving window edges require residual computation" + ); + for offset in [0, 1] { + let restored = restore(schema.clone(), &panes[offset..offset + 5]); + let evaluation_time_ms = (5 + offset as i64) * 60_000; + asap_types::post_asap::validate_pane_coverage( + &layout, + Some(evaluation_time_ms), + &asap_types::post_asap::WindowEdgeCoverage::PaneAligned, + ) + .unwrap(); + let inputs: BTreeMap<_, _> = (0..5) + .map(|id| { + ( + id as u64, + restore(schema.clone(), &panes[offset + id..offset + id + 1]), + ) + }) + .collect(); + // A five-pane deployment cannot bind only four state slots. + let incomplete: BTreeMap<_, _> = inputs + .iter() + .take(4) + .map(|(&id, batch)| { + ( + id, + Box::new(Operator::source(schema.clone(), vec![batch.clone()]).unwrap()) + as Source<'_>, + ) + }) + .collect(); + assert!(compiled.instantiate(incomplete).is_err()); + let result = physical_common::execute( + &compiled, + inputs, + Scope::Query { + evaluation_time_ms, + revision: 1, + }, + ); + let Value::Summary { state, .. } = &result[0][0].rows()[0][0] else { + panic!("missing merged state") + }; + let kll = state + .as_any() + .downcast_ref::() + .unwrap(); + assert_eq!(kll.inner.count(), 100); + let value = |index: usize| match result[index][0].rows()[0][0] { + Value::Float64(value) => value, + _ => panic!("missing quantile"), + }; + assert!((value(1) - (50 + offset * 20) as f64).abs() <= 1.); + assert!((value(2) - (99 + offset * 20) as f64).abs() <= 1.); + let starts = Arc::new(AtomicUsize::new(0)); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(schema.clone(), vec![restored]).unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + CountStarts { + operator: merge.clone(), + starts: starts.clone(), + }, + ) + .unwrap(); + dag.add(2, vec![1], readout(schema.clone(), 0.5)).unwrap(); + dag.add(3, vec![1], readout(schema.clone(), 0.99)).unwrap(); + let outputs = block_on(futures::future::join_all( + dag.execute( + &[2, 3], + RunContext::new(query_scope(), Limits::default()).unwrap(), + ) + .unwrap() + .into_iter() + .map(|stream| stream.collect::>()), + )); + assert!(outputs + .iter() + .all(|output| output.len() == 1 && output[0].is_ok())); + assert_eq!(starts.load(Ordering::SeqCst), 1); + } +} + +/// Relabelled KLL parameters and missing bindings fail explicitly. +#[test] +fn panes_reject_parameters_schema_and_missing_binding() { + let (_maintenance, merge, schema) = fixture(); + let wrong = Value::Summary { + family: family(200), + state: Arc::new(DatasketchesKLLAccumulator::new(128)), + }; + assert!(Batch::try_new(schema.clone(), vec![vec![wrong]]).is_err()); + let compiled = CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema))]), + BTreeMap::from([(1, (vec![0], merge))]), + vec![1], + ) + .unwrap(); + assert!(compiled.instantiate(BTreeMap::new()).is_err()); + let raw = raw_schema(); + let source = Operator::source( + raw.clone(), + vec![Batch::try_new(raw, vec![vec![Value::Float64(1.)]]).unwrap()], + ) + .unwrap(); + assert!(compiled + .instantiate(BTreeMap::from([(0, Box::new(source) as Source<'_>)])) + .is_err()); +} diff --git a/crates/integration-tests/tests/physical_common/mod.rs b/crates/integration-tests/tests/physical_common/mod.rs new file mode 100644 index 00000000..aca4d4ef --- /dev/null +++ b/crates/integration-tests/tests/physical_common/mod.rs @@ -0,0 +1,42 @@ +use asap_physical_operators::{ + operators::Operator, + physical_planner::{CompiledPhysicalDag, Source}, + runtime::{Limits, RunContext, Scope}, + values::Batch, +}; +use futures::{executor::block_on, StreamExt}; +use std::collections::BTreeMap; + +pub fn execute( + plan: &CompiledPhysicalDag, + inputs: BTreeMap, + scope: Scope, +) -> Vec> { + let sources = inputs + .into_iter() + .map(|(id, batch)| { + ( + id, + Box::new(Operator::source(batch.schema().clone(), vec![batch]).unwrap()) + as Source<'_>, + ) + }) + .collect(); + let dag = plan.instantiate(sources).unwrap(); + block_on(async { + let streams = dag + .execute( + plan.roots(), + RunContext::new(scope, Limits::default()).unwrap(), + ) + .unwrap(); + futures::future::join_all(streams.into_iter().map(|mut stream| async move { + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push((*batch.unwrap()).clone()); + } + batches + })) + .await + }) +} diff --git a/crates/integration-tests/tests/sql_to_physical.rs b/crates/integration-tests/tests/sql_to_physical.rs new file mode 100644 index 00000000..be96107e --- /dev/null +++ b/crates/integration-tests/tests/sql_to_physical.rs @@ -0,0 +1,165 @@ +//! SQL frontend, candidate selection, physical compilation and fresh-run execution. +use asap_aware_mapping::{search_workload, DefaultCostModel}; +use asap_frontend_sql::{lower_sql, SqlCatalog}; +use asap_physical_operators::{ + physical_planner::{compile, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + sources::{DataSources, MemorySource}, + values::{Batch, Value}, +}; +use asap_types::{ + post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, + pre_asap::{Column, DataType, QueryExpr, Schema}, + types::AccuracyTarget, +}; +use futures::StreamExt; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; + +/// SQL filtering and grouped aggregation survive logical/physical lowering; +/// rebinding the compiled DAG runs against new data rather than cached results. +#[tokio::test] +async fn sql_filter_grouped_sum_executes_and_rebinds() { + let catalog = SqlCatalog::new().with_table( + "metrics", + Schema::new(vec![ + Column::new("service", DataType::Utf8, false), + Column::new("value", DataType::Float64, true), + ]), + ); + for query in [ + "SELECT service, SUM(value) AS total FROM metrics WHERE value > 1 GROUP BY service", + "SELECT service, SUM(value) AS total FROM metrics GROUP BY service", + ] { + let logical = Rc::new( + lower_sql(query, &catalog, AccuracyTarget::Exact) + .await + .unwrap(), + ); + let space = search_workload(vec![("sql", logical)]); + let selected = space + .global_selection(&DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + let dag = compile_post_asap_dag(&selected).unwrap(); + let scan = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::Scan { .. } + } + ) + }) + .expect("raw SQL scan"); + let schema = Arc::new(scan.output_schema.clone()); + assert!(schema + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + let plan = compile( + &dag, + BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + for multiplier in [1., 2.] { + let rows = [ + ("api", Some(2.)), + ("api", Some(3.)), + ("api", None), + ("batch", Some(4.)), + ("batch", Some(1.)), + ] + .into_iter() + .map(|(service, value)| { + schema + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "value" => { + value.map_or(Value::Null, |value| Value::Float64(value * multiplier)) + } + _ => panic!("unexpected field {field:?}"), + }) + .collect() + }) + .collect(); + let PostAsapOperatorPayload::Fallback { expression } = &scan.payload else { + unreachable!() + }; + let QueryExpr::Scan { source, .. } = expression else { + unreachable!() + }; + let mut sources = DataSources::default(); + sources + .register( + source.clone(), + Arc::new( + MemorySource::new( + schema.clone(), + vec![Batch::try_new(schema.clone(), rows).unwrap()], + ) + .unwrap(), + ), + ) + .unwrap(); + let bound = plan + .instantiate(BTreeMap::from([( + u64::from(scan.id.0), + Box::new(sources.bind(expression).unwrap()) as Source<'_>, + )])) + .unwrap(); + let mut stream = bound + .execute( + plan.roots(), + RunContext::new( + Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(), + ) + .unwrap() + .remove(0); + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push(batch.unwrap()); + } + let mut actual: Vec<_> = batches + .iter() + .flat_map(|batch| batch.rows()) + .map(|row| { + let Value::Utf8(service) = &row[0] else { + panic!("missing service") + }; + let Value::Float64(value) = row[1] else { + panic!("missing sum") + }; + (service.to_string(), value) + }) + .collect(); + actual.sort_by(|a, b| a.0.cmp(&b.0)); + assert_eq!( + actual, + vec![ + ("api".into(), 5. * multiplier), + ( + "batch".into(), + 4. * multiplier + + if query.contains("WHERE") && multiplier == 1. { + 0. + } else { + multiplier + } + ) + ] + ); + } + } +} diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index 186dd04b..41e5f1b3 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -119,52 +119,7 @@ fn dashboard_workload() -> PlanningWorkload { #[test] fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() { let workload = dashboard_workload(); - workload.validate().unwrap(); - - let lowered = lower_promql_workload(&workload, 0) - .expect("valid PromQL workload") - .into_iter() - .next() - .expect("one normalized workload entry"); - let root = Rc::new(lowered); - let strategies = asap_aware_mapping::default_strategies_with(&FullyCostedRuntime); - let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); - let target = Rc::clone(&space.roots[0].1); - let capabilities = SummaryMaintenanceLifecycleCapabilities { - supports_ephemeral: true, - supports_prepared: false, - supports_shared: false, - supports_continuously_maintained: true, - }; - - let selection = global_selection_with_summary_maintenance_lifecycles( - &space, - WorkloadDemand { - workload: &workload.query_workload, - data_workload: workload.data_workload.as_ref(), - entry_indices: &[1], - }, - NOW_MS, - Some(Horizon(100.0)), - capabilities, - &FullyCostedRuntime, - ) - .unwrap(); - let plan = assemble_selected_dag_with_summary_maintenance_lifecycles( - &selection, - &target, - WorkloadDemand::new_with_data( - &workload.query_workload, - workload.data_workload.as_ref().unwrap(), - &[1], - ), - NOW_MS, - Some(Horizon(100.0)), - capabilities, - &FullyCostedRuntime, - ) - .unwrap() - .expect("selected summary plan"); + let plan = selected_plan(&workload); assert!(!plan.selected_raw_recompute); assert_eq!(plan.expected_reads, Some(100.0)); @@ -231,3 +186,686 @@ fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() "continuously_maintained" ); } + +fn selected_plan( + workload: &PlanningWorkload, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + selected_plan_with_model(workload, &FullyCostedRuntime) +} + +fn selected_plan_with_model( + workload: &PlanningWorkload, + model: &dyn CostModel, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + selected_plan_with_horizon(workload, model, Horizon(100.)) +} + +fn selected_plan_with_horizon( + workload: &PlanningWorkload, + model: &dyn CostModel, + horizon: Horizon, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + workload.validate().unwrap(); + + let lowered = lower_promql_workload(workload, 0) + .expect("valid PromQL workload") + .into_iter() + .next() + .expect("one normalized workload entry"); + let root = Rc::new(lowered); + let strategies = asap_aware_mapping::default_strategies_with(model); + let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); + let target = Rc::clone(&space.roots[0].1); + let capabilities = SummaryMaintenanceLifecycleCapabilities { + supports_ephemeral: true, + supports_prepared: false, + supports_shared: false, + supports_continuously_maintained: true, + }; + + let selection = global_selection_with_summary_maintenance_lifecycles( + &space, + WorkloadDemand { + workload: &workload.query_workload, + data_workload: workload.data_workload.as_ref(), + entry_indices: &[1], + }, + NOW_MS, + Some(horizon), + capabilities, + model, + ) + .unwrap(); + assemble_selected_dag_with_summary_maintenance_lifecycles( + &selection, + &target, + WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + NOW_MS, + Some(horizon), + capabilities, + model, + ) + .unwrap() + .expect("selected summary plan") +} + +mod physical_common; + +/// A selected continuous lifecycle supplies a materialization boundary; its +/// maintenance and query DAGs execute the selected KLL computation in fresh runs. +#[test] +fn continuous_lifecycle_compiles_and_executes_spatial_kll() { + use asap_physical_operators::{ + physical_planner::{compile_candidate, InputContract}, + runtime::Scope, + values::{Batch, Value}, + }; + use asap_types::{ + post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, + pre_asap::DataType, + }; + use std::{collections::BTreeMap, sync::Arc}; + let mut workload = dashboard_workload(); + workload.query_workload.query_batch.as_mut().unwrap()[0].query = + Query("quantile(0.99, latency)".into()); + workload.query_workload.repeating_queries.as_mut().unwrap()[0].query = + Query("quantile(0.99, latency)".into()); + let selected = selected_plan(&workload); + assert_eq!( + selected.deployments[0] + .summary_maintenance_lifecycle_guarantee + .as_ref() + .unwrap() + .summary_maintenance_lifecycle, + SummaryMaintenanceLifecycle::ContinuouslyMaintained + ); + let dag = compile_post_asap_dag(&selected.root).unwrap(); + let build = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .unwrap(); + let input = dag + .edges + .iter() + .find(|edge| edge.consumer == build.id) + .unwrap() + .producer; + let raw = dag.nodes.iter().find(|node| node.id == input).unwrap(); + let schema = Arc::new(raw.output_schema.clone()); + let candidate = compile_candidate( + &dag, + BTreeMap::from([(u64::from(input.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &[u64::from(build.id.0)], + ) + .unwrap(); + + // A continuous input without a finite pane boundary cannot implement this + // blocking builder. Retain lifecycle ownership in the candidate payload; + // only the legal bounded request candidate reaches workload pricing. + let mut unbounded = InputContract::bounded(schema.clone()); + unbounded.properties.boundedness = asap_physical_operators::plan::Boundedness::Unbounded; + let rejected = compile_candidate( + &dag, + BTreeMap::from([(u64::from(input.0), unbounded)]), + &[u64::from(dag.root.0)], + &[u64::from(build.id.0)], + ); + assert!(rejected.is_err()); + let request = compile_candidate( + &dag, + BTreeMap::from([(u64::from(input.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &[], + ) + .unwrap(); + let mut priced = 0; + let feedback = asap_physical_operators::physical_planner::select_candidate( + vec![ + rejected.map(|candidate| { + ( + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + candidate, + ) + }), + Ok((SummaryMaintenanceLifecycle::Ephemeral, request)), + ], + |_| { + priced += 1; + Ok(Some( + asap_physical_operators::physical_planner::CandidateCost { + workload_scope: "dashboard".into(), + horizon_seconds: 100., + total_cost: 1000., + }, + )) + }, + ) + .unwrap(); + assert_eq!(priced, 1); + assert_eq!(feedback.candidate.0, SummaryMaintenanceLifecycle::Ephemeral); + for revision in [1, 2] { + let rows = (1..=100) + .map(|value| { + schema + .fields + .iter() + .map(|field| match field.dtype { + SummaryFamilyType::Plain(DataType::Float64) => { + Value::Float64(f64::from(value)) + } + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), + _ => panic!("unexpected field {field:?}"), + }) + .collect() + }) + .collect(); + let raw_batch = Batch::try_new(schema.clone(), rows).unwrap(); + let direct = physical_common::execute( + &feedback.candidate.1.query, + BTreeMap::from([(u64::from(input.0), raw_batch.clone())]), + Scope::Query { + evaluation_time_ms: 300_000, + revision, + }, + ); + let state = physical_common::execute( + candidate.precompute.as_ref().unwrap(), + BTreeMap::from([(u64::from(input.0), raw_batch)]), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 300_000, + revision, + }, + ); + let result = physical_common::execute( + &candidate.query, + BTreeMap::from([(u64::from(build.id.0), state[0][0].clone())]), + Scope::Query { + evaluation_time_ms: 300_000, + revision, + }, + ); + let values: Vec<_> = result[0] + .iter() + .flat_map(|batch| batch.rows()) + .flat_map(|row| row.iter()) + .filter_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .collect(); + let direct_values: Vec<_> = direct[0] + .iter() + .flat_map(|batch| batch.rows()) + .flat_map(|row| row.iter()) + .filter_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .collect(); + assert_eq!( + values, direct_values, + "maintenance and request candidates preserve the same population" + ); + assert_eq!(values.len(), 1); + assert!( + (98. ..=100.).contains(&values[0]), + "p99 rank must reflect the supplied population" + ); + } +} + +struct SlidingPaneModel; +impl CostModel for SlidingPaneModel { + fn raw_query_recompute_total_cost( + &self, + target: &asap_types::pre_asap::QueryExpr, + reads: f64, + ) -> Option { + let _ = (target, reads); + Some(Cost(100_000.0)) + } + fn rank_candidates( + &self, + intent: &AggIntent, + candidates: &[asap_types::post_asap::SketchAlgorithm], + ) -> Vec { + FullyCostedRuntime.rank_candidates(intent, candidates) + } + fn summary_maintenance_lifecycle_cost_inputs( + &self, + summary: &SummaryNode, + ) -> SummaryMaintenanceLifecycleCostInputs { + let mut costs = FullyCostedRuntime.summary_maintenance_lifecycle_cost_inputs(summary); + // Controlled workload evidence makes repeated raw construction more + // expensive than retaining and updating the same temporal population. + costs.build_cost = Some(Cost(1000.)); + costs + } + fn summary_maintenance_capabilities( + &self, + summary: &SummaryNode, + ) -> SummaryMaintenanceCapabilities { + FullyCostedRuntime.summary_maintenance_capabilities(summary) + } + fn complete_summary_candidate_estimate( + &self, + _root: &SummaryNode, + _target: Option<&asap_types::pre_asap::QueryExpr>, + deployments: &[asap_aware_mapping::cost_model::CostedSummaryDeployment<'_>], + _horizon: Option, + _reads: Option, + _accuracy: &[AccuracyTarget], + ) -> Option { + Some(asap_aware_mapping::CompleteSummaryCandidateEstimate { + cost: Cost( + deployments + .iter() + .map(|deployment| deployment.selected_cost.0) + .sum(), + ), + physical_plan_id: Some("bounded-sliding-pane-evidence".into()), + window_frameworks: deployments + .iter() + .map(|deployment| { + (deployment.guarantee.summary_maintenance_lifecycle + == SummaryMaintenanceLifecycle::ContinuouslyMaintained) + .then_some(asap_types::post_asap::SummaryWindowFramework::Sliding) + }) + .collect(), + window_accuracy_guarantee: None, + }) + } +} + +/// Workload and optimizer-selected lifecycle generate both physical DAGs. +/// No computational operators or graph edges are constructed by this fixture. +#[test] +fn selected_temporal_lifecycle_compiles_panes_and_executes() { + use asap_physical_operators::{ + operators::Operator, + physical_planner::{ + compile_temporal_pane_candidate, InputContract, Source, TemporalEntityIdentity, + TemporalPaneMaintenance, + }, + runtime::{Limits, RunContext, Scope}, + summary_kernels::datasketches_kll::DatasketchesKLLAccumulator, + values::{Batch, Value}, + }; + use asap_types::{ + post_asap::{ + compile_post_asap_dag, plan_pane_phase, PostAsapOperatorPayload, SummaryFamilyType, + SummaryWindowFramework, + }, + pre_asap::DataType, + workload::TimestampMs, + }; + use futures::{executor::block_on, StreamExt}; + use std::{collections::BTreeMap, sync::Arc}; + for quantile in [0.5, 0.99] { + let mut workload = dashboard_workload(); + let query = Query(format!( + "quantile_over_time({quantile}, latency{{job=\"api\"}}[5m])" + )); + workload.query_workload.query_batch.as_mut().unwrap()[0].query = query.clone(); + workload.query_workload.repeating_queries.as_mut().unwrap()[0].query = query; + + workload + .data_workload + .as_mut() + .unwrap() + .data_ingestion_interval + .value = Some(DurationMs(60_000)); + workload.query_workload.repeating_queries.as_mut().unwrap()[0].demand = + RepeatedDemand::FixedIntervalAt { + interval: RepetitionInterval(60_000), + evaluation_phase: TimestampMs(300_000), + }; + let plan = selected_plan_with_horizon(&workload, &SlidingPaneModel, Horizon(1000.)); + assert!(!plan.selected_raw_recompute); + assert_eq!(plan.deployments.len(), 1); + let deployment = &plan.deployments[0]; + assert_eq!( + deployment.selected_window_framework, + Some(SummaryWindowFramework::Sliding) + ); + let dag = compile_post_asap_dag(&plan.root).unwrap(); + let build = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .unwrap(); + assert_eq!(build.id, deployment.post_asap_node_id); + let raw = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .unwrap(); + let schema = Arc::new(raw.output_schema.clone()); + let width = workload + .data_workload + .as_ref() + .unwrap() + .data_ingestion_interval + .value + .unwrap() + .0; + let layout = plan_pane_phase( + &workload.query_workload.repeating_queries.as_ref().unwrap()[0].demand, + width, + ) + .unwrap(); + let maintenance = TemporalPaneMaintenance { + summary_node: u64::from(build.id.0), + lifecycle: deployment + .summary_maintenance_lifecycle_guarantee + .clone() + .unwrap(), + framework: deployment.selected_window_framework.clone().unwrap(), + layout, + // The memory source has exactly the declared label columns; a + // schemaless deployment must resolve all entity keys first. + entity_identity: TemporalEntityIdentity::Columns( + schema + .fields + .iter() + .enumerate() + .filter(|(_, field)| field.name == "job") + .map(|(index, _)| index) + .collect(), + ), + }; + let candidate = compile_temporal_pane_candidate( + &dag, + BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &maintenance, + ) + .unwrap(); + assert_eq!(candidate.window_width_ms, 300_000); + assert_eq!(candidate.pane_inputs.len(), 5); + assert_ne!( + candidate.physical.precompute.as_ref().unwrap().roots()[0], + maintenance.summary_node, + "a one-minute pane is not the logical five-minute summary output" + ); + let source_opens = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let mut stored = Vec::new(); + for pane in 0..6 { + let rows = (0..20) + .flat_map(|sample| { + ["api", "batch"] + .into_iter() + .map(move |entity| (sample, entity)) + }) + .map(|(sample, entity)| { + schema + .fields + .iter() + .map(|field| match field.dtype { + SummaryFamilyType::Plain(DataType::Timestamp) => { + Value::Timestamp(pane * 60_000 + (sample + 1) * 3000) + } + SummaryFamilyType::Plain(DataType::Float64) => Value::Float64( + (pane * 20 + sample) as f64 + + if entity == "batch" { 100_000. } else { 0. }, + ), + SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8(entity.into()), + _ => panic!("unexpected raw field {field:?}"), + }) + .collect() + }) + .collect(); + let result = physical_common::execute( + candidate.physical.precompute.as_ref().unwrap(), + BTreeMap::from([( + u64::from(raw.id.0), + Batch::try_new(schema.clone(), rows).unwrap(), + )]), + Scope::Ingestion { + window_start_ms: pane * 60_000, + window_end_ms: (pane + 1) * 60_000, + revision: 1, + }, + ); + let batch = &result[0][0]; + stored.push(batch.clone()); + } + for offset in [0, 1] { + let inputs: BTreeMap<_, _> = candidate + .pane_inputs + .iter() + .enumerate() + .map(|(index, &id)| (id, stored[offset + index].clone())) + .collect(); + let scope = Scope::Query { + evaluation_time_ms: (5 + offset as i64) * 60_000, + revision: 1, + }; + let result = + physical_common::execute(&candidate.physical.query, inputs.clone(), scope.clone()); + let rows = result[0][0].rows(); + assert_eq!(rows.len(), 1); + assert!(rows[0] + .iter() + .any(|value| matches!(value, Value::Utf8(label) if label.as_ref() == "api"))); + let Value::Float64(value) = rows[0][1] else { + panic!("missing p99") + }; + assert!( + (value - ((if quantile == 0.5 { 50 } else { 99 }) + offset * 20) as f64).abs() + <= 1. + ); + assert!( + matches!(rows[0][0], Value::Timestamp(timestamp) if timestamp == (5 + offset as i64) * 60_000) + ); + let sources = |inputs: BTreeMap| -> BTreeMap> { + inputs + .into_iter() + .map(|(id, batch)| { + ( + id, + Box::new(TemporalCountingSource { + operator: Operator::source(batch.schema().clone(), vec![batch]) + .unwrap(), + opens: source_opens.clone(), + }) as Source<'static>, + ) + }) + .collect() + }; + let bound = candidate + .physical + .query + .instantiate(sources(inputs.clone())) + .unwrap(); + let population = block_on( + bound + .execute( + &[candidate.merged_state], + RunContext::new(scope.clone(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0) + .collect::>(), + ); + let Value::Summary { state, .. } = population[0].as_ref().unwrap().rows()[0] + .iter() + .find(|value| matches!(value, Value::Summary { .. })) + .unwrap() + else { + panic!("missing merged state") + }; + assert_eq!( + state + .as_any() + .downcast_ref::() + .unwrap() + .inner + .count(), + 100 + ); + let mut missing = inputs.clone(); + missing.remove(&candidate.pane_inputs[0]); + assert!(candidate + .physical + .query + .instantiate(sources(missing)) + .is_err()); + let mut duplicate = inputs.clone(); + duplicate.insert(candidate.pane_inputs[1], stored[offset].clone()); + let bad = candidate + .physical + .query + .instantiate(sources(duplicate)) + .unwrap(); + let errors = block_on( + bad.execute( + candidate.physical.query.roots(), + RunContext::new(scope.clone(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0) + .collect::>(), + ); + assert!( + errors.iter().any(Result::is_err), + "duplicate pane must not be merged twice" + ); + let mut duplicate_entity = inputs.clone(); + let pane = &stored[offset]; + duplicate_entity.insert( + candidate.pane_inputs[0], + Batch::try_new( + pane.schema().clone(), + vec![pane.rows()[0].clone(), pane.rows()[0].clone()], + ) + .unwrap(), + ); + let bad = candidate + .physical + .query + .instantiate(sources(duplicate_entity)) + .unwrap(); + let errors = block_on( + bad.execute( + candidate.physical.query.roots(), + RunContext::new(scope.clone(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0) + .collect::>(), + ); + assert!( + errors.iter().any(Result::is_err), + "duplicate snapshots within a pane must fail" + ); + let bound = candidate + .physical + .query + .instantiate(sources(inputs)) + .unwrap(); + source_opens.store(0, std::sync::atomic::Ordering::SeqCst); + assert!(bound + .execute( + candidate.physical.query.roots(), + RunContext::new( + Scope::Query { + evaluation_time_ms: 330_000, + revision: 1 + }, + Limits::default() + ) + .unwrap() + ) + .is_err()); + assert_eq!( + source_opens.load(std::sync::atomic::Ordering::SeqCst), + 0, + "invalid phase must fail before opening readers" + ); + } + // A selected framework cannot be silently replaced by physical planning. + let mut wrong_framework = maintenance.clone(); + wrong_framework.framework = SummaryWindowFramework::ExponentialHistogram; + let mut wrong_identity = maintenance.clone(); + wrong_identity.entity_identity = TemporalEntityIdentity::SingleEntity; + let mut unknown_phase = maintenance.clone(); + unknown_phase.layout.pane_origin_ms = None; + let mut partial_panes = maintenance.clone(); + partial_panes.layout.pane_width_ms = 90_000; + let mut wrong_lifecycle = maintenance.clone(); + wrong_lifecycle.lifecycle.summary_maintenance_lifecycle = + SummaryMaintenanceLifecycle::Ephemeral; + for unsupported in [ + wrong_framework, + wrong_identity, + unknown_phase, + partial_panes, + wrong_lifecycle, + ] { + assert!(compile_temporal_pane_candidate( + &dag, + BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &unsupported + ) + .is_err()); + } + } +} + +struct TemporalCountingSource { + operator: asap_physical_operators::operators::Operator, + opens: std::sync::Arc, +} +impl + asap_physical_operators::plan::PhysicalOperator< + asap_physical_operators::values::Batch, + asap_physical_operators::values::Schema, + > for TemporalCountingSource +{ + fn name(&self) -> &str { + "TemporalCountingSource" + } + fn properties( + &self, + inputs: &[asap_physical_operators::plan::PlanProperties], + ) -> asap_physical_operators::plan::PlanProperties { + self.operator.properties(inputs) + } + fn input_schemas(&self) -> Vec { + self.operator.input_schemas() + } + fn output_schema(&self) -> asap_physical_operators::values::Schema { + self.operator.output_schema() + } + fn output_bytes(&self, batch: &asap_physical_operators::values::Batch) -> usize { + batch.bytes() + } + fn start<'a>( + &'a self, + inputs: Vec< + asap_physical_operators::runtime::Input<'a, asap_physical_operators::values::Batch>, + >, + context: asap_physical_operators::runtime::RunContext, + ) -> Result< + asap_physical_operators::runtime::OutputStream<'a, asap_physical_operators::values::Batch>, + asap_physical_operators::Error, + > { + self.opens.fetch_add(1, std::sync::atomic::Ordering::SeqCst); + self.operator.start(inputs, context) + } +} diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md new file mode 100644 index 00000000..72313a9d --- /dev/null +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -0,0 +1,512 @@ +# Physical Planning, Summary Maintenance, and Deployment + +## 1. Architecture + +A Post-ASAP computation is progressively realized through four layers: + +```mermaid +flowchart LR + L["Logical Post-ASAP DAG
What computation?"] + M["Summary Maintenance Lifecycle
How is state maintained?"] + P["Physical DAG(s)
How is it executed?"] + D["Deployment Plan / DAG
How is it instantiated?"] + + L -->|"Summary Maintenance
Candidate Generation"| M + M -->|"Physical Plan
Compiler"| P + P -->|"Deployment Plan
Compiler"| D +``` + +| Layer | Defines | +| --- | --- | +| **Logical Post-ASAP DAG** | Computation semantics | +| **Summary Maintenance Lifecycle** | Build, retention, reuse, and window strategy | +| **Physical DAG(s)** | Supported physical candidates, executable operators and typed input boundaries | +| **Deployment Plan / DAG** | Selected candidate, concrete data/state bindings and operational lifecycle | + +ASAPPlanner owns the first three layers and the shared physical operator +implementation library. Deployment systems such as ASAPQuery and asap-fusion +own deployment compilation and operation. The lifecycle is a planning contract +associated with the logical DAG, not a separate computation IR. + +The Logical Post-ASAP DAG is preceded by the Pre-ASAP DAG (`QueryExpr`), the +language-independent query semantics before summary selection. Both are +logical. Planning builds Post-ASAP `SummaryNode` trees; `compile_post_asap_dag` +exports the selected tree as a `PostAsapDag`, which is the Physical Plan +Compiler's input. Its per-node execution phase (ingestion or query time) is an +initial placement: compilation places ingestion-time nodes in the precompute DAG, +while frontier enumeration proposes alternative materialization splits. Which +layer owns placement is an open design question, deferred to a later change. + +### Candidate generation and deployment selection + +Planner exposes the supported, semantically legal **physical plan candidates**. +It does not discard a computation family or materialization placement merely +because a deployment-independent cost estimate prefers another candidate. +Logical candidates are an internal search stage, not the deployment handoff. + +```text +Query semantics + accuracy and lifecycle requirements + ↓ Planner +Supported Physical DAG candidates + typed inputs/outputs + requirements + ↓ backend +Binding feasibility + runtime statistics + resource limits + ERP + ↓ backend deployment compiler +Selected PrecomputePlan + QueryPlan + StoredOutputReferences +``` + +Planner owns operators, dependencies, sharing, and each candidate's +materialization frontier. The backend rejects candidates it cannot realize and +prices feasible candidates over a comparable workload and time horizon. It binds +the selected candidate; it does not lower the logical computation again, exchange +operators, or move an operator across the selected frontier. A missing quote is +not a zero-cost implementation. ERP evidence cannot authorize an illegal rewrite. + +The candidate inventory must identify its supported search scope and budget. +If a configured exhaustive enumeration exceeds its budget, planning fails +explicitly instead of selecting from an undisclosed partial inventory. Reports +separate unsupported compilation, deployment infeasibility, missing evidence, +and a feasible candidate that loses on cost. Absence is not a cost comparison. + +For `sum by(job)(rate(m[1m]))`, Rate remains per series before grouped Sum. +When lifecycle requirements permit it, a candidate may finalize Rate and Sum +within a bounded precompute run and persist the grouped value. Another may leave +those operators in the query DAG. Storing a value requires its exact evaluation +window, revision, readiness and serving cadence to match the query contract. + +For instant-vector TopK, CMS/CountSketch with a candidate heap requires explicit +series identity and a supported latest-value input protocol. Appending historical +sample values does not preserve instant-vector semantics. Replacement, rank +decrease, expiry, grouping and the required approximation guarantee must be +validated before admitting that physical candidate. + +This is the target ownership contract. A backend path that still reconstructs +operators from logical candidates has not completed this integration. + +### Input semantics and summary semantics + +`source`, `filter`, `grouping` and `window` describe input-data semantics: +where records originate, which records qualify, how they are grouped and which +time interval applies. They are not a complete description of arbitrary summary +computation. In particular, the same four fields can summarize different value +expressions or produce different states. + +| Concern | Required semantic information | +| --- | --- | +| Input computation | Source identities and schemas, filters, joins/transforms and their order, or a reference to the canonical input sub-DAG | +| Values and grouping | Value expressions, item identities and weights where applicable, group keys and types, and operation-defined null/duplicate handling | +| Time | Time column and interpretation, interval bounds, evaluation alignment, and distinction between query range and maintained panes | +| Summary computation | Exact operation or sketch family, algorithm and parameters, and supported build/merge behavior | +| Output | State versus finalized value, output schema/type, and readout parameters when part of the output computation | + +For example, KLL over `latency_seconds` and KLL over `log(latency_seconds)` differ +even with identical source, filter, grouping and window. Likewise, weighted +frequency state needs both item and weight expressions. More complex inputs +must retain their computation DAG; four descriptive fields cannot replace it. + +The canonical selected computation is authoritative. These categories describe +what must be preserved, not a new flat IR or a second expression language. +Operator-defined behavior should be referenced through its canonical contract, +not independently configured in deployment metadata. Unsupported or unresolved +semantics cannot be treated as compatible. + +Logical planning defines the semantics; physical compilation realizes them as +operators and typed boundaries. Deployment binds concrete readers and state +records that satisfy those requirements. A stored summary definition records or +references the relevant semantics for compatibility checks. Matching a definition +alone does not establish actual window coverage, revision compatibility or +readiness; those require runtime checks. Physical location, encoding, scheduling +and retention are separate execution/deployment contracts. + +Persisted semantic identity, its wire format and any tenant or dataset binding +belong to the deployment. Planner provides the typed `PostAsapDag` that a +deployment canonicalizes; it does not define a stored-definition format. + +### Running example + +Suppose p50 and p99 are requested over the same latency samples in a five-minute window, +and one Planner candidate uses KLL with `k=200`. Assume query windows align with one-minute +pane boundaries and that the selected parameters satisfy the required guarantees. +Operator names below are illustrative; the example defines the design, not a +claim that the entire deployment integration is implemented. + +The data source identifies where samples come from. Filters, grouping and the +window determine which samples enter each summary. Here `pane_duration: 1m` +means each stored pane covers one minute; the query range is five minutes. +Neither duration specifies how often maintenance runs or how long state is kept. + +The example evolves through the architecture as follows: + +```text +1. Logical Post-ASAP DAG + +raw latency + ↓ +KLLBuild(k=200) + ↓ +KLLMerge + ┌─┴─────┐ + ↓ ↓ + p50 p99 + + │ + │ Summary Maintenance Candidate Generation + ▼ + +2. Summary Maintenance Lifecycle + +KLLBuild(k=200) + strategy = continuously maintain + window = 1-minute panes + reuse = p50 + p99 + query = merge panes covering requested aligned 5 minutes + + │ + │ Physical Plan Compiler + ▼ + +3. Physical DAGs + +Precompute DAG: +RawInput + ↓ +NativeKllBuild(k=200) + ↓ +KllStateOutput + +Query DAG: +InputSlot[5 panes] + ↓ +NativeKllMerge(k=200) + ┌─┴────────┐ + ↓ ↓ + NativeP50 NativeP99 + + │ + │ Deployment Plan Compiler + ▼ + +4. Deployment Plan / DAG + +Precompute: +OTLP latency source + ↓ +run KLL build over each complete 1-minute input pane + ↓ +store as latency-kll-1m/ + +Query: +resolve five latency-kll-1m states + ↓ +execute query Physical DAG + ↓ +return p50 / p99 +``` + +Each stage adds a different class of decision while preserving the preceding +contracts. Here, continuous maintenance means recurring production of pane state; +the bounded build DAG does not itself implement an unbounded streaming window. + +## 2. Logical Post-ASAP DAG → Summary Maintenance Lifecycle + +The **Logical Post-ASAP DAG** defines computation semantics: + +```text +Scan(latency) + ↓ +KLLBuild(k=200) + ↓ +KLLMerge + ┌─┴────────────┐ + ↓ ↓ +Quantile(.5) Quantile(.99) +``` + +It establishes that KLL with `k=200` is used and that the merge is shared by the +two readouts. It does not determine when KLL states are built or retained. + +**Summary Maintenance Candidate Generation** enumerates legal lifecycle choices +using workload demand, window/freshness requirements and supported physical +implementations. Backend selection uses runtime feasibility and cost after +physical compilation. The following example follows one candidate. + +For the running example, assume it selects: + +```text +producer: KLLBuild(k=200) + +strategy: + continuously maintain + +window realization: + 1-minute panes + +query requirement: + combine panes covering the requested aligned 5-minute range + +reuse: + one merged state serves p50 and p99 +``` + +This produces the **Summary Maintenance Lifecycle**. + +The lifecycle specifies how the selected logical summary should be maintained, +but not its concrete operator implementation or storage location. + +Physical feasibility may feed back into selection. For example, if the required +pane-based maintenance cannot be implemented, this lifecycle candidate cannot be +selected. One-minute panes alone also cannot cover an arbitrarily phased query +window; that requires supported boundary handling or a different candidate. + +## 3. Summary Maintenance Lifecycle → Physical DAG + +The **Physical Plan Compiler** consumes both computation semantics and maintenance +requirements: + +```text +Logical Post-ASAP DAG (PostAsapDag) ++ Summary Maintenance Lifecycle ++ physical capabilities + ↓ +Physical Plan Compiler + ↓ +Physical DAG(s) +``` + +For the running example, the lifecycle creates two execution boundaries. + +These two halves are named as `PhysicalCandidate` names them, `precompute` +and `query`. *Maintenance* stays the lifecycle's word (section 2): it covers +how state is built, retained, reused and scheduled. A precompute DAG is the +physical object that a maintenance lifecycle compiles to, so reusing +*maintenance* for it collapses two layers that the crates keep apart: +`asap-aware-mapping::summary_maintenance_*` owns the lifecycle, and +`asap-physical-operators::physical_planner` owns the DAGs. + +### Precompute Physical DAG + +```text +RawInputSlot( + window = 1m, + bounded = true +) + ↓ +NativeKllBuild(k=200) + ↓ +KllStateOutput(k=200) +``` + +This DAG implements construction of each maintained one-minute pane. Its input +contract requires all input samples matching the source, filters and group within that pane; the deployment supplies that +bounded input from its source integration. + +### Query Physical DAG + +```text +InputSlot( + k = 200, + coverage = requested aligned 5m +) + ↓ +NativeKllMerge(k=200) + ┌─┴──────────────────┐ + ↓ ↓ +NativeQuantile(.50) NativeQuantile(.99) +``` + +The Physical Plan Compiler chooses `NativeKllBuild`, `NativeKllMerge`, and the +physical quantile implementations, validates state compatibility, and preserves +the shared merge. It also resolves expressions, schemas, ordered dependencies +and execution properties. + +The resulting Physical DAGs know that compatible KLL states are required, but +do not know where those states are stored. + +For example: + +```text +InputSlot +``` + +is physical, while: + +```text +s3://.../latency-kll/12:01 +``` + +is deployment-specific. Placement and scheduling also remain outside the Physical +DAG. If the required behavior cannot be realized, physical compilation fails. + +### Physical candidates include precompute computation + +Materialization frontiers are Planner decisions. A candidate records both the +precompute Physical DAG and the query Physical DAG, with typed outputs connecting +them. The deployment compiler binds those outputs; it does not move operators. + +For `sum by(job)(rate(m[1m]))`, legal physical candidates can include: + +```text +Candidate A: + precompute: compatible per-series counter states → per-series Rate + materialized output: per-series rate values for window/evaluation/revision + query: stored per-series rate values → grouped Sum + +Candidate B: + precompute: compatible per-series counter states → per-series Rate → grouped Sum + materialized output: grouped values for window/evaluation/revision + query: stored grouped values → result +``` + +Both preserve reset-aware Rate before Sum. Summing raw counters before Rate is +not equivalent. The counter-state build may be another precompute DAG; typed +state inputs do not imply that a deployment can construct or bind those states. + +The shared library exposes `physical_planner::compile_candidates(...)` to lower +explicit frontier candidates to `PhysicalCandidate { precompute, query, +materialized_outputs }`. `select_candidate(...)` accepts deployment feasibility +and scoped complete-workload costs and chooses the lowest-cost feasible +candidate. Costs must describe the same workload and planning horizon; missing +feasibility is rejected before pricing. The optimizer supplies candidate +frontiers and cost evidence, including updates, retention, recurrence and sharing. +`enumerate_frontiers` constructs bounded, reachable antichain frontiers above explicit input boundaries, including query-only and fully precomputed results. It fails explicitly when the candidate budget is exceeded. Maintenance selection must still reject frontiers that violate window, freshness, or reuse requirements; deployment feasibility is checked before pricing. + +Physical compilation opens no readers. Bounded precompute outputs become typed +query inputs. Their source, filters, grouping, build window, evaluation time, readiness and +revision contracts must accompany the selected lifecycle and be checked during +deployment binding. Type compatibility alone does not establish reuse legality. + +The Planner integration test executes both candidates through the shared runtime +and reverses the selected frontier with two controlled cost fixtures. It also +rejects shadowed/duplicate boundaries and incomparable planning horizons. This +establishes Planner capability; it does not establish that ASAPQuery currently +supports persisting every scalar/result-output frontier. + +## 4. Physical DAG → Deployment Plan / DAG + +The **Deployment Plan Compiler** binds the Physical DAGs to the concrete deployment: + +```text +Physical DAGs ++ Summary Maintenance Lifecycle ++ deployment catalog/state ++ sources/materializations ++ operational policy + ↓ +Deployment Plan Compiler + ↓ +Deployment Plan / DAG +``` + +For the precompute DAG, it may produce: + +```text +Source: + RawInputSlot + → complete bounded panes from the OTLP latency source + +Schedule: + each 1-minute pane, once its completion requirements are met + +Execution: + RawInput → NativeKllBuild(k=200) + +Output: + KllStateOutput + → latency-kll-1m/ +``` + +For a query over `(12:00, 12:05]`, its input-binding rule resolves: + +```text +InputSlot[5 panes] + ├── latency-kll-1m/(12:00,12:01] + ├── latency-kll-1m/(12:01,12:02] + ├── latency-kll-1m/(12:02,12:03] + ├── latency-kll-1m/(12:03,12:04] + └── latency-kll-1m/(12:04,12:05] + ↓ + Query Physical DAG + ↓ + p50, p99 +``` + +The Deployment Plan Compiler establishes bindings and checks that their contracts +satisfy the physical inputs and selected lifecycle, including KLL parameters, +source, filters, grouping, window coverage and revision scope. The deployment engine +resolves request-specific states and checks their actual coverage, revisions and +readiness at execution time. A compiled plan cannot establish future readiness. + +The compiler does not replace `NativeKllMerge`, choose another sketch, or decide +to maintain different windows. Such changes require replanning. A Deployment +Plan / DAG is an operational instantiation, not another computation IR. + +## 5. Responsibility Boundary + +The complete example makes the ownership boundary explicit: + +| Stage | KLL example decision | +| --- | --- | +| **Logical Post-ASAP DAG** | Use `KLL(k=200)` with shared merge for p50/p99 | +| **Summary Maintenance Candidate Generation** | Maintain 1-minute panes and reuse them for aligned five-minute queries | +| **Summary Maintenance Lifecycle** | Record pane/window/freshness/reuse requirements | +| **Physical Plan Compiler** | Lower to native KLL build, merge, and readout operators | +| **Physical DAG** | Define precompute and query DAGs with typed input/output boundaries | +| **Deployment Plan Compiler** | Bind raw input and KLL state slots to concrete sources/materializations | +| **Deployment Plan / DAG** | Specify maintenance schedules, stored-pane resolution and query execution | + +```text +Logical: + "Use KLL for p50/p99." + +Lifecycle: + "Maintain reusable 1-minute KLL panes." + +Physical: + "Execute NativeKllBuild and + NativeKllMerge → {p50, p99}." + +Deployment: + "Read OTLP here, store panes here, + and bind these five panes for this aligned query." +``` + +The deployment engine executes the bound Physical DAGs through ASAPPlanner's +shared physical operator implementation library, `asap-physical-operators`, and +its DAG runtime. The merge executes once per run for both consumers. Execution +does not introduce additional planning decisions. + +Each maintained pane contributes its input samples once. A replacement snapshot +replaces that pane's state; query merging must not count both the old and new +snapshots as separate inputs. + +## 6. Executable acceptance coverage + +The tests cover optimizer-selected lifecycle execution and automatic temporal +pane compilation, alongside independent operator/runtime fixtures: + +| Test | Contract exercised | +| --- | --- | +| `summary_maintenance_lifecycle_e2e::selected_temporal_lifecycle_compiles_panes_and_executes` | PromQL p50/p99 workloads → selected continuous lifecycle and Sliding framework → automatically generated precompute/query DAGs → real codec round-trip → adjacent aligned windows; checks filters, entity identity, sample counts, missing/duplicate panes and phase rejection before opening readers | +| `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled precompute/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | +| `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute precompute DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | +| `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | +| `precompute_candidates::grouped_rate_can_be_materialized_before_or_after_grouped_sum` | Cost changes select different legal precompute frontiers; both selected candidates execute with the same reset-sensitive result; uncompilable candidates are not priced | +| `sql_to_physical::sql_filter_grouped_sum_executes_and_rebinds` | SQL text → candidate search → physical compilation → shared Scan predicates and grouped summary execution; NULL samples are ignored and fresh bindings produce new results | + +`physical_planner::compile_temporal_pane_candidate` consumes the logical DAG, +selected lifecycle/framework and a generic pane/entity input contract. It +generates pane construction, scan predicates, ordered state slots, a shared +merge, quantile readouts and run-scoped timestamps. A physical pane output has +its own identity: one minute of state cannot masquerade as the logical +five-minute summary. The returned candidate retains the maintenance contract. + +This initial realization supports bounded, complete KLL panes with known phase +and resolved entity identity, for Sliding windows or a single Tumbling window. +Source capability evidence must declare all entity keys or isolate one entity; +usage-derived PromQL columns alone cannot establish that identity. Partial edge +panes, exponential histograms and cross-run delta accumulation require further +physical candidates and are rejected by this entry point. + +Physical execution checks pane timestamps and duplicate entity states. Concrete +stored identity, revisions, readiness and complete coverage of required input samples remain +deployment responsibilities. Real storage and HTTP execution belong to +deployment-repository E2E tests. diff --git a/docs/develop_docs/native-promql-inputs.md b/docs/develop_docs/native-promql-inputs.md new file mode 100644 index 00000000..aa1a52b6 --- /dev/null +++ b/docs/develop_docs/native-promql-inputs.md @@ -0,0 +1,55 @@ +# Native PromQL source rows + +Audience: source-adapter and physical-executor developers. + +A PromQL query only names some labels. Those columns cannot establish series +identity for Rate or TopK: two series with the same `job` may have different +unreferenced instance labels. + +`physical_planner::promql_rows::with_series_identity` resolves supported unary +PromQL computations to a bounded row representation before candidate search. +It appends `$promql_series_identity`, a non-null UTF-8 column containing the +canonical JSON encoding of the full label map. The name cannot collide with a +legal PromQL label. The resulting schema is closed over physical columns; the +label map remains dynamic and is not restricted to labels named in the query. +This realization rejects unsupported label rewriting, implicit vector matching, +and `without` operations rather than dropping hidden labels. + +Source adapters construct batches with `series_row`. Named label columns are +projections of the same complete identity; absent named labels project to empty +strings. `decode_series_identity` restores all labels on result conversion and +rejects noncanonical encodings. A query adapter must still apply the selected +operator's metric-name/result-label rules. Source selection, complete window +coverage and revision admission remain deployment responsibilities. + +Planner's maintained-population candidate recognizes this explicit identity +representation. Its TopK readout compiles automatically to `CurrentSeries`, +`Sort`, and `Limit`; deployment supplies the raw boundary or an already maintained +population boundary. Compilation does not open either source. + +The native `CurrentSeries` operator selects the latest sample per complete +identity in `(evaluation_time - lookback, evaluation_time]`. It removes stale +markers after selecting the latest sample, so an older value cannot reappear. +It rejects conflicting values at one series timestamp and emits the evaluation +timestamp. Each run builds a new snapshot; decreased values and expired series +cannot retain earlier heap weights. It reserves workspace and observes the +run's cancellation and byte budget. Precompute scopes must match the declared +lookback before any input is polled. + +CMS/CountSketch heap operators can consume this snapshot. CMS still requires +nonnegative weights; legal approximate TopK admission still requires the +Planner's accuracy/membership evidence. Executing a heap does not establish +that its result satisfies a query's accuracy requirements. + +Tests cover open-label Rate → CMS/CountSketch heaps, hidden-label round trips, +reset and zero-rate cases, snapshot replacement/decrease/expiry/staleness, +serialized physical recovery, and resource rejection. These are shared-library +tests, not proof of Backend candidate selection or durable deployment execution. + +Spatial heap candidates use the same complete series identity. Planner's +`current_series_topk_candidates` explores a CountSketch-with-heap realization +of canonical Sort/Limit under an explicit accuracy target. The physical graph +selects the latest eligible samples before building a fresh heap. A maintained +population boundary can supply that snapshot directly. Arbitrary signed metric +values do not authorize CMS; counter Rate's non-negative proof is separate. +These candidates still require membership/score evidence for deployment admission. From ee9e0b15cafb9c39dfb3bc724eaa1ee3ed8da39b Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:22:53 +0000 Subject: [PATCH 04/59] refactor!: move pane construction out of the physical layer Pane geometry, pane population checks and per-pane scheduling are deployment concerns. The physical layer keeps the computation the deployment binds: per-input summary build, union, shared merge and readout. - Remove `physical_planner::compile_temporal_pane_candidate` and its `TemporalPaneMaintenance`/`TemporalPaneCandidate`/`TemporalEntityIdentity` contract. - Remove the `PaneInput` operator; `ScopeTimestamp` remains as a general operator in `operators/scope_timestamp`. - Drop the `selected_temporal_lifecycle_compiles_panes_and_executes` E2E test and update the crate README and design doc. Co-Authored-By: Claude Opus 5.5 --- crates/asap-physical-operators/README.md | 15 +- .../src/operators/mod.rs | 18 +- .../src/operators/panes.rs | 228 --------- .../src/operators/scope_timestamp.rs | 91 ++++ .../src/operators/unchecked.rs | 5 - .../src/physical_planner/mod.rs | 6 - .../src/physical_planner/temporal_panes.rs | 331 ------------- .../summary_maintenance_lifecycle_e2e.rs | 443 ------------------ .../physical-planning-and-deployment.md | 27 +- 9 files changed, 103 insertions(+), 1061 deletions(-) delete mode 100644 crates/asap-physical-operators/src/operators/panes.rs create mode 100644 crates/asap-physical-operators/src/operators/scope_timestamp.rs delete mode 100644 crates/asap-physical-operators/src/physical_planner/temporal_panes.rs diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index 5111d316..5f638465 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -116,15 +116,6 @@ the shared runtime with independent per-run state. Window coverage, revision and maintenance-policy admission remain deployment/planning contracts; this compiler does not discover storage or silently change a selected maintenance strategy. -`physical_planner::compile_temporal_pane_candidate` lowers a selected continuous -KLL lifecycle and Sliding/Tumbling framework into maintenance and query DAGs. -`TemporalPaneMaintenance` supplies pane geometry and a resolved complete entity -identity contract. The compiler inserts population guards, scan predicates, -pane construction, ordered state slots, a shared merge and quantile readouts. -Pane outputs have distinct physical identities from the logical whole-window -summary, and the returned candidate retains the maintenance contract for binding. -Each run checks phase, pane timestamps and duplicate entity states. The initial -realization uses complete bounded snapshots; partial edges, exponential -histograms and cross-run delta accumulation are unsupported. Storage identities, -revision selection, completeness/readiness evidence and scheduling stay with -deployment. +Pane construction and geometry belong to deployment. A deployment runs the +precompute DAG once per pane it constructs and binds the selected pane states to +query input slots; the query DAG merges and reads them out as computation. diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index 0f20fa3d..9ae0772c 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -21,8 +21,8 @@ mod current_series; mod filter; mod joins; mod limit; -mod panes; mod projection; +mod scope_timestamp; mod sort; mod source; mod summary; @@ -40,11 +40,6 @@ enum Kind { value: Value, dtype: DataType, }, - PaneInput { - coordinate: usize, - layout: planner_types::post_asap::PaneLayout, - offset_ms: Option, - }, ScopeTimestamp { columns: Vec>, }, @@ -260,10 +255,7 @@ impl PhysicalOperator for Operator { }; PlanProperties { boundedness, - emission: if matches!( - self.kind, - Kind::PaneInput { .. } | Kind::ScopeTimestamp { .. } - ) { + emission: if matches!(self.kind, Kind::ScopeTimestamp { .. }) { inputs .first() .map_or(Emission::Unknown, |input| input.emission) @@ -279,7 +271,6 @@ impl PhysicalOperator for Operator { match self.kind { Kind::Source(_) => "Source", Kind::Constant { .. } => "Constant", - Kind::PaneInput { .. } => "PaneInput", Kind::ScopeTimestamp { .. } => "ScopeTimestamp", Kind::Union => "Union", Kind::CurrentSeries { .. } => "CurrentSeries", @@ -303,7 +294,6 @@ impl PhysicalOperator for Operator { } } fn validate_context(&self, context: &RunContext) -> Result<(), Error> { - panes::validate_context(self, context)?; current_series::validate_context(self, context)?; self.readout_range(context).map(|_| ()) } @@ -332,9 +322,7 @@ impl PhysicalOperator for Operator { } Kind::Project(_) => projection::execute(self, inputs, context), Kind::CurrentSeries { .. } => current_series::execute(self, inputs, context), - Kind::PaneInput { .. } | Kind::ScopeTimestamp { .. } => { - panes::execute(self, inputs, context) - } + Kind::ScopeTimestamp { .. } => scope_timestamp::execute(self, inputs, context), Kind::Filter(_) => filter::execute(self, inputs, context), Kind::Limit { .. } => limit::execute(self, inputs, context), Kind::Sort { .. } => sort::execute(self, inputs, context), diff --git a/crates/asap-physical-operators/src/operators/panes.rs b/crates/asap-physical-operators/src/operators/panes.rs deleted file mode 100644 index 4748db09..00000000 --- a/crates/asap-physical-operators/src/operators/panes.rs +++ /dev/null @@ -1,228 +0,0 @@ -//! Run-scoped pane population checks and timestamp restoration after reduction. -use super::*; -use crate::runtime::Scope; -use planner_types::post_asap::{validate_pane_coverage, PaneLayout, WindowEdgeCoverage}; - -impl Operator { - pub(crate) fn pane_input( - input: Schema, - coordinate: usize, - layout: PaneLayout, - offset_ms: Option, - ) -> Result { - if plain(&input, coordinate)? != (&DataType::Timestamp, false) { - return Err(invalid("pane input requires a non-null timestamp")); - } - if layout.pane_width_ms > i64::MAX as u64 { - return Err(invalid("pane width exceeds timestamp range")); - } - validate_pane_coverage( - &layout, - layout.pane_origin_ms, - &WindowEdgeCoverage::PaneAligned, - ) - .map_err(|error| Error::Invalid(format!("invalid pane layout: {error:?}")))?; - if offset_ms.is_some_and(|offset| offset < 0) { - return Err(invalid("negative pane offset")); - } - Ok(Self { - kind: Kind::PaneInput { - coordinate, - layout, - offset_ms, - }, - inputs: vec![input.clone()], - output: input, - }) - } - - pub(crate) fn scope_timestamp(input: Schema, output: Schema) -> Result { - crate::values::validate_schema(&output)?; - let coordinate = output - .time_index - .ok_or_else(|| invalid("temporal output requires a time index"))?; - if plain(&output, coordinate)? != (&DataType::Timestamp, false) { - return Err(invalid("temporal output requires a non-null timestamp")); - } - let mut columns = Vec::new(); - let mut used = std::collections::BTreeSet::new(); - for (index, field) in output.fields.iter().enumerate() { - if index == coordinate { - columns.push(None); - continue; - } - let matches: Vec<_> = input - .fields - .iter() - .enumerate() - .filter(|(_, candidate)| { - candidate.dtype == field.dtype - && candidate.nullable == field.nullable - && (candidate.name == field.name - || !matches!(field.dtype, SummaryFamilyType::Plain(_))) - }) - .map(|(index, _)| index) - .collect(); - let [column] = matches.as_slice() else { - return Err(invalid("temporal output column missing or ambiguous")); - }; - if !used.insert(*column) { - return Err(invalid("temporal output repeats an input column")); - } - columns.push(Some(*column)); - } - if used.len() != input.fields.len() { - return Err(invalid("temporal output drops an input column")); - } - Ok(Self { - kind: Kind::ScopeTimestamp { columns }, - inputs: vec![input], - output, - }) - } -} - -pub(super) fn validate_context(operator: &Operator, context: &RunContext) -> Result<(), Error> { - let Kind::PaneInput { - layout, offset_ms, .. - } = &operator.kind - else { - return Ok(()); - }; - let end = match (&context.scope, offset_ms) { - ( - Scope::Ingestion { - window_start_ms, - window_end_ms, - .. - }, - None, - ) => { - if window_end_ms.checked_sub(*window_start_ms) != Some(layout.pane_width_ms as i64) { - return Err(invalid("maintenance run must cover exactly one pane")); - } - *window_end_ms - } - ( - Scope::Query { - evaluation_time_ms, .. - }, - Some(offset), - ) => { - let end = evaluation_time_ms - .checked_sub(*offset) - .ok_or_else(|| invalid("query pane timestamp overflows"))?; - end.checked_sub(layout.pane_width_ms as i64) - .ok_or_else(|| invalid("query pane start overflows"))?; - end - } - _ => return Err(invalid("pane operator received the wrong execution scope")), - }; - validate_pane_coverage(layout, Some(end), &WindowEdgeCoverage::PaneAligned).map_err(|error| { - Error::Invalid(format!( - "query requires aligned panes or boundary residuals: {error:?}" - )) - }) -} - -pub(super) fn execute<'a>( - operator: &'a Operator, - mut inputs: Vec>, - context: RunContext, -) -> Result, Error> { - validate_context(operator, &context)?; - let input = inputs.pop().ok_or_else(|| invalid("pane input missing"))?; - let output = operator.output.clone(); - let mut seen = std::collections::BTreeSet::new(); - let mut memory = context.reserve(0)?; - let mut key_bytes = 0; - Ok(input - .map(move |batch| { - if context.is_cancelled() { - return Err(Error::Cancelled); - } - let batch = batch?; - match &operator.kind { - Kind::PaneInput { - coordinate, - offset_ms, - .. - } => { - let groups: Vec<_> = output - .fields - .iter() - .enumerate() - .filter(|(index, field)| { - *index != *coordinate - && matches!(field.dtype, SummaryFamilyType::Plain(_)) - }) - .map(|(index, _)| index) - .collect(); - for row in batch.rows() { - let Value::Timestamp(timestamp) = row[*coordinate] else { - return Err(invalid("pane timestamp type mismatch")); - }; - match (&context.scope, offset_ms) { - ( - Scope::Ingestion { - window_start_ms, - window_end_ms, - .. - }, - None, - ) if timestamp > *window_start_ms && timestamp <= *window_end_ms => {} - ( - Scope::Query { - evaluation_time_ms, .. - }, - Some(offset), - ) if timestamp - == evaluation_time_ms - .checked_sub(*offset) - .ok_or_else(|| invalid("pane timestamp overflows"))? => - { - let key = group_key(row, &groups)?; - if seen.contains(&key) { - return Err(invalid("duplicate entity state within a pane")); - } - key_bytes += key.iter().map(Vec::len).sum::() - + key.len() * std::mem::size_of::>() - + 64; - memory.resize(key_bytes)?; - seen.insert(key); - } - _ => { - return Err(invalid("input population differs from required pane")) - } - } - } - Ok(batch.value().clone()) - } - Kind::ScopeTimestamp { columns } => { - let timestamp = match context.scope { - Scope::Ingestion { window_end_ms, .. } => window_end_ms, - Scope::Query { - evaluation_time_ms, .. - } => evaluation_time_ms, - }; - let rows = batch - .rows() - .iter() - .map(|row| { - columns - .iter() - .map(|column| { - column.map_or(Value::Timestamp(timestamp), |column| { - row[column].clone() - }) - }) - .collect() - }) - .collect(); - Batch::try_new(output.clone(), rows) - } - _ => unreachable!(), - } - }) - .boxed_local()) -} diff --git a/crates/asap-physical-operators/src/operators/scope_timestamp.rs b/crates/asap-physical-operators/src/operators/scope_timestamp.rs new file mode 100644 index 00000000..3561d6a6 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/scope_timestamp.rs @@ -0,0 +1,91 @@ +//! Run-scoped timestamp restoration after reduction. +use super::*; +use crate::runtime::Scope; + +impl Operator { + pub(crate) fn scope_timestamp(input: Schema, output: Schema) -> Result { + crate::values::validate_schema(&output)?; + let coordinate = output + .time_index + .ok_or_else(|| invalid("temporal output requires a time index"))?; + if plain(&output, coordinate)? != (&DataType::Timestamp, false) { + return Err(invalid("temporal output requires a non-null timestamp")); + } + let mut columns = Vec::new(); + let mut used = std::collections::BTreeSet::new(); + for (index, field) in output.fields.iter().enumerate() { + if index == coordinate { + columns.push(None); + continue; + } + let matches: Vec<_> = input + .fields + .iter() + .enumerate() + .filter(|(_, candidate)| { + candidate.dtype == field.dtype + && candidate.nullable == field.nullable + && (candidate.name == field.name + || !matches!(field.dtype, SummaryFamilyType::Plain(_))) + }) + .map(|(index, _)| index) + .collect(); + let [column] = matches.as_slice() else { + return Err(invalid("temporal output column missing or ambiguous")); + }; + if !used.insert(*column) { + return Err(invalid("temporal output repeats an input column")); + } + columns.push(Some(*column)); + } + if used.len() != input.fields.len() { + return Err(invalid("temporal output drops an input column")); + } + Ok(Self { + kind: Kind::ScopeTimestamp { columns }, + inputs: vec![input], + output, + }) + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::ScopeTimestamp { columns } = &operator.kind else { + return Err(invalid("scope timestamp operator required")); + }; + let input = inputs + .pop() + .ok_or_else(|| invalid("scope timestamp input missing"))?; + let output = operator.output.clone(); + let timestamp = match context.scope { + Scope::Ingestion { window_end_ms, .. } => window_end_ms, + Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + }; + Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + let rows = batch + .rows() + .iter() + .map(|row| { + columns + .iter() + .map(|column| { + column.map_or(Value::Timestamp(timestamp), |column| row[column].clone()) + }) + .collect() + }) + .collect(); + Batch::try_new(output.clone(), rows) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/unchecked.rs b/crates/asap-physical-operators/src/operators/unchecked.rs index 2ecbaabe..f927a4e4 100644 --- a/crates/asap-physical-operators/src/operators/unchecked.rs +++ b/crates/asap-physical-operators/src/operators/unchecked.rs @@ -30,11 +30,6 @@ impl TryFrom for Operator { let op = match kind { Kind::Source(_) => return Err(invalid("physical plans cannot serialize live sources")), Kind::Constant { value, dtype } => Operator::scalar(value, dtype)?, - Kind::PaneInput { - coordinate, - layout, - offset_ms, - } => Operator::pane_input(input(0)?, coordinate, layout, offset_ms)?, Kind::ScopeTimestamp { .. } => Operator::scope_timestamp(input(0)?, output.clone())?, Kind::Union => Operator::union(input(0)?, inputs.len())?, Kind::CurrentSeries { diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 142a2c25..112e1715 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -40,12 +40,6 @@ pub use candidates::{ CandidateSelection, PhysicalCandidate, }; -mod temporal_panes; -pub use temporal_panes::{ - compile_temporal_pane_candidate, TemporalEntityIdentity, TemporalPaneCandidate, - TemporalPaneMaintenance, -}; - mod compiled; pub use compiled::{CompiledPhysicalDag, InputContract}; diff --git a/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs b/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs deleted file mode 100644 index d535c777..00000000 --- a/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs +++ /dev/null @@ -1,331 +0,0 @@ -//! Lower a selected temporal maintenance contract; deployment supplies readers. -use super::*; -use planner_types::post_asap::{ - EvaluationSchedule, OutputRepresentation, PaneLayout, SketchAlgorithm, - SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, - SummaryWindowFramework, -}; - -/// Resolved source identity, supplied with physical capability evidence. -/// A schemaless PromQL projection cannot establish the complete label set. -#[derive(Clone, Debug)] -pub enum TemporalEntityIdentity { - /// The input resolver guarantees that the slot contains one entity. - SingleEntity, - /// All entity keys are represented by these columns; there are no hidden - /// labels distinguishing two rows with the same key. - Columns(Vec), -} - -/// Planner-selected lifecycle/window requirements for one temporal producer. -/// Pane geometry is semantic input, not a storage identity or scheduling policy. -#[derive(Clone, Debug)] -pub struct TemporalPaneMaintenance { - pub summary_node: NodeId, - pub lifecycle: SummaryMaintenanceLifecycleGuarantee, - pub framework: SummaryWindowFramework, - pub layout: PaneLayout, - pub entity_identity: TemporalEntityIdentity, -} - -/// Generated precompute and query computation. `pane_inputs` is ordered from -/// the oldest complete pane to the newest; each run checks actual timestamps. -#[derive(Clone)] -pub struct TemporalPaneCandidate { - pub physical: PhysicalCandidate, - pub maintenance: TemporalPaneMaintenance, - pub pane_inputs: Vec, - pub merged_state: NodeId, - pub window_width_ms: u64, -} - -/// Compile bounded pane construction and a shared pane merge for temporal KLL -/// quantile roots. The selected contract remains authoritative; unsupported -/// lifecycle/framework/operator shapes fail rather than being substituted. -/// This initial realization consumes complete pane populations and emits full -/// state snapshots. Cross-run delta accumulation belongs to other candidates. -pub fn compile_temporal_pane_candidate( - dag: &PostAsapDag, - inputs: BTreeMap, - roots: &[NodeId], - maintenance: &TemporalPaneMaintenance, -) -> Result { - dag.validate().map_err(|error| invalid(error.to_string()))?; - if maintenance.lifecycle.summary_maintenance_lifecycle - != SummaryMaintenanceLifecycle::ContinuouslyMaintained - || maintenance.lifecycle.summary_maintenance_mode != SummaryMaintenanceMode::Incremental - || maintenance.lifecycle.evaluation_schedule != EvaluationSchedule::PerUpdate - || maintenance.lifecycle.output_representation != OutputRepresentation::SummaryState - { - return Err(invalid( - "pane candidate requires continuous incremental summary maintenance", - )); - } - let build = dag - .nodes - .iter() - .find(|node| u64::from(node.id.0) == maintenance.summary_node) - .ok_or_else(|| invalid("unknown maintained producer"))?; - let Payload::SummaryAgg { - family, - input: update, - reduction: PlannerReduction::PerEntity, - grouping, - } = &build.payload - else { - return Err(invalid( - "pane candidate requires a temporal per-entity summary", - )); - }; - if !matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll) - || update.item.is_some() - { - return Err(invalid("pane candidate supports unkeyed temporal KLL only")); - } - crate::capability::validate_summary_kernel(family, update, grouping).map_err(Error::Invalid)?; - let dependencies: Vec<_> = dag - .edges - .iter() - .filter(|edge| edge.consumer == build.id) - .map(|edge| edge.producer) - .collect(); - let [raw_id] = dependencies.as_slice() else { - return Err(invalid("temporal producer requires one raw input")); - }; - let raw = dag - .nodes - .iter() - .find(|node| node.id == *raw_id) - .ok_or_else(|| invalid("missing raw input"))?; - let Payload::Fallback { - expression: QueryExpr::TimeRange { range, child }, - } = &raw.payload - else { - return Err(invalid( - "temporal producer requires an explicit logical time range", - )); - }; - let QueryExpr::Scan { predicates, .. } = child.as_ref() else { - return Err(invalid("temporal pane source requires a raw scan")); - }; - let window_width_ms: u64 = range - .as_millis() - .try_into() - .map_err(|_| invalid("temporal window overflows"))?; - if window_width_ms == 0 - || window_width_ms > i64::MAX as u64 - || range.subsec_nanos() % 1_000_000 != 0 - { - return Err(invalid( - "temporal window requires positive integral milliseconds", - )); - } - let width = maintenance.layout.pane_width_ms; - if width == 0 || width > window_width_ms || !window_width_ms.is_multiple_of(width) { - return Err(invalid("temporal window must contain whole panes")); - } - match maintenance.framework { - SummaryWindowFramework::Sliding => {} - SummaryWindowFramework::Tumbling if width == window_width_ms => {} - _ => return Err(invalid("unsupported temporal window realization")), - } - let count = window_width_ms / width; - if count > 4096 { - return Err(invalid("temporal pane candidate exceeds input budget")); - } - let raw_id = u64::from(raw_id.0); - if inputs.len() != 1 { - return Err(invalid( - "pane candidate requires exactly its raw input contract", - )); - } - let contract = inputs - .get(&raw_id) - .ok_or_else(|| invalid("missing raw input contract"))?; - let raw_schema = Arc::new(raw.output_schema.clone()); - if contract.schema != raw_schema || contract.properties.boundedness != Boundedness::Bounded { - return Err(invalid("pane source requires its declared bounded schema")); - } - let coordinate = raw_schema - .time_index - .ok_or_else(|| invalid("temporal source requires a time index"))?; - let SummaryInputExpr::Column(value) = &update.weight else { - return Err(invalid("pane builder requires a value column")); - }; - let value = named_column(&raw_schema, value)?; - let groups: Vec<_> = (0..raw_schema.fields.len()) - .filter(|&index| index != coordinate && index != value) - .collect(); - match &maintenance.entity_identity { - TemporalEntityIdentity::SingleEntity if groups.is_empty() => {} - TemporalEntityIdentity::Columns(columns) - if !columns.is_empty() - && columns.len() == columns.iter().collect::>().len() - && columns.iter().copied().collect::>() - == groups.iter().copied().collect() => {} - _ => { - return Err(invalid( - "pane input requires its complete resolved entity identity", - )) - } - } - let mut next = dag - .nodes - .iter() - .map(|node| u64::from(node.id.0)) - .max() - .unwrap_or(0) - + 1; - let mut allocate = || { - let id = next; - next += 1; - id - }; - let mut operators = BTreeMap::new(); - let guard = allocate(); - operators.insert( - guard, - ( - vec![raw_id], - Operator::pane_input( - raw_schema.clone(), - coordinate, - maintenance.layout.clone(), - None, - )?, - ), - ); - let mut previous = guard; - for predicate in predicates { - let id = allocate(); - operators.insert( - id, - ( - vec![previous], - Operator::filter(raw_schema.clone(), expression(&predicate.0, &raw_schema)?)?, - ), - ); - previous = id; - } - let native = - Operator::summary_build(raw_schema, family.clone(), value, Some(coordinate), groups)?; - let compact_state = native.schema(); - let native_id = allocate(); - operators.insert(native_id, (vec![previous], native)); - let state_schema = Arc::new(build.output_schema.clone()); - let pane_output = allocate(); - operators.insert( - pane_output, - ( - vec![native_id], - Operator::scope_timestamp(compact_state, state_schema.clone())?, - ), - ); - let precompute = CompiledPhysicalDag::from_operators(inputs, operators, vec![pane_output])?; - let state_coordinate = state_schema - .time_index - .ok_or_else(|| invalid("pane state requires a time index"))?; - let state_column = summary_column(&state_schema)?; - let mut query_inputs = BTreeMap::new(); - let mut operators = BTreeMap::new(); - let mut pane_inputs = Vec::new(); - let mut guarded_inputs = Vec::new(); - for pane in 0..count { - let input = allocate(); - let guard = allocate(); - query_inputs.insert(input, InputContract::bounded(state_schema.clone())); - let offset = ((count - 1 - pane) * width) as i64; - operators.insert( - guard, - ( - vec![input], - Operator::pane_input( - state_schema.clone(), - state_coordinate, - maintenance.layout.clone(), - Some(offset), - )?, - ), - ); - pane_inputs.push(input); - guarded_inputs.push(guard); - } - let union = allocate(); - operators.insert( - union, - ( - guarded_inputs, - Operator::union(state_schema.clone(), count as usize)?, - ), - ); - let merge = Operator::summary_merge( - state_schema.clone(), - state_column, - (0..state_schema.fields.len()) - .filter(|&index| index != state_coordinate && index != state_column) - .collect(), - )?; - let merged_schema = merge.schema(); - let merged_state = allocate(); - operators.insert(merged_state, (vec![union], merge)); - if roots.is_empty() || roots.iter().copied().collect::>().len() != roots.len() { - return Err(invalid("temporal query requires distinct output roots")); - } - for &root in roots { - let node = dag - .nodes - .iter() - .find(|node| u64::from(node.id.0) == root) - .ok_or_else(|| invalid("unknown temporal output root"))?; - let Payload::SummaryEstimate { - query: SketchQuery::Quantile { q }, - } = node.payload - else { - return Err(invalid("temporal pane root must be a KLL quantile")); - }; - if !q.is_finite() || !(0. ..=1.).contains(&q) { - return Err(invalid("invalid temporal quantile")); - } - let dependencies: Vec<_> = dag - .edges - .iter() - .filter(|edge| edge.consumer == node.id) - .map(|edge| u64::from(edge.producer.0)) - .collect(); - if dependencies != [maintenance.summary_node] { - return Err(invalid( - "temporal readout must consume the maintained producer", - )); - } - let readout = Operator::readout( - merged_schema.clone(), - summary_column(&merged_schema)?, - ReadoutQuery::Sketch(SketchQuery::Quantile { q }), - )?; - let readout_schema = readout.schema(); - let readout_id = allocate(); - operators.insert(readout_id, (vec![merged_state], readout)); - operators.insert( - root, - ( - vec![readout_id], - Operator::scope_timestamp(readout_schema, Arc::new(node.output_schema.clone()))?, - ), - ); - } - let query = CompiledPhysicalDag::from_operators(query_inputs, operators, roots.to_vec())?; - let mut output = precompute.output_contract(pane_output)?; - // Persisted readers have independent timing from the blocking builder. - output.properties.emission = Emission::Unknown; - Ok(TemporalPaneCandidate { - physical: PhysicalCandidate { - precompute: Some(precompute), - query, - materialized_outputs: BTreeMap::from([(pane_output, output)]), - }, - maintenance: maintenance.clone(), - pane_inputs, - merged_state, - window_width_ms, - }) -} diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index 41e5f1b3..d5406f23 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -426,446 +426,3 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { ); } } - -struct SlidingPaneModel; -impl CostModel for SlidingPaneModel { - fn raw_query_recompute_total_cost( - &self, - target: &asap_types::pre_asap::QueryExpr, - reads: f64, - ) -> Option { - let _ = (target, reads); - Some(Cost(100_000.0)) - } - fn rank_candidates( - &self, - intent: &AggIntent, - candidates: &[asap_types::post_asap::SketchAlgorithm], - ) -> Vec { - FullyCostedRuntime.rank_candidates(intent, candidates) - } - fn summary_maintenance_lifecycle_cost_inputs( - &self, - summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - let mut costs = FullyCostedRuntime.summary_maintenance_lifecycle_cost_inputs(summary); - // Controlled workload evidence makes repeated raw construction more - // expensive than retaining and updating the same temporal population. - costs.build_cost = Some(Cost(1000.)); - costs - } - fn summary_maintenance_capabilities( - &self, - summary: &SummaryNode, - ) -> SummaryMaintenanceCapabilities { - FullyCostedRuntime.summary_maintenance_capabilities(summary) - } - fn complete_summary_candidate_estimate( - &self, - _root: &SummaryNode, - _target: Option<&asap_types::pre_asap::QueryExpr>, - deployments: &[asap_aware_mapping::cost_model::CostedSummaryDeployment<'_>], - _horizon: Option, - _reads: Option, - _accuracy: &[AccuracyTarget], - ) -> Option { - Some(asap_aware_mapping::CompleteSummaryCandidateEstimate { - cost: Cost( - deployments - .iter() - .map(|deployment| deployment.selected_cost.0) - .sum(), - ), - physical_plan_id: Some("bounded-sliding-pane-evidence".into()), - window_frameworks: deployments - .iter() - .map(|deployment| { - (deployment.guarantee.summary_maintenance_lifecycle - == SummaryMaintenanceLifecycle::ContinuouslyMaintained) - .then_some(asap_types::post_asap::SummaryWindowFramework::Sliding) - }) - .collect(), - window_accuracy_guarantee: None, - }) - } -} - -/// Workload and optimizer-selected lifecycle generate both physical DAGs. -/// No computational operators or graph edges are constructed by this fixture. -#[test] -fn selected_temporal_lifecycle_compiles_panes_and_executes() { - use asap_physical_operators::{ - operators::Operator, - physical_planner::{ - compile_temporal_pane_candidate, InputContract, Source, TemporalEntityIdentity, - TemporalPaneMaintenance, - }, - runtime::{Limits, RunContext, Scope}, - summary_kernels::datasketches_kll::DatasketchesKLLAccumulator, - values::{Batch, Value}, - }; - use asap_types::{ - post_asap::{ - compile_post_asap_dag, plan_pane_phase, PostAsapOperatorPayload, SummaryFamilyType, - SummaryWindowFramework, - }, - pre_asap::DataType, - workload::TimestampMs, - }; - use futures::{executor::block_on, StreamExt}; - use std::{collections::BTreeMap, sync::Arc}; - for quantile in [0.5, 0.99] { - let mut workload = dashboard_workload(); - let query = Query(format!( - "quantile_over_time({quantile}, latency{{job=\"api\"}}[5m])" - )); - workload.query_workload.query_batch.as_mut().unwrap()[0].query = query.clone(); - workload.query_workload.repeating_queries.as_mut().unwrap()[0].query = query; - - workload - .data_workload - .as_mut() - .unwrap() - .data_ingestion_interval - .value = Some(DurationMs(60_000)); - workload.query_workload.repeating_queries.as_mut().unwrap()[0].demand = - RepeatedDemand::FixedIntervalAt { - interval: RepetitionInterval(60_000), - evaluation_phase: TimestampMs(300_000), - }; - let plan = selected_plan_with_horizon(&workload, &SlidingPaneModel, Horizon(1000.)); - assert!(!plan.selected_raw_recompute); - assert_eq!(plan.deployments.len(), 1); - let deployment = &plan.deployments[0]; - assert_eq!( - deployment.selected_window_framework, - Some(SummaryWindowFramework::Sliding) - ); - let dag = compile_post_asap_dag(&plan.root).unwrap(); - let build = dag - .nodes - .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) - .unwrap(); - assert_eq!(build.id, deployment.post_asap_node_id); - let raw = dag - .nodes - .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) - .unwrap(); - let schema = Arc::new(raw.output_schema.clone()); - let width = workload - .data_workload - .as_ref() - .unwrap() - .data_ingestion_interval - .value - .unwrap() - .0; - let layout = plan_pane_phase( - &workload.query_workload.repeating_queries.as_ref().unwrap()[0].demand, - width, - ) - .unwrap(); - let maintenance = TemporalPaneMaintenance { - summary_node: u64::from(build.id.0), - lifecycle: deployment - .summary_maintenance_lifecycle_guarantee - .clone() - .unwrap(), - framework: deployment.selected_window_framework.clone().unwrap(), - layout, - // The memory source has exactly the declared label columns; a - // schemaless deployment must resolve all entity keys first. - entity_identity: TemporalEntityIdentity::Columns( - schema - .fields - .iter() - .enumerate() - .filter(|(_, field)| field.name == "job") - .map(|(index, _)| index) - .collect(), - ), - }; - let candidate = compile_temporal_pane_candidate( - &dag, - BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], - &maintenance, - ) - .unwrap(); - assert_eq!(candidate.window_width_ms, 300_000); - assert_eq!(candidate.pane_inputs.len(), 5); - assert_ne!( - candidate.physical.precompute.as_ref().unwrap().roots()[0], - maintenance.summary_node, - "a one-minute pane is not the logical five-minute summary output" - ); - let source_opens = Arc::new(std::sync::atomic::AtomicUsize::new(0)); - let mut stored = Vec::new(); - for pane in 0..6 { - let rows = (0..20) - .flat_map(|sample| { - ["api", "batch"] - .into_iter() - .map(move |entity| (sample, entity)) - }) - .map(|(sample, entity)| { - schema - .fields - .iter() - .map(|field| match field.dtype { - SummaryFamilyType::Plain(DataType::Timestamp) => { - Value::Timestamp(pane * 60_000 + (sample + 1) * 3000) - } - SummaryFamilyType::Plain(DataType::Float64) => Value::Float64( - (pane * 20 + sample) as f64 - + if entity == "batch" { 100_000. } else { 0. }, - ), - SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8(entity.into()), - _ => panic!("unexpected raw field {field:?}"), - }) - .collect() - }) - .collect(); - let result = physical_common::execute( - candidate.physical.precompute.as_ref().unwrap(), - BTreeMap::from([( - u64::from(raw.id.0), - Batch::try_new(schema.clone(), rows).unwrap(), - )]), - Scope::Ingestion { - window_start_ms: pane * 60_000, - window_end_ms: (pane + 1) * 60_000, - revision: 1, - }, - ); - let batch = &result[0][0]; - stored.push(batch.clone()); - } - for offset in [0, 1] { - let inputs: BTreeMap<_, _> = candidate - .pane_inputs - .iter() - .enumerate() - .map(|(index, &id)| (id, stored[offset + index].clone())) - .collect(); - let scope = Scope::Query { - evaluation_time_ms: (5 + offset as i64) * 60_000, - revision: 1, - }; - let result = - physical_common::execute(&candidate.physical.query, inputs.clone(), scope.clone()); - let rows = result[0][0].rows(); - assert_eq!(rows.len(), 1); - assert!(rows[0] - .iter() - .any(|value| matches!(value, Value::Utf8(label) if label.as_ref() == "api"))); - let Value::Float64(value) = rows[0][1] else { - panic!("missing p99") - }; - assert!( - (value - ((if quantile == 0.5 { 50 } else { 99 }) + offset * 20) as f64).abs() - <= 1. - ); - assert!( - matches!(rows[0][0], Value::Timestamp(timestamp) if timestamp == (5 + offset as i64) * 60_000) - ); - let sources = |inputs: BTreeMap| -> BTreeMap> { - inputs - .into_iter() - .map(|(id, batch)| { - ( - id, - Box::new(TemporalCountingSource { - operator: Operator::source(batch.schema().clone(), vec![batch]) - .unwrap(), - opens: source_opens.clone(), - }) as Source<'static>, - ) - }) - .collect() - }; - let bound = candidate - .physical - .query - .instantiate(sources(inputs.clone())) - .unwrap(); - let population = block_on( - bound - .execute( - &[candidate.merged_state], - RunContext::new(scope.clone(), Limits::default()).unwrap(), - ) - .unwrap() - .remove(0) - .collect::>(), - ); - let Value::Summary { state, .. } = population[0].as_ref().unwrap().rows()[0] - .iter() - .find(|value| matches!(value, Value::Summary { .. })) - .unwrap() - else { - panic!("missing merged state") - }; - assert_eq!( - state - .as_any() - .downcast_ref::() - .unwrap() - .inner - .count(), - 100 - ); - let mut missing = inputs.clone(); - missing.remove(&candidate.pane_inputs[0]); - assert!(candidate - .physical - .query - .instantiate(sources(missing)) - .is_err()); - let mut duplicate = inputs.clone(); - duplicate.insert(candidate.pane_inputs[1], stored[offset].clone()); - let bad = candidate - .physical - .query - .instantiate(sources(duplicate)) - .unwrap(); - let errors = block_on( - bad.execute( - candidate.physical.query.roots(), - RunContext::new(scope.clone(), Limits::default()).unwrap(), - ) - .unwrap() - .remove(0) - .collect::>(), - ); - assert!( - errors.iter().any(Result::is_err), - "duplicate pane must not be merged twice" - ); - let mut duplicate_entity = inputs.clone(); - let pane = &stored[offset]; - duplicate_entity.insert( - candidate.pane_inputs[0], - Batch::try_new( - pane.schema().clone(), - vec![pane.rows()[0].clone(), pane.rows()[0].clone()], - ) - .unwrap(), - ); - let bad = candidate - .physical - .query - .instantiate(sources(duplicate_entity)) - .unwrap(); - let errors = block_on( - bad.execute( - candidate.physical.query.roots(), - RunContext::new(scope.clone(), Limits::default()).unwrap(), - ) - .unwrap() - .remove(0) - .collect::>(), - ); - assert!( - errors.iter().any(Result::is_err), - "duplicate snapshots within a pane must fail" - ); - let bound = candidate - .physical - .query - .instantiate(sources(inputs)) - .unwrap(); - source_opens.store(0, std::sync::atomic::Ordering::SeqCst); - assert!(bound - .execute( - candidate.physical.query.roots(), - RunContext::new( - Scope::Query { - evaluation_time_ms: 330_000, - revision: 1 - }, - Limits::default() - ) - .unwrap() - ) - .is_err()); - assert_eq!( - source_opens.load(std::sync::atomic::Ordering::SeqCst), - 0, - "invalid phase must fail before opening readers" - ); - } - // A selected framework cannot be silently replaced by physical planning. - let mut wrong_framework = maintenance.clone(); - wrong_framework.framework = SummaryWindowFramework::ExponentialHistogram; - let mut wrong_identity = maintenance.clone(); - wrong_identity.entity_identity = TemporalEntityIdentity::SingleEntity; - let mut unknown_phase = maintenance.clone(); - unknown_phase.layout.pane_origin_ms = None; - let mut partial_panes = maintenance.clone(); - partial_panes.layout.pane_width_ms = 90_000; - let mut wrong_lifecycle = maintenance.clone(); - wrong_lifecycle.lifecycle.summary_maintenance_lifecycle = - SummaryMaintenanceLifecycle::Ephemeral; - for unsupported in [ - wrong_framework, - wrong_identity, - unknown_phase, - partial_panes, - wrong_lifecycle, - ] { - assert!(compile_temporal_pane_candidate( - &dag, - BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], - &unsupported - ) - .is_err()); - } - } -} - -struct TemporalCountingSource { - operator: asap_physical_operators::operators::Operator, - opens: std::sync::Arc, -} -impl - asap_physical_operators::plan::PhysicalOperator< - asap_physical_operators::values::Batch, - asap_physical_operators::values::Schema, - > for TemporalCountingSource -{ - fn name(&self) -> &str { - "TemporalCountingSource" - } - fn properties( - &self, - inputs: &[asap_physical_operators::plan::PlanProperties], - ) -> asap_physical_operators::plan::PlanProperties { - self.operator.properties(inputs) - } - fn input_schemas(&self) -> Vec { - self.operator.input_schemas() - } - fn output_schema(&self) -> asap_physical_operators::values::Schema { - self.operator.output_schema() - } - fn output_bytes(&self, batch: &asap_physical_operators::values::Batch) -> usize { - batch.bytes() - } - fn start<'a>( - &'a self, - inputs: Vec< - asap_physical_operators::runtime::Input<'a, asap_physical_operators::values::Batch>, - >, - context: asap_physical_operators::runtime::RunContext, - ) -> Result< - asap_physical_operators::runtime::OutputStream<'a, asap_physical_operators::values::Batch>, - asap_physical_operators::Error, - > { - self.opens.fetch_add(1, std::sync::atomic::Ordering::SeqCst); - self.operator.start(inputs, context) - } -} diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 72313a9d..d9c91009 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -295,7 +295,7 @@ NativeKllBuild(k=200) KllStateOutput(k=200) ``` -This DAG implements construction of each maintained one-minute pane. Its input +This DAG computes the state of one maintained one-minute pane. Its input contract requires all input samples matching the source, filters and group within that pane; the deployment supplies that bounded input from its source integration. @@ -480,33 +480,18 @@ snapshots as separate inputs. ## 6. Executable acceptance coverage -The tests cover optimizer-selected lifecycle execution and automatic temporal -pane compilation, alongside independent operator/runtime fixtures: +The tests cover optimizer-selected lifecycle execution alongside independent +operator/runtime fixtures: | Test | Contract exercised | | --- | --- | -| `summary_maintenance_lifecycle_e2e::selected_temporal_lifecycle_compiles_panes_and_executes` | PromQL p50/p99 workloads → selected continuous lifecycle and Sliding framework → automatically generated precompute/query DAGs → real codec round-trip → adjacent aligned windows; checks filters, entity identity, sample counts, missing/duplicate panes and phase rejection before opening readers | | `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled precompute/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | | `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute precompute DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | | `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | | `precompute_candidates::grouped_rate_can_be_materialized_before_or_after_grouped_sum` | Cost changes select different legal precompute frontiers; both selected candidates execute with the same reset-sensitive result; uncompilable candidates are not priced | | `sql_to_physical::sql_filter_grouped_sum_executes_and_rebinds` | SQL text → candidate search → physical compilation → shared Scan predicates and grouped summary execution; NULL samples are ignored and fresh bindings produce new results | -`physical_planner::compile_temporal_pane_candidate` consumes the logical DAG, -selected lifecycle/framework and a generic pane/entity input contract. It -generates pane construction, scan predicates, ordered state slots, a shared -merge, quantile readouts and run-scoped timestamps. A physical pane output has -its own identity: one minute of state cannot masquerade as the logical -five-minute summary. The returned candidate retains the maintenance contract. - -This initial realization supports bounded, complete KLL panes with known phase -and resolved entity identity, for Sliding windows or a single Tumbling window. -Source capability evidence must declare all entity keys or isolate one entity; -usage-derived PromQL columns alone cannot establish that identity. Partial edge -panes, exponential histograms and cross-run delta accumulation require further -physical candidates and are rejected by this entry point. - -Physical execution checks pane timestamps and duplicate entity states. Concrete -stored identity, revisions, readiness and complete coverage of required input samples remain -deployment responsibilities. Real storage and HTTP execution belong to +Pane construction, pane timestamp checks, stored identity, revisions, readiness +and complete coverage of required input samples are deployment +responsibilities. Real storage and HTTP execution belong to deployment-repository E2E tests. From fb92998a97bd8910d104023c8ea6aa2e16890eab Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:23:03 +0000 Subject: [PATCH 05/59] docs: state what the physical layer does not own Co-Authored-By: Claude Opus 5.5 --- docs/design_docs/physical-planning-and-deployment.md | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index d9c91009..fe71c8c3 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -474,6 +474,13 @@ shared physical operator implementation library, `asap-physical-operators`, and its DAG runtime. The merge executes once per run for both consumers. Execution does not introduce additional planning decisions. +The physical layer does not own raw ingestion, pane construction or geometry, +storage formats, or decoding persisted bytes into typed state. It compiles +computation over typed input contracts: summary build, merge (for example KLL +merge), sketch estimates and exact finalization. The deployment constructs panes, +reads and decodes stored state, and binds the typed values to input slots. +Compiled physical plans are Planner outputs and keep their own serialized form. + Each maintained pane contributes its input samples once. A replacement snapshot replaces that pane's state; query merging must not count both the old and new snapshots as separate inputs. From ea0966da91174d40f90394834d06fc49028bc680 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 22:04:38 +0000 Subject: [PATCH 06/59] refactor: move PromQL series-identity resolution into asap-types Planner search needs the physical row representation to propose whole-root physical alternatives, and asap-aware-mapping cannot depend on the physical runtime crate. promql_rows::with_series_identity keeps its signature and delegates. Co-Authored-By: Claude Opus 5.5 --- .../src/physical_planner/promql_rows.rs | 76 +------------------ crates/types/src/pre_asap/schema.rs | 65 ++++++++++++++++ 2 files changed, 69 insertions(+), 72 deletions(-) diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index cb680096..8b887580 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -1,7 +1,7 @@ //! A bounded PromQL source row carries the entire label set, not just labels //! mentioned by the query. The source adapter owns this lossless encoding. use super::*; -use planner_types::pre_asap::{Column, DataType, Source as LogicalSource}; +use planner_types::pre_asap::DataType; use std::rc::Rc; /// Not a legal PromQL label name, so it cannot shadow a user label. @@ -22,78 +22,10 @@ pub fn decode_series_identity(encoded: &str) -> Result, Ok(labels) } -/// Resolve the row representation before candidate search. `closed` describes -/// physical columns here: the final column contains every dynamic source label. -/// It does not assert that the query's projected labels are the full label set. -/// -/// This realization supports explicit `by` grouping and per-series computation. -/// Operators that rewrite or implicitly match dynamic label sets require their -/// own realization; they must not accidentally treat the opaque identity as a -/// user label or silently discard it. +/// Resolve the row representation before candidate search; see +/// [`planner_types::pre_asap::schema::with_promql_series_identity`]. pub fn with_series_identity(root: &QueryExpr) -> Result { - let mut root = root.clone(); - fn visit(node: &mut QueryExpr) -> Result<(), Error> { - use planner_types::pre_asap::Reduction; - match node { - QueryExpr::Scan { - source: LogicalSource::TimeSeries { .. }, - schema, - .. - } => { - if schema - .columns - .iter() - .any(|column| column.name == SERIES_IDENTITY_COLUMN) - { - return Err(invalid( - "source already contains a physical series identity", - )); - } - if schema.closed { - return Err(invalid( - "dynamic series identity requires an open PromQL source", - )); - } - schema - .columns - .push(Column::new(SERIES_IDENTITY_COLUMN, DataType::Utf8, false)); - schema.closed = true; - Ok(()) - } - QueryExpr::TimeRange { child, .. } | QueryExpr::Limit { child, .. } => { - visit(Rc::make_mut(child)) - } - QueryExpr::Aggregate { - child, reduction, .. - } => { - if matches!(reduction, Reduction::Reduce(keys) if keys.is_without()) { - return Err(invalid( - "dynamic without grouping requires label-set projection", - )); - } - visit(Rc::make_mut(child)) - } - QueryExpr::Sort { - child, - partition_by, - .. - } => { - if partition_by.is_without() { - return Err(invalid( - "dynamic without ranking requires label-set projection", - )); - } - visit(Rc::make_mut(child)) - } - _ => Err(invalid( - "operator has no dynamic series-identity realization", - )), - } - } - visit(&mut root)?; - root.output_schema() - .map_err(|error| invalid(error.to_string()))?; - Ok(root) + planner_types::pre_asap::schema::with_promql_series_identity(root).map_err(invalid) } /// Construct source rows only from full identities. The named label columns diff --git a/crates/types/src/pre_asap/schema.rs b/crates/types/src/pre_asap/schema.rs index fbcb9d67..938b9b7f 100644 --- a/crates/types/src/pre_asap/schema.rs +++ b/crates/types/src/pre_asap/schema.rs @@ -158,6 +158,71 @@ pub struct Schema { /// label map. `$` cannot occur in a user PromQL label name. pub const PROMQL_SERIES_IDENTITY: &str = "$promql_series_identity"; +/// Resolve a PromQL root to rows carrying [`PROMQL_SERIES_IDENTITY`] before +/// candidate search. `closed` describes physical columns here: the final +/// column contains every dynamic source label. It does not assert that the +/// query's projected labels are the full label set. +/// +/// This realization supports explicit `by` grouping and per-series computation. +/// Operators that rewrite or implicitly match dynamic label sets require their +/// own realization; they must not accidentally treat the opaque identity as a +/// user label or silently discard it. +pub fn with_promql_series_identity(root: &super::QueryExpr) -> Result { + use super::{QueryExpr, Reduction, Source}; + use std::rc::Rc; + fn visit(node: &mut QueryExpr) -> Result<(), String> { + match node { + QueryExpr::Scan { + source: Source::TimeSeries { .. }, + schema, + .. + } => { + if schema + .columns + .iter() + .any(|column| column.name == PROMQL_SERIES_IDENTITY) + { + return Err("source already contains a physical series identity".into()); + } + if schema.closed { + return Err("dynamic series identity requires an open PromQL source".into()); + } + schema + .columns + .push(Column::new(PROMQL_SERIES_IDENTITY, DataType::Utf8, false)); + schema.closed = true; + Ok(()) + } + QueryExpr::TimeRange { child, .. } | QueryExpr::Limit { child, .. } => { + visit(Rc::make_mut(child)) + } + QueryExpr::Aggregate { + child, reduction, .. + } => { + if matches!(reduction, Reduction::Reduce(keys) if keys.is_without()) { + return Err("dynamic without grouping requires label-set projection".into()); + } + visit(Rc::make_mut(child)) + } + QueryExpr::Sort { + child, + partition_by, + .. + } => { + if partition_by.is_without() { + return Err("dynamic without ranking requires label-set projection".into()); + } + visit(Rc::make_mut(child)) + } + _ => Err("operator has no dynamic series-identity realization".into()), + } + } + let mut root = root.clone(); + visit(&mut root)?; + root.output_schema().map_err(|error| error.to_string())?; + Ok(root) +} + impl Schema { pub fn has_promql_series_identity(&self) -> bool { self.closed From fb824f91e0a7ac1eba6996fd04ac95e325edc23b Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 22:30:51 +0000 Subject: [PATCH 07/59] feat: list PromQL physical alternatives in PlanSpace The backend built one workload forest per per-query physical alternative by calling SketchAlgorithmStrategy's proposal methods outside PlanSpace. Planner now owns them: search_workload_with_targets asks each strategy's new ReplacementStrategy::propose_for_root once per targeted root. SketchAlgorithmStrategy resolves the series identity and proposes current-series TopK, fixed-window and query-time Rate aggregation, and realizations over per-series Rate state, finalized and deduplicated. The candidates carry ReplacementProvenance::RootPhysicalRealization: DAG assembly uses them verbatim, and global selection never commits them. Existing candidates and their order are unchanged. Co-Authored-By: Claude Opus 5.5 --- crates/asap-aware-mapping/src/replacement.rs | 114 +++++++- .../tests/planspace_physical_alternatives.rs | 272 ++++++++++++++++++ 2 files changed, 385 insertions(+), 1 deletion(-) create mode 100644 crates/asap-physical-operators/tests/planspace_physical_alternatives.rs diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index f8f55a1d..9f2444f9 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -560,6 +560,12 @@ pub enum ReplacementProvenance { /// [`Replacement::ExactComposition`] with /// [`OperationPlacement::Maintenance`] (issue #171). ValueOperationAtIngestionTime, + /// A finalized whole-query result over a physical row representation + /// the logical children do not share (see + /// [`ReplacementStrategy::propose_for_root`]). Assembly uses it verbatim, + /// and default selection never commits it: only deployment compiles and + /// prices it. + RootPhysicalRealization, } /// A candidate a strategy considered for a target but refused to propose on @@ -634,6 +640,14 @@ pub trait ReplacementStrategy { domain_error: None, } } + + /// Whole-query alternatives for a workload root under its end-to-end + /// `target`. These may change the root's physical row representation, so + /// [`search_workload_with_targets`] asks only workload roots, once each. + /// Default: none. + fn propose_for_root(&self, _root: &Rc, _target: &AccuracyTarget) -> Proposals { + Proposals::default() + } } // ── Realization: how one AggIntent may be realised ─────────────────────── @@ -1705,6 +1719,73 @@ impl ReplacementStrategy for SketchAlgorithmStrategy<'_> { fn propose(&self, target: &TargetSubDAG<'_>) -> Proposals { self.propose_with(target.root, None) } + + /// Physical alternatives over rows that carry the complete PromQL series + /// identity: current-series TopK, fixed-window and query-time Rate + /// aggregation, and ordinary realizations over per-series Rate state. + /// Each is a finalized query result for the identity-carrying root. + fn propose_for_root(&self, root: &Rc, target: &AccuracyTarget) -> Proposals { + let Ok(typed) = asap_types::pre_asap::schema::with_promql_series_identity(root) else { + return Proposals::default(); + }; + let typed = Rc::new(typed); + let mut proposals = self.current_series_topk_candidates(&typed, target); + let mut direct = self.propose_with(&typed, None).candidates; + // Other identity-carrying realizations duplicate the logical root's. + direct.retain(|candidate| { + matches!(&candidate.replacement, Replacement::Summary(node) if has_series_rate_frontier(node)) + }); + // Rejections from the shared enumeration repeat the logical root's own. + let candidates = std::mem::take(&mut proposals.candidates) + .into_iter() + .chain(self.fixed_window_rate_candidates(&typed).candidates) + .chain( + self.query_time_rate_aggregation_candidates(&typed) + .candidates, + ) + .chain(direct); + for mut candidate in candidates { + let Replacement::Summary(node) = candidate.replacement else { + continue; + }; + let Ok(node) = finalize_query_candidate(node, &typed) else { + continue; + }; + let duplicate = proposals.candidates.iter().any(|existing| { + matches!(&existing.replacement, Replacement::Summary(other) if *other == node) + }); + if !duplicate { + candidate.replacement = Replacement::Summary(node); + candidate.provenance = ReplacementProvenance::RootPhysicalRealization; + proposals.candidates.push(candidate); + } + } + proposals + } +} + +/// Exact per-series Rate state over raw samples: the frontier at which +/// deployment binds complete, identity-keyed counter states. +fn has_series_rate_frontier(node: &SummaryNode) -> bool { + match &node.expr { + SummaryExpr::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + reduction: Reduction::PerEntity, + child, + .. + } if matches!(&child.expr, SummaryExpr::KeepPreAsap(source) + if matches!(source.as_ref(), QueryExpr::TimeRange { .. })) => + { + true + } + SummaryExpr::ValueOperation { child, .. } | SummaryExpr::SummaryAgg { child, .. } => { + has_series_rate_frontier(child) + } + SummaryExpr::SummaryEstimate { summary_input, .. } => { + has_series_rate_frontier(summary_input) + } + _ => false, + } } /// A human-readable rationale for one candidate `Realization`, for @@ -5203,6 +5284,14 @@ impl<'a> GlobalSelection<'a> { if let Some(node) = self.assembled_nodes.borrow().get(&ptr) { return Ok(Rc::clone(node)); } + if let Some(ReplacementSubDAG { + replacement: Replacement::Summary(node), + provenance: ReplacementProvenance::RootPhysicalRealization, + .. + }) = self.groups.get(&ptr).and_then(|selection| selection.chosen) + { + return Ok(Rc::clone(node)); + } let selected_composed_summary = self .groups .get(&ptr) @@ -5966,7 +6055,8 @@ fn is_cse_candidate(candidate: &ReplacementSubDAG) -> bool { } fn is_automatically_selectable(candidate: &ReplacementSubDAG, cost_model: &dyn CostModel) -> bool { - !candidate.has_missing_accuracy_evidence() + candidate.provenance != ReplacementProvenance::RootPhysicalRealization + && !candidate.has_missing_accuracy_evidence() && candidate.runtime_support_evidence(cost_model) != Some(false) } @@ -6468,6 +6558,28 @@ pub fn search_workload_with_targets<'s, Id>( .zip(targets) .filter_map(|((_, root), target)| target.map(|t| (Rc::as_ptr(root), t))) .collect(); + // Whole-root proposals join the root group before its target check. + for (index, (ptr, target)) in root_ptrs.iter().enumerate() { + if root_ptrs[..index].contains(&(*ptr, target.clone())) { + continue; + } + let group = space.groups.get_mut(ptr).expect("every root has a group"); + let root = Rc::clone(&group.target); + for strategy in strategies { + let name = strategy.name(); + let proposals = strategy.propose_for_root(&root, target); + for mut candidate in proposals.candidates { + candidate.strategy = name; + group.add_candidate(candidate); + } + group + .rejected + .extend(proposals.rejected.into_iter().map(|mut rejection| { + rejection.strategy = name; + rejection + })); + } + } let mut composition_targets: HashMap<_, Vec<_>> = HashMap::new(); for (ptr, target) in root_ptrs { composition_targets diff --git a/crates/asap-physical-operators/tests/planspace_physical_alternatives.rs b/crates/asap-physical-operators/tests/planspace_physical_alternatives.rs new file mode 100644 index 00000000..04d684f7 --- /dev/null +++ b/crates/asap-physical-operators/tests/planspace_physical_alternatives.rs @@ -0,0 +1,272 @@ +//! Physical row-representation alternatives are part of Planner's search space: +//! `enumerate_candidate_dags_for_root` lists them without a caller-side +//! series-identity pass, cost ranking, or workload Cartesian expansion. +use asap_aware_mapping::{ + accuracy::{AccuracyEvidenceProvider, DefaultAccuracyModel, PropagationStats}, + cost_model::DefaultCostModel, + replacement::{default_strategies_with_evidence, ReplacementProvenance}, + search_workload_with_targets, Proposals, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, +}; +use asap_physical_operators::physical_planner::promql_rows::{ + compile_current_series_readout, compile_fixed_window_rate_aggregation, compile_rate_ranking, + SERIES_IDENTITY_COLUMN, +}; +use planner_types::{ + post_asap::*, + pre_asap::QueryExpr, + types::AccuracyTarget, + workload::{ + AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence as WorkloadEvidence, + PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, + TimeSelection, + }, +}; +use std::rc::Rc; + +struct Evidence; +impl AccuracyEvidenceProvider for Evidence { + fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + Some(1000) + } + fn propagation_stats( + &self, + op: &CompositionOperator, + _: &SummaryFamilyType, + _: Option<&SketchQuery>, + ) -> PropagationStats { + if matches!(op, CompositionOperator::TopKSelection) { + PropagationStats { + topk_selected_lower_bound: Some(101.), + topk_excluded_upper_bound: Some(100.), + topk_interval_failure_probability: Some(0.001), + ..Default::default() + } + } else { + Default::default() + } + } +} + +/// Forwards everything except whole-root proposals: the pre-change search. +struct LogicalOnly(Box); +impl ReplacementStrategy for LogicalOnly { + fn name(&self) -> &'static str { + self.0.name() + } + fn matches(&self, target: &TargetSubDAG<'_>) -> bool { + self.0.matches(target) + } + fn replacements(&self, target: &TargetSubDAG<'_>) -> Vec { + self.0.replacements(target) + } + fn propose(&self, target: &TargetSubDAG<'_>) -> Proposals { + self.0.propose(target) + } +} + +fn lower(query: &str, accuracy: &AccuracyTarget) -> Rc { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy.clone()), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: WorkloadEvidence { + value: Some(DurationMs(1_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + Rc::new( + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0), + ) +} + +type Dag = Vec<(usize, Rc)>; + +/// Candidate DAGs for query 1 of a two-query workload, with and without +/// whole-root proposals. Query 0 is a bystander that must not multiply them. +fn inventories(query: &str, accuracy: AccuracyTarget) -> (Vec, Vec) { + let roots = vec![ + ( + 0, + lower("sum by(job)(m)", &AccuracyTarget::Exact), + Some(AccuracyTarget::Exact), + ), + (1, lower(query, &accuracy), Some(accuracy)), + ]; + let full = default_strategies_with_evidence(&DefaultCostModel, &Evidence); + let logical: Vec> = + default_strategies_with_evidence(&DefaultCostModel, &Evidence) + .into_iter() + .map(|strategy| Box::new(LogicalOnly(strategy)) as Box) + .collect(); + let enumerate = |strategies: &[Box]| { + search_workload_with_targets(roots.clone(), strategies, &DefaultAccuracyModel) + .enumerate_candidate_dags_for_root(&1, 65_536) + .unwrap() + .candidates + }; + (enumerate(&full), enumerate(&logical)) +} + +fn carries_identity(dag: &Dag) -> bool { + dag.iter().any(|(_, root)| { + compile_post_asap_dag(root) + .unwrap() + .nodes + .iter() + .any(|node| { + node.output_schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + }) + }) +} + +/// Shared acceptance checks; returns the added physical alternatives. +fn added_alternatives(query: &str, accuracy: AccuracyTarget) -> Vec> { + let (full, logical) = inventories(query, accuracy); + for (index, dag) in full.iter().enumerate() { + assert_eq!(dag.len(), 1, "one root per candidate, no workload product"); + assert!( + !full[..index].contains(dag), + "{query}: identical DAG listed twice" + ); + } + let (added, kept): (Vec<_>, Vec<_>) = full.into_iter().partition(carries_identity); + assert_eq!( + kept, logical, + "{query}: existing candidates must be unchanged" + ); + added.into_iter().map(|mut dag| dag.remove(0).1).collect() +} + +// Grouped counter rates list both ingestion-time and query-time grouped Sum. +#[test] +fn grouped_rate_lists_fixed_window_and_query_time_aggregation() { + let added = added_alternatives("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact); + assert!(added + .iter() + .any(|root| compile_fixed_window_rate_aggregation(root).is_ok())); + assert!(added.iter().any(|root| compile_rate_ranking(root).is_ok())); +} + +// Ranking over counter rates lists heap ranking and fixed-window heaps. +#[test] +fn rate_topk_lists_rate_ranking_and_fixed_window_heaps() { + let added = added_alternatives("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)); + assert!(added.iter().any(|root| compile_rate_ranking(root).is_ok())); + assert!(added + .iter() + .any(|root| compile_fixed_window_rate_aggregation(root).is_ok())); +} + +// Instant-vector TopK lists the current-series heap readout. +#[test] +fn current_series_topk_lists_current_series_readout() { + let added = added_alternatives("topk by(job)(1, m)", AccuracyTarget::Epsilon(0.1)); + assert!(added + .iter() + .any(|root| compile_current_series_readout(root).is_ok())); +} + +// Every added alternative is a finalized query result the physical layer binds. +#[test] +fn added_alternatives_are_finalized_and_physically_bindable() { + for (query, accuracy) in [ + ("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact), + ("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)), + ("topk by(job)(1, m)", AccuracyTarget::Epsilon(0.1)), + ("rate(m[1m])", AccuracyTarget::Exact), + ] { + let added = added_alternatives(query, accuracy); + assert!(!added.is_empty(), "{query}"); + for root in added { + assert!(!matches!(root.expr, SummaryExpr::SummaryAgg { .. })); + assert!( + compile_current_series_readout(&root).is_ok() + || compile_rate_ranking(&root).is_ok() + || compile_fixed_window_rate_aggregation(&root).is_ok(), + "{query}: unbindable alternative {root:?}" + ); + } + } +} + +// Queries without a series-identity physical realization are unchanged. +#[test] +fn unrelated_queries_keep_their_inventory() { + for (query, accuracy) in [ + ("sum by(job)(m)", AccuracyTarget::Exact), + ( + "quantile_over_time(0.9, m[1m])", + AccuracyTarget::Epsilon(0.05), + ), + ("max_over_time(m[1m])", AccuracyTarget::Exact), + ] { + assert!(added_alternatives(query, accuracy).is_empty(), "{query}"); + } +} +// Default cost-based selection keeps the logical plan; deployment prices alternatives. +#[test] +fn global_selection_never_commits_a_physical_alternative() { + let accuracy = AccuracyTarget::Exact; + let root = lower("sum by(job)(rate(m[1m]))", &accuracy); + let strategies = default_strategies_with_evidence(&DefaultCostModel, &Evidence); + let space = search_workload_with_targets( + vec![(0, root, Some(accuracy))], + &strategies, + &DefaultAccuracyModel, + ); + let selected = space + .global_selection(&DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + assert!(!carries_identity(&vec![(0, selected)])); +} + +// A query repeated in the workload is proposed once, not once per copy. +#[test] +fn repeated_roots_do_not_duplicate_alternatives() { + let accuracy = AccuracyTarget::Exact; + let strategies = default_strategies_with_evidence(&DefaultCostModel, &Evidence); + let count = |copies: usize| { + let roots = (0..copies) + .map(|id| { + ( + id, + lower("sum by(job)(rate(m[1m]))", &accuracy), + Some(accuracy.clone()), + ) + }) + .collect(); + let space = search_workload_with_targets(roots, &strategies, &DefaultAccuracyModel); + space + .candidates_for_target(&space.roots[0].1) + .unwrap() + .candidates + .iter() + .filter(|candidate| { + candidate.provenance == ReplacementProvenance::RootPhysicalRealization + }) + .count() + }; + assert_eq!(count(2), count(1)); +} From c57d43811aadbee72825ff06a13fe2d2c9c3d88c Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 22:30:55 +0000 Subject: [PATCH 08/59] docs: describe per-root physical alternatives in candidate enumeration Co-Authored-By: Claude Opus 5.5 --- .../physical-planning-and-deployment.md | 8 +++++++ docs/develop_docs/library-api.md | 23 +++++++++++++++++++ 2 files changed, 31 insertions(+) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index fe71c8c3..fad645d4 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -79,6 +79,14 @@ sample values does not preserve instant-vector semantics. Replacement, rank decrease, expiry, grouping and the required approximation guarantee must be validated before admitting that physical candidate. +Whole-query alternatives that need a different physical row representation are +part of Planner's candidate space. For PromQL, Planner resolves rows that carry +the complete series identity, then proposes the families above for that root: +fixed-window Rate aggregation, query-time ranking or aggregation over +per-series Rate readouts, and current-series TopK. These are listed per root +with the other candidates, unranked. Queries without such a realization are +unchanged. + This is the target ownership contract. A backend path that still reconstructs operators from logical candidates has not completed this integration. diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index bfeea034..e4845733 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -267,6 +267,29 @@ they are not necessarily a globally sortable physical-cost scalar. Unavailable cost alternatives may remain for explanation. Inspect eligibility and evidence before physical selection; do not treat their presence as deployment permission. +### Enumerate candidate DAGs per root + +```text +PlanSpace::enumerate_candidate_dags_for_root(&self, id: &Id, expansion_limit: usize) + -> Result, RealizationError> +``` + +Returns every distinct finalized DAG for one root, unranked; other roots' +choices are not multiplied in. Exceeding `expansion_limit` is an error, never a +partial inventory. + +For PromQL roots that carry a target, `search_workload_with_targets` also asks +each strategy's `ReplacementStrategy::propose_for_root`. `SketchAlgorithmStrategy` +answers with physical alternatives over rows carrying the complete series +identity (`$promql_series_identity`): current-series TopK, fixed-window and +query-time Rate aggregation, and realizations over per-series Rate state. They +are finalized, deduplicated, and marked +`ReplacementProvenance::RootPhysicalRealization`. Callers do not apply +`with_series_identity` or call the proposal methods themselves. Compile each +alternative with the matching `promql_rows::compile_*` function; queries without +such an alternative keep their previous inventory. `global_selection` never +commits these candidates; the backend compiles and prices them. + ## Choose strategies and models ### Strategy options From a3b6f15ee4cf862a605dcb7a1b0049257eb491b4 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:05:14 +0000 Subject: [PATCH 09/59] refactor: list only current-series heap alternatives in PlanSpace PlanSpace decides what to compute; the summary maintenance lifecycle decides node timing and the physical compiler reads it. The fixed-window and query-time Rate aggregation candidates only rewrote a finalization node's timing, so they are placement variants and no longer go through propose_for_root. The backend still calls those methods directly. SketchAlgorithmStrategy::propose_for_root now proposes only current-series TopK heaps, which rank identity-carrying rows the logical root lacks. The per-series Rate filter and the verbatim-assembly branch are removed: the latter is unnecessary because non-Aggregate roots are already assembled verbatim. RootPhysicalRealization stays so global selection never commits an unvalidated identity-carrying readout. Co-Authored-By: Claude Opus 5.5 --- crates/asap-aware-mapping/src/replacement.rs | 75 ++++------------- ...s.rs => planspace_series_identity_heap.rs} | 82 +++++++------------ 2 files changed, 46 insertions(+), 111 deletions(-) rename crates/asap-physical-operators/tests/{planspace_physical_alternatives.rs => planspace_series_identity_heap.rs} (73%) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 9f2444f9..fb697251 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -560,11 +560,11 @@ pub enum ReplacementProvenance { /// [`Replacement::ExactComposition`] with /// [`OperationPlacement::Maintenance`] (issue #171). ValueOperationAtIngestionTime, - /// A finalized whole-query result over a physical row representation - /// the logical children do not share (see - /// [`ReplacementStrategy::propose_for_root`]). Assembly uses it verbatim, - /// and default selection never commits it: only deployment compiles and - /// prices it. + /// A finalized whole-query result over rows carrying the PromQL series + /// identity, which the logical root does not expose (see + /// [`ReplacementStrategy::propose_for_root`]). Default selection never + /// commits it, because its readout must be validated and priced by + /// deployment; otherwise it would silently replace the logical plan. RootPhysicalRealization, } @@ -641,10 +641,11 @@ pub trait ReplacementStrategy { } } - /// Whole-query alternatives for a workload root under its end-to-end - /// `target`. These may change the root's physical row representation, so + /// Whole-query logical alternatives for a workload root under its + /// end-to-end `target`. These may need input rows the root does not expose + /// (for example, the PromQL series identity), so /// [`search_workload_with_targets`] asks only workload roots, once each. - /// Default: none. + /// They decide what to compute, never placement. Default: none. fn propose_for_root(&self, _root: &Rc, _target: &AccuracyTarget) -> Proposals { Proposals::default() } @@ -1720,31 +1721,19 @@ impl ReplacementStrategy for SketchAlgorithmStrategy<'_> { self.propose_with(target.root, None) } - /// Physical alternatives over rows that carry the complete PromQL series - /// identity: current-series TopK, fixed-window and query-time Rate - /// aggregation, and ordinary realizations over per-series Rate state. - /// Each is a finalized query result for the identity-carrying root. + /// Heap realizations of an instant-vector ranking (current-series TopK). + /// They rank rows that carry the complete PromQL series identity, which + /// the logical root does not expose, so each is a finalized query result + /// for the identity-carrying root. Placement variants (for example, + /// fixed-window or query-time Rate aggregation) are not listed here: the + /// lifecycle assigns timing and the physical compiler reads it. fn propose_for_root(&self, root: &Rc, target: &AccuracyTarget) -> Proposals { let Ok(typed) = asap_types::pre_asap::schema::with_promql_series_identity(root) else { return Proposals::default(); }; let typed = Rc::new(typed); let mut proposals = self.current_series_topk_candidates(&typed, target); - let mut direct = self.propose_with(&typed, None).candidates; - // Other identity-carrying realizations duplicate the logical root's. - direct.retain(|candidate| { - matches!(&candidate.replacement, Replacement::Summary(node) if has_series_rate_frontier(node)) - }); - // Rejections from the shared enumeration repeat the logical root's own. - let candidates = std::mem::take(&mut proposals.candidates) - .into_iter() - .chain(self.fixed_window_rate_candidates(&typed).candidates) - .chain( - self.query_time_rate_aggregation_candidates(&typed) - .candidates, - ) - .chain(direct); - for mut candidate in candidates { + for mut candidate in std::mem::take(&mut proposals.candidates) { let Replacement::Summary(node) = candidate.replacement else { continue; }; @@ -1764,30 +1753,6 @@ impl ReplacementStrategy for SketchAlgorithmStrategy<'_> { } } -/// Exact per-series Rate state over raw samples: the frontier at which -/// deployment binds complete, identity-keyed counter states. -fn has_series_rate_frontier(node: &SummaryNode) -> bool { - match &node.expr { - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), - reduction: Reduction::PerEntity, - child, - .. - } if matches!(&child.expr, SummaryExpr::KeepPreAsap(source) - if matches!(source.as_ref(), QueryExpr::TimeRange { .. })) => - { - true - } - SummaryExpr::ValueOperation { child, .. } | SummaryExpr::SummaryAgg { child, .. } => { - has_series_rate_frontier(child) - } - SummaryExpr::SummaryEstimate { summary_input, .. } => { - has_series_rate_frontier(summary_input) - } - _ => false, - } -} - /// A human-readable rationale for one candidate `Realization`, for /// [`ReplacementSubDAG::rationale`] text. fn describe_realization(intent: &AggIntent, realization: &Realization) -> String { @@ -5284,14 +5249,6 @@ impl<'a> GlobalSelection<'a> { if let Some(node) = self.assembled_nodes.borrow().get(&ptr) { return Ok(Rc::clone(node)); } - if let Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), - provenance: ReplacementProvenance::RootPhysicalRealization, - .. - }) = self.groups.get(&ptr).and_then(|selection| selection.chosen) - { - return Ok(Rc::clone(node)); - } let selected_composed_summary = self .groups .get(&ptr) diff --git a/crates/asap-physical-operators/tests/planspace_physical_alternatives.rs b/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs similarity index 73% rename from crates/asap-physical-operators/tests/planspace_physical_alternatives.rs rename to crates/asap-physical-operators/tests/planspace_series_identity_heap.rs index 04d684f7..2f8d5249 100644 --- a/crates/asap-physical-operators/tests/planspace_physical_alternatives.rs +++ b/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs @@ -1,6 +1,7 @@ -//! Physical row-representation alternatives are part of Planner's search space: -//! `enumerate_candidate_dags_for_root` lists them without a caller-side -//! series-identity pass, cost ranking, or workload Cartesian expansion. +//! Logical heap alternatives that need the PromQL series identity are part of +//! Planner's search space: `enumerate_candidate_dags_for_root` lists +//! current-series TopK heaps without a caller-side series-identity pass, cost +//! ranking, or workload Cartesian expansion. Placement variants are not listed. use asap_aware_mapping::{ accuracy::{AccuracyEvidenceProvider, DefaultAccuracyModel, PropagationStats}, cost_model::DefaultCostModel, @@ -8,8 +9,7 @@ use asap_aware_mapping::{ search_workload_with_targets, Proposals, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_physical_operators::physical_planner::promql_rows::{ - compile_current_series_readout, compile_fixed_window_rate_aggregation, compile_rate_ranking, - SERIES_IDENTITY_COLUMN, + compile_current_series_readout, SERIES_IDENTITY_COLUMN, }; use planner_types::{ post_asap::*, @@ -139,7 +139,7 @@ fn carries_identity(dag: &Dag) -> bool { }) } -/// Shared acceptance checks; returns the added physical alternatives. +/// Shared acceptance checks; returns the added identity-carrying alternatives. fn added_alternatives(query: &str, accuracy: AccuracyTarget) -> Vec> { let (full, logical) = inventories(query, accuracy); for (index, dag) in full.iter().enumerate() { @@ -157,59 +157,35 @@ fn added_alternatives(query: &str, accuracy: AccuracyTarget) -> Vec 0); assert_eq!(count(2), count(1)); } From c976d5ca09d3991de77743e21a30015b1feb29a2 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:05:14 +0000 Subject: [PATCH 10/59] docs: limit per-root PlanSpace alternatives to current-series heaps Co-Authored-By: Claude Opus 5.5 --- .../physical-planning-and-deployment.md | 13 ++++++------- docs/develop_docs/library-api.md | 17 ++++++++--------- 2 files changed, 14 insertions(+), 16 deletions(-) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index fad645d4..7e7b3bd0 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -79,13 +79,12 @@ sample values does not preserve instant-vector semantics. Replacement, rank decrease, expiry, grouping and the required approximation guarantee must be validated before admitting that physical candidate. -Whole-query alternatives that need a different physical row representation are -part of Planner's candidate space. For PromQL, Planner resolves rows that carry -the complete series identity, then proposes the families above for that root: -fixed-window Rate aggregation, query-time ranking or aggregation over -per-series Rate readouts, and current-series TopK. These are listed per root -with the other candidates, unranked. Queries without such a realization are -unchanged. +Planner's candidate space decides what to compute, not placement. For an +instant-vector PromQL TopK, Planner resolves rows that carry the complete series +identity and lists the current-series heap realizations per root with the other +candidates, unranked. Precompute or query-time placement of Rate and grouped Sum +is not a separate Planner candidate: the summary maintenance lifecycle assigns +each node's timing, and the physical compiler reads it. This is the target ownership contract. A backend path that still reconstructs operators from logical candidates has not completed this integration. diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index e4845733..92288067 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -280,15 +280,14 @@ partial inventory. For PromQL roots that carry a target, `search_workload_with_targets` also asks each strategy's `ReplacementStrategy::propose_for_root`. `SketchAlgorithmStrategy` -answers with physical alternatives over rows carrying the complete series -identity (`$promql_series_identity`): current-series TopK, fixed-window and -query-time Rate aggregation, and realizations over per-series Rate state. They -are finalized, deduplicated, and marked -`ReplacementProvenance::RootPhysicalRealization`. Callers do not apply -`with_series_identity` or call the proposal methods themselves. Compile each -alternative with the matching `promql_rows::compile_*` function; queries without -such an alternative keep their previous inventory. `global_selection` never -commits these candidates; the backend compiles and prices them. +answers an instant-vector TopK with current-series heap realizations over rows +carrying the complete series identity (`$promql_series_identity`). They are +finalized, deduplicated, and marked `ReplacementProvenance::RootPhysicalRealization`. +Callers do not apply `with_series_identity` themselves. Compile each with +`promql_rows::compile_current_series_readout`; other queries keep their previous +inventory. `global_selection` never commits these candidates; the backend +compiles and prices them. PlanSpace lists no placement variants: node timing +comes from the summary maintenance lifecycle. ## Choose strategies and models From 166f257d72ebc521780cae38fea6a75672e2e939 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 22:18:32 +0000 Subject: [PATCH 11/59] feat: expose summary-maintenance lifecycle candidates for explicit binding Split plan_summary_maintenance_lifecycles into enumeration and selection so a deployment can price every lifecycle alternative per unique summary state and bind its own choice. Planner selection is unchanged: it now enumerates and then selects the cheapest complete combination through the same path. SummaryMaintenanceLifecycleCandidates::select validates that each choice is an alternative Planner could select, enforces schedule compatibility, and obtains window frameworks and totals from the same complete-candidate hook. Co-Authored-By: Claude Opus 5.5 --- crates/asap-aware-mapping/src/lib.rs | 15 +- .../src/summary_maintenance_lifecycle.rs | 715 ++++++++++++++---- 2 files changed, 587 insertions(+), 143 deletions(-) diff --git a/crates/asap-aware-mapping/src/lib.rs b/crates/asap-aware-mapping/src/lib.rs index 76c3a88c..716c6ee4 100644 --- a/crates/asap-aware-mapping/src/lib.rs +++ b/crates/asap-aware-mapping/src/lib.rs @@ -221,13 +221,14 @@ pub use summary_maintenance_dag_export::{ }; pub use summary_maintenance_lifecycle::{ assemble_selected_dag_with_summary_maintenance_lifecycles, - global_selection_with_summary_maintenance_lifecycles, plan_summary_maintenance_lifecycles, - SummaryMaintenanceCapabilities, SummaryMaintenanceDeployment, - SummaryMaintenanceLifecycleAlternative, SummaryMaintenanceLifecycleAssemblyError, - SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecycleCostInputs, - SummaryMaintenanceLifecyclePlan, SummaryMaintenanceLifecyclePlanError, - SummaryMaintenanceLifecycleRejection, SummaryMaintenanceLifecycleSelectionError, - WorkloadDemand, + enumerate_summary_maintenance_lifecycles, global_selection_with_summary_maintenance_lifecycles, + plan_summary_maintenance_lifecycles, SummaryMaintenanceCapabilities, + SummaryMaintenanceDeployment, SummaryMaintenanceLifecycleAlternative, + SummaryMaintenanceLifecycleAssemblyError, SummaryMaintenanceLifecycleCandidates, + SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecycleChoiceError, + SummaryMaintenanceLifecycleCostInputs, SummaryMaintenanceLifecyclePlan, + SummaryMaintenanceLifecyclePlanError, SummaryMaintenanceLifecycleRejection, + SummaryMaintenanceLifecycleSelectionError, WorkloadDemand, }; pub use topk_reuse::TopKLimitReuseStrategy; diff --git a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs index b0d9d79f..7e40645f 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs @@ -287,6 +287,169 @@ pub enum SummaryMaintenanceLifecycleSelectionError { SummaryMaintenance(#[from] SummaryMaintenanceLifecyclePlanError), } +/// Every lifecycle alternative for each unique summary state of one fixed +/// root, before any lifecycle is chosen. +/// +/// Planner selection ([`plan_summary_maintenance_lifecycles`]) and a +/// deployment's explicit choice ([`Self::select`]) both finish from this value, +/// so they produce the same [`SummaryMaintenanceLifecyclePlan`] shape. +pub struct SummaryMaintenanceLifecycleCandidates<'a> { + /// Unselected plan: deployments carry alternatives but no guarantee or + /// window framework. + plan: SummaryMaintenanceLifecyclePlan, + components: Vec, + arrival: DataArrival, + required_accuracy: Vec, + cost_model: &'a dyn CostModel, + comparison_target: Option<&'a QueryExpr>, +} + +/// Why an explicit per-state lifecycle choice cannot be bound. +#[derive(Debug, thiserror::Error, PartialEq)] +pub enum SummaryMaintenanceLifecycleChoiceError { + #[error("summary {0:?} is not a deployment of this root")] + UnknownSummary(PostAsapNodeId), + #[error("summary {0:?} is chosen more than once")] + DuplicateChoice(PostAsapNodeId), + #[error("summary {0:?} has no chosen lifecycle")] + MissingChoice(PostAsapNodeId), + #[error("chosen lifecycle is not an enumerated alternative of summary {0:?}")] + NotAnAlternative(PostAsapNodeId), + #[error("chosen lifecycle of summary {post_asap_node_id:?} is rejected: {rejection:?}")] + Rejected { + post_asap_node_id: PostAsapNodeId, + rejection: Option, + }, + #[error("summary states on one maintenance path have different evaluation schedules")] + IncompatibleEvaluationSchedules, + #[error("the cost model supplied no complete estimate for the chosen combination")] + NoCompleteEstimate, +} + +impl SummaryMaintenanceLifecycleCandidates<'_> { + /// One entry per unique reachable `SummaryAgg`, with every alternative + /// and its rejection; no lifecycle or window framework is selected. + pub fn deployments(&self) -> &[SummaryMaintenanceDeployment] { + &self.plan.deployments + } + + /// Guarantee that binding `lifecycle` would attach under this workload's + /// data arrival, so a caller can price an alternative before choosing it. + pub fn guarantee( + &self, + lifecycle: &SummaryMaintenanceLifecycle, + ) -> SummaryMaintenanceLifecycleGuarantee { + lifecycle_guarantee(lifecycle, self.arrival) + } + + fn context(&self) -> CompleteCostContext<'_> { + CompleteCostContext { + root: &self.plan.root, + components: &self.components, + cost_model: self.cost_model, + comparison_target: self.comparison_target, + horizon: self.plan.horizon, + expected_reads: self.plan.expected_reads, + required_accuracy: &self.required_accuracy, + } + } + + fn finish( + mut self, + estimate: Option, + ) -> SummaryMaintenanceLifecyclePlan { + if let Some(estimate) = estimate { + self.plan.summary_total_cost = Some(estimate.cost); + self.plan.selected_window_implementation_id = estimate.physical_plan_id; + self.plan.window_accuracy_guarantee = estimate.window_accuracy_guarantee; + } + self.plan + } + + /// Planner's choice: the cheapest complete combination of eligible + /// alternatives. + fn select_cheapest(mut self) -> SummaryMaintenanceLifecyclePlan { + let estimate = select_complete_lifecycle_combination( + &self.plan.root, + &mut self.plan.deployments, + &self.components, + self.arrival, + self.cost_model, + self.comparison_target, + self.plan.horizon, + self.plan.expected_reads, + &self.required_accuracy, + ); + self.finish(estimate) + } + + /// Bind one caller-chosen lifecycle per summary state. Each choice must be + /// an alternative Planner itself could select; the complete estimate is + /// then obtained exactly as for Planner selection, so window framework and + /// cost are the model's and unknown cost is never replaced by zero. + pub fn select( + mut self, + choices: &[(PostAsapNodeId, SummaryMaintenanceLifecycle)], + ) -> Result { + use SummaryMaintenanceLifecycleChoiceError as E; + let deployments = &self.plan.deployments; + let mut chosen: Vec> = + vec![None; deployments.len()]; + let context = self.context(); + for (id, lifecycle) in choices { + let index = deployments + .iter() + .position(|deployment| deployment.post_asap_node_id == *id) + .ok_or(E::UnknownSummary(*id))?; + if chosen[index].is_some() { + return Err(E::DuplicateChoice(*id)); + } + let alternative = deployments[index] + .alternatives + .iter() + .find(|alternative| alternative.summary_maintenance_lifecycle == *lifecycle) + .ok_or(E::NotAnAlternative(*id))?; + if !context.eligible(alternative) { + return Err(E::Rejected { + post_asap_node_id: *id, + rejection: alternative.rejection.clone(), + }); + } + chosen[index] = Some(alternative); + } + let selected = chosen + .into_iter() + .enumerate() + .map(|(index, alternative)| { + let alternative = + alternative.ok_or(E::MissingChoice(deployments[index].post_asap_node_id))?; + Ok(( + index, + lifecycle_guarantee(&alternative.summary_maintenance_lifecycle, self.arrival), + // Reached only for costed alternatives or when the + // complete hook is authoritative, matching Planner search. + alternative.total_cost.unwrap_or(Cost::ZERO), + )) + }) + .collect::, E>>()?; + if selected.is_empty() { + return Ok(self.finish(None)); + } + if !context.schedules_compatible(&selected) { + return Err(E::IncompatibleEvaluationSchedules); + } + let estimate = context + .estimate(deployments, &selected) + .ok_or(E::NoCompleteEstimate)?; + let guarantees = selected + .into_iter() + .map(|(index, guarantee, _)| (index, guarantee)) + .collect(); + apply_selection(&mut self.plan.deployments, guarantees, &estimate); + Ok(self.finish(Some(estimate))) + } +} + /// Workload-wide evidence derived specifically for summary-maintenance /// lifecycle enumeration and costing. /// @@ -333,7 +496,30 @@ pub fn plan_summary_maintenance_lifecycles( capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &dyn CostModel, ) -> Result { - plan_summary_maintenance_lifecycles_with_profile( + Ok(enumerate_summary_maintenance_lifecycles( + root, + demand, + now_ms, + horizon, + capabilities, + cost_model, + )? + .select_cheapest()) +} + +/// Validate a materialized plan and enumerate lifecycle alternatives for each +/// unique summary state without choosing one. A deployment that prices the +/// alternatives itself binds its choice with +/// [`SummaryMaintenanceLifecycleCandidates::select`]. +pub fn enumerate_summary_maintenance_lifecycles<'a>( + root: Rc, + demand: WorkloadDemand<'_>, + now_ms: u64, + horizon: Option, + capabilities: SummaryMaintenanceLifecycleCapabilities, + cost_model: &'a dyn CostModel, +) -> Result, SummaryMaintenanceLifecyclePlanError> { + enumerate_with_profile( root, demand, now_ms, @@ -349,16 +535,16 @@ pub fn plan_summary_maintenance_lifecycles( /// eligibility and data-arrival facts; `profile` supplies effective uses after /// DAG path multiplicity has been propagated by `PlanSpace`. #[expect(clippy::too_many_arguments, reason = "internal bound planning context")] -fn plan_summary_maintenance_lifecycles_with_profile( +fn enumerate_with_profile<'a>( root: Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, capabilities: SummaryMaintenanceLifecycleCapabilities, - cost_model: &dyn CostModel, + cost_model: &'a dyn CostModel, profile: Option, - comparison_target: Option<&QueryExpr>, -) -> Result { + comparison_target: Option<&'a QueryExpr>, +) -> Result, SummaryMaintenanceLifecyclePlanError> { demand.workload.validate()?; if let Some(data) = demand.data_workload { data.validate()?; @@ -392,7 +578,7 @@ fn plan_summary_maintenance_lifecycles_with_profile( collect_summary_aggs(&root, &mut HashSet::new(), &mut summaries); let node_ids = compile_post_asap_dag_with_node_ids(&root)?.node_ids; let components = summary_state_components(&summaries); - let mut deployments: Vec = summaries + let deployments: Vec = summaries .into_iter() .map(|summary| { let alternatives = alternatives_for( @@ -413,37 +599,26 @@ fn plan_summary_maintenance_lifecycles_with_profile( } }) .collect(); - let complete_estimate = select_complete_lifecycle_combination( - &root, - &mut deployments, - &components, - facts.arrival, + let selected_raw_recompute = matches!(root.expr, SummaryExpr::KeepPreAsap(_)); + Ok(SummaryMaintenanceLifecycleCandidates { + plan: SummaryMaintenanceLifecyclePlan { + root, + deployments, + horizon, + evaluation_rate: facts.evaluation_rate, + update_rate: facts.update_rate, + expected_reads: facts.reads, + selected_raw_recompute, + selected_window_implementation_id: None, + summary_total_cost: None, + window_accuracy_guarantee: None, + raw_recompute_total_cost: None, + }, + components, + arrival: facts.arrival, + required_accuracy: facts.required_accuracy, cost_model, comparison_target, - horizon, - facts.reads, - &facts.required_accuracy, - ); - let summary_total_cost = complete_estimate.as_ref().map(|estimate| estimate.cost); - let selected_window_implementation_id = complete_estimate - .as_ref() - .and_then(|estimate| estimate.physical_plan_id.clone()); - let window_accuracy_guarantee = complete_estimate - .as_ref() - .and_then(|estimate| estimate.window_accuracy_guarantee.clone()); - let selected_raw_recompute = matches!(root.expr, SummaryExpr::KeepPreAsap(_)); - Ok(SummaryMaintenanceLifecyclePlan { - root, - deployments, - horizon, - evaluation_rate: facts.evaluation_rate, - update_rate: facts.update_rate, - expected_reads: facts.reads, - selected_raw_recompute, - selected_window_implementation_id, - summary_total_cost, - window_accuracy_guarantee, - raw_recompute_total_cost: None, }) } @@ -483,7 +658,7 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( continue; }; costs.finalize_target(&group.target); - let plan = plan_summary_maintenance_lifecycles_with_profile( + let plan = enumerate_with_profile( Rc::clone(summary), WorkloadDemand { workload, @@ -496,7 +671,8 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( cost_model, Some(profiles.for_target(&group.target)), Some(&group.target), - )?; + )? + .select_cheapest(); let raw = plan .expected_reads .and_then(|reads| cost_model.raw_query_recompute_total_cost(&group.target, reads)); @@ -529,7 +705,7 @@ pub fn assemble_selected_dag_with_summary_maintenance_lifecycles( selection .assemble_selected_dag(target)? .map(|root| { - let mut plan = plan_summary_maintenance_lifecycles_with_profile( + let mut plan = enumerate_with_profile( root, demand, now_ms, @@ -538,7 +714,8 @@ pub fn assemble_selected_dag_with_summary_maintenance_lifecycles( cost_model, None, Some(target), - )?; + )? + .select_cheapest(); plan.raw_recompute_total_cost = plan .expected_reads .and_then(|reads| cost_model.raw_query_recompute_total_cost(target, reads)); @@ -1104,6 +1281,99 @@ fn summary_state_components(summaries: &[Rc]) -> Vec { .collect() } +/// Inputs shared by every complete lifecycle-combination evaluation of one +/// root, whether Planner searches combinations or a caller supplies one. +struct CompleteCostContext<'a> { + root: &'a SummaryNode, + components: &'a [usize], + cost_model: &'a dyn CostModel, + comparison_target: Option<&'a QueryExpr>, + horizon: Option, + expected_reads: Option, + required_accuracy: &'a [AccuracyTarget], +} + +impl CompleteCostContext<'_> { + /// Planner's own admission rule for one alternative. Uncosted alternatives + /// are admitted only when the complete-candidate hook is authoritative. + fn eligible(&self, alternative: &SummaryMaintenanceLifecycleAlternative) -> bool { + alternative.selectable() + || (self + .cost_model + .complete_summary_candidate_estimate_covers_lifecycle_costs() + && alternative.rejection + == Some(SummaryMaintenanceLifecycleRejection::MissingCostEvidence)) + } + + /// `selected` holds one entry per deployment, in deployment order. + fn schedules_compatible( + &self, + selected: &[(usize, SummaryMaintenanceLifecycleGuarantee, Cost)], + ) -> bool { + !selected.iter().enumerate().any(|(left, (_, a, _))| { + selected.iter().enumerate().any(|(right, (_, b, _))| { + self.components[left] == self.components[right] + && a.evaluation_schedule != b.evaluation_schedule + }) + }) + } + + fn estimate( + &self, + deployments: &[SummaryMaintenanceDeployment], + selected: &[(usize, SummaryMaintenanceLifecycleGuarantee, Cost)], + ) -> Option { + if !self.schedules_compatible(selected) { + return None; + } + let costed: Vec<_> = selected + .iter() + .map(|(index, guarantee, cost)| CostedSummaryDeployment { + summary: &deployments[*index].summary, + guarantee, + selected_cost: *cost, + }) + .collect(); + let estimate = self.cost_model.complete_summary_candidate_estimate( + self.root, + self.comparison_target, + &costed, + self.horizon, + self.expected_reads, + self.required_accuracy, + )?; + (estimate.window_frameworks.len() == deployments.len()).then_some(estimate) + } +} + +fn lifecycle_guarantee( + lifecycle: &SummaryMaintenanceLifecycle, + arrival: DataArrival, +) -> SummaryMaintenanceLifecycleGuarantee { + SummaryMaintenanceLifecycleGuarantee { + summary_maintenance_mode: maintenance_mode(lifecycle, arrival), + evaluation_schedule: evaluation_schedule(lifecycle, arrival), + summary_maintenance_lifecycle: lifecycle.clone(), + output_representation: OutputRepresentation::SummaryState, + } +} + +fn apply_selection( + deployments: &mut [SummaryMaintenanceDeployment], + guarantees: Vec<(usize, SummaryMaintenanceLifecycleGuarantee)>, + estimate: &CompleteSummaryCandidateEstimate, +) { + for (index, guarantee) in guarantees { + deployments[index].summary_maintenance_lifecycle_guarantee = Some(guarantee); + } + for (deployment, framework) in deployments + .iter_mut() + .zip(estimate.window_frameworks.iter().cloned()) + { + deployment.selected_window_framework = framework; + } +} + #[expect(clippy::too_many_arguments, reason = "complete combination context")] fn select_complete_lifecycle_combination( root: &SummaryNode, @@ -1120,82 +1390,48 @@ fn select_complete_lifecycle_combination( if deployments.is_empty() { return None; } + let context = CompleteCostContext { + root, + components, + cost_model, + comparison_target, + horizon, + expected_reads, + required_accuracy, + }; // The whole-candidate hook is intentionally arbitrary, so partial costs // cannot soundly prune the search. Bound exhaustive enumeration and fail // closed instead of allowing an adversarial DAG to consume exponential // planner time. - let complete_costing = cost_model.complete_summary_candidate_estimate_covers_lifecycle_costs(); - let eligible = |alternative: &SummaryMaintenanceLifecycleAlternative| { - alternative.selectable() - || (complete_costing - && alternative.rejection - == Some(SummaryMaintenanceLifecycleRejection::MissingCostEvidence)) - }; let combinations = deployments .iter() .try_fold(1_usize, |product, deployment| { let selectable = deployment .alternatives .iter() - .filter(|alternative| eligible(alternative)) + .filter(|alternative| context.eligible(alternative)) .count(); product.checked_mul(selectable) })?; if combinations == 0 || combinations > MAX_COMPLETE_LIFECYCLE_COMBINATIONS { return None; } - #[expect( - clippy::too_many_arguments, - reason = "recursive lifecycle-combination search state" - )] + type Best = Option<( + CompleteSummaryCandidateEstimate, + Vec<(usize, SummaryMaintenanceLifecycleGuarantee)>, + )>; fn visit( index: usize, - root: &SummaryNode, + context: &CompleteCostContext<'_>, deployments: &[SummaryMaintenanceDeployment], - components: &[usize], arrival: DataArrival, - cost_model: &dyn CostModel, - comparison_target: Option<&QueryExpr>, - horizon: Option, - expected_reads: Option, - required_accuracy: &[AccuracyTarget], - complete_costing: bool, selected: &mut Vec<(usize, SummaryMaintenanceLifecycleGuarantee, Cost)>, - best: &mut Option<( - CompleteSummaryCandidateEstimate, - Vec<(usize, SummaryMaintenanceLifecycleGuarantee)>, - )>, + best: &mut Best, ) { if index == deployments.len() { - if selected.iter().enumerate().any(|(left, (_, a, _))| { - selected.iter().enumerate().any(|(right, (_, b, _))| { - components[left] == components[right] - && a.evaluation_schedule != b.evaluation_schedule - }) - }) { - return; - } - let costed: Vec<_> = selected - .iter() - .map(|(index, guarantee, cost)| CostedSummaryDeployment { - summary: &deployments[*index].summary, - guarantee, - selected_cost: *cost, - }) - .collect(); - let Some(estimate) = cost_model.complete_summary_candidate_estimate( - root, - comparison_target, - &costed, - horizon, - expected_reads, - required_accuracy, - ) else { + let Some(estimate) = context.estimate(deployments, selected) else { return; }; - if estimate.window_frameworks.len() != deployments.len() { - return; - } if best .as_ref() .is_none_or(|(best_estimate, _)| estimate.cost.0 < best_estimate.cost.0) @@ -1213,40 +1449,14 @@ fn select_complete_lifecycle_combination( for alternative in deployments[index] .alternatives .iter() - .filter(|alternative| { - alternative.selectable() - || (complete_costing - && alternative.rejection - == Some(SummaryMaintenanceLifecycleRejection::MissingCostEvidence)) - }) + .filter(|alternative| context.eligible(alternative)) { - let lifecycle = alternative.summary_maintenance_lifecycle.clone(); - let guarantee = SummaryMaintenanceLifecycleGuarantee { - summary_maintenance_mode: maintenance_mode(&lifecycle, arrival), - evaluation_schedule: evaluation_schedule(&lifecycle, arrival), - summary_maintenance_lifecycle: lifecycle, - output_representation: OutputRepresentation::SummaryState, - }; selected.push(( index, - guarantee, + lifecycle_guarantee(&alternative.summary_maintenance_lifecycle, arrival), alternative.total_cost.unwrap_or(Cost::ZERO), )); - visit( - index + 1, - root, - deployments, - components, - arrival, - cost_model, - comparison_target, - horizon, - expected_reads, - required_accuracy, - complete_costing, - selected, - best, - ); + visit(index + 1, context, deployments, arrival, selected, best); selected.pop(); } } @@ -1254,29 +1464,14 @@ fn select_complete_lifecycle_combination( let mut best = None; visit( 0, - root, + &context, deployments, - components, arrival, - cost_model, - comparison_target, - horizon, - expected_reads, - required_accuracy, - complete_costing, &mut Vec::new(), &mut best, ); let (estimate, guarantees) = best?; - for (index, guarantee) in guarantees { - deployments[index].summary_maintenance_lifecycle_guarantee = Some(guarantee); - } - for (deployment, framework) in deployments - .iter_mut() - .zip(estimate.window_frameworks.iter().cloned()) - { - deployment.selected_window_framework = framework; - } + apply_selection(deployments, guarantees, &estimate); Some(estimate) } @@ -2402,4 +2597,252 @@ mod tests { assert_eq!(batch.evaluation_rate, None); assert_eq!(batch.one_shot_consumers, 1); } + + fn continuous_candidates<'a>( + workload: &QueryWorkload, + data: &DataWorkload, + model: &'a dyn CostModel, + ) -> SummaryMaintenanceLifecycleCandidates<'a> { + enumerate_summary_maintenance_lifecycles( + summary(), + WorkloadDemand::new_with_data(workload, data, &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities { + supports_shared: false, + ..SummaryMaintenanceLifecycleCapabilities::ALL + }, + model, + ) + .unwrap() + } + + fn choose( + candidates: &SummaryMaintenanceLifecycleCandidates<'_>, + lifecycle: SummaryMaintenanceLifecycle, + ) -> Vec<(PostAsapNodeId, SummaryMaintenanceLifecycle)> { + candidates + .deployments() + .iter() + .map(|deployment| (deployment.post_asap_node_id, lifecycle.clone())) + .collect() + } + + // Enumeration reports all four lifecycle kinds with their rejections and + // selects nothing. + #[test] + fn enumeration_exposes_every_lifecycle_without_selecting() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let candidates = continuous_candidates(&workload, &data, &UnitCosts); + let [deployment] = candidates.deployments() else { + panic!("one summary state"); + }; + assert_eq!(deployment.summary_maintenance_lifecycle_guarantee, None); + assert_eq!(deployment.selected_window_framework, None); + let outcome: Vec<_> = deployment + .alternatives + .iter() + .map(|alternative| { + ( + &alternative.summary_maintenance_lifecycle, + alternative.rejection.clone(), + alternative.total_cost.is_some(), + ) + }) + .collect(); + assert!(matches!( + outcome.as_slice(), + [ + (SummaryMaintenanceLifecycle::Ephemeral, None, true), + ( + SummaryMaintenanceLifecycle::Prepared { .. }, + Some(SummaryMaintenanceLifecycleRejection::RequiresPredictableOneTimeQuery), + false + ), + ( + SummaryMaintenanceLifecycle::Shared { .. }, + Some(SummaryMaintenanceLifecycleRejection::UnsupportedByRuntime), + false + ), + ( + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + None, + true + ), + ] + )); + let guarantee = candidates.guarantee(&SummaryMaintenanceLifecycle::ContinuouslyMaintained); + assert_eq!( + guarantee.summary_maintenance_mode, + SummaryMaintenanceMode::Incremental + ); + assert_eq!(guarantee.evaluation_schedule, EvaluationSchedule::PerUpdate); + } + + // Explicitly choosing Planner's own selection reproduces Planner's plan. + #[test] + fn explicit_choice_of_planner_selection_reproduces_planner_plan() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let planned = plan_summary_maintenance_lifecycles( + summary(), + WorkloadDemand::new_with_data(&workload, &data, &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities { + supports_shared: false, + ..SummaryMaintenanceLifecycleCapabilities::ALL + }, + &UnitCosts, + ) + .unwrap(); + let candidates = continuous_candidates(&workload, &data, &UnitCosts); + let choice = choose( + &candidates, + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + ); + let chosen = candidates.select(&choice).unwrap(); + assert_eq!(format!("{chosen:?}"), format!("{planned:?}")); + } + + // A deployment may bind a legal alternative Planner's estimate does not + // prefer; the plan carries that alternative's guarantee and cost. + #[test] + fn explicit_choice_may_bind_a_costlier_legal_alternative() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let candidates = continuous_candidates(&workload, &data, &UnitCosts); + let ephemeral_cost = candidates.deployments()[0].alternatives[0].total_cost; + let choice = choose(&candidates, SummaryMaintenanceLifecycle::Ephemeral); + let plan = candidates.select(&choice).unwrap(); + assert_eq!( + selected_summary_maintenance_lifecycle(&plan.deployments[0]), + Some(&SummaryMaintenanceLifecycle::Ephemeral) + ); + assert_eq!(plan.summary_total_cost, ephemeral_cost); + } + + // Choices that Planner could not select, or that do not cover exactly the + // enumerated states, are refused rather than bound. + #[test] + fn explicit_choice_rejects_illegal_or_incomplete_choices() { + use SummaryMaintenanceLifecycleChoiceError as E; + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let select = |model: &dyn CostModel, choice: &dyn Fn(PostAsapNodeId) -> Vec<_>| { + let candidates = continuous_candidates(&workload, &data, model); + let id = candidates.deployments()[0].post_asap_node_id; + (id, candidates.select(&choice(id)).unwrap_err()) + }; + let shared = SummaryMaintenanceLifecycle::Shared { + retention: DurationMs(10_000), + }; + let (id, error) = select(&UnitCosts, &|id| vec![(id, shared.clone())]); + assert_eq!( + error, + E::Rejected { + post_asap_node_id: id, + rejection: Some(SummaryMaintenanceLifecycleRejection::UnsupportedByRuntime), + } + ); + let continuous = SummaryMaintenanceLifecycle::ContinuouslyMaintained; + let (id, error) = select(&crate::cost_model::DefaultCostModel, &|id| { + vec![(id, SummaryMaintenanceLifecycle::Ephemeral)] + }); + assert_eq!( + error, + E::Rejected { + post_asap_node_id: id, + rejection: Some(SummaryMaintenanceLifecycleRejection::MissingCostEvidence), + } + ); + let (id, error) = select(&UnitCosts, &|id| { + vec![( + id, + SummaryMaintenanceLifecycle::Shared { + retention: DurationMs(1), + }, + )] + }); + assert_eq!(error, E::NotAnAlternative(id)); + let (id, error) = select(&UnitCosts, &|_| vec![]); + assert_eq!(error, E::MissingChoice(id)); + let (id, error) = select(&UnitCosts, &|id| { + vec![(id, continuous.clone()), (id, continuous.clone())] + }); + assert_eq!(error, E::DuplicateChoice(id)); + let (_, error) = select(&UnitCosts, &|_| { + vec![(PostAsapNodeId(u32::MAX), continuous.clone())] + }); + assert_eq!(error, E::UnknownSummary(PostAsapNodeId(u32::MAX))); + } + + // Nested states on one maintenance path must share an evaluation schedule. + #[test] + fn explicit_choice_rejects_incompatible_nested_schedules() { + let workload = workload(vec![], vec![repeating()], continuous(1_000, 20_000)); + let candidates = enumerate_summary_maintenance_lifecycles( + nested_summary(), + WorkloadDemand::new_with_data(&workload, &continuous(1_000, 20_000), &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &IncompatibleNestedCosts, + ) + .unwrap(); + let [outer, inner] = candidates.deployments() else { + panic!("two summary states"); + }; + let choice = vec![ + ( + outer.post_asap_node_id, + SummaryMaintenanceLifecycle::Ephemeral, + ), + ( + inner.post_asap_node_id, + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + ), + ]; + assert_eq!( + candidates.select(&choice).unwrap_err(), + SummaryMaintenanceLifecycleChoiceError::IncompatibleEvaluationSchedules + ); + } + + // A multi-summary root yields one candidate entry per unique state, with + // a shared `Rc` state listed once. + #[test] + fn enumeration_lists_each_unique_summary_state_once() { + let shared = summary(); + let root = Rc::new(SummaryNode { + expr: SummaryExpr::SummaryMerge { + timing: asap_types::post_asap::ExecutionTiming::IngestionTime, + children: vec![Rc::clone(&shared), Rc::clone(&shared), summary()], + }, + schema: shared.schema.clone(), + guarantee: None, + }); + let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); + let candidates = enumerate_summary_maintenance_lifecycles( + root, + WorkloadDemand::new_with_data(&workload, &at_rest(), &[0]), + 1_000, + None, + SummaryMaintenanceLifecycleCapabilities::ALL, + &UnitCosts, + ) + .unwrap(); + let ids: HashSet<_> = candidates + .deployments() + .iter() + .map(|deployment| deployment.post_asap_node_id) + .collect(); + assert_eq!(candidates.deployments().len(), 2); + assert_eq!(ids.len(), 2); + assert!(candidates + .deployments() + .iter() + .any(|deployment| Rc::ptr_eq(&deployment.summary, &shared))); + } } From 07244ea8244be797d7d0a81d8f4d40994a148d87 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 22:18:34 +0000 Subject: [PATCH 12/59] docs: describe lifecycle candidate enumeration and binding Co-Authored-By: Claude Opus 5.5 --- .../physical-planning-and-deployment.md | 15 +++++++++++++++ docs/develop_docs/library-api.md | 9 +++++++++ 2 files changed, 24 insertions(+) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 7e7b3bd0..2aad9ca7 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -236,6 +236,21 @@ using workload demand, window/freshness requirements and supported physical implementations. Backend selection uses runtime feasibility and cost after physical compilation. The following example follows one candidate. +Candidate generation and selection are separate steps. For every unique summary +state, enumeration reports each lifecycle (ephemeral, prepared, shared, +continuously maintained) as legal, with a Planner cost or explicitly unknown +cost, or as rejected with a reason. Planner does not remove a legal alternative +because its own estimate prefers another. A deployment prices the legal +alternatives over the whole workload, counting shared state once, and binds one +lifecycle per state. Binding checks that the choice is legal and that states on +one maintenance path share an evaluation schedule. An alternative whose cost is +unknown can be bound only when the deployment's cost model is authoritative for +complete-candidate cost; unknown cost is never treated as zero. It then yields the same +lifecycle guarantee and window framework the physical compiler consumes when +Planner selects. Planner's own cheapest-alternative selection remains available +for callers without deployment pricing. The window framework is decided for the +complete combination, not for one alternative in isolation. + For the running example, assume it selects: ```text diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index 92288067..cb541b94 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -633,6 +633,8 @@ that prepared or retained shared state is supported. | `plan_summary_maintenance_lifecycles` | Assembled logical DAG root, `WorkloadDemand`, `now_ms`, optional horizon, runtime capabilities, cost model | `Result` for that fixed root; does not revisit all semantic candidates | | `global_selection_with_summary_maintenance_lifecycles` | `PlanSpace`, workload/root-entry associations, time, horizon, capabilities, cost model | Lifecycle-aware compatible selection/error, using eligible cost evidence | | `assemble_selected_dag_with_summary_maintenance_lifecycles` | Selection, target root and lifecycle context | Optional lifecycle plan/error; attaches state deployment decisions | +| `enumerate_summary_maintenance_lifecycles` | Same inputs as `plan_summary_maintenance_lifecycles` | `SummaryMaintenanceLifecycleCandidates`: per unique state, every alternative with its cost or rejection; nothing selected. `guarantee(&lifecycle)` gives the mode/schedule that alternative would carry | +| `SummaryMaintenanceLifecycleCandidates::select(choices)` | One `(PostAsapNodeId, SummaryMaintenanceLifecycle)` per state, copied from `deployments()` | The same `SummaryMaintenanceLifecyclePlan` Planner selection would produce for that combination, or `SummaryMaintenanceLifecycleChoiceError` when a choice is unknown, missing, duplicated, rejected, schedule-incompatible, or not completely estimable | Inspect `deployments`, their selected lifecycle/alternatives/rejections, `selected_raw_recompute`, and optional summary/raw costs. Success of a function @@ -644,6 +646,13 @@ lifecycle analysis after structural selection can evaluate the selected root, but does not make the earlier selection lifecycle-optimal. An application may consume ranked candidates and perform this comparison downstream instead. +A deployment that prices lifecycles itself calls +`enumerate_summary_maintenance_lifecycles`, prices the alternatives, and binds +its choice with `select`. A choice is accepted only if Planner could select it: +an alternative with `MissingCostEvidence` is accepted only when the cost model's +complete-candidate hook covers lifecycle costs. Window frameworks and totals come +from that hook, as in Planner selection. + ## Optional whole-plan selection and DAG assembly ### What does global selection mean? From 590ec6d427af8a26739f08b673de4d680b71fb80 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:10:14 +0000 Subject: [PATCH 13/59] feat: derive Post-ASAP execution timing from a lifecycle plan A chosen summary-maintenance lifecycle did not reach the DAG: timing was fixed by realization strategies. SummaryMaintenanceLifecyclePlan:: execution_timed_dag assigns every node's timing from the selection: retained (non-Ephemeral) states and their inputs at ingestion time, everything else at query time. States without a selected lifecycle, and maintained populations that lifecycle enumeration does not cover, are refused. Co-Authored-By: Claude Opus 5.5 --- crates/asap-aware-mapping/src/lib.rs | 2 +- .../src/summary_maintenance_lifecycle.rs | 377 +++++++++++++++++- 2 files changed, 374 insertions(+), 5 deletions(-) diff --git a/crates/asap-aware-mapping/src/lib.rs b/crates/asap-aware-mapping/src/lib.rs index 716c6ee4..921f504e 100644 --- a/crates/asap-aware-mapping/src/lib.rs +++ b/crates/asap-aware-mapping/src/lib.rs @@ -228,7 +228,7 @@ pub use summary_maintenance_lifecycle::{ SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecycleChoiceError, SummaryMaintenanceLifecycleCostInputs, SummaryMaintenanceLifecyclePlan, SummaryMaintenanceLifecyclePlanError, SummaryMaintenanceLifecycleRejection, - SummaryMaintenanceLifecycleSelectionError, WorkloadDemand, + SummaryMaintenanceLifecycleSelectionError, SummaryMaintenanceTimingError, WorkloadDemand, }; pub use topk_reuse::TopKLimitReuseStrategy; diff --git a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs index 7e40645f..43bacfcb 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs @@ -20,10 +20,11 @@ use std::collections::{HashMap, HashSet}; use std::rc::Rc; use asap_types::post_asap::{ - compile_post_asap_dag_with_node_ids, EvaluationSchedule, ExecutionDataStateError, - OutputRepresentation, PostAsapNodeId, ResultGuarantee, SummaryExpr, - SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, - SummaryNode, SummaryWindowFramework, + compile_post_asap_dag, compile_post_asap_dag_with_node_ids, EvaluationSchedule, + ExecutionDataStateError, ExecutionTiming, OutputRepresentation, PostAsapDag, + PostAsapDagValidationError, PostAsapNodeId, PostAsapOperatorPayload, ResultGuarantee, + SummaryExpr, SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, + SummaryMaintenanceMode, SummaryNode, SummaryWindowFramework, ValueOperation, }; use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; @@ -211,6 +212,84 @@ pub struct SummaryMaintenanceLifecyclePlan { pub raw_recompute_total_cost: Option, } +/// Why a lifecycle plan cannot assign execution timing to its DAG. +#[derive(Debug, thiserror::Error, PartialEq)] +pub enum SummaryMaintenanceTimingError { + #[error(transparent)] + InvalidPostAsapDag(#[from] ExecutionDataStateError), + #[error("summary {0:?} has no selected lifecycle")] + UnselectedLifecycle(PostAsapNodeId), + /// Lifecycle enumeration covers `SummaryAgg` states only; timing for other + /// retained state would otherwise be guessed. + #[error("node {0:?} maintains state that has no summary-maintenance lifecycle")] + UnplannedMaintainedState(PostAsapNodeId), + #[error(transparent)] + InvalidPhases(#[from] PostAsapDagValidationError), +} + +impl SummaryMaintenanceLifecyclePlan { + /// The post-ASAP DAG of [`Self::root`] with every node's timing derived + /// from the selected lifecycles, so physical compilation places it. + /// + /// A retained (non-`Ephemeral`) state outlives one query, so it and every + /// input it consumes run at ingestion time. Every other node runs at query + /// time: readouts and consumers of retained state, and each `Ephemeral` + /// state not consumed by retained state together with its inputs, whose + /// raw data the deployment must supply as a query source. Timings already + /// on the root are ignored. + pub fn execution_timed_dag(&self) -> Result { + let dag = compile_post_asap_dag(&self.root)?; + if let Some(node) = dag.nodes.iter().find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Value { + operation: ValueOperation::MaintainPopulation { .. } + } + ) + }) { + return Err(SummaryMaintenanceTimingError::UnplannedMaintainedState( + node.id, + )); + } + let mut pending = Vec::new(); + for deployment in &self.deployments { + let guarantee = deployment + .summary_maintenance_lifecycle_guarantee + .as_ref() + .ok_or(SummaryMaintenanceTimingError::UnselectedLifecycle( + deployment.post_asap_node_id, + ))?; + if guarantee.summary_maintenance_lifecycle != SummaryMaintenanceLifecycle::Ephemeral { + pending.push(deployment.post_asap_node_id); + } + } + let mut ingestion = HashSet::new(); + while let Some(id) = pending.pop() { + if ingestion.insert(id) { + pending.extend( + dag.edges + .iter() + .filter(|edge| edge.consumer == id) + .map(|edge| edge.producer), + ); + } + } + let phases = dag + .nodes + .iter() + .map(|node| { + let timing = if ingestion.contains(&node.id) { + ExecutionTiming::IngestionTime + } else { + ExecutionTiming::QueryTime + }; + (node.id, timing) + }) + .collect(); + Ok(dag.with_execution_phases(&phases)?) + } +} + /// Explicit association between a materialized target and the normalized /// workload entries whose demand consumes it. /// @@ -2845,4 +2924,294 @@ mod tests { .iter() .any(|deployment| Rc::ptr_eq(&deployment.summary, &shared))); } + + fn readout(state: &Rc) -> Rc { + Rc::new(SummaryNode { + expr: SummaryExpr::ValueOperation { + child: Rc::clone(state), + operation: ValueOperation::FinalizeExactAccumulator, + timing: ExecutionTiming::QueryTime, + }, + schema: SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }, + guarantee: Some(ResultGuarantee::exact("sum")), + }) + } + + fn lifecycle_matching( + alternatives: &[SummaryMaintenanceLifecycleAlternative], + kind: fn(&SummaryMaintenanceLifecycle) -> bool, + ) -> SummaryMaintenanceLifecycle { + alternatives + .iter() + .map(|alternative| &alternative.summary_maintenance_lifecycle) + .find(|lifecycle| kind(lifecycle)) + .expect("lifecycle kind is an alternative") + .clone() + } + + /// Bind the lifecycle `choose` picks for every state of `root`, then + /// derive the timed DAG. + fn timed_dag( + root: Rc, + workload: &QueryWorkload, + data: &DataWorkload, + horizon: Option, + choose: impl Fn(&SummaryMaintenanceDeployment) -> SummaryMaintenanceLifecycle, + ) -> PostAsapDag { + let candidates = enumerate_summary_maintenance_lifecycles( + root, + WorkloadDemand::new_with_data(workload, data, &[0]), + 1_000, + horizon, + SummaryMaintenanceLifecycleCapabilities::ALL, + &UnitCosts, + ) + .unwrap(); + let choice: Vec<_> = candidates + .deployments() + .iter() + .map(|deployment| (deployment.post_asap_node_id, choose(deployment))) + .collect(); + let dag = candidates + .select(&choice) + .unwrap() + .execution_timed_dag() + .unwrap(); + dag.validate().unwrap(); + dag + } + + /// Operator kinds in node-id order, each paired with its timing. + fn timings(dag: &PostAsapDag) -> Vec<(&'static str, ExecutionTiming)> { + dag.nodes + .iter() + .map(|node| { + let kind = match node.payload { + PostAsapOperatorPayload::Fallback { .. } => "raw", + PostAsapOperatorPayload::SummaryAgg { .. } => "state", + PostAsapOperatorPayload::Value { .. } => "readout", + PostAsapOperatorPayload::Binary { .. } => "binary", + _ => "other", + }; + (kind, node.output_state.timing) + }) + .collect() + } + + const INGEST: ExecutionTiming = ExecutionTiming::IngestionTime; + const QUERY: ExecutionTiming = ExecutionTiming::QueryTime; + + // Every retained lifecycle kind runs its state and inputs at ingestion + // time and its readout at query time. + #[test] + fn retained_lifecycles_time_state_and_inputs_at_ingestion() { + let mut scheduled = batch(Predictability::Predictable { + known_at: Some(TimestampMs(1_000)), + }); + scheduled.execute_at = Some(TimestampMs(11_000)); + type Case = ( + QueryWorkload, + DataWorkload, + Option, + fn(&SummaryMaintenanceLifecycle) -> bool, + ); + let cases: [Case; 3] = [ + ( + workload(vec![], vec![repeating()], continuous(1_000, 60_000)), + continuous(1_000, 60_000), + Some(Horizon(10.0)), + |lifecycle| { + matches!( + lifecycle, + SummaryMaintenanceLifecycle::ContinuouslyMaintained + ) + }, + ), + ( + workload(vec![], vec![repeating()], at_rest()), + at_rest(), + Some(Horizon(10.0)), + |lifecycle| matches!(lifecycle, SummaryMaintenanceLifecycle::Shared { .. }), + ), + ( + workload(vec![scheduled], vec![], at_rest()), + at_rest(), + None, + |lifecycle| matches!(lifecycle, SummaryMaintenanceLifecycle::Prepared { .. }), + ), + ]; + for (workload, data, horizon, kind) in cases { + let dag = timed_dag( + readout(&summary()), + &workload, + &data, + horizon, + |deployment| lifecycle_matching(&deployment.alternatives, kind), + ); + assert_eq!( + timings(&dag), + [("raw", INGEST), ("state", INGEST), ("readout", QUERY)] + ); + } + } + + // An Ephemeral state, its raw input, and its readout all run at query time. + #[test] + fn ephemeral_lifecycle_times_state_and_downstream_at_query() { + let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); + let dag = timed_dag(readout(&summary()), &workload, &at_rest(), None, |_| { + SummaryMaintenanceLifecycle::Ephemeral + }); + assert_eq!( + timings(&dag), + [("raw", QUERY), ("state", QUERY), ("readout", QUERY)] + ); + } + + // One state read by two consumers is one deployment; its timing follows + // that single choice while both consumers run at query time. + #[test] + fn shared_state_is_timed_once_for_all_consumers() { + let state = summary(); + let lhs = readout(&state); + let rhs = Rc::new(lhs.as_ref().clone()); + let root = Rc::new(SummaryNode { + expr: SummaryExpr::BinaryOp { + lhs, + rhs, + operator: asap_types::post_asap::BinaryOperator { + kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( + asap_types::pre_asap::ArithmeticOpKind::Add, + ), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + timing: QUERY, + }, + schema: readout(&state).schema.clone(), + guarantee: None, + }); + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let dag = timed_dag(root, &workload, &data, Some(Horizon(10.0)), |deployment| { + assert!(Rc::ptr_eq(&deployment.summary, &state)); + SummaryMaintenanceLifecycle::ContinuouslyMaintained + }); + assert_eq!( + timings(&dag), + [ + ("raw", INGEST), + ("state", INGEST), + ("readout", QUERY), + ("readout", QUERY), + ("binary", QUERY), + ] + ); + } + + // An Ephemeral state consumed by retained state is built on the retained + // state's ingestion path; it is not retained, but cannot run at query time. + #[test] + fn ephemeral_state_feeding_retained_state_runs_at_ingestion() { + let mut scheduled = batch(Predictability::Predictable { + known_at: Some(TimestampMs(1_000)), + }); + scheduled.execute_at = Some(TimestampMs(11_000)); + let root = nested_summary(); + let workload = workload(vec![scheduled], vec![], at_rest()); + let dag = timed_dag( + Rc::clone(&root), + &workload, + &at_rest(), + None, + |deployment| { + if Rc::ptr_eq(&deployment.summary, &root) { + lifecycle_matching(&deployment.alternatives, |lifecycle| { + matches!(lifecycle, SummaryMaintenanceLifecycle::Prepared { .. }) + }) + } else { + SummaryMaintenanceLifecycle::Ephemeral + } + }, + ); + assert_eq!( + timings(&dag), + [("raw", INGEST), ("state", INGEST), ("state", INGEST)] + ); + } + + // Timing is not derived for a state without a selected lifecycle, and a + // raw-recompute plan runs entirely at query time. + #[test] + fn timing_requires_a_selected_lifecycle_for_every_state() { + let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); + let data = at_rest(); + let demand = WorkloadDemand::new_with_data(&workload, &data, &[0]); + let plan = plan_summary_maintenance_lifecycles( + readout(&summary()), + demand, + 1_000, + None, + SummaryMaintenanceLifecycleCapabilities::ALL, + &crate::cost_model::DefaultCostModel, + ) + .unwrap(); + assert_eq!( + plan.execution_timed_dag().unwrap_err(), + SummaryMaintenanceTimingError::UnselectedLifecycle( + plan.deployments[0].post_asap_node_id + ) + ); + let raw = plan_summary_maintenance_lifecycles( + crate::replacement::keep_pre_asap(&sum_query()).unwrap(), + demand, + 1_000, + None, + SummaryMaintenanceLifecycleCapabilities::ALL, + &UnitCosts, + ) + .unwrap(); + assert_eq!( + timings(&raw.execution_timed_dag().unwrap()), + [("raw", QUERY)] + ); + } + + // A maintained population is retained state the lifecycle plan does not + // enumerate, so its timing is refused rather than guessed. + #[test] + fn timing_refuses_state_outside_the_lifecycle_plan() { + let target = Rc::new(crate::test_support::lower_promql( + "sum(a)", + AccuracyTarget::Exact, + )); + let root = crate::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&target), + ) + .candidate(&target) + .unwrap(); + let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); + let plan = plan_summary_maintenance_lifecycles( + root, + WorkloadDemand::new_with_data(&workload, &at_rest(), &[0]), + 1_000, + None, + SummaryMaintenanceLifecycleCapabilities::ALL, + &UnitCosts, + ) + .unwrap(); + assert!(plan.deployments.is_empty()); + assert!(matches!( + plan.execution_timed_dag(), + Err(SummaryMaintenanceTimingError::UnplannedMaintainedState(_)) + )); + } } From 9b6dd96a2eed44daa9098cfdb462d64d98a5a421 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:10:14 +0000 Subject: [PATCH 14/59] test: carry a chosen lifecycle through timing to physical compilation Rebase note: #483 removed the pane test this test was appended after; the new test is appended to the #483 version of the file. Co-Authored-By: Claude Opus 5.5 --- .../summary_maintenance_lifecycle_e2e.rs | 190 ++++++++++++++++++ 1 file changed, 190 insertions(+) diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index d5406f23..77cdd3d7 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -426,3 +426,193 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { ); } } + +fn quantile_workload(query: &str) -> PlanningWorkload { + let mut workload = dashboard_workload(); + workload.query_workload.query_batch.as_mut().unwrap()[0].query = Query(query.into()); + workload.query_workload.repeating_queries.as_mut().unwrap()[0].query = Query(query.into()); + workload +} + +/// Precompute outputs implied by timing: ingestion-time nodes read by a +/// query-time node, or the root when it is itself ingestion-timed. +fn ingestion_frontier(dag: &asap_types::post_asap::PostAsapDag) -> Vec { + use asap_types::post_asap::ExecutionTiming::IngestionTime; + let timing = |id| { + dag.nodes + .iter() + .find(|node| node.id == id) + .unwrap() + .output_state + .timing + }; + let mut frontier: Vec<_> = dag + .nodes + .iter() + .filter(|node| { + node.output_state.timing == IngestionTime + && (node.id == dag.root + || dag.edges.iter().any(|edge| { + edge.producer == node.id && timing(edge.consumer) != IngestionTime + })) + }) + .map(|node| u64::from(node.id.0)) + .collect(); + frontier.sort_unstable(); + frontier +} + +/// For existing PromQL fixtures, Planner's own retained lifecycle selection +/// reproduces the timing that realization strategies assign today. +#[test] +fn planner_lifecycle_selection_reproduces_strategy_timing() { + for query in [ + "quantile_over_time(0.99, latency[5m])", + "quantile(0.99, latency)", + "sum by(job)(rate(m[1m]))", + ] { + let plan = selected_plan(&quantile_workload(query)); + assert!(!plan.selected_raw_recompute, "{query}"); + assert!(plan.deployments.iter().all(|deployment| { + deployment + .summary_maintenance_lifecycle_guarantee + .as_ref() + .is_some_and(|guarantee| { + guarantee.summary_maintenance_lifecycle + != SummaryMaintenanceLifecycle::Ephemeral + }) + })); + let strategy = asap_types::post_asap::compile_post_asap_dag(&plan.root).unwrap(); + assert_eq!(plan.execution_timed_dag().unwrap(), strategy, "{query}"); + } +} + +/// An explicitly chosen lifecycle reaches physical compilation through timing: +/// ContinuouslyMaintained puts the state in precompute, Ephemeral leaves +/// precompute empty and reads the raw source at query time; both answer alike. +#[test] +fn chosen_lifecycle_timing_decides_precompute_contents() { + use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; + use asap_physical_operators::{ + physical_planner::{compile_candidate, InputContract}, + runtime::Scope, + values::{Batch, Value}, + }; + use asap_types::{ + post_asap::{PostAsapOperatorPayload, SummaryFamilyType}, + pre_asap::DataType, + }; + use std::{collections::BTreeMap, sync::Arc}; + + let workload = quantile_workload("quantile(0.99, latency)"); + let root = selected_plan(&workload).root; + let mut answers = Vec::new(); + for lifecycle in [ + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + SummaryMaintenanceLifecycle::Ephemeral, + ] { + let candidates = enumerate_summary_maintenance_lifecycles( + Rc::clone(&root), + WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + NOW_MS, + Some(Horizon(100.)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &FullyCostedRuntime, + ) + .unwrap(); + let [deployment] = candidates.deployments() else { + panic!("one summary state"); + }; + let id = deployment.post_asap_node_id; + let state = u64::from(id.0); + let dag = candidates + .select(&[(id, lifecycle.clone())]) + .unwrap() + .execution_timed_dag() + .unwrap(); + let raw = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .unwrap(); + let (raw_id, schema) = (u64::from(raw.id.0), Arc::new(raw.output_schema.clone())); + let frontier = ingestion_frontier(&dag); + let candidate = compile_candidate( + &dag, + BTreeMap::from([(raw_id, InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &frontier, + ) + .unwrap(); + let rows = (1..=100) + .map(|value| { + schema + .fields + .iter() + .map(|field| match field.dtype { + SummaryFamilyType::Plain(DataType::Float64) => { + Value::Float64(f64::from(value)) + } + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), + _ => panic!("unexpected field {field:?}"), + }) + .collect() + }) + .collect(); + let raw_batch = Batch::try_new(schema.clone(), rows).unwrap(); + let query_scope = Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + }; + let result = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { + assert!(frontier.is_empty()); + assert!(candidate.precompute.is_none()); + physical_common::execute( + &candidate.query, + BTreeMap::from([(raw_id, raw_batch)]), + query_scope, + ) + } else { + assert_eq!(frontier, [state]); + assert_eq!( + candidate + .materialized_outputs + .keys() + .copied() + .collect::>(), + [state] + ); + let stored = physical_common::execute( + candidate.precompute.as_ref().unwrap(), + BTreeMap::from([(raw_id, raw_batch)]), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 300_000, + revision: 1, + }, + ); + physical_common::execute( + &candidate.query, + BTreeMap::from([(state, stored[0][0].clone())]), + query_scope, + ) + }; + answers.push( + result[0] + .iter() + .flat_map(|batch| batch.rows()) + .flat_map(|row| row.iter()) + .filter_map(|value| match value { + Value::Float64(value) => Some(*value), + _ => None, + }) + .collect::>(), + ); + } + assert_eq!(answers[0], answers[1]); + assert_eq!(answers[0].len(), 1); +} From 3b281da8e4ef688841fc7ca164ae811aab71cfa3 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:10:58 +0000 Subject: [PATCH 15/59] docs: state the lifecycle-owned timing layer contract Replace the open question about placement ownership with the agreed four-layer contract: logical Post-ASAP decides what to compute, the chosen lifecycle decides timing, physical compilation partitions by timing, and the backend prices lifecycle choices including store cost. Co-Authored-By: Claude Opus 5.5 --- .../physical-planning-and-deployment.md | 36 ++++++++++++++++--- 1 file changed, 31 insertions(+), 5 deletions(-) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 2aad9ca7..69e9b90a 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -32,10 +32,31 @@ The Logical Post-ASAP DAG is preceded by the Pre-ASAP DAG (`QueryExpr`), the language-independent query semantics before summary selection. Both are logical. Planning builds Post-ASAP `SummaryNode` trees; `compile_post_asap_dag` exports the selected tree as a `PostAsapDag`, which is the Physical Plan -Compiler's input. Its per-node execution phase (ingestion or query time) is an -initial placement: compilation places ingestion-time nodes in the precompute DAG, -while frontier enumeration proposes alternative materialization splits. Which -layer owns placement is an open design question, deferred to a later change. +Compiler's input. Its per-node execution phase (ingestion or query time) is +decided by the selected summary maintenance lifecycle, as the layer contract +below states. + +### Layer contract + +1. **Logical Post-ASAP** (`PlanSpace`) decides what to compute: summary + families, readouts and sharing. It does not decide placement; timing that a + realization strategy writes while building a candidate is provisional. +2. **Summary maintenance lifecycle** (Planner) lists the lifecycle choices for + each unique summary state. A chosen assignment determines every node's + `ExecutionTiming`, plus window framework and retention. + `SummaryMaintenanceLifecyclePlan::execution_timed_dag` applies it: a retained + (non-`Ephemeral`) state and all of its inputs run at ingestion time; + readouts, other consumers, and `Ephemeral` states not consumed by retained + state run at query time. +3. **Physical compile** (Planner) reads timing: ingestion-time nodes form the + precompute DAG and the rest form the query DAG, joined by typed outputs. It + does not see raw ingestion, panes, storage or stored-state readout. +4. **Backend** chooses the lifecycle assignment with its own `CostModel`: + precompute CPU (`maintenance_cost_per_update`), sketch/summary store cost + (`retention_cost_rate`), query reads (`summary_read_cost`) and per-query + builds (`build_cost`, for `Ephemeral`), counting shared state once. + `Ephemeral` requires the deployment to supply the state's raw input as a + query-time source. ### Candidate generation and deployment selection @@ -363,6 +384,9 @@ DAG. If the required behavior cannot be realized, physical compilation fails. Materialization frontiers are Planner decisions. A candidate records both the precompute Physical DAG and the query Physical DAG, with typed outputs connecting them. The deployment compiler binds those outputs; it does not move operators. +Lifecycle timing gives the frontier: ingestion-time nodes read by query-time +nodes. Moving further bounded consumers into precompute, as in Candidate B +below, is not yet expressed as a lifecycle choice. For `sum by(job)(rate(m[1m]))`, legal physical candidates can include: @@ -469,7 +493,7 @@ The complete example makes the ownership boundary explicit: | --- | --- | | **Logical Post-ASAP DAG** | Use `KLL(k=200)` with shared merge for p50/p99 | | **Summary Maintenance Candidate Generation** | Maintain 1-minute panes and reuse them for aligned five-minute queries | -| **Summary Maintenance Lifecycle** | Record pane/window/freshness/reuse requirements | +| **Summary Maintenance Lifecycle** | Record pane/window/freshness/reuse requirements and each node's execution timing | | **Physical Plan Compiler** | Lower to native KLL build, merge, and readout operators | | **Physical DAG** | Define precompute and query DAGs with typed input/output boundaries | | **Deployment Plan Compiler** | Bind raw input and KLL state slots to concrete sources/materializations | @@ -515,6 +539,8 @@ operator/runtime fixtures: | Test | Contract exercised | | --- | --- | | `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled precompute/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | +| `summary_maintenance_lifecycle_e2e::chosen_lifecycle_timing_decides_precompute_contents` | PromQL workload → enumerated lifecycles → explicit choice → timed DAG → compiled candidate; ContinuouslyMaintained stores the state in precompute, Ephemeral leaves precompute empty and reads the raw source at query time; both return the same p99 | +| `summary_maintenance_lifecycle_e2e::planner_lifecycle_selection_reproduces_strategy_timing` | For PromQL fixtures, the timed DAG from Planner's retained selection equals the DAG realization strategies produce today | | `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute precompute DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | | `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | | `precompute_candidates::grouped_rate_can_be_materialized_before_or_after_grouped_sum` | Cost changes select different legal precompute frontiers; both selected candidates execute with the same reset-sensitive result; uncompilable candidates are not priced | From 729333265b7cf474738fddc2f647f9b6e2653910 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 22:58:56 +0000 Subject: [PATCH 16/59] feat: derive physical candidates from one compilation `compile_candidate` re-lowered the whole Post-ASAP DAG for every materialization frontier, and `enumerate_frontiers` compiled it once more. Number helper operators from their Planner node (`u64::MAX - node_id`, at most one helper per node) so every boundary choice is a subgraph of one lowering. `cut_candidate` partitions a `compile` result for one frontier and `enumerate_compiled_frontiers` enumerates over it; both produce candidates byte-identical to per-frontier recompilation. Co-Authored-By: Claude Opus 5.5 --- .../src/physical_planner/candidates.rs | 152 ++++++++++++--- .../src/physical_planner/compiled.rs | 56 +++++- .../src/physical_planner/mod.rs | 20 +- .../tests/precompute_candidates.rs | 183 +++++++++++++++++- 4 files changed, 366 insertions(+), 45 deletions(-) diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs index a8f476f3..ebab0d31 100644 --- a/crates/asap-physical-operators/src/physical_planner/candidates.rs +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -25,24 +25,42 @@ pub fn compile_candidate( inputs: BTreeMap, roots: &[NodeId], frontier: &[NodeId], +) -> Result { + cut_candidate(&compile(dag, inputs, roots)?, frontier) +} + +/// Derive one frontier's candidate from a complete [`compile`] result by +/// partitioning its operators; nothing is lowered again. A deployment compiles +/// each query DAG once and derives every placement choice from that result. +/// The candidate is identical to [`compile_candidate`] for the same frontier. +pub fn cut_candidate( + compiled: &CompiledPhysicalDag, + frontier: &[NodeId], ) -> Result { if frontier.is_empty() { return Ok(PhysicalCandidate { precompute: None, - query: compile(dag, inputs, roots)?, + query: compiled.clone(), materialized_outputs: BTreeMap::new(), }); } let frontier_set: BTreeSet<_> = frontier.iter().copied().collect(); - if frontier_set.len() != frontier.len() || frontier.iter().any(|id| inputs.contains_key(id)) { + // `compile` retains only reachable nodes and numbers its helper operators + // above the u32 Planner ID range; only Planner outputs are boundaries. + if frontier_set.len() != frontier.len() + || frontier + .iter() + .any(|&id| !compiled.is_operator(id) || u32::try_from(id).is_err()) + { return Err(invalid("frontier must contain distinct computed outputs")); } - let full = compile(dag, inputs.clone(), roots)?; - let precompute = compile(dag, inputs.clone(), frontier)?; + let inputs: BTreeMap<_, _> = compiled + .input_contracts() + .map(|(id, contract)| (id, contract.clone())) + .collect(); + let precompute = compiled.cut(&inputs, frontier)?; let mut materialized_outputs = BTreeMap::new(); for &id in frontier { - // Also proves that the frontier is reachable from the requested roots. - full.output_contract(id)?; let mut output = precompute.output_contract(id)?; if output.properties.boundedness != Boundedness::Bounded { return Err(invalid("materialized output requires bounded execution")); @@ -54,7 +72,7 @@ pub fn compile_candidate( } let mut query_inputs = inputs; query_inputs.extend(materialized_outputs.clone()); - let query = compile(dag, query_inputs, roots)?; + let query = compiled.cut(&query_inputs, compiled.roots())?; let used: BTreeSet<_> = query.input_contracts().map(|(id, _)| id).collect(); if !frontier.iter().all(|id| used.contains(id)) { return Err(invalid( @@ -78,43 +96,40 @@ pub fn enumerate_frontiers( inputs: &BTreeMap, roots: &[NodeId], max_candidates: usize, +) -> Result>, Error> { + enumerate_compiled_frontiers(&compile(dag, inputs.clone(), roots)?, max_candidates) +} + +/// [`enumerate_frontiers`] over an existing [`compile`] result, so enumeration +/// and [`cut_candidate`] share one lowering. +pub fn enumerate_compiled_frontiers( + compiled: &CompiledPhysicalDag, + max_candidates: usize, ) -> Result>, Error> { if max_candidates == 0 { return Err(invalid( "frontier search requires a positive candidate budget", )); } - let compiled = compile(dag, inputs.clone(), roots)?; let mut ancestors = BTreeMap::>::new(); let mut eligible = Vec::new(); - for node in &dag.nodes { - let id = u64::from(node.id.0); - if inputs.contains_key(&id) { - continue; - } - let Ok(contract) = compiled.output_contract(id) else { - continue; - }; - if contract.properties.boundedness != Boundedness::Bounded { + for (id, properties) in compiled.output_properties()? { + if !compiled.is_operator(id) + || u32::try_from(id).is_err() + || properties.boundedness != Boundedness::Bounded + { continue; } let mut seen = BTreeSet::new(); let mut pending = vec![id]; while let Some(current) = pending.pop() { - if !seen.insert(current) || inputs.contains_key(¤t) { - continue; + if seen.insert(current) { + pending.extend(compiled.dependencies(current)); } - pending.extend( - dag.edges - .iter() - .filter(|edge| u64::from(edge.consumer.0) == current) - .map(|edge| u64::from(edge.producer.0)), - ); } ancestors.insert(id, seen); eligible.push(id); } - eligible.sort_unstable(); let mut frontiers = vec![vec![]]; for id in eligible { let additions = frontiers @@ -142,16 +157,20 @@ pub fn enumerate_frontiers( /// Lower every maintenance candidate before feasibility/cost evaluation. Keep /// individual failures visible; do not substitute another computation on error. +/// The DAG is lowered once; each frontier is a [`cut_candidate`] of it. pub fn compile_candidates( dag: &PostAsapDag, inputs: BTreeMap, roots: &[NodeId], frontiers: &[Vec], ) -> Vec> { - frontiers - .iter() - .map(|frontier| compile_candidate(dag, inputs.clone(), roots, frontier)) - .collect() + match compile(dag, inputs, roots) { + Ok(compiled) => frontiers + .iter() + .map(|frontier| cut_candidate(&compiled, frontier)) + .collect(), + Err(error) => frontiers.iter().map(|_| Err(error.clone())).collect(), + } } /// Complete workload cost supplied by scoped optimizer/deployment evidence. @@ -273,3 +292,76 @@ impl PhysicalCandidate { Ok(()) } } + +#[cfg(test)] +mod tests { + use super::*; + use planner_types::workload::*; + + fn grouped_rate() -> (PostAsapDag, BTreeMap, NodeId) { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("sum by(job)(rate(m[1m]))".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit( + planner_types::types::AccuracyTarget::Exact, + ), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let root = asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0); + let root = std::rc::Rc::new(promql_rows::with_series_identity(&root).unwrap()); + let space = asap_aware_mapping::search_workload(vec![("q", root)]); + let selected = space + .global_selection(&asap_aware_mapping::cost_model::DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + let dag = planner_types::post_asap::compile_post_asap_dag(&selected).unwrap(); + let state = dag + .nodes + .iter() + .find(|node| matches!(node.payload, Payload::SummaryAgg { .. })) + .unwrap(); + let inputs = BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + (dag.clone(), inputs, u64::from(dag.root.0)) + } + + /// Enumerating and cutting every frontier lowers each Planner node once. + #[test] + fn candidates_for_all_frontiers_share_one_lowering() { + let (dag, inputs, root) = grouped_rate(); + let lowered = || crate::physical_planner::LOWERED_NODES.with(|count| count.get()); + let before = lowered(); + let compiled = compile(&dag, inputs, &[root]).unwrap(); + let once = lowered() - before; + let frontiers = enumerate_compiled_frontiers(&compiled, 4096).unwrap(); + assert!(frontiers.len() >= 3, "{frontiers:?}"); + for frontier in &frontiers { + cut_candidate(&compiled, frontier).unwrap(); + } + assert!(once > 0); + assert_eq!(lowered() - before, once); + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs index 70d6a9ff..af0bbe39 100644 --- a/crates/asap-physical-operators/src/physical_planner/compiled.rs +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -212,13 +212,8 @@ impl CompiledPhysicalDag { } /// Derive a reachable output contract without opening deployment readers. pub fn output_contract(&self, id: NodeId) -> Result { - let sources = self - .input_contracts() - .map(|(id, contract)| (id, Box::new(contract.clone()) as Source<'_>)) - .collect(); - let graph = self.instantiate(sources)?; - let properties = graph.properties(&self.roots)?; - let properties = *properties + let properties = *self + .output_properties()? .get(&id) .ok_or_else(|| invalid("output is not reachable"))?; let schema = match self @@ -231,6 +226,53 @@ impl CompiledPhysicalDag { }; Ok(InputContract { schema, properties }) } + /// Properties of every reachable node, derived in one contract-only pass. + pub(super) fn output_properties(&self) -> Result, Error> { + let sources = self + .input_contracts() + .map(|(id, contract)| (id, Box::new(contract.clone()) as Source<'_>)) + .collect(); + self.instantiate(sources)?.properties(&self.roots) + } + /// Direct physical dependencies; empty for inputs and unknown IDs. + pub(super) fn dependencies(&self, id: NodeId) -> &[NodeId] { + match self.nodes.get(&id) { + Some(Node::Operator { inputs, .. }) => inputs, + _ => &[], + } + } + pub(super) fn is_operator(&self, id: NodeId) -> bool { + matches!(self.nodes.get(&id), Some(Node::Operator { .. })) + } + /// Keep the already-lowered operators reachable from `roots`, replacing + /// each node in `boundaries` by a typed input. Nothing is lowered again. + pub(super) fn cut( + &self, + boundaries: &BTreeMap, + roots: &[NodeId], + ) -> Result { + let mut result = Self::new(roots.to_vec()); + let mut pending = roots.to_vec(); + while let Some(id) = pending.pop() { + if result.nodes.contains_key(&id) { + continue; + } + let node = match boundaries.get(&id) { + Some(contract) => Node::Input(contract.clone()), + None => self + .nodes + .get(&id) + .cloned() + .ok_or_else(|| invalid(format!("missing physical node {id}")))?, + }; + if let Node::Operator { inputs, .. } = &node { + pending.extend(inputs); + } + result.nodes.insert(id, node); + } + result.validate()?; + Ok(result) + } /// Validate using contract-only sources. No deployment reader is available. pub fn validate(&self) -> Result<(), Error> { let sources = self diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 112e1715..e852add8 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -36,8 +36,8 @@ pub mod promql_values; mod candidates; pub use candidates::{ - compile_candidate, compile_candidates, enumerate_frontiers, select_candidate, CandidateCost, - CandidateSelection, PhysicalCandidate, + compile_candidate, compile_candidates, cut_candidate, enumerate_compiled_frontiers, + enumerate_frontiers, select_candidate, CandidateCost, CandidateSelection, PhysicalCandidate, }; mod compiled; @@ -103,6 +103,12 @@ pub fn bind_with_data_sources<'a>( bind(dag, sources, roots) } +#[cfg(test)] +thread_local! { + /// Planner nodes lowered by this thread, for compile-once tests. + static LOWERED_NODES: std::cell::Cell = const { std::cell::Cell::new(0) }; +} + fn compile_internal( dag: &PostAsapDag, mut sources: BTreeMap, @@ -159,9 +165,12 @@ fn compile_internal( } } let mut graph = CompiledPhysicalDag::new(roots.to_vec()); - let mut auxiliary = u64::MAX; for id in ordered { let node = nodes[&id]; + // At most one helper operator per node, numbered above the u32 Planner + // ID range by its node alone, so every boundary choice yields a subgraph + // of the same lowering and candidate cuts need not renumber operators. + let auxiliary = u64::MAX - id; let output = Arc::new(node.output_schema.clone()); crate::values::validate_schema(&output)?; if let Some(source) = sources.remove(&id) { @@ -170,6 +179,8 @@ fn compile_internal( } graph.add_input(id, source)?; } else { + #[cfg(test)] + LOWERED_NODES.with(|count| count.set(count.get() + 1)); let mut inputs = dependencies.get(&id).cloned().unwrap_or_default(); let mut schemas = inputs .iter() @@ -185,7 +196,6 @@ fn compile_internal( Operator::union(schemas[0].clone(), schemas.len())?, )?; inputs = vec![auxiliary]; - auxiliary -= 1; schemas.truncate(1); } if let Payload::Value { @@ -280,7 +290,6 @@ fn compile_internal( vec![auxiliary], Operator::limit(input, *k as u64, 0, groups)?.with_output_schema(output)?, )?; - auxiliary -= 1; continue; } // A closed row must include either all source labels or the explicit @@ -340,7 +349,6 @@ fn compile_internal( vec![auxiliary], Operator::scope_timestamp(compact, output)?, )?; - auxiliary -= 1; continue; } let mut operator = compile_node(node, &schemas) diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index c00b388f..f402e3dc 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -4,8 +4,9 @@ use asap_physical_operators::{ factory::create_planner_accumulator, operators::Operator, physical_planner::{ - compile_candidates, select_candidate, CandidateCost, CompiledPhysicalDag, InputContract, - Source, + compile, compile_candidate, compile_candidates, cut_candidate, + enumerate_compiled_frontiers, enumerate_frontiers, select_candidate, CandidateCost, + CompiledPhysicalDag, InputContract, PhysicalCandidate, Source, }, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, @@ -546,3 +547,181 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { "must execute both stored and query-time grouped Rate candidates: {executed}" ); } + +/// The per-frontier lowering used before compile-once cuts: each boundary +/// choice lowers the precompute and query DAGs from the logical DAG again. +fn recompiled_candidate( + dag: &PostAsapDag, + inputs: &BTreeMap, + roots: &[u64], + frontier: &[u64], +) -> Result { + use asap_physical_operators::plan::Emission; + if frontier.is_empty() { + return Ok(PhysicalCandidate { + precompute: None, + query: compile(dag, inputs.clone(), roots)?, + materialized_outputs: BTreeMap::new(), + }); + } + let precompute = compile(dag, inputs.clone(), frontier)?; + let mut materialized_outputs = BTreeMap::new(); + for &id in frontier { + let mut output = precompute.output_contract(id)?; + output.properties.emission = Emission::Unknown; + materialized_outputs.insert(id, output); + } + let mut query_inputs = inputs.clone(); + query_inputs.extend(materialized_outputs.clone()); + Ok(PhysicalCandidate { + precompute: Some(precompute), + query: compile(dag, query_inputs, roots)?, + materialized_outputs, + }) +} + +fn assert_cuts_match_recompilation( + dag: &PostAsapDag, + inputs: BTreeMap, + roots: &[u64], + min_frontiers: usize, +) { + let compiled = compile(dag, inputs.clone(), roots).unwrap(); + let frontiers = enumerate_compiled_frontiers(&compiled, 4096).unwrap(); + assert_eq!( + frontiers, + enumerate_frontiers(dag, &inputs, roots, 4096).unwrap() + ); + assert!(frontiers.len() >= min_frontiers, "{frontiers:?}"); + for frontier in &frontiers { + let cut = cut_candidate(&compiled, frontier).unwrap(); + let expected = recompiled_candidate(dag, &inputs, roots, frontier).unwrap(); + assert_eq!( + cut.encode().unwrap(), + expected.encode().unwrap(), + "{frontier:?}" + ); + } +} + +/// Every enumerated grouped Rate→Sum frontier (query-only, stored Rate, +/// stored Sum) cuts to exactly the candidate that per-frontier lowering builds. +#[test] +fn grouped_rate_cuts_equal_per_frontier_compilation() { + let dag = grouped_rate(); + let state = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .unwrap(); + let inputs = BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + assert_cuts_match_recompilation(&dag, inputs, &[u64::from(dag.root.0)], 3); +} + +/// Cuts of a DAG whose nodes lower to helper operators (current-series +/// population read by Sort→Limit) keep the same operator IDs as recompilation. +#[test] +fn population_topk_cuts_equal_per_frontier_compilation() { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("topk by(job)(1, m)".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(60_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let original = asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0); + let root = Rc::new( + asap_physical_operators::physical_planner::promql_rows::with_series_identity(&original) + .unwrap(), + ); + let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&root), + ) + .candidate(&root) + .unwrap(); + let dag = compile_post_asap_dag(&selected).unwrap(); + let raw = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .unwrap(); + let inputs = BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(Arc::new(raw.output_schema.clone())), + )]); + let roots = [u64::from(dag.root.0)]; + let compiled = compile(&dag, inputs.clone(), &roots).unwrap(); + // The root reads its population through a Sort helper numbered by the root. + let helper = u64::MAX - roots[0]; + assert_eq!(compiled.operator_name(helper), Some("Sort")); + assert!( + cut_candidate(&compiled, &[helper]).is_err(), + "helper operators are not Planner boundaries" + ); + assert_cuts_match_recompilation(&dag, inputs, &roots, 2); +} + +/// Cuts reject frontiers that recompilation rejects: duplicates, inputs, +/// unknown IDs, and an output shadowed by its descendant. +#[test] +fn cut_candidate_rejects_invalid_frontiers() { + let dag = grouped_rate(); + let state = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .unwrap(); + let readout = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator + } + ) + }) + .unwrap(); + let (state_id, rate_id, root) = ( + u64::from(state.id.0), + u64::from(readout.id.0), + u64::from(dag.root.0), + ); + let inputs = BTreeMap::from([( + state_id, + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + let compiled = compile(&dag, inputs.clone(), &[root]).unwrap(); + for frontier in [ + vec![rate_id, rate_id], + vec![state_id], + vec![999], + vec![root, rate_id], + ] { + assert!(cut_candidate(&compiled, &frontier).is_err(), "{frontier:?}"); + assert!(compile_candidate(&dag, inputs.clone(), &[root], &frontier).is_err()); + } +} From d1a82ae3720656ffa6b0639a29da2a7c3004a671 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 22:58:56 +0000 Subject: [PATCH 17/59] docs: describe compile-once candidate cuts Co-Authored-By: Claude Opus 5.5 --- .../physical-planning-and-deployment.md | 11 +++++++++ docs/develop_docs/library-api.md | 24 +++++++++++++++++++ 2 files changed, 35 insertions(+) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 69e9b90a..c3995b52 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -415,6 +415,17 @@ feasibility is rejected before pricing. The optimizer supplies candidate frontiers and cost evidence, including updates, retention, recurrence and sharing. `enumerate_frontiers` constructs bounded, reachable antichain frontiers above explicit input boundaries, including query-only and fully precomputed results. It fails explicitly when the candidate budget is exceeded. Maintenance selection must still reject frontiers that violate window, freshness, or reuse requirements; deployment feasibility is checked before pricing. +Lowering a node does not depend on the chosen frontier, so Planner lowers each +query DAG once. `compile(dag, inputs, roots)` yields the complete +`CompiledPhysicalDag`; `enumerate_compiled_frontiers(&compiled, max)` and +`cut_candidate(&compiled, frontier)` then derive each placement by partitioning +its operators: nodes above the frontier and below the roots form the query DAG, +and the frontier's ancestors form the precompute DAG. Helper operators are +numbered by their Planner node, so a cut is byte-identical to lowering that +frontier directly. `compile_candidate(s)` and `enumerate_frontiers` are wrappers +over this path. A deployment compiles each query DAG once, not once per +placement choice. Temporal pane candidates remain a separate lowering. + Physical compilation opens no readers. Bounded precompute outputs become typed query inputs. Their source, filters, grouping, build window, evaluation time, readiness and revision contracts must accompany the selected lifecycle and be checked during diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index cb541b94..54bcdce7 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -653,6 +653,30 @@ an alternative with `MissingCostEvidence` is accepted only when the cost model's complete-candidate hook covers lifecycle costs. Window frameworks and totals come from that hook, as in Planner selection. +A lifecycle choice then fixes each physical placement: for example, a +continuously maintained state places its producer in precompute, while an +ephemeral one keeps it in the query. Compile each query's `PostAsapDag` once +and derive every placement from that result: + +```rust +use asap_physical_operators::physical_planner::{ + compile, cut_candidate, enumerate_compiled_frontiers, +}; + +let compiled = compile(&dag, inputs, &roots)?; // each node lowered once +for frontier in enumerate_compiled_frontiers(&compiled, 4096)? { + // Precompute/query DAGs split at `frontier`; no logical lowering. + let candidate = cut_candidate(&compiled, &frontier)?; + // Check feasibility and price `candidate`; bind the selected one as is. +} +``` + +`cut_candidate` returns exactly the `PhysicalCandidate` that +`compile_candidate(&dag, inputs, &roots, &frontier)` returns, and rejects the +same invalid frontiers. The mapping from lifecycle choice to frontier stays with +the caller. Temporal pane candidates are a different lowering and still use +`compile_temporal_pane_candidate`. + ## Optional whole-plan selection and DAG assembly ### What does global selection mean? From 8b59b4b420cd11b1f150cfc15c5b2d10a4c66083 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:14:40 +0000 Subject: [PATCH 18/59] test: compare cut candidates by their serialized form The #462 split no longer exposes PhysicalCandidate::encode; its serde form gives the same byte-for-byte comparison. Co-Authored-By: Claude Opus 5.5 --- crates/asap-physical-operators/tests/precompute_candidates.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index f402e3dc..f0d63488 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -597,8 +597,8 @@ fn assert_cuts_match_recompilation( let cut = cut_candidate(&compiled, frontier).unwrap(); let expected = recompiled_candidate(dag, &inputs, roots, frontier).unwrap(); assert_eq!( - cut.encode().unwrap(), - expected.encode().unwrap(), + serde_json::to_vec(&cut).unwrap(), + serde_json::to_vec(&expected).unwrap(), "{frontier:?}" ); } From d9c794dc30e831d2a1058c1a3392f35c107de582 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:47:53 +0000 Subject: [PATCH 19/59] feat: derive the physical frontier from lifecycle timing The lifecycle layer assigns each node's timing; physical compilation now reads it. `frontier_from_timing` returns the ingestion-time nodes read by query-time nodes (or an ingestion-time root) and rejects a query-time node feeding an ingestion-time one, so each lifecycle assignment is a `cut_candidate` of one `compile` result. `enumerate_compiled_frontiers` is private: placement comes from timing, and its only caller is `enumerate_frontiers`. Co-Authored-By: Claude Opus 5.5 --- .../src/physical_planner/candidates.rs | 122 +++++++++++++++++- .../src/physical_planner/mod.rs | 4 +- .../tests/precompute_candidates.rs | 12 +- 3 files changed, 125 insertions(+), 13 deletions(-) diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs index ebab0d31..f5658a9a 100644 --- a/crates/asap-physical-operators/src/physical_planner/candidates.rs +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -86,6 +86,41 @@ pub fn cut_candidate( }) } +/// Materialization frontier implied by lifecycle-assigned timing: ingestion-time +/// nodes read by a query-time node, plus the root when it is ingestion-timed. +/// `cut_candidate` of one [`compile`] result with this frontier realizes the +/// assignment, so different assignments are different cuts of one lowering. +/// That holds while timing-dependent lowering (an ingestion-time `Binary` +/// aligns by value column) has the same timing at compile time as here. +/// A query-time node feeding an ingestion-time node has no valid placement. +pub fn frontier_from_timing(dag: &PostAsapDag) -> Result, Error> { + use planner_types::post_asap::ExecutionTiming::IngestionTime; + let timing = dag + .nodes + .iter() + .map(|node| (node.id, node.output_state.timing)) + .collect::>(); + let mut frontier = BTreeSet::new(); + if timing.get(&dag.root) == Some(&IngestionTime) { + frontier.insert(u64::from(dag.root.0)); + } + for edge in &dag.edges { + let (Some(&producer), Some(&consumer)) = + (timing.get(&edge.producer), timing.get(&edge.consumer)) + else { + return Err(invalid("timed DAG edge names an unknown node")); + }; + match (producer == IngestionTime, consumer == IngestionTime) { + (true, false) => { + frontier.insert(u64::from(edge.producer.0)); + } + (false, true) => return Err(invalid("query-time node feeds an ingestion-time node")), + _ => {} + } + } + Ok(frontier.into_iter().collect()) +} + /// Enumerate bounded, reachable materialization frontiers above explicit inputs. /// Each frontier is an antichain: storing an output and its ancestor together /// would leave the ancestor unused by query execution. Lifecycle eligibility @@ -100,9 +135,7 @@ pub fn enumerate_frontiers( enumerate_compiled_frontiers(&compile(dag, inputs.clone(), roots)?, max_candidates) } -/// [`enumerate_frontiers`] over an existing [`compile`] result, so enumeration -/// and [`cut_candidate`] share one lowering. -pub fn enumerate_compiled_frontiers( +fn enumerate_compiled_frontiers( compiled: &CompiledPhysicalDag, max_candidates: usize, ) -> Result>, Error> { @@ -364,4 +397,87 @@ mod tests { assert!(once > 0); assert_eq!(lowered() - before, once); } + + fn with_timing( + dag: &PostAsapDag, + timing: impl Fn(&PostAsapDagNode) -> planner_types::post_asap::ExecutionTiming, + ) -> PostAsapDag { + let mut timed = dag.clone(); + for node in &mut timed.nodes { + node.output_state.timing = timing(node); + } + for edge in &mut timed.edges { + let producer = timed.nodes.iter().find(|node| node.id == edge.producer); + edge.data_state = producer.unwrap().output_state; + } + timed + } + + fn raw_input(dag: &PostAsapDag) -> BTreeMap { + let raw = dag + .nodes + .iter() + .find(|node| matches!(node.payload, Payload::Fallback { .. })) + .unwrap(); + BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(Arc::new(raw.output_schema.clone())), + )]) + } + + /// Cutting one compilation by a retained-state timing and by the all + /// query-time timing (what ContinuouslyMaintained and Ephemeral assign) + /// lowers each Planner node once and matches `compile_candidate`. + #[test] + fn timing_cuts_share_one_lowering() { + use planner_types::post_asap::ExecutionTiming::QueryTime; + let (retained, _, root) = grouped_rate(); + let ephemeral = with_timing(&retained, |_| QueryTime); + let inputs = raw_input(&retained); + let lowered = || crate::physical_planner::LOWERED_NODES.with(|count| count.get()); + let before = lowered(); + let compiled = compile(&ephemeral, inputs.clone(), &[root]).unwrap(); + let once = lowered() - before; + let cuts = [&retained, &ephemeral].map(|timed| { + let frontier = frontier_from_timing(timed).unwrap(); + let cut = cut_candidate(&compiled, &frontier).unwrap(); + (timed, frontier, cut) + }); + assert!(once > 0); + assert_eq!(lowered() - before, once); + assert_eq!(cuts[0].1.len(), 1); + assert!(cuts[1].1.is_empty()); + for (timed, frontier, cut) in cuts { + let expected = compile_candidate(timed, inputs.clone(), &[root], &frontier).unwrap(); + assert_eq!( + serde_json::to_vec(&cut).unwrap(), + serde_json::to_vec(&expected).unwrap() + ); + } + } + + /// The frontier is the ingestion-time nodes read at query time; an + /// ingestion-time root is itself the frontier. + #[test] + fn frontier_from_timing_includes_ingestion_root() { + use planner_types::post_asap::ExecutionTiming::IngestionTime; + let (dag, _, root) = grouped_rate(); + let timed = with_timing(&dag, |_| IngestionTime); + assert_eq!(frontier_from_timing(&timed).unwrap(), [root]); + } + + /// A query-time node feeding an ingestion-time node is rejected. + #[test] + fn frontier_from_timing_rejects_query_time_input_to_ingestion() { + use planner_types::post_asap::ExecutionTiming::{IngestionTime, QueryTime}; + let (dag, _, _) = grouped_rate(); + let timed = with_timing(&dag, |node| { + if node.id == dag.root { + IngestionTime + } else { + QueryTime + } + }); + assert!(frontier_from_timing(&timed).is_err()); + } } diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index e852add8..732cdf28 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -36,8 +36,8 @@ pub mod promql_values; mod candidates; pub use candidates::{ - compile_candidate, compile_candidates, cut_candidate, enumerate_compiled_frontiers, - enumerate_frontiers, select_candidate, CandidateCost, CandidateSelection, PhysicalCandidate, + compile_candidate, compile_candidates, cut_candidate, enumerate_frontiers, + frontier_from_timing, select_candidate, CandidateCost, CandidateSelection, PhysicalCandidate, }; mod compiled; diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index f0d63488..65345c76 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -4,9 +4,9 @@ use asap_physical_operators::{ factory::create_planner_accumulator, operators::Operator, physical_planner::{ - compile, compile_candidate, compile_candidates, cut_candidate, - enumerate_compiled_frontiers, enumerate_frontiers, select_candidate, CandidateCost, - CompiledPhysicalDag, InputContract, PhysicalCandidate, Source, + compile, compile_candidate, compile_candidates, cut_candidate, enumerate_frontiers, + select_candidate, CandidateCost, CompiledPhysicalDag, InputContract, PhysicalCandidate, + Source, }, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, @@ -587,11 +587,7 @@ fn assert_cuts_match_recompilation( min_frontiers: usize, ) { let compiled = compile(dag, inputs.clone(), roots).unwrap(); - let frontiers = enumerate_compiled_frontiers(&compiled, 4096).unwrap(); - assert_eq!( - frontiers, - enumerate_frontiers(dag, &inputs, roots, 4096).unwrap() - ); + let frontiers = enumerate_frontiers(dag, &inputs, roots, 4096).unwrap(); assert!(frontiers.len() >= min_frontiers, "{frontiers:?}"); for frontier in &frontiers { let cut = cut_candidate(&compiled, frontier).unwrap(); From 437a5dc31012bd6e134d3cd106a8b7588d79aa14 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:47:53 +0000 Subject: [PATCH 20/59] test: cut one compilation by chosen lifecycle timings For the KLL quantile and grouped Rate->Sum fixtures, ContinuouslyMaintained and Ephemeral timed DAGs cut one compilation into exactly the candidates `compile_candidate` builds. The hand-written timing frontier in the chosen lifecycle test now uses `frontier_from_timing`. Co-Authored-By: Claude Opus 5.5 --- .../summary_maintenance_lifecycle_e2e.rs | 186 +++++++++++------- 1 file changed, 119 insertions(+), 67 deletions(-) diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index 77cdd3d7..49be34ce 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -212,6 +212,15 @@ fn selected_plan_with_horizon( .into_iter() .next() .expect("one normalized workload entry"); + selected_plan_for_lowered(workload, lowered, model, horizon) +} + +fn selected_plan_for_lowered( + workload: &PlanningWorkload, + lowered: asap_types::pre_asap::QueryExpr, + model: &dyn CostModel, + horizon: Horizon, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { let root = Rc::new(lowered); let strategies = asap_aware_mapping::default_strategies_with(model); let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); @@ -434,32 +443,70 @@ fn quantile_workload(query: &str) -> PlanningWorkload { workload } -/// Precompute outputs implied by timing: ingestion-time nodes read by a -/// query-time node, or the root when it is itself ingestion-timed. -fn ingestion_frontier(dag: &asap_types::post_asap::PostAsapDag) -> Vec { - use asap_types::post_asap::ExecutionTiming::IngestionTime; - let timing = |id| { - dag.nodes - .iter() - .find(|node| node.id == id) - .unwrap() - .output_state - .timing - }; - let mut frontier: Vec<_> = dag +/// Timed DAG for `query` after binding every summary state to `lifecycle`. +/// Grouped queries carry a physical series identity, as per-entity state needs. +fn lifecycle_timed_dag( + query: &str, + lifecycle: &SummaryMaintenanceLifecycle, +) -> (asap_types::post_asap::PostAsapDag, Vec) { + use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; + let workload = quantile_workload(query); + let mut lowered = lower_promql_workload(&workload, 0).unwrap().remove(0); + if query.contains(" by(") { + lowered = + asap_physical_operators::physical_planner::promql_rows::with_series_identity(&lowered) + .unwrap(); + } + let root = + selected_plan_for_lowered(&workload, lowered, &FullyCostedRuntime, Horizon(100.)).root; + let candidates = enumerate_summary_maintenance_lifecycles( + root, + WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + NOW_MS, + Some(Horizon(100.)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &FullyCostedRuntime, + ) + .unwrap(); + let choices: Vec<_> = candidates + .deployments() + .iter() + .map(|deployment| (deployment.post_asap_node_id, lifecycle.clone())) + .collect(); + let mut states: Vec<_> = choices.iter().map(|(id, _)| u64::from(id.0)).collect(); + states.sort_unstable(); + let dag = candidates + .select(&choices) + .unwrap() + .execution_timed_dag() + .unwrap(); + (dag, states) +} + +/// Compile inputs for a timed DAG: its raw source, available at either phase. +fn raw_inputs( + dag: &asap_types::post_asap::PostAsapDag, +) -> std::collections::BTreeMap { + let raw = dag .nodes .iter() - .filter(|node| { - node.output_state.timing == IngestionTime - && (node.id == dag.root - || dag.edges.iter().any(|edge| { - edge.producer == node.id && timing(edge.consumer) != IngestionTime - })) + .find(|node| { + matches!( + node.payload, + asap_types::post_asap::PostAsapOperatorPayload::Fallback { .. } + ) }) - .map(|node| u64::from(node.id.0)) - .collect(); - frontier.sort_unstable(); - frontier + .unwrap(); + std::collections::BTreeMap::from([( + u64::from(raw.id.0), + asap_physical_operators::physical_planner::InputContract::bounded(std::sync::Arc::new( + raw.output_schema.clone(), + )), + )]) } /// For existing PromQL fixtures, Planner's own retained lifecycle selection @@ -492,62 +539,29 @@ fn planner_lifecycle_selection_reproduces_strategy_timing() { /// precompute empty and reads the raw source at query time; both answer alike. #[test] fn chosen_lifecycle_timing_decides_precompute_contents() { - use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; use asap_physical_operators::{ - physical_planner::{compile_candidate, InputContract}, + physical_planner::{compile_candidate, frontier_from_timing}, runtime::Scope, values::{Batch, Value}, }; - use asap_types::{ - post_asap::{PostAsapOperatorPayload, SummaryFamilyType}, - pre_asap::DataType, - }; - use std::{collections::BTreeMap, sync::Arc}; + use asap_types::{post_asap::SummaryFamilyType, pre_asap::DataType}; + use std::collections::BTreeMap; - let workload = quantile_workload("quantile(0.99, latency)"); - let root = selected_plan(&workload).root; let mut answers = Vec::new(); for lifecycle in [ SummaryMaintenanceLifecycle::ContinuouslyMaintained, SummaryMaintenanceLifecycle::Ephemeral, ] { - let candidates = enumerate_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data( - &workload.query_workload, - workload.data_workload.as_ref().unwrap(), - &[1], - ), - NOW_MS, - Some(Horizon(100.)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &FullyCostedRuntime, - ) - .unwrap(); - let [deployment] = candidates.deployments() else { + let (dag, states) = lifecycle_timed_dag("quantile(0.99, latency)", &lifecycle); + let [state] = states[..] else { panic!("one summary state"); }; - let id = deployment.post_asap_node_id; - let state = u64::from(id.0); - let dag = candidates - .select(&[(id, lifecycle.clone())]) - .unwrap() - .execution_timed_dag() - .unwrap(); - let raw = dag - .nodes - .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) - .unwrap(); - let (raw_id, schema) = (u64::from(raw.id.0), Arc::new(raw.output_schema.clone())); - let frontier = ingestion_frontier(&dag); - let candidate = compile_candidate( - &dag, - BTreeMap::from([(raw_id, InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], - &frontier, - ) - .unwrap(); + let inputs = raw_inputs(&dag); + let (&raw_id, contract) = inputs.iter().next().unwrap(); + let schema = contract.schema.clone(); + let frontier = frontier_from_timing(&dag).unwrap(); + let candidate = + compile_candidate(&dag, inputs, &[u64::from(dag.root.0)], &frontier).unwrap(); let rows = (1..=100) .map(|value| { schema @@ -616,3 +630,41 @@ fn chosen_lifecycle_timing_decides_precompute_contents() { assert_eq!(answers[0], answers[1]); assert_eq!(answers[0].len(), 1); } + +/// One compilation, cut by each lifecycle assignment's timing, yields exactly +/// the candidate `compile_candidate` builds for that timed DAG: the retained +/// state is the frontier under ContinuouslyMaintained, and nothing under +/// Ephemeral. Covers the KLL quantile fixture and grouped Rate→Sum. +#[test] +fn lifecycle_timing_cuts_one_compilation() { + use asap_physical_operators::physical_planner::{ + compile, compile_candidate, cut_candidate, frontier_from_timing, + }; + for query in ["quantile(0.99, latency)", "sum by(job)(rate(m[1m]))"] { + let ephemeral = SummaryMaintenanceLifecycle::Ephemeral; + let (compiled_dag, _) = lifecycle_timed_dag(query, &ephemeral); + let inputs = raw_inputs(&compiled_dag); + let roots = [u64::from(compiled_dag.root.0)]; + let compiled = compile(&compiled_dag, inputs.clone(), &roots).unwrap(); + for lifecycle in [ + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + ephemeral, + ] { + let (dag, states) = lifecycle_timed_dag(query, &lifecycle); + let frontier = frontier_from_timing(&dag).unwrap(); + let expected_frontier = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { + vec![] + } else { + states + }; + assert_eq!(frontier, expected_frontier, "{query} {lifecycle:?}"); + let cut = cut_candidate(&compiled, &frontier).unwrap(); + let expected = compile_candidate(&dag, inputs.clone(), &roots, &frontier).unwrap(); + assert_eq!( + serde_json::to_vec(&cut).unwrap(), + serde_json::to_vec(&expected).unwrap(), + "{query} {lifecycle:?}" + ); + } + } +} From d133d35465eff96bbe2f13bfbe9da9a2e1cfaf63 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:47:54 +0000 Subject: [PATCH 21/59] docs: describe timing-derived candidate cuts Co-Authored-By: Claude Opus 5.5 --- .../physical-planning-and-deployment.md | 26 ++++++++++++------- docs/develop_docs/library-api.md | 24 ++++++++++------- 2 files changed, 30 insertions(+), 20 deletions(-) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index c3995b52..f64eb0cf 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -415,16 +415,21 @@ feasibility is rejected before pricing. The optimizer supplies candidate frontiers and cost evidence, including updates, retention, recurrence and sharing. `enumerate_frontiers` constructs bounded, reachable antichain frontiers above explicit input boundaries, including query-only and fully precomputed results. It fails explicitly when the candidate budget is exceeded. Maintenance selection must still reject frontiers that violate window, freshness, or reuse requirements; deployment feasibility is checked before pricing. -Lowering a node does not depend on the chosen frontier, so Planner lowers each -query DAG once. `compile(dag, inputs, roots)` yields the complete -`CompiledPhysicalDag`; `enumerate_compiled_frontiers(&compiled, max)` and -`cut_candidate(&compiled, frontier)` then derive each placement by partitioning -its operators: nodes above the frontier and below the roots form the query DAG, -and the frontier's ancestors form the precompute DAG. Helper operators are -numbered by their Planner node, so a cut is byte-identical to lowering that -frontier directly. `compile_candidate(s)` and `enumerate_frontiers` are wrappers -over this path. A deployment compiles each query DAG once, not once per -placement choice. Temporal pane candidates remain a separate lowering. +The lifecycle layer decides timing; physical compilation reads it. Lowering a +node does not depend on the frontier, so each query DAG is lowered once and +different lifecycle assignments are different cuts of that lowering. +`compile(dag, inputs, roots)` yields the complete `CompiledPhysicalDag`. +`frontier_from_timing(&timed_dag)` reads an assignment's timed DAG (from +`execution_timed_dag`) and returns its frontier: ingestion-time nodes read by +query-time nodes, or an ingestion-time root; a query-time node feeding an +ingestion-time node is rejected. `cut_candidate(&compiled, &frontier)` then +partitions the lowered operators: the frontier's ancestors form the precompute +DAG and the rest form the query DAG. Helper operators are numbered by their +Planner node (`u64::MAX - node_id`), so a cut is byte-identical to +`compile_candidate` for that frontier. One exception: an ingestion-time +`Binary` lowers differently, so its timing must match at compile time. +`compile_candidate(s)` and `enumerate_frontiers` wrap the same path. Temporal +pane candidates remain a separate lowering. Physical compilation opens no readers. Bounded precompute outputs become typed query inputs. Their source, filters, grouping, build window, evaluation time, readiness and @@ -551,6 +556,7 @@ operator/runtime fixtures: | --- | --- | | `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled precompute/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | | `summary_maintenance_lifecycle_e2e::chosen_lifecycle_timing_decides_precompute_contents` | PromQL workload → enumerated lifecycles → explicit choice → timed DAG → compiled candidate; ContinuouslyMaintained stores the state in precompute, Ephemeral leaves precompute empty and reads the raw source at query time; both return the same p99 | +| `summary_maintenance_lifecycle_e2e::lifecycle_timing_cuts_one_compilation` | KLL quantile and grouped Rate→Sum: one compilation cut by the ContinuouslyMaintained and Ephemeral timed DAGs equals `compile_candidate` for each; the frontier is the retained state or empty | | `summary_maintenance_lifecycle_e2e::planner_lifecycle_selection_reproduces_strategy_timing` | For PromQL fixtures, the timed DAG from Planner's retained selection equals the DAG realization strategies produce today | | `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute precompute DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | | `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index 54bcdce7..4c77535c 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -653,28 +653,32 @@ an alternative with `MissingCostEvidence` is accepted only when the cost model's complete-candidate hook covers lifecycle costs. Window frameworks and totals come from that hook, as in Planner selection. -A lifecycle choice then fixes each physical placement: for example, a -continuously maintained state places its producer in precompute, while an -ephemeral one keeps it in the query. Compile each query's `PostAsapDag` once -and derive every placement from that result: +A lifecycle choice then fixes each physical placement through timing: a +continuously maintained state and its inputs run at ingestion time, while an +ephemeral one stays at query time. Compile each query's `PostAsapDag` once and +cut every chosen assignment from that result: ```rust use asap_physical_operators::physical_planner::{ - compile, cut_candidate, enumerate_compiled_frontiers, + compile, cut_candidate, frontier_from_timing, }; let compiled = compile(&dag, inputs, &roots)?; // each node lowered once -for frontier in enumerate_compiled_frontiers(&compiled, 4096)? { +for plan in lifecycle_plans { + let frontier = frontier_from_timing(&plan.execution_timed_dag()?)?; // Precompute/query DAGs split at `frontier`; no logical lowering. let candidate = cut_candidate(&compiled, &frontier)?; // Check feasibility and price `candidate`; bind the selected one as is. } ``` -`cut_candidate` returns exactly the `PhysicalCandidate` that -`compile_candidate(&dag, inputs, &roots, &frontier)` returns, and rejects the -same invalid frontiers. The mapping from lifecycle choice to frontier stays with -the caller. Temporal pane candidates are a different lowering and still use +The frontier is the set of ingestion-time nodes read by query-time nodes (or an +ingestion-time root). `frontier_from_timing` rejects a query-time node feeding +an ingestion-time node. `cut_candidate` returns exactly what +`compile_candidate(&dag, inputs, &roots, &frontier)` returns and rejects the +same invalid frontiers. If the DAG has an ingestion-time `Binary`, compile with +the same timing for that node, because it lowers differently. Temporal pane +candidates are a different lowering and still use `compile_temporal_pane_candidate`. ## Optional whole-plan selection and DAG assembly From b89bd24e04d5718d736cc9c3f6e2cf701f40968a Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 03:14:09 +0000 Subject: [PATCH 22/59] refactor: allow several helper operators per Planner node Number helpers as u64::MAX - (node << 16) - index so a node that lowers to an operator chain keeps deterministic, traversal-independent helper IDs. Co-Authored-By: Claude Opus 5.5 --- .../src/physical_planner/mod.rs | 14 ++++++++++---- .../tests/precompute_candidates.rs | 2 +- .../physical-planning-and-deployment.md | 2 +- 3 files changed, 12 insertions(+), 6 deletions(-) diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 732cdf28..fede3777 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -109,6 +109,15 @@ thread_local! { static LOWERED_NODES: std::cell::Cell = const { std::cell::Cell::new(0) }; } +/// Helper operators are numbered from their Planner node alone, above the u32 +/// Planner ID range, so every boundary choice yields a subgraph of the same +/// lowering and candidate cuts need not renumber operators. A node lowering to +/// several helpers takes consecutive indices below its base. +fn helper_id(node: NodeId, index: u64) -> NodeId { + debug_assert!(node <= u64::from(u32::MAX) && index < 1 << 16); + u64::MAX - (node << 16) - index +} + fn compile_internal( dag: &PostAsapDag, mut sources: BTreeMap, @@ -167,10 +176,7 @@ fn compile_internal( let mut graph = CompiledPhysicalDag::new(roots.to_vec()); for id in ordered { let node = nodes[&id]; - // At most one helper operator per node, numbered above the u32 Planner - // ID range by its node alone, so every boundary choice yields a subgraph - // of the same lowering and candidate cuts need not renumber operators. - let auxiliary = u64::MAX - id; + let auxiliary = helper_id(id, 0); let output = Arc::new(node.output_schema.clone()); crate::values::validate_schema(&output)?; if let Some(source) = sources.remove(&id) { diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index 65345c76..e1bf9080 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -670,7 +670,7 @@ fn population_topk_cuts_equal_per_frontier_compilation() { let roots = [u64::from(dag.root.0)]; let compiled = compile(&dag, inputs.clone(), &roots).unwrap(); // The root reads its population through a Sort helper numbered by the root. - let helper = u64::MAX - roots[0]; + let helper = u64::MAX - (roots[0] << 16); assert_eq!(compiled.operator_name(helper), Some("Sort")); assert!( cut_candidate(&compiled, &[helper]).is_err(), diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index f64eb0cf..6bddfa6b 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -425,7 +425,7 @@ query-time nodes, or an ingestion-time root; a query-time node feeding an ingestion-time node is rejected. `cut_candidate(&compiled, &frontier)` then partitions the lowered operators: the frontier's ancestors form the precompute DAG and the rest form the query DAG. Helper operators are numbered by their -Planner node (`u64::MAX - node_id`), so a cut is byte-identical to +Planner node (`u64::MAX - (node_id << 16) - index`), so a cut is byte-identical to `compile_candidate` for that frontier. One exception: an ingestion-time `Binary` lowers differently, so its timing must match at compile time. `compile_candidate(s)` and `enumerate_frontiers` wrap the same path. Temporal From 1a7cbf27e23a397ae5b7c5b26174f8bc275e3ad5 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 03:19:33 +0000 Subject: [PATCH 23/59] test: expect retained frontier states read at query time or at the root With several retained states, only those read by a query-time node or forming the root are cut points. Co-Authored-By: Claude Opus 5.5 --- .../tests/summary_maintenance_lifecycle_e2e.rs | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index 49be34ce..f58e7ac7 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -652,10 +652,28 @@ fn lifecycle_timing_cuts_one_compilation() { ] { let (dag, states) = lifecycle_timed_dag(query, &lifecycle); let frontier = frontier_from_timing(&dag).unwrap(); + // Retained states read by a query-time consumer, or the root itself. + let query_time = |id: u64| { + dag.nodes.iter().any(|node| { + u64::from(node.id.0) == id + && node.output_state.timing + == asap_types::post_asap::ExecutionTiming::QueryTime + }) + }; let expected_frontier = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { vec![] } else { states + .iter() + .copied() + .filter(|state| { + *state == u64::from(dag.root.0) + || dag.edges.iter().any(|edge| { + u64::from(edge.producer.0) == *state + && query_time(u64::from(edge.consumer.0)) + }) + }) + .collect() }; assert_eq!(frontier, expected_frontier, "{query} {lifecycle:?}"); let cut = cut_candidate(&compiled, &frontier).unwrap(); From 0ca3062dce3cf81a873f26ece55b163a1eae5cc7 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 03:06:21 +0000 Subject: [PATCH 24/59] fix: keep a selected grouped Sum over realized Rate readouts DAG assembly replaced any selected outer Sum over an inner aggregate with a query-time exact Sum, even when the selected summary realizes the inner Rate itself. The grouped Sum state therefore never reached the inventory, and no lifecycle choice could move grouped Sum into precompute. Assembly now keeps such a selected summary; the query-time residual still applies when the outer summary would hide its inner aggregate in KeepPreAsap. Default selection for sum by(job)(rate(...)) now yields Rate -> grouped Sum state; the physical frontier test reads its query root accordingly. Conflicts with earlier stack changes resolved to the integration tree: - crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs: c98281a Merge remote-tracking branch 'origin/feat/compile-once-cuts' into integration/planner-for-backend Co-Authored-By: Claude Opus 5.5 --- crates/asap-aware-mapping/src/replacement.rs | 76 +++++++++- .../tests/precompute_candidates.rs | 7 +- .../summary_maintenance_lifecycle_e2e.rs | 130 ++++++++++++++++++ 3 files changed, 210 insertions(+), 3 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index fb697251..59fc05be 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -5249,6 +5249,9 @@ impl<'a> GlobalSelection<'a> { if let Some(node) = self.assembled_nodes.borrow().get(&ptr) { return Ok(Rc::clone(node)); } + // A selected summary that realizes its inner aggregate, instead of + // hiding it in `KeepPreAsap`, is kept; lifecycle assignment decides + // whether it runs in precompute or at query time. let selected_composed_summary = self .groups .get(&ptr) @@ -5256,7 +5259,7 @@ impl<'a> GlobalSelection<'a> { .is_some_and(|candidate| matches!(&candidate.replacement, Replacement::Summary(node) if matches!(&node.expr, SummaryExpr::SummaryAgg { child, .. } - if matches!(&child.expr, SummaryExpr::KeepPreAsap(raw) if !contains_aggregate(raw))))); + if !matches!(&child.expr, SummaryExpr::KeepPreAsap(raw) if contains_aggregate(raw))))); let node = if query_time_nested_sum(target) && !selected_composed_summary { self.assemble_residual(target)? } else { @@ -6955,6 +6958,77 @@ mod tests { use asap_types::types::AccuracyTarget; use std::collections::HashMap; + // Candidate shape without execution timing: what is computed, not where. + fn timing_free_shape(node: &Rc) -> serde_json::Value { + fn strip(value: &mut serde_json::Value) { + match value { + serde_json::Value::Object(fields) => { + fields.remove("timing"); + fields.values_mut().for_each(strip); + } + serde_json::Value::Array(values) => values.iter_mut().for_each(strip), + _ => {} + } + } + let mut shape = + serde_json::to_value(asap_types::post_asap::compile_post_asap_dag(node).unwrap()) + .unwrap(); + strip(&mut shape); + shape + } + + // Rate inventories never offer two candidates that differ only in timing. + #[test] + fn rate_candidate_inventories_have_no_timing_only_duplicates() { + for (query, accuracy) in [ + ("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact), + ("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)), + ] { + let root = Rc::new(lower_promql(query, accuracy)); + let inventory = search_workload(vec![(0usize, root)]) + .enumerate_candidate_dags(4096) + .unwrap(); + let shapes = inventory + .candidates + .iter() + .map(|forest| timing_free_shape(&forest[0].1)) + .collect::>(); + for (i, shape) in shapes.iter().enumerate() { + assert!(!shapes[..i].contains(shape), "{query}: duplicate {i}"); + } + } + } + + // Grouped Sum over Rate readouts stays a summary state in the inventory, + // so lifecycle assignment can place it in precompute or at query time. + #[test] + fn grouped_rate_sum_inventory_keeps_sum_state_for_lifecycle_placement() { + let root = Rc::new(lower_promql( + "sum by(job)(rate(m[1m]))", + AccuracyTarget::Exact, + )); + let inventory = search_workload(vec![(0usize, root)]) + .enumerate_candidate_dags(4096) + .unwrap(); + let is_exact = |node: &SummaryNode, kind: ExactKind| { + matches!(&node.expr, SummaryExpr::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(k, _), .. + } if *k == kind) + }; + assert!(inventory.candidates.iter().any(|forest| { + let SummaryExpr::ValueOperation { child: sum, .. } = &forest[0].1.expr else { + return false; + }; + let SummaryExpr::SummaryAgg { child: rate, .. } = &sum.expr else { + return false; + }; + is_exact(sum, ExactKind::Sum) + && matches!(&rate.expr, SummaryExpr::ValueOperation { + child, operation: ValueOperation::FinalizeExactAccumulator, .. + } if is_exact(child, ExactKind::Rate)) + })); + } + // Every exposed query result has a readout; internal accumulator frontiers stay states. #[test] fn query_candidate_roots_do_not_leak_exact_accumulator_state() { diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index e1bf9080..f8050118 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -56,7 +56,7 @@ fn grouped_rate() -> PostAsapDag { let space = grouped_rate_space(); let selected = space .global_selection(&DefaultCostModel) - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_query(&space.roots[0].1) .unwrap() .unwrap(); compile_post_asap_dag(&selected).unwrap() @@ -108,7 +108,10 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { PostAsapOperatorPayload::Value { operation: ValueOperation::FinalizeExactAccumulator } - ) + ) && dag + .edges + .iter() + .any(|edge| edge.producer == state.id && edge.consumer == node.id) }) .unwrap(); let input_schema = Arc::new(state.output_schema.clone()); diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index f58e7ac7..e1a1f57b 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -686,3 +686,133 @@ fn lifecycle_timing_cuts_one_compilation() { } } } + +/// Grouped Rate→Sum is one inventory candidate: retaining the Sum state puts +/// Rate and Sum in precompute, while an `Ephemeral` Sum over a retained Rate +/// state leaves Sum in the query DAG. +#[test] +fn grouped_rate_sum_placement_is_a_lifecycle_choice() { + use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; + use asap_physical_operators::physical_planner::{compile_candidate, InputContract}; + use asap_types::post_asap::{ + ExactKind, PostAsapOperatorPayload, SummaryExpr, SummaryFamilyType, + }; + use std::{collections::BTreeMap, sync::Arc}; + + let workload = quantile_workload("sum by(job)(rate(m[1m]))"); + let root = Rc::new( + asap_physical_operators::physical_planner::promql_rows::with_series_identity( + &lower_promql_workload(&workload, 0).unwrap().remove(0), + ) + .unwrap(), + ); + let is_exact = |node: &SummaryNode, kind: ExactKind| { + matches!(&node.expr, SummaryExpr::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(k, _), .. + } if *k == kind) + }; + let inventory = asap_aware_mapping::search_workload(vec![("q", root)]) + .enumerate_candidate_dags(4096) + .unwrap(); + let candidates = inventory + .candidates + .into_iter() + .map(|mut forest| forest.remove(0).1) + .filter(|candidate| { + matches!(&candidate.expr, SummaryExpr::ValueOperation { child, .. } + if is_exact(child, ExactKind::Sum)) + }) + .collect::>(); + let [candidate] = candidates.as_slice() else { + panic!("one grouped Sum candidate, got {}", candidates.len()); + }; + let mut placements = Vec::new(); + for sum_lifecycle in [ + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + SummaryMaintenanceLifecycle::Ephemeral, + ] { + let lifecycles = enumerate_summary_maintenance_lifecycles( + Rc::clone(candidate), + WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + NOW_MS, + Some(Horizon(100.)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &FullyCostedRuntime, + ) + .unwrap(); + let choices = lifecycles + .deployments() + .iter() + .map(|deployment| { + let lifecycle = if is_exact(&deployment.summary, ExactKind::Sum) { + sum_lifecycle.clone() + } else { + SummaryMaintenanceLifecycle::ContinuouslyMaintained + }; + (deployment.post_asap_node_id, lifecycle) + }) + .collect::>(); + assert_eq!(choices.len(), 2, "Rate and Sum states"); + let dag = lifecycles + .select(&choices) + .unwrap() + .execution_timed_dag() + .unwrap(); + let raw = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .unwrap(); + let frontier = + asap_physical_operators::physical_planner::frontier_from_timing(&dag).unwrap(); + let [boundary] = frontier.as_slice() else { + panic!("one precompute output, got {frontier:?}"); + }; + let boundary = dag + .nodes + .iter() + .find(|node| u64::from(node.id.0) == *boundary) + .unwrap(); + let physical = compile_candidate( + &dag, + BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(Arc::new(raw.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &frontier, + ) + .unwrap(); + let json = |value| String::from_utf8(serde_json::to_vec(value).unwrap()).unwrap(); + placements.push(( + boundary.payload.clone(), + json(physical.precompute.as_ref().unwrap()), + json(&physical.query), + )); + } + let builds = |json: &str, kind: &str| { + json.contains(&format!( + r#"{{"SummaryBuild":{{"family":{{"ExactAggregate":["{kind}","{kind}"]}}"# + )) + }; + let [(retained, retained_pre, retained_query), (ephemeral, ephemeral_pre, ephemeral_query)] = + placements.as_slice() + else { + unreachable!() + }; + let state = |payload: &PostAsapOperatorPayload, kind: ExactKind| { + matches!(payload, PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(k, _), .. + } if *k == kind) + }; + assert!(state(retained, ExactKind::Sum)); + assert!(builds(retained_pre, "Rate") && builds(retained_pre, "Sum")); + assert!(!retained_query.contains("SummaryBuild")); + assert!(state(ephemeral, ExactKind::Rate)); + assert!(builds(ephemeral_pre, "Rate") && !builds(ephemeral_pre, "Sum")); + assert!(builds(ephemeral_query, "Sum")); +} From ff18c033503145f6bc174f5cf9bf44afd41fef52 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 03:06:22 +0000 Subject: [PATCH 25/59] refactor!: remove timing-only Rate placement candidates SketchAlgorithmStrategy::fixed_window_rate_candidates and query_time_rate_aggregation_candidates returned the same logical DAG as the ordinary heap or grouped Sum candidate with Rate finalization flipped between ingestion and query time. Placement now comes only from a chosen lifecycle via SummaryMaintenanceLifecyclePlan::execution_timed_dag. compile_fixed_window_rate_aggregation takes that lifecycle-timed PostAsapDag instead of a SummaryNode with baked-in timing. The fixed-window heap test binds continuously maintained lifecycles; the grouped Sum placement pair is covered by the lifecycle end-to-end test. Co-Authored-By: Claude Opus 5.5 --- crates/asap-aware-mapping/src/replacement.rs | 117 +----------- .../src/physical_planner/promql_rows.rs | 15 +- .../tests/weighted_topk_binding.rs | 178 +++++++++++------- 3 files changed, 118 insertions(+), 192 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 59fc05be..8c826b37 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -1356,114 +1356,6 @@ impl<'a> SketchAlgorithmStrategy<'a> { self.propose_with(&ranked, None) } - /// Fixed-window maintenance can finalize each series' counter state and - /// build a fresh heap or grouped Sum for that evaluation window. Deployment must provide - /// a complete, synchronized population and bind the matching window; this - /// candidate never incrementally adds one window's rates to another. - pub fn fixed_window_rate_candidates(&self, root: &Rc) -> Proposals { - fn place(node: &Rc) -> Option> { - let mut next = node.as_ref().clone(); - match &mut next.expr { - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing, - } if matches!(&child.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), - reduction: Reduction::PerEntity, child: source, .. - } if matches!(&source.expr, SummaryExpr::KeepPreAsap(source) if matches!(source.as_ref(), QueryExpr::TimeRange { .. }))) => - { - *timing = ExecutionTiming::IngestionTime; - } - SummaryExpr::ValueOperation { child, .. } - | SummaryExpr::SummaryAgg { child, .. } => *child = place(child)?, - SummaryExpr::SummaryEstimate { summary_input, .. } => { - *summary_input = place(summary_input)? - } - _ => return None, - } - Some(Rc::new(next)) - } - let mut proposals = self.propose_with(root, None); - proposals.candidates.retain_mut(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { - return false; - }; - let Ok(dag) = asap_types::post_asap::compile_post_asap_dag(node) else { - return false; - }; - if !dag.nodes.iter().any(|node| match &node.payload { - asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), - .. - } => matches!( - kind.algorithm(), - SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap - ), - asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), - .. - } => true, - _ => false, - }) { - return false; - } - let Some(placed) = place(node) else { - return false; - }; - if asap_types::post_asap::compile_post_asap_dag(&placed).is_err() { - return false; - } - let Ok(placed) = finalize_query_candidate(placed, root) else { - return false; - }; - candidate.replacement = Replacement::Summary(placed); - candidate - .rationale - .push_str("; fixed-window precompute over complete per-series counter states"); - true - }); - proposals - } - - /// Retain grouped Sum after a per-series Rate readout as a query-time - /// candidate alongside its complete-window maintenance placement. - pub fn query_time_rate_aggregation_candidates(&self, root: &Rc) -> Proposals { - fn query_time(node: &Rc) -> Rc { - let mut next = node.as_ref().clone(); - match &mut next.expr { - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing, - } if matches!( - &child.expr, - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), - .. - } - ) => - { - *timing = ExecutionTiming::QueryTime; - } - SummaryExpr::ValueOperation { child, .. } - | SummaryExpr::SummaryAgg { child, .. } => *child = query_time(child), - _ => {} - } - Rc::new(next) - } - let mut proposals = self.fixed_window_rate_candidates(root); - proposals.candidates.retain_mut(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { return false }; - if !matches!(&node.expr, SummaryExpr::ValueOperation { child, operation: ValueOperation::FinalizeExactAccumulator, .. } - if matches!(&child.expr, SummaryExpr::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), .. })) { return false; } - candidate.replacement = Replacement::Summary(query_time(node)); - candidate.rationale = "query-time grouped Sum over complete per-series Rate readouts".into(); - true - }); - proposals - } - pub(crate) fn from_planning_inputs(planning_inputs: CandidatePlanningInputs<'a>) -> Self { Self { planning_inputs } } @@ -2764,7 +2656,9 @@ fn realize_physical_summary_input( /// Emit `SummaryAgg` (recursively binding the child), plus the /// `SummaryEstimate` readout when `estimate` is set. // Retain the exact expression and schema while placing its value production -// on the update path. Read-time consumers keep their original shared nodes. +// on the update path. This is the initial layout for values feeding a summary; +// lifecycle timing is authoritative. Read-time consumers keep their original +// shared nodes. fn maintenance_exact_values(node: Rc) -> Option> { let expr = match &node.expr { // These guards can fall back at read time, but cannot recover a parent @@ -2993,8 +2887,9 @@ fn construct_summary_agg( }; Rc::clone(child) } else if snapshot_weighted { - // A fresh query-time summary consumes this evaluation's finalized rates. - // Moving rate snapshots must never accumulate across evaluations. + // Each evaluation's finalized rates feed a fresh summary; rate snapshots + // must never accumulate across evaluations. Query time is only the + // initial layout; a retained summary's lifecycle moves it to ingestion. finalize_query_candidate(bound_child, &input.child)? } else { let child = finalize_exact_accumulator_at( diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index 8b887580..d2c13328 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -249,16 +249,13 @@ pub fn compile_rate_ranking( Ok((source, program)) } -/// The selected logical placement requires fresh aggregate state per closed window. -/// Compile both physical graphs before deployment chooses storage or scheduling. -/// The input is the complete collection of per-series exact counter states. +/// Compile a lifecycle-timed DAG whose heap or grouped Sum over per-series +/// Rate readouts runs at ingestion time: fresh aggregate state per closed +/// window. The input is the complete collection of per-series counter states. pub fn compile_fixed_window_rate_aggregation( - selected: &Rc, + dag: &planner_types::post_asap::PostAsapDag, ) -> Result { - use planner_types::post_asap::{ - compile_post_asap_dag, ExactKind, ExecutionTiming, SketchAlgorithm, - }; - let dag = compile_post_asap_dag(selected).map_err(|e| invalid(e.to_string()))?; + use planner_types::post_asap::{ExactKind, ExecutionTiming, SketchAlgorithm}; let sources = dag .nodes .iter() @@ -310,7 +307,7 @@ pub fn compile_fixed_window_rate_aggregation( )); } compile_candidate( - &dag, + dag, BTreeMap::from([( u64::from(source.id.0), InputContract::bounded(Arc::new(source.output_schema.clone())), diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index 664ae799..ef086b1b 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -756,10 +756,102 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { } } -// Placement changes execution ownership only. Every fixed-window candidate -// contains Rate finalization before a fresh heap, with query readout downstream. +/// Deployment-side lifecycle choice: every summary state of `candidate` is +/// continuously maintained, and the chosen lifecycles set execution timing. +fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDag { + use asap_aware_mapping::{ + cost_model::{Cost, CostModel}, + enumerate_summary_maintenance_lifecycles, CostRate, Horizon, + SummaryMaintenanceCapabilities, SummaryMaintenanceLifecycleCapabilities, + SummaryMaintenanceLifecycleCostInputs, WorkloadDemand, + }; + use planner_types::workload::{ + DataArrival, Rate, RepeatedDemand, RepeatingEntry, RepetitionInterval, + }; + struct Costed; + impl CostModel for Costed { + fn rank_candidates( + &self, + _: &planner_types::pre_asap::agg_intent::AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + fn summary_maintenance_lifecycle_cost_inputs( + &self, + _: &SummaryNode, + ) -> SummaryMaintenanceLifecycleCostInputs { + SummaryMaintenanceLifecycleCostInputs { + build_cost: Some(Cost(10.)), + maintenance_cost_per_update: Some(Cost(1.)), + summary_read_cost: Some(Cost(1.)), + retention_cost_rate: Some(CostRate(0.1)), + retirement_cost: Some(Cost(1.)), + } + } + fn summary_maintenance_capabilities( + &self, + _: &SummaryNode, + ) -> SummaryMaintenanceCapabilities { + SummaryMaintenanceCapabilities { + incremental_update: true, + merge: true, + delete: true, + } + } + } + const NOW_MS: u64 = 1_000_000; + let queries = QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: None, + repeating_queries: Some(vec![RepeatingEntry { + query: Query("topk by(job)(2, rate(m[1m]))".into()), + demand: RepeatedDemand::FixedInterval(RepetitionInterval(60_000)), + requirements: QueryRequirements::default(), + predictability: Predictability::Predictable { known_at: None }, + time_selection: TimeSelection::default(), + }]), + }; + let data = DataWorkload { + arrival: DataArrival::ContinuouslyIngesting, + ingestion_rate: WorkloadEvidence { + value: Some(Rate(1.)), + source: planner_types::workload::EvidenceSource::Observed, + observed_at_ms: Some(NOW_MS), + valid_for_ms: Some(60_000), + }, + ..Default::default() + }; + let lifecycles = enumerate_summary_maintenance_lifecycles( + Rc::clone(candidate), + WorkloadDemand::new_with_data(&queries, &data, &[0]), + NOW_MS, + Some(Horizon(100.)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &Costed, + ) + .unwrap(); + let choices = lifecycles + .deployments() + .iter() + .map(|deployment| { + ( + deployment.post_asap_node_id, + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + ) + }) + .collect::>(); + lifecycles + .select(&choices) + .unwrap() + .execution_timed_dag() + .unwrap() +} + +// A maintained heap over finalized per-series Rate is the fixed-window +// placement: lifecycle timing, not a separate candidate, puts it in precompute. #[test] -fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { +fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { use asap_physical_operators::physical_planner::{ compile_candidate, promql_rows::with_series_identity, }; @@ -775,13 +867,17 @@ fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { &EqualSplitAllocator, &Evidence, ); - let candidates = strategy.fixed_window_rate_candidates(&root).candidates; + let candidates = strategy + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .filter_map(|candidate| match candidate.replacement { + Replacement::Summary(root) if candidate.rationale.contains("WithHeap") => Some(root), + _ => None, + }) + .collect::>(); assert_eq!(candidates.len(), 2); - for candidate in candidates { - let Replacement::Summary(root) = candidate.replacement else { - panic!() - }; - let dag = compile_post_asap_dag(&root).unwrap(); + for root in candidates { + let dag = continuously_maintained_dag(&root); let state = dag .nodes .iter() @@ -819,16 +915,11 @@ fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { &[u64::from(heap.id.0)], ) .unwrap(); - let exported = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_aggregation(&root).unwrap(); + let exported = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_aggregation(&dag).unwrap(); assert_eq!( serde_json::to_vec(&exported).unwrap(), serde_json::to_vec(&physical).unwrap() ); - assert!( - asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking(&root) - .is_err(), - "query binding must not move the selected precompute frontier" - ); // Execute the selected split across a state serialization boundary. // Each run builds fresh weights from that window's counters. let execute = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDag, @@ -959,60 +1050,3 @@ fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { ); } } - -// Grouped Rate has a legal stored Sum candidate as well as query-time reduction. -#[test] -fn grouped_rate_exposes_precomputed_sum_with_query_readout() { - let root = Rc::new( - asap_physical_operators::physical_planner::promql_rows::with_series_identity( - &lower_promql("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact).unwrap(), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, - &DefaultAccuracyModel, - &EqualSplitAllocator, - &Evidence, - ); - let direct = strategy.query_time_rate_aggregation_candidates(&root); - assert!( - direct.candidates.iter().any(|candidate| { - let Replacement::Summary(root) = &candidate.replacement else { - return false; - }; - let Ok((_, program)) = - asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking(root) - else { - return false; - }; - let output = program.output_contract(program.roots()[0]).unwrap(); - output - .schema - .fields - .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_))) - }), - "query-time grouped Rate must finalize Sum inside the physical graph" - ); - let candidates = strategy.fixed_window_rate_candidates(&root).candidates; - assert!( - !candidates.is_empty(), - "Planner must expose Rate -> grouped Sum at ingestion" - ); - for candidate in candidates { - let Replacement::Summary(root) = candidate.replacement else { - panic!() - }; - let physical = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_aggregation(&root).unwrap(); - let precompute = - String::from_utf8(serde_json::to_vec(&physical.precompute.unwrap()).unwrap()).unwrap(); - assert!( - precompute.contains("SummaryBuild") - && precompute.contains("Rate") - && precompute.contains("Sum") - ); - let query = String::from_utf8(serde_json::to_vec(&physical.query).unwrap()).unwrap(); - assert!(query.contains("Readout") && !query.contains("SummaryBuild")); - } -} From eb5885ca7cb3a6a25ae01f5d4652c2cde5e547c9 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 03:06:22 +0000 Subject: [PATCH 26/59] docs: describe grouped Rate->Sum placement as a lifecycle choice Co-Authored-By: Claude Opus 5.5 --- .../architecture/input-output-workflow.md | 4 +- .../physical-planning-and-deployment.md | 38 ++++++++++--------- 2 files changed, 24 insertions(+), 18 deletions(-) diff --git a/docs/design_docs/architecture/input-output-workflow.md b/docs/design_docs/architecture/input-output-workflow.md index fb9a7f42..32c0b093 100644 --- a/docs/design_docs/architecture/input-output-workflow.md +++ b/docs/design_docs/architecture/input-output-workflow.md @@ -34,7 +34,9 @@ fields and [frontend dependencies](#frontend-specific-dependencies). DAG assembly](#selection-and-dag-assembly), and [summary-maintenance lifecycle](#summary-maintenance-lifecycle-aware-helper) APIs operate on this `PlanSpace`. These are alternative uses of the candidate space, not mandatory sequential -stages. `PlanSpace` itself has no selected summary-maintenance lifecycle. +stages. `PlanSpace` itself has no selected summary-maintenance lifecycle, and +its candidates do not choose precompute versus query-time placement: a chosen +lifecycle assignment sets each node's execution timing. The candidate DAGs are logical planning artifacts. ASAPPlanner does **not** produce a deployed executable plan; downstream systems bind physical operators, diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 6bddfa6b..b53ee835 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -89,10 +89,12 @@ separate unsupported compilation, deployment infeasibility, missing evidence, and a feasible candidate that loses on cost. Absence is not a cost comparison. For `sum by(job)(rate(m[1m]))`, Rate remains per series before grouped Sum. -When lifecycle requirements permit it, a candidate may finalize Rate and Sum -within a bounded precompute run and persist the grouped value. Another may leave -those operators in the query DAG. Storing a value requires its exact evaluation -window, revision, readiness and serving cadence to match the query contract. +`PlanSpace` offers one such candidate, with a per-series Rate state and a grouped +Sum state. Its lifecycle assignment places it: a retained Sum state finalizes +Rate and builds Sum within a bounded precompute run; an `Ephemeral` Sum over a +retained Rate state leaves the Rate readout and Sum in the query DAG. Storing a +value requires its exact evaluation window, revision, readiness and serving +cadence to match the query contract. For instant-vector TopK, CMS/CountSketch with a candidate heap requires explicit series identity and a supported latest-value input protocol. Appending historical @@ -385,23 +387,25 @@ Materialization frontiers are Planner decisions. A candidate records both the precompute Physical DAG and the query Physical DAG, with typed outputs connecting them. The deployment compiler binds those outputs; it does not move operators. Lifecycle timing gives the frontier: ingestion-time nodes read by query-time -nodes. Moving further bounded consumers into precompute, as in Candidate B -below, is not yet expressed as a lifecycle choice. - -For `sum by(job)(rate(m[1m]))`, legal physical candidates can include: +nodes. For `sum by(job)(rate(m[1m]))`, the two lifecycle choices of the single +logical candidate give: ```text -Candidate A: - precompute: compatible per-series counter states → per-series Rate - materialized output: per-series rate values for window/evaluation/revision - query: stored per-series rate values → grouped Sum - -Candidate B: - precompute: compatible per-series counter states → per-series Rate → grouped Sum - materialized output: grouped values for window/evaluation/revision - query: stored grouped values → result +Candidate A (Rate state retained, Sum Ephemeral): + precompute: counter samples → per-series Rate state + materialized output: per-series Rate states for window/evaluation/revision + query: stored Rate states → Rate readout → grouped Sum → result + +Candidate B (Rate and Sum states retained): + precompute: counter samples → per-series Rate → grouped Sum state + materialized output: grouped Sum states for window/evaluation/revision + query: stored grouped Sum states → Sum readout → result ``` +Explicit frontiers passed to `compile_candidates` can also persist per-series +rate values; deriving that frontier from timing inside the physical planner is +not yet implemented. + Both preserve reset-aware Rate before Sum. Summing raw counters before Rate is not equivalent. The counter-state build may be another precompute DAG; typed state inputs do not imply that a deployment can construct or bind those states. From 0ed79d983a891aa2379d6a2e4e450f426d3f7658 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 12:19:24 +0000 Subject: [PATCH 27/59] feat: plan maintained populations through summary lifecycles execution_timed_dag refused any plan with a MaintainPopulation node because lifecycle enumeration covered SummaryAgg states only, so population timing stayed fixed by the realization strategy. Enumeration now also emits one deployment per unique maintained population that does not feed a SummaryAgg, with the usual alternatives costed through the caller's lifecycle hooks (unknown stays unknown). select and execution_timed_dag treat it like summary state: retained at ingestion with its raw input, Ephemeral rebuilt from raw input at query time. A population feeding a SummaryAgg is that state's input and follows its timing. A plan whose population deployment was removed is still refused. Co-Authored-By: Claude Opus 5.5 --- .../src/summary_maintenance_lifecycle.rs | 366 +++++++++++++++--- 1 file changed, 311 insertions(+), 55 deletions(-) diff --git a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs index 43bacfcb..18f48876 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs @@ -11,7 +11,8 @@ //! //! This module enumerates and costs `Ephemeral`, `Prepared`, `Shared`, and //! `ContinuouslyMaintained` alternatives for every unique `SummaryAgg` in a -//! materialized plan. [`SummaryMaintenanceMode`] is an orthogonal detail of +//! materialized plan, and for every maintained population (`MaintainPopulation`) +//! that is not an input of a `SummaryAgg`. [`SummaryMaintenanceMode`] is an orthogonal detail of //! the selected deployment: state is either built directly or updated //! incrementally. Unknown evidence stays unknown and therefore cannot make a //! long-lived alternative win. @@ -20,11 +21,11 @@ use std::collections::{HashMap, HashSet}; use std::rc::Rc; use asap_types::post_asap::{ - compile_post_asap_dag, compile_post_asap_dag_with_node_ids, EvaluationSchedule, - ExecutionDataStateError, ExecutionTiming, OutputRepresentation, PostAsapDag, - PostAsapDagValidationError, PostAsapNodeId, PostAsapOperatorPayload, ResultGuarantee, - SummaryExpr, SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, - SummaryMaintenanceMode, SummaryNode, SummaryWindowFramework, ValueOperation, + compile_post_asap_dag_with_node_ids, EvaluationSchedule, ExecutionDataStateError, + ExecutionTiming, OutputRepresentation, PostAsapDag, PostAsapDagValidationError, PostAsapNodeId, + ResultGuarantee, SummaryExpr, SummaryMaintenanceLifecycle, + SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, SummaryNode, + SummaryWindowFramework, ValueOperation, }; use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; @@ -159,14 +160,16 @@ impl SummaryMaintenanceLifecycleAlternative { } } -/// One unique summary-state deployment. Shared `Rc` nodes are emitted once. +/// One unique retained-state deployment. Shared `Rc` nodes are emitted once. #[derive(Debug, Clone)] pub struct SummaryMaintenanceDeployment { /// Identity of this summary in the exported post-ASAP semantic DAG. /// It is scoped to one plan version and is not a summary definition or /// summary instance identity. pub post_asap_node_id: PostAsapNodeId, - /// The unique materialized `SummaryAgg` represented by this deployment. + /// The unique materialized `SummaryAgg`, or maintained population + /// (`MaintainPopulation`) not consumed by a `SummaryAgg`, represented by + /// this deployment. Cost-model lifecycle hooks receive this node. pub summary: Rc, /// Lifecycle, evaluation, and representation commitment selected for this /// state, or `None` when no alternative is selectable. @@ -184,8 +187,9 @@ pub struct SummaryMaintenanceDeployment { pub struct SummaryMaintenanceLifecyclePlan { /// Root of the materialized post-ASAP DAG being deployed. pub root: Rc, - /// One entry per unique reachable `SummaryAgg`; shared `Rc` nodes appear - /// only once. + /// One entry per unique reachable `SummaryAgg`, then per unique + /// maintained population outside any `SummaryAgg`'s inputs; shared `Rc` + /// nodes appear only once. pub deployments: Vec, /// Caller-supplied optimization horizon used to turn rates into total /// costs. `None` keeps horizon-dependent alternatives unselectable. @@ -219,8 +223,9 @@ pub enum SummaryMaintenanceTimingError { InvalidPostAsapDag(#[from] ExecutionDataStateError), #[error("summary {0:?} has no selected lifecycle")] UnselectedLifecycle(PostAsapNodeId), - /// Lifecycle enumeration covers `SummaryAgg` states only; timing for other - /// retained state would otherwise be guessed. + /// A maintained population outside any `SummaryAgg`'s inputs has no + /// deployment, so its timing would be guessed. Enumeration always emits + /// one; this arises only for a plan whose deployments were edited. #[error("node {0:?} maintains state that has no summary-maintenance lifecycle")] UnplannedMaintainedState(PostAsapNodeId), #[error(transparent)] @@ -235,21 +240,27 @@ impl SummaryMaintenanceLifecyclePlan { /// input it consumes run at ingestion time. Every other node runs at query /// time: readouts and consumers of retained state, and each `Ephemeral` /// state not consumed by retained state together with its inputs, whose - /// raw data the deployment must supply as a query source. Timings already - /// on the root are ignored. + /// raw data the deployment must supply as a query source. This applies to + /// maintained populations as to `SummaryAgg` states; a population feeding + /// a `SummaryAgg` is one of its inputs. Timings already on the root are + /// ignored. pub fn execution_timed_dag(&self) -> Result { - let dag = compile_post_asap_dag(&self.root)?; - if let Some(node) = dag.nodes.iter().find(|node| { - matches!( - node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::MaintainPopulation { .. } - } - ) - }) { - return Err(SummaryMaintenanceTimingError::UnplannedMaintainedState( - node.id, - )); + let compiled = compile_post_asap_dag_with_node_ids(&self.root)?; + let dag = compiled.dag; + let mut populations = Vec::new(); + collect_states(&self.root, &mut HashSet::new(), &mut populations, true); + for population in &populations { + let id = compiled + .node_ids + .node_id(population) + .expect("collected population belongs to the compiled DAG"); + if !self + .deployments + .iter() + .any(|deployment| deployment.post_asap_node_id == id) + { + return Err(SummaryMaintenanceTimingError::UnplannedMaintainedState(id)); + } } let mut pending = Vec::new(); for deployment in &self.deployments { @@ -406,8 +417,10 @@ pub enum SummaryMaintenanceLifecycleChoiceError { } impl SummaryMaintenanceLifecycleCandidates<'_> { - /// One entry per unique reachable `SummaryAgg`, with every alternative - /// and its rejection; no lifecycle or window framework is selected. + /// One entry per unique retained state (see + /// [`SummaryMaintenanceLifecyclePlan::deployments`]), with every + /// alternative and its rejection; no lifecycle or window framework is + /// selected. pub fn deployments(&self) -> &[SummaryMaintenanceDeployment] { &self.plan.deployments } @@ -654,7 +667,8 @@ fn enumerate_with_profile<'a>( }; } let mut summaries = Vec::new(); - collect_summary_aggs(&root, &mut HashSet::new(), &mut summaries); + collect_states(&root, &mut HashSet::new(), &mut summaries, false); + collect_states(&root, &mut HashSet::new(), &mut summaries, true); let node_ids = compile_post_asap_dag_with_node_ids(&root)?.node_ids; let components = summary_state_components(&summaries); let deployments: Vec = summaries @@ -1249,20 +1263,33 @@ fn rejected( } } -fn collect_summary_aggs( +/// Collect unique `SummaryAgg` states, or with `populations` the unique +/// maintained populations outside every `SummaryAgg`'s inputs. A population +/// feeding a `SummaryAgg` is on that state's maintenance path, so that +/// state's lifecycle times it. +fn collect_states( node: &Rc, seen: &mut HashSet<*const SummaryNode>, output: &mut Vec>, + populations: bool, ) { if !seen.insert(Rc::as_ptr(node)) { return; } match &node.expr { + SummaryExpr::SummaryAgg { .. } if populations => {} SummaryExpr::SummaryAgg { child, .. } => { output.push(Rc::clone(node)); - collect_summary_aggs(child, seen, output); + collect_states(child, seen, output, populations); + } + SummaryExpr::ValueOperation { + child, operation, .. + } => { + if populations && matches!(operation, ValueOperation::MaintainPopulation { .. }) { + output.push(Rc::clone(node)); + } + collect_states(child, seen, output, populations) } - SummaryExpr::ValueOperation { child, .. } => collect_summary_aggs(child, seen, output), SummaryExpr::SummaryJoin { outer, inner, .. } | SummaryExpr::RelationalJoin { left: outer, @@ -1278,16 +1305,16 @@ fn collect_summary_aggs( left: outer, right: inner, } => { - collect_summary_aggs(outer, seen, output); - collect_summary_aggs(inner, seen, output); + collect_states(outer, seen, output, populations); + collect_states(inner, seen, output, populations); } SummaryExpr::SummaryDelete { summary_input, .. } | SummaryExpr::SummaryEstimate { summary_input, .. } => { - collect_summary_aggs(summary_input, seen, output) + collect_states(summary_input, seen, output, populations) } SummaryExpr::SummaryMerge { children, .. } => { for child in children { - collect_summary_aggs(child, seen, output); + collect_states(child, seen, output, populations); } } SummaryExpr::KeepPreAsap(_) => {} @@ -1316,7 +1343,7 @@ pub(crate) fn evaluation_schedule( } /// Summary states composed on one maintenance path must be produced on the -/// same schedule. Return a component id for each collected `SummaryAgg`. +/// same schedule. Return a component id for each collected state. fn summary_state_components(summaries: &[Rc]) -> Vec { let indices: HashMap<_, _> = summaries .iter() @@ -1336,6 +1363,21 @@ fn summary_state_components(summaries: &[Rc]) -> Vec { let SummaryExpr::SummaryAgg { child, .. } = &summary.expr else { continue; }; + // A population that is also read directly is one deployment; the + // summary state built from it shares its schedule. + if let ( + SummaryExpr::ValueOperation { + operation: ValueOperation::MaintainPopulation { .. }, + .. + }, + Some(&population), + ) = (&child.expr, indices.get(&Rc::as_ptr(child))) + { + let parent_root = find(&mut parents, parent_index); + let child_root = find(&mut parents, population); + parents[child_root] = parent_root; + continue; + } if !matches!( child.expr, SummaryExpr::SummaryAgg { .. } @@ -1347,7 +1389,7 @@ fn summary_state_components(summaries: &[Rc]) -> Vec { continue; } let mut descendants = Vec::new(); - collect_summary_aggs(child, &mut HashSet::new(), &mut descendants); + collect_states(child, &mut HashSet::new(), &mut descendants, false); for descendant in descendants { let child_index = indices[&Rc::as_ptr(&descendant)]; let parent_root = find(&mut parents, parent_index); @@ -1597,8 +1639,8 @@ mod tests { } use super::*; use asap_types::post_asap::{ - ExactKind, ExactParams, GroupingStrategy, ResultGuarantee, SketchAlgorithm, - SummaryFamilyType, SummaryField, SummarySchema, + ExactKind, ExactParams, GroupingStrategy, PostAsapOperatorPayload, ResultGuarantee, + SketchAlgorithm, SummaryFamilyType, SummaryField, SummarySchema, }; use asap_types::pre_asap::AggIntent; use asap_types::pre_asap::{Column, ColumnRef, DataType, QueryExpr, Reduction, Schema, Source}; @@ -3185,33 +3227,247 @@ mod tests { ); } - // A maintained population is retained state the lifecycle plan does not - // enumerate, so its timing is refused rather than guessed. - #[test] - fn timing_refuses_state_outside_the_lifecycle_plan() { + /// A strategy-built `sum(a)` over one maintained current-series population. + fn population_readout() -> Rc { let target = Rc::new(crate::test_support::lower_promql( "sum(a)", AccuracyTarget::Exact, )); - let root = crate::maintained_population::MaintainedPopulationStrategy::new( - std::slice::from_ref(&target), - ) + crate::maintained_population::MaintainedPopulationStrategy::new(std::slice::from_ref( + &target, + )) .candidate(&target) + .unwrap() + } + + fn is_population(node: &SummaryNode) -> bool { + matches!( + node.expr, + SummaryExpr::ValueOperation { + operation: ValueOperation::MaintainPopulation { .. }, + .. + } + ) + } + + fn population_timings(dag: &PostAsapDag) -> Vec<(&'static str, ExecutionTiming)> { + dag.nodes + .iter() + .zip(timings(dag)) + .map(|(node, (kind, timing))| match node.payload { + PostAsapOperatorPayload::Value { + operation: ValueOperation::MaintainPopulation { .. }, + } => ("population", timing), + _ => (kind, timing), + }) + .collect() + } + + // A maintained population is enumerated as retained state, with costs + // from the caller's model for both the maintained and the rebuilt choice. + #[test] + fn enumeration_includes_maintained_population() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let candidates = enumerate_summary_maintenance_lifecycles( + population_readout(), + WorkloadDemand::new_with_data(&workload, &data, &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &UnitCosts, + ) .unwrap(); - let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); + let [deployment] = candidates.deployments() else { + panic!("one population state"); + }; + assert!(is_population(&deployment.summary)); + let cost = |lifecycle: SummaryMaintenanceLifecycle| { + deployment + .alternatives + .iter() + .find(|alternative| alternative.summary_maintenance_lifecycle == lifecycle) + .and_then(|alternative| alternative.total_cost) + }; + // Ephemeral: (build 10 + read 1 + retire 1) x 10 reads. Maintained over + // 10 s at 1 update/s: build 10 + updates 10 + reads 10 + retention 1 + retire 1. + assert_eq!( + cost(SummaryMaintenanceLifecycle::Ephemeral), + Some(Cost(120.0)) + ); + assert_eq!( + cost(SummaryMaintenanceLifecycle::ContinuouslyMaintained), + Some(Cost(32.0)) + ); + } + + // Without cost evidence a population's alternatives stay unknown: Planner + // selects none and timing is refused rather than guessed. + #[test] + fn population_without_cost_evidence_stays_unselected() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); let plan = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &at_rest(), &[0]), + population_readout(), + WorkloadDemand::new_with_data(&workload, &data, &[0]), 1_000, - None, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &crate::cost_model::DefaultCostModel, + ) + .unwrap(); + let [deployment] = plan.deployments.as_slice() else { + panic!("one population state"); + }; + assert!(deployment + .alternatives + .iter() + .all(|alternative| alternative.total_cost.is_none())); + assert!(deployment.summary_maintenance_lifecycle_guarantee.is_none()); + assert_eq!( + plan.execution_timed_dag().unwrap_err(), + SummaryMaintenanceTimingError::UnselectedLifecycle(deployment.post_asap_node_id) + ); + } + + // A retained population and its raw input run at ingestion time; an + // Ephemeral population is rebuilt from raw input at query time. + #[test] + fn population_lifecycle_choice_decides_its_timing() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let timed = |lifecycle: SummaryMaintenanceLifecycle| { + population_timings(&timed_dag( + population_readout(), + &workload, + &data, + Some(Horizon(10.0)), + |_| lifecycle.clone(), + )) + }; + assert_eq!( + timed(SummaryMaintenanceLifecycle::ContinuouslyMaintained), + [("raw", INGEST), ("population", INGEST), ("readout", QUERY)] + ); + assert_eq!( + timed(SummaryMaintenanceLifecycle::Ephemeral), + [("raw", QUERY), ("population", QUERY), ("readout", QUERY)] + ); + } + + // When retaining is cheaper, Planner's own selection keeps the population + // maintained at ingestion time, as realization strategies placed it before + // population timing became a lifecycle decision. + #[test] + fn planner_selection_retains_population_at_ingestion() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let plan = plan_summary_maintenance_lifecycles( + population_readout(), + WorkloadDemand::new_with_data(&workload, &data, &[0]), + 1_000, + Some(Horizon(10.0)), SummaryMaintenanceLifecycleCapabilities::ALL, &UnitCosts, ) .unwrap(); - assert!(plan.deployments.is_empty()); - assert!(matches!( + assert_ne!( + selected_summary_maintenance_lifecycle(&plan.deployments[0]), + Some(&SummaryMaintenanceLifecycle::Ephemeral) + ); + assert!(plan.deployments[0] + .summary_maintenance_lifecycle_guarantee + .is_some()); + assert_eq!( + population_timings(&plan.execution_timed_dag().unwrap()), + [("raw", INGEST), ("population", INGEST), ("readout", QUERY)] + ); + } + + // A population feeding summary state is that state's input, not a separate + // deployment: the state's lifecycle times it. + #[test] + fn population_feeding_summary_state_follows_that_state() { + let SummaryExpr::ValueOperation { + child: population, .. + } = &population_readout().expr + else { + unreachable!() + }; + let state = summary(); + let SummaryExpr::SummaryAgg { + family, + input, + reduction, + grouping, + .. + } = &state.expr + else { + unreachable!() + }; + let state = Rc::new(SummaryNode { + expr: SummaryExpr::SummaryAgg { + child: Rc::clone(population), + family: family.clone(), + input: input.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + }, + ..state.as_ref().clone() + }); + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let timed = |lifecycle: SummaryMaintenanceLifecycle| { + population_timings(&timed_dag( + readout(&state), + &workload, + &data, + Some(Horizon(10.0)), + |deployment| { + assert!(Rc::ptr_eq(&deployment.summary, &state)); + lifecycle.clone() + }, + )) + }; + assert_eq!( + timed(SummaryMaintenanceLifecycle::ContinuouslyMaintained), + [ + ("raw", INGEST), + ("population", INGEST), + ("state", INGEST), + ("readout", QUERY) + ] + ); + assert_eq!( + timed(SummaryMaintenanceLifecycle::Ephemeral), + [ + ("raw", QUERY), + ("population", QUERY), + ("state", QUERY), + ("readout", QUERY) + ] + ); + } + + // A plan whose population deployment was removed after enumeration is + // refused rather than timed by a guess. + #[test] + fn timing_refuses_population_without_deployment() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let mut plan = plan_summary_maintenance_lifecycles( + population_readout(), + WorkloadDemand::new_with_data(&workload, &data, &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &UnitCosts, + ) + .unwrap(); + let id = plan.deployments.remove(0).post_asap_node_id; + assert_eq!( plan.execution_timed_dag(), - Err(SummaryMaintenanceTimingError::UnplannedMaintainedState(_)) - )); + Err(SummaryMaintenanceTimingError::UnplannedMaintainedState(id)) + ); } } From 4e0131171f74e0de39c4faafe7968d6fa6bf3c31 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 12:19:25 +0000 Subject: [PATCH 28/59] refactor: make maintained-population timing a lifecycle decision The execution-data-state validator required MaintainPopulation at ingestion time, and MaintainedPopulationStrategy hard-coded it there. Timing is now the lifecycle's choice: the validator accepts either timing and keeps the structural contracts (the population reads its matching raw input; its readout is query-time over a population that supports it). The strategy writes query time as the initial layout, as other realization strategies do. Co-Authored-By: Claude Opus 5.5 --- .../src/maintained_population.rs | 31 ++++++++++++++++++- .../src/post_asap/execution_data_state.rs | 10 +++--- 2 files changed, 36 insertions(+), 5 deletions(-) diff --git a/crates/asap-aware-mapping/src/maintained_population.rs b/crates/asap-aware-mapping/src/maintained_population.rs index 504c8100..29cd3c54 100644 --- a/crates/asap-aware-mapping/src/maintained_population.rs +++ b/crates/asap-aware-mapping/src/maintained_population.rs @@ -258,11 +258,15 @@ impl MaintainedPopulationStrategy { schema: input_schema.clone(), guarantee: Some(ResultGuarantee::exact("source samples")), }); + // Query time is only the initial layout: whether the population is + // retained at ingestion or rebuilt per query is its lifecycle choice + // (`SummaryMaintenanceLifecyclePlan::execution_timed_dag`). The readout + // and projection above it are query-time by construction. let maintained = Rc::new(SummaryNode { expr: SummaryExpr::ValueOperation { child: scan, operation: ValueOperation::MaintainPopulation { population }, - timing: ExecutionTiming::IngestionTime, + timing: ExecutionTiming::QueryTime, }, schema: input_schema, guarantee: Some(ResultGuarantee::exact( @@ -435,6 +439,31 @@ mod tests { assert_eq!(p.grouping, ["instance"]); assert_eq!(p.matchers[0].operation, CurrentSeriesMatch::Regex); } + // Population timing is a lifecycle choice: a retained or rebuilt + // population both validate, while its readout must stay at query time. + #[test] + fn population_timing_is_not_structural() { + let root = lower("topk(5,a)"); + let candidate = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) + .candidate(&root) + .unwrap(); + let with_timings = |population: ExecutionTiming, readout: ExecutionTiming| { + let mut node = (*candidate).clone(); + let SummaryExpr::ValueOperation { child, timing, .. } = &mut node.expr else { + unreachable!() + }; + *timing = readout; + let SummaryExpr::ValueOperation { timing, .. } = &mut Rc::make_mut(child).expr else { + unreachable!() + }; + *timing = population; + compile_post_asap_dag(&Rc::new(node)) + }; + use ExecutionTiming::{IngestionTime, QueryTime}; + assert!(with_timings(IngestionTime, QueryTime).is_ok()); + assert!(with_timings(QueryTime, QueryTime).is_ok()); + assert!(with_timings(IngestionTime, IngestionTime).is_err()); + } // A readout cannot reinterpret arbitrary rows as maintained state or exceed its producer's contract. #[test] fn malformed_population_dags_fail_closed() { diff --git a/crates/types/src/post_asap/execution_data_state.rs b/crates/types/src/post_asap/execution_data_state.rs index 43f2cd05..3bb56264 100644 --- a/crates/types/src/post_asap/execution_data_state.rs +++ b/crates/types/src/post_asap/execution_data_state.rs @@ -483,14 +483,17 @@ fn visit( operation, timing, } => { + // Population timing is a lifecycle decision: a retained population + // is maintained at ingestion time, an ephemeral one is rebuilt + // from raw input per query. Its input and readout contracts are + // structural and hold either way. let valid_population = match operation { ValueOperation::MaintainPopulation { population } => { - *timing == ExecutionTiming::IngestionTime - && matches!(&child.expr, SummaryExpr::KeepPreAsap(input) if population.matches_input(input)) + matches!(&child.expr, SummaryExpr::KeepPreAsap(input) if population.matches_input(input)) } ValueOperation::ReadPopulation { readout } => { *timing == ExecutionTiming::QueryTime - && matches!(&child.expr, SummaryExpr::ValueOperation { operation: ValueOperation::MaintainPopulation { population }, timing: ExecutionTiming::IngestionTime, .. } if population.supports(readout)) + && matches!(&child.expr, SummaryExpr::ValueOperation { operation: ValueOperation::MaintainPopulation { population }, .. } if population.supports(readout)) } _ => true, }; @@ -513,7 +516,6 @@ fn visit( &child.expr, SummaryExpr::ValueOperation { operation: ValueOperation::MaintainPopulation { .. }, - timing: ExecutionTiming::IngestionTime, .. } ); From 761524ecc65b47f971889bdf2410233cb30d28d2 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 12:19:25 +0000 Subject: [PATCH 29/59] test: carry a chosen population lifecycle to physical compilation Conflicts with earlier stack changes resolved to the integration tree: - crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs: 51fd198 Merge #491 population lifecycle into Planner integration Co-Authored-By: Claude Opus 5.5 --- .../summary_maintenance_lifecycle_e2e.rs | 146 ++++++++++++++++++ 1 file changed, 146 insertions(+) diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index e1a1f57b..f9802b40 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -687,6 +687,152 @@ fn lifecycle_timing_cuts_one_compilation() { } } + +/// A maintained current-series population is placed by its lifecycle choice: +/// ContinuouslyMaintained stores the population in precompute, Ephemeral +/// rebuilds it from the raw source at query time; both rank alike. +#[test] +fn chosen_population_lifecycle_decides_precompute_contents() { + use asap_aware_mapping::{ + enumerate_summary_maintenance_lifecycles, + maintained_population::MaintainedPopulationStrategy, + }; + use asap_physical_operators::{ + physical_planner::{ + compile_candidate, + promql_rows::{series_row, with_series_identity}, + InputContract, + }, + runtime::Scope, + values::{Batch, Value}, + }; + use asap_types::post_asap::{ + maintained_population::PopulationInput, PostAsapOperatorPayload, ValueOperation, + }; + use std::{collections::BTreeMap, sync::Arc}; + + let workload = quantile_workload("topk by(job)(1, m)"); + let root = Rc::new( + with_series_identity(&lower_promql_workload(&workload, 0).unwrap().remove(0)).unwrap(), + ); + let root = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) + .candidate(&root) + .unwrap(); + let mut answers = Vec::new(); + for lifecycle in [ + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + SummaryMaintenanceLifecycle::Ephemeral, + ] { + let candidates = enumerate_summary_maintenance_lifecycles( + Rc::clone(&root), + WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + NOW_MS, + Some(Horizon(100.)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &FullyCostedRuntime, + ) + .unwrap(); + let [deployment] = candidates.deployments() else { + panic!("one population state"); + }; + let id = deployment.post_asap_node_id; + let dag = candidates + .select(&[(id, lifecycle.clone())]) + .unwrap() + .execution_timed_dag() + .unwrap(); + let population = dag.nodes.iter().find(|node| node.id == id).unwrap(); + let PostAsapOperatorPayload::Value { + operation: ValueOperation::MaintainPopulation { population }, + } = &population.payload + else { + panic!("the deployment is the maintained population"); + }; + let PopulationInput::CurrentSeries(spec) = &population.input else { + panic!("current-series population"); + }; + let lookback = i64::try_from(spec.lookback_ms).unwrap(); + let raw = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .unwrap(); + let (raw_id, schema) = (u64::from(raw.id.0), Arc::new(raw.output_schema.clone())); + let frontier = ingestion_frontier(&dag); + let candidate = compile_candidate( + &dag, + BTreeMap::from([(raw_id, InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &frontier, + ) + .unwrap(); + let end = 60_000; + let rows = [("a", end - 1, 100.), ("a", end, 1.), ("b", end, 20.)] + .into_iter() + .map(|(instance, at, value)| { + series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("instance".into(), instance.into()), + ]), + at, + value, + ) + .unwrap() + }) + .collect(); + let raw_batch = Batch::try_new(schema.clone(), rows).unwrap(); + let query_scope = Scope::Query { + evaluation_time_ms: end, + revision: 1, + }; + let result = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { + assert!(frontier.is_empty()); + assert!(candidate.precompute.is_none()); + physical_common::execute( + &candidate.query, + BTreeMap::from([(raw_id, raw_batch)]), + query_scope, + ) + } else { + let state = u64::from(id.0); + assert_eq!(frontier, [state]); + let stored = physical_common::execute( + candidate.precompute.as_ref().unwrap(), + BTreeMap::from([(raw_id, raw_batch)]), + Scope::Ingestion { + window_start_ms: end - lookback, + window_end_ms: end, + revision: 1, + }, + ); + physical_common::execute( + &candidate.query, + BTreeMap::from([(state, stored[0][0].clone())]), + query_scope, + ) + }; + answers.push( + result[0] + .iter() + .flat_map(|batch| batch.rows()) + .flat_map(|row| row.iter()) + .filter_map(|value| match value { + Value::Float64(value) => Some(*value), + _ => None, + }) + .collect::>(), + ); + } + assert_eq!(answers[0], answers[1]); + assert_eq!(answers[0], [20.]); +} + /// Grouped Rate→Sum is one inventory candidate: retaining the Sum state puts /// Rate and Sum in precompute, while an `Ephemeral` Sum over a retained Rate /// state leaves Sum in the query DAG. From 76c503ec9386da97ef14b5ba2d3acf9846268a24 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 12:19:26 +0000 Subject: [PATCH 30/59] docs: describe maintained populations as lifecycle-planned state Rebase note: keeps the #479 compile-once documentation and test row beside the population text, as the integration branch does. Co-Authored-By: Claude Opus 5.5 --- .../physical-planning-and-deployment.md | 19 ++++++++++++++++--- docs/develop_docs/library-api.md | 12 +++++++++++- 2 files changed, 27 insertions(+), 4 deletions(-) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index b53ee835..b250e946 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -42,12 +42,15 @@ below states. families, readouts and sharing. It does not decide placement; timing that a realization strategy writes while building a candidate is provisional. 2. **Summary maintenance lifecycle** (Planner) lists the lifecycle choices for - each unique summary state. A chosen assignment determines every node's + each unique retained state: every summary state (`SummaryAgg`) and every + maintained current-series population that does not feed a summary state. + A chosen assignment determines every node's `ExecutionTiming`, plus window framework and retention. `SummaryMaintenanceLifecyclePlan::execution_timed_dag` applies it: a retained (non-`Ephemeral`) state and all of its inputs run at ingestion time; readouts, other consumers, and `Ephemeral` states not consumed by retained - state run at query time. + state run at query time. A population that feeds a summary state is one of + that state's inputs and follows its timing. 3. **Physical compile** (Planner) reads timing: ingestion-time nodes form the precompute DAG and the rest form the query DAG, joined by typed outputs. It does not see raw ingestion, panes, storage or stored-state readout. @@ -259,7 +262,7 @@ using workload demand, window/freshness requirements and supported physical implementations. Backend selection uses runtime feasibility and cost after physical compilation. The following example follows one candidate. -Candidate generation and selection are separate steps. For every unique summary +Candidate generation and selection are separate steps. For every unique retained state, enumeration reports each lifecycle (ephemeral, prepared, shared, continuously maintained) as legal, with a Planner cost or explicitly unknown cost, or as rejected with a reason. Planner does not remove a legal alternative @@ -274,6 +277,14 @@ Planner selects. Planner's own cheapest-alternative selection remains available for callers without deployment pricing. The window framework is decided for the complete combination, not for one alternative in isolation. +A maintained population (for example, the current series of `topk by(job)(1, m)`) +is retained state like a summary. Retaining it maintains the latest sample per +series at ingestion and leaves only the readout at query time. Choosing +`Ephemeral` rebuilds that snapshot from raw samples for each query, so the +deployment must supply the raw source at query time. The caller's `CostModel` +prices both through the same lifecycle hooks; a model without population +evidence leaves them unknown, and they are not selected. + For the running example, assume it selects: ```text @@ -561,6 +572,8 @@ operator/runtime fixtures: | `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled precompute/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | | `summary_maintenance_lifecycle_e2e::chosen_lifecycle_timing_decides_precompute_contents` | PromQL workload → enumerated lifecycles → explicit choice → timed DAG → compiled candidate; ContinuouslyMaintained stores the state in precompute, Ephemeral leaves precompute empty and reads the raw source at query time; both return the same p99 | | `summary_maintenance_lifecycle_e2e::lifecycle_timing_cuts_one_compilation` | KLL quantile and grouped Rate→Sum: one compilation cut by the ContinuouslyMaintained and Ephemeral timed DAGs equals `compile_candidate` for each; the frontier is the retained state or empty | + +| `summary_maintenance_lifecycle_e2e::chosen_population_lifecycle_decides_precompute_contents` | PromQL `topk by(job)` over a maintained population → explicit choice → timed DAG → compiled candidate; ContinuouslyMaintained stores the population in precompute, Ephemeral rebuilds it from raw samples at query time; both rank alike | | `summary_maintenance_lifecycle_e2e::planner_lifecycle_selection_reproduces_strategy_timing` | For PromQL fixtures, the timed DAG from Planner's retained selection equals the DAG realization strategies produce today | | `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute precompute DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | | `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index 4c77535c..17687e5b 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -633,7 +633,7 @@ that prepared or retained shared state is supported. | `plan_summary_maintenance_lifecycles` | Assembled logical DAG root, `WorkloadDemand`, `now_ms`, optional horizon, runtime capabilities, cost model | `Result` for that fixed root; does not revisit all semantic candidates | | `global_selection_with_summary_maintenance_lifecycles` | `PlanSpace`, workload/root-entry associations, time, horizon, capabilities, cost model | Lifecycle-aware compatible selection/error, using eligible cost evidence | | `assemble_selected_dag_with_summary_maintenance_lifecycles` | Selection, target root and lifecycle context | Optional lifecycle plan/error; attaches state deployment decisions | -| `enumerate_summary_maintenance_lifecycles` | Same inputs as `plan_summary_maintenance_lifecycles` | `SummaryMaintenanceLifecycleCandidates`: per unique state, every alternative with its cost or rejection; nothing selected. `guarantee(&lifecycle)` gives the mode/schedule that alternative would carry | +| `enumerate_summary_maintenance_lifecycles` | Same inputs as `plan_summary_maintenance_lifecycles` | `SummaryMaintenanceLifecycleCandidates`: per unique retained state, every alternative with its cost or rejection; nothing selected. `guarantee(&lifecycle)` gives the mode/schedule that alternative would carry | | `SummaryMaintenanceLifecycleCandidates::select(choices)` | One `(PostAsapNodeId, SummaryMaintenanceLifecycle)` per state, copied from `deployments()` | The same `SummaryMaintenanceLifecyclePlan` Planner selection would produce for that combination, or `SummaryMaintenanceLifecycleChoiceError` when a choice is unknown, missing, duplicated, rejected, schedule-incompatible, or not completely estimable | Inspect `deployments`, their selected lifecycle/alternatives/rejections, @@ -681,6 +681,16 @@ the same timing for that node, because it lowers differently. Temporal pane candidates are a different lowering and still use `compile_temporal_pane_candidate`. +Retained states are `SummaryAgg` nodes and `MaintainPopulation` nodes that do +not feed a `SummaryAgg`; a population that does feed one is part of that +state's input. The lifecycle cost hooks (`summary_maintenance_capabilities`, +`summary_maintenance_lifecycle_cost_inputs_for_horizon`) and the complete-candidate +hook therefore also receive `MaintainPopulation` nodes. A model that does not +recognize one should return unknown costs, which keep its alternatives +unselected. `SummaryMaintenanceLifecyclePlan::execution_timed_dag` times a +population as it times a summary state: retained at ingestion, `Ephemeral` at +query time from the raw source. + ## Optional whole-plan selection and DAG assembly ### What does global selection mean? From 066bb72c1194f8445c756a3f2a1652d09ba9d0e0 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 12:27:40 +0000 Subject: [PATCH 31/59] fix: time a shared population by its summary consumer A population read directly and also consumed by a SummaryAgg became its own deployment or not depending on traversal order, and its own lifecycle could disagree with the retained state built from it. Any population reachable from a SummaryAgg is now that state's input and never a separate deployment, so the state's lifecycle times it in either order. Also clarify review-noted wording: retained-state docs, the uniform-pricing consequence for cost models, and stale population-timing notes in the design proposals. Co-Authored-By: Claude Opus 5.5 --- .../src/summary_maintenance_dag_export.rs | 2 +- .../src/summary_maintenance_lifecycle.rs | 199 ++++++++++++++---- .../src/post_asap/execution_data_state.rs | 1 + .../physical-planning-and-deployment.md | 4 +- .../maintained-populations.md | 7 +- .../design_docs/proposals/operator-sharing.md | 2 +- docs/develop_docs/library-api.md | 3 +- 7 files changed, 167 insertions(+), 51 deletions(-) diff --git a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs index efc8cf13..129d06ed 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs @@ -117,7 +117,7 @@ pub fn export_summary_maintenance_plan( } /// Walk in the same post-order as `dag_export::export_summary` and attach a -/// deployment directly to every flattened occurrence of its `SummaryAgg`. +/// deployment directly to every flattened occurrence of its state node. /// This makes the decision visible to graph consumers without asking them to /// reconstruct pointer identity from graph position. fn annotate_lifecycle_deployments( diff --git a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs index 18f48876..cb1dd9db 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs @@ -225,7 +225,7 @@ pub enum SummaryMaintenanceTimingError { UnselectedLifecycle(PostAsapNodeId), /// A maintained population outside any `SummaryAgg`'s inputs has no /// deployment, so its timing would be guessed. Enumeration always emits - /// one; this arises only for a plan whose deployments were edited. + /// one; this arises only for a plan whose root or deployments were edited. #[error("node {0:?} maintains state that has no summary-maintenance lifecycle")] UnplannedMaintainedState(PostAsapNodeId), #[error(transparent)] @@ -247,9 +247,7 @@ impl SummaryMaintenanceLifecyclePlan { pub fn execution_timed_dag(&self) -> Result { let compiled = compile_post_asap_dag_with_node_ids(&self.root)?; let dag = compiled.dag; - let mut populations = Vec::new(); - collect_states(&self.root, &mut HashSet::new(), &mut populations, true); - for population in &populations { + for population in &standalone_populations(&self.root) { let id = compiled .node_ids .node_id(population) @@ -377,7 +375,7 @@ pub enum SummaryMaintenanceLifecycleSelectionError { SummaryMaintenance(#[from] SummaryMaintenanceLifecyclePlanError), } -/// Every lifecycle alternative for each unique summary state of one fixed +/// Every lifecycle alternative for each unique retained state of one fixed /// root, before any lifecycle is chosen. /// /// Planner selection ([`plan_summary_maintenance_lifecycles`]) and a @@ -667,8 +665,13 @@ fn enumerate_with_profile<'a>( }; } let mut summaries = Vec::new(); - collect_states(&root, &mut HashSet::new(), &mut summaries, false); - collect_states(&root, &mut HashSet::new(), &mut summaries, true); + collect_states( + &root, + &mut HashSet::new(), + &mut summaries, + StateKind::SummaryAgg, + ); + summaries.extend(standalone_populations(&root)); let node_ids = compile_post_asap_dag_with_node_ids(&root)?.node_ids; let components = summary_state_components(&summaries); let deployments: Vec = summaries @@ -1263,32 +1266,38 @@ fn rejected( } } -/// Collect unique `SummaryAgg` states, or with `populations` the unique -/// maintained populations outside every `SummaryAgg`'s inputs. A population -/// feeding a `SummaryAgg` is on that state's maintenance path, so that -/// state's lifecycle times it. +#[derive(Clone, Copy, PartialEq)] +enum StateKind { + SummaryAgg, + Population, +} + +/// Collect every unique node of `kind` reachable from `node`. fn collect_states( node: &Rc, seen: &mut HashSet<*const SummaryNode>, output: &mut Vec>, - populations: bool, + kind: StateKind, ) { if !seen.insert(Rc::as_ptr(node)) { return; } match &node.expr { - SummaryExpr::SummaryAgg { .. } if populations => {} SummaryExpr::SummaryAgg { child, .. } => { - output.push(Rc::clone(node)); - collect_states(child, seen, output, populations); + if kind == StateKind::SummaryAgg { + output.push(Rc::clone(node)); + } + collect_states(child, seen, output, kind); } SummaryExpr::ValueOperation { child, operation, .. } => { - if populations && matches!(operation, ValueOperation::MaintainPopulation { .. }) { + if kind == StateKind::Population + && matches!(operation, ValueOperation::MaintainPopulation { .. }) + { output.push(Rc::clone(node)); } - collect_states(child, seen, output, populations) + collect_states(child, seen, output, kind) } SummaryExpr::SummaryJoin { outer, inner, .. } | SummaryExpr::RelationalJoin { @@ -1305,22 +1314,50 @@ fn collect_states( left: outer, right: inner, } => { - collect_states(outer, seen, output, populations); - collect_states(inner, seen, output, populations); + collect_states(outer, seen, output, kind); + collect_states(inner, seen, output, kind); } SummaryExpr::SummaryDelete { summary_input, .. } | SummaryExpr::SummaryEstimate { summary_input, .. } => { - collect_states(summary_input, seen, output, populations) + collect_states(summary_input, seen, output, kind) } SummaryExpr::SummaryMerge { children, .. } => { for child in children { - collect_states(child, seen, output, populations); + collect_states(child, seen, output, kind); } } SummaryExpr::KeepPreAsap(_) => {} } } +/// Maintained populations that are not an input of any `SummaryAgg`. A +/// population feeding summary state is on that state's maintenance path, so +/// that state's lifecycle times it, even when a readout also reads it directly. +fn standalone_populations(root: &Rc) -> Vec> { + let mut summaries = Vec::new(); + collect_states( + root, + &mut HashSet::new(), + &mut summaries, + StateKind::SummaryAgg, + ); + let mut nested = Vec::new(); + let mut seen = HashSet::new(); + for summary in &summaries { + collect_states(summary, &mut seen, &mut nested, StateKind::Population); + } + let nested: HashSet<_> = nested.iter().map(Rc::as_ptr).collect(); + let mut populations = Vec::new(); + collect_states( + root, + &mut HashSet::new(), + &mut populations, + StateKind::Population, + ); + populations.retain(|population| !nested.contains(&Rc::as_ptr(population))); + populations +} + pub(crate) fn evaluation_schedule( lifecycle: &SummaryMaintenanceLifecycle, arrival: DataArrival, @@ -1363,21 +1400,6 @@ fn summary_state_components(summaries: &[Rc]) -> Vec { let SummaryExpr::SummaryAgg { child, .. } = &summary.expr else { continue; }; - // A population that is also read directly is one deployment; the - // summary state built from it shares its schedule. - if let ( - SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { .. }, - .. - }, - Some(&population), - ) = (&child.expr, indices.get(&Rc::as_ptr(child))) - { - let parent_root = find(&mut parents, parent_index); - let child_root = find(&mut parents, population); - parents[child_root] = parent_root; - continue; - } if !matches!( child.expr, SummaryExpr::SummaryAgg { .. } @@ -1389,7 +1411,12 @@ fn summary_state_components(summaries: &[Rc]) -> Vec { continue; } let mut descendants = Vec::new(); - collect_states(child, &mut HashSet::new(), &mut descendants, false); + collect_states( + child, + &mut HashSet::new(), + &mut descendants, + StateKind::SummaryAgg, + ); for descendant in descendants { let child_index = indices[&Rc::as_ptr(&descendant)]; let parent_root = find(&mut parents, parent_index); @@ -3371,13 +3398,11 @@ mod tests { &UnitCosts, ) .unwrap(); - assert_ne!( + // Shared and ContinuouslyMaintained tie at 32; the first wins. + assert!(matches!( selected_summary_maintenance_lifecycle(&plan.deployments[0]), - Some(&SummaryMaintenanceLifecycle::Ephemeral) - ); - assert!(plan.deployments[0] - .summary_maintenance_lifecycle_guarantee - .is_some()); + Some(SummaryMaintenanceLifecycle::Shared { .. }) + )); assert_eq!( population_timings(&plan.execution_timed_dag().unwrap()), [("raw", INGEST), ("population", INGEST), ("readout", QUERY)] @@ -3449,6 +3474,94 @@ mod tests { ); } + // A population both read directly and consumed by summary state is that + // state's input in either traversal order: not a separate deployment, and + // timed by the state's lifecycle. + #[test] + fn shared_population_follows_its_summary_consumer() { + let direct = population_readout(); + let SummaryExpr::ValueOperation { + child: population, .. + } = &direct.expr + else { + unreachable!() + }; + let state = summary(); + let SummaryExpr::SummaryAgg { + family, + input, + reduction, + grouping, + .. + } = &state.expr + else { + unreachable!() + }; + let state = Rc::new(SummaryNode { + expr: SummaryExpr::SummaryAgg { + child: Rc::clone(population), + family: family.clone(), + input: input.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + }, + ..state.as_ref().clone() + }); + let binary = |lhs: Rc, rhs: Rc| { + Rc::new(SummaryNode { + schema: lhs.schema.clone(), + expr: SummaryExpr::BinaryOp { + lhs, + rhs, + operator: asap_types::post_asap::BinaryOperator { + kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( + asap_types::pre_asap::ArithmeticOpKind::Add, + ), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + timing: QUERY, + }, + guarantee: None, + }) + }; + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + for root in [ + binary(Rc::clone(&direct), readout(&state)), + binary(readout(&state), Rc::clone(&direct)), + ] { + for lifecycle in [ + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + SummaryMaintenanceLifecycle::Ephemeral, + ] { + let dag = timed_dag( + Rc::clone(&root), + &workload, + &data, + Some(Horizon(10.0)), + |deployment| { + assert!(Rc::ptr_eq(&deployment.summary, &state)); + lifecycle.clone() + }, + ); + let expected = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { + QUERY + } else { + INGEST + }; + for (kind, timing) in population_timings(&dag) { + if matches!(kind, "raw" | "population" | "state") { + assert_eq!(timing, expected, "{kind}"); + } else { + assert_eq!(timing, QUERY, "{kind}"); + } + } + } + } + } + // A plan whose population deployment was removed after enumeration is // refused rather than timed by a guess. #[test] diff --git a/crates/types/src/post_asap/execution_data_state.rs b/crates/types/src/post_asap/execution_data_state.rs index 3bb56264..a5d87c44 100644 --- a/crates/types/src/post_asap/execution_data_state.rs +++ b/crates/types/src/post_asap/execution_data_state.rs @@ -510,6 +510,7 @@ fn visit( && s.primitive == DataPrimitive::SummaryState && (*timing == ExecutionTiming::QueryTime || s.timing == *timing) && is_exact_accumulator_state(&child.schema).is_ok(); + // A query-time readout may read a population retained at ingestion. let population_readout = matches!(operation, ValueOperation::ReadPopulation { .. }) && *timing == ExecutionTiming::QueryTime && matches!( diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index b250e946..81afc40b 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -43,7 +43,7 @@ below states. realization strategy writes while building a candidate is provisional. 2. **Summary maintenance lifecycle** (Planner) lists the lifecycle choices for each unique retained state: every summary state (`SummaryAgg`) and every - maintained current-series population that does not feed a summary state. + maintained population that does not feed a summary state. A chosen assignment determines every node's `ExecutionTiming`, plus window framework and retention. `SummaryMaintenanceLifecyclePlan::execution_timed_dag` applies it: a retained @@ -574,7 +574,7 @@ operator/runtime fixtures: | `summary_maintenance_lifecycle_e2e::lifecycle_timing_cuts_one_compilation` | KLL quantile and grouped Rate→Sum: one compilation cut by the ContinuouslyMaintained and Ephemeral timed DAGs equals `compile_candidate` for each; the frontier is the retained state or empty | | `summary_maintenance_lifecycle_e2e::chosen_population_lifecycle_decides_precompute_contents` | PromQL `topk by(job)` over a maintained population → explicit choice → timed DAG → compiled candidate; ContinuouslyMaintained stores the population in precompute, Ephemeral rebuilds it from raw samples at query time; both rank alike | -| `summary_maintenance_lifecycle_e2e::planner_lifecycle_selection_reproduces_strategy_timing` | For PromQL fixtures, the timed DAG from Planner's retained selection equals the DAG realization strategies produce today | +| `summary_maintenance_lifecycle_e2e::planner_lifecycle_selection_reproduces_strategy_timing` | For PromQL summary fixtures, the timed DAG from Planner's retained selection equals the DAG realization strategies produce | | `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute precompute DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | | `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | | `precompute_candidates::grouped_rate_can_be_materialized_before_or_after_grouped_sum` | Cost changes select different legal precompute frontiers; both selected candidates execute with the same reset-sensitive result; uncompilable candidates are not priced | diff --git a/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md b/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md index 99501e97..06555965 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md +++ b/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md @@ -70,7 +70,7 @@ columns and multi-measure aggregates need additional rules. ```text KeepPreAsap(input) - -> MaintainPopulation { input, max_k, quantiles } [maintenance] + -> MaintainPopulation { input, max_k, quantiles } [lifecycle-timed] -> ReadPopulation { Quantile(q1) } [read] -> ReadPopulation { Quantile(q2) } [read] -> ReadPopulation { TopK(k1) } [read] @@ -115,8 +115,9 @@ because their source names or numeric values happen to agree. ## Validation, selection and execution responsibilities -Planner validates the declared input, maintenance/read phases and readout -compatibility. Its intended guarantee is exact membership and exact readout; +Planner validates the declared input, the query-time readout and readout +compatibility; the population's lifecycle decides whether it is maintained at +ingestion or rebuilt per query. Its intended guarantee is exact membership and exact readout; a physical implementation still must preserve the language's numeric and empty-input semantics. In particular, SQL global COUNT over an empty population returns a row with zero, while PromQL COUNT over an empty vector returns an empty vector. diff --git a/docs/design_docs/proposals/operator-sharing.md b/docs/design_docs/proposals/operator-sharing.md index 03b36cf5..3d075105 100644 --- a/docs/design_docs/proposals/operator-sharing.md +++ b/docs/design_docs/proposals/operator-sharing.md @@ -287,7 +287,7 @@ Per node kind: | `SummaryAgg` | from the child; ingestion time under `KeepPreAsap` | **set** by binding, as today's fallback: `IngestionTime`, or `QueryTime` over a query-time child | | `FinalizeExactAccumulator` | a stored field, set by the planner | **set** by the planner: the same position allows either time | | `SummaryEstimate` | query time, fixed by the kind | **derived** from the kind: query time | -| `MaintainPopulation` / `ReadPopulation` | a stored field, always ingestion / query time | **derived** from the kind: ingestion / query time | +| `MaintainPopulation` / `ReadPopulation` | a stored field: population timing set by its lifecycle; readout always query time | population: **set** by its lifecycle; readout: **derived**, query time | | unused variants | `SummaryMerge`: a stored field; `Join` / `Subtract` / `Delete`: ingestion time | unimplemented (§1.3) | Unlike a guarantee, a timing depends on the parents, so `derive_timings` needs the whole diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index 17687e5b..a4119122 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -687,7 +687,8 @@ state's input. The lifecycle cost hooks (`summary_maintenance_capabilities`, `summary_maintenance_lifecycle_cost_inputs_for_horizon`) and the complete-candidate hook therefore also receive `MaintainPopulation` nodes. A model that does not recognize one should return unknown costs, which keep its alternatives -unselected. `SummaryMaintenanceLifecyclePlan::execution_timed_dag` times a +unselected; a model that prices every node uniformly now also prices +populations, so population candidates can win lifecycle-aware selection. `SummaryMaintenanceLifecyclePlan::execution_timed_dag` times a population as it times a summary state: retained at ingestion, `Ephemeral` at query time from the raw source. From d3372f54e9b80bb6fb751b391bdb3807d6712b90 Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:08:53 +0000 Subject: [PATCH 32/59] integrate: read the population test frontier from lifecycle timing #479 replaced the test-local `ingestion_frontier` helper with `physical_planner::frontier_from_timing`; the population lifecycle test from this PR still called the removed helper. Taken from integration commit 0d132b9. Co-Authored-By: Claude Opus 5.5 --- .../tests/summary_maintenance_lifecycle_e2e.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index f9802b40..08e3e3b1 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -687,7 +687,6 @@ fn lifecycle_timing_cuts_one_compilation() { } } - /// A maintained current-series population is placed by its lifecycle choice: /// ContinuouslyMaintained stores the population in precompute, Ephemeral /// rebuilds it from the raw source at query time; both rank alike. @@ -762,7 +761,8 @@ fn chosen_population_lifecycle_decides_precompute_contents() { .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) .unwrap(); let (raw_id, schema) = (u64::from(raw.id.0), Arc::new(raw.output_schema.clone())); - let frontier = ingestion_frontier(&dag); + let frontier = + asap_physical_operators::physical_planner::frontier_from_timing(&dag).unwrap(); let candidate = compile_candidate( &dag, BTreeMap::from([(raw_id, InputContract::bounded(schema.clone()))]), From 7cc2d17cd6b6a8cc85b508431e4ca1aa480f3173 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:24:21 +0000 Subject: [PATCH 33/59] docs: inventory backend computation against physical compile coverage Co-Authored-By: Claude Opus 5.5 --- docs/develop_docs/README.md | 1 + .../develop_docs/physical-compile-coverage.md | 69 +++++++++++++++++++ 2 files changed, 70 insertions(+) create mode 100644 docs/develop_docs/physical-compile-coverage.md diff --git a/docs/develop_docs/README.md b/docs/develop_docs/README.md index 2593f3f5..817722ad 100644 --- a/docs/develop_docs/README.md +++ b/docs/develop_docs/README.md @@ -15,5 +15,6 @@ formats, evidence, and verification workflows. - [Metrics-observability corpora](metrics-observability-corpora.md) - [Physical handoff cost references](physical-handoff-costs.md), [storage operations](storage-operation-costs.md) - [Replacement explanations](replacement-explanations.md) +- [Physical compile coverage for deployment computation](physical-compile-coverage.md) - [Planner vocabulary migration (#427)](planner-vocabulary-migration.md) diff --git a/docs/develop_docs/physical-compile-coverage.md b/docs/develop_docs/physical-compile-coverage.md new file mode 100644 index 00000000..4dc309ae --- /dev/null +++ b/docs/develop_docs/physical-compile-coverage.md @@ -0,0 +1,69 @@ +# Physical compile coverage for deployment computation + +Audience: developers moving computation from ASAPQuery-backend into +`asap_physical_operators::physical_planner`. + +## Contract + +Logical selection decides what to compute. The maintenance lifecycle sets node +timing. `physical_planner::compile` turns a timed `PostAsapDag` into physical +operator DAGs. The backend owns ingestion, panes, storage, stored-state +readout, external exact engines, pricing/selection, and execution scheduling. + +A backend lowering is *covered* when `compile` accepts the corresponding +`PostAsapDag` node and produces operators with the same result. The backend +should then pass the timed DAG and its input contracts to `compile`. It should +not rebuild operator choices from PromQL text or construct operators itself. + +## Inventory + +Surveyed backend: `ASAPQuery-backend` branch `perf/788-startup-search`. +Planner base: `split/462-f-physical-planner` (#475). + +Status values: + +- **Supported**: `compile` or `compile_node` already covers this computation. +- **Partial**: some shapes are covered. The Notes column lists the gap. +- **Missing**: `compile` rejects this computation. +- **Backend**: not computation, or owned by the backend. + +| # | Backend site | Computation | Planner node | Status at #475 | Notes | +|---|---|---|---|---|---| +| 1 | `query_time.rs` `Lower::lower`, `compile_logical` | PromQL AST → `QueryTimeOperator` graph for a native query | `Fallback { QueryExpr }` subtrees plus value payloads | Missing | `compile` lowers `Fallback` only as a raw `Scan` source. | +| 2 | `QueryTimeOperator::Aggregate` (sum/min/max/avg/count) | Grouped value aggregation | `Value::Exact(Aggregate)`; `SummaryAgg{ExactAggregate, Reduce}` over finalized values | Supported | Also `promql_values::compile_aggregate`. | +| 3 | `QueryTimeOperator::Sort`, `Limit` (topk, sort, sort_desc) | Ordering and per-group limits | `Value::Sort`, `Value::Limit` | Supported | | +| 4 | `QueryTimeOperator::Binary`, `QueryPlanNode::Binary` (vector ⊗ scalar) | Arithmetic with a scalar operand | `Binary` whose operand is `Fallback{PromqlScalarBridge(Literal)}` | Missing | Query-time `Binary` accepts only label-map vector schemas. The literal node has no native binding. | +| 5 | `QueryTimeOperator::Binary` (vector ⊗ vector) | One-to-one label matching and arithmetic | `Binary` over grouped value rows | Missing | Only the ingestion-time `aligned_binary` and label-map `vector_binary` exist. | +| 6 | `binary_operator` CheckedDiv / FiniteDiv | Guarded division | `BinaryOperator` checked flags | Partial | Flags are evaluated, but only where rows 4/5 are covered. | +| 7 | `QueryTimeOperator::Binary` comparisons, `bool` | Filter or 0/1 comparison | `Binary{Compare}` | Missing | `Payload::Binary` does not carry `return_bool`. | +| 8 | `QueryTimeOperator::UnaryNegate` | Negation | `Binary{Mul}` by literal `-1` | Missing | The frontend emits `* -1`; same gap as row 4. | +| 9 | `QueryTimeOperator::VectorToScalar` | `scalar()` | `Fallback{PromqlScalarFromVector}` | Missing | Only `promql_values::compile_vector_to_scalar`. | +| 10 | `QueryTimeOperator::HistogramQuantile` | Bucket interpolation | `Fallback` / `AggIntent::HistogramQuantile` | Missing | Only `promql_values::compile_histogram_quantile`. | +| 11 | `QueryTimeOperator::Temporal` (rate, increase, `*_over_time`) | Per-series window functions | `SummaryAgg{PerEntity}` over `TimeRange(Scan)` | Partial | Supported with closed series identity. Not supported over `Fallback` matrices (`compile_temporal` only). | +| 12 | `logical_dag.rs` `Subquery`, `subquery_grid`, `expanded_inputs` | Re-evaluate the child on a step grid and assemble a matrix | `Fallback{PromqlSubquery}` | Missing | No Planner operator. | +| 13 | `QueryPlanNode::Scalar`, `DagCompiler::lower` scalar literal | Scalar constant | `Fallback{PromqlScalarBridge(Literal)}` | Missing | Only `promql_values::compile_scalar`. | +| 14 | `DagCompiler::lower` `ReduceSum`; `physical_values.rs` PerEntity projection | Sum over finalized values; per-entity identity | `SummaryAgg{ExactAggregate(Sum)}` | Supported | The backend builds an identity `Operator::project` itself for PerEntity. | +| 15 | `DagCompiler::lower` `ExactReadout`; `post_asap_readout.rs` ExactReadout | Finalize exact state (sum/count/min/max/rate/increase) | `Value::FinalizeExactAccumulator` | Partial | Count yields Int64 against a declared Float64 PromQL value. `compile` rejects it. | +| 16 | `post_asap_readout.rs` SummaryEstimate (`readout_bound`, `expand_item_rows`) | Sketch estimate per group; TopK item expansion | `SummaryEstimate` | Partial | The backend's label-map state layout and MetricsQL `__name__` rules have no Planner equivalent. `compile_exact_readout` has no sketch counterpart. | +| 17 | `post_asap_readout.rs` SummaryMerge (`merge_bound_states`) | Merge states by group | `SummaryMerge` | Supported | Union plus `summary_merge`. | +| 18 | `post_asap_readout.rs` counter range parameters | Counter lookback for rate/increase | `TimeRange` ancestor of finalization | Supported | Applied through `with_counter_lookback`. | +| 19 | `post_asap_readout.rs` `execute_value_fragment` | Per-timestamp binding of a value fragment | n/a | Backend | Evaluation scheduling. | +| 20 | `DagCompiler::lower` SummaryJoin / Subtract / Delete | Summary algebra | `SummaryJoin`, `SummarySubtract`, `SummaryDelete` | Missing | The backend also rejects these (`ExactFallback`). | +| 21 | `current_series.rs` Snapshot + TopK | Current-series ranking | `ReadPopulation{TopK}` | Supported | | +| 22 | `current_series.rs` Sum / Count / Average | Current-series aggregates | `ReadPopulation{Sum,Count,Average}` | Missing | `compile` accepts only TopK. | +| 23 | `current_series.rs` Quantile | Current-series quantile | `ReadPopulation{Quantile}` | Missing | No exact quantile reduction. | +| 24 | `raw_dag.rs` weight `Column` | Summary update from a sample/projected value | `SummaryAgg` | Supported | | +| 25 | `raw_dag.rs` weight `Constant` | Unit/constant-weight update | `SummaryAgg` | Missing | `compile_node` requires a column weight. | +| 26 | `raw_dag.rs` item `Column` / `Tuple` | Keyed update item | `SummaryAgg{item}` | Supported | `keyed_summary_build`. | +| 27 | `raw_dag.rs` item `EntityIdentity` | Series-identity item | `SummaryAgg{item}` | Missing | Needs the series-identity column. | +| 28 | `physical_values.rs` `compile`, `combine` | Translate `QueryTimeOperator` to `promql_values::*`; compose fragments | n/a | Supported | Exists only because of row 1. `CompiledPhysicalDag::compose` is Planner API. | +| 29 | `query_plan.rs` `compile_native_fragment` (Semi join, Exact aggregate, Sort, Limit, Filter) | Relational value ops | `RelationalJoin`, `Value::*` | Supported | Already calls `compile`. | +| 30 | `query_time.rs` `selected_query_time_nodes`, `selected_native_expression`, `selected_aggregate_operator` | Recover operator identity from original PromQL text | Payload variants (`ExactKind::Min`/`Max`, `AggIntent`) | Supported | Payloads already carry the identity. These witnesses are needed only while row 1 remains. | +| 31 | Scan, ExactSubquery, CandidateExactSubquery, CurrentSeries ingest, ReadMaterialization, ExternalExact | Storage reads and external engines | Input contracts | Backend | | + +Totals at #475: 11 Supported, 4 Partial, 14 Missing, 2 Backend. + +## Remaining + +Every Missing or Partial row above, in order of backend usage. Rows 1 and 28 +are removed from the backend only after rows 4–13 are covered. From a8c5d3f6c4584878491767d6d8821736b2c8cc41 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:44:41 +0000 Subject: [PATCH 34/59] feat(physical): compile maintained-population aggregate readouts ReadPopulation Sum/Count/Average/Quantile now compile to a grouped aggregate over the population snapshot, so deployments no longer evaluate these readouts in their current-series store. Quantile uses a new exact Reduction::Quantile with PromQL rank interpolation. Co-Authored-By: Claude Opus 5.5 --- .../src/operators/aggregate/mod.rs | 55 +++++ .../src/physical_planner/mod.rs | 13 +- .../src/physical_planner/row_values.rs | 31 +++ .../tests/deployment_computation.rs | 200 ++++++++++++++++++ 4 files changed, 294 insertions(+), 5 deletions(-) create mode 100644 crates/asap-physical-operators/src/physical_planner/row_values.rs create mode 100644 crates/asap-physical-operators/tests/deployment_computation.rs diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs index 71e7134c..0df3c157 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/mod.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -27,6 +27,12 @@ impl Operator { false, ) } + Reduction::Quantile { column, q } => { + if plain(&input, *column)?.0 != &DataType::Float64 || q.is_nan() { + return Err(invalid("quantile requires Float64 input and a numeric q")); + } + (DataType::Float64, false) + } Reduction::Min(i) | Reduction::Max(i) => { let (t, nullable) = plain(&input, *i)?; if !ordered(t) { @@ -120,6 +126,11 @@ pub enum Reduction { Avg(usize), Min(usize), Max(usize), + /// PromQL `quantile`: linear interpolation between closest ranks. + Quantile { + column: usize, + q: f64, + }, } pub(super) fn execute<'a>( operator: &'a Operator, @@ -198,6 +209,25 @@ async fn reduce( Ok(output) } +// Matches Prometheus `quantile`: NaN for no values, ±Inf outside [0, 1]. +fn quantile(q: f64, mut values: Vec) -> f64 { + if values.is_empty() { + return f64::NAN; + } + if q < 0. { + return f64::NEG_INFINITY; + } + if q > 1. { + return f64::INFINITY; + } + values.sort_by(f64::total_cmp); + let rank = q * (values.len() - 1) as f64; + let low = rank.floor() as usize; + let high = (low + 1).min(values.len() - 1); + let weight = rank - low as f64; + values[low] * (1. - weight) + values[high] * weight +} + async fn reduce_one( rows: &[Vec], measure: &Reduction, @@ -211,6 +241,18 @@ async fn reduce_one( )) } Reduction::Sum(i) | Reduction::Avg(i) | Reduction::Min(i) | Reduction::Max(i) => *i, + Reduction::Quantile { column, q } => { + let mut values = Vec::with_capacity(rows.len()); + for row in rows { + work.checkpoint().await?; + match &row[*column] { + Value::Float64(value) => values.push(*value), + Value::Null => {} + _ => return Err(invalid("floating quantile value required")), + } + } + return Ok(Value::Float64(quantile(*q, values))); + } }; let values = rows .iter() @@ -291,3 +333,16 @@ async fn reduce_one( sum })) } + +#[cfg(test)] +mod tests { + // Quantile follows Prometheus: interpolate ranks, NaN when empty, ±Inf outside [0, 1]. + #[test] + fn quantile_matches_prometheus_edge_cases() { + assert!(super::quantile(0.5, vec![]).is_nan()); + assert_eq!(super::quantile(0.5, vec![3.]), 3.); + assert_eq!(super::quantile(0.75, vec![4., 1., 2., 3.]), 3.25); + assert_eq!(super::quantile(-0.1, vec![1.]), f64::NEG_INFINITY); + assert_eq!(super::quantile(1.1, vec![1.]), f64::INFINITY); + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index fede3777..3896e2a2 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -43,6 +43,8 @@ pub use candidates::{ mod compiled; pub use compiled::{CompiledPhysicalDag, InputContract}; +mod row_values; + /// Compile computation without opening or retaining deployment readers. /// Input contracts identify explicit boundaries selected by maintenance planning. pub fn compile( @@ -247,11 +249,6 @@ fn compile_internal( use planner_types::post_asap::maintained_population::{ PopulationInput, PopulationReadout, }; - let PopulationReadout::TopK { k } = readout else { - return Err(invalid( - "native population readout does not support this operation", - )); - }; let [producer] = inputs.as_slice() else { return Err(invalid("population readout requires one input")); }; @@ -272,6 +269,12 @@ fn compile_internal( )); } let input = schemas[0].clone(); + let PopulationReadout::TopK { k } = readout else { + let aggregate = + row_values::population_aggregate(&input, &spec.grouping, readout)?; + graph.add(id, inputs, aggregate.with_output_schema(output)?)?; + continue; + }; let groups = spec .grouping .iter() diff --git a/crates/asap-physical-operators/src/physical_planner/row_values.rs b/crates/asap-physical-operators/src/physical_planner/row_values.rs new file mode 100644 index 00000000..954da893 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/row_values.rs @@ -0,0 +1,31 @@ +//! Query-time PromQL value computation over logical row schemas. +use super::*; +use planner_types::post_asap::maintained_population::PopulationReadout; + +/// Aggregate readouts of a maintained current-series population. +pub(super) fn population_aggregate( + input: &Schema, + grouping: &[String], + readout: &PopulationReadout, +) -> Result { + let groups = grouping + .iter() + .map(|name| named_column(input, &ColumnRef::Named(name.clone()))) + .collect::, _>>()?; + let value = named_column(input, &ColumnRef::SampleValue)?; + let reduction = match readout { + PopulationReadout::Sum => Reduction::Sum(value), + PopulationReadout::Count => Reduction::Count, + PopulationReadout::Average => Reduction::Avg(value), + PopulationReadout::Quantile { q } => Reduction::Quantile { + column: value, + q: *q, + }, + PopulationReadout::TopK { .. } => { + return Err(invalid( + "TopK population readout ranks; it does not aggregate", + )) + } + }; + Operator::aggregate(input.clone(), groups, vec![("value".into(), reduction)]) +} diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs new file mode 100644 index 00000000..70880fd1 --- /dev/null +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -0,0 +1,200 @@ +//! Planner-selected PromQL computation compiles from the timed DAG alone; +//! the deployment supplies only raw rows at the ingestion frontier. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{compile, promql_rows, CompiledPhysicalDag, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{post_asap::*, pre_asap::QueryExpr, types::AccuracyTarget, workload::*}; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; + +fn lower(query: &str) -> QueryExpr { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(60_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0) +} + +fn population_dag(query: &str) -> PostAsapDag { + let root = Rc::new(promql_rows::with_series_identity(&lower(query)).unwrap()); + let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&root), + ) + .candidate(&root) + .unwrap(); + compile_post_asap_dag(&selected).unwrap() +} + +/// Raw scan nodes are the frontier; everything above them is compiled. +fn raw_inputs(dag: &PostAsapDag) -> Vec<(u64, Arc, String)> { + dag.nodes + .iter() + .filter_map(|node| match &node.payload { + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::TimeRange { child, .. }, + } => match child.as_ref() { + QueryExpr::Scan { + source: planner_types::pre_asap::Source::TimeSeries { metric }, + .. + } => Some(( + u64::from(node.id.0), + Arc::new(node.output_schema.clone()), + metric.clone(), + )), + _ => None, + }, + _ => None, + }) + .collect() +} + +type Sample = (&'static str, &'static str, &'static str, i64, f64); + +/// Compile, round-trip, bind raw `(metric, job, instance, ts, value)` samples, +/// and return `(job, value)` rows of the root. +fn run(dag: &PostAsapDag, samples: &[Sample], end: i64) -> Result, String> { + let inputs = raw_inputs(dag); + let program = compile( + dag, + inputs + .iter() + .map(|(id, schema, _)| (*id, InputContract::bounded(schema.clone()))) + .collect(), + &[u64::from(dag.root.0)], + ) + .map_err(|e| e.to_string())?; + let program: CompiledPhysicalDag = + serde_json::from_slice(&serde_json::to_vec(&program).unwrap()).unwrap(); + let sources = inputs + .iter() + .map(|(id, schema, metric)| { + let rows = samples + .iter() + .filter(|sample| sample.0 == metric) + .map(|(name, job, instance, at, value)| { + let labels = BTreeMap::from([ + ("__name__".to_string(), name.to_string()), + ("job".into(), job.to_string()), + ("instance".into(), instance.to_string()), + ]); + if schema + .fields + .iter() + .any(|f| f.name == promql_rows::SERIES_IDENTITY_COLUMN) + { + promql_rows::series_row(schema, &labels, *at, *value).unwrap() + } else { + schema + .fields + .iter() + .enumerate() + .map(|(i, f)| match f.name.as_str() { + _ if Some(i) == schema.time_index => Value::Timestamp(*at), + "value" => Value::Float64(*value), + label => Value::Utf8(labels[label].clone().into()), + }) + .collect() + } + }) + .collect(); + let batch = Batch::try_new(schema.clone(), rows).unwrap(); + ( + *id, + Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect(); + let graph = program.instantiate(sources).map_err(|e| e.to_string())?; + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: end, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = graph + .execute(program.roots(), context) + .map_err(|e| e.to_string())? + .remove(0); + let mut rows = BTreeMap::new(); + while let Some(batch) = stream.next().await { + let batch = batch.map_err(|e| e.to_string())?; + let job = batch.schema().fields.iter().position(|f| f.name == "job"); + for row in batch.rows() { + let key = match job.map(|i| &row[i]) { + Some(Value::Utf8(job)) => job.to_string(), + _ => String::new(), + }; + let value = match row.last() { + Some(Value::Float64(v)) => *v, + Some(Value::Int64(v)) => *v as f64, + other => return Err(format!("unexpected value {other:?}")), + }; + assert!(rows.insert(key, value).is_none(), "duplicate output group"); + } + } + Ok(rows) + }) +} + +const SAMPLES: &[Sample] = &[ + ("m", "api", "a", 10_000, 4.), + ("m", "api", "a", 50_000, 1.), + ("m", "api", "b", 40_000, 7.), + ("m", "api", "c", 30_000, 2.), + ("m", "db", "d", 20_000, 5.), +]; + +fn reference(pairs: &[(&str, f64)]) -> BTreeMap { + pairs.iter().map(|(k, v)| (k.to_string(), *v)).collect() +} + +// Current-series aggregates read the latest member values, matching the +// backend CurrentSeriesStore formulas (PromQL quantile interpolation). +#[test] +fn population_aggregates_match_current_series_reference() { + // Latest values: api = {a: 1, b: 7, c: 2}; db = {d: 5}. + for (query, expected) in [ + ("sum by (job) (m)", reference(&[("api", 10.), ("db", 5.)])), + ("count by (job) (m)", reference(&[("api", 3.), ("db", 1.)])), + ( + "avg by (job) (m)", + reference(&[("api", 10. / 3.), ("db", 5.)]), + ), + // Sorted api = [1, 2, 7]; rank 0.25 * 2 = 0.5 → 1.5. + ( + "quantile by (job) (0.25, m)", + reference(&[("api", 1.5), ("db", 5.)]), + ), + ] { + let dag = population_dag(query); + assert_eq!(run(&dag, SAMPLES, 60_000).unwrap(), expected, "{query}"); + } +} From 362203a43d8dc4040a8794708a6263055350f4ba Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:45:59 +0000 Subject: [PATCH 35/59] feat(physical): compile query-time arithmetic over grouped value rows A query-time Binary with a PromQL scalar-literal operand folds the literal into a projection, which also covers unary negation. Two grouped row inputs match one-to-one on equal label columns through an inner equi-join before the operator is applied. Comparisons and per-series rows still fail at compile time. Co-Authored-By: Claude Opus 5.5 --- .../src/physical_planner/mod.rs | 56 +++++- .../src/physical_planner/row_values.rs | 170 +++++++++++++++++- .../tests/deployment_computation.rs | 84 +++++++++ 3 files changed, 308 insertions(+), 2 deletions(-) diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 3896e2a2..102f32b0 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -14,7 +14,8 @@ use planner_types::{ SketchQuery, SummaryFamilyType, SummaryInputExpr, ValueOperation, }, pre_asap::{ - AggIntent, ColumnRef, CompareOpKind, GroupKeys, QueryExpr, Reduction as PlannerReduction, + AggIntent, ColumnRef, CompareOpKind, DataType, GroupKeys, QueryExpr, + Reduction as PlannerReduction, }, }; use std::{ @@ -145,7 +146,28 @@ fn compile_internal( }, ) }); + // Scalar literal operands of query-time arithmetic are folded into the consumer. + let mut literals = BTreeMap::::new(); for edge in edges { + let consumer = u64::from(edge.consumer.0); + if let ( + Payload::Fallback { expression }, + Some(PostAsapDagNode { + payload: Payload::Binary { .. }, + .. + }), + ) = ( + &nodes[&u64::from(edge.producer.0)].payload, + nodes.get(&consumer), + ) { + if let Some(value) = row_values::scalar_literal(expression) { + let left = edge.role == planner_types::post_asap::EdgeRole::Left; + if literals.insert(consumer, (value, left)).is_some() { + return Err(invalid("binary with two scalar literals is not folded")); + } + continue; + } + } dependencies .entry(u64::from(edge.consumer.0)) .or_default() @@ -360,6 +382,38 @@ fn compile_internal( )?; continue; } + if let Payload::Binary { operator } = &node.payload { + let query_time = node.output_state.timing + == planner_types::post_asap::ExecutionTiming::QueryTime; + if let Some(&(value, left)) = literals.get(&id) { + let [input] = schemas.as_slice() else { + return Err(invalid("scalar binary requires one row input")); + }; + if !query_time { + return Err(invalid("scalar literal binary must run at query time")); + } + let project = row_values::scalar_binary(input, operator, value, left) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + graph.add(id, inputs, project.with_output_schema(output)?)?; + continue; + } + let label_map = |schema: &Schema| { + schema + .fields + .iter() + .any(|f| matches!(f.dtype, SummaryFamilyType::Plain(DataType::Map { .. }))) + }; + if let (true, [left, right]) = (query_time, schemas.as_slice()) { + if !label_map(left) && !label_map(right) { + let (join, project) = row_values::grouped_binary(left, right, operator) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + graph.add(auxiliary, inputs, join)?; + graph.add(id, vec![auxiliary], project.with_output_schema(output)?)?; + auxiliary -= 1; + continue; + } + } + } let mut operator = compile_node(node, &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; if operator.is_counter_readout() { diff --git a/crates/asap-physical-operators/src/physical_planner/row_values.rs b/crates/asap-physical-operators/src/physical_planner/row_values.rs index 954da893..10c93c63 100644 --- a/crates/asap-physical-operators/src/physical_planner/row_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/row_values.rs @@ -1,6 +1,174 @@ //! Query-time PromQL value computation over logical row schemas. use super::*; -use planner_types::post_asap::maintained_population::PopulationReadout; +use planner_types::post_asap::{maintained_population::PopulationReadout, BinaryOperator}; +use planner_types::pre_asap::{BinaryOpKind, DataType, Predicate, ScalarValue}; +use std::rc::Rc; + +/// A PromQL number literal has no row schema; its consumer folds it in. +pub(super) fn scalar_literal(expression: &QueryExpr) -> Option { + match expression { + QueryExpr::PromqlScalarBridge(child) => scalar_literal(child), + QueryExpr::Literal(ScalarValue::Float64(value)) => Some(*value), + _ => None, + } +} + +/// Rows without a time column, label map, or series identity carry only +/// their group labels, so those labels are the complete PromQL identity. +fn grouped_value(input: &Schema) -> Result<(usize, Vec), Error> { + if input.time_index.is_some() + || input + .fields + .iter() + .any(|field| field.name == promql_rows::SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "row binary requires grouped rows; per-series matching needs a name-free identity", + )); + } + let mut value = None; + let mut labels = Vec::new(); + for (i, field) in input.fields.iter().enumerate() { + match &field.dtype { + SummaryFamilyType::Plain(DataType::Float64) if value.is_none() => value = Some(i), + SummaryFamilyType::Plain(DataType::Utf8) => labels.push(i), + _ => { + return Err(invalid( + "row binary requires Utf8 labels and one Float64 value", + )) + } + } + } + Ok(( + value.ok_or_else(|| invalid("row binary requires a Float64 value"))?, + labels, + )) +} + +fn arithmetic(operator: &BinaryOperator) -> Result<(), Error> { + if !matches!(operator.kind, BinaryOpKind::Arithmetic(_)) { + return Err(invalid( + "row comparison requires filter or bool semantics, which Binary does not carry", + )); + } + Ok(()) +} + +/// Apply `vector op scalar` (or `scalar op vector`) to each row's value. +pub(super) fn scalar_binary( + input: &Schema, + operator: &BinaryOperator, + literal: f64, + literal_left: bool, +) -> Result { + arithmetic(operator)?; + let (value, _) = grouped_value(input)?; + let literal = Expression::Literal { + value: crate::values::Value::Float64(literal), + dtype: DataType::Float64, + }; + let columns = input + .fields + .iter() + .enumerate() + .map(|(i, field)| { + let expression = if i != value { + Expression::Column(i) + } else if literal_left { + binary(operator, literal.clone(), Expression::Column(i)) + } else { + binary(operator, Expression::Column(i), literal.clone()) + }; + (field.name.clone(), expression) + }) + .collect(); + Operator::project(input.clone(), columns) +} + +/// One-to-one PromQL matching of grouped rows on equal label sets. Returns +/// the inner equi-join and the projection that applies the operator. +pub(super) fn grouped_binary( + left: &Schema, + right: &Schema, + operator: &BinaryOperator, +) -> Result<(Operator, Operator), Error> { + arithmetic(operator)?; + let (left_value, left_labels) = grouped_value(left)?; + let (right_value, right_labels) = grouped_value(right)?; + if left_labels.len() != right_labels.len() { + return Err(invalid("row binary inputs have different label sets")); + } + let width = left.fields.len(); + let keys = left_labels + .iter() + .map(|&l| { + let name = &left.fields[l].name; + let r = right_labels + .iter() + .copied() + .find(|&r| &right.fields[r].name == name) + .ok_or_else(|| invalid("row binary inputs have different label sets"))?; + let (a, b) = ( + Rc::new(QueryExpr::Column(l)), + Rc::new(QueryExpr::Column(width + r)), + ); + let equal = QueryExpr::Compare { + left: a.clone(), + op: CompareOpKind::Eq, + right: b.clone(), + }; + // A nullable label compares like PromQL's empty label: absent on both sides matches. + Ok(if left.fields[l].nullable || right.fields[r].nullable { + QueryExpr::BoolOr(vec![ + equal, + QueryExpr::BoolAnd(vec![QueryExpr::IsNull(a), QueryExpr::IsNull(b)]), + ]) + } else { + equal + }) + }) + .collect::, Error>>()?; + let predicate = Predicate(Rc::new(QueryExpr::BoolAnd(keys))); + let mut joined = left.fields.clone(); + joined.extend(right.fields.iter().cloned()); + let join = Operator::relational_join( + left.clone(), + right.clone(), + planner_types::pre_asap::JoinKind::Inner, + &predicate, + Arc::new(planner_types::post_asap::SummarySchema { + fields: joined, + time_index: None, + }), + )?; + let columns = left + .fields + .iter() + .enumerate() + .map(|(i, field)| { + let expression = if i == left_value { + binary( + operator, + Expression::Column(i), + Expression::Column(width + right_value), + ) + } else { + Expression::Column(i) + }; + (field.name.clone(), expression) + }) + .collect(); + let project = Operator::project(join.schema(), columns)?; + Ok((join, project)) +} + +fn binary(operator: &BinaryOperator, left: Expression, right: Expression) -> Expression { + Expression::Binary { + operator: operator.clone(), + left: Box::new(left), + right: Box::new(right), + } +} /// Aggregate readouts of a maintained current-series population. pub(super) fn population_aggregate( diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index 70880fd1..25899ec6 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -40,6 +40,27 @@ fn lower(query: &str) -> QueryExpr { .remove(0) } +/// The first exact summary candidate, as Planner selection would hand it over. +fn exact_dag(query: &str) -> PostAsapDag { + use asap_aware_mapping::{Replacement, ReplacementStrategy, TargetSubDAG}; + let expression = lower(query); + let root = Rc::new(promql_rows::with_series_identity(&expression).unwrap_or(expression)); + asap_aware_mapping::SketchAlgorithmStrategy::new(&asap_aware_mapping::DefaultCostModel) + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .find_map(|candidate| match candidate.replacement { + Replacement::Summary(node) => { + let dag = compile_post_asap_dag(&node).ok()?; + dag.nodes + .iter() + .all(|n| !matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { family, .. } if !matches!(family, SummaryFamilyType::ExactAggregate(..)))) + .then_some(dag) + } + _ => None, + }) + .unwrap() +} + fn population_dag(query: &str) -> PostAsapDag { let root = Rc::new(promql_rows::with_series_identity(&lower(query)).unwrap()); let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( @@ -198,3 +219,66 @@ fn population_aggregates_match_current_series_reference() { assert_eq!(run(&dag, SAMPLES, 60_000).unwrap(), expected, "{query}"); } } + +// Scalar operands on either side apply to every grouped value, including negation. +#[test] +fn scalar_literal_arithmetic_applies_to_grouped_values() { + // sum_over_time over 5m per job: api = 4 + 1 + 7 + 2 = 14, db = 5. + for (query, expected) in [ + ( + "sum by (job) (sum_over_time(m[5m])) * 2", + reference(&[("api", 28.), ("db", 10.)]), + ), + ( + "100 - sum by (job) (sum_over_time(m[5m]))", + reference(&[("api", 86.), ("db", 95.)]), + ), + ( + "-sum by (job) (sum_over_time(m[5m]))", + reference(&[("api", -14.), ("db", -5.)]), + ), + ] { + assert_eq!( + run(&exact_dag(query), SAMPLES, 60_000).unwrap(), + expected, + "{query}" + ); + } +} + +// Grouped vectors match one-to-one on labels; unmatched groups are dropped and +// unchecked division by zero yields +Inf as in PromQL. +#[test] +fn grouped_vector_arithmetic_matches_labels() { + let samples: &[Sample] = &[ + ("a", "api", "x", 10_000, 6.), + ("a", "api", "y", 20_000, 3.), + ("a", "db", "x", 10_000, 1.), + ("a", "web", "x", 10_000, 1.), + ("b", "api", "x", 10_000, 3.), + ("b", "db", "x", 10_000, 0.), + ("b", "cache", "x", 10_000, 1.), + ]; + let dag = + exact_dag("sum by (job) (sum_over_time(a[5m])) / sum by (job) (sum_over_time(b[5m]))"); + assert_eq!( + run(&dag, samples, 60_000).unwrap(), + reference(&[("api", 3.), ("db", f64::INFINITY)]) + ); +} + +// Comparisons need filter/bool semantics that `Binary` does not carry, so +// they fail at compile time instead of emitting 0/1 values. +#[test] +fn row_comparison_fails_closed() { + let mut dag = exact_dag("sum by (job) (sum_over_time(m[5m])) * 2"); + for node in &mut dag.nodes { + if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { + operator.kind = planner_types::pre_asap::BinaryOpKind::Compare( + planner_types::pre_asap::CompareOpKind::Gt, + ); + } + } + let error = run(&dag, SAMPLES, 60_000).unwrap_err(); + assert!(error.contains("comparison"), "{error}"); +} From 49cf0446dfdb062bff27ed27c872df5b2e8f1367 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:45:59 +0000 Subject: [PATCH 36/59] fix(physical): finalize exact counts to declared Float64 values Exact count readout yields Int64, but PromQL declares a Float64 sample, so compile rejected count finalization. Convert exactly. Co-Authored-By: Claude Opus 5.5 --- .../src/physical_planner/mod.rs | 35 +++++++++++++++ .../tests/deployment_computation.rs | 44 +++++++++++++++++++ 2 files changed, 79 insertions(+) diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 102f32b0..afc1ff1f 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -414,6 +414,41 @@ fn compile_internal( } } } + if let Payload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + } = &node.payload + { + // Exact counts read out as Int64; PromQL declares a Float64 sample. + let readout = bind_operation(node, &schemas) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + let actual = readout.schema(); + let converted = actual.fields.iter().zip(&output.fields).position(|(a, d)| { + a.dtype == SummaryFamilyType::Plain(DataType::Int64) + && d.dtype == SummaryFamilyType::Plain(DataType::Float64) + }); + if let Some(column) = converted { + let columns = actual + .fields + .iter() + .enumerate() + .map(|(i, field)| { + ( + field.name.clone(), + if i == column { + Expression::ExactFloat64(i) + } else { + Expression::Column(i) + }, + ) + }) + .collect(); + let project = Operator::project(actual, columns)?.with_output_schema(output)?; + graph.add(auxiliary, inputs, readout)?; + graph.add(id, vec![auxiliary], project)?; + auxiliary -= 1; + continue; + } + } let mut operator = compile_node(node, &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; if operator.is_counter_readout() { diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index 25899ec6..f08cfcae 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -267,6 +267,50 @@ fn grouped_vector_arithmetic_matches_labels() { ); } +// Exact observation counts finalize to the Float64 value PromQL declares, +// then roll up per job: api has 2 + 1 + 1 samples in 5m, db has 1. +#[test] +fn exact_count_finalizes_to_declared_float_value() { + let mut dag = exact_dag("sum by (job) (count_over_time(m[5m]))"); + let finalize = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator + } + ) + }) + .unwrap() + .clone(); + let root = dag.nodes.iter().find(|n| n.id == dag.root).unwrap().clone(); + let mut edge = dag + .edges + .iter() + .find(|e| e.producer == finalize.id) + .unwrap() + .clone(); + // Read the rolled-up exact state the same way the query path does. + let mut read = finalize.clone(); + read.id = PostAsapNodeId(root.id.0 + 1); + read.output_schema = root.output_schema.clone(); + read.output_schema.fields.last_mut().unwrap().dtype = + SummaryFamilyType::Plain(planner_types::pre_asap::DataType::Float64); + edge.producer = root.id; + edge.consumer = read.id; + edge.intermediate_schema = root.output_schema.clone(); + edge.data_state = root.output_state.clone(); + dag.root = read.id; + dag.nodes.push(read); + dag.edges.push(edge); + assert_eq!( + run(&dag, SAMPLES, 60_000).unwrap(), + reference(&[("api", 4.), ("db", 1.)]) + ); +} + // Comparisons need filter/bool semantics that `Binary` does not carry, so // they fail at compile time instead of emitting 0/1 values. #[test] From 3e2609a14cfba141cc0f9aa7d299ec18e0a3129b Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:45:59 +0000 Subject: [PATCH 37/59] docs: record covered and remaining physical compile gaps Co-Authored-By: Claude Opus 5.5 --- .../develop_docs/physical-compile-coverage.md | 27 +++++++++++++++++-- 1 file changed, 25 insertions(+), 2 deletions(-) diff --git a/docs/develop_docs/physical-compile-coverage.md b/docs/develop_docs/physical-compile-coverage.md index 4dc309ae..5db0edce 100644 --- a/docs/develop_docs/physical-compile-coverage.md +++ b/docs/develop_docs/physical-compile-coverage.md @@ -63,7 +63,30 @@ Status values: Totals at #475: 11 Supported, 4 Partial, 14 Missing, 2 Backend. +## Covered after this change + +| Row | Change | +|---|---| +| 4, 8, 13 | Query-time `Binary` folds a scalar-literal operand into a projection over grouped value rows. | +| 5 | Query-time `Binary` over grouped value rows performs an inner equi-join on equal label columns, then applies the operator. Per-series rows remain Partial. | +| 15 | Count finalization converts exactly to the declared Float64 value. | +| 22, 23 | `ReadPopulation` Sum/Count/Average/Quantile compile to grouped aggregation. `Reduction::Quantile` implements PromQL interpolation. | + +Totals after this change: 17 Supported, 4 Partial, 8 Missing, 2 Backend. + ## Remaining -Every Missing or Partial row above, in order of backend usage. Rows 1 and 28 -are removed from the backend only after rows 4–13 are covered. +In order of backend usage: + +1. Row 1 and rows 9–12: lower PromQL-shaped `Fallback{QueryExpr}` subtrees + (range functions over matrices, `scalar()`, `histogram_quantile`, `sort`, + subquery grids). After that, rows 28 and 30 can be deleted from the backend. +2. Row 7: comparison filters and `bool` comparisons. This needs `return_bool` + in the `Binary` payload. `compile` currently rejects comparisons. +3. Row 5 for per-series rows: matching needs a metric-name-free series + identity, not the full `$promql_series_identity`. +4. Rows 25 and 27: constant weights and `EntityIdentity` items for precompute + `SummaryAgg`. +5. Row 16: a label-map sketch-state readout, the counterpart of + `compile_exact_readout`, and MetricsQL `__name__` retention rules. +6. Row 20: summary join, subtract, and delete. From 644cd95802f6ca1c552f3084f00cd83587dde610 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 02:50:21 +0000 Subject: [PATCH 38/59] fix(physical): empty global population readouts and NaN quantile order A global population aggregate with no members emitted one row; PromQL returns an empty vector. Quantile now orders NaN samples first, as Prometheus does. Found in independent review. Co-Authored-By: Claude Opus 5.5 --- .../src/operators/aggregate/mod.rs | 11 ++++-- .../src/physical_planner/mod.rs | 11 ++++-- .../src/physical_planner/row_values.rs | 34 +++++++++++++++++-- .../tests/deployment_computation.rs | 14 +++++++- 4 files changed, 62 insertions(+), 8 deletions(-) diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs index 0df3c157..37b4e96d 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/mod.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -209,7 +209,8 @@ async fn reduce( Ok(output) } -// Matches Prometheus `quantile`: NaN for no values, ±Inf outside [0, 1]. +// Matches Prometheus `quantile`: NaN for no values, ±Inf outside [0, 1], +// and NaN samples ordered first. fn quantile(q: f64, mut values: Vec) -> f64 { if values.is_empty() { return f64::NAN; @@ -220,7 +221,12 @@ fn quantile(q: f64, mut values: Vec) -> f64 { if q > 1. { return f64::INFINITY; } - values.sort_by(f64::total_cmp); + values.sort_by(|a, b| match (a.is_nan(), b.is_nan()) { + (true, true) => std::cmp::Ordering::Equal, + (true, false) => std::cmp::Ordering::Less, + (false, true) => std::cmp::Ordering::Greater, + _ => a.total_cmp(b), + }); let rank = q * (values.len() - 1) as f64; let low = rank.floor() as usize; let high = (low + 1).min(values.len() - 1); @@ -344,5 +350,6 @@ mod tests { assert_eq!(super::quantile(0.75, vec![4., 1., 2., 3.]), 3.25); assert_eq!(super::quantile(-0.1, vec![1.]), f64::NEG_INFINITY); assert_eq!(super::quantile(1.1, vec![1.]), f64::INFINITY); + assert_eq!(super::quantile(1., vec![2., f64::NAN, 1.]), 2.); } } diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index afc1ff1f..10fae689 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -292,9 +292,16 @@ fn compile_internal( } let input = schemas[0].clone(); let PopulationReadout::TopK { k } = readout else { - let aggregate = + let mut chain = row_values::population_aggregate(&input, &spec.grouping, readout)?; - graph.add(id, inputs, aggregate.with_output_schema(output)?)?; + let last = chain.pop().expect("nonempty chain"); + let mut inputs = inputs; + for operator in chain { + graph.add(auxiliary, inputs, operator)?; + inputs = vec![auxiliary]; + auxiliary -= 1; + } + graph.add(id, inputs, last.with_output_schema(output)?)?; continue; }; let groups = spec diff --git a/crates/asap-physical-operators/src/physical_planner/row_values.rs b/crates/asap-physical-operators/src/physical_planner/row_values.rs index 10c93c63..737c8065 100644 --- a/crates/asap-physical-operators/src/physical_planner/row_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/row_values.rs @@ -170,12 +170,12 @@ fn binary(operator: &BinaryOperator, left: Expression, right: Expression) -> Exp } } -/// Aggregate readouts of a maintained current-series population. +/// Aggregate readouts of a maintained current-series population, as a chain. pub(super) fn population_aggregate( input: &Schema, grouping: &[String], readout: &PopulationReadout, -) -> Result { +) -> Result, Error> { let groups = grouping .iter() .map(|name| named_column(input, &ColumnRef::Named(name.clone()))) @@ -195,5 +195,33 @@ pub(super) fn population_aggregate( )) } }; - Operator::aggregate(input.clone(), groups, vec![("value".into(), reduction)]) + if !groups.is_empty() { + return Ok(vec![Operator::aggregate( + input.clone(), + groups, + vec![("value".into(), reduction)], + )?]); + } + // A global aggregate over no members is an empty PromQL vector, not one row. + let aggregate = Operator::aggregate( + input.clone(), + vec![], + vec![ + ("value".into(), reduction), + ("members".into(), Reduction::Count), + ], + )?; + let zero = Expression::Literal { + value: crate::values::Value::Int64(0), + dtype: DataType::Int64, + }; + let filter = Operator::filter( + aggregate.schema(), + Expression::Less(Box::new(zero), Box::new(Expression::Column(1))), + )?; + let project = Operator::project( + filter.schema(), + vec![("value".into(), Expression::Column(0))], + )?; + Ok(vec![aggregate, filter, project]) } diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index f08cfcae..b026f47f 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -220,6 +220,18 @@ fn population_aggregates_match_current_series_reference() { } } +// A global readout of an empty population is an empty vector, as in PromQL. +#[test] +fn global_population_aggregate_of_no_members_is_empty() { + // Latest values are [1, 2, 5, 7] at 60s; every member has expired by 1000s. + for (query, expected) in [("sum(m)", 15.), ("count(m)", 4.), ("quantile(0.5, m)", 3.5)] { + let dag = population_dag(query); + let live = run(&dag, SAMPLES, 60_000).unwrap(); + assert_eq!(live, reference(&[("", expected)]), "{query}"); + assert!(run(&dag, SAMPLES, 1_000_000).unwrap().is_empty(), "{query}"); + } +} + // Scalar operands on either side apply to every grouped value, including negation. #[test] fn scalar_literal_arithmetic_applies_to_grouped_values() { @@ -301,7 +313,7 @@ fn exact_count_finalizes_to_declared_float_value() { edge.producer = root.id; edge.consumer = read.id; edge.intermediate_schema = root.output_schema.clone(); - edge.data_state = root.output_state.clone(); + edge.data_state = root.output_state; dag.root = read.id; dag.nodes.push(read); dag.edges.push(edge); From f9d8a3a80bf55d716c77cdaee67a7d344ca2d8d1 Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:09:28 +0000 Subject: [PATCH 39/59] integrate: let coverage lowering chains use per-node helper indices #479 (d41f201) allows several helper operators per Planner node, so the coverage lowerings added here advance `auxiliary`; make it mutable. Taken from integration commit da77b78. Co-Authored-By: Claude Opus 5.5 --- crates/asap-physical-operators/src/physical_planner/mod.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 10fae689..c7967288 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -200,7 +200,7 @@ fn compile_internal( let mut graph = CompiledPhysicalDag::new(roots.to_vec()); for id in ordered { let node = nodes[&id]; - let auxiliary = helper_id(id, 0); + let mut auxiliary = helper_id(id, 0); let output = Arc::new(node.output_schema.clone()); crate::values::validate_schema(&output)?; if let Some(source) = sources.remove(&id) { From 2416901b8d26ec902822176fade17736ab5d0b39 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 03:45:46 +0000 Subject: [PATCH 40/59] fix(promql): retain subquery offset and @ as a TimeShift Co-Authored-By: Claude Opus 5.5 --- crates/frontend-promql/src/promql.rs | 22 ++++++++++++++----- .../frontend-promql/tests/promql_lowering.rs | 17 ++++++++++++++ crates/types/src/pre_asap/query_expr.rs | 3 ++- 3 files changed, 36 insertions(+), 6 deletions(-) diff --git a/crates/frontend-promql/src/promql.rs b/crates/frontend-promql/src/promql.rs index c3df6c67..b71a8c1e 100644 --- a/crates/frontend-promql/src/promql.rs +++ b/crates/frontend-promql/src/promql.rs @@ -307,11 +307,23 @@ fn walk(expr: &Expr) -> Result { vector_match: None, }), }, - Expr::Subquery(sq) => Ok(Unresolved::PromqlSubquery { - range: sq.range, - resolution: sq.step, - child: Rc::new(walk(&sq.expr)?), - }), + Expr::Subquery(sq) => { + let subquery = Unresolved::PromqlSubquery { + range: sq.range, + resolution: sq.step, + child: Rc::new(walk(&sq.expr)?), + }; + // `offset`/`@` move the whole subquery, including its step grid. + let shift = time_shift(sq.offset.as_ref(), sq.at.as_ref())?; + Ok(if shift.is_identity() { + subquery + } else { + Unresolved::TimeShift { + shift, + child: Rc::new(subquery), + } + }) + } Expr::VectorSelector(vs) => { let (metric, matchers, shift) = vs_parts(vs)?; Ok(instant_source(metric, matchers, shift)) diff --git a/crates/frontend-promql/tests/promql_lowering.rs b/crates/frontend-promql/tests/promql_lowering.rs index 2dd56173..3c92b7ba 100644 --- a/crates/frontend-promql/tests/promql_lowering.rs +++ b/crates/frontend-promql/tests/promql_lowering.rs @@ -1202,3 +1202,20 @@ fn histogram_quantiles_rejects_an_out_of_range_quantile() { ); } } + +// A subquery's `offset`/`@` shift the whole subquery, so the tree keeps them. +#[test] +fn subquery_time_shift_is_retained() { + let QueryExpr::Aggregate { child, .. } = lower("max_over_time(m[5m:1m] offset 1m)") else { + panic!("expected a range function"); + }; + let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + panic!("subquery offset was dropped: {child:?}"); + }; + assert_eq!(shift.offset_ms, 60_000); + assert!(matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. })); + assert!(matches!( + lower("max_over_time(m[5m:1m] @ 100)"), + QueryExpr::Aggregate { child, .. } if matches!(child.as_ref(), QueryExpr::TimeShift { .. }) + )); +} diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs index dec0107a..5b63e5a2 100644 --- a/crates/types/src/pre_asap/query_expr.rs +++ b/crates/types/src/pre_asap/query_expr.rs @@ -878,7 +878,8 @@ pub enum QueryExpr { /// unchanged. Wraps the shifted selector directly — `m offset 1h` → /// `TimeShift { Scan }`; a ranged selector `m[5m] offset 1h` → /// `TimeRange { 5m, TimeShift { Scan } }` (the range is taken at the shifted - /// time). Never carries the identity shift (the converter emits a bare + /// time). A shifted subquery wraps the `PromqlSubquery`, moving its step + /// grid. Never carries the identity shift (the converter emits a bare /// selector when neither modifier is present). TimeShift { shift: TimeShift, From 2cd0eecd2731b0e012165e3597e053bb7ac5a586 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 03:45:46 +0000 Subject: [PATCH 41/59] feat(physical): add a PromQL per-series window operator Evaluates instant selection and range functions over left-open windows at the query time or on a subquery step grid. Conflicts with earlier stack changes resolved to the integration tree: - crates/asap-physical-operators/src/operators/mod.rs: a7ff3ae Merge remote-tracking branch 'origin/feat/physical-compile-promql-fallback' into integration/planner-for-backend Co-Authored-By: Claude Opus 5.5 --- .../src/operators/aggregate/temporal.rs | 79 +++--- .../src/operators/mod.rs | 14 ++ .../src/operators/series_window.rs | 226 ++++++++++++++++++ .../src/operators/unchecked.rs | 13 + 4 files changed, 297 insertions(+), 35 deletions(-) create mode 100644 crates/asap-physical-operators/src/operators/series_window.rs diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs index dbf66108..c14abcbb 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -65,38 +65,7 @@ pub(in crate::operators) async fn reduce( "duplicate or out-of-window timestamp".into(), )); } - match intent { - AggIntent::Rate => rate(&points, start, end).map(Value::Float64), - AggIntent::Increase => rate(&points, start, end) - .map(|v| Value::Float64(v * (end as f64 - start as f64) / 1000.)), - AggIntent::Count { .. } => Some(Value::Int64( - i64::try_from(points.len()) - .map_err(|_| Error::Invalid("count overflow".into()))?, - )), - AggIntent::Sum { .. } => Some(Value::Float64(points.iter().map(|p| p.1).sum())), - AggIntent::Avg { .. } => Some(Value::Float64( - points.iter().map(|p| p.1).sum::() / points.len() as f64, - )), - AggIntent::Min { .. } => { - Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { - if a.is_nan() || p.1 < a { - p.1 - } else { - a - } - }))) - } - AggIntent::Max { .. } => { - Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { - if a.is_nan() || p.1 > a { - p.1 - } else { - a - } - }))) - } - _ => return Err(Error::Invalid("unsupported temporal intent".into())), - } + window_value(intent, &points, start, end)? }; if let Some(result) = result { keys.push(result); @@ -106,7 +75,47 @@ pub(in crate::operators) async fn reduce( Ok(output) } -fn rate(points: &[(i64, f64)], start: i64, end: i64) -> Option { +/// One series' value over its sorted samples in the window `(start, end]`. +/// `None` means PromQL emits no sample for this series. +pub(in crate::operators) fn window_value( + intent: &AggIntent, + points: &[(i64, f64)], + start: i64, + end: i64, +) -> Result, Error> { + Ok(match intent { + AggIntent::Rate => rate(points, start, end, true).map(Value::Float64), + AggIntent::Delta => rate(points, start, end, false) + .map(|v| Value::Float64(v * (end as f64 - start as f64) / 1000.)), + AggIntent::Increase => rate(points, start, end, true) + .map(|v| Value::Float64(v * (end as f64 - start as f64) / 1000.)), + AggIntent::Count { .. } => Some(Value::Int64( + i64::try_from(points.len()).map_err(|_| Error::Invalid("count overflow".into()))?, + )), + AggIntent::Sum { .. } => Some(Value::Float64(points.iter().map(|p| p.1).sum())), + AggIntent::Avg { .. } => Some(Value::Float64( + points.iter().map(|p| p.1).sum::() / points.len() as f64, + )), + AggIntent::Min { .. } => Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { + if a.is_nan() || p.1 < a { + p.1 + } else { + a + } + }))), + AggIntent::Max { .. } => Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { + if a.is_nan() || p.1 > a { + p.1 + } else { + a + } + }))), + _ => return Err(Error::Invalid("unsupported temporal intent".into())), + }) +} + +/// Prometheus `extrapolatedRate`; `counter` enables reset correction and the zero bound. +fn rate(points: &[(i64, f64)], start: i64, end: i64, counter: bool) -> Option { if points.len() < 2 { return None; } @@ -118,7 +127,7 @@ fn rate(points: &[(i64, f64)], start: i64, end: i64) -> Option { } let mut delta = last - first; for pair in points.windows(2) { - if pair[1].1 < pair[0].1 { + if counter && pair[1].1 < pair[0].1 { delta += pair[0].1; } } @@ -129,7 +138,7 @@ fn rate(points: &[(i64, f64)], start: i64, end: i64) -> Option { to_start = average / 2.; } // Apply the zero bound after the sparse-window half-interval cap. - if delta > 0. && first >= 0. { + if counter && delta > 0. && first >= 0. { to_start = to_start.min(span * first / delta); } if to_end >= average * 1.1 { diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index 9ae0772c..cb7c5433 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -23,6 +23,7 @@ mod joins; mod limit; mod projection; mod scope_timestamp; +mod series_window; mod sort; mod source; mod summary; @@ -30,6 +31,7 @@ mod unchecked; pub(crate) mod vector_binary; pub(crate) mod vector_window; pub use aggregate::Reduction; +pub use series_window::SubquerySteps; pub use sort::SortKey; pub use summary::ReadoutQuery; #[derive(Clone, serde::Serialize, serde::Deserialize)] @@ -66,6 +68,14 @@ enum Kind { intent: Box>, }, HistogramQuantile, + SeriesWindow { + function: Option>>, + coordinate: usize, + value: usize, + range_ms: i64, + offset_ms: i64, + steps: Option, + }, Project(Vec), Filter(Expression), Limit { @@ -237,6 +247,7 @@ impl PhysicalOperator for Operator { | Kind::RangeWindow { .. } | Kind::HistogramQuantile | Kind::CurrentSeries { .. } + | Kind::SeriesWindow { .. } | Kind::Aggregate { .. } | Kind::Window { .. } | Kind::Join { .. } @@ -279,6 +290,7 @@ impl PhysicalOperator for Operator { Kind::AlignedBinary { .. } => "AlignedBinary", Kind::RangeWindow { .. } => "RangeWindow", Kind::HistogramQuantile => "HistogramQuantile", + Kind::SeriesWindow { .. } => "SeriesWindow", Kind::Project(_) => "Project", Kind::Filter(_) => "Filter", Kind::Limit { .. } => "Limit", @@ -295,6 +307,7 @@ impl PhysicalOperator for Operator { } fn validate_context(&self, context: &RunContext) -> Result<(), Error> { current_series::validate_context(self, context)?; + series_window::validate_context(self, context)?; self.readout_range(context).map(|_| ()) } fn input_schemas(&self) -> Vec { @@ -323,6 +336,7 @@ impl PhysicalOperator for Operator { Kind::Project(_) => projection::execute(self, inputs, context), Kind::CurrentSeries { .. } => current_series::execute(self, inputs, context), Kind::ScopeTimestamp { .. } => scope_timestamp::execute(self, inputs, context), + Kind::SeriesWindow { .. } => series_window::execute(self, inputs, context), Kind::Filter(_) => filter::execute(self, inputs, context), Kind::Limit { .. } => limit::execute(self, inputs, context), Kind::Sort { .. } => sort::execute(self, inputs, context), diff --git a/crates/asap-physical-operators/src/operators/series_window.rs b/crates/asap-physical-operators/src/operators/series_window.rs new file mode 100644 index 00000000..ad56db0b --- /dev/null +++ b/crates/asap-physical-operators/src/operators/series_window.rs @@ -0,0 +1,226 @@ +//! PromQL per-series evaluation over the samples before an evaluation instant. +use super::*; +use planner_types::pre_asap::AggIntent; + +/// A PromQL subquery grid: every multiple of `step_ms` in +/// `(T - offset_ms - range_ms, T - offset_ms]`, where `T` is the query time. +#[derive(Clone, Copy, Debug, PartialEq, Eq, serde::Serialize, serde::Deserialize)] +pub struct SubquerySteps { + pub range_ms: i64, + pub step_ms: i64, + pub offset_ms: i64, +} + +const STALE_MARKER: u64 = 0x7ff0_0000_0000_0002; +const MAX_SUBQUERY_STEPS: i64 = 100_000; + +impl Operator { + /// Evaluate each series at instant `t` over its samples in + /// `(t - offset_ms - range_ms, t - offset_ms]`. `t` is the query time, or + /// each step of `steps`. `function: None` is instant selection: the latest + /// sample, absent if it is a stale marker. Range functions ignore stale + /// markers. A series is every column except the time and `value` columns. + /// Output rows keep the input schema, with time `t` and the result value. + pub fn series_window( + input: Schema, + function: Option>, + range_ms: i64, + offset_ms: i64, + steps: Option, + ) -> Result { + let coordinate = input + .time_index + .ok_or_else(|| invalid("series window requires a time column"))?; + let values = input + .fields + .iter() + .enumerate() + .filter(|(_, f)| f.name == "value") + .map(|(i, _)| i) + .collect::>(); + let [value] = values.as_slice() else { + return Err(invalid("series window requires one value column")); + }; + if plain(&input, coordinate)? != (&DataType::Timestamp, false) + || plain(&input, *value)? != (&DataType::Float64, false) + { + return Err(invalid( + "series window requires non-null time and Float64 value", + )); + } + if range_ms <= 0 || steps.is_some_and(|s| s.range_ms <= 0 || s.step_ms <= 0) { + return Err(invalid("series window ranges and steps must be positive")); + } + // Bounds per-run work independently of the data, as the backend's grid does. + if steps.is_some_and(|s| s.range_ms / s.step_ms > MAX_SUBQUERY_STEPS) { + return Err(invalid("subquery exceeds 100000 steps")); + } + if !matches!( + function, + None | Some( + AggIntent::Rate + | AggIntent::Increase + | AggIntent::Delta + | AggIntent::Count { .. } + | AggIntent::Sum { col: None } + | AggIntent::Avg { col: None } + | AggIntent::Min { col: None } + | AggIntent::Max { col: None } + ) + ) { + return Err(invalid("unsupported PromQL range function")); + } + Ok(Self { + kind: Kind::SeriesWindow { + function: function.map(Box::new), + coordinate, + value: *value, + range_ms, + offset_ms, + steps, + }, + inputs: vec![input.clone()], + output: input, + }) + } +} + +/// The first and last evaluation instants and the step between them. The grid +/// is iterated, not allocated: its size depends only on the query. +fn evaluation_times( + context: &RunContext, + steps: Option, +) -> Result<(i64, i64, i64), Error> { + let crate::runtime::Scope::Query { + evaluation_time_ms, .. + } = context.scope + else { + return Err(invalid("series window requires a query evaluation time")); + }; + let Some(steps) = steps else { + return Ok((evaluation_time_ms, evaluation_time_ms, 1)); + }; + let overflow = || invalid("subquery grid overflows"); + let end = evaluation_time_ms + .checked_sub(steps.offset_ms) + .ok_or_else(overflow)?; + let start = end.checked_sub(steps.range_ms).ok_or_else(overflow)?; + let first = (start.div_euclid(steps.step_ms) + 1) + .checked_mul(steps.step_ms) + .ok_or_else(overflow)?; + Ok((first, end, steps.step_ms)) +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::SeriesWindow { + function, + coordinate, + value, + range_ms, + offset_ms, + steps, + } = &operator.kind + else { + unreachable!() + }; + let (coordinate, value) = (*coordinate, *value); + let input = inputs + .pop() + .ok_or_else(|| invalid("series window input missing"))?; + let times = evaluation_times(&context, *steps)?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + let identity = (0..operator.output.fields.len()) + .filter(|&i| i != coordinate && i != value) + .collect::>(); + let mut series = BTreeMap::>, (usize, Vec<(i64, f64)>)>::new(); + for (index, row) in rows.iter().enumerate() { + work.checkpoint().await?; + let key = group_key(row, &identity)?; + let (Value::Timestamp(time), Value::Float64(sample)) = (&row[coordinate], &row[value]) + else { + return Err(invalid("series window requires time and value samples")); + }; + workspace.grow(16)?; + if !series.contains_key(&key) { + workspace.grow(key_bytes(&key) + 64)?; + } + series + .entry(key) + .or_insert_with(|| (index, Vec::new())) + .1 + .push((*time, *sample)); + } + for (_, points) in series.values_mut() { + work.checkpoint().await?; + points.sort_by_key(|p| p.0); + if points.windows(2).any(|p| p[0].0 == p[1].0) { + return Err(invalid("duplicate sample timestamp for one series")); + } + } + let mut output = Vec::new(); + let (mut time, last, step) = times; + while time <= last && !series.is_empty() { + work.checkpoint().await?; + let overflow = || invalid("series window overflows"); + let end = time.checked_sub(*offset_ms).ok_or_else(overflow)?; + let start = end.checked_sub(*range_ms).ok_or_else(overflow)?; + for (template, points) in series.values() { + work.checkpoint().await?; + // PromQL ranges are left-open: a sample at `start` is outside. + let first = points.partition_point(|p| p.0 <= start); + let last = points.partition_point(|p| p.0 <= end); + let points = &points[first..last]; + let result = match function { + None => points + .last() + .filter(|p| p.1.to_bits() != STALE_MARKER) + .map(|p| p.1), + Some(intent) => { + let fresh = points + .iter() + .copied() + .filter(|p| p.1.to_bits() != STALE_MARKER) + .collect::>(); + if fresh.is_empty() { + None + } else { + match aggregate::temporal::window_value(intent, &fresh, start, end)? { + Some(Value::Float64(v)) => Some(v), + Some(Value::Int64(v)) => Some(v as f64), + Some(_) => return Err(invalid("invalid range function result")), + None => None, + } + } + } + }; + if let Some(result) = result { + let mut row = rows[*template].clone(); + row[coordinate] = Value::Timestamp(time); + row[value] = Value::Float64(result); + workspace.grow(row_bytes(&row))?; + output.push(row); + } + } + let Some(next) = time.checked_add(step) else { + break; + }; + time = next; + } + Batch::try_new(operator.output.clone(), output) + }) + .boxed_local()) +} + +pub(super) fn validate_context(operator: &Operator, context: &RunContext) -> Result<(), Error> { + if let Kind::SeriesWindow { steps, .. } = operator.kind { + evaluation_times(context, steps)?; + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/operators/unchecked.rs b/crates/asap-physical-operators/src/operators/unchecked.rs index f927a4e4..80a4544a 100644 --- a/crates/asap-physical-operators/src/operators/unchecked.rs +++ b/crates/asap-physical-operators/src/operators/unchecked.rs @@ -50,6 +50,19 @@ impl TryFrom for Operator { } => Operator::aligned_binary(input(0)?, input(1)?, keys, values, operator)?, Kind::RangeWindow { intent } => Operator::range_window(*intent)?, Kind::HistogramQuantile => Operator::histogram_quantile(), + Kind::SeriesWindow { + function, + range_ms, + offset_ms, + steps, + .. + } => Operator::series_window( + input(0)?, + function.map(|f| *f), + range_ms, + offset_ms, + steps, + )?, Kind::Project(expressions) => { if expressions.len() != output.fields.len() { return Err(invalid("projection width mismatch")); From 64a9cb63d2d5a5863c409cdcad45fddb4c3df457 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 03:45:46 +0000 Subject: [PATCH 42/59] feat(physical): compile PromQL fallback subtrees from typed expressions Conflicts with earlier stack changes resolved to the integration tree: - crates/asap-physical-operators/src/physical_planner/promql_rows.rs: a7ff3ae Merge remote-tracking branch 'origin/feat/physical-compile-promql-fallback' into integration/planner-for-backend Co-Authored-By: Claude Opus 5.5 --- .../src/physical_planner/mod.rs | 55 ++- .../src/physical_planner/promql_fallback.rs | 403 ++++++++++++++++++ .../tests/physical_dag.rs | 4 +- .../tests/promql_fallback.rs | 385 +++++++++++++++++ 4 files changed, 845 insertions(+), 2 deletions(-) create mode 100644 crates/asap-physical-operators/src/physical_planner/promql_fallback.rs create mode 100644 crates/asap-physical-operators/tests/promql_fallback.rs diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index c7967288..21520150 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -32,6 +32,7 @@ fn invalid(message: impl Into) -> Error { pub type Source<'a> = Box + 'a>; pub mod precompute; +pub mod promql_fallback; pub mod promql_rows; pub mod promql_values; @@ -173,7 +174,19 @@ fn compile_internal( .or_default() .push(u64::from(edge.producer.0)); } - if sources.keys().any(|id| !nodes.contains_key(id)) { + let known = |id: &NodeId| { + nodes.contains_key(id) + || promql_fallback::raw_series_owner(*id).is_some_and(|owner| { + matches!( + nodes.get(&owner), + Some(PostAsapDagNode { + payload: Payload::Fallback { .. }, + .. + }) + ) + }) + }; + if !sources.keys().all(known) { return Err(invalid("source binding names an unknown node")); } let mut ordered = Vec::new(); @@ -228,6 +241,46 @@ fn compile_internal( inputs = vec![auxiliary]; schemas.truncate(1); } + // A consumed bare selector supplies raw range rows (e.g. to a + // per-entity summary), not an instant vector, so only its consumer computes. + let raw_rows = matches!( + &node.payload, + Payload::Fallback { + expression: QueryExpr::TimeRange { .. } + } + ) && dag.edges.iter().any(|e| u64::from(e.producer.0) == id); + if let (Payload::Fallback { expression }, false) = (&node.payload, raw_rows) { + let slot = promql_fallback::raw_series_input(id); + let (leaf, mut chain) = promql_fallback::lower(expression) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + let mut inputs = match (leaf, sources.remove(&slot)) { + (Some((_, schema)), Some(contract)) if contract.schema == schema => { + graph.add_input(slot, contract)?; + vec![slot] + } + (None, None) => vec![], + (Some(_), None) => { + return Err(invalid(format!( + "node {id}: PromQL fallback requires raw series input {slot}" + ))) + } + _ => { + return Err(invalid(format!( + "node {id}: raw series input differs from the selector schema" + ))) + } + }; + let last = chain + .pop() + .ok_or_else(|| invalid("empty PromQL lowering"))?; + for operator in chain { + graph.add(auxiliary, inputs, operator)?; + inputs = vec![auxiliary]; + auxiliary -= 1; + } + graph.add(id, inputs, last.with_output_schema(output)?)?; + continue; + } if let Payload::Value { operation: ValueOperation::MaintainPopulation { population }, } = &node.payload diff --git a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs new file mode 100644 index 00000000..78dd1b06 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs @@ -0,0 +1,403 @@ +//! Compile a retained PromQL subtree (`Fallback`) from its typed expression. +//! The deployment supplies the raw series of its one selector; the Planner +//! computes selection, range functions, subqueries and aggregation. +use super::*; +use crate::operators::SubquerySteps; +use planner_types::post_asap::execution_data_state::lift_plain; + +/// Input slot for the raw series read by Fallback node `node`'s selector. +/// The node's own ID names its computed output, so the raw rows need another. +pub fn raw_series_input(node: NodeId) -> NodeId { + node | (1 << 32) +} + +/// The Fallback node that owns a raw-series input slot. +pub(super) fn raw_series_owner(slot: NodeId) -> Option { + (slot >> 32 == 1).then_some(slot & u64::from(u32::MAX)) +} + +/// A selector expression and its raw-series row schema. +pub type Selector = (QueryExpr, Schema); + +/// The selector a Fallback expression reads, and the row schema of the raw +/// series the deployment supplies at [`raw_series_input`]. `None` means the +/// expression reads no series. The rows must cover the selector's window at +/// every evaluation instant; under a subquery `[R:S] offset O` that is +/// `(T - O - R - offset - range, T - O - offset]`. +pub fn raw_series(expression: &QueryExpr) -> Result, Error> { + Ok(lower(expression)?.0) +} + +/// Operators computing `expression`, in order, after its raw-series input. +pub(super) fn lower(expression: &QueryExpr) -> Result<(Option, Vec), Error> { + let mut chain = Chain::default(); + chain.value(expression)?; + Ok((chain.leaf, chain.operators)) +} + +fn declared(expression: &QueryExpr) -> Result { + let schema = expression + .output_schema() + .map_err(|error| invalid(error.to_string()))?; + Ok(Arc::new(lift_plain(&schema))) +} + +fn millis(duration: &std::time::Duration) -> Result { + i64::try_from(duration.as_millis()).map_err(|_| invalid("PromQL duration exceeds Int64")) +} + +/// `TimeRange { range, [TimeShift { offset }], Scan }`: range and offset. +fn selector(expression: &QueryExpr) -> Result<(i64, i64), Error> { + let QueryExpr::TimeRange { range, child } = expression else { + return Err(invalid("PromQL operand must be a series selector")); + }; + let (offset, scan) = match child.as_ref() { + QueryExpr::TimeShift { shift, child } if shift.at.is_none() => { + (shift.offset_ms, child.as_ref()) + } + scan => (0, scan), + }; + if !matches!(scan, QueryExpr::Scan { .. }) { + return Err(invalid( + "PromQL selector must read one scan; @ is unsupported", + )); + } + Ok((millis(range)?, offset)) +} + +#[derive(Default)] +struct Chain { + leaf: Option, + operators: Vec, +} + +impl Chain { + fn schema(&self) -> Result { + self.operators + .last() + .map(Operator::schema) + .or_else(|| self.leaf.as_ref().map(|(_, schema)| schema.clone())) + .ok_or_else(|| invalid("PromQL operator has no input")) + } + + /// Conform `operator` to the logical schema of the expression it computes. + fn push(&mut self, operator: Operator, logical: &QueryExpr) -> Result<(), Error> { + self.operators + .push(operator.with_output_schema(declared(logical)?)?); + Ok(()) + } + + fn read(&mut self, selector: &QueryExpr) -> Result { + let schema = declared(selector)?; + if !schema + .fields + .iter() + .any(|f| f.name == promql_rows::SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "PromQL fallback requires the complete series identity", + )); + } + if self + .leaf + .replace((selector.clone(), schema.clone())) + .is_some() + { + return Err(invalid("PromQL fallback reads more than one selector")); + } + Ok(schema) + } + + /// An instant vector, or a scalar for scalar-valued expressions. + fn value(&mut self, expression: &QueryExpr) -> Result<(), Error> { + match expression { + QueryExpr::TimeRange { .. } => { + let (range, offset) = selector(expression)?; + let input = self.read(expression)?; + self.push( + Operator::series_window(input, None, range, offset, None)?, + expression, + ) + } + QueryExpr::Aggregate { + reduction: planner_types::pre_asap::Reduction::PerEntity, + measures, + having: None, + child, + .. + } => { + let [function] = measures.as_slice() else { + return Err(invalid("range function requires one measure")); + }; + self.range_function(function, child, expression) + } + QueryExpr::Aggregate { + reduction: planner_types::pre_asap::Reduction::Reduce(keys), + measures, + having: None, + child, + .. + } => { + let [measure] = measures.as_slice() else { + return Err(invalid("vector aggregation requires one measure")); + }; + self.value(child)?; + self.aggregate(measure, keys, expression) + } + QueryExpr::Sort { + keys, + partition_by, + child, + } => { + self.value(child)?; + let input = self.schema()?; + let keys = keys + .iter() + .map(|key| match key.expr { + QueryExpr::Column(column) => Ok(SortKey { + column, + descending: !key.ascending, + nulls_first: key.nulls_first, + }), + _ => Err(invalid("sort key must be a column")), + }) + .collect::>()?; + let groups = groups(&input, partition_by)?; + self.push(Operator::sort(input, keys, groups)?, expression) + } + QueryExpr::Limit { n, offset, child } => { + self.value(child)?; + let input = self.schema()?; + // `topk by (...)` partitions through the Sort it limits. + let groups = match child.as_ref() { + QueryExpr::Sort { partition_by, .. } => groups(&input, partition_by)?, + _ => vec![], + }; + self.push( + Operator::limit(input, *n as u64, *offset as u64, groups)?, + expression, + ) + } + QueryExpr::BinaryOp { + op: planner_types::pre_asap::BinaryOpKind::Arithmetic(op), + lhs, + rhs, + vector_match: None, + } => { + let (vector, literal, literal_left) = match ( + row_values::scalar_literal(lhs), + row_values::scalar_literal(rhs), + ) { + (None, Some(value)) => (lhs, value, false), + (Some(value), None) => (rhs, value, true), + _ => { + return Err(invalid( + "PromQL fallback arithmetic requires one literal operand", + )) + } + }; + self.value(vector)?; + let input = self.schema()?; + let value = named_column(&input, &ColumnRef::SampleValue)?; + let literal = Expression::Literal { + value: crate::values::Value::Float64(literal), + dtype: DataType::Float64, + }; + let operator = planner_types::post_asap::BinaryOperator { + kind: planner_types::pre_asap::BinaryOpKind::Arithmetic(op.clone()), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }; + let (left, right) = if literal_left { + (literal, Expression::Column(value)) + } else { + (Expression::Column(value), literal) + }; + let columns = input + .fields + .iter() + .enumerate() + .map(|(i, field)| { + let expression = if i == value { + Expression::Binary { + operator: operator.clone(), + left: Box::new(left.clone()), + right: Box::new(right.clone()), + } + } else { + Expression::Column(i) + }; + (field.name.clone(), expression) + }) + .collect(); + self.push(Operator::project(input, columns)?, expression) + } + QueryExpr::PromqlScalarFromVector(child) => { + self.value(child)?; + let input = self.schema()?; + let value = named_column(&input, &ColumnRef::SampleValue)?; + self.push(Operator::vector_to_scalar(input, value)?, expression) + } + QueryExpr::PromqlVectorFromScalar(child) => { + self.value(child)?; + let input = self.schema()?; + self.operators + .push(Operator::scope_timestamp(input, declared(expression)?)?); + Ok(()) + } + QueryExpr::PromqlScalarBridge(_) => { + let value = row_values::scalar_literal(expression) + .ok_or_else(|| invalid("PromQL scalar must be a literal"))?; + self.push( + Operator::scalar(crate::values::Value::Float64(value), DataType::Float64)?, + expression, + ) + } + _ => Err(invalid("PromQL expression has no native fallback lowering")), + } + } + + /// `function(matrix)`, where the matrix is a range selector or a subquery. + fn range_function( + &mut self, + function: &AggIntent, + matrix: &QueryExpr, + logical: &QueryExpr, + ) -> Result<(), Error> { + let function = unbound(function)?; + let (subquery, offset) = match matrix { + QueryExpr::TimeShift { shift, child } if shift.at.is_none() => { + (child.as_ref(), shift.offset_ms) + } + other => (other, 0), + }; + let QueryExpr::PromqlSubquery { + range: outer, + resolution, + child, + } = subquery + else { + let (range, offset) = selector(matrix)?; + let input = self.read(matrix)?; + return self.push( + Operator::series_window(input, Some(function), range, offset, None)?, + logical, + ); + }; + let step = resolution.as_ref().ok_or_else(|| { + invalid("subquery resolution defaults to the deployment evaluation interval") + })?; + let steps = SubquerySteps { + range_ms: millis(outer)?, + step_ms: millis(step)?, + offset_ms: offset, + }; + // Each step evaluates a per-series selection or range function. + let (inner, selected) = match child.as_ref() { + QueryExpr::Aggregate { + reduction: planner_types::pre_asap::Reduction::PerEntity, + measures, + having: None, + child: selected, + .. + } => match measures.as_slice() { + [inner] => (Some(unbound(inner)?), selected.as_ref()), + _ => return Err(invalid("range function requires one measure")), + }, + selected => (None, selected), + }; + let (range, inner_offset) = selector(selected)?; + let input = self.read(selected)?; + self.push( + Operator::series_window(input, inner, range, inner_offset, Some(steps))?, + child, + )?; + let input = self.schema()?; + self.push( + Operator::series_window(input, Some(function), steps.range_ms, offset, None)?, + logical, + ) + } + + /// Cross-series aggregation. A global aggregate groups by one constant so + /// that no input series yields an empty vector, not one row. + fn aggregate( + &mut self, + measure: &AggIntent, + keys: &GroupKeys, + logical: &QueryExpr, + ) -> Result<(), Error> { + let mut input = self.schema()?; + let value = named_column(&input, &ColumnRef::SampleValue)?; + let reduction = match measure { + AggIntent::Sum { col: None } => Reduction::Sum(value), + AggIntent::Avg { col: None } => Reduction::Avg(value), + AggIntent::Min { col: None } => Reduction::Min(value), + AggIntent::Max { col: None } => Reduction::Max(value), + AggIntent::Count { .. } => Reduction::Count, + _ => return Err(invalid("vector aggregate has no native lowering")), + }; + let mut groups = groups(&input, keys)?; + let global = groups.is_empty(); + if global { + let mut columns = (0..input.fields.len()) + .map(|i| (input.fields[i].name.clone(), Expression::Column(i))) + .collect::>(); + columns.push(( + "$promql_global_group".into(), + Expression::Literal { + value: crate::values::Value::Utf8("".into()), + dtype: DataType::Utf8, + }, + )); + let project = Operator::project(input, columns)?; + input = project.schema(); + groups = vec![input.fields.len() - 1]; + self.operators.push(project); + } + let output = declared(logical)?; + let name = output + .fields + .last() + .ok_or_else(|| invalid("aggregate output lacks a value"))? + .name + .clone(); + let aggregate = Operator::aggregate(input, groups, vec![(name, reduction)])?; + let actual = aggregate.schema(); + self.operators.push(aggregate); + // Drop the constant group; convert counts where PromQL declares Float64. + let skip = usize::from(global); + let columns = actual.fields[skip..] + .iter() + .zip(&output.fields) + .enumerate() + .map(|(i, (field, declared))| { + let column = i + skip; + let expression = if field.dtype != declared.dtype { + Expression::ExactFloat64(column) + } else { + Expression::Column(column) + }; + (field.name.clone(), expression) + }) + .collect(); + self.push(Operator::project(actual, columns)?, logical) + } +} + +fn unbound(intent: &AggIntent) -> Result, Error> { + Ok(match intent { + AggIntent::Rate => AggIntent::Rate, + AggIntent::Increase => AggIntent::Increase, + AggIntent::Delta => AggIntent::Delta, + AggIntent::Count { accuracy } => AggIntent::Count { + accuracy: accuracy.clone(), + }, + AggIntent::Sum { col: None } => AggIntent::Sum { col: None }, + AggIntent::Avg { col: None } => AggIntent::Avg { col: None }, + AggIntent::Min { col: None } => AggIntent::Min { col: None }, + AggIntent::Max { col: None } => AggIntent::Max { col: None }, + _ => return Err(invalid("unsupported PromQL range function")), + }) +} diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 96e74e54..652880e0 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -546,7 +546,9 @@ fn bind_post_asap_before_execution() { }; let native = bind(&dag, sources(), &[1]).unwrap(); assert_eq!(floats(&run(&native, 1, query()), 0), vec![3.]); - assert!(bind(&dag, BTreeMap::new(), &[1]).is_err()); + // A literal Fallback needs no deployment input. + let literal = bind(&dag, BTreeMap::new(), &[1]).unwrap(); + assert_eq!(floats(&run(&literal, 1, query()), 0), vec![3.]); dag.nodes[1].payload = PostAsapOperatorPayload::Value { operation: ValueOperation::Extension { name: "unknown".into(), diff --git a/crates/asap-physical-operators/tests/promql_fallback.rs b/crates/asap-physical-operators/tests/promql_fallback.rs new file mode 100644 index 00000000..c31a57d4 --- /dev/null +++ b/crates/asap-physical-operators/tests/promql_fallback.rs @@ -0,0 +1,385 @@ +//! A retained PromQL subtree (`Fallback`) compiles from its typed expression. +//! The deployment supplies only its selector's raw series; expected values are +//! hand-computed with Prometheus semantics. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{compile, promql_fallback, promql_rows, CompiledPhysicalDag, InputContract}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{execution_data_state::lift_plain, *}, + pre_asap::QueryExpr, + types::AccuracyTarget, + workload::*, +}; +use std::{collections::BTreeMap, rc::Rc}; + +/// Bare selectors look back one ingestion interval: 60s. +fn lower(query: &str) -> QueryExpr { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(60_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let expression = asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0); + promql_rows::with_series_identity(&expression).unwrap() +} + +/// The whole query retained as one pre-ASAP node. +fn fallback_dag(expression: QueryExpr) -> PostAsapDag { + let schema = lift_plain(&expression.output_schema().unwrap()); + compile_post_asap_dag(&Rc::new(SummaryNode { + expr: SummaryExpr::KeepPreAsap(Rc::new(expression)), + schema, + guarantee: None, + })) + .unwrap() +} + +/// `(job, seconds, value)`; every sample belongs to metric `m`. +type Sample = (&'static str, i64, f64); + +fn compile_query(query: &str) -> Result { + let expression = lower(query); + let dag = fallback_dag(expression.clone()); + let root = u64::from(dag.root.0); + let inputs = promql_fallback::raw_series(&expression) + .map_err(|e| e.to_string())? + .map(|(_, schema)| { + ( + promql_fallback::raw_series_input(root), + InputContract::bounded(schema), + ) + }) + .into_iter() + .collect(); + let program = compile(&dag, inputs, &[root]).map_err(|e| e.to_string())?; + Ok(serde_json::from_slice(&serde_json::to_vec(&program).unwrap()).unwrap()) +} + +/// Evaluate at `at` seconds; returns `(job or "", timestamp ms, value)` rows in order. +fn run(query: &str, samples: &[Sample], at: i64) -> Result, String> { + let expression = lower(query); + let program = compile_query(query)?; + let mut sources = BTreeMap::new(); + if let Some((_, schema)) = promql_fallback::raw_series(&expression).unwrap() { + let rows = samples + .iter() + .map(|(job, seconds, value)| { + let labels = BTreeMap::from([ + ("__name__".to_string(), "m".to_string()), + ("job".into(), job.to_string()), + ]); + promql_rows::series_row(&schema, &labels, seconds * 1000, *value).unwrap() + }) + .collect(); + let batch = Batch::try_new(schema.clone(), rows).unwrap(); + sources.insert( + promql_fallback::raw_series_input(program.roots()[0]), + Box::new(Operator::source(schema, vec![batch]).unwrap()) as _, + ); + } + let graph = program.instantiate(sources).map_err(|e| e.to_string())?; + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: at * 1000, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = graph + .execute(program.roots(), context) + .map_err(|e| e.to_string())? + .remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + let batch = batch.map_err(|e| e.to_string())?; + let schema = batch.schema().clone(); + for row in batch.rows() { + let mut job = String::new(); + let mut time = -1; + let mut value = None; + for (field, cell) in schema.fields.iter().zip(row) { + match (field.name.as_str(), cell) { + (promql_rows::SERIES_IDENTITY_COLUMN, Value::Utf8(id)) => { + job = promql_rows::decode_series_identity(id).unwrap()["job"].clone() + } + ("job", Value::Utf8(label)) => job = label.to_string(), + (_, Value::Timestamp(t)) => time = *t, + (_, Value::Float64(v)) => value = Some(*v), + (_, Value::Int64(v)) => value = Some(*v as f64), + other => return Err(format!("unexpected cell {other:?}")), + } + } + rows.push((job, time, value.ok_or("missing value")?)); + } + } + Ok(rows) + }) +} + +fn values(query: &str, samples: &[Sample], at: i64) -> Vec<(String, f64)> { + run(query, samples, at) + .unwrap_or_else(|e| panic!("{query}: {e}")) + .into_iter() + .map(|(job, _, value)| (job, value)) + .collect() +} + +fn one(query: &str, samples: &[Sample], at: i64) -> f64 { + match values(query, samples, at).as_slice() { + [(_, value)] => *value, + other => panic!("{query}: expected one sample, got {other:?}"), + } +} + +const COUNTER: &[Sample] = &[ + ("a", 60, 10.), + ("a", 120, 20.), + ("a", 180, 5.), + ("a", 240, 15.), +]; + +// rate/increase correct the reset at 180s and extrapolate half an interval at +// most; delta treats the same samples as a gauge. +#[test] +fn range_functions_follow_prometheus_extrapolation_and_resets() { + // Reset-corrected increase is 25 over 180s of samples; 60s on each side extrapolates. + let increase = 25. * (180. + 60. + 60.) / 180.; + assert!((one("increase(m[5m])", COUNTER, 300) - increase).abs() < 1e-9); + assert!((one("rate(m[5m])", COUNTER, 300) - increase / 300.).abs() < 1e-12); + let delta = 5. * (180. + 60. + 60.) / 180.; + assert!((one("delta(m[5m])", COUNTER, 300) - delta).abs() < 1e-9); + // Fewer than two samples yield no rate. + assert!(values("rate(m[2m])", COUNTER, 300).is_empty()); + for (query, expected) in [ + ("sum_over_time(m[5m])", 50.), + ("avg_over_time(m[5m])", 12.5), + ("min_over_time(m[5m])", 5.), + ("max_over_time(m[5m])", 20.), + ("count_over_time(m[5m])", 4.), + ] { + assert_eq!(one(query, COUNTER, 300), expected, "{query}"); + } +} + +// Ranges are left-open: a sample at `t - range` is excluded, one at `t` is included. +#[test] +fn ranges_exclude_their_start_and_offsets_shift_them() { + let samples = &[("a", 60, 1.), ("a", 90, 1.), ("a", 120, 1.), ("a", 150, 1.)]; + assert_eq!(one("count_over_time(m[1m])", samples, 120), 2.); + // offset 1m reads (60s, 120s] at 180s; output keeps the evaluation time. + let rows = run("count_over_time(m[1m] offset 1m)", samples, 180).unwrap(); + assert_eq!(rows, vec![("a".into(), 180_000, 2.)]); +} + +// A bare selector takes the latest sample within the lookback; a stale marker +// hides the series rather than exposing an older value. +#[test] +fn instant_selection_uses_lookback_and_stale_markers() { + let stale = f64::from_bits(0x7ff0_0000_0000_0002); + let samples = &[("a", 0, 1.), ("a", 30, 2.), ("b", 30, 3.), ("b", 50, stale)]; + assert_eq!(values("m", samples, 60), vec![("a".into(), 2.)]); + // The lookback (30s, 90s] excludes the sample at 30s. + assert!(values("m", samples, 90).is_empty()); + // Range functions skip stale markers. + assert_eq!( + values("sum_over_time(m[1m])", samples, 60), + vec![("a".into(), 2.), ("b".into(), 3.)] + ); +} + +// NaN samples follow Prometheus: min/max skip them, sums propagate them. +#[test] +fn nan_samples() { + let samples = &[("a", 10, f64::NAN), ("a", 20, 3.), ("a", 30, 1.)]; + assert_eq!(one("max_over_time(m[1m])", samples, 60), 3.); + assert_eq!(one("min_over_time(m[1m])", samples, 60), 1.); + assert!(one("sum_over_time(m[1m])", samples, 60).is_nan()); +} + +// Aggregation over no series is an empty vector, not one zero or null row; +// sort_desc orders the selected series. +#[test] +fn cross_series_aggregates_and_empty_inputs() { + let samples = &[("a", 50, 1.), ("b", 40, 2.), ("b", 55, 4.)]; + assert_eq!(values("sum(m)", samples, 60), vec![(String::new(), 5.)]); + assert_eq!(values("count(m)", samples, 60), vec![(String::new(), 2.)]); + assert_eq!( + values("max by (job) (m)", samples, 60), + vec![("a".into(), 1.), ("b".into(), 4.)] + ); + assert_eq!( + values("sort_desc(m)", samples, 60), + vec![("b".into(), 4.), ("a".into(), 1.)] + ); + // topk by (job) keeps the top series of each job, not one overall. + let jobs = &[("a", 50, 1.), ("b", 50, 2.)]; + let mut top = values("topk by (job) (1, m)", jobs, 60); + top.sort_by(|x, y| x.0.cmp(&y.0)); + assert_eq!(top, vec![("a".into(), 1.), ("b".into(), 2.)]); + assert_eq!(values("topk(1, m)", jobs, 60), vec![("b".into(), 2.)]); + for query in ["sum(m)", "count(m)", "max(m)", "sum by (job) (rate(m[5m]))"] { + assert!(values(query, &[], 60).is_empty(), "{query}"); + } +} + +// scalar() is the single series' value and NaN otherwise; vector() needs no input. +#[test] +fn scalar_and_vector_bridges() { + assert_eq!(one("scalar(m)", &[("a", 50, 7.)], 60), 7.); + assert!(one("scalar(m)", &[("a", 50, 7.), ("b", 50, 8.)], 60).is_nan()); + assert!(one("scalar(m)", &[], 60).is_nan()); + assert_eq!( + run("vector(3)", &[], 60).unwrap(), + vec![(String::new(), 60_000, 3.)] + ); + assert_eq!( + values("2 - m", &[("a", 50, 7.)], 60), + vec![("a".into(), -5.)] + ); + assert_eq!( + values("m * 2", &[("a", 50, 7.)], 60), + vec![("a".into(), 14.)] + ); +} + +// Subquery steps are absolute multiples of the resolution in the left-open +// range; each step evaluates the operand, and the outer function reduces them. +#[test] +fn subqueries_evaluate_their_operand_on_the_aligned_grid() { + // Steps 60..300: selections 1, 7, 3, (none at 240s), 4. + let samples = &[ + ("a", 50, 1.), + ("a", 110, 7.), + ("a", 170, 3.), + ("a", 290, 4.), + ]; + assert_eq!(one("max_over_time(m[5m:1m])", samples, 300), 7.); + assert_eq!(one("count_over_time(m[5m:1m])", samples, 300), 4.); + // At 190s the steps are 60, 120, 180 (not 70, 130, 190): counts 1 + 2 + 2. + let samples = &[("a", 30, 1.), ("a", 90, 1.), ("a", 150, 1.), ("a", 185, 1.)]; + assert_eq!( + one("sum_over_time(count_over_time(m[2m])[3m:1m])", samples, 190), + 5. + ); + // offset 1m moves the grid to (-50s, 130s]: steps 0, 60, 120 count 0 + 1 + 2. + assert_eq!( + one( + "sum_over_time(count_over_time(m[2m])[3m:1m] offset 1m)", + samples, + 190 + ), + 3. + ); + assert!(compile_query("max_over_time(m[5m:1m] @ 100)").is_err()); +} + +// Subquery work is bounded by the query: at most 100000 steps. +#[test] +fn dense_subquery_grids_are_rejected() { + assert!(compile_query("max_over_time(m[100s:1ms])").is_ok()); + assert!(compile_query("max_over_time(m[30d:1ms])").is_err()); +} + +// The deployment must supply the selector's raw rows under the documented slot +// with the exact selector schema; unsupported shapes stay rejected. +#[test] +fn raw_series_contract_is_explicit() { + let expression = lower("rate(m[5m])"); + let dag = fallback_dag(expression.clone()); + let root = u64::from(dag.root.0); + let (selector, schema) = promql_fallback::raw_series(&expression).unwrap().unwrap(); + assert!(matches!(selector, QueryExpr::TimeRange { .. })); + let missing = compile(&dag, BTreeMap::new(), &[root]).err().unwrap(); + assert!(missing.to_string().contains("raw series input")); + let mut wrong = (*schema).clone(); + wrong.fields.pop(); + let wrong = compile( + &dag, + BTreeMap::from([( + promql_fallback::raw_series_input(root), + InputContract::bounded(std::sync::Arc::new(wrong)), + )]), + &[root], + ); + assert!(wrong.is_err()); + // A consumed bare selector is raw range rows for its consumer; it is not + // turned into instant selection. + let selector = lower("m"); + let schema = lift_plain(&selector.output_schema().unwrap()); + let node = |id, payload| PostAsapDagNode { + id: PostAsapNodeId(id), + payload, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: schema.clone(), + guarantee: None, + }; + let consumed = PostAsapDag { + nodes: vec![ + node( + 0, + PostAsapOperatorPayload::Fallback { + expression: selector.clone(), + }, + ), + node( + 1, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Limit { + n: 1, + offset: 0, + partition_by: Default::default(), + }, + }, + ), + ], + edges: vec![PostAsapDagEdge { + producer: PostAsapNodeId(0), + consumer: PostAsapNodeId(1), + role: EdgeRole::Input, + intermediate_schema: schema.clone(), + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }], + root: PostAsapNodeId(1), + }; + let raw = promql_fallback::raw_series(&selector).unwrap().unwrap().1; + assert!(compile( + &consumed, + BTreeMap::from([( + promql_fallback::raw_series_input(0), + InputContract::bounded(raw) + )]), + &[1], + ) + .is_err()); + // Implicit subquery resolution belongs to the deployment's evaluation interval. + assert!(compile_query("max_over_time(m[5m:])").is_err()); +} From fd0bb0a1bbb163800df6edfe949d8dfcd1fd5b93 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 03:45:46 +0000 Subject: [PATCH 43/59] docs: record PromQL fallback compile coverage Co-Authored-By: Claude Opus 5.5 --- .../develop_docs/physical-compile-coverage.md | 42 +++++++++++++++++-- 1 file changed, 39 insertions(+), 3 deletions(-) diff --git a/docs/develop_docs/physical-compile-coverage.md b/docs/develop_docs/physical-compile-coverage.md index 5db0edce..ce8c3817 100644 --- a/docs/develop_docs/physical-compile-coverage.md +++ b/docs/develop_docs/physical-compile-coverage.md @@ -74,13 +74,49 @@ Totals at #475: 11 Supported, 4 Partial, 14 Missing, 2 Backend. Totals after this change: 17 Supported, 4 Partial, 8 Missing, 2 Backend. +## Covered by PromQL fallback compilation + +`compile` now lowers a `Fallback{QueryExpr}` node from its typed expression. +The expression must read at most one selector, realized with +`promql_rows::with_series_identity`. The deployment supplies that selector's raw +rows at `promql_fallback::raw_series_input(node)`, with the schema returned by +`promql_fallback::raw_series`. The node's own ID still names its output, so a +deployment may instead supply the whole result, for example from an external +exact engine. A Fallback that reads no selector, such as `vector(1)`, needs no +input. + +`Operator::series_window` evaluates each series at the query time, or at each +subquery step, over the left-open window `(t - offset - range, t - offset]`. +Instant selection takes the latest sample and omits the series if that sample is +a stale marker. Range functions ignore stale markers. The raw rows must cover +every window the node evaluates; `raw_series` documents the subquery extent. +A subquery has at most 100000 steps. Output rows keep the full series identity; +the query adapter still applies PromQL's metric-name rules. A bare selector +consumed by another node, such as a per-entity summary, remains raw range rows +and is not compiled as instant selection. + +| Row | Change | +|---|---| +| 1 | Supported shapes: selectors with `offset`; `rate`, `increase`, `delta`, and `sum`/`avg`/`min`/`max`/`count_over_time`; `by` aggregation; `sort`; `topk`/`limit`; `scalar()`; `vector(literal)`; arithmetic with one literal. Now Partial. | +| 9 | `scalar()` compiles to `VectorToScalar`. | +| 11 | Range functions over the raw selector rows. | +| 12 | `f(sel[R:S])` and `f(g(sel[r])[R:S])`, with subquery `offset`, evaluate on the aligned step grid. The frontend now retains subquery `offset`/`@` as a `TimeShift`; it previously dropped them. Now Partial. | + +Totals after this change: 19 Supported, 5 Partial, 5 Missing, 2 Backend. + ## Remaining In order of backend usage: -1. Row 1 and rows 9–12: lower PromQL-shaped `Fallback{QueryExpr}` subtrees - (range functions over matrices, `scalar()`, `histogram_quantile`, `sort`, - subquery grids). After that, rows 28 and 30 can be deleted from the backend. +1. Rows 1, 10, and 12, the remaining `Fallback` shapes: + - Subtrees with more than one selector, such as vector-vector binaries. + - `histogram_quantile`: the frontend emits `by ()` grouping with no + output labels. The IR must group `without (le)` and keep the labels. + - Subquery operands other than one per-series function; implicit + subquery resolution, which is a deployment default. + - `@` on selectors and subqueries; `without` grouping; `irate`, + `changes`, and other range functions. + After these shapes are covered, the backend can delete rows 28 and 30. 2. Row 7: comparison filters and `bool` comparisons. This needs `return_bool` in the `Binary` payload. `compile` currently rejects comparisons. 3. Row 5 for per-series rows: matching needs a metric-name-free series From 5f2a37cc48cb794a01d12538a08cf9d53c55a3fa Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 04:16:40 +0000 Subject: [PATCH 44/59] feat(physical): compile irate, idelta, changes, resets, last and quantile over time Per-series windows evaluate these PromQL range functions with Prometheus semantics over fresh (non-stale) samples. Co-Authored-By: Claude Opus 5.5 --- .../src/operators/aggregate/mod.rs | 2 +- .../src/operators/aggregate/temporal.rs | 29 +++++++++++++ .../src/operators/series_window.rs | 6 +++ .../src/physical_planner/promql_fallback.rs | 14 +++++++ .../tests/promql_fallback.rs | 42 +++++++++++++++++++ 5 files changed, 92 insertions(+), 1 deletion(-) diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs index 37b4e96d..d929fc07 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/mod.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -211,7 +211,7 @@ async fn reduce( // Matches Prometheus `quantile`: NaN for no values, ±Inf outside [0, 1], // and NaN samples ordered first. -fn quantile(q: f64, mut values: Vec) -> f64 { +pub(super) fn quantile(q: f64, mut values: Vec) -> f64 { if values.is_empty() { return f64::NAN; } diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs index c14abcbb..3ac0a3d1 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -110,6 +110,35 @@ pub(in crate::operators) fn window_value( a } }))), + AggIntent::IRate | AggIntent::IDelta => { + let [.., (t0, v0), (t1, v1)] = points else { + return Ok(None); + }; + let rate = matches!(intent, AggIntent::IRate); + // A counter reset makes the last value the increase. + let delta = if rate && v1 < v0 { *v1 } else { v1 - v0 }; + match (rate, t1 - t0) { + (_, 0) => None, + (true, interval) => Some(Value::Float64(delta / (interval as f64 / 1000.))), + (false, _) => Some(Value::Float64(delta)), + } + } + AggIntent::Changes | AggIntent::Resets => { + let changed = |(a, b): (f64, f64)| match intent { + AggIntent::Changes => a != b && !(a.is_nan() && b.is_nan()), + _ => b < a, + }; + let count = points + .windows(2) + .filter(|pair| changed((pair[0].1, pair[1].1))) + .count(); + Some(Value::Float64(count as f64)) + } + AggIntent::LastOverTime => points.last().map(|p| Value::Float64(p.1)), + AggIntent::Quantile { col: None, q, .. } => Some(Value::Float64(super::quantile( + *q, + points.iter().map(|p| p.1).collect(), + ))), _ => return Err(Error::Invalid("unsupported temporal intent".into())), }) } diff --git a/crates/asap-physical-operators/src/operators/series_window.rs b/crates/asap-physical-operators/src/operators/series_window.rs index ad56db0b..38189aa6 100644 --- a/crates/asap-physical-operators/src/operators/series_window.rs +++ b/crates/asap-physical-operators/src/operators/series_window.rs @@ -66,6 +66,12 @@ impl Operator { | AggIntent::Avg { col: None } | AggIntent::Min { col: None } | AggIntent::Max { col: None } + | AggIntent::IRate + | AggIntent::IDelta + | AggIntent::Changes + | AggIntent::Resets + | AggIntent::LastOverTime + | AggIntent::Quantile { col: None, .. } ) ) { return Err(invalid("unsupported PromQL range function")); diff --git a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs index 78dd1b06..ad950780 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs @@ -398,6 +398,20 @@ fn unbound(intent: &AggIntent) -> Result, Error> { AggIntent::Avg { col: None } => AggIntent::Avg { col: None }, AggIntent::Min { col: None } => AggIntent::Min { col: None }, AggIntent::Max { col: None } => AggIntent::Max { col: None }, + AggIntent::IRate => AggIntent::IRate, + AggIntent::IDelta => AggIntent::IDelta, + AggIntent::Changes => AggIntent::Changes, + AggIntent::Resets => AggIntent::Resets, + AggIntent::LastOverTime => AggIntent::LastOverTime, + AggIntent::Quantile { + col: None, + q, + accuracy, + } => AggIntent::Quantile { + col: None, + q: *q, + accuracy: accuracy.clone(), + }, _ => return Err(invalid("unsupported PromQL range function")), }) } diff --git a/crates/asap-physical-operators/tests/promql_fallback.rs b/crates/asap-physical-operators/tests/promql_fallback.rs index c31a57d4..e5c7fb76 100644 --- a/crates/asap-physical-operators/tests/promql_fallback.rs +++ b/crates/asap-physical-operators/tests/promql_fallback.rs @@ -383,3 +383,45 @@ fn raw_series_contract_is_explicit() { // Implicit subquery resolution belongs to the deployment's evaluation interval. assert!(compile_query("max_over_time(m[5m:])").is_err()); } + +// irate/idelta use the last two samples (irate corrects a reset to the last +// value); changes/resets count value changes and decreases; quantile_over_time +// interpolates; all skip stale markers. +#[test] +fn instant_and_counting_range_functions() { + // COUNTER in (0s, 300s]: 10, 20, 5, 15. + assert!((one("irate(m[5m])", COUNTER, 300) - 10. / 60.).abs() < 1e-12); + assert_eq!(one("idelta(m[5m])", COUNTER, 300), 10.); + // At 200s the last pair 20 -> 5 is a reset: irate uses 5 as the increase. + assert!((one("irate(m[5m])", COUNTER, 200) - 5. / 60.).abs() < 1e-12); + assert_eq!(one("idelta(m[5m])", COUNTER, 200), -15.); + assert!(values("irate(m[1m])", COUNTER, 300).is_empty()); + assert_eq!(one("changes(m[5m])", COUNTER, 300), 3.); + assert_eq!(one("resets(m[5m])", COUNTER, 300), 1.); + assert_eq!(one("changes(m[2m])", COUNTER, 300), 0.); + // NaN to NaN is not a change; any other transition involving NaN is. + let flat = &[ + ("a", 10, 1.), + ("a", 20, 1.), + ("a", 30, 2.), + ("a", 40, f64::NAN), + ("a", 50, f64::NAN), + ("a", 55, 1.), + ]; + assert_eq!(one("changes(m[1m])", flat, 60), 3.); + let stale = f64::from_bits(0x7ff0_0000_0000_0002); + let ended = &[("a", 240, 15.), ("a", 250, stale)]; + assert_eq!(one("last_over_time(m[5m])", ended, 300), 15.); + assert!(values("m", ended, 300).is_empty()); + // Sorted 5, 10, 15, 20: rank 1.5 and 0.75; outside [0, 1] is +-Inf. + assert_eq!(one("quantile_over_time(0.5, m[5m])", COUNTER, 300), 12.5); + assert_eq!(one("quantile_over_time(0.25, m[5m])", COUNTER, 300), 8.75); + assert_eq!( + one("quantile_over_time(2, m[5m])", COUNTER, 300), + f64::INFINITY + ); + assert_eq!( + one("quantile_over_time(-1, m[5m])", COUNTER, 300), + f64::NEG_INFINITY + ); +} From bd0b4cbd8e5e7eb3f1dc6b89f538a28e1bba5c4f Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 04:16:46 +0000 Subject: [PATCH 45/59] feat(physical): compile multi-selector PromQL fallbacks with matching, without and @ Each Fallback selector reads its own raw-series slot. Vector-vector arithmetic uses PromQL one-to-one matching with on/ignoring via new series_labels and series_binary operators; without grouping rewrites the series identity; @ fixes selector and subquery evaluation. Conflicts with earlier stack changes resolved to the integration tree: - crates/asap-physical-operators/src/operators/mod.rs: 9a13ae4 integrate: extend asap-types series identity with #486/#487 shapes - crates/asap-physical-operators/src/physical_planner/promql_rows.rs: 9a13ae4 integrate: extend asap-types series identity with #486/#487 shapes Co-Authored-By: Claude Opus 5.5 --- .../src/operators/mod.rs | 16 + .../src/operators/series_labels.rs | 204 ++++++++++++ .../src/operators/series_window.rs | 21 +- .../src/operators/unchecked.rs | 8 + .../src/physical_planner/mod.rs | 63 ++-- .../src/physical_planner/promql_fallback.rs | 302 ++++++++++++------ .../tests/promql_fallback.rs | 251 +++++++++++++-- 7 files changed, 709 insertions(+), 156 deletions(-) create mode 100644 crates/asap-physical-operators/src/operators/series_labels.rs diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index cb7c5433..196e9e8a 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -23,6 +23,7 @@ mod joins; mod limit; mod projection; mod scope_timestamp; +mod series_labels; mod series_window; mod sort; mod source; @@ -74,8 +75,16 @@ enum Kind { value: usize, range_ms: i64, offset_ms: i64, + at_ms: Option, steps: Option, }, + SeriesLabels { + kind: planner_types::pre_asap::VectorMatchKind, + labels: Vec, + }, + SeriesBinary { + operator: planner_types::post_asap::BinaryOperator, + }, Project(Vec), Filter(Expression), Limit { @@ -248,6 +257,8 @@ impl PhysicalOperator for Operator { | Kind::HistogramQuantile | Kind::CurrentSeries { .. } | Kind::SeriesWindow { .. } + | Kind::SeriesLabels { .. } + | Kind::SeriesBinary { .. } | Kind::Aggregate { .. } | Kind::Window { .. } | Kind::Join { .. } @@ -291,6 +302,8 @@ impl PhysicalOperator for Operator { Kind::RangeWindow { .. } => "RangeWindow", Kind::HistogramQuantile => "HistogramQuantile", Kind::SeriesWindow { .. } => "SeriesWindow", + Kind::SeriesLabels { .. } => "SeriesLabels", + Kind::SeriesBinary { .. } => "SeriesBinary", Kind::Project(_) => "Project", Kind::Filter(_) => "Filter", Kind::Limit { .. } => "Limit", @@ -337,6 +350,9 @@ impl PhysicalOperator for Operator { Kind::CurrentSeries { .. } => current_series::execute(self, inputs, context), Kind::ScopeTimestamp { .. } => scope_timestamp::execute(self, inputs, context), Kind::SeriesWindow { .. } => series_window::execute(self, inputs, context), + Kind::SeriesLabels { .. } | Kind::SeriesBinary { .. } => { + series_labels::execute(self, inputs, context) + } Kind::Filter(_) => filter::execute(self, inputs, context), Kind::Limit { .. } => limit::execute(self, inputs, context), Kind::Sort { .. } => sort::execute(self, inputs, context), diff --git a/crates/asap-physical-operators/src/operators/series_labels.rs b/crates/asap-physical-operators/src/operators/series_labels.rs new file mode 100644 index 00000000..be248319 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/series_labels.rs @@ -0,0 +1,204 @@ +//! PromQL label-set rewriting and one-to-one vector matching over rows that +//! carry a series identity or plain label columns. +use super::*; +use planner_types::{ + post_asap::BinaryOperator, + pre_asap::{schema::PROMQL_SERIES_IDENTITY, BinaryOpKind, VectorMatchKind}, +}; + +type Labels = BTreeMap; + +/// Where a row's label set lives: the encoded series identity when present, +/// otherwise the non-empty Utf8 label columns. The one Float64 column is the +/// sample value, whatever an aggregate named it. +#[derive(Clone, Debug)] +struct Layout { + identity: Option, + labels: Vec, + value: usize, +} + +fn layout(input: &Schema) -> Result { + let mut identity = None; + let mut labels = Vec::new(); + let mut value = None; + for (i, field) in input.fields.iter().enumerate() { + match plain(input, i)? { + (DataType::Utf8, false) if field.name == PROMQL_SERIES_IDENTITY => identity = Some(i), + (DataType::Utf8, _) if field.name != PROMQL_SERIES_IDENTITY => labels.push(i), + (DataType::Float64, false) if value.is_none() => value = Some(i), + (DataType::Timestamp, false) if input.time_index == Some(i) => {} + _ => return Err(invalid("PromQL vector requires labels, time and one value")), + } + } + Ok(Layout { + identity, + labels, + value: value.ok_or_else(|| invalid("PromQL vector requires one value"))?, + }) +} + +impl Layout { + fn read(&self, input: &Schema, row: &[Value]) -> Result { + if let Some(i) = self.identity { + let Value::Utf8(encoded) = &row[i] else { + return Err(invalid("series identity must be Utf8")); + }; + let mut labels: Labels = + serde_json::from_str(encoded).map_err(|e| invalid(&e.to_string()))?; + // PromQL treats an empty label value as an absent label. + labels.retain(|_, v| !v.is_empty()); + return Ok(labels); + } + let mut labels = Labels::new(); + for &i in &self.labels { + match &row[i] { + Value::Utf8(v) if !v.is_empty() => { + labels.insert(input.fields[i].name.clone(), v.to_string()); + } + Value::Utf8(_) | Value::Null => {} + _ => return Err(invalid("label must be Utf8")), + } + } + Ok(labels) + } + + /// Replace the row's labels; a label column absent from `labels` is empty. + fn write(&self, input: &Schema, row: &mut [Value], labels: &Labels) -> Result<(), Error> { + if let Some(i) = self.identity { + let encoded = serde_json::to_string(labels).map_err(|e| invalid(&e.to_string()))?; + row[i] = Value::Utf8(encoded.into()); + } + for &i in &self.labels { + let value = labels.get(&input.fields[i].name).map_or("", String::as_str); + row[i] = Value::Utf8(value.into()); + } + Ok(()) + } +} + +impl Operator { + /// Rewrite each row's label set to PromQL's matching labels: `On` keeps + /// only `labels`; `Ignoring` drops `labels` and the metric name. + pub fn series_labels( + input: Schema, + kind: VectorMatchKind, + labels: Vec, + ) -> Result { + layout(&input)?; + Ok(Self { + kind: Kind::SeriesLabels { kind, labels }, + inputs: vec![input.clone()], + output: input, + }) + } + + /// PromQL one-to-one arithmetic between rows with equal label sets. The + /// result keeps the left row, without the metric name. + pub fn series_binary( + left: Schema, + right: Schema, + operator: BinaryOperator, + ) -> Result { + layout(&left)?; + layout(&right)?; + if !matches!(operator.kind, BinaryOpKind::Arithmetic(_)) || operator.vector_match.is_some() + { + return Err(invalid("series binary requires unmatched arithmetic")); + } + Ok(Self { + kind: Kind::SeriesBinary { operator }, + inputs: vec![left.clone(), right], + output: left, + }) + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let left_layout = layout(&operator.inputs[0])?; + let right = match operator.kind { + Kind::SeriesBinary { .. } => { + Some(inputs.pop().ok_or_else(|| invalid("missing right input"))?) + } + _ => None, + }; + let left = inputs.pop().ok_or_else(|| invalid("missing left input"))?; + Ok(futures::stream::once(async move { + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + let (rows, _memory) = collect_rows(left, &context).await?; + let mut result = Vec::new(); + match (&operator.kind, right) { + (Kind::SeriesLabels { kind, labels }, None) => { + for mut row in rows { + work.checkpoint().await?; + let mut set = left_layout.read(&output, &row)?; + match kind { + VectorMatchKind::On => set.retain(|k, _| labels.contains(k)), + VectorMatchKind::Ignoring => { + set.retain(|k, _| k != "__name__" && !labels.contains(k)) + } + } + left_layout.write(&output, &mut row, &set)?; + workspace.grow(row_bytes(&row))?; + result.push(row); + } + } + (Kind::SeriesBinary { operator: binary }, Some(right)) => { + let right_schema = &operator.inputs[1]; + let right_layout = layout(right_schema)?; + let (right, _right_memory) = collect_rows(right, &context).await?; + // Prometheus returns before matching when either side is empty. + if rows.is_empty() || right.is_empty() { + return Batch::try_new(output.clone(), vec![]); + } + let mut matches = BTreeMap::new(); + for row in &right { + work.checkpoint().await?; + let set = right_layout.read(right_schema, row)?; + workspace.grow(set.iter().map(|(k, v)| 64 + k.len() + v.len()).sum())?; + let Value::Float64(value) = row[right_layout.value] else { + return Err(invalid("vector value must be Float64")); + }; + // Prometheus rejects a duplicate on the one side. + if matches.insert(set, (value, false)).is_some() { + return Err(invalid( + "duplicate series for a match group on the right-hand side", + )); + } + } + for mut row in rows { + work.checkpoint().await?; + let mut set = left_layout.read(&output, &row)?; + let Some((value, matched)) = matches.get_mut(&set) else { + continue; + }; + // Only a left duplicate that finds a match is ambiguous. + if std::mem::replace(matched, true) { + return Err(invalid( + "many-to-one matching must be explicit (group_left/group_right)", + )); + } + let Value::Float64(left_value) = row[left_layout.value] else { + return Err(invalid("vector value must be Float64")); + }; + row[left_layout.value] = crate::expressions::arithmetic::evaluate_binary( + binary, left_value, *value, + )?; + set.remove("__name__"); + left_layout.write(&output, &mut row, &set)?; + workspace.grow(row_bytes(&row))?; + result.push(row); + } + } + _ => return Err(invalid("series label operator inputs mismatch")), + } + Batch::try_new(output.clone(), result) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/series_window.rs b/crates/asap-physical-operators/src/operators/series_window.rs index 38189aa6..20ab1aed 100644 --- a/crates/asap-physical-operators/src/operators/series_window.rs +++ b/crates/asap-physical-operators/src/operators/series_window.rs @@ -3,12 +3,14 @@ use super::*; use planner_types::pre_asap::AggIntent; /// A PromQL subquery grid: every multiple of `step_ms` in -/// `(T - offset_ms - range_ms, T - offset_ms]`, where `T` is the query time. +/// `(T - offset_ms - range_ms, T - offset_ms]`. `T` is `at_ms` (the subquery's +/// `@`) when present, else the query time. #[derive(Clone, Copy, Debug, PartialEq, Eq, serde::Serialize, serde::Deserialize)] pub struct SubquerySteps { pub range_ms: i64, pub step_ms: i64, pub offset_ms: i64, + pub at_ms: Option, } const STALE_MARKER: u64 = 0x7ff0_0000_0000_0002; @@ -16,8 +18,9 @@ const MAX_SUBQUERY_STEPS: i64 = 100_000; impl Operator { /// Evaluate each series at instant `t` over its samples in - /// `(t - offset_ms - range_ms, t - offset_ms]`. `t` is the query time, or - /// each step of `steps`. `function: None` is instant selection: the latest + /// `(e - offset_ms - range_ms, e - offset_ms]`. `t` is the query time, or + /// each step of `steps`; `e` is `at_ms` (the selector's `@`) when present, + /// else `t`. `function: None` is instant selection: the latest /// sample, absent if it is a stale marker. Range functions ignore stale /// markers. A series is every column except the time and `value` columns. /// Output rows keep the input schema, with time `t` and the result value. @@ -26,6 +29,7 @@ impl Operator { function: Option>, range_ms: i64, offset_ms: i64, + at_ms: Option, steps: Option, ) -> Result { let coordinate = input @@ -83,6 +87,7 @@ impl Operator { value: *value, range_ms, offset_ms, + at_ms, steps, }, inputs: vec![input.clone()], @@ -107,7 +112,9 @@ fn evaluation_times( return Ok((evaluation_time_ms, evaluation_time_ms, 1)); }; let overflow = || invalid("subquery grid overflows"); - let end = evaluation_time_ms + let end = steps + .at_ms + .unwrap_or(evaluation_time_ms) .checked_sub(steps.offset_ms) .ok_or_else(overflow)?; let start = end.checked_sub(steps.range_ms).ok_or_else(overflow)?; @@ -128,6 +135,7 @@ pub(super) fn execute<'a>( value, range_ms, offset_ms, + at_ms, steps, } = &operator.kind else { @@ -175,7 +183,10 @@ pub(super) fn execute<'a>( while time <= last && !series.is_empty() { work.checkpoint().await?; let overflow = || invalid("series window overflows"); - let end = time.checked_sub(*offset_ms).ok_or_else(overflow)?; + let end = at_ms + .unwrap_or(time) + .checked_sub(*offset_ms) + .ok_or_else(overflow)?; let start = end.checked_sub(*range_ms).ok_or_else(overflow)?; for (template, points) in series.values() { work.checkpoint().await?; diff --git a/crates/asap-physical-operators/src/operators/unchecked.rs b/crates/asap-physical-operators/src/operators/unchecked.rs index 80a4544a..46bc6a0c 100644 --- a/crates/asap-physical-operators/src/operators/unchecked.rs +++ b/crates/asap-physical-operators/src/operators/unchecked.rs @@ -54,6 +54,7 @@ impl TryFrom for Operator { function, range_ms, offset_ms, + at_ms, steps, .. } => Operator::series_window( @@ -61,8 +62,15 @@ impl TryFrom for Operator { function.map(|f| *f), range_ms, offset_ms, + at_ms, steps, )?, + Kind::SeriesLabels { kind, labels } => { + Operator::series_labels(input(0)?, kind, labels)? + } + Kind::SeriesBinary { operator } => { + Operator::series_binary(input(0)?, input(1)?, operator)? + } Kind::Project(expressions) => { if expressions.len() != output.fields.len() { return Err(invalid("projection width mismatch")); diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 21520150..655c024a 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -250,35 +250,54 @@ fn compile_internal( } ) && dag.edges.iter().any(|e| u64::from(e.producer.0) == id); if let (Payload::Fallback { expression }, false) = (&node.payload, raw_rows) { - let slot = promql_fallback::raw_series_input(id); - let (leaf, mut chain) = promql_fallback::lower(expression) + let promql_fallback::Lowering { + selectors, + mut steps, + } = promql_fallback::lower(expression) .map_err(|error| invalid(format!("node {id}: {error}")))?; - let mut inputs = match (leaf, sources.remove(&slot)) { - (Some((_, schema)), Some(contract)) if contract.schema == schema => { - graph.add_input(slot, contract)?; - vec![slot] - } - (None, None) => vec![], - (Some(_), None) => { - return Err(invalid(format!( - "node {id}: PromQL fallback requires raw series input {slot}" - ))) - } - _ => { - return Err(invalid(format!( - "node {id}: raw series input differs from the selector schema" + let mut slots = Vec::new(); + for (i, (_, schema)) in selectors.iter().enumerate() { + let slot = promql_fallback::raw_series_input(id, i); + match sources.remove(&slot) { + Some(contract) if &contract.schema == schema => { + graph.add_input(slot, contract)? + } + Some(_) => { + return Err(invalid(format!( + "node {id}: raw series input {slot} differs from the selector schema" ))) + } + None => { + return Err(invalid(format!( + "node {id}: PromQL fallback requires raw series input {slot}" + ))) + } } - }; - let last = chain + slots.push(slot); + } + let (last, last_inputs) = steps .pop() .ok_or_else(|| invalid("empty PromQL lowering"))?; - for operator in chain { - graph.add(auxiliary, inputs, operator)?; - inputs = vec![auxiliary]; + let mut ids = Vec::new(); + let resolve = |inputs: Vec, ids: &[NodeId]| { + inputs + .into_iter() + .map(|input| match input { + promql_fallback::Input::Raw(i) => slots[i], + promql_fallback::Input::Step(i) => ids[i], + }) + .collect::>() + }; + for (operator, inputs) in steps { + graph.add(auxiliary, resolve(inputs, &ids), operator)?; + ids.push(auxiliary); auxiliary -= 1; } - graph.add(id, inputs, last.with_output_schema(output)?)?; + graph.add( + id, + resolve(last_inputs, &ids), + last.with_output_schema(output)?, + )?; continue; } if let Payload::Value { diff --git a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs index ad950780..12c60ad1 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs @@ -1,38 +1,52 @@ //! Compile a retained PromQL subtree (`Fallback`) from its typed expression. -//! The deployment supplies the raw series of its one selector; the Planner -//! computes selection, range functions, subqueries and aggregation. +//! The deployment supplies the raw series of each selector; the Planner +//! computes selection, range functions, subqueries, matching and aggregation. use super::*; use crate::operators::SubquerySteps; use planner_types::post_asap::execution_data_state::lift_plain; +use planner_types::pre_asap::{AtModifier, BinaryOpKind, VectorMatch, VectorMatchKind}; -/// Input slot for the raw series read by Fallback node `node`'s selector. -/// The node's own ID names its computed output, so the raw rows need another. -pub fn raw_series_input(node: NodeId) -> NodeId { - node | (1 << 32) +/// Input slot for the raw series read by the `selector`th selector (in +/// [`raw_series`] order) of Fallback node `node`. The node's own ID names its +/// computed output, so the raw rows need another. +pub fn raw_series_input(node: NodeId, selector: usize) -> NodeId { + node | ((selector as u64 + 1) << 32) } /// The Fallback node that owns a raw-series input slot. pub(super) fn raw_series_owner(slot: NodeId) -> Option { - (slot >> 32 == 1).then_some(slot & u64::from(u32::MAX)) + (slot >> 32 != 0).then_some(slot & u64::from(u32::MAX)) } /// A selector expression and its raw-series row schema. pub type Selector = (QueryExpr, Schema); -/// The selector a Fallback expression reads, and the row schema of the raw -/// series the deployment supplies at [`raw_series_input`]. `None` means the -/// expression reads no series. The rows must cover the selector's window at -/// every evaluation instant; under a subquery `[R:S] offset O` that is -/// `(T - O - R - offset - range, T - O - offset]`. -pub fn raw_series(expression: &QueryExpr) -> Result, Error> { - Ok(lower(expression)?.0) +/// The selectors a Fallback expression reads, left to right, and the row +/// schema of the raw series the deployment supplies for each at +/// [`raw_series_input`]. The rows must cover the selector's window at every +/// evaluation instant `T`, or at its `@` time: `(T - offset - range, T - offset]`; +/// under a subquery `[R:S] offset O` that is `(T - O - R - offset - range, T - O - offset]`. +pub fn raw_series(expression: &QueryExpr) -> Result, Error> { + Ok(lower(expression)?.selectors) } -/// Operators computing `expression`, in order, after its raw-series input. -pub(super) fn lower(expression: &QueryExpr) -> Result<(Option, Vec), Error> { - let mut chain = Chain::default(); - chain.value(expression)?; - Ok((chain.leaf, chain.operators)) +/// An operator input: a selector's raw rows or an earlier step. +pub(super) enum Input { + Raw(usize), + Step(usize), +} + +/// Operators computing an expression; the last step is its result. +#[derive(Default)] +pub(super) struct Lowering { + pub selectors: Vec, + pub steps: Vec<(Operator, Vec)>, +} + +pub(super) fn lower(expression: &QueryExpr) -> Result { + let mut lowering = Lowering::default(); + lowering.value(expression)?; + Ok(lowering) } fn declared(expression: &QueryExpr) -> Result { @@ -46,48 +60,64 @@ fn millis(duration: &std::time::Duration) -> Result { i64::try_from(duration.as_millis()).map_err(|_| invalid("PromQL duration exceeds Int64")) } -/// `TimeRange { range, [TimeShift { offset }], Scan }`: range and offset. -fn selector(expression: &QueryExpr) -> Result<(i64, i64), Error> { +/// A fixed `@` time. `start()`/`end()` depend on the deployment's range query. +fn at(shift: &planner_types::pre_asap::TimeShift) -> Result, Error> { + match shift.at { + None => Ok(None), + Some(AtModifier::Timestamp(at)) => Ok(Some(at)), + Some(_) => Err(invalid("@ start() and @ end() depend on the range query")), + } +} + +/// `TimeRange { range, [TimeShift { offset, @ }], Scan }`: range, offset, `@`. +fn selector(expression: &QueryExpr) -> Result<(i64, i64, Option), Error> { let QueryExpr::TimeRange { range, child } = expression else { return Err(invalid("PromQL operand must be a series selector")); }; - let (offset, scan) = match child.as_ref() { - QueryExpr::TimeShift { shift, child } if shift.at.is_none() => { - (shift.offset_ms, child.as_ref()) - } - scan => (0, scan), + let (offset, at, scan) = match child.as_ref() { + QueryExpr::TimeShift { shift, child } => (shift.offset_ms, at(shift)?, child.as_ref()), + scan => (0, None, scan), }; if !matches!(scan, QueryExpr::Scan { .. }) { - return Err(invalid( - "PromQL selector must read one scan; @ is unsupported", - )); + return Err(invalid("PromQL selector must read one scan")); } - Ok((millis(range)?, offset)) + Ok((millis(range)?, offset, at)) } -#[derive(Default)] -struct Chain { - leaf: Option, - operators: Vec, +/// PromQL scalar-valued expressions have no labels to match. +fn scalar(expression: &QueryExpr) -> bool { + matches!( + expression, + QueryExpr::PromqlScalarBridge(_) + | QueryExpr::PromqlScalarFromVector(_) + | QueryExpr::EvalTimestamp + ) } -impl Chain { - fn schema(&self) -> Result { - self.operators - .last() - .map(Operator::schema) - .or_else(|| self.leaf.as_ref().map(|(_, schema)| schema.clone())) - .ok_or_else(|| invalid("PromQL operator has no input")) +impl Lowering { + fn schema(&self, input: &Input) -> Schema { + match input { + Input::Raw(i) => self.selectors[*i].1.clone(), + Input::Step(i) => self.steps[*i].0.schema(), + } + } + + fn add(&mut self, operator: Operator, inputs: Vec) -> Input { + self.steps.push((operator, inputs)); + Input::Step(self.steps.len() - 1) } /// Conform `operator` to the logical schema of the expression it computes. - fn push(&mut self, operator: Operator, logical: &QueryExpr) -> Result<(), Error> { - self.operators - .push(operator.with_output_schema(declared(logical)?)?); - Ok(()) + fn push( + &mut self, + operator: Operator, + inputs: Vec, + logical: &QueryExpr, + ) -> Result { + Ok(self.add(operator.with_output_schema(declared(logical)?)?, inputs)) } - fn read(&mut self, selector: &QueryExpr) -> Result { + fn read(&mut self, selector: &QueryExpr) -> Result { let schema = declared(selector)?; if !schema .fields @@ -98,24 +128,20 @@ impl Chain { "PromQL fallback requires the complete series identity", )); } - if self - .leaf - .replace((selector.clone(), schema.clone())) - .is_some() - { - return Err(invalid("PromQL fallback reads more than one selector")); - } - Ok(schema) + self.selectors.push((selector.clone(), schema)); + Ok(Input::Raw(self.selectors.len() - 1)) } /// An instant vector, or a scalar for scalar-valued expressions. - fn value(&mut self, expression: &QueryExpr) -> Result<(), Error> { + fn value(&mut self, expression: &QueryExpr) -> Result { match expression { QueryExpr::TimeRange { .. } => { - let (range, offset) = selector(expression)?; + let (range, offset, at) = selector(expression)?; let input = self.read(expression)?; + let schema = self.schema(&input); self.push( - Operator::series_window(input, None, range, offset, None)?, + Operator::series_window(schema, None, range, offset, at, None)?, + vec![input], expression, ) } @@ -141,16 +167,16 @@ impl Chain { let [measure] = measures.as_slice() else { return Err(invalid("vector aggregation requires one measure")); }; - self.value(child)?; - self.aggregate(measure, keys, expression) + let input = self.value(child)?; + self.aggregate(input, measure, keys, expression) } QueryExpr::Sort { keys, partition_by, child, } => { - self.value(child)?; - let input = self.schema()?; + let step = self.value(child)?; + let input = self.schema(&step); let keys = keys .iter() .map(|key| match key.expr { @@ -163,11 +189,11 @@ impl Chain { }) .collect::>()?; let groups = groups(&input, partition_by)?; - self.push(Operator::sort(input, keys, groups)?, expression) + self.push(Operator::sort(input, keys, groups)?, vec![step], expression) } QueryExpr::Limit { n, offset, child } => { - self.value(child)?; - let input = self.schema()?; + let step = self.value(child)?; + let input = self.schema(&step); // `topk by (...)` partitions through the Sort it limits. let groups = match child.as_ref() { QueryExpr::Sort { partition_by, .. } => groups(&input, partition_by)?, @@ -175,14 +201,15 @@ impl Chain { }; self.push( Operator::limit(input, *n as u64, *offset as u64, groups)?, + vec![step], expression, ) } QueryExpr::BinaryOp { - op: planner_types::pre_asap::BinaryOpKind::Arithmetic(op), + op: BinaryOpKind::Arithmetic(op), lhs, rhs, - vector_match: None, + vector_match, } => { let (vector, literal, literal_left) = match ( row_values::scalar_literal(lhs), @@ -190,14 +217,11 @@ impl Chain { ) { (None, Some(value)) => (lhs, value, false), (Some(value), None) => (rhs, value, true), - _ => { - return Err(invalid( - "PromQL fallback arithmetic requires one literal operand", - )) - } + (None, None) => return self.match_vectors(expression, vector_match), + _ => return Err(invalid("PromQL arithmetic between two literals")), }; - self.value(vector)?; - let input = self.schema()?; + let step = self.value(vector)?; + let input = self.schema(&step); let value = named_column(&input, &ColumnRef::SampleValue)?; let literal = Expression::Literal { value: crate::values::Value::Float64(literal), @@ -231,26 +255,32 @@ impl Chain { (field.name.clone(), expression) }) .collect(); - self.push(Operator::project(input, columns)?, expression) + self.push(Operator::project(input, columns)?, vec![step], expression) } QueryExpr::PromqlScalarFromVector(child) => { - self.value(child)?; - let input = self.schema()?; + let step = self.value(child)?; + let input = self.schema(&step); let value = named_column(&input, &ColumnRef::SampleValue)?; - self.push(Operator::vector_to_scalar(input, value)?, expression) + self.push( + Operator::vector_to_scalar(input, value)?, + vec![step], + expression, + ) } QueryExpr::PromqlVectorFromScalar(child) => { - self.value(child)?; - let input = self.schema()?; - self.operators - .push(Operator::scope_timestamp(input, declared(expression)?)?); - Ok(()) + let step = self.value(child)?; + let input = self.schema(&step); + Ok(self.add( + Operator::scope_timestamp(input, declared(expression)?)?, + vec![step], + )) } QueryExpr::PromqlScalarBridge(_) => { let value = row_values::scalar_literal(expression) .ok_or_else(|| invalid("PromQL scalar must be a literal"))?; self.push( Operator::scalar(crate::values::Value::Float64(value), DataType::Float64)?, + vec![], expression, ) } @@ -258,19 +288,65 @@ impl Chain { } } + /// One-to-one vector arithmetic: both sides reduce to their matching + /// labels, which are also the result's labels. + fn match_vectors( + &mut self, + logical: &QueryExpr, + vector_match: &Option, + ) -> Result { + let QueryExpr::BinaryOp { + op: kind, lhs, rhs, .. + } = logical + else { + unreachable!() + }; + if scalar(lhs) || scalar(rhs) { + return Err(invalid("PromQL arithmetic with a non-literal scalar")); + } + let (matching, labels) = match vector_match { + None => (VectorMatchKind::Ignoring, vec![]), + Some(VectorMatch { + kind, + labels, + grouping: None, + }) => (kind.clone(), labels.clone()), + Some(_) => return Err(invalid("group_left/group_right matching is unsupported")), + }; + let mut sides = Vec::new(); + for side in [lhs, rhs] { + let step = self.value(side)?; + let input = self.schema(&step); + sides.push(self.add( + Operator::series_labels(input, matching.clone(), labels.clone())?, + vec![step], + )); + } + let (left, right) = (self.schema(&sides[0]), self.schema(&sides[1])); + let operator = planner_types::post_asap::BinaryOperator { + kind: kind.clone(), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }; + self.push( + Operator::series_binary(left, right, operator)?, + sides, + logical, + ) + } + /// `function(matrix)`, where the matrix is a range selector or a subquery. fn range_function( &mut self, function: &AggIntent, matrix: &QueryExpr, logical: &QueryExpr, - ) -> Result<(), Error> { + ) -> Result { let function = unbound(function)?; - let (subquery, offset) = match matrix { - QueryExpr::TimeShift { shift, child } if shift.at.is_none() => { - (child.as_ref(), shift.offset_ms) - } - other => (other, 0), + let (subquery, offset, at_ms) = match matrix { + QueryExpr::TimeShift { shift, child } => (child.as_ref(), shift.offset_ms, at(shift)?), + other => (other, 0, None), }; let QueryExpr::PromqlSubquery { range: outer, @@ -278,10 +354,12 @@ impl Chain { child, } = subquery else { - let (range, offset) = selector(matrix)?; + let (range, offset, at) = selector(matrix)?; let input = self.read(matrix)?; + let schema = self.schema(&input); return self.push( - Operator::series_window(input, Some(function), range, offset, None)?, + Operator::series_window(schema, Some(function), range, offset, at, None)?, + vec![input], logical, ); }; @@ -292,6 +370,7 @@ impl Chain { range_ms: millis(outer)?, step_ms: millis(step)?, offset_ms: offset, + at_ms, }; // Each step evaluates a per-series selection or range function. let (inner, selected) = match child.as_ref() { @@ -307,15 +386,18 @@ impl Chain { }, selected => (None, selected), }; - let (range, inner_offset) = selector(selected)?; - let input = self.read(selected)?; - self.push( - Operator::series_window(input, inner, range, inner_offset, Some(steps))?, + let (range, inner_offset, inner_at) = selector(selected)?; + let raw = self.read(selected)?; + let schema = self.schema(&raw); + let step = self.push( + Operator::series_window(schema, inner, range, inner_offset, inner_at, Some(steps))?, + vec![raw], child, )?; - let input = self.schema()?; + let input = self.schema(&step); self.push( - Operator::series_window(input, Some(function), steps.range_ms, offset, None)?, + Operator::series_window(input, Some(function), steps.range_ms, offset, at_ms, None)?, + vec![step], logical, ) } @@ -324,11 +406,12 @@ impl Chain { /// that no input series yields an empty vector, not one row. fn aggregate( &mut self, + mut step: Input, measure: &AggIntent, keys: &GroupKeys, logical: &QueryExpr, - ) -> Result<(), Error> { - let mut input = self.schema()?; + ) -> Result { + let mut input = self.schema(&step); let value = named_column(&input, &ColumnRef::SampleValue)?; let reduction = match measure { AggIntent::Sum { col: None } => Reduction::Sum(value), @@ -338,7 +421,22 @@ impl Chain { AggIntent::Count { .. } => Reduction::Count, _ => return Err(invalid("vector aggregate has no native lowering")), }; - let mut groups = groups(&input, keys)?; + let mut groups = if keys.is_without() { + // Group by every remaining label, including the rewritten identity. + let excluded = keys.keys(); + if excluded.iter().any(|&i| i >= input.fields.len()) { + return Err(invalid("grouping column out of range")); + } + let names = excluded.iter().map(|&i| input.fields[i].name.clone()); + let relabel = + Operator::series_labels(input.clone(), VectorMatchKind::Ignoring, names.collect())?; + step = self.add(relabel, vec![step]); + (0..input.fields.len()) + .filter(|&i| Some(i) != input.time_index && i != value && !excluded.contains(&i)) + .collect() + } else { + groups(&input, keys)? + }; let global = groups.is_empty(); if global { let mut columns = (0..input.fields.len()) @@ -354,7 +452,7 @@ impl Chain { let project = Operator::project(input, columns)?; input = project.schema(); groups = vec![input.fields.len() - 1]; - self.operators.push(project); + step = self.add(project, vec![step]); } let output = declared(logical)?; let name = output @@ -365,7 +463,7 @@ impl Chain { .clone(); let aggregate = Operator::aggregate(input, groups, vec![(name, reduction)])?; let actual = aggregate.schema(); - self.operators.push(aggregate); + let step = self.add(aggregate, vec![step]); // Drop the constant group; convert counts where PromQL declares Float64. let skip = usize::from(global); let columns = actual.fields[skip..] @@ -382,7 +480,7 @@ impl Chain { (field.name.clone(), expression) }) .collect(); - self.push(Operator::project(actual, columns)?, logical) + self.push(Operator::project(actual, columns)?, vec![step], logical) } } diff --git a/crates/asap-physical-operators/tests/promql_fallback.rs b/crates/asap-physical-operators/tests/promql_fallback.rs index e5c7fb76..d274495c 100644 --- a/crates/asap-physical-operators/tests/promql_fallback.rs +++ b/crates/asap-physical-operators/tests/promql_fallback.rs @@ -17,7 +17,7 @@ use planner_types::{ use std::{collections::BTreeMap, rc::Rc}; /// Bare selectors look back one ingestion interval: 60s. -fn lower(query: &str) -> QueryExpr { +fn parse(query: &str) -> QueryExpr { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -42,10 +42,13 @@ fn lower(query: &str) -> QueryExpr { ..Default::default() }), }; - let expression = asap_frontend_promql::lower_promql_workload(&workload, 0) + asap_frontend_promql::lower_promql_workload(&workload, 0) .unwrap() - .remove(0); - promql_rows::with_series_identity(&expression).unwrap() + .remove(0) +} + +fn lower(query: &str) -> QueryExpr { + promql_rows::with_series_identity(&parse(query)).unwrap() } /// The whole query retained as one pre-ASAP node. @@ -59,46 +62,79 @@ fn fallback_dag(expression: QueryExpr) -> PostAsapDag { .unwrap() } -/// `(job, seconds, value)`; every sample belongs to metric `m`. +/// `(labels, seconds, value)`. `labels` is `k=v,...`, or a bare `job` value. type Sample = (&'static str, i64, f64); +fn labels(spec: &str) -> BTreeMap { + if !spec.contains('=') { + return BTreeMap::from([("job".into(), spec.into())]); + } + spec.split(',') + .map(|pair| { + let (k, v) = pair.split_once('=').unwrap(); + (k.to_string(), v.to_string()) + }) + .collect() +} + +/// The metric a selector reads. +fn metric(selector: &QueryExpr) -> String { + match selector { + QueryExpr::Scan { + source: planner_types::pre_asap::Source::TimeSeries { metric }, + .. + } => metric.clone(), + QueryExpr::TimeRange { child, .. } | QueryExpr::TimeShift { child, .. } => metric(child), + other => panic!("not a selector: {other:?}"), + } +} + fn compile_query(query: &str) -> Result { let expression = lower(query); let dag = fallback_dag(expression.clone()); let root = u64::from(dag.root.0); let inputs = promql_fallback::raw_series(&expression) .map_err(|e| e.to_string())? - .map(|(_, schema)| { + .into_iter() + .enumerate() + .map(|(i, (_, schema))| { ( - promql_fallback::raw_series_input(root), + promql_fallback::raw_series_input(root, i), InputContract::bounded(schema), ) }) - .into_iter() .collect(); let program = compile(&dag, inputs, &[root]).map_err(|e| e.to_string())?; Ok(serde_json::from_slice(&serde_json::to_vec(&program).unwrap()).unwrap()) } -/// Evaluate at `at` seconds; returns `(job or "", timestamp ms, value)` rows in order. -fn run(query: &str, samples: &[Sample], at: i64) -> Result, String> { +/// Evaluate at `at` seconds over samples of each named metric; returns +/// `(output labels, timestamp ms, value)` rows in order. +#[allow(clippy::type_complexity)] +fn evaluate( + query: &str, + metrics: &[(&str, &[Sample])], + at: i64, +) -> Result, i64, f64)>, String> { let expression = lower(query); let program = compile_query(query)?; let mut sources = BTreeMap::new(); - if let Some((_, schema)) = promql_fallback::raw_series(&expression).unwrap() { - let rows = samples + let selectors = promql_fallback::raw_series(&expression).unwrap(); + for (i, (selector, schema)) in selectors.into_iter().enumerate() { + let name = metric(&selector); + let rows = metrics .iter() - .map(|(job, seconds, value)| { - let labels = BTreeMap::from([ - ("__name__".to_string(), "m".to_string()), - ("job".into(), job.to_string()), - ]); + .filter(|(m, _)| *m == name) + .flat_map(|(_, samples)| samples.iter()) + .map(|(spec, seconds, value)| { + let mut labels = labels(spec); + labels.insert("__name__".into(), name.clone()); promql_rows::series_row(&schema, &labels, seconds * 1000, *value).unwrap() }) .collect(); let batch = Batch::try_new(schema.clone(), rows).unwrap(); sources.insert( - promql_fallback::raw_series_input(program.roots()[0]), + promql_fallback::raw_series_input(program.roots()[0], i), Box::new(Operator::source(schema, vec![batch]).unwrap()) as _, ); } @@ -121,28 +157,67 @@ fn run(query: &str, samples: &[Sample], at: i64) -> Result { - job = promql_rows::decode_series_identity(id).unwrap()["job"].clone() + labels = promql_rows::decode_series_identity(id).unwrap() } - ("job", Value::Utf8(label)) => job = label.to_string(), + (_, Value::Utf8(_) | Value::Null) => {} (_, Value::Timestamp(t)) => time = *t, (_, Value::Float64(v)) => value = Some(*v), (_, Value::Int64(v)) => value = Some(*v as f64), other => return Err(format!("unexpected cell {other:?}")), } } - rows.push((job, time, value.ok_or("missing value")?)); + if !schema + .fields + .iter() + .any(|f| f.name == promql_rows::SERIES_IDENTITY_COLUMN) + { + for (field, cell) in schema.fields.iter().zip(row) { + if let Value::Utf8(label) = cell { + if !label.is_empty() { + labels.insert(field.name.clone(), label.to_string()); + } + } + } + } + rows.push((labels, time, value.ok_or("missing value")?)); } } Ok(rows) }) } +/// Evaluate at `at` seconds over metric `m`; returns `(job or "", timestamp ms, value)`. +fn run(query: &str, samples: &[Sample], at: i64) -> Result, String> { + Ok(evaluate(query, &[("m", samples)], at)? + .into_iter() + .map(|(labels, time, value)| (labels.get("job").cloned().unwrap_or_default(), time, value)) + .collect()) +} + +/// Output rows as `(k=v,... sorted, value)`, including any `__name__`. +fn labeled(query: &str, metrics: &[(&str, &[Sample])], at: i64) -> Vec<(String, f64)> { + let mut rows = evaluate(query, metrics, at) + .unwrap_or_else(|e| panic!("{query}: {e}")) + .into_iter() + .map(|(labels, _, value)| { + let spec = labels + .iter() + .map(|(k, v)| format!("{k}={v}")) + .collect::>() + .join(","); + (spec, value) + }) + .collect::>(); + rows.sort_by(|a, b| a.0.cmp(&b.0)); + rows +} + fn values(query: &str, samples: &[Sample], at: i64) -> Vec<(String, f64)> { run(query, samples, at) .unwrap_or_else(|e| panic!("{query}: {e}")) @@ -297,7 +372,6 @@ fn subqueries_evaluate_their_operand_on_the_aligned_grid() { ), 3. ); - assert!(compile_query("max_over_time(m[5m:1m] @ 100)").is_err()); } // Subquery work is bounded by the query: at most 100000 steps. @@ -314,7 +388,10 @@ fn raw_series_contract_is_explicit() { let expression = lower("rate(m[5m])"); let dag = fallback_dag(expression.clone()); let root = u64::from(dag.root.0); - let (selector, schema) = promql_fallback::raw_series(&expression).unwrap().unwrap(); + let [(selector, schema)] = promql_fallback::raw_series(&expression) + .unwrap() + .try_into() + .unwrap(); assert!(matches!(selector, QueryExpr::TimeRange { .. })); let missing = compile(&dag, BTreeMap::new(), &[root]).err().unwrap(); assert!(missing.to_string().contains("raw series input")); @@ -323,7 +400,7 @@ fn raw_series_contract_is_explicit() { let wrong = compile( &dag, BTreeMap::from([( - promql_fallback::raw_series_input(root), + promql_fallback::raw_series_input(root, 0), InputContract::bounded(std::sync::Arc::new(wrong)), )]), &[root], @@ -370,11 +447,11 @@ fn raw_series_contract_is_explicit() { }], root: PostAsapNodeId(1), }; - let raw = promql_fallback::raw_series(&selector).unwrap().unwrap().1; + let raw = promql_fallback::raw_series(&selector).unwrap().remove(0).1; assert!(compile( &consumed, BTreeMap::from([( - promql_fallback::raw_series_input(0), + promql_fallback::raw_series_input(0, 0), InputContract::bounded(raw) )]), &[1], @@ -425,3 +502,123 @@ fn instant_and_counting_range_functions() { f64::NEG_INFINITY ); } + +// `@ ` evaluates the selector or subquery at `t`, minus any offset, and the +// result keeps the query's evaluation time. +#[test] +fn at_modifier_fixes_the_evaluation_instant() { + let samples = &[("a", 60, 1.), ("a", 120, 2.), ("a", 180, 3.)]; + assert_eq!( + run("m @ 120", samples, 1000).unwrap(), + vec![("a".into(), 1_000_000, 2.)] + ); + assert!(values("m", samples, 1000).is_empty()); + assert_eq!(one("count_over_time(m[2m] @ 180)", samples, 1000), 2.); + assert_eq!(one("m @ 180 offset 1m", samples, 1000), 2.); + // The subquery grid is (60s, 180s]: steps 120 and 180 select 2 and 3. + assert_eq!(one("max_over_time(m[2m:1m] @ 180)", samples, 1000), 3.); + assert_eq!( + one("sum_over_time(m[2m:1m] @ 180 offset 1m)", samples, 1000), + 3. + ); + // An inner @ pins every step to the same instant. + assert_eq!(one("sum_over_time((m @ 60)[2m:1m])", samples, 180), 2.); + // start() and end() depend on the range query, which is the deployment's. + assert!(compile_query("m @ start()").is_err()); +} + +const A: &[Sample] = &[("job=x", 50, 10.), ("job=y", 50, 20.), ("job=w", 50, 0.)]; +const B: &[Sample] = &[("job=x", 50, 2.), ("job=z", 50, 5.), ("job=w", 50, 0.)]; + +// Vector-vector arithmetic matches series one-to-one on label sets without +// the metric name, and the result drops the metric name. +#[test] +fn vector_arithmetic_matches_label_sets() { + let metrics = &[("a", A), ("b", B)]; + let quotient = labeled("a / b", metrics, 60); + assert_eq!(quotient.len(), 2); + assert_eq!(quotient[0].0, "job=w"); + assert!(quotient[0].1.is_nan(), "0 / 0 is NaN"); + assert_eq!(quotient[1], ("job=x".into(), 5.)); + // Each selector reads its own raw rows, even a repeated metric. + assert_eq!( + labeled("(a - b) * a", metrics, 60), + vec![("job=w".into(), 0.), ("job=x".into(), 80.)] + ); + assert_eq!( + labeled("sum by (job) (a) - sum by (job) (b)", metrics, 60), + vec![("job=w".into(), 0.), ("job=x".into(), 8.)] + ); + // Rates of two counters over their own windows. + let up: &[Sample] = &[("job=x", 0, 0.), ("job=x", 60, 60.)]; + let down: &[Sample] = &[("job=x", 0, 0.), ("job=x", 60, 30.)]; + assert_eq!( + labeled("rate(a[2m]) / rate(b[2m])", &[("a", up), ("b", down)], 60), + vec![("job=x".into(), 2.)] + ); +} + +// on() keeps only the listed labels and ignoring() drops them; a duplicate +// match group is an error unless the left duplicates never match. +#[test] +fn on_and_ignoring_select_the_matching_labels() { + let a: &[Sample] = &[("job=x,inst=1", 50, 10.)]; + let b: &[Sample] = &[("job=x,inst=2", 50, 4.)]; + let metrics = &[("a", a), ("b", b)]; + assert!(labeled("a - b", metrics, 60).is_empty()); + assert_eq!( + labeled("a - on(job) b", metrics, 60), + vec![("job=x".into(), 6.)] + ); + assert_eq!( + labeled("a - ignoring(inst) b", metrics, 60), + vec![("job=x".into(), 6.)] + ); + let pair: &[Sample] = &[("job=x,inst=1", 50, 1.), ("job=x,inst=2", 50, 2.)]; + let other: &[Sample] = &[("job=y", 50, 1.)]; + assert!(evaluate("a + on(job) b", &[("a", a), ("b", pair)], 60).is_err()); + assert!(evaluate("a + on(job) b", &[("a", pair), ("b", b)], 60).is_err()); + assert!(labeled("a + on(job) b", &[("a", pair), ("b", other)], 60).is_empty()); + assert!(promql_rows::with_series_identity(&parse("a + on(job) group_left b")).is_err()); +} + +// without() groups by every label except the listed ones and the metric name. +#[test] +fn without_grouping_drops_labels_and_the_name() { + let a: &[Sample] = &[ + ("job=x,inst=1", 50, 1.), + ("job=x,inst=2", 50, 2.), + ("job=y,inst=1", 50, 4.), + ]; + let metrics = &[("a", a)]; + assert_eq!( + labeled("sum without (inst) (a)", metrics, 60), + vec![("job=x".into(), 3.), ("job=y".into(), 4.)] + ); + assert_eq!( + labeled("count without (inst) (a)", metrics, 60), + vec![("job=x".into(), 2.), ("job=y".into(), 1.)] + ); + assert_eq!( + labeled("max without (job, inst) (a)", metrics, 60), + vec![(String::new(), 4.)] + ); + assert!(labeled("sum without (inst) (a)", &[], 60).is_empty()); +} + +// An empty label value is an absent label, and an empty side yields an empty +// result before any duplicate check, as in Prometheus. +#[test] +fn empty_labels_and_empty_sides_match_prometheus() { + let a: &[Sample] = &[("job=x,env=", 50, 3.)]; + let b: &[Sample] = &[("job=x", 50, 1.)]; + assert_eq!( + labeled("a + b", &[("a", a), ("b", b)], 60), + vec![("job=x".into(), 4.)] + ); + let pair: &[Sample] = &[("job=x,inst=1", 50, 1.), ("job=x,inst=2", 50, 2.)]; + assert!(labeled("a + on(job) b", &[("b", pair)], 60).is_empty()); + assert!(labeled("b + on(job) a", &[("b", pair)], 60).is_empty()); + // A non-literal scalar operand has no identity realization yet. + assert!(promql_rows::with_series_identity(&parse("a + scalar(b)")).is_err()); +} From e93bfaf88cf7231520006e068e17846fd59c312a Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 04:16:46 +0000 Subject: [PATCH 46/59] docs: record multi-selector PromQL fallback coverage and histogram_quantile IR needs Co-Authored-By: Claude Opus 5.5 --- .../develop_docs/physical-compile-coverage.md | 68 +++++++++++++++---- 1 file changed, 53 insertions(+), 15 deletions(-) diff --git a/docs/develop_docs/physical-compile-coverage.md b/docs/develop_docs/physical-compile-coverage.md index ce8c3817..31c69782 100644 --- a/docs/develop_docs/physical-compile-coverage.md +++ b/docs/develop_docs/physical-compile-coverage.md @@ -76,14 +76,14 @@ Totals after this change: 17 Supported, 4 Partial, 8 Missing, 2 Backend. ## Covered by PromQL fallback compilation -`compile` now lowers a `Fallback{QueryExpr}` node from its typed expression. -The expression must read at most one selector, realized with -`promql_rows::with_series_identity`. The deployment supplies that selector's raw -rows at `promql_fallback::raw_series_input(node)`, with the schema returned by -`promql_fallback::raw_series`. The node's own ID still names its output, so a -deployment may instead supply the whole result, for example from an external -exact engine. A Fallback that reads no selector, such as `vector(1)`, needs no -input. +`compile` now lowers a `Fallback{QueryExpr}` node from its typed expression, +realized with `promql_rows::with_series_identity`. The deployment supplies the +raw rows of the `i`th selector returned by `promql_fallback::raw_series` at +`promql_fallback::raw_series_input(node, i)`, with that selector's schema. Each +selector has its own slot, even when two selectors read the same metric. The +node's own ID still names its output, so a deployment may instead supply the +whole result, for example from an external exact engine. A Fallback that reads +no selector, such as `vector(1)`, needs no input. `Operator::series_window` evaluates each series at the query time, or at each subquery step, over the left-open window `(t - offset - range, t - offset]`. @@ -104,23 +104,61 @@ and is not compiled as instant selection. Totals after this change: 19 Supported, 5 Partial, 5 Missing, 2 Backend. +## Covered by multi-selector fallback compilation + +| Row | Change | +|---|---| +| 1 | Vector-vector arithmetic with one-to-one matching, `on`, and `ignoring`. Each side is reduced to its matching labels, then matched on equal label sets. A duplicate match group is an error, as in Prometheus. The result drops `__name__`. `without` aggregation. `irate`, `idelta`, `changes`, `resets`, `last_over_time`, and exact `quantile_over_time`. `@ ` on selectors. Still Partial. | +| 11 | The range functions above, over raw selector rows. | +| 12 | `@ ` on subqueries anchors the step grid. An inner selector's `@` pins every step. Still Partial. | + +`@` fixes the instant a window ends at, before `offset`; the output keeps the +query's evaluation time. The raw rows must cover the window at that instant. +Vector matching and `without` rewrite the series identity, so their results +already lack `__name__`. Other results keep it; the query adapter still drops +it. + +Totals are unchanged: 19 Supported, 5 Partial, 5 Missing, 2 Backend. + ## Remaining In order of backend usage: 1. Rows 1, 10, and 12, the remaining `Fallback` shapes: - - Subtrees with more than one selector, such as vector-vector binaries. - - `histogram_quantile`: the frontend emits `by ()` grouping with no - output labels. The IR must group `without (le)` and keep the labels. + - `histogram_quantile` (row 10). The compiler cannot recover the grouping + from today's IR, which is `Aggregate{Reduce(by ()), [HistogramQuantile{q}]}`. + It computes one quantile over all buckets and drops the output labels. The IR + must carry: + - The bucket label: the `le` column of the child's schema, named by the + intent, for example `HistogramQuantile{q, le: C}`. The frontend must + add `le` to the selector's schema even when no matcher names it. + - The grouping: `Reduce(without([le]))`, so that each histogram is + one set of series that differ only in `le`. An explicit + `sum by (x, le)` inside the argument still yields `without (le)` over + those rows. + - The output labels: every input label except `le` and `__name__`. The + `without` output schema already carries them, and the series identity + with `le` removed. `series_labels(Ignoring, [le])` computes the latter. + The operator then applies Prometheus `bucketQuantile`. It parses `le` + as a float, skips unparsable values, and requires a `+Inf` bucket, else + it returns NaN. It forces cumulative counts to be monotonic and returns + NaN for fewer than two buckets. For q < 0 it returns -Inf; for q > 1, + +Inf. + - Comparisons and set operators (`and`, `or`, `unless`); `group_left` and + `group_right`; arithmetic with a non-literal scalar, such as + `scalar(x)` or `time()`. - Subquery operands other than one per-series function; implicit subquery resolution, which is a deployment default. - - `@` on selectors and subqueries; `without` grouping; `irate`, - `changes`, and other range functions. + - `@ start()` and `@ end()`, which need the range query's bounds in the + run scope. + - Other functions, such as `deriv`, `predict_linear`, + `stddev_over_time`, `absent`, `label_replace`, and math functions. After these shapes are covered, the backend can delete rows 28 and 30. 2. Row 7: comparison filters and `bool` comparisons. This needs `return_bool` in the `Binary` payload. `compile` currently rejects comparisons. -3. Row 5 for per-series rows: matching needs a metric-name-free series - identity, not the full `$promql_series_identity`. +3. Row 5 for per-series rows in a `Binary` payload node: its grouped-row join + still rejects `$promql_series_identity`. It could reuse the Fallback's + `series_labels` and `series_binary` operators. 4. Rows 25 and 27: constant weights and `EntityIdentity` items for precompute `SummaryAgg`. 5. Row 16: a label-map sketch-state readout, the counterpart of From d4375b3e8231c645a357318bbbe00ea2df1684fc Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:10:40 +0000 Subject: [PATCH 47/59] integrate: extend asap-types series identity with #486/#487 shapes #477 moved PromQL series-identity resolution into asap-types. The fallback shapes compiled by #486 and #487 (time shifts, subqueries, scalar bridges, and arithmetic between series) also need identity realization there. Taken from integration commit 9a13ae4. Co-Authored-By: Claude Opus 5.5 --- crates/types/src/pre_asap/schema.rs | 45 +++++++++++++++++++++-------- 1 file changed, 33 insertions(+), 12 deletions(-) diff --git a/crates/types/src/pre_asap/schema.rs b/crates/types/src/pre_asap/schema.rs index 938b9b7f..fdac93de 100644 --- a/crates/types/src/pre_asap/schema.rs +++ b/crates/types/src/pre_asap/schema.rs @@ -168,8 +168,16 @@ pub const PROMQL_SERIES_IDENTITY: &str = "$promql_series_identity"; /// own realization; they must not accidentally treat the opaque identity as a /// user label or silently discard it. pub fn with_promql_series_identity(root: &super::QueryExpr) -> Result { - use super::{QueryExpr, Reduction, Source}; + use super::{QueryExpr, Source}; use std::rc::Rc; + fn scalar_literal(expression: &QueryExpr) -> Option { + match expression { + QueryExpr::PromqlScalarBridge(child) => scalar_literal(child), + QueryExpr::Literal(super::ScalarValue::Float64(value)) => Some(*value), + _ => None, + } + } + let mut root = root.clone(); fn visit(node: &mut QueryExpr) -> Result<(), String> { match node { QueryExpr::Scan { @@ -193,17 +201,31 @@ pub fn with_promql_series_identity(root: &super::QueryExpr) -> Result { - visit(Rc::make_mut(child)) - } - QueryExpr::Aggregate { - child, reduction, .. - } => { - if matches!(reduction, Reduction::Reduce(keys) if keys.is_without()) { - return Err("dynamic without grouping requires label-set projection".into()); - } - visit(Rc::make_mut(child)) + QueryExpr::TimeRange { child, .. } + | QueryExpr::Limit { child, .. } + | QueryExpr::TimeShift { child, .. } + | QueryExpr::PromqlSubquery { child, .. } + | QueryExpr::PromqlScalarFromVector(child) => visit(Rc::make_mut(child)), + // Constants read no series. + QueryExpr::PromqlScalarBridge(_) => Ok(()), + QueryExpr::PromqlVectorFromScalar(child) if scalar_literal(child).is_some() => Ok(()), + QueryExpr::BinaryOp { + op: super::BinaryOpKind::Arithmetic(_), + lhs, + rhs, + vector_match, + } if vector_match.as_ref().is_none_or(|m| m.grouping.is_none()) + && ![&*lhs, &*rhs].into_iter().any(|side| { + matches!( + side.as_ref(), + QueryExpr::PromqlScalarFromVector(_) | QueryExpr::EvalTimestamp + ) + }) => + { + visit(Rc::make_mut(lhs))?; + visit(Rc::make_mut(rhs)) } + QueryExpr::Aggregate { child, .. } => visit(Rc::make_mut(child)), QueryExpr::Sort { child, partition_by, @@ -217,7 +239,6 @@ pub fn with_promql_series_identity(root: &super::QueryExpr) -> Result Err("operator has no dynamic series-identity realization".into()), } } - let mut root = root.clone(); visit(&mut root)?; root.output_schema().map_err(|error| error.to_string())?; Ok(root) From f9502890a71086386a8eb403b443db0eba6d29ce Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 06:29:30 +0000 Subject: [PATCH 48/59] feat(physical): compile summaries over raw-sample precompute boundaries A precompute boundary at a raw time-series scan binds as raw sample rows ($population label map, $timestamp, value). The label map is the complete source identity, so per-series and grouped SummaryAgg lower over it with constant, column or unit-frequency (HLL) updates, and keyed heaps resolve their items from labels, the sample value or the canonical label identity. Co-Authored-By: Claude Opus 5.5 --- .../src/expressions/mod.rs | 57 ++ .../src/physical_planner/precompute.rs | 198 ++++++- .../tests/precompute_raw_samples.rs | 495 ++++++++++++++++++ 3 files changed, 733 insertions(+), 17 deletions(-) create mode 100644 crates/integration-tests/tests/precompute_raw_samples.rs diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs index 2916cca7..1ff7b4db 100644 --- a/crates/asap-physical-operators/src/expressions/mod.rs +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -23,6 +23,16 @@ pub enum Expression { labels: Vec, without: bool, }, + /// One label of a label map; an absent label reads as empty, as in PromQL. + Label { + column: usize, + name: String, + }, + /// Canonical encoding of a complete label map, identical to + /// `promql_rows::encode_series_identity`. + LabelIdentity { + column: usize, + }, Literal { value: Value, dtype: DataType, @@ -119,6 +129,21 @@ impl Expression { } Ok((expected, false)) } + Label { column, .. } | LabelIdentity { column } => { + if plain(input, *column)? + != ( + &DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }, + false, + ) + { + return Err(invalid("label read requires a non-null Utf8 map")); + } + Ok((DataType::Utf8, false)) + } Column(i) => { let (t, n) = plain(input, *i)?; Ok((t.clone(), n)) @@ -251,6 +276,38 @@ impl Expression { } } Planner(expression) => expression.evaluate(row)?, + Label { column, name } => { + let Value::Map(entries) = &row[*column] else { + return Err(invalid("label read requires a map")); + }; + let mut found = None; + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("label read requires Utf8 entries")); + }; + if key.as_ref() == name.as_str() && found.replace(value.clone()).is_some() { + return Err(invalid("duplicate label name")); + } + } + Value::Utf8(found.unwrap_or_else(|| "".into())) + } + LabelIdentity { column } => { + let Value::Map(entries) = &row[*column] else { + return Err(invalid("label identity requires a map")); + }; + let mut labels = std::collections::BTreeMap::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("label identity requires Utf8 entries")); + }; + if labels.insert(key.to_string(), value.to_string()).is_some() { + return Err(invalid("duplicate label name")); + } + } + Value::Utf8( + crate::physical_planner::promql_rows::encode_series_identity(&labels)?.into(), + ) + } Column(i) => row[*i].clone(), Literal { value, .. } => value.clone(), Negate(v) => match v.evaluate(row)? { diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index 063bd0b8..dd66f62c 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -34,6 +34,83 @@ pub fn population_schema(family: SummaryFamilyType) -> Schema { }) } +/// Raw sample rows at a precompute boundary. `$population` holds the series' +/// complete label set, so it is the complete source identity of per-series +/// summaries; `$timestamp` is the sample time. Rows are what the boundary's +/// source scan selected; the deployment decides which rows and panes they are. +pub fn raw_sample_schema() -> Schema { + let mut schema = (*population_schema(SummaryFamilyType::Plain(DataType::Float64))).clone(); + schema.fields[1].name = "$timestamp".into(); + Arc::new(schema) +} + +pub fn raw_sample_row( + labels: &BTreeMap, + timestamp_ms: i64, + value: f64, +) -> Vec { + use crate::values::Value; + vec![ + Value::Map( + labels + .iter() + .map(|(k, v)| { + ( + Value::Utf8(k.as_str().into()), + Value::Utf8(v.as_str().into()), + ) + }) + .collect::>() + .into(), + ), + Value::Timestamp(timestamp_ms), + Value::Float64(value), + ] +} + +/// Input contract of a precompute boundary: raw sample rows for a raw time +/// series scan, otherwise the stored population of its summary state. +pub fn boundary_schema(node: &PostAsapDagNode) -> Result { + let Payload::Fallback { expression } = &node.payload else { + return source_schema(&node.output_schema); + }; + let scan = match expression { + planner_types::pre_asap::QueryExpr::TimeRange { child, .. } => child.as_ref(), + expression => expression, + }; + if !matches!( + scan, + planner_types::pre_asap::QueryExpr::Scan { + source: planner_types::pre_asap::Source::TimeSeries { .. }, + .. + } + ) { + return source_schema(&node.output_schema); + } + let logical = &node.output_schema; + // Labels may be absent from a series; its label map then omits them. + let valid = logical + .fields + .iter() + .enumerate() + .all(|(i, field)| match &field.dtype { + SummaryFamilyType::Plain(DataType::Timestamp) => { + Some(i) == logical.time_index && !field.nullable + } + SummaryFamilyType::Plain(DataType::Float64) => field.name == "value" && !field.nullable, + SummaryFamilyType::Plain(DataType::Utf8) => true, + _ => false, + }) + && logical.time_index.is_some() + && logical.fields.iter().filter(|f| f.name == "value").count() == 1; + if !valid { + return Err(invalid( + "raw sample boundary requires labels, a timestamp and one Float64 value", + )); + } + Ok(raw_sample_schema()) +} + /// Validate the adapter layout during installed-plan recovery without lowering operators. pub fn source_schema(logical: &SummarySchema) -> Result { let states = logical @@ -132,7 +209,7 @@ pub fn compile( for id in ordered { let node = nodes[&id]; if frontier.contains(&id) { - let schema = source_schema(&node.output_schema)?; + let schema = boundary_schema(node)?; sources.insert(id, InputContract::bounded(schema.clone())); outputs.insert(id, schema); continue; @@ -281,15 +358,39 @@ fn fragment( let [input] = schemas else { return Err(invalid("summary update requires one input")); }; - if update.item.is_some() - || !matches!(grouping, GroupingStrategy::PerSubpopulationInstance) - { + // Item identities resolve against the complete label set of raw + // samples; finalized readouts carry no such identity. + let raw = *input == raw_sample_schema(); + let unit_frequency = crate::capability::is_unit_sample_frequency(update); + let keyed = update.item.is_some() && !unit_frequency; + if (keyed && !raw) || !matches!(grouping, GroupingStrategy::PerSubpopulationInstance) { return Err(invalid( "precompute keyed/shared update needs its dedicated physical candidate", )); } crate::capability::validate_summary_kernel(family, update, grouping) .map_err(Error::Invalid)?; + if raw + && matches!( + update.weight_domain, + planner_types::post_asap::WeightDomain::NonNegative { + proof: planner_types::post_asap::NonNegativeWeightProof::ResetAwareCounterDerivative + } + ) + { + return Err(invalid( + "a counter-derivative weight cannot be read from raw cumulative samples", + )); + } + if keyed + && matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::CmsWithHeap) + && !matches!( + update.weight_domain, + planner_types::post_asap::WeightDomain::NonNegative { .. } + ) + { + return Err(invalid("CMS requires a nonnegative weight contract")); + } let labels = match reduction { PlannerReduction::PerEntity => Expression::Column(0), PlannerReduction::Reduce(keys) => Expression::LabelSet { @@ -302,8 +403,9 @@ fn fragment( .output_schema .fields .get(*key) + // A raw label map omits absent labels. .filter(|field| { - !field.nullable + (raw || !field.nullable) && field.dtype == SummaryFamilyType::Plain(DataType::Utf8) }) .map(|f| f.name.clone()) @@ -316,6 +418,8 @@ fn fragment( }, }; let weight = match &update.weight { + // A unit-frequency summary (HLL) observes the sample value itself. + _ if unit_frequency => Expression::Column(2), SummaryInputExpr::Constant(value) => Expression::Literal { value: crate::values::Value::Float64(*value), dtype: DataType::Float64, @@ -334,20 +438,41 @@ fn fragment( )) } }; - let project = Operator::project( - input.clone(), - vec![ - ("$population".into(), labels), - ("$window_end".into(), Expression::Column(1)), - ("value".into(), Expression::FiniteFloat64(Box::new(weight))), - ], - )? - .with_output_schema(population_schema(SummaryFamilyType::Plain( - DataType::Float64, - )))?; + let mut columns = vec![ + ("$population".into(), labels), + ("$window_end".into(), Expression::Column(1)), + ("value".into(), Expression::FiniteFloat64(Box::new(weight))), + ]; + let mut fields = population_schema(SummaryFamilyType::Plain(DataType::Float64)) + .fields + .clone(); + if keyed { + let mut items = Vec::new(); + raw_items(update.item.as_ref().expect("keyed item"), &mut items)?; + for (index, (expression, dtype)) in items.into_iter().enumerate() { + let name = format!("$item{index}"); + fields.push(planner_types::post_asap::SummaryField { + name: name.clone(), + dtype: SummaryFamilyType::Plain(dtype), + nullable: false, + }); + columns.push((name, expression)); + } + } + let item_columns = (3..fields.len()).collect::>(); + let project = Operator::project(input.clone(), columns)?.with_output_schema( + Arc::new(SummarySchema { + fields, + time_index: Some(1), + }), + )?; let projected = project.schema(); let project = add(vec![0], project)?; - let build = Operator::summary_build(projected, family.clone(), 2, Some(1), vec![0])?; + let build = if keyed { + Operator::keyed_summary_build(projected, family.clone(), 2, item_columns, vec![0])? + } else { + Operator::summary_build(projected, family.clone(), 2, Some(1), vec![0])? + }; let built = build.schema(); let build = add(vec![project], build)?; add( @@ -382,3 +507,42 @@ fn fragment( }; CompiledPhysicalDag::from_operators(sources, operators, vec![root]) } + +/// Resolve keyed item identities over raw sample rows: labels by name (absent +/// labels read as empty, as in PromQL), the sample value, or the canonical +/// encoding of the complete label set. +fn raw_items( + expr: &SummaryInputExpr, + items: &mut Vec<(Expression, DataType)>, +) -> Result<(), Error> { + match expr { + SummaryInputExpr::Column(ColumnRef::SampleValue) => { + items.push((Expression::Column(2), DataType::Float64)) + } + SummaryInputExpr::Column(ColumnRef::Named(name) | ColumnRef::Qualified { name, .. }) => { + items.push(( + Expression::Label { + column: 0, + name: name.clone(), + }, + DataType::Utf8, + )) + } + SummaryInputExpr::EntityIdentity( + planner_types::post_asap::EntityIdentity::PromqlLabelSet { excluding }, + ) if excluding.is_empty() => { + items.push((Expression::LabelIdentity { column: 0 }, DataType::Utf8)) + } + SummaryInputExpr::Tuple(parts) if !parts.is_empty() => { + for part in parts { + raw_items(part, items)?; + } + } + _ => { + return Err(invalid( + "keyed summary item does not resolve over raw samples", + )) + } + } + Ok(()) +} diff --git a/crates/integration-tests/tests/precompute_raw_samples.rs b/crates/integration-tests/tests/precompute_raw_samples.rs new file mode 100644 index 00000000..c2730be6 --- /dev/null +++ b/crates/integration-tests/tests/precompute_raw_samples.rs @@ -0,0 +1,495 @@ +//! Planner-selected summaries over raw samples compile as precompute graphs +//! and produce the same estimates as feeding their kernel sample by sample. +use std::{collections::BTreeMap, collections::BTreeSet, rc::Rc, sync::Arc}; + +use asap_aware_mapping::cost_model::DefaultCostModel; +use asap_aware_mapping::{ + search_workload, Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, + TargetSubDAG, +}; +use asap_integration_tests::fixtures::lower_promql; +use asap_physical_operators::{ + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{precompute, Source}, + runtime::{Limits, RunContext, Scope}, + summary_kernels::{exact::ExactAccumulator, weighted_frequency::WeightedFrequency}, + values::{Batch, Value}, + AggregateCore, KeyByLabelValues, Statistic, +}; +use asap_types::post_asap::{ + compile_post_asap_dag, EntityIdentity, ExactKind, PostAsapDag, PostAsapOperatorPayload, + SketchAlgorithm, SketchQuery, SummaryFamilyType, SummaryInputExpr, SummaryNode, SummaryUpdate, +}; +use asap_types::pre_asap::{expr_ir::ColumnRef, query_expr::Reduction}; +use asap_types::types::AccuracyTarget; +use futures::{executor::block_on, StreamExt}; + +type Series = BTreeMap; + +fn series(service: &str, instance: &str) -> Series { + [ + ("__name__", "m"), + ("service", service), + ("instance", instance), + ] + .into_iter() + .map(|(k, v)| (k.to_owned(), v.to_owned())) + .collect() +} + +/// Every Planner candidate for `query`: the searched selection plus each +/// summary replacement of the root. +fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { + let root = Rc::new(lower_promql(query, accuracy).expect("lowering failed")); + let mut result = SketchAlgorithmStrategy::default_cost_model() + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .filter_map(|candidate| match candidate { + ReplacementSubDAG { + replacement: Replacement::Summary(node), + .. + } => Some(node), + _ => None, + }) + .collect::>(); + let space = search_workload(vec![("query", root)]); + if let Ok(Some(selected)) = space + .global_selection(&DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + { + result.push(selected); + } + result +} + +/// Raw-input summary nodes: `(dag, raw source id, summary id)`. +fn raw_summaries(dag: &PostAsapDag) -> Vec<(u64, u64)> { + dag.nodes + .iter() + .filter(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .filter_map(|node| { + let inputs = dag + .edges + .iter() + .filter(|edge| edge.consumer == node.id) + .collect::>(); + let [edge] = inputs.as_slice() else { + return None; + }; + let source = dag.nodes.iter().find(|n| n.id == edge.producer)?; + matches!(source.payload, PostAsapOperatorPayload::Fallback { .. }) + .then_some((u64::from(source.id.0), u64::from(node.id.0))) + }) + .collect() +} + +fn samples() -> Vec<(Series, i64, f64)> { + let mut rows = Vec::new(); + for (index, (service, instance)) in [("a", "1"), ("a", "2"), ("b", "1")].iter().enumerate() { + for step in 1..=5i64 { + let value = (index as f64 + 1.0) * step as f64 + (step % 2) as f64; + rows.push((series(service, instance), step * 1000, value)); + } + } + rows +} + +fn execute( + dag: &PostAsapDag, + source: u64, + root: u64, + rows: &[(Series, i64, f64)], +) -> Vec<(Series, Arc)> { + let program = precompute::compile(dag, &[source], &[root]).unwrap_or_else(|error| { + panic!( + "raw summary {root} does not compile: {error}; source {:?}", + dag.nodes + .iter() + .find(|n| u64::from(n.id.0) == source) + .map(|n| (&n.output_schema, &n.payload)) + ); + }); + let program = serde_json::from_slice::< + asap_physical_operators::physical_planner::CompiledPhysicalDag, + >(&serde_json::to_vec(&program).unwrap()) + .unwrap(); + let schema = precompute::raw_sample_schema(); + let batch = Batch::try_new( + schema.clone(), + rows.iter() + .map(|(labels, time, value)| precompute::raw_sample_row(labels, *time, *value)) + .collect(), + ) + .unwrap(); + let sources = BTreeMap::from([( + source, + Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, + )]); + let graph = program.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 6000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut result = Vec::new(); + while let Some(batch) = stream.next().await { + for row in batch.unwrap().rows() { + let [Value::Map(labels), Value::Timestamp(6000), Value::Summary { state, .. }] = + row.as_slice() + else { + panic!("unexpected population row {row:?}"); + }; + let labels = labels + .iter() + .map(|(k, v)| match (k, v) { + (Value::Utf8(k), Value::Utf8(v)) => (k.to_string(), v.to_string()), + _ => panic!("non-label population entry"), + }) + .collect(); + result.push((labels, state.clone())); + } + } + result + }) +} + +fn population(reduction: &Reduction, dag_labels: &[String], labels: &Series) -> Series { + match reduction { + Reduction::PerEntity => labels.clone(), + Reduction::Reduce(_) => labels + .iter() + .filter(|(k, v)| dag_labels.contains(k) && !v.is_empty()) + .map(|(k, v)| (k.clone(), v.clone())) + .collect(), + } +} + +/// Item identity used by a keyed summary: labels by name, the sample value, +/// or the canonical label-set identity. +fn item(expr: &SummaryInputExpr, labels: &Series, value: f64, out: &mut Vec) { + match expr { + SummaryInputExpr::Column(ColumnRef::SampleValue) => out.push(Value::Float64(value)), + SummaryInputExpr::Column(ColumnRef::Named(name)) => out.push(Value::Utf8( + labels.get(name).cloned().unwrap_or_default().into(), + )), + SummaryInputExpr::EntityIdentity(EntityIdentity::PromqlLabelSet { excluding }) + if excluding.is_empty() => + { + out.push(Value::Utf8(serde_json::to_string(labels).unwrap().into())) + } + SummaryInputExpr::Tuple(items) => items.iter().for_each(|i| item(i, labels, value, out)), + other => panic!("unsupported fixture item {other:?}"), + } +} + +fn weight(update: &SummaryUpdate, value: f64) -> f64 { + match update.weight { + SummaryInputExpr::Constant(weight) => weight, + _ => value, + } +} + +/// Estimates that identify a state's content for comparison. +fn readouts(state: &dyn AggregateCore, family: &SummaryFamilyType) -> Vec { + if let Some(exact) = state.as_any().downcast_ref::() { + let SummaryFamilyType::ExactAggregate(kind, _) = family else { + unreachable!() + }; + let statistic = match kind { + ExactKind::Sum => Statistic::Sum, + ExactKind::Count => Statistic::Count, + ExactKind::Min => Statistic::Min, + ExactKind::Max => Statistic::Max, + ExactKind::Rate => Statistic::Rate, + ExactKind::Increase => Statistic::Increase, + other => panic!("unexpected exact kind {other:?}"), + }; + return vec![exact + .readout(statistic, None, None::<&KeyByLabelValues>) + .unwrap() + .unwrap()]; + } + let SummaryFamilyType::Sketch(kind, _) = family else { + panic!("sketch state for exact family") + }; + match kind.algorithm() { + SketchAlgorithm::Kll | SketchAlgorithm::DDSketch => [0.1, 0.5, 0.9] + .into_iter() + .map(|q| state.estimate(&SketchQuery::Quantile { q }).unwrap()) + .collect(), + SketchAlgorithm::Hll => vec![state.estimate(&SketchQuery::Cardinality).unwrap()], + other => panic!("unexpected unkeyed sketch {other:?}"), + } +} + +/// Compile one raw-input summary, execute it over `rows`, and compare each +/// population with its kernel fed sample by sample. Returns the family label, +/// or the family when it has no native state. +fn check( + query: &str, + dag: &PostAsapDag, + source: u64, + root: u64, + rows: &[(Series, i64, f64)], +) -> Result { + let node = dag + .nodes + .iter() + .find(|n| u64::from(n.id.0) == root) + .unwrap(); + let PostAsapOperatorPayload::SummaryAgg { + family, + input, + reduction, + grouping, + } = &node.payload + else { + unreachable!() + }; + let source_node = dag + .nodes + .iter() + .find(|n| u64::from(n.id.0) == source) + .unwrap(); + let keys = match reduction { + Reduction::Reduce(keys) => keys + .keys() + .iter() + .map(|i| source_node.output_schema.fields[*i].name.clone()) + .collect(), + Reduction::PerEntity => vec![], + }; + if asap_physical_operators::capability::validate_native_family(family).is_err() { + // Families without a native state (e.g. plain CMS, UnivMon) + // are outside precompute execution; their compile must fail. + assert!(precompute::compile(dag, &[source], &[root]).is_err()); + return Err(format!("{family:?}")); + } + let actual = execute(dag, source, root, rows); + let label = match family { + SummaryFamilyType::ExactAggregate(kind, _) => format!("{kind:?}"), + SummaryFamilyType::Sketch(kind, _) => format!("{:?}", kind.algorithm()), + other => format!("{other:?}"), + }; + if let SummaryFamilyType::Sketch(kind, _) = family { + if let (Some(keyed), false) = (&input.item, kind.algorithm() == &SketchAlgorithm::Hll) { + // Keyed heaps: every item's estimated weight is its exact + // total at this scale (no collisions in the fixture). + let mut expected = BTreeMap::>::new(); + for (labels, _, value) in rows { + let mut items = Vec::new(); + item(keyed, labels, *value, &mut items); + *expected + .entry(population(reduction, &keys, labels)) + .or_default() + .entry(format!("{items:?}")) + .or_default() += weight(input, *value); + } + assert_eq!(actual.len(), expected.len(), "{query}"); + for (labels, state) in &actual { + let heap = state.as_any().downcast_ref::().unwrap(); + let got = heap + .rows(usize::MAX >> 1) + .into_iter() + .map(|mut row| { + let Some(Value::Float64(score)) = row.pop() else { + panic!("heap score") + }; + (format!("{row:?}"), score) + }) + .collect::>(); + assert_eq!(&got, &expected[labels], "{query}"); + } + return Ok(label); + } + } + let mut expected = + BTreeMap::>::new(); + for (labels, time, value) in rows { + let updater = expected + .entry(population(reduction, &keys, labels)) + .or_insert_with(|| create_planner_accumulator(family, input, grouping).unwrap()); + let unit = input.item.is_some(); + updater.update_single(if unit { *value } else { weight(input, *value) }, *time); + } + assert_eq!(actual.len(), expected.len(), "{query}"); + for (labels, state) in actual { + let reference = expected[&labels].snapshot_accumulator(); + assert_eq!( + readouts(state.as_ref(), family), + readouts(reference.as_ref(), family), + "{query}: {labels:?}" + ); + } + Ok(label) +} + +// Every raw-input summary selected by Planner compiles over raw sample rows, +// and each population's estimates equal feeding its kernel sample by sample. +#[test] +fn raw_sample_summaries_compile_and_match_their_kernels() { + let exact = AccuracyTarget::Exact; + let sketch = AccuracyTarget::Epsilon(0.02); + let queries = [ + ("sum_over_time(m[5m])", &exact), + ("count_over_time(m[5m])", &exact), + ("min_over_time(m[5m])", &exact), + ("max_over_time(m[5m])", &exact), + ("rate(m[5m])", &exact), + ("increase(m[5m])", &exact), + ("sum by (service) (sum_over_time(m[5m]))", &exact), + ("sum by (service) (rate(m[5m]))", &exact), + ("topk(2, sum_over_time(m[5m]))", &exact), + ("quantile_over_time(0.9, m[5m])", &sketch), + ("sum by (service) (quantile_over_time(0.9, m[5m]))", &sketch), + ("quantile by (service) (0.9, m)", &sketch), + ("distinct_over_time(m[5m])", &sketch), + ("count(m)", &sketch), + ("topk(2, m)", &sketch), + ( + "topk(2, sum by (service) (count_over_time(m[5m])))", + &sketch, + ), + ]; + let rows = samples(); + let mut families = BTreeSet::new(); + let mut unsupported = BTreeSet::new(); + for (query, accuracy) in queries { + for candidate in candidates(query, accuracy.clone()) { + let dag = compile_post_asap_dag(&candidate).unwrap(); + for (source, root) in raw_summaries(&dag) { + match check(query, &dag, source, root, &rows) { + Ok(family) => families.insert(family), + Err(family) => unsupported.insert(family), + }; + } + } + } + println!("raw summary families: {families:?}; without native state: {unsupported:?}"); + for family in [ + "Sum", "Count", "Min", "Max", "Rate", "Increase", "Kll", "DDSketch", "Hll", + ] { + assert!( + families.contains(family), + "no {family} fixture: {families:?}" + ); + } +} + +/// Replace the raw summary of `sum by (service) (sum_over_time(m[5m]))` with +/// another update, keeping its raw input and grouping. +fn grouped_raw_summary(family: SummaryFamilyType, input: SummaryUpdate) -> (PostAsapDag, u64, u64) { + let candidate = candidates( + "sum by (service) (sum_over_time(m[5m]))", + AccuracyTarget::Exact, + ) + .pop() + .unwrap(); + let mut dag = compile_post_asap_dag(&candidate).unwrap(); + let (source, root) = raw_summaries(&dag)[0]; + let node = dag + .nodes + .iter_mut() + .find(|n| u64::from(n.id.0) == root) + .unwrap(); + let PostAsapOperatorPayload::SummaryAgg { + family: old, + input: update, + .. + } = &mut node.payload + else { + unreachable!() + }; + for field in &mut node.output_schema.fields { + if field.dtype == *old { + field.dtype = family.clone(); + } + } + *old = family; + *update = input; + let schema = node.output_schema.clone(); + for edge in dag + .edges + .iter_mut() + .filter(|e| u64::from(e.producer.0) == root) + { + edge.intermediate_schema = schema.clone(); + } + (dag, source, root) +} + +// Keyed heaps over raw samples resolve items from the series label set and +// estimate each item's exact total; invalid weight contracts do not compile. +#[test] +fn raw_sample_heaps_resolve_items_from_labels() { + use asap_types::post_asap::{ + GroupingStrategy::PerSubpopulationInstance, NonNegativeWeightProof, SketchKind, + SketchParams, WeightDomain, + }; + let heap = |algorithm, params| { + SummaryFamilyType::Sketch(SketchKind::new(algorithm, params), PerSubpopulationInstance) + }; + let cms = heap( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 64, + depth: 3, + heap_size: 8, + }, + ); + let count_sketch = heap( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 64, + depth: 3, + heap_size: 8, + }, + ); + let identity = SummaryUpdate { + item: Some(SummaryInputExpr::EntityIdentity( + EntityIdentity::PromqlLabelSet { excluding: vec![] }, + )), + weight: SummaryInputExpr::Constant(1.0), + weight_domain: WeightDomain::NonNegative { + proof: NonNegativeWeightProof::UnitCount, + }, + }; + let by_instance = SummaryUpdate { + item: Some(SummaryInputExpr::Tuple(vec![SummaryInputExpr::Column( + ColumnRef::Named("instance".into()), + )])), + weight: SummaryInputExpr::Column(ColumnRef::SampleValue), + weight_domain: WeightDomain::UnknownOrSigned, + }; + let rows = samples(); + for (family, input) in [ + (cms.clone(), identity.clone()), + (count_sketch.clone(), identity), + (count_sketch, by_instance.clone()), + ] { + let (dag, source, root) = grouped_raw_summary(family, input); + assert!(check("heap", &dag, source, root, &rows).is_ok()); + } + let signed_cms = grouped_raw_summary(cms.clone(), by_instance.clone()); + assert!(precompute::compile(&signed_cms.0, &[signed_cms.1], &[signed_cms.2]).is_err()); + let derivative = grouped_raw_summary( + cms, + SummaryUpdate { + weight_domain: WeightDomain::NonNegative { + proof: NonNegativeWeightProof::ResetAwareCounterDerivative, + }, + ..by_instance + }, + ); + assert!( + precompute::compile(&derivative.0, &[derivative.1], &[derivative.2]).is_err(), + "raw samples are cumulative counters, not their derivative" + ); +} From c5bc826f2236659acce6009e0c2b89b78604b6d9 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 06:43:12 +0000 Subject: [PATCH 49/59] fix(physical): canonical raw identities and checked raw heap items Raw sample rows drop empty label values so one series has one population. Heap items resolve names against the scan (the value column, the series identity, labels; other scan columns are rejected), and identity items may exclude labels, as `topk by` emits. Unit-frequency HLL updates are raw-only and never apply to keyed families. Tests cover Planner-generated heaps, missing and empty labels, and `without` grouping. Co-Authored-By: Claude Opus 5.5 --- .../src/expressions/mod.rs | 11 +- .../src/physical_planner/precompute.rs | 102 +++++++++--- .../tests/precompute_raw_samples.rs | 151 +++++++++++++++--- 3 files changed, 217 insertions(+), 47 deletions(-) diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs index 1ff7b4db..a16a3bde 100644 --- a/crates/asap-physical-operators/src/expressions/mod.rs +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -28,10 +28,12 @@ pub enum Expression { column: usize, name: String, }, - /// Canonical encoding of a complete label map, identical to + /// Canonical encoding of a label map less `excluding`, identical to /// `promql_rows::encode_series_identity`. LabelIdentity { column: usize, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + excluding: Vec, }, Literal { value: Value, @@ -129,7 +131,7 @@ impl Expression { } Ok((expected, false)) } - Label { column, .. } | LabelIdentity { column } => { + Label { column, .. } | LabelIdentity { column, .. } => { if plain(input, *column)? != ( &DataType::Map { @@ -291,7 +293,7 @@ impl Expression { } Value::Utf8(found.unwrap_or_else(|| "".into())) } - LabelIdentity { column } => { + LabelIdentity { column, excluding } => { let Value::Map(entries) = &row[*column] else { return Err(invalid("label identity requires a map")); }; @@ -300,6 +302,9 @@ impl Expression { let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { return Err(invalid("label identity requires Utf8 entries")); }; + if excluding.iter().any(|label| label.as_str() == key.as_ref()) { + continue; + } if labels.insert(key.to_string(), value.to_string()).is_some() { return Err(invalid("duplicate label name")); } diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index dd66f62c..38663542 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -1,4 +1,5 @@ //! Compile immutable summary-input computation with explicit population and pane identity. +use super::promql_rows::SERIES_IDENTITY_COLUMN as SERIES_IDENTITY; use super::*; use planner_types::{ post_asap::{ExecutionTiming, GroupingStrategy, SummarySchema}, @@ -36,14 +37,18 @@ pub fn population_schema(family: SummaryFamilyType) -> Schema { /// Raw sample rows at a precompute boundary. `$population` holds the series' /// complete label set, so it is the complete source identity of per-series -/// summaries; `$timestamp` is the sample time. Rows are what the boundary's -/// source scan selected; the deployment decides which rows and panes they are. +/// summaries; `$timestamp` is the sample time and `value` a finite sample +/// (stale markers are not samples). Rows are what the boundary's source scan +/// selected; the deployment decides which rows and panes they are. Build rows +/// with [`raw_sample_row`], which makes the label set canonical. pub fn raw_sample_schema() -> Schema { let mut schema = (*population_schema(SummaryFamilyType::Plain(DataType::Float64))).clone(); schema.fields[1].name = "$timestamp".into(); Arc::new(schema) } +/// A raw sample row whose label set is sorted, unique and omits empty values, +/// so one series always has one population identity. pub fn raw_sample_row( labels: &BTreeMap, timestamp_ms: i64, @@ -54,6 +59,7 @@ pub fn raw_sample_row( Value::Map( labels .iter() + .filter(|(_, v)| !v.is_empty()) .map(|(k, v)| { ( Value::Utf8(k.as_str().into()), @@ -101,6 +107,10 @@ pub fn boundary_schema(node: &PostAsapDagNode) -> Result { SummaryFamilyType::Plain(DataType::Utf8) => true, _ => false, }) + && !logical + .fields + .iter() + .any(|f| f.name.starts_with('$') && f.name != SERIES_IDENTITY) && logical.time_index.is_some() && logical.fields.iter().filter(|f| f.name == "value").count() == 1; if !valid { @@ -361,7 +371,16 @@ fn fragment( // Item identities resolve against the complete label set of raw // samples; finalized readouts carry no such identity. let raw = *input == raw_sample_schema(); - let unit_frequency = crate::capability::is_unit_sample_frequency(update); + // A unit-frequency summary (HLL) observes each raw sample value. + let unit_frequency = raw + && crate::capability::is_unit_sample_frequency(update) + && !matches!(family, SummaryFamilyType::Sketch(kind, _) if matches!( + kind.algorithm(), + planner_types::post_asap::SketchAlgorithm::Cms + | planner_types::post_asap::SketchAlgorithm::CountSketch + | planner_types::post_asap::SketchAlgorithm::CmsWithHeap + | planner_types::post_asap::SketchAlgorithm::CountSketchWithHeap + )); let keyed = update.item.is_some() && !unit_frequency; if (keyed && !raw) || !matches!(grouping, GroupingStrategy::PerSubpopulationInstance) { return Err(invalid( @@ -403,9 +422,11 @@ fn fragment( .output_schema .fields .get(*key) - // A raw label map omits absent labels. + // A raw label map omits absent labels; the + // series identity is not one of its labels. .filter(|field| { (raw || !field.nullable) + && field.name != SERIES_IDENTITY && field.dtype == SummaryFamilyType::Plain(DataType::Utf8) }) .map(|f| f.name.clone()) @@ -418,7 +439,6 @@ fn fragment( }, }; let weight = match &update.weight { - // A unit-frequency summary (HLL) observes the sample value itself. _ if unit_frequency => Expression::Column(2), SummaryInputExpr::Constant(value) => Expression::Literal { value: crate::values::Value::Float64(*value), @@ -448,7 +468,11 @@ fn fragment( .clone(); if keyed { let mut items = Vec::new(); - raw_items(update.item.as_ref().expect("keyed item"), &mut items)?; + raw_items( + update.item.as_ref().expect("keyed item"), + &parents[0].output_schema, + &mut items, + )?; for (index, (expression, dtype)) in items.into_iter().enumerate() { let name = format!("$item{index}"); fields.push(planner_types::post_asap::SummaryField { @@ -508,34 +532,70 @@ fn fragment( CompiledPhysicalDag::from_operators(sources, operators, vec![root]) } -/// Resolve keyed item identities over raw sample rows: labels by name (absent -/// labels read as empty, as in PromQL), the sample value, or the canonical -/// encoding of the complete label set. +/// Resolve keyed item identities over raw sample rows: labels (absent labels +/// read as empty, as in PromQL), the sample value, or the canonical encoding +/// of the label set less excluded labels. fn raw_items( expr: &SummaryInputExpr, + scan: &SummarySchema, items: &mut Vec<(Expression, DataType)>, ) -> Result<(), Error> { + // Open PromQL scans need not list every label, so any name that is not + // another scan column (value, time, series identity) reads as a label. + let label = |column: &ColumnRef| match column { + ColumnRef::Named(name) | ColumnRef::Qualified { name, .. } + if !name.starts_with('$') + && scan.fields.iter().all(|f| { + &f.name != name || f.dtype == SummaryFamilyType::Plain(DataType::Utf8) + }) => + { + Some(name.clone()) + } + _ => None, + }; + let identity = |excluding: Vec| { + ( + Expression::LabelIdentity { + column: 0, + excluding, + }, + DataType::Utf8, + ) + }; match expr { SummaryInputExpr::Column(ColumnRef::SampleValue) => { items.push((Expression::Column(2), DataType::Float64)) } - SummaryInputExpr::Column(ColumnRef::Named(name) | ColumnRef::Qualified { name, .. }) => { - items.push(( - Expression::Label { - column: 0, - name: name.clone(), - }, - DataType::Utf8, - )) + SummaryInputExpr::Column(ColumnRef::Named(name) | ColumnRef::Qualified { name, .. }) + if name == "value" => + { + items.push((Expression::Column(2), DataType::Float64)) + } + SummaryInputExpr::Column(ColumnRef::Named(name) | ColumnRef::Qualified { name, .. }) + if name == SERIES_IDENTITY => + { + items.push(identity(vec![])) } + SummaryInputExpr::Column(column) if label(column).is_some() => items.push(( + Expression::Label { + column: 0, + name: label(column).expect("resolved label"), + }, + DataType::Utf8, + )), SummaryInputExpr::EntityIdentity( planner_types::post_asap::EntityIdentity::PromqlLabelSet { excluding }, - ) if excluding.is_empty() => { - items.push((Expression::LabelIdentity { column: 0 }, DataType::Utf8)) - } + ) => items.push(identity( + excluding + .iter() + .map(|column| { + label(column).ok_or_else(|| invalid("excluded identity label is not a label")) + }) + .collect::>()?, + )), SummaryInputExpr::Tuple(parts) if !parts.is_empty() => { for part in parts { - raw_items(part, items)?; + raw_items(part, scan, items)?; } } _ => { diff --git a/crates/integration-tests/tests/precompute_raw_samples.rs b/crates/integration-tests/tests/precompute_raw_samples.rs index c2730be6..dde29fb1 100644 --- a/crates/integration-tests/tests/precompute_raw_samples.rs +++ b/crates/integration-tests/tests/precompute_raw_samples.rs @@ -27,17 +27,27 @@ use futures::{executor::block_on, StreamExt}; type Series = BTreeMap; -fn series(service: &str, instance: &str) -> Series { +/// A series label set; `None` omits the label. An empty value is present +/// in the input but is not part of the series identity. +fn series(service: Option<&str>, instance: &str) -> Series { [ - ("__name__", "m"), + ("__name__", Some("m")), ("service", service), - ("instance", instance), + ("instance", Some(instance)), ] .into_iter() - .map(|(k, v)| (k.to_owned(), v.to_owned())) + .filter_map(|(k, v)| Some((k.to_owned(), v?.to_owned()))) .collect() } +fn canonical(labels: &Series) -> Series { + labels + .iter() + .filter(|(_, v)| !v.is_empty()) + .map(|(k, v)| (k.clone(), v.clone())) + .collect() +} + /// Every Planner candidate for `query`: the searched selection plus each /// summary replacement of the root. fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { @@ -86,10 +96,17 @@ fn raw_summaries(dag: &PostAsapDag) -> Vec<(u64, u64)> { fn samples() -> Vec<(Series, i64, f64)> { let mut rows = Vec::new(); - for (index, (service, instance)) in [("a", "1"), ("a", "2"), ("b", "1")].iter().enumerate() { + let series_set = [ + (Some("a"), "1"), + (Some("a"), "2"), + (Some("b"), "1"), + (None, "3"), + (Some("b"), ""), + ]; + for (index, (service, instance)) in series_set.iter().enumerate() { for step in 1..=5i64 { let value = (index as f64 + 1.0) * step as f64 + (step % 2) as f64; - rows.push((series(service, instance), step * 1000, value)); + rows.push((series(*service, instance), step * 1000, value)); } } rows @@ -160,13 +177,21 @@ fn execute( }) } +/// PromQL grouping of a canonical label set: `by` keeps the named labels, +/// `without` drops them and `__name__`. fn population(reduction: &Reduction, dag_labels: &[String], labels: &Series) -> Series { + let labels = canonical(labels); match reduction { - Reduction::PerEntity => labels.clone(), - Reduction::Reduce(_) => labels - .iter() - .filter(|(k, v)| dag_labels.contains(k) && !v.is_empty()) - .map(|(k, v)| (k.clone(), v.clone())) + Reduction::PerEntity => labels, + Reduction::Reduce(keys) => labels + .into_iter() + .filter(|(k, _)| { + if keys.is_without() { + k != "__name__" && !dag_labels.contains(k) + } else { + dag_labels.contains(k) + } + }) .collect(), } } @@ -176,13 +201,27 @@ fn population(reduction: &Reduction, dag_labels: &[String], labels: &Series) -> fn item(expr: &SummaryInputExpr, labels: &Series, value: f64, out: &mut Vec) { match expr { SummaryInputExpr::Column(ColumnRef::SampleValue) => out.push(Value::Float64(value)), + SummaryInputExpr::Column(ColumnRef::Named(name)) if name == "value" => { + out.push(Value::Float64(value)) + } SummaryInputExpr::Column(ColumnRef::Named(name)) => out.push(Value::Utf8( - labels.get(name).cloned().unwrap_or_default().into(), + canonical(labels) + .get(name) + .cloned() + .unwrap_or_default() + .into(), )), - SummaryInputExpr::EntityIdentity(EntityIdentity::PromqlLabelSet { excluding }) - if excluding.is_empty() => - { - out.push(Value::Utf8(serde_json::to_string(labels).unwrap().into())) + SummaryInputExpr::EntityIdentity(EntityIdentity::PromqlLabelSet { excluding }) => { + let mut identity = canonical(labels); + for column in excluding { + let ColumnRef::Named(name) = column else { + panic!("unsupported fixture exclusion {column:?}") + }; + identity.remove(name); + } + out.push(Value::Utf8( + serde_json::to_string(&identity).unwrap().into(), + )) } SummaryInputExpr::Tuple(items) => items.iter().for_each(|i| item(i, labels, value, out)), other => panic!("unsupported fixture item {other:?}"), @@ -357,24 +396,41 @@ fn raw_sample_summaries_compile_and_match_their_kernels() { "topk(2, sum by (service) (count_over_time(m[5m])))", &sketch, ), + ("topk(2, sum_over_time(m[5m]))", &sketch), + ("topk by (service) (2, sum_over_time(m[5m]))", &sketch), ]; let rows = samples(); let mut families = BTreeSet::new(); let mut unsupported = BTreeSet::new(); + let mut checked = BTreeMap::new(); for (query, accuracy) in queries { for candidate in candidates(query, accuracy.clone()) { let dag = compile_post_asap_dag(&candidate).unwrap(); for (source, root) in raw_summaries(&dag) { match check(query, &dag, source, root, &rows) { - Ok(family) => families.insert(family), - Err(family) => unsupported.insert(family), - }; + Ok(family) => { + families.insert(family); + *checked.entry(query).or_insert(0) += 1; + } + Err(family) => { + unsupported.insert(family); + } + } } } } - println!("raw summary families: {families:?}; without native state: {unsupported:?}"); + println!("checked {checked:?}; families {families:?}; without native state {unsupported:?}"); for family in [ - "Sum", "Count", "Min", "Max", "Rate", "Increase", "Kll", "DDSketch", "Hll", + "Sum", + "Count", + "Min", + "Max", + "Rate", + "Increase", + "Kll", + "DDSketch", + "Hll", + "CountSketchWithHeap", ] { assert!( families.contains(family), @@ -384,7 +440,7 @@ fn raw_sample_summaries_compile_and_match_their_kernels() { } /// Replace the raw summary of `sum by (service) (sum_over_time(m[5m]))` with -/// another update, keeping its raw input and grouping. +/// another update, keeping its raw input and reduction. fn grouped_raw_summary(family: SummaryFamilyType, input: SummaryUpdate) -> (PostAsapDag, u64, u64) { let candidate = candidates( "sum by (service) (sum_over_time(m[5m]))", @@ -485,11 +541,60 @@ fn raw_sample_heaps_resolve_items_from_labels() { weight_domain: WeightDomain::NonNegative { proof: NonNegativeWeightProof::ResetAwareCounterDerivative, }, - ..by_instance + ..by_instance.clone() }, ); assert!( precompute::compile(&derivative.0, &[derivative.1], &[derivative.2]).is_err(), "raw samples are cumulative counters, not their derivative" ); + // A scan's time column is not a label; it cannot silently read as empty. + let time_item = grouped_raw_summary( + heap( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 64, + depth: 3, + heap_size: 8, + }, + ), + SummaryUpdate { + item: Some(SummaryInputExpr::Column(ColumnRef::Named("ts".into()))), + ..by_instance + }, + ); + assert!(precompute::compile(&time_item.0, &[time_item.1], &[time_item.2]).is_err()); +} + +// `without` grouping over raw samples drops the listed labels and `__name__`. +#[test] +fn raw_sample_without_grouping_drops_labels_and_name() { + use asap_types::pre_asap::query_expr::GroupKeys; + let family = + SummaryFamilyType::ExactAggregate(ExactKind::Sum, asap_types::post_asap::ExactParams::Sum); + let (mut dag, source, root) = + grouped_raw_summary(family, SummaryUpdate::column(ColumnRef::SampleValue)); + let service = dag + .nodes + .iter() + .find(|n| u64::from(n.id.0) == source) + .unwrap() + .output_schema + .fields + .iter() + .position(|f| f.name == "service") + .unwrap(); + let node = dag + .nodes + .iter_mut() + .find(|n| u64::from(n.id.0) == root) + .unwrap(); + let PostAsapOperatorPayload::SummaryAgg { reduction, .. } = &mut node.payload else { + unreachable!() + }; + *reduction = Reduction::Reduce(GroupKeys::without(vec![service])); + assert_eq!( + check("without", &dag, source, root, &samples()), + Ok("Sum".into()) + ); } From f574c00aeddcb88ddd8c1a79fa9a40252dbdf421 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 06:45:57 +0000 Subject: [PATCH 50/59] fix(physical): keep raw unit-frequency updates to unkeyed sketches State the canonical label-set obligation and assert per-query coverage. Co-Authored-By: Claude Opus 5.5 --- .../src/physical_planner/precompute.rs | 7 ++++--- .../integration-tests/tests/precompute_raw_samples.rs | 10 ++++++++++ 2 files changed, 14 insertions(+), 3 deletions(-) diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index 38663542..4fa40c68 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -39,8 +39,9 @@ pub fn population_schema(family: SummaryFamilyType) -> Schema { /// complete label set, so it is the complete source identity of per-series /// summaries; `$timestamp` is the sample time and `value` a finite sample /// (stale markers are not samples). Rows are what the boundary's source scan -/// selected; the deployment decides which rows and panes they are. Build rows -/// with [`raw_sample_row`], which makes the label set canonical. +/// selected; the deployment decides which rows and panes they are. Label sets +/// must be canonical (sorted, unique, no empty values), since they are the +/// population identity: build rows with [`raw_sample_row`]. pub fn raw_sample_schema() -> Schema { let mut schema = (*population_schema(SummaryFamilyType::Plain(DataType::Float64))).clone(); schema.fields[1].name = "$timestamp".into(); @@ -374,7 +375,7 @@ fn fragment( // A unit-frequency summary (HLL) observes each raw sample value. let unit_frequency = raw && crate::capability::is_unit_sample_frequency(update) - && !matches!(family, SummaryFamilyType::Sketch(kind, _) if matches!( + && matches!(family, SummaryFamilyType::Sketch(kind, _) if !matches!( kind.algorithm(), planner_types::post_asap::SketchAlgorithm::Cms | planner_types::post_asap::SketchAlgorithm::CountSketch diff --git a/crates/integration-tests/tests/precompute_raw_samples.rs b/crates/integration-tests/tests/precompute_raw_samples.rs index dde29fb1..ade52ec8 100644 --- a/crates/integration-tests/tests/precompute_raw_samples.rs +++ b/crates/integration-tests/tests/precompute_raw_samples.rs @@ -420,6 +420,16 @@ fn raw_sample_summaries_compile_and_match_their_kernels() { } } println!("checked {checked:?}; families {families:?}; without native state {unsupported:?}"); + for query in [ + "topk by (service) (2, sum_over_time(m[5m]))", + "quantile by (service) (0.9, m)", + "distinct_over_time(m[5m])", + ] { + assert!( + checked.contains_key(query), + "{query} has no checked raw summary" + ); + } for family in [ "Sum", "Count", From d8b185e2a2174e2b6211bbe00a948347456ed786 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 13:31:11 +0000 Subject: [PATCH 51/59] feat(kernels): validate exact state on deserialization Deployments persisting ExactAccumulator had to mirror its serde shape to reject payloads whose population states differ from the declared family. Deserialization now performs that check itself. Co-Authored-By: Claude Opus 5.5 --- Cargo.lock | 1 + crates/asap-physical-operators/Cargo.toml | 1 + .../src/summary_kernels/exact.rs | 102 ++++++++++++++++++ 3 files changed, 104 insertions(+) diff --git a/Cargo.lock b/Cargo.lock index 7f0a3755..a3fe727e 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -385,6 +385,7 @@ dependencies = [ "asap-types", "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369)", "futures", + "rmp-serde", "serde", "serde_json", "thiserror 2.0.18", diff --git a/crates/asap-physical-operators/Cargo.toml b/crates/asap-physical-operators/Cargo.toml index 74c172e3..cec452e5 100644 --- a/crates/asap-physical-operators/Cargo.toml +++ b/crates/asap-physical-operators/Cargo.toml @@ -14,5 +14,6 @@ thiserror = "2" [dev-dependencies] +rmp-serde = "1" asap-aware-mapping = { path = "../asap-aware-mapping" } asap-frontend-promql = { path = "../frontend-promql" } diff --git a/crates/asap-physical-operators/src/summary_kernels/exact.rs b/crates/asap-physical-operators/src/summary_kernels/exact.rs index cb9d6099..aeb0c6db 100644 --- a/crates/asap-physical-operators/src/summary_kernels/exact.rs +++ b/crates/asap-physical-operators/src/summary_kernels/exact.rs @@ -19,13 +19,50 @@ enum ScalarState { /// Both the family and population layout survive persistence. Sharing counter /// arithmetic never authorizes a Rate state to answer an Increase readout. +/// +/// Deserialization validates the payload against its declared family, so +/// deployments can persist this state with any serde format without mirroring +/// its shape. #[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(try_from = "ExactPayload")] pub struct ExactAccumulator { family: SummaryFamilyType, scalar: ScalarState, keyed: Option>, } +#[derive(Deserialize)] +struct ExactPayload { + family: SummaryFamilyType, + scalar: ScalarState, + keyed: Option>, +} + +impl TryFrom for ExactAccumulator { + type Error = String; + + fn try_from(payload: ExactPayload) -> Result { + let empty = Self::new(payload.family, payload.keyed.is_some())?; + let same_kind = |state: &ScalarState| { + std::mem::discriminant(state) == std::mem::discriminant(&empty.scalar) + }; + if !same_kind(&payload.scalar) + || payload + .keyed + .iter() + .flat_map(HashMap::values) + .any(|s| !same_kind(s)) + { + return Err("exact payload differs from its declared Planner family".into()); + } + Ok(Self { + scalar: payload.scalar, + keyed: payload.keyed, + ..empty + }) + } +} + /// Planned readout of an exact summary. `lookback_ms` is the logical PromQL /// counter window; the evaluation range is resolved from it at run time. #[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] @@ -235,3 +272,68 @@ impl AggregateCore for ExactAccumulator { }) } } + +#[cfg(test)] +mod tests { + use super::*; + use serde::Serialize; + + #[derive(Serialize)] + struct Payload { + family: SummaryFamilyType, + scalar: ScalarState, + keyed: Option>, + } + + fn decode(payload: &Payload) -> Result { + rmp_serde::from_slice(&rmp_serde::to_vec_named(payload).unwrap()) + } + + fn sum() -> SummaryFamilyType { + SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + } + + // A persisted exact state decodes back to the same family, layout and readout. + #[test] + fn serialized_state_round_trips() { + let mut state = ExactAccumulator::new(sum(), true).unwrap(); + let key = KeyByLabelValues::new_with_labels(vec!["a".into()]); + state.update(Some(&key), 2.5, 0); + let bytes = rmp_serde::to_vec_named(&state).unwrap(); + let restored: ExactAccumulator = rmp_serde::from_slice(&bytes).unwrap(); + assert_eq!(restored.family(), &sum()); + assert_eq!( + restored.readout(Statistic::Sum, None, Some(&key)).unwrap(), + Some(2.5) + ); + } + + // Decoding rejects population states that differ from the declared family. + #[test] + fn decode_rejects_state_of_another_family() { + let scalar = Payload { + family: sum(), + scalar: ScalarState::Count(3), + keyed: None, + }; + assert!(decode(&scalar).is_err()); + let key = KeyByLabelValues::new_with_labels(vec!["a".into()]); + let keyed = Payload { + family: sum(), + scalar: ScalarState::Sum(0.0), + keyed: Some(HashMap::from([(key, ScalarState::Max(Some(1.0)))])), + }; + assert!(decode(&keyed).is_err()); + } + + // Decoding rejects families that have no exact Planner state. + #[test] + fn decode_rejects_unsupported_family() { + let payload = Payload { + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Count), + scalar: ScalarState::Sum(0.0), + keyed: None, + }; + assert!(decode(&payload).is_err()); + } +} From 210cbe5ecb0f301a058e2ff2cf67fde608f4025c Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 13:31:11 +0000 Subject: [PATCH 52/59] feat(kernels): adopt and expose UnivMon sketches, answer UnivMon readouts The backend kept a UnivMon shim because Planner's accumulator hid its sketch and answered no readouts. from_sketch/sketch let a deployment encode and restore it with sketchlib's codec; estimate answers count, distinct, L2 and entropy. Co-Authored-By: Claude Opus 5.5 --- .../src/summary_kernels/univmon.rs | 119 +++++++++++++++++- 1 file changed, 113 insertions(+), 6 deletions(-) diff --git a/crates/asap-physical-operators/src/summary_kernels/univmon.rs b/crates/asap-physical-operators/src/summary_kernels/univmon.rs index bffd8afa..27ca7fcc 100644 --- a/crates/asap-physical-operators/src/summary_kernels/univmon.rs +++ b/crates/asap-physical-operators/src/summary_kernels/univmon.rs @@ -2,9 +2,25 @@ use crate::AggregateCore; use asap_sketchlib::{DataInput, UnivMon}; +use planner_types::{post_asap::SketchQuery, pre_asap::ColumnRef}; type Error = Box; +fn check_dimensions( + heap_size: usize, + rows: usize, + cols: usize, + layers: usize, +) -> Result<(), Error> { + if heap_size == 0 || cols == 0 || !(1..=20).contains(&rows) || !(1..=64).contains(&layers) { + return Err("invalid UnivMon dimensions".into()); + } + rows.checked_mul(cols) + .and_then(|n| n.checked_mul(layers)) + .ok_or("UnivMon dimensions overflow")?; + Ok(()) +} + #[derive(Debug, Clone)] pub struct UnivMonAccumulator { inner: UnivMon, @@ -17,17 +33,33 @@ impl UnivMonAccumulator { } pub fn new(heap_size: usize, rows: usize, cols: usize, layers: usize) -> Result { - if heap_size == 0 || cols == 0 || !(1..=20).contains(&rows) || !(1..=64).contains(&layers) { - return Err("invalid UnivMon dimensions".into()); - } - rows.checked_mul(cols) - .and_then(|n| n.checked_mul(layers)) - .ok_or("UnivMon dimensions overflow")?; + check_dimensions(heap_size, rows, cols, layers)?; Ok(Self { inner: UnivMon::init_univmon(heap_size, rows, cols, layers), }) } + /// Adopt a decoded sketch, e.g. one a deployment restored from its stored + /// bytes. Rejects dimensions [`Self::new`] rejects and terminal-mode + /// sketches, which do not accept this accumulator's updates. + pub fn from_sketch(sketch: UnivMon) -> Result { + check_dimensions( + sketch.heap_size, + sketch.sketch_row, + sketch.sketch_col, + sketch.layer_size, + )?; + if !sketch.accepts_standard_updates() { + return Err("terminal-mode UnivMon state cannot accept standard updates".into()); + } + Ok(Self { inner: sketch }) + } + + /// The underlying sketch, so a deployment can encode it with sketchlib's codec. + pub fn sketch(&self) -> &UnivMon { + &self.inner + } + /// Each non-NaN sample is one occurrence. Signed zero has one identity. pub fn insert_sample(&mut self, value: f64) -> Result<(), Error> { if value.is_nan() { @@ -106,4 +138,79 @@ impl AggregateCore for UnivMonAccumulator { merged.merge_in_place(other)?; Ok(Box::new(merged)) } + + /// Sample count (a bare `PointCount`), distinct count, L2 norm and entropy + /// of the sample-value frequencies. + fn estimate(&self, query: &SketchQuery) -> Result { + Ok(match query { + SketchQuery::PointCount { + key: ColumnRef::SampleValue, + value: None, + } => self.inner.calc_l1(), + SketchQuery::Cardinality => self.inner.calc_card(), + SketchQuery::FrequencyL2 => self.inner.calc_l2(), + SketchQuery::FrequencyEntropy => self.inner.calc_entropy(), + other => return Err(format!("UnivMon does not answer {other:?}").into()), + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn count() -> SketchQuery { + SketchQuery::PointCount { + key: ColumnRef::SampleValue, + value: None, + } + } + + // Count, distinct, L2 and entropy readouts count each non-NaN sample once; + // signed zero is one identity. + #[test] + fn frequency_readouts() { + let mut state = UnivMonAccumulator::new(32, 5, 1024, 4).unwrap(); + for value in [0.0, -0.0, 2.0, 2.0, f64::NAN] { + state.insert_sample(value).unwrap(); + } + let read = |query| state.estimate(&query).unwrap(); + assert_eq!(read(count()), 4.0); + assert!((read(SketchQuery::Cardinality) - 2.0).abs() < 0.01); + assert!((read(SketchQuery::FrequencyL2) - 8.0f64.sqrt()).abs() < 0.01); + assert!((read(SketchQuery::FrequencyEntropy) - 1.0).abs() < 0.01); + assert!(state.estimate(&SketchQuery::Quantile { q: 0.5 }).is_err()); + } + + // A sketch taken out and adopted back answers the same readouts. + #[test] + fn adopted_sketch_keeps_readouts() { + let mut state = UnivMonAccumulator::new(32, 5, 1024, 4).unwrap(); + for value in [1.0, 2.0, 2.0] { + state.insert_sample(value).unwrap(); + } + let adopted = UnivMonAccumulator::from_sketch(state.sketch().clone()).unwrap(); + assert_eq!(adopted.dimensions(), state.dimensions()); + for query in [ + count(), + SketchQuery::Cardinality, + SketchQuery::FrequencyEntropy, + ] { + assert_eq!( + adopted.estimate(&query).unwrap(), + state.estimate(&query).unwrap() + ); + } + } + + // Adoption rejects terminal-mode sketches and dimensions `new` rejects. + #[test] + fn adoption_rejects_terminal_or_invalid_sketches() { + let mut terminal = UnivMon::init_univmon(4, 3, 16, 2); + terminal.fast_insert(&DataInput::U64(1), 1); + assert!(UnivMonAccumulator::from_sketch(terminal.clone()).is_err()); + terminal.free(); + assert!(UnivMonAccumulator::from_sketch(terminal).is_ok()); + assert!(UnivMonAccumulator::from_sketch(UnivMon::init_univmon(4, 21, 16, 2)).is_err()); + } } From f77a5fc93d213b974b886b487b5b5c7eefe772c7 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 13:31:35 +0000 Subject: [PATCH 53/59] feat(kernels): expose weighted frequency shape and byte codec The backend reached the sketchlib kernel through a msgpack round trip to persist and validate it. Public algorithm/shape and to_bytes/from_bytes (delegating to sketchlib's versioned format) remove that. Co-Authored-By: Claude Opus 5.5 --- .../src/summary_kernels/weighted_frequency.rs | 36 +++++++++++++++++-- 1 file changed, 34 insertions(+), 2 deletions(-) diff --git a/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs index 6f008532..43528929 100644 --- a/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs +++ b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs @@ -75,12 +75,25 @@ impl WeightedFrequency { Ok((algorithm, width as usize, depth as usize, capacity as usize)) } - pub(crate) fn algorithm(&self) -> FrequencyAlgorithm { + /// Algorithm and `(width, depth, capacity)` shape, so a deployment can check + /// a decoded state against its declared family. + pub fn algorithm(&self) -> FrequencyAlgorithm { self.inner.algorithm() } - pub(crate) fn shape(&self) -> (usize, usize, usize) { + pub fn shape(&self) -> (usize, usize, usize) { self.inner.shape() } + /// Sketchlib's versioned `WeightedFrequencyV1` bytes, for deployments that + /// persist this state. + pub fn to_bytes(&self) -> Vec { + self.inner.to_bytes() + } + /// Decode and validate bytes written by [`Self::to_bytes`]. + pub fn from_bytes(bytes: &[u8]) -> Result { + Kernel::from_bytes(bytes) + .map(|inner| Self { inner }) + .map_err(adapt_error) + } pub fn new( algorithm: FrequencyAlgorithm, width: usize, @@ -149,4 +162,23 @@ mod tests { .merge_with(&WeightedFrequency::new(FrequencyAlgorithm::Cms, 32, 5, 8).unwrap()) .is_err()); } + + // Stored bytes restore the algorithm, shape and ranked rows; foreign bytes are rejected. + #[test] + fn bytes_round_trip() { + let mut state = WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 64, 3, 4).unwrap(); + state.update(&[Value::Utf8("a".into())], 2.5).unwrap(); + state.update(&[Value::Utf8("b".into())], 1.0).unwrap(); + let restored = WeightedFrequency::from_bytes(&state.to_bytes()).unwrap(); + assert!(matches!( + restored.algorithm(), + FrequencyAlgorithm::CountSketch + )); + assert_eq!(restored.shape(), (64, 3, 4)); + assert_eq!( + format!("{:?}", restored.rows(2)), + format!("{:?}", state.rows(2)) + ); + assert!(WeightedFrequency::from_bytes(b"not a frequency state").is_err()); + } } From aebe2ea1048bfc8ccc4ddef445387fbce40b9f9f Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 13:32:02 +0000 Subject: [PATCH 54/59] feat(kernels): add AggregateCore::as_any_mut Backend ingest deltas and read-path merges cloned each cached sketch to update it. A mutable downcast lets them update it in place. Co-Authored-By: Claude Opus 5.5 --- .../src/summary_kernels/count_min_sketch.rs | 3 +++ .../count_min_sketch_with_heap.rs | 3 +++ .../src/summary_kernels/count_sketch.rs | 3 +++ .../summary_kernels/count_sketch_with_heap.rs | 3 +++ .../src/summary_kernels/datasketches_kll.rs | 3 +++ .../src/summary_kernels/dd_sketch.rs | 3 +++ .../src/summary_kernels/exact.rs | 3 +++ .../src/summary_kernels/hll_sketch.rs | 3 +++ .../src/summary_kernels/hydra_kll.rs | 3 +++ .../src/summary_kernels/increase.rs | 3 +++ .../src/summary_kernels/traits.rs | 26 +++++++++++++++++++ .../src/summary_kernels/univmon.rs | 3 +++ .../src/summary_kernels/weighted_frequency.rs | 3 +++ 13 files changed, 62 insertions(+) diff --git a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs index 5a78fc48..bb090691 100644 --- a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs +++ b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs @@ -28,6 +28,9 @@ impl AggregateCore for CountMinSketchAccumulator { fn as_any(&self) -> &dyn std::any::Any { self } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { let other = other diff --git a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs index da07aeee..f2d8786c 100644 --- a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs +++ b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs @@ -80,6 +80,9 @@ impl AggregateCore for CountMinSketchWithHeapAccumulator { fn as_any(&self) -> &dyn std::any::Any { self } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } fn merge_with( &self, diff --git a/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs index 2728b5f2..c3aaa40c 100644 --- a/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs +++ b/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs @@ -34,6 +34,9 @@ impl AggregateCore for CountSketchAccumulator { fn as_any(&self) -> &dyn std::any::Any { self } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } fn merge_with( &self, diff --git a/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs b/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs index a735e4bd..9f15e504 100644 --- a/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs +++ b/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs @@ -68,6 +68,9 @@ impl AggregateCore for CountSketchWithHeapAccumulator { fn as_any(&self) -> &dyn std::any::Any { self } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } fn merge_with( &self, diff --git a/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs b/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs index 0f740a9d..3417bb8c 100644 --- a/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs +++ b/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs @@ -46,6 +46,9 @@ impl AggregateCore for DatasketchesKLLAccumulator { fn as_any(&self) -> &dyn std::any::Any { self } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { let other = other diff --git a/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs index 1d6be98c..dcda7ad4 100644 --- a/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs +++ b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs @@ -24,6 +24,9 @@ impl AggregateCore for DDSketchAccumulator { fn as_any(&self) -> &dyn std::any::Any { self } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { let other = other diff --git a/crates/asap-physical-operators/src/summary_kernels/exact.rs b/crates/asap-physical-operators/src/summary_kernels/exact.rs index aeb0c6db..a211b7f3 100644 --- a/crates/asap-physical-operators/src/summary_kernels/exact.rs +++ b/crates/asap-physical-operators/src/summary_kernels/exact.rs @@ -251,6 +251,9 @@ impl AggregateCore for ExactAccumulator { fn as_any(&self) -> &dyn std::any::Any { self } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } fn merge_with(&self, other: &dyn AggregateCore) -> Result, Error> { let other = other .as_any() diff --git a/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs index 089261c5..f675da23 100644 --- a/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs +++ b/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs @@ -24,6 +24,9 @@ impl AggregateCore for HllSketchAccumulator { fn as_any(&self) -> &dyn std::any::Any { self } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { let other = other diff --git a/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs b/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs index c0347e69..8d6f6a79 100644 --- a/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs +++ b/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs @@ -31,6 +31,9 @@ impl AggregateCore for HydraKllSketchAccumulator { fn as_any(&self) -> &dyn std::any::Any { self } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } fn merge_with( &self, diff --git a/crates/asap-physical-operators/src/summary_kernels/increase.rs b/crates/asap-physical-operators/src/summary_kernels/increase.rs index f92ee582..63095601 100644 --- a/crates/asap-physical-operators/src/summary_kernels/increase.rs +++ b/crates/asap-physical-operators/src/summary_kernels/increase.rs @@ -99,6 +99,9 @@ impl AggregateCore for IncreaseAccumulator { fn as_any(&self) -> &dyn std::any::Any { self } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } fn merge_with( &self, diff --git a/crates/asap-physical-operators/src/summary_kernels/traits.rs b/crates/asap-physical-operators/src/summary_kernels/traits.rs index 7d866077..3902a296 100644 --- a/crates/asap-physical-operators/src/summary_kernels/traits.rs +++ b/crates/asap-physical-operators/src/summary_kernels/traits.rs @@ -13,6 +13,10 @@ pub trait AggregateCore: Send + Sync { fn as_any(&self) -> &dyn std::any::Any; + /// Mutable downcast, so a deployment can apply ingest deltas and merges to + /// a cached state in place instead of copying it for every frame. + fn as_any_mut(&mut self) -> &mut dyn std::any::Any; + /// Merge with a state of the same family and shape, leaving both inputs unchanged. fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError>; @@ -33,3 +37,25 @@ impl Clone for Box { self.clone_boxed_core() } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::summary_kernels::DDSketchAccumulator; + + // A state behind a trait object can be updated in place through the mutable downcast. + #[test] + fn mutable_downcast_updates_in_place() { + let mut state: Box = Box::new(DDSketchAccumulator::new(0.01)); + let dd = state + .as_any_mut() + .downcast_mut::() + .unwrap(); + dd.inner.update(3.0); + let count = SketchQuery::PointCount { + key: planner_types::pre_asap::ColumnRef::SampleValue, + value: None, + }; + assert_eq!(state.estimate(&count).unwrap(), 1.0); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/univmon.rs b/crates/asap-physical-operators/src/summary_kernels/univmon.rs index 27ca7fcc..6d42c336 100644 --- a/crates/asap-physical-operators/src/summary_kernels/univmon.rs +++ b/crates/asap-physical-operators/src/summary_kernels/univmon.rs @@ -128,6 +128,9 @@ impl AggregateCore for UnivMonAccumulator { fn as_any(&self) -> &dyn std::any::Any { self } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } fn merge_with(&self, other: &dyn AggregateCore) -> Result, Error> { let other = other diff --git a/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs index 43528929..bac6b7b6 100644 --- a/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs +++ b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs @@ -127,6 +127,9 @@ impl AggregateCore for WeightedFrequency { fn as_any(&self) -> &dyn std::any::Any { self } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } fn merge_with( &self, other: &dyn AggregateCore, From 6715f96cb393e8eb4bccff1c8c5b787c30b77369 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 08:00:55 +0000 Subject: [PATCH 55/59] fix(mapping): type the summary state column of without aggregations A without grouping's keys are the excluded labels, so the state column index was the excluded-key count, which could fall past the kept labels. The family was then never written into the schema and exporting the candidate panicked with SummaryFamilySchemaMismatch. Co-Authored-By: Claude Opus 5.5 --- crates/asap-aware-mapping/src/replacement.rs | 30 ++++++++++--------- .../tests/promql_to_post_asap.rs | 27 +++++++++++++++++ 2 files changed, 43 insertions(+), 14 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 8c826b37..9cb81e14 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -2804,13 +2804,12 @@ fn construct_summary_agg( } else { reduction.clone() }; - let per_series = matches!(reduction, Reduction::PerEntity); - let by: Vec = reduction - .group_keys() - .map(|g| g.to_vec()) - .unwrap_or_default(); let out_schema = node.output_schema()?; - let state_idx = summary_col_index(&out_schema, &by, per_series); + let measures = match node { + QueryExpr::Aggregate { measures, .. } => measures.len(), + _ => 1, + }; + let state_idx = summary_col_index(&out_schema, reduction, measures); let readout_schema = if keyed_heap && matches!(node, QueryExpr::Aggregate { child, .. } if is_snapshot_weighted_topk(intent, child)) @@ -3558,16 +3557,19 @@ fn compose_guarantee( /// cross-series output is `by ++ [agg]` (the column after the keys); /// a per-series reduction keeps every label and replaces the sample value /// (named `value` — mirror `per_series_reduction_schema`'s fallback). -/// `per_series` is the caller's already-read `Reduction` (issue #165) — -/// this never re-derives it, so it can't disagree with the caller. -fn summary_col_index(out_schema: &Schema, by: &[usize], per_series: bool) -> usize { - if per_series { - out_schema +/// `reduction` is the caller's already-read `Reduction` (issue #165). +/// `without` output is `kept labels ++ measures`, and its keys are the +/// *excluded* labels, so the state column follows the kept labels instead. +fn summary_col_index(out_schema: &Schema, reduction: &Reduction, measures: usize) -> usize { + match reduction { + Reduction::PerEntity => out_schema .column_id("value") .or_else(|| (0..out_schema.columns.len()).find(|&i| Some(i) != out_schema.time_index)) - .unwrap_or(0) - } else { - by.len() + .unwrap_or(0), + Reduction::Reduce(keys) if keys.is_without() => { + out_schema.columns.len().saturating_sub(measures) + } + Reduction::Reduce(keys) => keys.len(), } } diff --git a/crates/integration-tests/tests/promql_to_post_asap.rs b/crates/integration-tests/tests/promql_to_post_asap.rs index 539db7bf..a0e71a08 100644 --- a/crates/integration-tests/tests/promql_to_post_asap.rs +++ b/crates/integration-tests/tests/promql_to_post_asap.rs @@ -1436,3 +1436,30 @@ fn ddsketch_ratio_requires_a_supported_population_size() { assert!(strategy.replacements(&TargetSubDAG::new(&pre)).is_empty()); } } + +// Every `without` aggregation candidate exports a valid DAG: its summary state +// column carries the family instead of the readout's Float64 value. +#[test] +fn without_aggregation_candidates_export_valid_dags() { + for accuracy in [ + AccuracyTarget::Exact, + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ] { + for query in ["sum without (pod) (m)", "quantile without (pod) (0.5, m)"] { + let root = Rc::new(lower_promql(query, accuracy.clone()).unwrap()); + let space = search_workload_with_targets( + vec![(0, root, Some(accuracy.clone()))], + &asap_aware_mapping::default_strategies(), + &DefaultAccuracyModel, + ); + let inventory = space.enumerate_candidate_dags_for_root(&0, 65_536).unwrap(); + assert!(!inventory.candidates.is_empty(), "{query}"); + for (_, node) in inventory.candidates.iter().flatten() { + compile_post_asap_dag(node).unwrap_or_else(|e| panic!("{query}: {e}")); + } + } + } +} From 5e651f7b857abe9c645811843849d2dbfef74f27 Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 08:06:33 +0000 Subject: [PATCH 56/59] feat(physical): compile per-series vector arithmetic over stored readouts Query-time Binary nodes over rows with a series identity, such as avg_over_time as stored sum/count or rate(a)/rate(b), now compile to series_labels and series_binary: one-to-one matching on labels without the metric name, honoring on/ignoring when the IR carries them. A literal operand drops the metric name and applies to each value. Co-Authored-By: Claude Opus 5.5 --- .../src/physical_planner/mod.rs | 25 ++ .../src/physical_planner/row_values.rs | 77 ++++++- .../tests/deployment_computation.rs | 214 ++++++++++++++++-- 3 files changed, 296 insertions(+), 20 deletions(-) diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 655c024a..2e244692 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -471,6 +471,20 @@ fn compile_internal( if !query_time { return Err(invalid("scalar literal binary must run at query time")); } + if row_values::per_series(input) { + let mut chain = + row_values::series_scalar_binary(input, operator, value, left) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + let last = chain.pop().expect("nonempty chain"); + let mut inputs = inputs; + for operator in chain { + graph.add(auxiliary, inputs, operator)?; + inputs = vec![auxiliary]; + auxiliary -= 1; + } + graph.add(id, inputs, last.with_output_schema(output)?)?; + continue; + } let project = row_values::scalar_binary(input, operator, value, left) .map_err(|error| invalid(format!("node {id}: {error}")))?; graph.add(id, inputs, project.with_output_schema(output)?)?; @@ -483,6 +497,17 @@ fn compile_internal( .any(|f| matches!(f.dtype, SummaryFamilyType::Plain(DataType::Map { .. }))) }; if let (true, [left, right]) = (query_time, schemas.as_slice()) { + if row_values::per_series(left) || row_values::per_series(right) { + let [left, right, binary] = + row_values::series_vector_binary(left, right, operator) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + let sides = [auxiliary, auxiliary - 1]; + graph.add(sides[0], vec![inputs[0]], left)?; + graph.add(sides[1], vec![inputs[1]], right)?; + graph.add(id, sides.to_vec(), binary.with_output_schema(output)?)?; + auxiliary -= 2; + continue; + } if !label_map(left) && !label_map(right) { let (join, project) = row_values::grouped_binary(left, right, operator) .map_err(|error| invalid(format!("node {id}: {error}")))?; diff --git a/crates/asap-physical-operators/src/physical_planner/row_values.rs b/crates/asap-physical-operators/src/physical_planner/row_values.rs index 737c8065..61553927 100644 --- a/crates/asap-physical-operators/src/physical_planner/row_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/row_values.rs @@ -1,7 +1,9 @@ //! Query-time PromQL value computation over logical row schemas. use super::*; use planner_types::post_asap::{maintained_population::PopulationReadout, BinaryOperator}; -use planner_types::pre_asap::{BinaryOpKind, DataType, Predicate, ScalarValue}; +use planner_types::pre_asap::{ + BinaryOpKind, DataType, Predicate, ScalarValue, VectorMatch, VectorMatchKind, +}; use std::rc::Rc; /// A PromQL number literal has no row schema; its consumer folds it in. @@ -23,7 +25,7 @@ fn grouped_value(input: &Schema) -> Result<(usize, Vec), Error> { .any(|field| field.name == promql_rows::SERIES_IDENTITY_COLUMN) { return Err(invalid( - "row binary requires grouped rows; per-series matching needs a name-free identity", + "row binary requires grouped rows or rows with a series identity", )); } let mut value = None; @@ -54,6 +56,15 @@ fn arithmetic(operator: &BinaryOperator) -> Result<(), Error> { Ok(()) } +/// Rows whose labels are the encoded PromQL series identity, such as +/// per-series readouts of stored state. +pub(super) fn per_series(input: &Schema) -> bool { + input + .fields + .iter() + .any(|field| field.name == promql_rows::SERIES_IDENTITY_COLUMN) +} + /// Apply `vector op scalar` (or `scalar op vector`) to each row's value. pub(super) fn scalar_binary( input: &Schema, @@ -63,6 +74,68 @@ pub(super) fn scalar_binary( ) -> Result { arithmetic(operator)?; let (value, _) = grouped_value(input)?; + literal_projection(input, value, operator, literal, literal_left) +} + +/// `scalar_binary` over per-series rows: PromQL arithmetic also drops the +/// metric name from the series identity. Returns a chain. +pub(super) fn series_scalar_binary( + input: &Schema, + operator: &BinaryOperator, + literal: f64, + literal_left: bool, +) -> Result, Error> { + arithmetic(operator)?; + let relabel = Operator::series_labels(input.clone(), VectorMatchKind::Ignoring, vec![])?; + let value = input + .fields + .iter() + .position(|field| { + field.dtype == SummaryFamilyType::Plain(DataType::Float64) && !field.nullable + }) + .ok_or_else(|| invalid("per-series rows require a Float64 value"))?; + let project = literal_projection(input, value, operator, literal, literal_left)?; + Ok(vec![relabel, project]) +} + +/// PromQL one-to-one arithmetic where either side carries a series identity. +/// Returns the left and right relabelings to the matching labels, and the +/// binary over their outputs. +pub(super) fn series_vector_binary( + left: &Schema, + right: &Schema, + operator: &BinaryOperator, +) -> Result<[Operator; 3], Error> { + arithmetic(operator)?; + let (kind, labels) = match &operator.vector_match { + None => (VectorMatchKind::Ignoring, vec![]), + Some(VectorMatch { + kind, + labels, + grouping: None, + }) => (kind.clone(), labels.clone()), + Some(_) => return Err(invalid("group_left/group_right matching is unsupported")), + }; + let left_labels = Operator::series_labels(left.clone(), kind.clone(), labels.clone())?; + let right_labels = Operator::series_labels(right.clone(), kind, labels)?; + let binary = Operator::series_binary( + left_labels.schema(), + right_labels.schema(), + BinaryOperator { + vector_match: None, + ..operator.clone() + }, + )?; + Ok([left_labels, right_labels, binary]) +} + +fn literal_projection( + input: &Schema, + value: usize, + operator: &BinaryOperator, + literal: f64, + literal_left: bool, +) -> Result { let literal = Expression::Literal { value: crate::values::Value::Float64(literal), dtype: DataType::Float64, diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index b026f47f..ff573382 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -97,8 +97,12 @@ fn raw_inputs(dag: &PostAsapDag) -> Vec<(u64, Arc, String)> { type Sample = (&'static str, &'static str, &'static str, i64, f64); /// Compile, round-trip, bind raw `(metric, job, instance, ts, value)` samples, -/// and return `(job, value)` rows of the root. -fn run(dag: &PostAsapDag, samples: &[Sample], end: i64) -> Result, String> { +/// and return the root's batches. +fn execute( + dag: &PostAsapDag, + samples: &[Sample], + end: i64, +) -> Result>, String> { let inputs = raw_inputs(dag); let program = compile( dag, @@ -164,27 +168,35 @@ fn run(dag: &PostAsapDag, samples: &[Sample], end: i64) -> Result job.to_string(), - _ => String::new(), - }; - let value = match row.last() { - Some(Value::Float64(v)) => *v, - Some(Value::Int64(v)) => *v as f64, - other => return Err(format!("unexpected value {other:?}")), - }; - assert!(rows.insert(key, value).is_none(), "duplicate output group"); - } + batches.push(batch.map_err(|e| e.to_string())?); } - Ok(rows) + Ok(batches) }) } +/// [`execute`], returning `(job, value)` rows of the root. +fn run(dag: &PostAsapDag, samples: &[Sample], end: i64) -> Result, String> { + let mut rows = BTreeMap::new(); + for batch in execute(dag, samples, end)? { + let job = batch.schema().fields.iter().position(|f| f.name == "job"); + for row in batch.rows() { + let key = match job.map(|i| &row[i]) { + Some(Value::Utf8(job)) => job.to_string(), + _ => String::new(), + }; + let value = match row.last() { + Some(Value::Float64(v)) => *v, + Some(Value::Int64(v)) => *v as f64, + other => return Err(format!("unexpected value {other:?}")), + }; + assert!(rows.insert(key, value).is_none(), "duplicate output group"); + } + } + Ok(rows) +} + const SAMPLES: &[Sample] = &[ ("m", "api", "a", 10_000, 4.), ("m", "api", "a", 50_000, 1.), @@ -338,3 +350,169 @@ fn row_comparison_fails_closed() { let error = run(&dag, SAMPLES, 60_000).unwrap_err(); assert!(error.contains("comparison"), "{error}"); } + +/// [`execute`], returning per-series `(identity, value)` rows of the root, +/// with NaN-aware formatting for comparison. +fn run_series(dag: &PostAsapDag, samples: &[Sample], end: i64) -> Result { + let mut rows = BTreeMap::new(); + for batch in execute(dag, samples, end)? { + let schema = batch.schema(); + let column = |name: &str| schema.fields.iter().position(|f| f.name == name).unwrap(); + let (identity, value) = (column(promql_rows::SERIES_IDENTITY_COLUMN), column("value")); + for row in batch.rows() { + let (Value::Utf8(identity), Value::Float64(value)) = (&row[identity], &row[value]) + else { + return Err(format!("unexpected row {row:?}")); + }; + assert!(rows.insert(identity.to_string(), *value).is_none()); + } + } + Ok(format!("{rows:?}")) +} + +fn series(pairs: &[(&str, &str, f64)]) -> String { + let rows = pairs + .iter() + .map(|(job, instance, value)| { + ( + format!(r#"{{"instance":"{instance}","job":"{job}"}}"#), + *value, + ) + }) + .collect::>(); + format!("{rows:?}") +} + +/// Five counter samples per series, one per minute up to 300s, at `base + step * i`. +fn counter( + metric: &'static str, + job: &'static str, + base: f64, + step: f64, +) -> impl Iterator { + (1..=5).map(move |i| (metric, job, "x", i * 60_000, base + step * (i - 1) as f64)) +} + +// avg_over_time over stored per-series sum/count state divides per series and +// drops the metric name, as Prometheus does. An overflowing stored sum fails +// the checked division instead of returning +Inf. +#[test] +fn per_series_average_divides_stored_sum_by_count() { + let dag = exact_dag("avg_over_time(m[5m])"); + assert_eq!( + run_series(&dag, SAMPLES, 60_000).unwrap(), + series(&[ + ("api", "a", 2.5), + ("api", "b", 7.), + ("api", "c", 2.), + ("db", "d", 5.) + ]) + ); + let huge: &[Sample] = &[ + ("m", "api", "a", 10_000, 1.7e308), + ("m", "api", "a", 20_000, 1.7e308), + ]; + assert!(run_series(&dag, huge, 60_000).is_err()); +} + +// rate(a) / rate(b) matches series on their labels without the metric name. +// Unmatched series are dropped; x/0 is +Inf and 0/0 is NaN; an empty side +// yields an empty vector. +#[test] +fn per_series_rate_ratio_matches_prometheus() { + // Window (0, 300s]: first sample at 60s, extrapolated 60s to the start + // (durationToZero is exactly 60s too), so rate = (last - first) * 1.25 / 300. + // a{api} = 40 * 1.25 / 300, b{api} = 20 * 1.25 / 300, so the ratio is 2. + let samples = counter("a", "api", 10., 10.) + .chain(counter("a", "db", 10., 10.)) + .chain(counter("a", "web", 10., 10.)) + .chain(counter("a", "cache", 7., 0.)) + .chain(counter("b", "api", 5., 5.)) + .chain(counter("b", "web", 7., 0.)) + .chain(counter("b", "cache", 7., 0.)) + .chain(counter("b", "other", 5., 5.)) + .collect::>(); + let dag = exact_dag("rate(a[5m]) / rate(b[5m])"); + assert_eq!( + run_series(&dag, &samples, 300_000).unwrap(), + series(&[ + ("api", "x", 2.), + ("cache", "x", f64::NAN), + ("web", "x", f64::INFINITY), + ]) + ); + let only_a = samples + .iter() + .filter(|s| s.0 == "a") + .copied() + .collect::>(); + assert_eq!(run_series(&dag, &only_a, 300_000).unwrap(), series(&[])); +} + +// A literal operand applies to every stored per-series value, on either side, +// and drops the metric name. +#[test] +fn per_series_scalar_arithmetic_applies_to_stored_readouts() { + let samples = counter("m", "api", 10., 10.).collect::>(); + // rate = 40 * 1.25 / 300 = 1/6. + for (query, expected) in [ + ("rate(m[5m]) * 2", 50. / 300. * 2.), + ("1 - rate(m[5m])", 1. - 50. / 300.), + ] { + assert_eq!( + run_series(&exact_dag(query), &samples, 300_000).unwrap(), + series(&[("api", "x", expected)]), + "{query}" + ); + } +} + +fn with_vector_match( + mut dag: PostAsapDag, + kind: planner_types::pre_asap::VectorMatchKind, + labels: &[&str], +) -> PostAsapDag { + for node in &mut dag.nodes { + if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { + operator.vector_match = Some(planner_types::pre_asap::VectorMatch { + kind: kind.clone(), + labels: labels.iter().map(|l| l.to_string()).collect(), + grouping: None, + }); + } + } + dag +} + +// `on` and `ignoring` reduce each side to the matching labels, which become +// the result's labels; a duplicate match group on either side is an error. +#[test] +fn per_series_vector_matching_follows_on_and_ignoring() { + use planner_types::pre_asap::VectorMatchKind; + let samples = counter("a", "api", 10., 10.) + .chain(counter("b", "api", 5., 5.)) + .chain(counter("b", "db", 5., 5.)) + .collect::>(); + let dag = exact_dag("rate(a[5m]) / rate(b[5m])"); + for (kind, labels) in [ + (VectorMatchKind::On, ["job"]), + (VectorMatchKind::Ignoring, ["instance"]), + ] { + let dag = with_vector_match(dag.clone(), kind, &labels); + assert_eq!( + run_series(&dag, &samples, 300_000).unwrap(), + format!( + "{:?}", + BTreeMap::from([(r#"{"job":"api"}"#.to_string(), 2.)]) + ) + ); + let mut duplicate = samples.clone(); + duplicate.extend(counter("b", "api", 5., 5.).map(|s| (s.0, s.1, "y", s.3, s.4))); + let error = run_series(&dag, &duplicate, 300_000).unwrap_err(); + assert!(error.contains("duplicate series"), "{error}"); + let mut duplicate = samples.clone(); + duplicate.extend(counter("a", "api", 5., 5.).map(|s| (s.0, s.1, "y", s.3, s.4))); + let error = run_series(&dag, &duplicate, 300_000).unwrap_err(); + assert!(error.contains("many-to-one"), "{error}"); + } +} From 88c0478b35cbfb1949ebf1f725e2de5ceebfa10e Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 08:07:59 +0000 Subject: [PATCH 57/59] fix(physical): compensate PromQL sums and averages like Prometheus Grouped sum/avg (including current-series readouts) and sum/avg_over_time now use Kahan-Neumaier summation. An average switches to Prometheus' incremental mean once the running sum would overflow, so it no longer becomes +Inf where Prometheus returns a finite mean. Co-Authored-By: Claude Opus 5.5 --- .../src/operators/aggregate/mod.rs | 70 +++++++++++++++++-- .../src/operators/aggregate/temporal.rs | 12 ++-- .../tests/deployment_computation.rs | 27 +++++++ .../tests/promql_fallback.rs | 17 +++++ 4 files changed, 116 insertions(+), 10 deletions(-) diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs index d929fc07..3b065172 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/mod.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -302,9 +302,9 @@ async fn reduce_one( } return Ok(best.cloned().unwrap_or(Value::Null)); } - let mut count = 0usize; let dtype = plain(input, column)?.0; if dtype == &DataType::Int64 { + let mut count = 0usize; let mut sum = 0i128; for v in values { work.checkpoint().await?; @@ -324,22 +324,80 @@ async fn reduce_one( )) }; } - let mut sum = -0.0; + let mut floats = Vec::with_capacity(rows.len()); for v in values { work.checkpoint().await?; let Value::Float64(v) = v else { return Err(invalid("floating aggregate value required")); }; - sum += v; - count += 1; + floats.push(*v); } Ok(Value::Float64(if matches!(measure, Reduction::Avg(_)) { - sum / count as f64 + promql_avg(&floats) } else { - sum + // Prometheus starts `sum` from the first value; -0.0 keeps an + // all-negative-zero group's sign. + promql_sum(-0.0, &floats) })) } +/// Prometheus `kahansum.Inc`: Kahan-Neumaier compensated addition, with the +/// compensation cleared once the sum is infinite. +fn kahan_inc(inc: f64, sum: f64, c: f64) -> (f64, f64) { + let t = sum + inc; + let c = if t.is_infinite() { + 0. + } else if sum.abs() >= inc.abs() { + c + ((sum - t) + inc) + } else { + c + ((inc - t) + sum) + }; + (t, c) +} + +/// Prometheus `sum` and `sum_over_time` from `start`. +pub(in crate::operators) fn promql_sum(start: f64, values: &[f64]) -> f64 { + let (sum, c) = values + .iter() + .fold((start, 0.), |(sum, c), &v| kahan_inc(v, sum, c)); + if sum.is_infinite() { + sum + } else { + sum + c + } +} + +/// Prometheus `avg` and `avg_over_time`: a compensated sum divided by the +/// count until the running sum would overflow, then an incremental mean. +/// NaN for no values. +pub(in crate::operators) fn promql_avg(values: &[f64]) -> f64 { + let Some((&first, rest)) = values.split_first() else { + return f64::NAN; + }; + let (mut sum, mut c, mut mean, mut incremental) = (first, 0., 0., false); + for (i, &v) in rest.iter().enumerate() { + let count = (i + 2) as f64; + if !incremental { + let (next, next_c) = kahan_inc(v, sum, c); + if !next.is_infinite() { + (sum, c) = (next, next_c); + continue; + } + incremental = true; + mean = sum / (count - 1.); + c /= count - 1.; + } + let q = (count - 1.) / count; + (mean, c) = kahan_inc(v / count, q * mean, q * c); + } + let count = values.len() as f64; + if incremental { + mean + c + } else { + sum / count + c / count + } +} + #[cfg(test)] mod tests { // Quantile follows Prometheus: interpolate ranks, NaN when empty, ±Inf outside [0, 1]. diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs index 3ac0a3d1..2971d575 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -92,10 +92,14 @@ pub(in crate::operators) fn window_value( AggIntent::Count { .. } => Some(Value::Int64( i64::try_from(points.len()).map_err(|_| Error::Invalid("count overflow".into()))?, )), - AggIntent::Sum { .. } => Some(Value::Float64(points.iter().map(|p| p.1).sum())), - AggIntent::Avg { .. } => Some(Value::Float64( - points.iter().map(|p| p.1).sum::() / points.len() as f64, - )), + AggIntent::Sum { .. } | AggIntent::Avg { .. } => { + let values = points.iter().map(|p| p.1).collect::>(); + Some(Value::Float64(if matches!(intent, AggIntent::Sum { .. }) { + super::promql_sum(0., &values) + } else { + super::promql_avg(&values) + })) + } AggIntent::Min { .. } => Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { if a.is_nan() || p.1 < a { p.1 diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index ff573382..58adc8c2 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -516,3 +516,30 @@ fn per_series_vector_matching_follows_on_and_ignoring() { assert!(error.contains("many-to-one"), "{error}"); } } + +// Current-series sums and averages are compensated like Prometheus, and an +// overflowing running sum does not turn the average into +Inf. +#[test] +fn population_sums_and_averages_are_compensated() { + let cancel: &[Sample] = &[ + ("m", "api", "a", 50_000, 1e100), + ("m", "api", "b", 50_000, 1.), + ("m", "api", "c", 50_000, -1e100), + ]; + let huge: &[Sample] = &[ + ("m", "api", "a", 50_000, 1.7e308), + ("m", "api", "b", 50_000, 1.7e308), + ]; + for (query, samples, expected) in [ + ("sum by (job) (m)", cancel, 1.), + ("avg by (job) (m)", cancel, 1. / 3.), + ("avg by (job) (m)", huge, 1.7e308), + ] { + let dag = population_dag(query); + assert_eq!( + run(&dag, samples, 60_000).unwrap(), + reference(&[("api", expected)]), + "{query}" + ); + } +} diff --git a/crates/asap-physical-operators/tests/promql_fallback.rs b/crates/asap-physical-operators/tests/promql_fallback.rs index d274495c..48603791 100644 --- a/crates/asap-physical-operators/tests/promql_fallback.rs +++ b/crates/asap-physical-operators/tests/promql_fallback.rs @@ -622,3 +622,20 @@ fn empty_labels_and_empty_sides_match_prometheus() { // A non-literal scalar operand has no identity realization yet. assert!(promql_rows::with_series_identity(&parse("a + scalar(b)")).is_err()); } + +// Sums and averages use Prometheus' Kahan-Neumaier compensation, and an +// average whose running sum overflows switches to an incremental mean. +#[test] +fn sums_and_averages_are_compensated_like_prometheus() { + let cancel = &[("a", 10, 1e100), ("a", 20, 1.), ("a", 30, -1e100)]; + assert_eq!(one("sum_over_time(m[1m])", cancel, 60), 1.); + assert_eq!(one("avg_over_time(m[1m])", cancel, 60), 1. / 3.); + let huge = &[("a", 10, 1.7e308), ("a", 20, 1.7e308)]; + assert_eq!(one("avg_over_time(m[1m])", huge, 60), 1.7e308); + assert_eq!(one("sum_over_time(m[1m])", huge, 60), f64::INFINITY); + let cancel = &[("a", 50, 1e100), ("b", 50, 1.), ("c", 50, -1e100)]; + assert_eq!(one("sum(m)", cancel, 60), 1.); + assert_eq!(one("avg(m)", cancel, 60), 1. / 3.); + let huge = &[("a", 50, 1.7e308), ("b", 50, 1.7e308)]; + assert_eq!(one("avg(m)", huge, 60), 1.7e308); +} From c543d8c7e20cfb7bad43e3402ff797cea89d472e Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 08:08:12 +0000 Subject: [PATCH 58/59] docs: record per-series arithmetic and compensated summation coverage Co-Authored-By: Claude Opus 5.5 --- .../develop_docs/physical-compile-coverage.md | 27 ++++++++++++++----- 1 file changed, 21 insertions(+), 6 deletions(-) diff --git a/docs/develop_docs/physical-compile-coverage.md b/docs/develop_docs/physical-compile-coverage.md index 31c69782..5011f61e 100644 --- a/docs/develop_docs/physical-compile-coverage.md +++ b/docs/develop_docs/physical-compile-coverage.md @@ -120,6 +120,22 @@ it. Totals are unchanged: 19 Supported, 5 Partial, 5 Missing, 2 Backend. +## Covered by per-series arithmetic + +| Row | Change | +|---|---| +| 5 | Query-time `Binary` over rows with a series identity, such as per-series readouts of stored state, uses the Fallback's `series_labels` and `series_binary`. Examples: `avg_over_time` as stored sum/count, and `rate(a) / rate(b)`. Matching drops `__name__` and honors `on`/`ignoring` when the payload carries them. Now Supported. | +| 4, 8 | A literal operand also applies to per-series rows and drops `__name__`. | + +Grouped `sum`/`avg`, current-series `Sum`/`Average` readouts, and +`sum_over_time`/`avg_over_time` use Prometheus' Kahan-Neumaier summation. An +average switches to an incremental mean once the running sum would overflow. +Stored exact `Sum` state still sums without compensation, because a +compensation term would change the stored state layout. Its checked +`avg_over_time` division therefore fails instead of returning a finite mean. + +Totals after this change: 20 Supported, 4 Partial, 5 Missing, 2 Backend. + ## Remaining In order of backend usage: @@ -156,11 +172,10 @@ In order of backend usage: After these shapes are covered, the backend can delete rows 28 and 30. 2. Row 7: comparison filters and `bool` comparisons. This needs `return_bool` in the `Binary` payload. `compile` currently rejects comparisons. -3. Row 5 for per-series rows in a `Binary` payload node: its grouped-row join - still rejects `$promql_series_identity`. It could reuse the Fallback's - `series_labels` and `series_binary` operators. -4. Rows 25 and 27: constant weights and `EntityIdentity` items for precompute +3. Rows 25 and 27: constant weights and `EntityIdentity` items for precompute `SummaryAgg`. -5. Row 16: a label-map sketch-state readout, the counterpart of +4. Row 16: a label-map sketch-state readout, the counterpart of `compile_exact_readout`, and MetricsQL `__name__` retention rules. -6. Row 20: summary join, subtract, and delete. +5. Row 20: summary join, subtract, and delete. +6. Compensated stored exact `Sum` state, a state-layout change shared with the + backend's stored-state decoding. From 000c1ef6350d1ae1f6d6c1c0759fba8030026cca Mon Sep 17 00:00:00 2001 From: zzylol Date: Wed, 30 Sep 2026 08:13:29 +0000 Subject: [PATCH 59/59] test: cover infinite sums and name removal on stored readouts; clarify docs Addresses review: pin Inf/-Inf sum and average cases, show that literal arithmetic drops __name__ from a stored readout, correct the zero-start comment, and note one-to-one-only coverage and the SQL SUM/AVG change. Co-Authored-By: Claude Opus 5.5 --- .../asap-physical-operators/src/operators/aggregate/mod.rs | 6 +++--- .../asap-physical-operators/tests/deployment_computation.rs | 2 ++ crates/asap-physical-operators/tests/promql_fallback.rs | 6 ++++++ docs/develop_docs/physical-compile-coverage.md | 4 +++- 4 files changed, 14 insertions(+), 4 deletions(-) diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs index 3b065172..e7e9ed22 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/mod.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -335,9 +335,9 @@ async fn reduce_one( Ok(Value::Float64(if matches!(measure, Reduction::Avg(_)) { promql_avg(&floats) } else { - // Prometheus starts `sum` from the first value; -0.0 keeps an - // all-negative-zero group's sign. - promql_sum(-0.0, &floats) + // Prometheus starts `sum` from the first value; adding it to 0 gives + // the same sum and compensation. + promql_sum(0., &floats) })) } diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index 58adc8c2..d11de58e 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -458,6 +458,8 @@ fn per_series_scalar_arithmetic_applies_to_stored_readouts() { for (query, expected) in [ ("rate(m[5m]) * 2", 50. / 300. * 2.), ("1 - rate(m[5m])", 1. - 50. / 300.), + // The stored sum readout keeps `__name__`; the arithmetic drops it. + ("sum_over_time(m[5m]) * 2", 150. * 2.), ] { assert_eq!( run_series(&exact_dag(query), &samples, 300_000).unwrap(), diff --git a/crates/asap-physical-operators/tests/promql_fallback.rs b/crates/asap-physical-operators/tests/promql_fallback.rs index 48603791..a9decb52 100644 --- a/crates/asap-physical-operators/tests/promql_fallback.rs +++ b/crates/asap-physical-operators/tests/promql_fallback.rs @@ -633,6 +633,12 @@ fn sums_and_averages_are_compensated_like_prometheus() { let huge = &[("a", 10, 1.7e308), ("a", 20, 1.7e308)]; assert_eq!(one("avg_over_time(m[1m])", huge, 60), 1.7e308); assert_eq!(one("sum_over_time(m[1m])", huge, 60), f64::INFINITY); + let infinite = &[("a", 10, f64::INFINITY), ("a", 20, 1.)]; + assert_eq!(one("sum_over_time(m[1m])", infinite, 60), f64::INFINITY); + assert_eq!(one("avg_over_time(m[1m])", infinite, 60), f64::INFINITY); + let opposite = &[("a", 10, f64::INFINITY), ("a", 20, f64::NEG_INFINITY)]; + assert!(one("sum_over_time(m[1m])", opposite, 60).is_nan()); + assert!(one("avg_over_time(m[1m])", opposite, 60).is_nan()); let cancel = &[("a", 50, 1e100), ("b", 50, 1.), ("c", 50, -1e100)]; assert_eq!(one("sum(m)", cancel, 60), 1.); assert_eq!(one("avg(m)", cancel, 60), 1. / 3.); diff --git a/docs/develop_docs/physical-compile-coverage.md b/docs/develop_docs/physical-compile-coverage.md index 5011f61e..f992412b 100644 --- a/docs/develop_docs/physical-compile-coverage.md +++ b/docs/develop_docs/physical-compile-coverage.md @@ -124,12 +124,14 @@ Totals are unchanged: 19 Supported, 5 Partial, 5 Missing, 2 Backend. | Row | Change | |---|---| -| 5 | Query-time `Binary` over rows with a series identity, such as per-series readouts of stored state, uses the Fallback's `series_labels` and `series_binary`. Examples: `avg_over_time` as stored sum/count, and `rate(a) / rate(b)`. Matching drops `__name__` and honors `on`/`ignoring` when the payload carries them. Now Supported. | +| 5 | Query-time `Binary` over rows with a series identity, such as per-series readouts of stored state, uses the Fallback's `series_labels` and `series_binary`. Examples: `avg_over_time` as stored sum/count, and `rate(a) / rate(b)`. Matching drops `__name__` and honors `on`/`ignoring` when the payload carries them. Only one-to-one arithmetic is covered; `group_left`/`group_right` stay rejected and comparisons are row 7. Now Supported. | | 4, 8 | A literal operand also applies to per-series rows and drops `__name__`. | Grouped `sum`/`avg`, current-series `Sum`/`Average` readouts, and `sum_over_time`/`avg_over_time` use Prometheus' Kahan-Neumaier summation. An average switches to an incremental mean once the running sum would overflow. +The grouped path also serves SQL `SUM`/`AVG` over Float64, which are now +compensated the same way. Stored exact `Sum` state still sums without compensation, because a compensation term would change the stored state layout. Its checked `avg_over_time` division therefore fails instead of returning a finite mean.