From 874d86a4ca247aec378df2aa0578a89a8048afe1 Mon Sep 17 00:00:00 2001 From: zz_y Date: Wed, 23 Sep 2026 20:51:56 +0000 Subject: [PATCH 01/90] feat: own shared physical DAG execution in Planner --- Cargo.lock | 61 +- Cargo.toml | 2 + crates/asap-physical-operators/Cargo.toml | 26 + crates/asap-physical-operators/README.md | 69 + .../count_min_sketch_accumulator.rs | 1323 +++++++++++++++++ .../count_min_sketch_with_heap_accumulator.rs | 832 +++++++++++ .../accumulators/count_sketch_accumulator.rs | 678 +++++++++ .../count_sketch_with_heap_accumulator.rs | 575 +++++++ .../datasketches_kll_accumulator.rs | 727 +++++++++ .../src/accumulators/dd_sketch_accumulator.rs | 665 +++++++++ .../src/accumulators/exact_accumulator.rs | 326 ++++ .../accumulators/hll_sketch_accumulator.rs | 788 ++++++++++ .../src/accumulators/hydra_kll_accumulator.rs | 165 ++ .../src/accumulators/increase_accumulator.rs | 742 +++++++++ .../src/accumulators/keyed_counter_state.rs | 529 +++++++ .../src/accumulators/keyed_max_state.rs | 335 +++++ .../src/accumulators/keyed_min_state.rs | 335 +++++ .../keyed_sum_count_accumulator.rs | 558 +++++++ .../src/accumulators/max_accumulator.rs | 248 +++ .../src/accumulators/min_accumulator.rs | 253 ++++ .../src/accumulators/mod.rs | 37 + .../sketch_envelope_accumulator.rs | 154 ++ .../src/accumulators/sum_accumulator.rs | 413 +++++ .../src/accumulators/univmon_accumulator.rs | 234 +++ .../src/aggregation_type.rs | 215 +++ .../asap-physical-operators/src/arithmetic.rs | 19 + .../asap-physical-operators/src/capability.rs | 129 ++ .../src/dag/batch_execution.rs | 200 +++ crates/asap-physical-operators/src/dag/mod.rs | 517 +++++++ .../src/dag/operators.rs | 1170 +++++++++++++++ .../src/dag/planner.rs | 558 +++++++ .../asap-physical-operators/src/dag/tests.rs | 260 ++++ .../asap-physical-operators/src/dag/values.rs | 327 ++++ crates/asap-physical-operators/src/factory.rs | 1190 +++++++++++++++ .../src/key_by_label_values.rs | 164 ++ crates/asap-physical-operators/src/lib.rs | 23 + .../src/measurement.rs | 94 ++ .../asap-physical-operators/src/statistic.rs | 67 + crates/asap-physical-operators/src/traits.rs | 357 +++++ .../tests/deployment.rs | 96 ++ .../tests/physical_dag.rs | 814 ++++++++++ crates/asap_sketch_codec/Cargo.toml | 8 + crates/asap_sketch_codec/src/lib.rs | 84 ++ docs/design_docs/physical-operators.md | 85 ++ 44 files changed, 16447 insertions(+), 5 deletions(-) create mode 100644 crates/asap-physical-operators/Cargo.toml create mode 100644 crates/asap-physical-operators/README.md create mode 100644 crates/asap-physical-operators/src/accumulators/count_min_sketch_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/count_min_sketch_with_heap_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/count_sketch_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/count_sketch_with_heap_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/datasketches_kll_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/dd_sketch_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/exact_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/hll_sketch_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/hydra_kll_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/increase_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/keyed_counter_state.rs create mode 100644 crates/asap-physical-operators/src/accumulators/keyed_max_state.rs create mode 100644 crates/asap-physical-operators/src/accumulators/keyed_min_state.rs create mode 100644 crates/asap-physical-operators/src/accumulators/keyed_sum_count_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/max_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/min_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/mod.rs create mode 100644 crates/asap-physical-operators/src/accumulators/sketch_envelope_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/sum_accumulator.rs create mode 100644 crates/asap-physical-operators/src/accumulators/univmon_accumulator.rs create mode 100644 crates/asap-physical-operators/src/aggregation_type.rs create mode 100644 crates/asap-physical-operators/src/arithmetic.rs create mode 100644 crates/asap-physical-operators/src/capability.rs create mode 100644 crates/asap-physical-operators/src/dag/batch_execution.rs create mode 100644 crates/asap-physical-operators/src/dag/mod.rs create mode 100644 crates/asap-physical-operators/src/dag/operators.rs create mode 100644 crates/asap-physical-operators/src/dag/planner.rs create mode 100644 crates/asap-physical-operators/src/dag/tests.rs create mode 100644 crates/asap-physical-operators/src/dag/values.rs create mode 100644 crates/asap-physical-operators/src/factory.rs create mode 100644 crates/asap-physical-operators/src/key_by_label_values.rs create mode 100644 crates/asap-physical-operators/src/lib.rs create mode 100644 crates/asap-physical-operators/src/measurement.rs create mode 100644 crates/asap-physical-operators/src/statistic.rs create mode 100644 crates/asap-physical-operators/src/traits.rs create mode 100644 crates/asap-physical-operators/tests/deployment.rs create mode 100644 crates/asap-physical-operators/tests/physical_dag.rs create mode 100644 crates/asap_sketch_codec/Cargo.toml create mode 100644 crates/asap_sketch_codec/src/lib.rs create mode 100644 docs/design_docs/physical-operators.md diff --git a/Cargo.lock b/Cargo.lock index d28e1bc5..6cd77629 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -162,7 +162,7 @@ dependencies = [ "arrow-schema", "arrow-select", "atoi", - "base64", + "base64 0.22.1", "chrono", "comfy-table", "half", @@ -309,7 +309,7 @@ version = "0.1.0" dependencies = [ "asap-frontend-promql", "asap-types", - "asap_sketchlib", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", "serde", "serde_json", "thiserror 2.0.18", @@ -369,11 +369,31 @@ dependencies = [ "asap-frontend-promql", "asap-frontend-sql", "asap-types", - "asap_sketchlib", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", "serde_json", "tokio", ] +[[package]] +name = "asap-physical-operators" +version = "0.1.0" +dependencies = [ + "asap-types", + "asap_sketch_codec", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=026cd18c7b8c23ae6c46d4d683151ba562b8cd3a)", + "base64 0.21.7", + "bincode", + "futures", + "hex", + "prost", + "rmp-serde", + "serde", + "serde_json", + "thiserror 2.0.18", + "tracing", + "xxhash-rust", +] + [[package]] name = "asap-sql-function-catalog" version = "0.1.0" @@ -387,6 +407,31 @@ dependencies = [ "thiserror 2.0.18", ] +[[package]] +name = "asap_sketch_codec" +version = "0.1.0" +dependencies = [ + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=026cd18c7b8c23ae6c46d4d683151ba562b8cd3a)", + "prost", +] + +[[package]] +name = "asap_sketchlib" +version = "0.3.0" +source = "git+https://github.com/ProjectASAP/asap_sketchlib?rev=026cd18c7b8c23ae6c46d4d683151ba562b8cd3a#026cd18c7b8c23ae6c46d4d683151ba562b8cd3a" +dependencies = [ + "bytes", + "prost", + "rand 0.9.5", + "rmp-serde", + "serde", + "serde-big-array", + "serde_bytes", + "smallvec", + "twox-hash 2.1.2", + "xxhash-rust", +] + [[package]] name = "asap_sketchlib" version = "0.3.0" @@ -448,6 +493,12 @@ version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c08606f8c3cbf4ce6ec8e28fb0014a2c086708fe954eaa885384a6165172e7e8" +[[package]] +name = "base64" +version = "0.21.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9d297deb1925b89f2ccc13d7635fa0714f12c87adce1c75356b39ca9b7178567" + [[package]] name = "base64" version = "0.22.1" @@ -942,7 +993,7 @@ checksum = "f52c4012648b34853e40a2c6bcaa8772f837831019b68aca384fb38436dba162" dependencies = [ "arrow", "arrow-buffer", - "base64", + "base64 0.22.1", "blake2", "blake3", "chrono", @@ -2168,7 +2219,7 @@ dependencies = [ "arrow-ipc", "arrow-schema", "arrow-select", - "base64", + "base64 0.22.1", "brotli", "bytes", "chrono", diff --git a/Cargo.toml b/Cargo.toml index 9d44070d..037a4d8b 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,5 +1,7 @@ [workspace] members = [ + "crates/asap-physical-operators", + "crates/asap_sketch_codec", "crates/types", "crates/sql-function-catalog", "crates/asap-aware-mapping", diff --git a/crates/asap-physical-operators/Cargo.toml b/crates/asap-physical-operators/Cargo.toml new file mode 100644 index 00000000..8fe7eb31 --- /dev/null +++ b/crates/asap-physical-operators/Cargo.toml @@ -0,0 +1,26 @@ +[package] +name = "asap-physical-operators" +version = "0.1.0" +edition = "2021" + +[dependencies] +futures = "0.3" +planner-types = { package = "asap-types", path = "../types" } +asap_sketch_codec = { path = "../asap_sketch_codec" } +asap_sketchlib = { git = "https://github.com/ProjectASAP/asap_sketchlib", rev = "026cd18c7b8c23ae6c46d4d683151ba562b8cd3a" } +serde = { version = "1", features = ["derive", "rc"] } +serde_json = "1" +tracing = "0.1" +thiserror = "2" +base64 = "0.21" +bincode = "1.3" +rmp-serde = "1.3" +prost = "0.13" +xxhash-rust = { version = "0.8", features = ["xxh32", "xxh64"] } + +[features] +default = [] +extra_debugging = [] + +[dev-dependencies] +hex = "0.4" diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md new file mode 100644 index 00000000..8ff91704 --- /dev/null +++ b/crates/asap-physical-operators/README.md @@ -0,0 +1,69 @@ +# ASAP physical operators + +An independent Rust physical operator DAG runtime shared by ingestion time and +query time execution. The library requires neither backend engine, a server, +a storage implementation, Arrow nor DataFusion. DataFusion informed the design; +it is not the execution framework. + +`dag::PhysicalDag` binds typed operator inputs to node IDs. Each execution starts +one producer per reachable node, shares output batches among its consumers, and +bounds buffering. Dropping one consumer does not cancel other consumers. A +`RunContext` carries query or ingestion scope, cancellation and byte accounting. +Executions use the caller's worker and worker-local streams, with no internal +thread pool. Poll multiple root streams concurrently when they share inputs. + +`dag::operators::Operator` implements native batch sources, scalar values, +projection, filtering, grouped exact aggregation, semi-join, grouped Sort and +Limit, vector-to-scalar conversion, Union, and summary construction/merge/readout. +Sort followed by Limit implements grouped ranking; no dedicated TopK physical +operator is needed. Summary construction updates state batch by batch. End of +input means the supplied query range or ingestion window is complete. + +```rust +use asap_physical_operators::dag::{ + operators::{Expression, Operator}, + values::Value, + Limits, PhysicalDag, RunContext, Scope, +}; +use asap_physical_operators::planner::pre_asap::DataType; +use futures::{executor::block_on, StreamExt}; + +let source = Operator::scalar(Value::Int64(7), DataType::Int64)?; +let negate = Operator::project(source.schema(), vec![ + ("value".into(), Expression::Negate(Box::new(Expression::Column(0)))), +])?; +let mut plan = PhysicalDag::default(); +plan.add(0, vec![], source)?; +plan.add(1, vec![0], negate)?; +let run = RunContext::new( + Scope::Query { evaluation_time_ms: 1000, revision: 1 }, + Limits::default(), +)?; +let mut output = plan.execute(&[1], run)?.remove(0); +let batch = block_on(output.next()).unwrap()?; +assert!(matches!(batch.rows()[0][0], Value::Int64(-7))); +# Ok::<(), asap_physical_operators::dag::Error>(()) +``` + +`dag::planner::bind` accepts a post-ASAP DAG and explicit source bindings for +installed ingestion/storage frontiers. It rejects unsupported operations and +schema mismatches before starting a source. Implement `PhysicalOperator` for a +deployment source, including asynchronous I/O; computation operators remain in +the library. The public `planner` export identifies the exact Planner types used +by the crate. The native binder currently supports a subset of those types and +operations; it does not interpret an unknown node as external fallback. + +Plain values preserve Planner scalar/collection types and nullability. Numeric +arithmetic uses matching Int64 or Float64 inputs; integer overflow is an error. +Boolean predicates use three-valued logic. Native summary states currently cover +exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch and HLL. Binding checks family, +parameters and readout compatibility; source batches also validate state payloads. +Existing accumulator algorithms are reused as kernels behind these operators. + +This crate is owned by ASAPPlanner. Its `planner-types` dependency is the local +IR crate, so a contract change and its execution tests belong in the same PR. +Deployments supply storage/ingestion sources and adapt output protocols. The +library has no ASAPQuery-backend dependency. Backend raw Scan remains a separate +deployment capability. + +See [the design](../../docs/design_docs/physical-operators.md). diff --git a/crates/asap-physical-operators/src/accumulators/count_min_sketch_accumulator.rs b/crates/asap-physical-operators/src/accumulators/count_min_sketch_accumulator.rs new file mode 100644 index 00000000..e9d7fd64 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/count_min_sketch_accumulator.rs @@ -0,0 +1,1323 @@ +use crate::accumulators::dd_sketch_accumulator::normalize_sample_p; +use crate::{ + AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, + MultipleSubpopulationAggregate, SerializableToSink, +}; +use asap_sketchlib::{CountMinSketch, CountMinSketchDelta, MessagePackCodec}; +use serde_json::Value; +use std::collections::HashMap; + +use crate::Statistic; + +/// Count-Min Sketch accumulator — wraps asap_sketchlib::CountMinSketch. +/// Core struct, update/merge/serde logic live in `asap_sketchlib::sketches`. +/// This file retains QE-specific trait impls, legacy deserializers, and JSON output. +#[derive(Debug, Clone)] +pub struct CountMinSketchAccumulator { + pub inner: CountMinSketch, + /// Edge sampling probability `p ∈ (0,1]` carried on the producer's + /// `SketchEnvelope.sample_p`. The edge admits each insert with + /// probability `p`, so every stored cell count is ~`p`× the true count. + /// CMS is L1/additive and linear, so the unbiased rescale of BOTH a + /// point-frequency estimate (`query_key`) and the aggregate + /// total-event statistics (`Count`/`Sum`/`Increase`/`Rate`) is `×1/p`. + /// `1.0` (and the proto3 default `0.0`, dual-read as `1.0`) means no + /// sampling, so the rescale is a no-op and the behaviour is identical + /// to before. Mirrors `DDSketchAccumulator::sample_p`; set from the + /// envelope at the `from_sketchlib_proto_bytes` decode site and + /// preserved across `reset_to_empty` and `merge_with`. + pub sample_p: f64, +} + +impl CountMinSketchAccumulator { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + inner: CountMinSketch::new(row_num, col_num), + sample_p: 1.0, + } + } + + // Marked as _update and kept private; only called internally. + fn _update(&mut self, key: &KeyByLabelValues, value: f64) { + self.inner.update(&key.to_semicolon_str(), value); + } + + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + // The edge sampled inserts with probability `sample_p`, so the + // stored point-frequency estimate is ~`p`× the true frequency. + // CMS is linear/additive, so `×1/p` is the unbiased rescale. + // `sample_p == 1.0` (unsampled / legacy) makes this a no-op. + self.inner.estimate(&key.to_semicolon_str()) / self.sample_p + } + + pub fn deserialize_from_json(data: &Value) -> Result> { + let row_num = data["row_num"] + .as_f64() + .ok_or("Missing or invalid 'row_num' field")? as usize; + let col_num = data["col_num"] + .as_f64() + .ok_or("Missing or invalid 'col_num' field")? as usize; + + let sketch_data = data["sketch"] + .as_array() + .ok_or("Missing or invalid 'sketch' field")?; + + let mut sketch = Vec::new(); + for row in sketch_data { + let row_array = row.as_array().ok_or("Invalid row in sketch data")?; + let mut sketch_row = Vec::new(); + for cell in row_array { + let value = cell.as_f64().ok_or("Invalid cell value in sketch data")?; + sketch_row.push(value); + } + sketch.push(sketch_row); + } + + Ok(Self { + inner: CountMinSketch::from_legacy_matrix(sketch, row_num, col_num), + sample_p: 1.0, + }) + } + + /// Decode from the modified OTLP wire format's + /// `CountMinSketchDataPoint.sketch` bytes when + /// `encoding = COUNT_MIN_SKETCH_ENCODING_MSGPACK`. The bytes are the + /// MessagePack serialization of the cross-language sketch-core + /// `CountMinSketch` wire struct (same format the legacy Arroyo path + /// uses — this method is the modified-OTLP entrypoint for PR I). + pub fn from_msgpack_bytes(buffer: &[u8]) -> Result> { + Ok(Self { + inner: CountMinSketch::from_msgpack(buffer) + .map_err(|e| -> Box { e.to_string().into() })?, + // The msgpack CountMinSketch struct carries no envelope/sample_p; + // the msgpack path is parity/test-only and is never edge-sampled. + sample_p: 1.0, + }) + } + + /// Decode from the modified OTLP wire format's + /// `CountMinSketchDataPoint.sketch` bytes — i.e. the protobuf-encoded + /// `asap_sketchlib::proto::sketchlib::CountMinState` message used by + /// DataCollector's `countminsketchprocessor` when emitting via + /// `Metric.data = CountMinSketch{…}` with + /// `encoding = COUNT_MIN_SKETCH_ENCODING_PROTO`. + /// + /// The resulting accumulator is constructed via + /// `CountMinSketch::from_legacy_matrix` after reshaping the flat + /// `counts_int` / `counts_float` field into a `Vec>`. + pub fn from_sketchlib_proto_bytes(buffer: &[u8]) -> Result> { + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, CountMinState, CounterType, SketchEnvelope, + }; + use prost::Message; + + // DataCollector's countminsketchprocessor wraps the state in a + // `SketchEnvelope{count_min: CountMinState}` via + // `SerializePortableFO` + `proto.Marshal`. Try decoding as envelope + // first, fall back to bare `CountMinState` for callers (e.g. unit + // tests) that encode the state directly. Capture the envelope's + // `sample_p` alongside the state so the point-frequency + // (`query_key`) and aggregate statistics rescale by `1/p`. Bare + // `CountMinState` bytes (no envelope) carry no sampling info → + // `sample_p` 1.0 (no rescale). Mirrors `DDSketchAccumulator`. + let (state, sample_p) = match SketchEnvelope::decode(buffer) { + Ok(env) => { + let sp = env.sample_p; + match env.sketch_state { + Some(sketch_envelope::SketchState::CountMin(st)) => (st, sp), + Some(other) => { + return Err(format!( + "SketchEnvelope contains non-CountMin sketch: {:?}", + std::mem::discriminant(&other) + ) + .into()); + } + // Envelope decoded but was empty (e.g. the buffer is a + // bare CountMinState that happened to parse as a default + // envelope). Fall through to bare decode. + None => ( + CountMinState::decode(buffer) + .map_err(|e| format!("decode CountMinState: {e}"))?, + 1.0, + ), + } + } + Err(_) => ( + CountMinState::decode(buffer).map_err(|e| format!("decode CountMinState: {e}"))?, + 1.0, + ), + }; + let rows = state.rows as usize; + let cols = state.cols as usize; + // Defensive dim validation BEFORE reconstructing the matrix: + // reject degenerate / narrow-hash-budget-violating / absurdly + // oversized dims so a malformed payload fails gracefully (the + // ingest caller skips the data point) instead of building a + // degenerate or huge matrix. + validate_sketch_dims("CountMinState", rows, cols)?; + let expected_len = rows * cols; + let counter_type = CounterType::try_from(state.counter_type).map_err(|_| { + format!( + "CountMinState has unknown counter_type tag {}", + state.counter_type + ) + })?; + let flat: Vec = match counter_type { + CounterType::Int32 | CounterType::Int64 => { + if state.counts_int.len() != expected_len { + return Err(format!( + "CountMinState counts_int has {} entries, expected rows*cols = {}", + state.counts_int.len(), + expected_len + ) + .into()); + } + state.counts_int.iter().map(|&v| v as f64).collect() + } + CounterType::Float64 => { + if state.counts_float.len() != expected_len { + return Err(format!( + "CountMinState counts_float has {} entries, expected rows*cols = {}", + state.counts_float.len(), + expected_len + ) + .into()); + } + state.counts_float.clone() + } + // INT128 stores (hi, lo) pairs and would have 2 * rows * cols + // entries in counts_int; defer to PR C if a producer ever uses it. + other => { + return Err(format!( + "CountMinState counter_type {other:?} not yet supported \ + (PR C will extend coverage)" + ) + .into()); + } + }; + let mut matrix = Vec::with_capacity(rows); + for r in 0..rows { + let start = r * cols; + matrix.push(flat[start..start + cols].to_vec()); + } + Ok(Self { + inner: CountMinSketch::from_legacy_matrix(matrix, rows, cols), + sample_p: normalize_sample_p(sample_p), + }) + } + + /// Apply a proto-encoded `CountMinDelta` frame to this + /// accumulator's inner sketch — the decode path for + /// `COUNT_MIN_SKETCH_ENCODING_PROTO_DELTA` (paper §6.2 B3 / B4). + pub fn apply_proto_delta_bytes( + &mut self, + buffer: &[u8], + ) -> Result<(), Box> { + use asap_sketchlib::proto::sketchlib::CountMinDelta as PbDelta; + use prost::Message; + + let pb = PbDelta::decode(buffer).map_err(|e| format!("decode CountMinDelta: {e}"))?; + + if pb.cell_rows.len() != pb.cell_cols.len() || pb.cell_rows.len() != pb.d_counts.len() { + return Err(format!( + "CountMinDelta packed-array length mismatch: \ + cell_rows={}, cell_cols={}, d_counts={}", + pb.cell_rows.len(), + pb.cell_cols.len(), + pb.d_counts.len() + ) + .into()); + } + let cells = pb + .cell_rows + .iter() + .zip(pb.cell_cols.iter()) + .zip(pb.d_counts.iter()) + .map(|((r, c), dc)| (*r, *c, *dc)) + .collect(); + let delta = CountMinSketchDelta { + rows: pb.rows, + cols: pb.cols, + cells, + l1: pb.l1, + l2: pb.l2, + // The Go-side CountMinDelta proto now carries an hh_keys field + // (heavy-hitter candidates), mirrored on asap_sketchlib's + // CountMinSketchDelta. The vendored Rust proto bindings here don't + // decode it yet, and CountMin has no TopK to rebuild, so pass an + // empty set — same handling as CountSketch's hh_keys. + hh_keys: Vec::new(), + }; + self.inner + .apply_delta(&delta) + .map_err(|e| format!("apply CountMinDelta: {e}"))?; + Ok(()) + } + + pub fn deserialize_from_bytes(buffer: &[u8]) -> Result> { + if buffer.len() < 8 { + return Err("Buffer too short for row_num and col_num".into()); + } + + // TODO: this logic will need to be checked for i32 -> f64 + // Github Issue #11 + + let row_num = u32::from_le_bytes([buffer[0], buffer[1], buffer[2], buffer[3]]) as usize; + let col_num = u32::from_le_bytes([buffer[4], buffer[5], buffer[6], buffer[7]]) as usize; + + let expected_size = 8 + (row_num * col_num * 4); + if buffer.len() < expected_size { + return Err("Buffer too short for sketch data".into()); + } + + let mut sketch = Vec::new(); + let mut offset = 8; + + for _ in 0..row_num { + let mut row = Vec::new(); + for _ in 0..col_num { + let value = f64::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + buffer[offset + 4], + buffer[offset + 5], + buffer[offset + 6], + buffer[offset + 7], + ]); + row.push(value); + offset += 8; + } + sketch.push(row); + } + + Ok(Self { + inner: CountMinSketch::from_legacy_matrix(sketch, row_num, col_num), + sample_p: 1.0, + }) + } + + /// Merge multiple accumulators efficiently without cloning all of them. + pub fn merge_multiple( + accumulators: &[Box], + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + + let mut cms_accumulators = Vec::with_capacity(accumulators.len()); + for acc in accumulators { + if acc.get_accumulator_type() != AggregationType::CountMinSketch { + return Err(format!( + "Cannot merge CountMinSketchAccumulator with {:?}", + acc.get_accumulator_type() + ) + .into()); + } + let cms_acc = acc + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountMinSketchAccumulator")?; + cms_accumulators.push(cms_acc); + } + + // Check dimensions are consistent + let rows = cms_accumulators[0].inner.rows(); + let cols = cms_accumulators[0].inner.cols(); + for acc in &cms_accumulators { + if acc.inner.rows() != rows || acc.inner.cols() != cols { + return Err( + "Cannot merge CountMinSketch accumulators with different dimensions".into(), + ); + } + } + + let inner_refs: Vec<&CountMinSketch> = + cms_accumulators.iter().map(|acc| &acc.inner).collect(); + let merged_inner = CountMinSketch::merge_refs(&inner_refs)?; + // sample_p is a per-series config constant, so all operands carry the + // same value in practice. Mirror DDSketch's merge policy: prefer a + // sampled factor (< 1.0) over the no-sampling default so a merge with + // a freshly-reset (1.0) base keeps the series' sampling rate. + let sample_p = cms_accumulators + .iter() + .map(|acc| acc.sample_p) + .find(|&p| p < 1.0) + .unwrap_or(cms_accumulators[0].sample_p); + Ok(Self { + inner: merged_inner, + sample_p, + }) + } +} + +/// Defensive upper bound on the number of matrix cells (`rows * cols`) +/// we'll reconstruct from an inbound wire-declared CMS / CountSketch +/// dimension pair. A malformed / hostile payload could declare absurd +/// dims (e.g. `rows = cols = u32::MAX`) and trick the decoder into a +/// huge `Vec` allocation before the `counts_*.len() != rows*cols` +/// check ever runs. Realistic sketches are at most a few hundred rows +/// by tens-of-thousands of columns, so 8M cells (~64 MiB of f64) is a +/// generous ceiling that no legitimate producer reaches. +pub(crate) const MAX_SKETCH_CELLS: usize = 8 * 1024 * 1024; + +/// Validate an inbound, wire-declared `(rows, cols)` pair for a +/// matrix-backed frequency sketch (CMS / CountSketch) BEFORE any matrix +/// is reconstructed from it. Returns `Ok(())` for dimensions a +/// legitimate producer could have emitted, and an `Err` (never a panic) +/// for malformed / degenerate ones so the ingest path can skip the data +/// point and fall through to its existing decode-failure accounting. +/// +/// Rejections: +/// 1. `rows < 1` or `cols < 1` — a zero-dim matrix has no cells. +/// 2. Narrow-hash-budget violation. The cross-language wire hasher +/// (`sketchlib`'s `MatrixHashType::Packed64`) derives every row's +/// column index from disjoint bit-fields of a single 64-bit hash +/// word: row `r` reads `mask_bits = ceil(log2(cols))` bits at offset +/// `r * mask_bits`. Once `rows * mask_bits > 64` the per-row column +/// slices overflow / alias the 64-bit word and the matrix-cell +/// layout is no longer the one the producer hashed into — the sketch +/// is internally degenerate. This mirrors sketchlib's own +/// `MatrixFastHash::assert_compatible` budget (`rows * (mask_bits + 1) <= 64`); we check the column-index bits alone so realistic +/// configs (5x2048, 5x4096, 5x2000) — for which the sign bits share +/// the top of the word without affecting the cell layout — still +/// pass. +/// 3. Obviously-oversized dims: `rows * cols > MAX_SKETCH_CELLS`, +/// guarding against a huge allocation from a malformed payload. +/// +/// `what` names the wire struct for the error message (e.g. +/// `"CountMinState"`). +pub(crate) fn validate_sketch_dims(what: &str, rows: usize, cols: usize) -> Result<(), String> { + if rows < 1 || cols < 1 { + return Err(format!( + "{what} has degenerate dims (rows={rows}, cols={cols}); rejecting" + )); + } + // mask_bits = ceil(log2(cols)); cols >= 1 here. ilog2 is floor(log2). + let mask_bits = if cols.is_power_of_two() { + cols.ilog2() as usize + } else { + cols.ilog2() as usize + 1 + }; + if rows.saturating_mul(mask_bits) > 64 { + return Err(format!( + "{what} dims (rows={rows}, cols={cols}) exceed the 64-bit \ + packed-hash column budget (rows * ceil(log2(cols)) = {} > 64); \ + the sketch's matrix-cell layout is degenerate, rejecting", + rows.saturating_mul(mask_bits) + )); + } + if rows.saturating_mul(cols) > MAX_SKETCH_CELLS { + return Err(format!( + "{what} dims (rows={rows}, cols={cols}) declare {} cells, \ + exceeding the {MAX_SKETCH_CELLS}-cell ingest cap; rejecting to \ + avoid a huge allocation from a malformed payload", + rows.saturating_mul(cols) + )); + } + Ok(()) +} + +impl SerializableToSink for CountMinSketchAccumulator { + fn serialize_to_json(&self) -> Value { + serde_json::json!({ + "row_num": self.inner.rows(), + "col_num": self.inner.cols(), + "sketch": self.inner.sketch() + }) + } + + fn serialize_to_bytes(&self) -> Vec { + self.inner.to_msgpack().unwrap_or_default() + } +} + +impl AggregateCore for CountMinSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "CountMinSketchAccumulator" + } + + /// Per-window base rotation: rebuild an empty counter matrix with + /// the same (rows, cols) so the next window's additive cell deltas + /// align to the identical hash geometry. `sample_p` is a per-series + /// config constant (not per-window data), so it is intentionally + /// preserved across the rotation — mirrors `DDSketchAccumulator`. + fn reset_to_empty(&mut self) { + self.inner = CountMinSketch::new(self.inner.rows(), self.inner.cols()); + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge CountMinSketchAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + + let other_cms = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountMinSketchAccumulator")?; + + let merged_inner = CountMinSketch::merge_refs(&[&self.inner, &other_cms.inner])?; + // Mirror DDSketchAccumulator's merge policy exactly: sample_p is a + // per-series config constant, so both operands carry the same value + // in practice. Prefer a sampled factor over the no-sampling default + // so a merge with a freshly-reset (1.0) base keeps the series' + // sampling rate. + let sample_p = if self.sample_p < 1.0 { + self.sample_p + } else { + other_cms.sample_p + }; + Ok(Box::new(Self { + inner: merged_inner, + sample_p, + })) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::CountMinSketch + } + + fn approx_memory_bytes(&self) -> usize { + // Conservative constant for the CountMinSketch counter matrix. + // Real per-instance sizing would require exposing rows/cols on + // the inner sketch; 16 KiB is a reasonable v1 default. + 16 * 1024 + } + + fn get_keys(&self) -> Option> { + None + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + key: &Option, + query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::MultipleSubpopulationAggregate; + use crate::Statistic; + + // Key-provided path: route to MultipleSubpopulationAggregate::query + // (the canonical "what's the count of this key?" lookup). + if let Some(key_val) = key.as_ref() { + return self.query(statistic, key_val, Some(query_kwargs)); + } + if let Some(k) = query_kwargs.get("key") { + let key_val = crate::KeyByLabelValues::new_with_labels(vec![k.clone()]); + return self.query(statistic, &key_val, Some(query_kwargs)); + } + + // No-key path: return total event volume. The min-row-sum is the + // canonical CMS estimator for "how many inserts were observed" — + // each insert increments exactly one cell per row, so every row + // sums to the true insert count (modulo collisions, which CMS + // never *underestimates*; min is the tightest upper bound). + // + // When the edge sampled this series (sample_p < 1.0), each insert + // was admitted w.p. `p`, so the stored min-row-sum is ~`p`× the + // true event count. CMS is L1/additive and linear, so rescale by + // `1/sample_p` for an unbiased estimate. `sample_p == 1.0` + // (unsampled / legacy) makes this a no-op. This rescales BOTH the + // Count/Sum/Increase statistics and (via the same closure) the + // Rate per-second readout. + let total_events = || -> f64 { + let matrix = self.inner.sketch(); + if matrix.is_empty() || matrix[0].is_empty() { + return 0.0; + } + let row_totals = matrix.iter().map(|r| r.iter().sum::()); + let min_total = row_totals.fold(f64::INFINITY, f64::min); + if min_total.is_finite() { + min_total / self.sample_p + } else { + 0.0 + } + }; + match statistic { + Statistic::Count | Statistic::Sum => Ok(total_events()), + // PR #111 honest-gap closure (in-the-bag for ASAP tier). + // CMS records insert counts but not timestamps, so per-second + // `rate(metric[range])` requires the engine to push the + // range duration via `query_kwargs["range_ms"]`. When + // present, divide the min-row-sum by `range_ms / 1000`. When + // absent (the engine has not been wired to inject range_ms + // for this query, e.g. instant `rate` calls outside the + // PromQL range-vector pattern), fall back to the raw event + // count so the answer is at least non-empty — the caller's + // caveat is that the units are events/window rather than + // events/second. Increase carries the same caveat. + Statistic::Rate => { + let total = total_events(); + let range_ms_str = query_kwargs.get("range_ms").map(String::as_str); + let Some(s) = range_ms_str else { + return Ok(total); + }; + let range_ms: f64 = s + .parse() + .map_err(|e| format!("CountMinSketchAccumulator: bad range_ms='{s}': {e}"))?; + if range_ms <= 0.0 { + return Err("CountMinSketchAccumulator: range_ms must be positive".into()); + } + Ok(total * 1000.0 / range_ms) + } + Statistic::Increase => Ok(total_events()), + other => Err(format!( + "CountMinSketchAccumulator: statistic {:?} not supported \ + without a key (only Count / Sum / Rate / Increase aggregate \ + over the whole sketch)", + other, + ) + .into()), + } + } +} + +impl MultipleSubpopulationAggregate for CountMinSketchAccumulator { + fn query( + &self, + _statistic: Statistic, + key: &KeyByLabelValues, + _query_kwargs: Option<&HashMap>, + ) -> Result> { + Ok(self.query_key(key)) + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +impl MergeableAccumulator for CountMinSketchAccumulator { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + let mut iter = accumulators.into_iter(); + let mut merged = iter.next().unwrap(); + for acc in iter { + merged.inner.merge(&acc.inner)?; + } + Ok(merged) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_count_min_sketch_creation() { + let cms = CountMinSketchAccumulator::new(4, 1000); + assert_eq!(cms.inner.rows(), 4); + assert_eq!(cms.inner.cols(), 1000); + let sketch = cms.inner.sketch(); + assert_eq!(sketch.len(), 4); + assert_eq!(sketch[0].len(), 1000); + + for row in &sketch { + for &value in row { + assert_eq!(value, 0.0); + } + } + } + + #[test] + fn test_count_min_sketch_update() { + let mut cms = CountMinSketchAccumulator::new(2, 10); + let key = KeyByLabelValues::new(); + cms._update(&key, 1.0); + let result = cms.query_key(&key); + assert!(result >= 1.0); + } + + #[test] + fn test_count_min_sketch_query() { + let cms = CountMinSketchAccumulator::new(2, 10); + let key = KeyByLabelValues::new(); + assert_eq!(cms.query_key(&key), 0.0); + + let multi_trait: &dyn MultipleSubpopulationAggregate = &cms; + assert_eq!(multi_trait.query(Statistic::Sum, &key, None).unwrap(), 0.0); + } + + #[test] + fn test_count_min_sketch_merge() { + // Build controlled state via from_legacy_matrix (works for both Legacy and Sketchlib backends). + let cms1 = CountMinSketchAccumulator { + inner: CountMinSketch::from_legacy_matrix( + vec![vec![5.0, 0.0, 0.0], vec![0.0, 0.0, 10.0]], + 2, + 3, + ), + sample_p: 1.0, + }; + let cms2 = CountMinSketchAccumulator { + inner: CountMinSketch::from_legacy_matrix( + vec![vec![3.0, 7.0, 0.0], vec![0.0, 0.0, 0.0]], + 2, + 3, + ), + sample_p: 1.0, + }; + + let merged = CountMinSketchAccumulator::merge_accumulators(vec![cms1, cms2]).unwrap(); + + let merged_sketch = merged.inner.sketch(); + assert_eq!(merged_sketch[0][0], 8.0); + assert_eq!(merged_sketch[0][1], 7.0); + assert_eq!(merged_sketch[1][2], 10.0); + } + + #[test] + fn test_count_min_sketch_merge_dimension_mismatch() { + let cms1 = CountMinSketchAccumulator::new(2, 3); + let cms2 = CountMinSketchAccumulator::new(3, 3); + let result = CountMinSketchAccumulator::merge_accumulators(vec![cms1, cms2]); + assert!(result.is_err()); + } + + #[test] + fn test_count_min_sketch_as_aggregate_core() { + let cms = CountMinSketchAccumulator::new(2, 3); + assert_eq!(cms.type_name(), "CountMinSketchAccumulator"); + } + + #[test] + fn test_trait_object() { + let cms = CountMinSketchAccumulator::new(2, 3); + let trait_obj: Box = Box::new(cms); + assert_eq!(trait_obj.type_name(), "CountMinSketchAccumulator"); + } + + #[test] + fn test_count_min_sketch_key_query() { + let mut cms = CountMinSketchAccumulator::new(4, 100); + let key = KeyByLabelValues::new(); + assert_eq!(cms.query_key(&key), 0.0); + cms._update(&key, 5.0); + let result = cms.query_key(&key); + assert!(result >= 5.0); + } + + #[test] + fn test_update_and_query_use_same_key_encoding() { + // Regression test: _update and query_key must hash the same key string. + // Previously _update went through serialize_to_json (which returns a JSON + // array, so as_object() is always None) and always stored under key "". + // query_key correctly used key.labels.join(";"), so they never matched. + let mut cms = CountMinSketchAccumulator::new(4, 1000); + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string(), "prod".to_string()]); + cms._update(&key, 5.0); + let result = cms.query_key(&key); + assert!( + result >= 5.0, + "_update and query_key used different key encodings: got {result}" + ); + + // Also verify a different key does not interfere. + let other_key = KeyByLabelValues::new_with_labels(vec!["api".to_string()]); + // other_key was never updated; its estimate should be lower than key's. + let other_result = cms.query_key(&other_key); + // In a sketch this large there should be no collision, so other_result == 0. + assert_eq!( + other_result, 0.0, + "unrelated key returned non-zero: {other_result}" + ); + } + + #[test] + fn test_multiple_subpopulation_aggregate() { + let mut cms = CountMinSketchAccumulator::new(3, 50); + let key = KeyByLabelValues::new(); + cms._update(&key, 10.0); + + let multi_trait: &dyn MultipleSubpopulationAggregate = &cms; + let result = multi_trait.query(Statistic::Sum, &key, None).unwrap(); + assert!(result >= 10.0); + + let keys = multi_trait.get_keys(); + assert!(keys.is_none()); + } + + #[test] + fn test_count_min_sketch_merge_multiple() { + // Build controlled state via from_legacy_matrix (works for both Legacy and Sketchlib backends). + let cms1 = CountMinSketchAccumulator { + inner: CountMinSketch::from_legacy_matrix( + vec![vec![5.0, 0.0, 0.0], vec![0.0, 0.0, 10.0]], + 2, + 3, + ), + sample_p: 1.0, + }; + let cms2 = CountMinSketchAccumulator { + inner: CountMinSketch::from_legacy_matrix( + vec![vec![3.0, 7.0, 0.0], vec![0.0, 0.0, 0.0]], + 2, + 3, + ), + sample_p: 1.0, + }; + let cms3 = CountMinSketchAccumulator { + inner: CountMinSketch::from_legacy_matrix( + vec![vec![2.0, 0.0, 0.0], vec![0.0, 0.0, 5.0]], + 2, + 3, + ), + sample_p: 1.0, + }; + + let boxed_accs: Vec> = + vec![Box::new(cms1), Box::new(cms2), Box::new(cms3)]; + + let merged = CountMinSketchAccumulator::merge_multiple(&boxed_accs).unwrap(); + + let merged_sketch = merged.inner.sketch(); + assert_eq!(merged_sketch[0][0], 10.0); + assert_eq!(merged_sketch[0][1], 7.0); + assert_eq!(merged_sketch[1][2], 15.0); + } + + #[test] + fn test_count_min_sketch_merge_multiple_error_cases() { + let empty: Vec> = vec![]; + assert!(CountMinSketchAccumulator::merge_multiple(&empty).is_err()); + + let cms1 = CountMinSketchAccumulator::new(2, 3); + let cms2 = CountMinSketchAccumulator::new(3, 3); + let boxed_accs: Vec> = vec![Box::new(cms1), Box::new(cms2)]; + assert!(CountMinSketchAccumulator::merge_multiple(&boxed_accs).is_err()); + + use crate::accumulators::sum_accumulator::SumAccumulator; + let cms = CountMinSketchAccumulator::new(2, 3); + let sum = SumAccumulator::new(); + let mixed_accs: Vec> = vec![Box::new(cms), Box::new(sum)]; + assert!(CountMinSketchAccumulator::merge_multiple(&mixed_accs).is_err()); + } + + #[test] + fn test_from_sketchlib_proto_bytes_int64() { + // Hand-build a CountMinState proto with INT64 counters and verify + // round-tripping through from_sketchlib_proto_bytes yields the same + // matrix that the modified-OTLP wire format would carry. + use asap_sketchlib::proto::sketchlib::{CountMinState, CounterType}; + use prost::Message; + + let rows = 2u32; + let cols = 3u32; + // Row-major: row 0 = [1,2,3], row 1 = [4,5,6] + let counts_int: Vec = vec![1, 2, 3, 4, 5, 6]; + let state = CountMinState { + rows, + cols, + counter_type: CounterType::Int64 as i32, + counts_int: counts_int.clone(), + counts_float: Vec::new(), + sum_counts: Vec::new(), + sum2_counts: Vec::new(), + l1: Vec::new(), + l2: Vec::new(), + }; + let bytes = state.encode_to_vec(); + + let acc = CountMinSketchAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + let matrix = acc.inner.sketch(); + assert_eq!(matrix.len(), rows as usize); + assert_eq!(matrix[0], vec![1.0, 2.0, 3.0]); + assert_eq!(matrix[1], vec![4.0, 5.0, 6.0]); + } + + #[test] + fn test_from_sketchlib_proto_bytes_envelope_wrapped() { + // Mirrors what DataCollector's countminsketchprocessor emits: + // the state is wrapped in a `SketchEnvelope{count_min: ...}` + // via sketchlib-go's `SerializePortableFO` + `proto.Marshal`. + // Before the fix, the Rust decoder decoded the envelope bytes as + // a bare CountMinState, which produced "invalid wire type" + // errors on field `cols` and silently fell through to §5.2. + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, CountMinState, CounterType, SketchEnvelope, + }; + use prost::Message; + + let state = CountMinState { + rows: 2, + cols: 3, + counter_type: CounterType::Int64 as i32, + counts_int: vec![7, 8, 9, 10, 11, 12], + counts_float: Vec::new(), + sum_counts: Vec::new(), + sum2_counts: Vec::new(), + l1: Vec::new(), + l2: Vec::new(), + }; + let env = SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::CountMin(state)), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + + let acc = CountMinSketchAccumulator::from_sketchlib_proto_bytes(&bytes) + .expect("envelope-wrapped decode should succeed"); + let matrix = acc.inner.sketch(); + assert_eq!(matrix[0], vec![7.0, 8.0, 9.0]); + assert_eq!(matrix[1], vec![10.0, 11.0, 12.0]); + } + + #[test] + fn test_from_sketchlib_proto_bytes_envelope_wrong_sketch_type() { + // An envelope carrying a non-CountMin sketch should be rejected + // with a clear error rather than silently producing garbage. + use asap_sketchlib::proto::sketchlib::{sketch_envelope, KllState, SketchEnvelope}; + use prost::Message; + + let kll = KllState::default(); + let env = SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Kll(kll)), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + + let result = CountMinSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err(), "wrong-sketch envelope should error"); + } + + #[test] + fn test_from_sketchlib_proto_bytes_float64() { + use asap_sketchlib::proto::sketchlib::{CountMinState, CounterType}; + use prost::Message; + + let state = CountMinState { + rows: 2, + cols: 2, + counter_type: CounterType::Float64 as i32, + counts_int: Vec::new(), + counts_float: vec![1.5, 2.5, 3.5, 4.5], + sum_counts: Vec::new(), + sum2_counts: Vec::new(), + l1: Vec::new(), + l2: Vec::new(), + }; + let bytes = state.encode_to_vec(); + + let acc = CountMinSketchAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + let matrix = acc.inner.sketch(); + assert_eq!(matrix[0], vec![1.5, 2.5]); + assert_eq!(matrix[1], vec![3.5, 4.5]); + } + + #[test] + fn test_from_sketchlib_proto_bytes_dimension_mismatch() { + // counts_int has 5 entries but rows*cols = 6 → expect error + use asap_sketchlib::proto::sketchlib::{CountMinState, CounterType}; + use prost::Message; + + let state = CountMinState { + rows: 2, + cols: 3, + counter_type: CounterType::Int64 as i32, + counts_int: vec![1, 2, 3, 4, 5], + counts_float: Vec::new(), + sum_counts: Vec::new(), + sum2_counts: Vec::new(), + l1: Vec::new(), + l2: Vec::new(), + }; + let bytes = state.encode_to_vec(); + + let result = CountMinSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err()); + assert!( + result.unwrap_err().to_string().contains("counts_int"), + "error should mention counts_int dim mismatch" + ); + } + + #[test] + fn test_from_sketchlib_proto_bytes_zero_dims_rejected() { + use asap_sketchlib::proto::sketchlib::CountMinState; + use prost::Message; + + let state = CountMinState::default(); + let bytes = state.encode_to_vec(); + + let result = CountMinSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err()); + assert!(result.unwrap_err().to_string().contains("degenerate dims")); + } + + #[test] + fn test_apply_proto_delta_bytes_round_trip() { + use asap_sketchlib::proto::sketchlib::CountMinDelta as PbDelta; + use prost::Message; + + let mut acc = CountMinSketchAccumulator { + inner: CountMinSketch::from_legacy_matrix( + vec![vec![1.0, 2.0, 3.0], vec![4.0, 5.0, 6.0]], + 2, + 3, + ), + sample_p: 1.0, + }; + let bytes = PbDelta { + rows: 2, + cols: 3, + cell_rows: vec![0, 1], + cell_cols: vec![0, 2], + d_counts: vec![10, 100], + l1: vec![], + l2: vec![], + ..Default::default() + } + .encode_to_vec(); + + acc.apply_proto_delta_bytes(&bytes).expect("apply ok"); + assert_eq!( + acc.inner.sketch(), + vec![vec![11.0, 2.0, 3.0], vec![4.0, 5.0, 106.0]] + ); + } + + #[test] + fn test_apply_proto_delta_bytes_rejects_garbage() { + let mut acc = CountMinSketchAccumulator::new(2, 3); + assert!(acc.apply_proto_delta_bytes(b"not valid proto").is_err()); + } + + // ---------------------------------------------------------------- + // Statistic::Rate / Statistic::Increase — PR #111 honest-gap closure. + // CMS records insert counts but not timestamps. The Rate readout + // requires the engine to push `range_ms` via query_kwargs; without + // it the accumulator falls back to the raw event count (units of + // events/window) so the answer is at least non-empty. + // ---------------------------------------------------------------- + + #[test] + fn test_query_statistic_rate_with_range_ms() { + // Build a CMS whose min-row-sum is 100 events. With a 5-minute + // (300_000 ms) range, the per-second rate is 100 / 300 ≈ 0.333. + let cms = CountMinSketchAccumulator { + inner: CountMinSketch::from_legacy_matrix( + vec![vec![100.0, 0.0], vec![100.0, 0.0]], + 2, + 2, + ), + sample_p: 1.0, + }; + let mut kwargs = HashMap::new(); + kwargs.insert("range_ms".to_string(), "300000".to_string()); + let trait_obj: &dyn AggregateCore = &cms; + let v = trait_obj + .query_statistic(Statistic::Rate, &None, &kwargs) + .expect("Rate with range_ms is supported"); + assert!( + (v - (100.0 / 300.0)).abs() < 1e-9, + "expected 100/300 = {}, got {v}", + 100.0 / 300.0, + ); + } + + #[test] + fn test_query_statistic_rate_without_range_ms_falls_back_to_count() { + // Without `range_ms` in kwargs the accumulator returns the raw + // event volume (events/window units). Caller is responsible for + // surfacing that caveat to the user; this avoids `status=error` + // for instant rate-shape queries that bypass the matrix-selector + // code path. + let cms = CountMinSketchAccumulator { + inner: CountMinSketch::from_legacy_matrix(vec![vec![42.0, 0.0], vec![42.0, 0.0]], 2, 2), + sample_p: 1.0, + }; + let trait_obj: &dyn AggregateCore = &cms; + let v = trait_obj + .query_statistic(Statistic::Rate, &None, &HashMap::new()) + .expect("Rate without range_ms still answers (fallback)"); + assert_eq!(v, 42.0); + } + + #[test] + fn test_query_statistic_increase_returns_total_count() { + // Increase semantics on CMS: total events in the window — the + // same min-row-sum as Sum / Count. Differs from Rate only in + // that it never divides by range. + let cms = CountMinSketchAccumulator { + inner: CountMinSketch::from_legacy_matrix(vec![vec![5.0, 7.0], vec![3.0, 9.0]], 2, 2), + sample_p: 1.0, + }; + let trait_obj: &dyn AggregateCore = &cms; + let v = trait_obj + .query_statistic(Statistic::Increase, &None, &HashMap::new()) + .expect("Increase is supported"); + // min-row-sum: row0 = 12, row1 = 12, min = 12. + assert_eq!(v, 12.0); + } + + // ---------------------------------------------------------------- + // Defensive inbound-dimension validation (harden/sketch-dim-validation). + // Malformed / degenerate / narrow-hash-budget-violating CMS dims must + // be rejected gracefully (Err, never a panic); valid configs the + // backend actually uses (5x2048, 5x4096, 5x2000) must still decode. + // ---------------------------------------------------------------- + + /// Build a bare `CountMinState` proto carrying the given dims and a + /// row-major INT64 counts vector sized to `rows*cols` so that, IF the + /// dims pass validation, the reshape also succeeds. Used to prove a + /// malformed-dim payload is rejected at the dim gate, not later. + fn cms_state_bytes(rows: u32, cols: u32) -> Vec { + use asap_sketchlib::proto::sketchlib::{CountMinState, CounterType}; + use prost::Message; + let n = (rows as usize).saturating_mul(cols as usize); + let state = CountMinState { + rows, + cols, + counter_type: CounterType::Int64 as i32, + counts_int: vec![0i64; n], + counts_float: Vec::new(), + sum_counts: Vec::new(), + sum2_counts: Vec::new(), + l1: Vec::new(), + l2: Vec::new(), + }; + state.encode_to_vec() + } + + #[test] + fn test_validate_sketch_dims_accepts_valid_configs() { + // The realistic configs the backend uses must pass unchanged. + for (r, c) in [(5usize, 2048usize), (5, 4096), (5, 2000), (4, 1000), (2, 3)] { + assert!( + validate_sketch_dims("CountMinState", r, c).is_ok(), + "valid config {r}x{c} was wrongly rejected" + ); + } + } + + #[test] + fn test_validate_sketch_dims_rejects_malformed() { + // Zero dims. + assert!(validate_sketch_dims("CountMinState", 0, 2048).is_err()); + assert!(validate_sketch_dims("CountMinState", 5, 0).is_err()); + // Narrow-hash-budget violation: 5 * ceil(log2(8192))=5*13=65 > 64. + let err = validate_sketch_dims("CountMinState", 5, 8192).unwrap_err(); + assert!(err.contains("budget"), "expected budget error, got: {err}"); + // Absurdly oversized: 1 x 16,777,216 = 16M cells > 8M cap. (1 row + // keeps the hash budget tiny — 1*24=24 — so the cap check, not the + // budget check, is what fires here.) + let err = validate_sketch_dims("CountMinState", 1, 16_777_216).unwrap_err(); + assert!(err.contains("cap"), "expected cell-cap error, got: {err}"); + // No panic on extreme dims (saturating_mul guards the products). + assert!(validate_sketch_dims("CountMinState", usize::MAX, usize::MAX).is_err()); + } + + #[test] + fn test_from_sketchlib_proto_bytes_rejects_bad_dims_no_panic() { + // A data point declaring narrow-hash-budget-violating dims must be + // skipped (Err returned, NOT a panic). The ingest caller turns + // this Err into a dropped data point + WARN log. + let bytes = cms_state_bytes(5, 8192); + let result = CountMinSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err(), "budget-violating dims should be rejected"); + assert!(result.unwrap_err().to_string().contains("rejecting")); + + // A valid neighbour (5x4096) on the same path still decodes fine. + let ok_bytes = cms_state_bytes(5, 4096); + let acc = CountMinSketchAccumulator::from_sketchlib_proto_bytes(&ok_bytes) + .expect("valid 5x4096 CMS should still decode"); + assert_eq!(acc.inner.rows(), 5); + assert_eq!(acc.inner.cols(), 4096); + } + + #[test] + fn test_query_statistic_rate_rejects_invalid_range_ms() { + let cms = CountMinSketchAccumulator::new(2, 2); + let mut kwargs = HashMap::new(); + kwargs.insert("range_ms".to_string(), "0".to_string()); + let trait_obj: &dyn AggregateCore = &cms; + let err = trait_obj + .query_statistic(Statistic::Rate, &None, &kwargs) + .expect_err("range_ms=0 should error"); + assert!(err.to_string().contains("positive")); + + let mut kwargs = HashMap::new(); + kwargs.insert("range_ms".to_string(), "not-a-number".to_string()); + let err = trait_obj + .query_statistic(Statistic::Rate, &None, &kwargs) + .expect_err("non-numeric range_ms should error"); + assert!(err.to_string().contains("bad range_ms")); + } + + // ---------------------------------------------------------------- + // sample_p rescale. The edge admits each insert with probability `p`, + // so every stored cell is ~p× the true count. CMS is L1/additive and + // linear, so BOTH the point-frequency (query_key) and the aggregate + // total-event statistics (Count/Sum/Increase/Rate) rescale by 1/p. + // ---------------------------------------------------------------- + + #[test] + fn test_query_key_rescaled_by_sample_p() { + // Same stored cell counts, two sample_p values: the p=0.25 sketch + // must report 4× the point-frequency of the unsampled one. + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + let mut unsampled = CountMinSketchAccumulator::new(4, 1000); + unsampled._update(&key, 10.0); + let mut sampled = CountMinSketchAccumulator::new(4, 1000); + sampled._update(&key, 10.0); + sampled.sample_p = 0.25; + + let raw = unsampled.query_key(&key); + let rescaled = sampled.query_key(&key); + assert!( + raw >= 10.0, + "raw estimate should be >= inserted 10, got {raw}" + ); + assert!( + (rescaled - raw * 4.0).abs() < 1e-9, + "expected point-frequency rescaled ≈ 4×raw ({}), got {rescaled}", + raw * 4.0 + ); + } + + #[test] + fn test_aggregate_statistics_rescaled_by_sample_p() { + use crate::Statistic; + // Build a CMS with a known min-row-sum of 12 events, sampled at + // p=0.25 → every aggregate statistic should report 12 / 0.25 = 48. + let cms = CountMinSketchAccumulator { + inner: CountMinSketch::from_legacy_matrix(vec![vec![5.0, 7.0], vec![3.0, 9.0]], 2, 2), + sample_p: 0.25, + }; + let trait_obj: &dyn AggregateCore = &cms; + for stat in [Statistic::Count, Statistic::Sum, Statistic::Increase] { + let v = trait_obj + .query_statistic(stat, &None, &HashMap::new()) + .unwrap_or_else(|e| panic!("{stat:?} should be supported: {e}")); + // min-row-sum = 12, rescaled by 1/0.25 = 48. + assert!( + (v - 48.0).abs() < 1e-9, + "{stat:?}: expected rescaled 48, got {v}" + ); + } + // Rate also divides through the rescaled total: 48 events over a + // 6-second (6000 ms) range = 8 events/s. + let mut kwargs = HashMap::new(); + kwargs.insert("range_ms".to_string(), "6000".to_string()); + let r = trait_obj + .query_statistic(Statistic::Rate, &None, &kwargs) + .expect("rate ok"); + assert!((r - 8.0).abs() < 1e-9, "expected rate 8.0, got {r}"); + } + + #[test] + fn test_sample_p_unset_behaves_as_one() { + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, CountMinState, CounterType, SketchEnvelope, + }; + use prost::Message; + // An envelope with no sample_p (proto3 default 0.0) must normalize + // to 1.0 (no rescale) — byte-compatible with legacy frames. + let state = CountMinState { + rows: 2, + cols: 2, + counter_type: CounterType::Int64 as i32, + counts_int: vec![1, 2, 3, 4], + counts_float: Vec::new(), + sum_counts: Vec::new(), + sum2_counts: Vec::new(), + l1: Vec::new(), + l2: Vec::new(), + }; + let env = SketchEnvelope { + // sample_p left at proto3 default 0.0. + sketch_state: Some(sketch_envelope::SketchState::CountMin(state)), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + let acc = CountMinSketchAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + assert_eq!(acc.sample_p, 1.0, "unset sample_p must normalize to 1.0"); + } + + #[test] + fn test_from_sketchlib_proto_bytes_reads_envelope_sample_p() { + use crate::Statistic; + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, CountMinState, CounterType, SketchEnvelope, + }; + use prost::Message; + // min-row-sum = 12 raw; sample_p 0.25 → Count = 48. + let state = CountMinState { + rows: 2, + cols: 2, + counter_type: CounterType::Float64 as i32, + counts_int: Vec::new(), + counts_float: vec![5.0, 7.0, 3.0, 9.0], + sum_counts: Vec::new(), + sum2_counts: Vec::new(), + l1: Vec::new(), + l2: Vec::new(), + }; + let env = SketchEnvelope { + sample_p: 0.25, + sketch_state: Some(sketch_envelope::SketchState::CountMin(state)), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + let acc = CountMinSketchAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + assert_eq!(acc.sample_p, 0.25); + let trait_obj: &dyn AggregateCore = &acc; + let v = trait_obj + .query_statistic(Statistic::Count, &None, &HashMap::new()) + .expect("count ok"); + assert!((v - 48.0).abs() < 1e-9, "expected rescaled 48, got {v}"); + } + + #[test] + fn test_reset_to_empty_preserves_sample_p() { + let mut acc = CountMinSketchAccumulator::new(2, 3); + acc.sample_p = 0.25; + acc.reset_to_empty(); + assert_eq!(acc.sample_p, 0.25, "window rotation must keep sample_p"); + } + + #[test] + fn test_merge_prefers_sampled_factor() { + let mut a = CountMinSketchAccumulator::new(2, 3); + a.sample_p = 0.25; + let b = CountMinSketchAccumulator::new(2, 3); // sample_p 1.0 + let merged = a.merge_with(&b).expect("merge ok"); + let merged = merged + .as_any() + .downcast_ref::() + .expect("downcast ok"); + assert_eq!(merged.sample_p, 0.25); + + // merge_multiple mirrors the same policy. + let mut c = CountMinSketchAccumulator::new(2, 3); + c.sample_p = 0.25; + let d = CountMinSketchAccumulator::new(2, 3); + let boxed: Vec> = vec![Box::new(d), Box::new(c)]; + let merged = CountMinSketchAccumulator::merge_multiple(&boxed).expect("merge ok"); + assert_eq!(merged.sample_p, 0.25); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/count_min_sketch_with_heap_accumulator.rs b/crates/asap-physical-operators/src/accumulators/count_min_sketch_with_heap_accumulator.rs new file mode 100644 index 00000000..f5d1369d --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/count_min_sketch_with_heap_accumulator.rs @@ -0,0 +1,832 @@ +use crate::{ + AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, + MultipleSubpopulationAggregate, SerializableToSink, +}; +use asap_sketchlib::{CmsHeapItem, CountMinSketchWithHeap, MessagePackCodec}; +use serde::Deserialize; +use serde_json::Value; +use std::collections::HashMap; + +use crate::Statistic; + +/// Local serde view of the DELTA-HEAP wire frame produced by sketchlib-go's +/// `CountSketch.SerializeMsgpackWithHeapDelta` (encoding `MSGPACK_DELTA`). +/// Decoded with `rmp_serde` directly in the backend so NO delta API needs to +/// be added to the public `asap_sketchlib`. +/// +/// rmp_serde compact layout — a 4-element positional array: +/// +/// [ +/// is_delta: bool (always true), +/// matrix_delta: ( rows:u32, cols:u32, cells: Vec<(u32,u32,i64)> ), +/// topk_heap: Vec<(String, f64)>, // FULL heap, [key, value] pairs +/// heap_size: u64, +/// ] +/// +/// Tuple structs deserialize from msgpack fixed arrays positionally, so this +/// matches the Go encoder's byte layout exactly (no field names on the wire). +#[derive(Debug, Deserialize)] +struct HeapDeltaWire { + is_delta: bool, + matrix_delta: MatrixDeltaWire, + topk_heap: Vec<(String, f64)>, + #[allow(dead_code)] + heap_size: u64, +} + +#[derive(Debug, Deserialize)] +struct MatrixDeltaWire { + rows: u32, + cols: u32, + cells: Vec<(u32, u32, i64)>, +} + +/// Validated/flattened view of a decoded DELTA-HEAP frame. +struct HeapDeltaFrame { + rows: u32, + cols: u32, + heap_size: u64, + cells: Vec<(u32, u32, i64)>, + heap: Vec<(String, f64)>, +} + +impl HeapDeltaFrame { + fn from_msgpack(buffer: &[u8]) -> Result> { + let wire: HeapDeltaWire = rmp_serde::from_slice(buffer) + .map_err(|e| format!("decode CountSketchWithHeap delta msgpack: {e}"))?; + if !wire.is_delta { + return Err("CountSketchWithHeap delta frame has is_delta=false".into()); + } + Ok(Self { + rows: wire.matrix_delta.rows, + cols: wire.matrix_delta.cols, + heap_size: wire.heap_size, + cells: wire.matrix_delta.cells, + heap: wire.topk_heap, + }) + } +} + +/// Count-Min Sketch with Heap accumulator — wraps `asap_sketchlib::CountMinSketchWithHeap`. +/// Core struct, update/merge/serde logic live in `asap_sketchlib::message_pack_format::portable::countminsketch_topk`. +/// This file retains QE-specific trait impls, legacy deserializers, and JSON output. +#[derive(Debug, Clone)] +pub struct CountMinSketchWithHeapAccumulator { + pub inner: CountMinSketchWithHeap, +} + +// Re-export HeapItem so existing code using CountMinSketchWithHeapAccumulator::HeapItem still works. +pub use asap_sketchlib::CmsHeapItem as HeapItemReexport; + +impl CountMinSketchWithHeapAccumulator { + pub fn new(row_num: usize, col_num: usize, heap_size: usize) -> Self { + Self { + inner: CountMinSketchWithHeap::new(row_num, col_num, heap_size), + } + } + + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + let key_string = key.labels.join(";"); + self.inner.estimate(&key_string) + } + + /// Decode a heap-bearing CountSketch FULL msgpack frame + /// (`{sketch:[matrix,rows,cols], topk_heap, heap_size}`) into a heap + /// accumulator. This is the window-1 / full-frame base for the + /// DELTA-HEAP delta path: the backend caches THIS accumulator as the + /// per-series base so a later `MSGPACK_DELTA` frame applies its sparse + /// matrix delta onto a heap accumulator (not a plain CountSketch). + /// + /// Delegates to the PUBLIC `asap_sketchlib::CountMinSketchWithHeap:: + /// from_msgpack` (both heap-bearing frequency variants share the wire + /// shape; the CountSketch-with-heap promotion is decided by the ingest + /// router, not the bytes). + pub fn from_msgpack_with_heap_bytes(buffer: &[u8]) -> Result> { + Ok(Self { + inner: CountMinSketchWithHeap::from_msgpack(buffer) + .map_err(|e| format!("deserialize CountMinSketchWithHeap msgpack: {e}"))?, + }) + } + + /// Apply a DELTA-HEAP msgpack frame (encoding `MSGPACK_DELTA`) onto this + /// accumulator IN PLACE, WITHOUT any change to the public + /// `asap_sketchlib`: the frame is decoded generically with `rmp_serde` + /// into local serde structs, the sparse signed cell deltas are added to + /// the stored matrix (read back via the public `sketch_matrix()`), and + /// the top-k heap is REPLACED with the frame's full heap. The rebuilt + /// inner is produced via the public `from_legacy_matrix`, which rounds + /// cells to the i64 storage and re-seeds the heap. + /// + /// Under the per-window-reset model (`docs/delta-baseline-contract.md` + /// §3) the ingest caller resets this accumulator to empty at a window + /// boundary before applying, so the delta — which is the window's own + /// matrix against an empty base — reconstructs the window's state. + pub fn apply_msgpack_heap_delta_bytes( + &mut self, + buffer: &[u8], + ) -> Result<(), Box> { + let frame = HeapDeltaFrame::from_msgpack(buffer)?; + + let rows = self.inner.rows(); + let cols = self.inner.cols(); + let heap_size = self.inner.heap_size; + + // Read the current (post-reset, possibly empty) matrix and apply the + // sparse signed deltas additively. Cells outside the stored + // dimensions are skipped defensively (mirrors the plain-CountSketch + // delta apply). + let mut matrix = self.inner.sketch_matrix(); + for (r, c, dc) in &frame.cells { + let (r, c) = (*r as usize, *c as usize); + if r >= rows || c >= cols { + continue; + } + matrix[r][c] += *dc as f64; + } + + // Replace the heap with the frame's full heap. `from_legacy_matrix` + // re-seeds both the matrix and the heap from these inputs. + let heap: Vec = frame + .heap + .into_iter() + .map(|(key, value)| CmsHeapItem { key, value }) + .collect(); + + self.inner = + CountMinSketchWithHeap::from_legacy_matrix(matrix, heap, rows, cols, heap_size); + Ok(()) + } + + /// Reconstruct a heap accumulator STANDALONE from a single DELTA-HEAP + /// msgpack frame (encoding `MSGPACK_DELTA`), with NO cached per-series + /// base. Used by the read-side reducer's `FrequencyTopk` path, where — + /// unlike the ingest accumulator — there is no rolling base to apply + /// onto: under the per-window-reset contract + /// (`docs/delta-baseline-contract.md` §3) each window's delta encodes + /// that window's own state against an EMPTY base, so reconstruction is + /// "empty(dims) + apply(delta)". + /// + /// Reuses the exact ingest-side apply logic: read the (rows, cols, + /// heap_size) the frame declares, build an empty accumulator of those + /// dims (equivalent to `reset_to_empty` on a same-shape base), then + /// fold the frame in via `apply_msgpack_heap_delta_bytes`. No + /// `asap_sketchlib` change — the frame is decoded generically with + /// `rmp_serde`. + pub fn from_msgpack_heap_delta_bytes( + buffer: &[u8], + ) -> Result> { + let frame = HeapDeltaFrame::from_msgpack(buffer)?; + if frame.rows == 0 || frame.cols == 0 { + return Err(format!( + "CountSketchWithHeap delta frame has zero dims (rows={}, cols={})", + frame.rows, frame.cols + ) + .into()); + } + let mut acc = Self::new( + frame.rows as usize, + frame.cols as usize, + frame.heap_size as usize, + ); + acc.apply_msgpack_heap_delta_bytes(buffer)?; + Ok(acc) + } + + /// This function seems will never be used anymore. Keep it for possible future use. + pub fn deserialize_from_json(data: &Value) -> Result> { + let row_num = data["row_num"] + .as_f64() + .ok_or("Missing or invalid 'row_num' field")? as usize; + let col_num = data["col_num"] + .as_f64() + .ok_or("Missing or invalid 'col_num' field")? as usize; + let heap_size = data["heap_size"] + .as_f64() + .ok_or("Missing or invalid 'heap_size' field")? as usize; + + let sketch_data = data["sketch"] + .as_array() + .ok_or("Missing or invalid 'sketch' field")?; + + let mut sketch = Vec::new(); + for row in sketch_data { + let row_array = row.as_array().ok_or("Invalid row in sketch data")?; + let mut sketch_row = Vec::new(); + for cell in row_array { + let value = cell.as_f64().ok_or("Invalid cell value in sketch data")?; + sketch_row.push(value); + } + sketch.push(sketch_row); + } + + let topk_heap_data = data["topk_heap"] + .as_array() + .ok_or("Missing or invalid 'topk_heap' field")?; + + let mut topk_heap = Vec::new(); + for item in topk_heap_data { + let key = item["key"] + .as_str() + .ok_or("Missing or invalid 'key' in heap item")? + .to_string(); + let value = item["value"] + .as_f64() + .ok_or("Missing or invalid 'value' in heap item")?; + topk_heap.push(CmsHeapItem { key, value }); + } + + Ok(Self { + inner: CountMinSketchWithHeap::from_legacy_matrix( + sketch, topk_heap, row_num, col_num, heap_size, + ), + }) + } + + pub fn deserialize_from_bytes(_buffer: &[u8]) -> Result> { + Err("deserialize_from_bytes for CountMinSketchWithHeapAccumulator not implemented".into()) + } + + /// VALUE-WEIGHTED heavy-hitter update (FIX: CountSketch/CMS topk + /// recall-0). The default ingest path inserts `+1` per occurrence keyed + /// by the raw `item`, so the heap ranks groups by OCCURRENCE COUNT — the + /// wrong answer for `topk(k, sum by (label) (metric))`, which asks for + /// the top groups by SUM OF VALUE. This update adds the sample `value` + /// (not `+1`) into both the CMS matrix and the top-k heap, keyed by the + /// GROUP LABEL (e.g. the `host` / `zone` value), so the heap's ranking is + /// by summed value. Repeated calls for the same `group_label` accumulate, + /// so after folding a window the heap holds Σvalue per group. + /// + /// Delegates to the library's value-weighted `CountMinSketchWithHeap:: + /// update(key, value)` (`sketchlib_cms_heap_update` → `insert_many(key, + /// round(value))`), which is the "separate update path" the evaluation + /// plan (Fig 3c) called for. + pub fn insert_value(&mut self, group_label: &str, value: f64) { + self.inner.update(group_label, value); + } + + /// Read the top-`k` GROUPS ranked by summed VALUE (descending), keyed by + /// the group label. Pairs with [`Self::insert_value`]: the heap built by + /// value-weighted updates ranks by Σvalue, so this returns the + /// value-weighted top-k (not the occurrence-count top-k the raw `item` + /// heap would give). Sorted descending by value; ties broken by key for + /// determinism; truncated to `k`. + pub fn topk_by_value(&self, k: usize) -> Vec<(String, f64)> { + let mut items: Vec<(String, f64)> = self + .inner + .topk_heap_items() + .into_iter() + .map(|it| (it.key, it.value)) + .collect(); + items.sort_by(|a, b| { + b.1.partial_cmp(&a.1) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.0.cmp(&b.0)) + }); + items.truncate(k); + items + } + + /// Get all keys from the top-k heap. + pub fn get_topk_keys(&self) -> Vec { + self.inner + .topk_heap_items() + .iter() + .map(|item| { + let labels: Vec = item.key.split(';').map(|s| s.to_string()).collect(); + KeyByLabelValues { labels } + }) + .collect() + } +} + +impl SerializableToSink for CountMinSketchWithHeapAccumulator { + fn serialize_to_json(&self) -> Value { + let heap_items: Vec = self + .inner + .topk_heap_items() + .iter() + .map(|item| { + serde_json::json!({ + "key": item.key, + "value": item.value + }) + }) + .collect(); + + serde_json::json!({ + "row_num": self.inner.rows(), + "col_num": self.inner.cols(), + "heap_size": self.inner.heap_size, + "sketch": self.inner.sketch_matrix(), + "topk_heap": heap_items + }) + } + + fn serialize_to_bytes(&self) -> Vec { + self.inner.to_msgpack().unwrap_or_default() + } +} + +impl AggregateCore for CountMinSketchWithHeapAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "CountMinSketchWithHeapAccumulator" + } + + /// Per-window base rotation (`docs/delta-baseline-contract.md` §3): + /// rebuild an empty heap accumulator with the same (rows, cols, + /// heap_size) so the next window's DELTA-HEAP frame applies onto a clean, + /// same-shape base. Without this override the trait default is a no-op, + /// which would let the additive matrix delta accumulate across windows + /// (over-counting). Mirrors `CountSketchAccumulator::reset_to_empty`. + fn reset_to_empty(&mut self) { + self.inner = + CountMinSketchWithHeap::new(self.inner.rows(), self.inner.cols(), self.inner.heap_size); + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge CountMinSketchWithHeapAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + + let other_cms = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountMinSketchWithHeapAccumulator")?; + + let merged = Self::merge_accumulators(vec![self.clone(), other_cms.clone()])?; + Ok(Box::new(merged)) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::CountMinSketchWithHeap + } + + fn get_keys(&self) -> Option> { + Some(self.get_topk_keys()) + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + key: &Option, + query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::MultipleSubpopulationAggregate; + let key_val = key + .as_ref() + .ok_or("Key required for CountMinSketchWithHeapAccumulator")?; + self.query(statistic, key_val, Some(query_kwargs)) + } +} + +impl MultipleSubpopulationAggregate for CountMinSketchWithHeapAccumulator { + fn query( + &self, + _statistic: Statistic, + key: &KeyByLabelValues, + _query_kwargs: Option<&HashMap>, + ) -> Result> { + Ok(self.query_key(key)) + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +impl MergeableAccumulator for CountMinSketchWithHeapAccumulator { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + let mut iter = accumulators.into_iter(); + let mut merged = iter.next().unwrap(); + for acc in iter { + merged.inner.merge(&acc.inner)?; + } + Ok(merged) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_count_min_sketch_with_heap_creation() { + let cms = CountMinSketchWithHeapAccumulator::new(4, 1000, 20); + assert_eq!(cms.inner.rows(), 4); + assert_eq!(cms.inner.cols(), 1000); + assert_eq!(cms.inner.heap_size, 20); + assert_eq!(cms.inner.topk_heap_items().len(), 0); + } + + #[test] + fn test_count_min_sketch_with_heap_query() { + let cms = CountMinSketchWithHeapAccumulator::new(2, 10, 5); + let key = KeyByLabelValues::new(); + assert_eq!(cms.query_key(&key), 0.0); + + let multi_trait: &dyn MultipleSubpopulationAggregate = &cms; + assert_eq!(multi_trait.query(Statistic::Sum, &key, None).unwrap(), 0.0); + } + + #[test] + fn test_count_min_sketch_with_heap_merge() { + // Build controlled state via from_legacy_matrix (works regardless of backend config). + let sketch1 = vec![ + vec![10.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0], + vec![0.0, 20.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0], + ]; + let heap1 = vec![ + CmsHeapItem { + key: "key1".to_string(), + value: 100.0, + }, + CmsHeapItem { + key: "key2".to_string(), + value: 50.0, + }, + ]; + let sketch2 = vec![ + vec![5.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0], + vec![0.0, 15.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0], + ]; + let heap2 = vec![ + CmsHeapItem { + key: "key3".to_string(), + value: 75.0, + }, + CmsHeapItem { + key: "key1".to_string(), + value: 80.0, + }, + ]; + + let cms1 = CountMinSketchWithHeapAccumulator { + inner: CountMinSketchWithHeap::from_legacy_matrix(sketch1, heap1, 2, 10, 5), + }; + let cms2 = CountMinSketchWithHeapAccumulator { + inner: CountMinSketchWithHeap::from_legacy_matrix(sketch2, heap2, 2, 10, 3), + }; + + let result = CountMinSketchWithHeapAccumulator::merge_accumulators(vec![cms1, cms2]); + assert!(result.is_ok()); + let merged = result.unwrap(); + assert_eq!(merged.inner.sketch_matrix()[0][0], 15.0); + assert_eq!(merged.inner.sketch_matrix()[1][1], 35.0); + assert_eq!(merged.inner.heap_size, 3); + assert!(merged.inner.topk_heap_items().len() <= 3); + } + + #[test] + fn test_count_min_sketch_with_heap_merge_single() { + let cms = CountMinSketchWithHeapAccumulator::new(2, 3, 5); + let result = CountMinSketchWithHeapAccumulator::merge_accumulators(vec![cms.clone()]); + assert!(result.is_ok()); + let merged = result.unwrap(); + assert_eq!(merged.inner.rows(), cms.inner.rows()); + assert_eq!(merged.inner.cols(), cms.inner.cols()); + assert_eq!(merged.inner.heap_size, cms.inner.heap_size); + } + + #[test] + fn test_count_min_sketch_with_heap_merge_dimension_mismatch() { + let cms1 = CountMinSketchWithHeapAccumulator::new(2, 10, 5); + let cms2 = CountMinSketchWithHeapAccumulator::new(3, 10, 5); + let result = CountMinSketchWithHeapAccumulator::merge_accumulators(vec![cms1, cms2]); + assert!(result.is_err()); + assert!(result.unwrap_err().to_string().contains("dimension")); + } + + #[test] + fn test_count_min_sketch_with_heap_as_aggregate_core() { + let cms = CountMinSketchWithHeapAccumulator::new(2, 3, 5); + assert_eq!(cms.type_name(), "CountMinSketchWithHeapAccumulator"); + } + + #[test] + fn test_get_topk_keys() { + let mut cms = CountMinSketchWithHeapAccumulator::new(2, 3, 5); + cms.inner.update("label1;label2", 100.0); + cms.inner.update("label3;label4", 50.0); + + let keys = cms.get_topk_keys(); + assert_eq!(keys.len(), 2); + // Top-k order can differ between Legacy and Sketchlib backends (heap ordering / estimates). + let label_sets: std::collections::HashSet<_> = + keys.iter().map(|k| k.labels.clone()).collect(); + assert!(label_sets.contains(&vec!["label1".to_string(), "label2".to_string()])); + assert!(label_sets.contains(&vec!["label3".to_string(), "label4".to_string()])); + } + + #[test] + fn test_multiple_subpopulation_aggregate() { + let cms = CountMinSketchWithHeapAccumulator::new(3, 50, 10); + let key = KeyByLabelValues::new(); + + let multi_trait: &dyn MultipleSubpopulationAggregate = &cms; + let result = multi_trait.query(Statistic::Sum, &key, None).unwrap(); + assert_eq!(result, 0.0); + + let keys = multi_trait.get_keys(); + assert!(keys.is_some()); + assert_eq!(keys.unwrap().len(), 0); + } + + // ---------------------------------------------------------------- + // DELTA-HEAP wire form (encoding MSGPACK_DELTA): apply a sparse matrix + // delta + replace the heap, decoded generically (rmp_serde) WITHOUT any + // asap_sketchlib delta API. The first test feeds a frame produced by the + // Go encoder (sketchlib-go `MarshalCountSketchWithHeapDelta`) to prove + // cross-language byte parity — mirrors how the full-heap parity is + // proven. The second proves PWR full -> delta -> delta reconstruction. + // ---------------------------------------------------------------- + + /// Cross-language byte-parity: this hex is the exact output of + /// sketchlib-go's `asapmsgpack.MarshalCountSketchWithHeapDelta(5, 1024, + /// cells=[(0,1,50),(1,3,-4),(4,1023,1_000_000)], + /// heap=[("/checkout",50),("/cart",20)], heap_size=20)` (captured via a + /// throw-away Go print test, identical methodology to the full-heap + /// golden in `sketchlib-go/.../count_sketch_with_heap_test.go`). If the + /// Go encoder or the rmp_serde layout ever shifts, this decode fails + /// loudly. + const GO_DELTA_HEAP_GOLDEN_HEX: &str = "94c39305cd04009393000132930103fc9304cd03ffce000f42409292a92f636865636b6f7574cb404900000000000092a52f63617274cb403400000000000014"; + + #[test] + fn test_apply_go_produced_delta_heap_frame_matrix_and_heap() { + let bytes = hex::decode(GO_DELTA_HEAP_GOLDEN_HEX).expect("hex"); + + // Base = empty heap accumulator with the frame's dims (what the + // ingest caller holds after the per-window base rotation). + let mut acc = CountMinSketchWithHeapAccumulator::new(5, 1024, 20); + acc.apply_msgpack_heap_delta_bytes(&bytes) + .expect("apply Go delta-heap frame"); + + // Matrix: the three sparse cells landed onto the empty base. + let m = acc.inner.sketch_matrix(); + assert_eq!(m.len(), 5); + assert_eq!(m[0].len(), 1024); + assert_eq!(m[0][1], 50.0, "cell (0,1)"); + assert_eq!(m[1][3], -4.0, "cell (1,3)"); + assert_eq!(m[4][1023], 1_000_000.0, "cell (4,1023)"); + // Everything else stays zero. + assert_eq!(m[2][2], 0.0); + assert_eq!(m[0][0], 0.0); + + // Heap: the frame's full heap, with /checkout ranked above /cart. + let mut items = acc.inner.topk_heap_items(); + items.sort_by(|a, b| b.value.partial_cmp(&a.value).unwrap()); + assert_eq!(items.len(), 2); + assert_eq!(items[0].key, "/checkout"); + assert_eq!(items[0].value, 50.0); + assert_eq!(items[1].key, "/cart"); + assert_eq!(items[1].value, 20.0); + } + + #[test] + fn test_pwr_full_then_delta_then_delta_reconstructs_per_window() { + use asap_sketchlib::MessagePackCodec; + + // Window 1 (full frame): build a heap-bearing CountSketch with mass + // and serialize the FULL `{sketch,topk_heap,heap_size}` frame, then + // decode it into a heap accumulator (the cached per-series base). + let w1 = CountMinSketchWithHeap::from_legacy_matrix( + vec![vec![300.0; 4]; 5], + vec![CmsHeapItem { + key: "k".into(), + value: 300.0, + }], + 5, + 4, + 20, + ); + let w1_bytes = w1.to_msgpack().expect("w1 full msgpack"); + let mut base = CountMinSketchWithHeapAccumulator::from_msgpack_with_heap_bytes(&w1_bytes) + .expect("decode w1 full frame as heap accumulator"); + assert_eq!(base.inner.sketch_matrix()[0][0], 300.0); + + // Window 2 delta: this window's own state is matrix cells of value 50 + // against an EMPTY base + heap {k:50}. The DELTA-HEAP frame is encoded + // the same way the Go producer does (4-array, is_delta, sparse cells). + let w2_frame = encode_delta_heap(5, 4, &[(0, 0, 50), (1, 1, 50)], &[("k", 50.0)], 20); + // PWR: rotate base to empty at the window boundary, then apply. + base.reset_to_empty(); + assert_eq!( + base.inner.sketch_matrix()[0][0], + 0.0, + "reset_to_empty cleared matrix" + ); + base.apply_msgpack_heap_delta_bytes(&w2_frame) + .expect("apply w2 delta"); + assert_eq!(base.inner.sketch_matrix()[0][0], 50.0, "window-2 cell"); + assert_eq!(base.inner.sketch_matrix()[1][1], 50.0); + // No cross-window leakage from window 1's 300s. + assert_eq!(base.inner.sketch_matrix()[2][2], 0.0); + let h2: Vec<_> = base.inner.topk_heap_items(); + assert_eq!(h2.len(), 1); + assert_eq!(h2[0].key, "k"); + assert_eq!(h2[0].value, 50.0); + + // Window 3 delta: 80s against empty + heap {k:80}. + let w3_frame = encode_delta_heap(5, 4, &[(0, 0, 80)], &[("k", 80.0)], 20); + base.reset_to_empty(); + base.apply_msgpack_heap_delta_bytes(&w3_frame) + .expect("apply w3 delta"); + assert_eq!(base.inner.sketch_matrix()[0][0], 80.0, "window-3 cell"); + assert_eq!(base.inner.sketch_matrix()[1][1], 0.0, "no window-2 leakage"); + let h3 = base.inner.topk_heap_items(); + assert_eq!(h3.len(), 1); + assert_eq!(h3[0].value, 80.0); + } + + #[test] + fn test_rmp_serde_layout_is_byte_identical_to_go_encoder() { + // The rmp_serde positional encoding of the delta-heap frame must be + // BYTE-IDENTICAL to sketchlib-go's hand-rolled + // `MarshalCountSketchWithHeapDelta`. This hex is the Go encoder's + // output for (5, 4, cells=[(0,0,50),(1,1,50)], heap=[("k",50)], + // heap_size=20) — the same inputs `encode_delta_heap` uses below. + // Equality here proves both encode AND decode are cross-language + // byte-compatible (the decode path is exercised by the Go-golden + // test above). + const GO_PARITY_HEX: &str = "94c39305049293000032930101329192a16bcb404900000000000014"; + let rust_bytes = encode_delta_heap(5, 4, &[(0, 0, 50), (1, 1, 50)], &[("k", 50.0)], 20); + assert_eq!(hex::encode(&rust_bytes), GO_PARITY_HEX); + } + + #[test] + fn test_apply_delta_rejects_full_frame_and_garbage() { + use asap_sketchlib::MessagePackCodec; + let mut acc = CountMinSketchWithHeapAccumulator::new(2, 4, 5); + // A FULL frame (3-array, no is_delta marker) must NOT decode as a + // delta — the routing relies on the two shapes being distinct. + let full = CountMinSketchWithHeap::from_legacy_matrix( + vec![vec![1.0; 4]; 2], + vec![CmsHeapItem { + key: "a".into(), + value: 1.0, + }], + 2, + 4, + 5, + ) + .to_msgpack() + .unwrap(); + assert!(acc.apply_msgpack_heap_delta_bytes(&full).is_err()); + assert!(acc.apply_msgpack_heap_delta_bytes(b"not msgpack").is_err()); + } + + /// Encode a DELTA-HEAP frame the same way sketchlib-go's + /// `MarshalCountSketchWithHeapDelta` does (rmp_serde positional layout), + /// so the test exercises the real decode path. Tuple structs serialize + /// as msgpack fixed arrays — byte-identical to the Go hand-rolled writer. + fn encode_delta_heap( + rows: u32, + cols: u32, + cells: &[(u32, u32, i64)], + heap: &[(&str, f64)], + heap_size: u64, + ) -> Vec { + #[derive(serde::Serialize)] + struct W<'a>( + bool, + (u32, u32, &'a [(u32, u32, i64)]), + Vec<(String, f64)>, + u64, + ); + let heap_owned: Vec<(String, f64)> = + heap.iter().map(|(k, v)| (k.to_string(), *v)).collect(); + let w = W(true, (rows, cols, cells), heap_owned, heap_size); + rmp_serde::to_vec(&w).expect("encode delta-heap") + } + + // ---------------------------------------------------------------- + // FIX 1 — VALUE-WEIGHTED top-k (recall 0 → correct). + // + // `topk(k, sum by (host) (cpu_load))` asks for the top-k hosts by + // SUM OF VALUE. The heavy-hitter heap built by the default `+1`-per- + // occurrence update ranks by COUNT keyed by `item`, so its recall + // against the value-weighted ground truth is 0 when the busiest host + // (most samples) is NOT the heaviest host (largest Σvalue). + // `insert_value(group_label, value)` adds the sample VALUE keyed by the + // GROUP LABEL, so `topk_by_value` ranks by Σvalue — correct recall. + // ---------------------------------------------------------------- + + /// Crafted adversarial dataset: the host with the MOST samples + /// (`h_chatty`, 100 tiny samples) is NOT the host with the largest + /// value-sum (`h_heavy`, a handful of huge samples). A COUNT-ranked + /// heap would surface `h_chatty`; the value-weighted top-k must surface + /// the true heavy hitters by Σvalue, giving recall 1.0 against the + /// ground-truth top-k-by-value-sum. + #[test] + fn value_weighted_topk_has_full_recall_vs_count_topk() { + // (host, per-sample value, sample count) → true Σvalue: + // h_heavy : 1000 × 3 = 3000 (few samples, huge value) + // h_mid : 200 × 5 = 1000 + // h_small : 50 × 6 = 300 + // h_chatty: 1 × 100 = 100 (MOST samples, tiny value) + let data: &[(&str, f64, usize)] = &[ + ("h_heavy", 1000.0, 3), + ("h_mid", 200.0, 5), + ("h_small", 50.0, 6), + ("h_chatty", 1.0, 100), + ]; + + // Wide CMS + heap large enough to hold every group exactly (4 groups) + // so the estimate equals the true Σvalue with no hash collisions. + let mut acc = CountMinSketchWithHeapAccumulator::new(5, 4096, 16); + let mut truth: std::collections::HashMap<&str, f64> = std::collections::HashMap::new(); + for (host, value, count) in data { + for _ in 0..*count { + acc.insert_value(host, *value); + } + *truth.entry(*host).or_insert(0.0) += value * (*count as f64); + } + + // Ground-truth top-2 by value-sum: h_heavy (3000), h_mid (1000). + let mut truth_ranked: Vec<(&str, f64)> = truth.into_iter().collect(); + truth_ranked.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap()); + let truth_top2: std::collections::HashSet<&str> = + truth_ranked.iter().take(2).map(|(k, _)| *k).collect(); + assert!( + truth_top2.contains("h_heavy") && truth_top2.contains("h_mid"), + "ground-truth top-2 by value-sum should be h_heavy + h_mid" + ); + + // Value-weighted top-2 from the heap. + let got = acc.topk_by_value(2); + assert_eq!(got.len(), 2, "k=2 → two groups: {got:?}"); + let got_keys: std::collections::HashSet<&str> = + got.iter().map(|(k, _)| k.as_str()).collect(); + + // RECALL = |got ∩ truth| / |truth| must be 1.0. + let hits = got_keys.intersection(&truth_top2).count(); + let recall = hits as f64 / truth_top2.len() as f64; + assert_eq!( + recall, 1.0, + "value-weighted top-k recall must be 1.0 (count-ranked heap would \ + surface h_chatty and miss h_heavy → recall < 1): got={got:?}" + ); + + // The busiest-by-count host (h_chatty) must NOT be in the top-2, + // proving we rank by value-sum, not occurrence count. + assert!( + !got_keys.contains("h_chatty"), + "h_chatty (most samples, smallest value-sum) must be excluded: {got:?}" + ); + + // Estimates are exact here (no collisions, heap holds all groups): + // top-1 must be h_heavy with Σvalue 3000. + assert_eq!(got[0].0, "h_heavy"); + assert!( + (got[0].1 - 3000.0).abs() < 1e-6, + "h_heavy value-sum estimate ≈ 3000, got {}", + got[0].1 + ); + assert_eq!(got[1].0, "h_mid"); + assert!( + (got[1].1 - 1000.0).abs() < 1e-6, + "h_mid value-sum estimate ≈ 1000, got {}", + got[1].1 + ); + } + + /// A single value-weighted insert must put the full value (not +1) into + /// the heap, and repeated inserts for the same group must accumulate. + #[test] + fn insert_value_accumulates_summed_value_in_heap() { + let mut acc = CountMinSketchWithHeapAccumulator::new(4, 1024, 8); + acc.insert_value("g", 10.0); + acc.insert_value("g", 25.0); + let top = acc.topk_by_value(1); + assert_eq!(top.len(), 1); + assert_eq!(top[0].0, "g"); + assert!( + (top[0].1 - 35.0).abs() < 1e-6, + "summed value should be 35 (10+25), got {}", + top[0].1 + ); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/count_sketch_accumulator.rs b/crates/asap-physical-operators/src/accumulators/count_sketch_accumulator.rs new file mode 100644 index 00000000..fc74a018 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/count_sketch_accumulator.rs @@ -0,0 +1,678 @@ +//! CountSketch accumulator backed by `asap_sketchlib::CountSketch`. +//! +//! Supports worker merge, persistence serialization, and modified-OTLP proto +//! decoding. Per-key queries delegate to sketchlib's median-of-signed-rows +//! estimator so query and ingest use the same hash specification. Top-k +//! requires the separate heap-bearing accumulator. + +use crate::{ + AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, + MultipleSubpopulationAggregate, SerializableToSink, +}; +use asap_sketchlib::{CountSketch, CountSketchDelta, MessagePackCodec}; +use serde_json::Value; +use std::collections::HashMap; + +use crate::Statistic; + +/// Count Sketch accumulator — inner matrix of signed counts. +#[derive(Debug, Clone)] +pub struct CountSketchAccumulator { + pub inner: CountSketch, +} + +impl CountSketchAccumulator { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + inner: CountSketch::new(row_num, col_num), + } + } + + /// Median-of-signed-rows point estimate for `key`, via the real + /// `asap_sketchlib::CountSketch::estimate` — the canonical, hash-spec- + /// compatible estimator (see `AggregateCore::query_statistic`'s doc for + /// why this replaced a hand-rolled, non-compatible hash). + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + self.inner.estimate(&key.to_semicolon_str()) + } + + /// Decode from the modified OTLP wire format's + /// `CountSketchDataPoint.sketch` bytes when + /// `encoding = COUNT_SKETCH_ENCODING_MSGPACK`. The bytes are the + /// MessagePack serialization of the cross-language sketch-core + /// `CountSketch` struct — PR I parity entrypoint. + pub fn from_msgpack_bytes(buffer: &[u8]) -> Result> { + Ok(Self { + inner: CountSketch::from_msgpack(buffer) + .map_err(|e| format!("deserialize CountSketch msgpack: {e}"))?, + }) + } + + /// Decode from the modified OTLP wire format's + /// `CountSketchDataPoint.sketch` bytes — the protobuf-encoded + /// `asap_sketchlib::proto::sketchlib::CountSketchState` message + /// that DataCollector's `countsketchprocessor` emits when + /// `encoding = COUNT_SKETCH_ENCODING_PROTO`. + /// + /// Mirrors `CountMinSketchAccumulator::from_sketchlib_proto_bytes` + /// but on the signed-counter `CountSketchState`. The resulting + /// accumulator is constructed via + /// `CountSketch::from_legacy_matrix` after reshaping the flat + /// `counts_int` / `counts_float` field into a `Vec>`. + pub fn from_sketchlib_proto_bytes(buffer: &[u8]) -> Result> { + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, CountSketchState, CounterType, SketchEnvelope, + }; + use prost::Message; + + // DataCollector's countsketchprocessor wraps the state in a + // `SketchEnvelope{count_sketch: CountSketchState}` via + // sketchlib-go's `SerializePortableFO` + `proto.Marshal`. Try + // decoding as envelope first, fall back to bare + // `CountSketchState` for callers (e.g. unit tests) that + // encode the state directly. Mirrors the PR #14 fix on + // `CountMinSketchAccumulator::from_sketchlib_proto_bytes`. + let state = match SketchEnvelope::decode(buffer) { + Ok(env) => match env.sketch_state { + Some(sketch_envelope::SketchState::CountSketch(st)) => st, + Some(other) => { + return Err(format!( + "SketchEnvelope contains non-CountSketch sketch: {:?}", + std::mem::discriminant(&other) + ) + .into()); + } + None => CountSketchState::decode(buffer) + .map_err(|e| format!("decode CountSketchState: {e}"))?, + }, + Err(_) => CountSketchState::decode(buffer) + .map_err(|e| format!("decode CountSketchState: {e}"))?, + }; + let rows = state.rows as usize; + let cols = state.cols as usize; + // Defensive dim validation BEFORE reconstructing the matrix: + // reject degenerate / narrow-hash-budget-violating / absurdly + // oversized dims so a malformed payload fails gracefully (the + // ingest caller skips the data point) instead of building a + // degenerate or huge matrix. Shares the CMS validator since the + // CountSketch matrix uses the same packed-hash column layout. + crate::accumulators::count_min_sketch_accumulator::validate_sketch_dims( + "CountSketchState", + rows, + cols, + )?; + let expected_len = rows * cols; + let counter_type = CounterType::try_from(state.counter_type).map_err(|_| { + format!( + "CountSketchState has unknown counter_type tag {}", + state.counter_type + ) + })?; + let flat: Vec = match counter_type { + CounterType::Int32 | CounterType::Int64 => { + if state.counts_int.len() != expected_len { + return Err(format!( + "CountSketchState counts_int has {} entries, expected rows*cols = {}", + state.counts_int.len(), + expected_len + ) + .into()); + } + state.counts_int.iter().map(|&v| v as f64).collect() + } + CounterType::Float64 => { + if state.counts_float.len() != expected_len { + return Err(format!( + "CountSketchState counts_float has {} entries, expected rows*cols = {}", + state.counts_float.len(), + expected_len + ) + .into()); + } + state.counts_float.clone() + } + other => { + return Err(format!( + "CountSketchState counter_type {other:?} not yet supported \ + (INT128 stores interleaved hi/lo pairs; will be added when needed)" + ) + .into()); + } + }; + let mut matrix = Vec::with_capacity(rows); + for r in 0..rows { + let start = r * cols; + matrix.push(flat[start..start + cols].to_vec()); + } + Ok(Self { + inner: CountSketch::from_legacy_matrix(matrix, rows, cols), + }) + } + + /// Apply a proto-encoded `CountSketchDelta` frame to this + /// accumulator's inner sketch — the decode path for + /// `COUNT_SKETCH_ENCODING_PROTO_DELTA` (paper §6.2 B3 / B4). + /// + /// Cells apply additively: `matrix[cell_rows[i]][cell_cols[i]] + /// += d_counts[i]`. Per-row L2 is parsed off the wire but + /// ignored at application time — it's a downstream error- + /// accounting signal, not a merge input. + pub fn apply_proto_delta_bytes( + &mut self, + buffer: &[u8], + ) -> Result<(), Box> { + use asap_sketchlib::proto::sketchlib::CountSketchDelta as PbDelta; + use prost::Message; + + let pb = PbDelta::decode(buffer).map_err(|e| format!("decode CountSketchDelta: {e}"))?; + + if pb.cell_rows.len() != pb.cell_cols.len() || pb.cell_rows.len() != pb.d_counts.len() { + return Err(format!( + "CountSketchDelta packed-array length mismatch: \ + cell_rows={}, cell_cols={}, d_counts={}", + pb.cell_rows.len(), + pb.cell_cols.len(), + pb.d_counts.len() + ) + .into()); + } + let cells = pb + .cell_rows + .iter() + .zip(pb.cell_cols.iter()) + .zip(pb.d_counts.iter()) + .map(|((r, c), dc)| (*r, *c, *dc)) + .collect(); + // This is the heap-less matrix kernel; ranked membership is handled + // by the explicit heap-bearing operator, not inferred from delta keys. + let delta = CountSketchDelta { + rows: pb.rows, + cols: pb.cols, + cells, + l2: pb.l2, + hh_keys: Vec::new(), + }; + self.inner + .apply_delta(&delta) + .map_err(|e| format!("apply CountSketchDelta: {e}"))?; + Ok(()) + } +} + +impl SerializableToSink for CountSketchAccumulator { + fn serialize_to_json(&self) -> Value { + serde_json::json!({ + "row_num": self.inner.rows, + "col_num": self.inner.cols, + "sketch": self.inner.sketch(), + }) + } + + fn serialize_to_bytes(&self) -> Vec { + self.inner.to_msgpack().unwrap_or_default() + } +} + +impl AggregateCore for CountSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "CountSketchAccumulator" + } + + /// Per-window base rotation: rebuild an empty signed-counter matrix + /// with the same (rows, cols) so the next window's additive cell + /// deltas align to the identical hash geometry. + fn reset_to_empty(&mut self) { + self.inner = CountSketch::new(self.inner.rows, self.inner.cols); + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge CountSketchAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + let other_cs = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountSketchAccumulator")?; + + let merged_inner = CountSketch::merge_refs(&[&self.inner, &other_cs.inner])?; + Ok(Box::new(Self { + inner: merged_inner, + })) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::CountSketch + } + + fn get_keys(&self) -> Option> { + None + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + key: &Option, + query_kwargs: &HashMap, + ) -> Result> { + use crate::Statistic; + // Key-provided path: route to MultipleSubpopulationAggregate::query + // (the canonical "what's the count of this key?" lookup), same + // pattern as CountMinSketchAccumulator. Fixed from a hand-rolled + // `DefaultHasher`-based estimator that did NOT use the sketchlib + // hash spec (its own doc admitted this — "not the sketchlib hash + // spec... the canonical compatibility path requires plumbing the + // sketchlib seeds through") — `asap_sketchlib::CountSketch::estimate` + // already hashes against the correct portable spec, so this is a + // genuine correctness fix, not just a refactor. + if let Some(key_val) = key.as_ref() { + return self.query(statistic, key_val, Some(query_kwargs)); + } + if let Some(k) = query_kwargs.get("key") { + let key_val = KeyByLabelValues::new_with_labels(vec![k.clone()]); + return self.query(statistic, &key_val, Some(query_kwargs)); + } + // No-key path: unchanged from before this fix -- CountSketch's + // signed rows have no CMS-style "min-row-sum = true total" + // property, so these are documented approximations, not a + // heavy-hitter answer. Not touched by this fix (only the + // key-provided path above had the hash-compatibility bug). + match statistic { + Statistic::Topk | Statistic::Count => { + let matrix = self.inner.sketch(); + let total: f64 = matrix.iter().flatten().map(|v| v.abs()).sum(); + let rows = matrix.len() as f64; + Ok(if rows > 0.0 { total / rows } else { 0.0 }) + } + Statistic::Sum => { + let matrix = self.inner.sketch(); + let total: f64 = matrix.iter().flatten().sum(); + let rows = matrix.len() as f64; + Ok(if rows > 0.0 { total / rows } else { 0.0 }) + } + other => Err(format!( + "CountSketchAccumulator: statistic {:?} not supported (only Topk / Count / Sum, with optional `key` in query_kwargs)", + other, + ) + .into()), + } + } +} + +impl MultipleSubpopulationAggregate for CountSketchAccumulator { + fn query( + &self, + _statistic: Statistic, + key: &KeyByLabelValues, + _query_kwargs: Option<&HashMap>, + ) -> Result> { + Ok(self.query_key(key)) + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +impl MergeableAccumulator for CountSketchAccumulator { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + let mut iter = accumulators.into_iter(); + let mut merged = iter.next().unwrap(); + for acc in iter { + merged.inner.merge(&acc.inner)?; + } + Ok(merged) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_query_key_uses_real_sketchlib_estimator() { + // `query_key` must match sketchlib's estimator and hash specification. + let mut cs = CountSketchAccumulator::new(4, 1000); + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + cs.inner.update(&key.to_semicolon_str(), 10.0); + assert_eq!( + cs.query_key(&key), + cs.inner.estimate(&key.to_semicolon_str()) + ); + } + + #[test] + fn test_multiple_subpopulation_aggregate_query() { + let mut cs = CountSketchAccumulator::new(4, 1000); + let key = KeyByLabelValues::new_with_labels(vec!["checkout".to_string()]); + cs.inner.update(&key.to_semicolon_str(), 25.0); + + let multi_trait: &dyn MultipleSubpopulationAggregate = &cs; + let result = multi_trait.query(Statistic::Sum, &key, None).unwrap(); + assert_eq!(result, cs.query_key(&key)); + + // query_statistic (the AggregateCore entry point) must route a + // provided key through the same path. + let core: &dyn AggregateCore = &cs; + let via_core = core + .query_statistic(Statistic::Sum, &Some(key.clone()), &HashMap::new()) + .unwrap(); + assert_eq!(via_core, cs.query_key(&key)); + } + + #[test] + fn test_mergeable_accumulator_merge_accumulators() { + let cs1 = CountSketchAccumulator { + inner: CountSketch::from_legacy_matrix(vec![vec![1.0, -2.0], vec![3.0, -4.0]], 2, 2), + }; + let cs2 = CountSketchAccumulator { + inner: CountSketch::from_legacy_matrix(vec![vec![-1.0, 2.0], vec![-3.0, 4.0]], 2, 2), + }; + let merged = CountSketchAccumulator::merge_accumulators(vec![cs1, cs2]).unwrap(); + assert_eq!(merged.inner.sketch(), &vec![vec![0.0, 0.0], vec![0.0, 0.0]]); + } + + #[test] + fn test_mergeable_accumulator_rejects_empty() { + let result = CountSketchAccumulator::merge_accumulators(vec![]); + assert!(result.is_err()); + } + + fn encode_state( + rows: u32, + cols: u32, + counter_type: i32, + counts_int: Vec, + counts_float: Vec, + ) -> Vec { + use asap_sketchlib::proto::sketchlib::CountSketchState; + use prost::Message; + let state = CountSketchState { + rows, + cols, + counter_type, + counts_int, + counts_float, + l2: Vec::new(), + topk: None, + }; + state.encode_to_vec() + } + + #[test] + fn test_from_sketchlib_proto_bytes_int64() { + use asap_sketchlib::proto::sketchlib::CounterType; + // Signed 2x3 matrix: row 0 = [1,-2,3], row 1 = [-4,5,-6] + let bytes = encode_state( + 2, + 3, + CounterType::Int64 as i32, + vec![1, -2, 3, -4, 5, -6], + Vec::new(), + ); + let acc = CountSketchAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + let matrix = acc.inner.sketch(); + assert_eq!(matrix[0], vec![1.0, -2.0, 3.0]); + assert_eq!(matrix[1], vec![-4.0, 5.0, -6.0]); + } + + #[test] + fn test_from_sketchlib_proto_bytes_envelope_wrapped() { + // Mirrors what DataCollector's countsketchprocessor emits: + // the state wrapped in a `SketchEnvelope{count_sketch: ...}` + // via sketchlib-go's `SerializePortableFO` + `proto.Marshal`. + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, CountSketchState, CounterType, SketchEnvelope, + }; + use prost::Message; + + let state = CountSketchState { + rows: 2, + cols: 3, + counter_type: CounterType::Int64 as i32, + counts_int: vec![1, -2, 3, -4, 5, -6], + counts_float: Vec::new(), + ..Default::default() + }; + let env = SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::CountSketch(state)), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + + let acc = CountSketchAccumulator::from_sketchlib_proto_bytes(&bytes) + .expect("envelope-wrapped decode should succeed"); + let matrix = acc.inner.sketch(); + assert_eq!(matrix[0], vec![1.0, -2.0, 3.0]); + assert_eq!(matrix[1], vec![-4.0, 5.0, -6.0]); + } + + #[test] + fn test_from_sketchlib_proto_bytes_envelope_wrong_sketch_type() { + // An envelope carrying a non-CountSketch sketch should be + // rejected with a clear error rather than silently producing + // garbage. + use asap_sketchlib::proto::sketchlib::{sketch_envelope, KllState, SketchEnvelope}; + use prost::Message; + + let env = SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Kll(KllState::default())), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + + let result = CountSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err(), "wrong-sketch envelope should error"); + } + + #[test] + fn test_from_sketchlib_proto_bytes_float64() { + use asap_sketchlib::proto::sketchlib::CounterType; + let bytes = encode_state( + 2, + 2, + CounterType::Float64 as i32, + Vec::new(), + vec![1.5, -2.5, 3.5, -4.5], + ); + let acc = CountSketchAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + let matrix = acc.inner.sketch(); + assert_eq!(matrix[0], vec![1.5, -2.5]); + assert_eq!(matrix[1], vec![3.5, -4.5]); + } + + #[test] + fn test_from_sketchlib_proto_bytes_dimension_mismatch() { + use asap_sketchlib::proto::sketchlib::CounterType; + // 2x3 declared but only 5 int entries + let bytes = encode_state( + 2, + 3, + CounterType::Int64 as i32, + vec![1, 2, 3, 4, 5], + Vec::new(), + ); + let result = CountSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err()); + assert!( + result.unwrap_err().to_string().contains("counts_int"), + "error should mention counts_int dim mismatch" + ); + } + + #[test] + fn test_from_sketchlib_proto_bytes_zero_dims_rejected() { + use asap_sketchlib::proto::sketchlib::CountSketchState; + use prost::Message; + let state = CountSketchState::default(); + let bytes = state.encode_to_vec(); + let result = CountSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err()); + assert!(result.unwrap_err().to_string().contains("degenerate dims")); + } + + #[test] + fn test_aggregate_core_merge_matches_matrix_add() { + let a = CountSketchAccumulator { + inner: CountSketch::from_legacy_matrix(vec![vec![1.0, -2.0], vec![3.0, -4.0]], 2, 2), + }; + let b = CountSketchAccumulator { + inner: CountSketch::from_legacy_matrix(vec![vec![-1.0, 2.0], vec![-3.0, 4.0]], 2, 2), + }; + let merged_box = a.merge_with(&b).expect("merge ok"); + let merged = merged_box + .as_any() + .downcast_ref::() + .expect("downcast ok"); + let m = merged.inner.sketch(); + assert_eq!(m[0], vec![0.0, 0.0]); + assert_eq!(m[1], vec![0.0, 0.0]); + } + + #[test] + fn test_aggregate_core_merge_wrong_type_rejects() { + use crate::accumulators::count_min_sketch_accumulator::CountMinSketchAccumulator; + let cs = CountSketchAccumulator::new(2, 3); + let cms = CountMinSketchAccumulator::new(2, 3); + let result = cs.merge_with(&cms); + assert!(result.is_err()); + } + + #[test] + fn test_from_msgpack_bytes_round_trip() { + let original = CountSketch::from_legacy_matrix( + vec![vec![1.0, -2.0, 3.0], vec![-4.0, 5.0, -6.0]], + 2, + 3, + ); + let bytes = original.to_msgpack().unwrap(); + let acc = CountSketchAccumulator::from_msgpack_bytes(&bytes).expect("decode ok"); + assert_eq!(acc.inner.rows, 2); + assert_eq!(acc.inner.cols, 3); + assert_eq!(acc.inner.sketch(), original.sketch()); + } + + #[test] + fn test_from_msgpack_bytes_rejects_garbage() { + let result = CountSketchAccumulator::from_msgpack_bytes(b"not valid msgpack"); + assert!(result.is_err()); + } + + #[test] + fn test_apply_proto_delta_bytes_round_trip() { + use asap_sketchlib::proto::sketchlib::CountSketchDelta as PbDelta; + use prost::Message; + + let mut acc = CountSketchAccumulator { + inner: CountSketch::from_legacy_matrix( + vec![vec![1.0, 2.0, 3.0], vec![4.0, 5.0, 6.0]], + 2, + 3, + ), + }; + let bytes = PbDelta { + rows: 2, + cols: 3, + cell_rows: vec![0, 1], + cell_cols: vec![0, 2], + d_counts: vec![10, -6], + l2: vec![], + ..Default::default() + } + .encode_to_vec(); + + acc.apply_proto_delta_bytes(&bytes).expect("apply ok"); + assert_eq!( + acc.inner.sketch(), + &vec![vec![11.0, 2.0, 3.0], vec![4.0, 5.0, 0.0]] + ); + } + + #[test] + fn test_apply_proto_delta_bytes_rejects_garbage() { + let mut acc = CountSketchAccumulator::new(2, 3); + assert!(acc.apply_proto_delta_bytes(b"not valid proto").is_err()); + } + + // ---------------------------------------------------------------- + // Defensive inbound-dimension validation (harden/sketch-dim-validation). + // Malformed / narrow-hash-budget-violating CountSketch dims must be + // rejected gracefully (Err, never a panic); valid configs the backend + // actually uses (5x2048, 5x4096, 5x2000) must still decode. + // ---------------------------------------------------------------- + + #[test] + fn test_from_sketchlib_proto_bytes_rejects_bad_dims_no_panic() { + use asap_sketchlib::proto::sketchlib::CounterType; + // 5 * ceil(log2(8192))=5*13=65 > 64 — narrow-hash-budget violation. + // counts sized to rows*cols so rejection is on dims, not length. + let n = 5usize * 8192usize; + let bytes = encode_state( + 5, + 8192, + CounterType::Int64 as i32, + vec![0i64; n], + Vec::new(), + ); + let result = CountSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err(), "budget-violating dims should be rejected"); + assert!(result.unwrap_err().to_string().contains("rejecting")); + + // A valid neighbour (5x4096) on the same path still decodes fine. + let n_ok = 5usize * 4096usize; + let ok_bytes = encode_state( + 5, + 4096, + CounterType::Int64 as i32, + vec![0i64; n_ok], + Vec::new(), + ); + let acc = CountSketchAccumulator::from_sketchlib_proto_bytes(&ok_bytes) + .expect("valid 5x4096 CountSketch should still decode"); + assert_eq!(acc.inner.rows, 5); + assert_eq!(acc.inner.cols, 4096); + } + + #[test] + fn test_from_sketchlib_proto_bytes_rejects_oversized_dims() { + use asap_sketchlib::proto::sketchlib::CounterType; + // Declare 1 x 16,777,216 = 16M cells (> 8M cap) but send an empty + // counts vector: validation must reject on the dim cap BEFORE the + // decoder tries to allocate/reshape a 16M-entry matrix. (1 row keeps + // the hash budget tiny so the cap check, not the budget check, fires.) + let bytes = encode_state( + 1, + 16_777_216, + CounterType::Int64 as i32, + Vec::new(), + Vec::new(), + ); + let result = CountSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err(), "oversized dims should be rejected"); + let msg = result.unwrap_err().to_string(); + assert!(msg.contains("cap"), "expected cell-cap error, got: {msg}"); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/count_sketch_with_heap_accumulator.rs b/crates/asap-physical-operators/src/accumulators/count_sketch_with_heap_accumulator.rs new file mode 100644 index 00000000..59dd3df1 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/count_sketch_with_heap_accumulator.rs @@ -0,0 +1,575 @@ +//! Count Sketch with Heap accumulator — wraps +//! `asap_sketchlib::CountSketchWithHeap`. +//! +//! Port of `count_min_sketch_with_heap_accumulator.rs` for the distinct +//! `CountSketchWithHeap` (median-of-signed-rows estimator) rather than +//! `CountMinSketchWithHeap` (min-over-rows estimator). The two are +//! different sketch algorithms that happen to share a storage shape and +//! wire layout -- see `asap_sketchlib::CountSketchWithHeap`'s own doc and +//! this session's `delta_apply.rs`/`decoders.rs` fix on the read side. +//! Before this file existed, `accumulator_factory.rs`'s raw-metric +//! ingest dispatch built a `CountMinSketchWithHeapAccumulator` (CMS math) +//! for `SketchAlgorithm::CountSketchWithHeap` sids -- the same conflation bug +//! already fixed on the read side, now closed on the write side too. + +use crate::{ + AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, + MultipleSubpopulationAggregate, SerializableToSink, +}; +use asap_sketchlib::{CountSketchWithHeap, CsHeapItem, MessagePackCodec}; +use serde::Deserialize; +use serde_json::Value; +use std::collections::HashMap; + +use crate::Statistic; + +/// Local serde view of the DELTA-HEAP wire frame (encoding `MSGPACK_DELTA`). +/// Identical shape to `count_min_sketch_with_heap_accumulator.rs`'s +/// `HeapDeltaWire`/`MatrixDeltaWire` -- the wire frame is generic (sparse +/// cell deltas + a full heap), not CMS-specific. See that file's doc for +/// the exact rmp_serde positional layout. +#[derive(Debug, Deserialize)] +struct HeapDeltaWire { + is_delta: bool, + matrix_delta: MatrixDeltaWire, + topk_heap: Vec<(String, f64)>, + #[allow(dead_code)] + heap_size: u64, +} + +#[derive(Debug, Deserialize)] +struct MatrixDeltaWire { + rows: u32, + cols: u32, + cells: Vec<(u32, u32, i64)>, +} + +/// Validated/flattened view of a decoded DELTA-HEAP frame. +struct HeapDeltaFrame { + rows: u32, + cols: u32, + heap_size: u64, + cells: Vec<(u32, u32, i64)>, + heap: Vec<(String, f64)>, +} + +impl HeapDeltaFrame { + fn from_msgpack(buffer: &[u8]) -> Result> { + let wire: HeapDeltaWire = rmp_serde::from_slice(buffer) + .map_err(|e| format!("decode CountSketchWithHeap delta msgpack: {e}"))?; + if !wire.is_delta { + return Err("CountSketchWithHeap delta frame has is_delta=false".into()); + } + Ok(Self { + rows: wire.matrix_delta.rows, + cols: wire.matrix_delta.cols, + heap_size: wire.heap_size, + cells: wire.matrix_delta.cells, + heap: wire.topk_heap, + }) + } +} + +/// Count Sketch with Heap accumulator — wraps `asap_sketchlib::CountSketchWithHeap`. +/// Core struct, update/merge/serde logic live in +/// `asap_sketchlib::message_pack_format::portable::countsketch_topk`. This +/// file retains QE-specific trait impls, legacy deserializers, and JSON +/// output -- same split as `CountMinSketchWithHeapAccumulator`. +#[derive(Debug, Clone)] +pub struct CountSketchWithHeapAccumulator { + pub inner: CountSketchWithHeap, +} + +impl CountSketchWithHeapAccumulator { + pub fn new(row_num: usize, col_num: usize, heap_size: usize) -> Self { + Self { + inner: CountSketchWithHeap::new(row_num, col_num, heap_size), + } + } + + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + let key_string = key.labels.join(";"); + self.inner.estimate(&key_string) + } + + /// Decode a heap-bearing CountSketch FULL msgpack frame into a heap + /// accumulator -- the window-1 / full-frame base for the DELTA-HEAP + /// delta path. Mirrors `CountMinSketchWithHeapAccumulator::from_msgpack_with_heap_bytes`. + pub fn from_msgpack_with_heap_bytes(buffer: &[u8]) -> Result> { + Ok(Self { + inner: CountSketchWithHeap::from_msgpack(buffer) + .map_err(|e| format!("deserialize CountSketchWithHeap msgpack: {e}"))?, + }) + } + + /// Apply a DELTA-HEAP msgpack frame (encoding `MSGPACK_DELTA`) onto this + /// accumulator IN PLACE. Mirrors + /// `CountMinSketchWithHeapAccumulator::apply_msgpack_heap_delta_bytes` + /// exactly -- the frame decode/apply logic is generic, not tied to + /// which estimator the rebuilt sketch uses. + pub fn apply_msgpack_heap_delta_bytes( + &mut self, + buffer: &[u8], + ) -> Result<(), Box> { + let frame = HeapDeltaFrame::from_msgpack(buffer)?; + + let rows = self.inner.rows(); + let cols = self.inner.cols(); + let heap_size = self.inner.heap_size; + + let mut matrix = self.inner.sketch_matrix(); + for (r, c, dc) in &frame.cells { + let (r, c) = (*r as usize, *c as usize); + if r >= rows || c >= cols { + continue; + } + matrix[r][c] += *dc as f64; + } + + let heap: Vec = frame + .heap + .into_iter() + .map(|(key, value)| CsHeapItem { key, value }) + .collect(); + + self.inner = CountSketchWithHeap::from_legacy_matrix(matrix, heap, rows, cols, heap_size); + Ok(()) + } + + /// Reconstruct a heap accumulator STANDALONE from a single DELTA-HEAP + /// msgpack frame, with no cached per-series base. Mirrors + /// `CountMinSketchWithHeapAccumulator::from_msgpack_heap_delta_bytes`. + pub fn from_msgpack_heap_delta_bytes( + buffer: &[u8], + ) -> Result> { + let frame = HeapDeltaFrame::from_msgpack(buffer)?; + if frame.rows == 0 || frame.cols == 0 { + return Err(format!( + "CountSketchWithHeap delta frame has zero dims (rows={}, cols={})", + frame.rows, frame.cols + ) + .into()); + } + let mut acc = Self::new( + frame.rows as usize, + frame.cols as usize, + frame.heap_size as usize, + ); + acc.apply_msgpack_heap_delta_bytes(buffer)?; + Ok(acc) + } + + /// Value-weighted heavy-hitter update -- see + /// `CountMinSketchWithHeapAccumulator::insert_value`'s doc for why + /// this (not a `+1`-per-occurrence update) is the correct semantics + /// for `topk(k, sum by (label) (metric))`-shaped queries. + pub fn insert_value(&mut self, group_label: &str, value: f64) { + self.inner.update(group_label, value); + } + + /// Read the top-`k` groups ranked by summed value (descending, tie-broken + /// by key for determinism). Mirrors `CountMinSketchWithHeapAccumulator::topk_by_value`. + pub fn topk_by_value(&self, k: usize) -> Vec<(String, f64)> { + let mut items: Vec<(String, f64)> = self + .inner + .topk_heap_items() + .into_iter() + .map(|it| (it.key, it.value)) + .collect(); + items.sort_by(|a, b| { + b.1.partial_cmp(&a.1) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.0.cmp(&b.0)) + }); + items.truncate(k); + items + } + + /// Get all keys from the top-k heap. + pub fn get_topk_keys(&self) -> Vec { + self.inner + .topk_heap_items() + .iter() + .map(|item| { + let labels: Vec = item.key.split(';').map(|s| s.to_string()).collect(); + KeyByLabelValues { labels } + }) + .collect() + } +} + +impl SerializableToSink for CountSketchWithHeapAccumulator { + fn serialize_to_json(&self) -> Value { + let heap_items: Vec = self + .inner + .topk_heap_items() + .iter() + .map(|item| { + serde_json::json!({ + "key": item.key, + "value": item.value + }) + }) + .collect(); + + serde_json::json!({ + "row_num": self.inner.rows(), + "col_num": self.inner.cols(), + "heap_size": self.inner.heap_size, + "sketch": self.inner.sketch_matrix(), + "topk_heap": heap_items + }) + } + + fn serialize_to_bytes(&self) -> Vec { + self.inner.to_msgpack().unwrap_or_default() + } +} + +impl AggregateCore for CountSketchWithHeapAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "CountSketchWithHeapAccumulator" + } + + /// Per-window base rotation -- mirrors + /// `CountMinSketchWithHeapAccumulator::reset_to_empty`. + fn reset_to_empty(&mut self) { + self.inner = + CountSketchWithHeap::new(self.inner.rows(), self.inner.cols(), self.inner.heap_size); + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge CountSketchWithHeapAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + + let other_cs = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountSketchWithHeapAccumulator")?; + + let merged = Self::merge_accumulators(vec![self.clone(), other_cs.clone()])?; + Ok(Box::new(merged)) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::CountSketchWithHeap + } + + fn get_keys(&self) -> Option> { + Some(self.get_topk_keys()) + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + key: &Option, + query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::MultipleSubpopulationAggregate; + let key_val = key + .as_ref() + .ok_or("Key required for CountSketchWithHeapAccumulator")?; + self.query(statistic, key_val, Some(query_kwargs)) + } +} + +impl MultipleSubpopulationAggregate for CountSketchWithHeapAccumulator { + fn query( + &self, + _statistic: Statistic, + key: &KeyByLabelValues, + _query_kwargs: Option<&HashMap>, + ) -> Result> { + Ok(self.query_key(key)) + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +impl MergeableAccumulator for CountSketchWithHeapAccumulator { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + let mut iter = accumulators.into_iter(); + let mut merged = iter.next().unwrap(); + for acc in iter { + merged.inner.merge(&acc.inner)?; + } + Ok(merged) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_count_sketch_with_heap_creation() { + let cs = CountSketchWithHeapAccumulator::new(4, 1000, 20); + assert_eq!(cs.inner.rows(), 4); + assert_eq!(cs.inner.cols(), 1000); + assert_eq!(cs.inner.heap_size, 20); + assert_eq!(cs.inner.topk_heap_items().len(), 0); + } + + #[test] + fn test_count_sketch_with_heap_query() { + let cs = CountSketchWithHeapAccumulator::new(2, 10, 5); + let key = KeyByLabelValues::new(); + assert_eq!(cs.query_key(&key), 0.0); + + let multi_trait: &dyn MultipleSubpopulationAggregate = &cs; + assert_eq!(multi_trait.query(Statistic::Sum, &key, None).unwrap(), 0.0); + } + + #[test] + fn test_count_sketch_with_heap_merge() { + let sketch1 = vec![ + vec![10.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0], + vec![0.0, 20.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0], + ]; + let heap1 = vec![ + CsHeapItem { + key: "key1".to_string(), + value: 100.0, + }, + CsHeapItem { + key: "key2".to_string(), + value: 50.0, + }, + ]; + let sketch2 = vec![ + vec![5.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0], + vec![0.0, 15.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0], + ]; + let heap2 = vec![ + CsHeapItem { + key: "key3".to_string(), + value: 75.0, + }, + CsHeapItem { + key: "key1".to_string(), + value: 80.0, + }, + ]; + + let cs1 = CountSketchWithHeapAccumulator { + inner: CountSketchWithHeap::from_legacy_matrix(sketch1, heap1, 2, 10, 5), + }; + let cs2 = CountSketchWithHeapAccumulator { + inner: CountSketchWithHeap::from_legacy_matrix(sketch2, heap2, 2, 10, 3), + }; + + let result = CountSketchWithHeapAccumulator::merge_accumulators(vec![cs1, cs2]); + assert!(result.is_ok()); + let merged = result.unwrap(); + assert_eq!(merged.inner.sketch_matrix()[0][0], 15.0); + assert_eq!(merged.inner.sketch_matrix()[1][1], 35.0); + assert_eq!(merged.inner.heap_size, 3); + assert!(merged.inner.topk_heap_items().len() <= 3); + } + + #[test] + fn test_count_sketch_with_heap_merge_single() { + let cs = CountSketchWithHeapAccumulator::new(2, 3, 5); + let result = CountSketchWithHeapAccumulator::merge_accumulators(vec![cs.clone()]); + assert!(result.is_ok()); + let merged = result.unwrap(); + assert_eq!(merged.inner.rows(), cs.inner.rows()); + assert_eq!(merged.inner.cols(), cs.inner.cols()); + assert_eq!(merged.inner.heap_size, cs.inner.heap_size); + } + + #[test] + fn test_count_sketch_with_heap_merge_dimension_mismatch() { + let cs1 = CountSketchWithHeapAccumulator::new(2, 10, 5); + let cs2 = CountSketchWithHeapAccumulator::new(3, 10, 5); + let result = CountSketchWithHeapAccumulator::merge_accumulators(vec![cs1, cs2]); + assert!(result.is_err()); + } + + #[test] + fn test_count_sketch_with_heap_as_aggregate_core() { + let cs = CountSketchWithHeapAccumulator::new(2, 3, 5); + assert_eq!(cs.type_name(), "CountSketchWithHeapAccumulator"); + } + + #[test] + fn test_get_topk_keys() { + let mut cs = CountSketchWithHeapAccumulator::new(2, 3, 5); + cs.inner.update("label1;label2", 100.0); + cs.inner.update("label3;label4", 50.0); + + let keys = cs.get_topk_keys(); + assert_eq!(keys.len(), 2); + let label_sets: std::collections::HashSet<_> = + keys.iter().map(|k| k.labels.clone()).collect(); + assert!(label_sets.contains(&vec!["label1".to_string(), "label2".to_string()])); + assert!(label_sets.contains(&vec!["label3".to_string(), "label4".to_string()])); + } + + #[test] + fn test_multiple_subpopulation_aggregate() { + let cs = CountSketchWithHeapAccumulator::new(3, 50, 10); + let key = KeyByLabelValues::new(); + + let multi_trait: &dyn MultipleSubpopulationAggregate = &cs; + let result = multi_trait.query(Statistic::Sum, &key, None).unwrap(); + assert_eq!(result, 0.0); + + let keys = multi_trait.get_keys(); + assert!(keys.is_some()); + assert_eq!(keys.unwrap().len(), 0); + } + + #[test] + fn test_pwr_full_then_delta_then_delta_reconstructs_per_window() { + use asap_sketchlib::MessagePackCodec; + + let w1 = CountSketchWithHeap::from_legacy_matrix( + vec![vec![300.0; 4]; 5], + vec![CsHeapItem { + key: "k".into(), + value: 300.0, + }], + 5, + 4, + 20, + ); + let w1_bytes = w1.to_msgpack().expect("w1 full msgpack"); + let mut base = CountSketchWithHeapAccumulator::from_msgpack_with_heap_bytes(&w1_bytes) + .expect("decode w1 full frame as heap accumulator"); + assert_eq!(base.inner.sketch_matrix()[0][0], 300.0); + + let w2_frame = encode_delta_heap(5, 4, &[(0, 0, 50), (1, 1, 50)], &[("k", 50.0)], 20); + base.reset_to_empty(); + assert_eq!( + base.inner.sketch_matrix()[0][0], + 0.0, + "reset_to_empty cleared matrix" + ); + base.apply_msgpack_heap_delta_bytes(&w2_frame) + .expect("apply w2 delta"); + assert_eq!(base.inner.sketch_matrix()[0][0], 50.0, "window-2 cell"); + assert_eq!(base.inner.sketch_matrix()[1][1], 50.0); + assert_eq!(base.inner.sketch_matrix()[2][2], 0.0); + let h2: Vec<_> = base.inner.topk_heap_items(); + assert_eq!(h2.len(), 1); + assert_eq!(h2[0].key, "k"); + assert_eq!(h2[0].value, 50.0); + + let w3_frame = encode_delta_heap(5, 4, &[(0, 0, 80)], &[("k", 80.0)], 20); + base.reset_to_empty(); + base.apply_msgpack_heap_delta_bytes(&w3_frame) + .expect("apply w3 delta"); + assert_eq!(base.inner.sketch_matrix()[0][0], 80.0, "window-3 cell"); + assert_eq!(base.inner.sketch_matrix()[1][1], 0.0, "no window-2 leakage"); + let h3 = base.inner.topk_heap_items(); + assert_eq!(h3.len(), 1); + assert_eq!(h3[0].value, 80.0); + } + + #[test] + fn test_apply_delta_rejects_full_frame_and_garbage() { + use asap_sketchlib::MessagePackCodec; + let mut acc = CountSketchWithHeapAccumulator::new(2, 4, 5); + let full = CountSketchWithHeap::from_legacy_matrix( + vec![vec![1.0; 4]; 2], + vec![CsHeapItem { + key: "a".into(), + value: 1.0, + }], + 2, + 4, + 5, + ) + .to_msgpack() + .unwrap(); + assert!(acc.apply_msgpack_heap_delta_bytes(&full).is_err()); + assert!(acc.apply_msgpack_heap_delta_bytes(b"not msgpack").is_err()); + } + + fn encode_delta_heap( + rows: u32, + cols: u32, + cells: &[(u32, u32, i64)], + heap: &[(&str, f64)], + heap_size: u64, + ) -> Vec { + #[derive(serde::Serialize)] + struct W<'a>( + bool, + (u32, u32, &'a [(u32, u32, i64)]), + Vec<(String, f64)>, + u64, + ); + let heap_owned: Vec<(String, f64)> = + heap.iter().map(|(k, v)| (k.to_string(), *v)).collect(); + let w = W(true, (rows, cols, cells), heap_owned, heap_size); + rmp_serde::to_vec(&w).expect("encode delta-heap") + } + + #[test] + fn insert_value_accumulates_summed_value_in_heap() { + let mut acc = CountSketchWithHeapAccumulator::new(4, 1024, 8); + acc.insert_value("g", 10.0); + acc.insert_value("g", 25.0); + let top = acc.topk_by_value(1); + assert_eq!(top.len(), 1); + assert_eq!(top[0].0, "g"); + assert!( + (top[0].1 - 35.0).abs() < 1e-6, + "summed value should be 35 (10+25), got {}", + top[0].1 + ); + } + + /// The core proof this file exists at all: `CountSketchWithHeapAccumulator` + /// wraps the real, distinct `asap_sketchlib::CountSketchWithHeap` -- + /// not the CMS-family `CountMinSketchWithHeap` a collapsed dispatch + /// used to substitute (the exact bug this file fixes on the ingest + /// side, mirroring the already-fixed read side). Two different Rust + /// types means `merge_with` rejects mixing them at the type-check + /// level, same as any other mismatched-family merge attempt -- + /// verified directly rather than via a numeric estimate comparison + /// (asap_sketchlib's own test suite already proves the median vs + /// min-over-rows divergence at the sketch-math level). + #[test] + fn test_rejects_merge_with_cms_family_accumulator() { + use crate::accumulators::count_min_sketch_with_heap_accumulator::CountMinSketchWithHeapAccumulator; + + let cs = CountSketchWithHeapAccumulator::new(4, 64, 10); + let cms = CountMinSketchWithHeapAccumulator::new(4, 64, 10); + let result = cs.merge_with(&cms); + assert!( + result.is_err(), + "CountSketchWithHeapAccumulator must not merge with CountMinSketchWithHeapAccumulator \ + -- different algorithms sharing only a storage shape" + ); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/datasketches_kll_accumulator.rs b/crates/asap-physical-operators/src/accumulators/datasketches_kll_accumulator.rs new file mode 100644 index 00000000..4ccfd3d0 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/datasketches_kll_accumulator.rs @@ -0,0 +1,727 @@ +use crate::{ + AggregateCore, AggregationType, AuxStats, MergeableAccumulator, SerializableToSink, + SingleSubpopulationAggregate, +}; +use asap_sketchlib::{KllSketch, MessagePackCodec}; +use base64::{engine::general_purpose, Engine as _}; +use serde_json::Value; +use std::collections::HashMap; +#[cfg(feature = "extra_debugging")] +use std::time::Instant; +use tracing::debug; + +use crate::Statistic; + +/// KLL sketch accumulator — wraps asap_sketchlib::KllSketch. +/// Core struct, update/merge/serde logic live in `asap_sketchlib::sketches`. +/// This file retains QE-specific trait impls and JSON output. +pub struct DatasketchesKLLAccumulator { + pub inner: KllSketch, +} + +impl DatasketchesKLLAccumulator { + pub fn new(k: u16) -> Self { + Self { + inner: KllSketch::new(k), + } + } + + pub fn update(&mut self, value: f64) { + self.inner.update(value); + } + + pub fn get_quantile(&self, quantile: f64) -> f64 { + self.inner.quantile(quantile) + } + + /// Decode from the modified OTLP wire format's + /// `KLLSketchDataPoint.sketch` bytes when + /// `encoding = KLL_SKETCH_ENCODING_MSGPACK`. The bytes are the + /// MessagePack serialization of the cross-language sketch-core + /// `KllSketch` struct — PR I parity entrypoint. Unlike the + /// `_ENCODING_PROTO` path (which does lossy statistical + /// reconstruction via `update()` replay), the msgpack path is a + /// bit-identical round-trip because sketch-core's `KllSketch` + /// serializes its full internal state to msgpack. + pub fn from_msgpack_bytes(buffer: &[u8]) -> Result> { + Ok(Self { + inner: KllSketch::from_msgpack(buffer) + .map_err(|e| -> Box { e.to_string().into() })?, + }) + } + + /// Decode from the modified OTLP wire format's + /// `KLLSketchDataPoint.sketch` bytes — the protobuf-encoded + /// `asap_sketchlib::proto::sketchlib::KllState` message that + /// DataCollector's `kllprocessor` emits when + /// `encoding = KLL_SKETCH_ENCODING_PROTO`. + /// + /// The neutral codec decodes the sketchlib envelope. + /// The level-aware constructor below preserves the supplied retained + /// sample layout without replaying updates. + pub fn from_sketchlib_proto_bytes(buffer: &[u8]) -> Result> { + let state = asap_sketch_codec::kll_state(buffer)?; + if state.k < 8 { + return Err(format!("KllState.k must be >= 8 (got {})", state.k).into()); + } + if state.k > u16::MAX as u32 { + return Err(format!( + "KllState.k does not fit in u16 (got {}, max {})", + state.k, + u16::MAX + ) + .into()); + } + // Validate the levels[] boundary array if it is populated. The + // proto contract says `levels[0] == 0` and + // `levels[num_levels] == items.len()`. If the producer left + // levels empty (common when num_levels is zero), skip. + if !state.levels.is_empty() { + if state.levels.len() as u32 != state.num_levels + 1 { + return Err(format!( + "KllState levels length = {}, expected num_levels+1 = {}", + state.levels.len(), + state.num_levels + 1 + ) + .into()); + } + if state.levels[0] != 0 { + return Err(format!("KllState.levels[0] = {}, expected 0", state.levels[0]).into()); + } + if *state.levels.last().unwrap() as usize != state.items.len() { + return Err(format!( + "KllState.levels[{}] = {}, expected items.len() = {}", + state.num_levels, + state.levels.last().unwrap(), + state.items.len() + ) + .into()); + } + } + let k = state.k as u16; + // Direct, bit-exact reconstruction from the portable state (no per-item + // `update()` replay) whenever the producer supplied the `levels[]` + // boundary array — which it does for any non-empty sketch. Falls back to + // the statistical replay only when `levels` is absent (empty sketch). + if !state.levels.is_empty() { + // KllState is highest-level first; the in-memory constructor + // expects L0 first. Replaying or copying the wire order changes + // retained-item weights after the first compaction. + let mut items = Vec::with_capacity(state.items.len()); + let mut levels = vec![0]; + if state + .levels + .windows(2) + .any(|bounds| bounds[0] > bounds[1] || bounds[1] as usize > state.items.len()) + { + return Err("KllState levels must be monotonic and within items".into()); + } + for bounds in state.levels.windows(2).rev() { + items.extend_from_slice(&state.items[bounds[0] as usize..bounds[1] as usize]); + levels.push(items.len()); + } + return Ok(Self { + inner: KllSketch::from_portable_state( + k, + &items, + &levels, + state.num_levels as usize, + ) + .map_err(|e| -> Box { e.into() })?, + }); + } + let mut acc = Self::new(k); + for item in &state.items { + acc.update(*item); + } + Ok(acc) + } + + /// Merge multiple accumulators efficiently without cloning all of them. + pub fn merge_multiple( + accumulators: &[Box], + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + + let mut kll_accumulators = Vec::with_capacity(accumulators.len()); + for acc in accumulators { + if acc.get_accumulator_type() != AggregationType::DatasketchesKLL { + return Err(format!( + "Cannot merge DatasketchesKLLAccumulator with {:?}", + acc.get_accumulator_type() + ) + .into()); + } + let kll_acc = acc + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to DatasketchesKLLAccumulator")?; + kll_accumulators.push(kll_acc); + } + + let inner_refs: Vec<&KllSketch> = kll_accumulators.iter().map(|acc| &acc.inner).collect(); + let merged_inner = KllSketch::merge_refs(&inner_refs)?; + Ok(Self { + inner: merged_inner, + }) + } +} + +// Manual trait implementations since the C++ library doesn't provide them +impl Clone for DatasketchesKLLAccumulator { + fn clone(&self) -> Self { + Self { + inner: self.inner.clone(), + } + } +} + +impl std::fmt::Debug for DatasketchesKLLAccumulator { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("DatasketchesKLLAccumulator") + .field("k", &self.inner.k) + .field("sketch_n", &self.inner.count()) + .finish() + } +} + +// TODO: verify this +// Thread safety: The C++ library is not thread-safe by default, but since we're using it +// in a single-threaded context per accumulator instance and only sharing read-only operations, +// this should be safe. +unsafe impl Send for DatasketchesKLLAccumulator {} +unsafe impl Sync for DatasketchesKLLAccumulator {} + +impl SerializableToSink for DatasketchesKLLAccumulator { + fn serialize_to_json(&self) -> Value { + // Mirror Python implementation: {"sketch": base64_encoded_string} + let sketch_bytes = self.inner.sketch_bytes(); + let sketch_b64 = general_purpose::STANDARD.encode(&sketch_bytes); + serde_json::json!({ "sketch": sketch_b64 }) + } + + fn serialize_to_bytes(&self) -> Vec { + self.inner.to_msgpack().unwrap_or_default() + } +} + +impl AggregateCore for DatasketchesKLLAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "DatasketchesKLLAccumulator" + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + #[cfg(feature = "extra_debugging")] + let merge_with_start = Instant::now(); + #[cfg(feature = "extra_debugging")] + debug!( + "[PERF] DatasketchesKLLAccumulator::merge_with() started - self.k={}, self.n={}", + self.inner.k, + self.inner.count() + ); + + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge DatasketchesKLLAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + + let other_kll = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to DatasketchesKLLAccumulator")?; + + let merged_inner = KllSketch::merge_refs(&[&self.inner, &other_kll.inner])?; + let merged = Self { + inner: merged_inner, + }; + + #[cfg(feature = "extra_debugging")] + debug!( + "[PERF] DatasketchesKLLAccumulator::merge_with() TOTAL TIME: {:?}", + merge_with_start.elapsed() + ); + + Ok(Box::new(merged)) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::DatasketchesKLL + } + + fn approx_memory_bytes(&self) -> usize { + // KLL with default k=200 holds ~2*k items (~3 KiB). Round up + // for overhead. + 4 * 1024 + } + + fn aux_stats(&self) -> AuxStats { + // KLL natively tracks `count` (n, samples observed). min/max + // are available from the underlying sketch but only via a + // O(k) quantile extraction at quantile=0/1, which is not + // a cheap trait-method call. sum is not retained by KLL. + // + // Surface only count here; follow-up PR may add min/max via a + // dedicated accessor on sketch-core. `sum_over_time` queries + // on KLL fall back to query_statistic as they do today. + AuxStats { + count: Some(self.inner.count()), + ..AuxStats::empty() + } + } + + fn get_keys(&self) -> Option> { + None + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + _key: &Option, + query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::SingleSubpopulationAggregate; + self.query(statistic, Some(query_kwargs)) + } +} + +impl SingleSubpopulationAggregate for DatasketchesKLLAccumulator { + fn query( + &self, + statistic: Statistic, + query_kwargs: Option<&HashMap>, + ) -> Result> { + match statistic { + Statistic::Quantile => { + debug!( + "Querying DatasketchesKLLAccumulator for quantile with kwargs: {:?}", + query_kwargs + ); + let quantile = query_kwargs + .and_then(|kwargs| kwargs.get("quantile")) + .ok_or("Missing quantile parameter for quantile query")? + .parse::() + .map_err(|_| "Invalid quantile parameter format")?; + + if !(0.0..=1.0).contains(&quantile) { + return Err("Quantile must be between 0.0 and 1.0".into()); + } + + Ok(self.get_quantile(quantile)) + } + _ => Err( + format!("Unsupported statistic in DatasketchesKLLAccumulator: {statistic:?}") + .into(), + ), + } + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +impl MergeableAccumulator for DatasketchesKLLAccumulator { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + let mut iter = accumulators.into_iter(); + let mut merged = iter.next().unwrap(); + for acc in iter { + merged.inner.merge(&acc.inner)?; + } + Ok(merged) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use prost::Message; + + fn encode_state(state: asap_sketchlib::proto::sketchlib::KllState) -> Vec { + use asap_sketchlib::proto::sketchlib::{sketch_envelope, SketchEnvelope}; + SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Kll(state)), + ..Default::default() + } + .encode_to_vec() + } + + #[test] + fn test_datasketches_kll_creation() { + let kll = DatasketchesKLLAccumulator::new(200); + assert!(kll.inner.count() == 0); + assert_eq!(kll.inner.k, 200); + } + + #[test] + fn test_datasketches_kll_update() { + let mut kll = DatasketchesKLLAccumulator::new(200); + kll.update(10.0); + kll.update(20.0); + kll.update(15.0); + assert_eq!(kll.inner.count(), 3); + } + + #[test] + fn test_datasketches_kll_quantile() { + let mut kll = DatasketchesKLLAccumulator::new(200); + for i in 1..=10 { + kll.update(i as f64); + } + assert_eq!(kll.get_quantile(0.0), 1.0); + assert_eq!(kll.get_quantile(1.0), 10.0); + // Sketchlib KLL is approximate; 0.5 quantile of 1..10 may be 5, 6, or 7. + let q50 = kll.get_quantile(0.5); + assert!((q50 - 6.0).abs() <= 1.0, "expected median ~6, got {q50}"); + } + + #[test] + fn test_datasketches_kll_query() { + let mut kll = DatasketchesKLLAccumulator::new(200); + for i in 1..=10 { + kll.update(i as f64); + } + + let mut query_kwargs = HashMap::new(); + query_kwargs.insert("quantile".to_string(), "0.5".to_string()); + let result = kll.query(Statistic::Quantile, Some(&query_kwargs)).unwrap(); + // Sketchlib KLL is approximate; 0.5 quantile of 1..10 may be 5, 6, or 7. + assert!( + (result - 6.0).abs() <= 1.0, + "expected median ~6, got {result}" + ); + + assert!(kll.query(Statistic::Sum, Some(&query_kwargs)).is_err()); + } + + #[test] + fn test_datasketches_kll_merge() { + let mut kll1 = DatasketchesKLLAccumulator::new(200); + let mut kll2 = DatasketchesKLLAccumulator::new(200); + + for i in 1..=5 { + kll1.update(i as f64); + } + for i in 6..=10 { + kll2.update(i as f64); + } + + let merged = DatasketchesKLLAccumulator::merge_accumulators(vec![kll1, kll2]).unwrap(); + assert_eq!(merged.inner.count(), 10); + assert_eq!(merged.get_quantile(0.0), 1.0); + assert_eq!(merged.get_quantile(1.0), 10.0); + } + + #[test] + fn test_datasketches_kll_get_keys() { + let kll = DatasketchesKLLAccumulator::new(200); + assert_eq!(kll.type_name(), "DatasketchesKLLAccumulator"); + } + + #[test] + fn test_trait_object() { + let mut kll = DatasketchesKLLAccumulator::new(200); + kll.update(5.0); + let trait_obj: Box = Box::new(kll); + assert_eq!(trait_obj.type_name(), "DatasketchesKLLAccumulator"); + } + + #[test] + fn test_datasketches_kll_query_with_kwargs() { + let mut kll = DatasketchesKLLAccumulator::new(200); + for i in 1..=10 { + kll.update(i as f64); + } + + let mut query_kwargs = HashMap::new(); + query_kwargs.insert("quantile".to_string(), "0.5".to_string()); + let result = kll.query(Statistic::Quantile, Some(&query_kwargs)).unwrap(); + // Sketchlib KLL is approximate; 0.5 quantile of 1..10 may be 5, 6, or 7. + assert!( + (result - 6.0).abs() <= 1.0, + "expected median ~6, got {result}" + ); + + query_kwargs.insert("quantile".to_string(), "0.9".to_string()); + let result = kll.query(Statistic::Quantile, Some(&query_kwargs)).unwrap(); + // Sketchlib KLL is approximate; 0.9 quantile of 1..10 may be 9 or 10. + assert!( + (9.0..=10.0).contains(&result), + "expected 0.9 quantile in [9,10], got {result}" + ); + + query_kwargs.insert("quantile".to_string(), "0.0".to_string()); + assert_eq!( + kll.query(Statistic::Quantile, Some(&query_kwargs)).unwrap(), + 1.0 + ); + + query_kwargs.insert("quantile".to_string(), "1.0".to_string()); + assert_eq!( + kll.query(Statistic::Quantile, Some(&query_kwargs)).unwrap(), + 10.0 + ); + + assert!(kll.query(Statistic::Quantile, None).is_err()); + + query_kwargs.insert("quantile".to_string(), "invalid".to_string()); + assert!(kll.query(Statistic::Quantile, Some(&query_kwargs)).is_err()); + + query_kwargs.insert("quantile".to_string(), "1.5".to_string()); + assert!(kll.query(Statistic::Quantile, Some(&query_kwargs)).is_err()); + + query_kwargs.insert("quantile".to_string(), "-0.1".to_string()); + assert!(kll.query(Statistic::Quantile, Some(&query_kwargs)).is_err()); + + query_kwargs.insert("quantile".to_string(), "0.5".to_string()); + assert!(kll.query(Statistic::Sum, Some(&query_kwargs)).is_err()); + } + + #[test] + fn test_datasketches_kll_merge_multiple() { + let mut kll1 = DatasketchesKLLAccumulator::new(200); + let mut kll2 = DatasketchesKLLAccumulator::new(200); + let mut kll3 = DatasketchesKLLAccumulator::new(200); + + for i in 1..=5 { + kll1.update(i as f64); + } + for i in 6..=10 { + kll2.update(i as f64); + } + for i in 11..=15 { + kll3.update(i as f64); + } + + let boxed_accs: Vec> = + vec![Box::new(kll1), Box::new(kll2), Box::new(kll3)]; + + let merged = DatasketchesKLLAccumulator::merge_multiple(&boxed_accs).unwrap(); + assert_eq!(merged.inner.count(), 15); + assert_eq!(merged.get_quantile(0.0), 1.0); + assert_eq!(merged.get_quantile(1.0), 15.0); + assert_eq!(merged.get_quantile(0.5), 8.0); + } + + #[test] + fn test_datasketches_kll_merge_multiple_error_cases() { + let empty: Vec> = vec![]; + assert!(DatasketchesKLLAccumulator::merge_multiple(&empty).is_err()); + + let kll1 = DatasketchesKLLAccumulator::new(200); + let kll2 = DatasketchesKLLAccumulator::new(100); + let boxed_accs: Vec> = vec![Box::new(kll1), Box::new(kll2)]; + assert!(DatasketchesKLLAccumulator::merge_multiple(&boxed_accs).is_err()); + + use crate::accumulators::sum_accumulator::SumAccumulator; + let kll = DatasketchesKLLAccumulator::new(200); + let sum = SumAccumulator::new(); + let mixed_accs: Vec> = vec![Box::new(kll), Box::new(sum)]; + assert!(DatasketchesKLLAccumulator::merge_multiple(&mixed_accs).is_err()); + } + + #[test] + fn test_from_sketchlib_proto_bytes_reconstructs_quantiles() { + // Build a KllState with 64 items in level order; the decoder + // replays every item through `update()` so the reconstructed + // sketch is statistically equivalent — quantile estimates + // match the ground truth (sorted items) within KLL's own + // rank-error bound for k=200. + use asap_sketchlib::proto::sketchlib::KllState; + + let items: Vec = (0..64).map(|i| i as f64).collect(); + let state = KllState { + k: 200, + m: 8, + num_levels: 1, + levels: vec![0, 64], + items: items.clone(), + coin: None, + offset: 0.0, + value_scale: 0, + residuals: Vec::new(), + }; + let bytes = encode_state(state); + + let acc = + DatasketchesKLLAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + assert_eq!(acc.inner.count(), 64); + // For 64 values 0..63, the true median is 31.5 and quantile + // error is ~1% × range = 0.63. KLL's own point query can + // legally be off by up to ε × N ~= 0.01 × 64 = 0.64. Allow a + // generous tolerance since the important invariant is "the + // decoded sketch is queryable and returns a sensible value". + let median = acc.get_quantile(0.5); + assert!( + (median - 31.5).abs() <= 10.0, + "reconstructed median {median} is outside tolerance of true median 31.5" + ); + let q01 = acc.get_quantile(0.01); + let q99 = acc.get_quantile(0.99); + assert!( + q01 <= q99, + "quantile monotonicity violated: q01={q01}, q99={q99}" + ); + } + + // Compacted portable state is highest-level first, unlike the runtime buffer. + #[test] + fn compacted_wire_state_preserves_count_and_quantiles() { + use asap_sketchlib::{proto::sketchlib::KllState, sketches::KLL}; + let mut source = KLL::::init_kll_with_seed(32, 123); + for i in 0..1000 { + source.update(&(((i * 7919 + 17) % 1009) as f64 / 1009.0)); + } + assert!(source.wire_num_levels() > 1); + let state = KllState { + k: 32, + m: source.wire_m(), + num_levels: source.wire_num_levels(), + levels: source.wire_levels(), + items: source.wire_items(), + coin: None, + offset: 0.0, + value_scale: 0, + residuals: vec![], + }; + let decoded = + DatasketchesKLLAccumulator::from_sketchlib_proto_bytes(&encode_state(state)).unwrap(); + assert_eq!(decoded.inner.count(), source.count() as u64); + for q in [0.0, 0.1, 0.5, 0.9, 1.0] { + assert_eq!(decoded.inner.quantile(q), source.quantile(q), "q={q}"); + } + } + + #[test] + fn test_from_sketchlib_proto_bytes_envelope_wrapped() { + // Mirrors what DataCollector's kllprocessor emits: the state + // wrapped in a `SketchEnvelope{kll: ...}` via sketchlib-go's + // `SerializePortableFO` + `proto.Marshal`. + use asap_sketchlib::proto::sketchlib::{sketch_envelope, KllState, SketchEnvelope}; + + let items: Vec = (0..64).map(|i| i as f64).collect(); + let state = KllState { + k: 200, + m: 8, + num_levels: 1, + levels: vec![0, 64], + items, + coin: None, + offset: 0.0, + value_scale: 0, + residuals: Vec::new(), + }; + let env = SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Kll(state)), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + + let acc = DatasketchesKLLAccumulator::from_sketchlib_proto_bytes(&bytes) + .expect("envelope-wrapped decode should succeed"); + assert_eq!(acc.inner.count(), 64); + } + + #[test] + fn test_from_sketchlib_proto_bytes_envelope_wrong_sketch_type() { + use asap_sketchlib::proto::sketchlib::{sketch_envelope, CountMinState, SketchEnvelope}; + + let env = SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::CountMin( + CountMinState::default(), + )), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + + let result = DatasketchesKLLAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err(), "wrong-sketch envelope should error"); + } + + #[test] + fn test_from_sketchlib_proto_bytes_rejects_small_k() { + use asap_sketchlib::proto::sketchlib::KllState; + let state = KllState { + k: 4, // < minimum of 8 + m: 2, + num_levels: 0, + levels: Vec::new(), + items: Vec::new(), + coin: None, + offset: 0.0, + value_scale: 0, + residuals: Vec::new(), + }; + let bytes = encode_state(state); + let result = DatasketchesKLLAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err()); + assert!(result.unwrap_err().to_string().contains("k must be >= 8")); + } + + #[test] + fn test_from_sketchlib_proto_bytes_rejects_inconsistent_levels() { + use asap_sketchlib::proto::sketchlib::KllState; + // num_levels=1 but levels array has 3 entries instead of 2 + let state = KllState { + k: 200, + m: 8, + num_levels: 1, + levels: vec![0, 5, 10], + items: vec![1.0, 2.0, 3.0, 4.0, 5.0], + coin: None, + offset: 0.0, + value_scale: 0, + residuals: Vec::new(), + }; + let bytes = encode_state(state); + let result = DatasketchesKLLAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err()); + assert!(result.unwrap_err().to_string().contains("levels length")); + } + + #[test] + fn aux_stats_exposes_count_via_kll_n() { + let mut acc = DatasketchesKLLAccumulator::new(200); + for i in 0..50 { + acc.update(i as f64); + } + let aux = acc.aux_stats(); + assert_eq!(aux.count, Some(50)); + // KLL doesn't natively expose min/max cheaply and doesn't + // track sum at all — those fields must be None so callers + // fall through to query_statistic. + assert_eq!(aux.sum, None); + assert_eq!(aux.min, None); + assert_eq!(aux.max, None); + } + + #[test] + fn aux_stats_empty_kll_has_zero_count() { + let acc = DatasketchesKLLAccumulator::new(200); + assert_eq!(acc.aux_stats().count, Some(0)); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/dd_sketch_accumulator.rs b/crates/asap-physical-operators/src/accumulators/dd_sketch_accumulator.rs new file mode 100644 index 00000000..5015b50b --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/dd_sketch_accumulator.rs @@ -0,0 +1,665 @@ +//! DDSketch accumulator — wraps `asap_sketchlib::DdSketch`. +//! +//! Concrete accumulator reached from the modified-OTLP +//! `Metric.data = DDSketch{…}` hot path (PR C-CountSketch follow-up). +//! Merge via bucket-index alignment on the inner sketch, serialize as +//! MessagePack for the sink, and decode from the sketchlib +//! `DDSketchState` proto. +//! +//! Query semantics follow the STRICT policy after the DataPoint-level +//! METRIC scalars were dropped from the wire format +//! (ProjectASAP/sketchlib-go#243 / asap_sketchlib#57): the sketch serves +//! Quantile (log-bucket estimation) and Count (sum of bucket counts). +//! Sum/Min/Max are no longer derivable from the wire bytes and are +//! served by controller-provisioned exact aggregations — `query_statistic` +//! returns the unavailable-statistic error for them. + +use crate::{AggregateCore, AggregationType, KeyByLabelValues, SerializableToSink}; +use asap_sketchlib::{DdSketch, DdSketchDelta, MessagePackCodec}; +use serde_json::Value; +use std::collections::HashMap; + +/// DDSketch accumulator — inner log-bucketed sketch. +#[derive(Debug, Clone)] +pub struct DDSketchAccumulator { + pub inner: DdSketch, + /// Edge sampling probability `p ∈ (0,1]` carried on the producer's + /// `SketchEnvelope.sample_p`. The edge admits each value with probability + /// `p` (NitroSketch geometric skip), so `inner.total_count()` is ~`p`× the + /// true count and a `Count` query must rescale by `1/p`. Quantiles are + /// rank-preserving and need NO rescale. `1.0` (and the proto3 default `0.0`, + /// dual-read as `1.0`) means no sampling, so the rescale is a no-op and the + /// behaviour is identical to before. The factor is a per-series config + /// constant: it is set from the first (always-full, otel.rs ingest + /// contract) frame and preserved across delta applies, window-boundary + /// `reset_to_empty`, and `merge_with`. + pub sample_p: f64, +} + +/// Normalize a wire `sample_p` to a usable rescale denominator. `0.0` (proto3 +/// default), `>= 1.0`, and non-finite all collapse to `1.0` (no sampling), so a +/// `Count` rescale by `1/p` is a no-op on unsampled / legacy frames. +pub(crate) fn normalize_sample_p(p: f64) -> f64 { + if p.is_finite() && p > 0.0 && p < 1.0 { + p + } else { + 1.0 + } +} + +impl DDSketchAccumulator { + pub fn new(alpha: f64) -> Self { + Self { + inner: DdSketch::new(alpha), + sample_p: 1.0, + } + } + + /// Read the normalized edge sampling probability from a full-frame + /// `SketchEnvelope`'s `sample_p`. Returns `1.0` (no sampling) for bare + /// `DdSketchState` bytes or any decode failure — the primary production + /// decode path (`reconstruct_via_runtime`) discards the envelope's + /// `sample_p`, so the ingest call site re-reads it from the same bytes. + pub fn sample_p_from_envelope_bytes(buffer: &[u8]) -> f64 { + use asap_sketchlib::proto::sketchlib::SketchEnvelope; + use prost::Message; + SketchEnvelope::decode(buffer) + .map(|env| normalize_sample_p(env.sample_p)) + .unwrap_or(1.0) + } + + /// Decode from the modified OTLP wire format's + /// `DDSketchDataPoint.sketch` bytes when + /// `encoding = DDSKETCH_ENCODING_MSGPACK`. The bytes are the + /// MessagePack serialization of the cross-language sketch-core + /// `DdSketch` struct — PR I parity entrypoint. + pub fn from_msgpack_bytes(buffer: &[u8]) -> Result> { + Ok(Self { + inner: DdSketch::from_msgpack(buffer) + .map_err(|e| format!("deserialize DdSketch msgpack: {e}"))?, + // The msgpack DdSketch struct carries no envelope/sample_p; the + // msgpack path is parity/test-only and is never edge-sampled. + sample_p: 1.0, + }) + } + + /// Decode from the modified OTLP wire format's + /// `DDSketchDataPoint.sketch` bytes — the protobuf-encoded + /// `asap_sketchlib::proto::sketchlib::DDSketchState` message that + /// DataCollector's `ddsketchprocessor` emits when + /// `encoding = DD_SKETCH_ENCODING_PROTO`. + pub fn from_sketchlib_proto_bytes(buffer: &[u8]) -> Result> { + let (state, sample_p) = asap_sketch_codec::ddsketch_state(buffer)?; + if !(state.alpha > 0.0 && state.alpha < 1.0) { + return Err(format!( + "DDSketchState alpha {} out of range (expected 0 < alpha < 1)", + state.alpha + ) + .into()); + } + // The DataPoint-level METRIC scalars (count/sum/min/max) were + // dropped from `DDSketchState` (ProjectASAP/sketchlib-go#243 / + // asap_sketchlib#57). Reconstruct from the bucket store only: + // `DdSketch::from_raw` now takes just (alpha, store_counts, + // store_offset) and recovers `count` by summing the bucket + // counts via `total_count()`. + let inner = DdSketch::from_raw(state.alpha, state.store_counts.clone(), state.store_offset); + Ok(Self { + inner, + sample_p: normalize_sample_p(sample_p), + }) + } + + /// Apply a proto-encoded `DDSketchDelta` frame to this + /// accumulator's inner sketch — the decode path for + /// `DD_SKETCH_ENCODING_PROTO_DELTA` (paper §6.2 B3 / B4). + /// + /// Called against an accumulator that already carries the base + /// sketch state; the caller is the per-series snapshot cache in + /// the ingest path. Bytes are the + /// `asap_sketchlib::proto::sketchlib::DdSketchDelta` message. + pub fn apply_proto_delta_bytes( + &mut self, + buffer: &[u8], + ) -> Result<(), Box> { + use asap_sketchlib::proto::sketchlib::DdSketchDelta as PbDelta; + use prost::Message; + + let pb = PbDelta::decode(buffer).map_err(|e| format!("decode DDSketchDelta: {e}"))?; + + // The delta no longer carries d_count/d_sum/min/max + // (ProjectASAP/sketchlib-go#243 / asap_sketchlib#57). Apply the + // bucket deltas only; `DdSketch` recomputes its total count from + // the merged bucket counts (`total_count()`). + let buckets = pb + .buckets + .into_iter() + .map(|b| (b.index, b.d_count)) + .collect(); + let delta = DdSketchDelta { + buckets, + ..Default::default() + }; + self.inner + .apply_delta(&delta) + .map_err(|error| format!("apply DDSketchDelta: {error}"))?; + Ok(()) + } +} + +impl SerializableToSink for DDSketchAccumulator { + fn serialize_to_json(&self) -> Value { + // The DataPoint-level scalars (sum/min/max) are no longer carried + // by `DdSketch` (ProjectASAP/sketchlib-go#243 / asap_sketchlib#57). + // `count` is the bucket-derived total via `total_count()`. + serde_json::json!({ + "alpha": self.inner.alpha, + "store_offset": self.inner.store_offset, + "bucket_count": self.inner.store_counts.len(), + // Raw bucket-derived count (admitted samples). `sample_p` is the + // scale factor a consumer applies (count / sample_p) to estimate + // the true count; `query_statistic(Count)` already does this. + "count": self.inner.total_count(), + "sample_p": self.sample_p, + }) + } + + fn serialize_to_bytes(&self) -> Vec { + self.inner.to_msgpack().unwrap_or_default() + } +} + +impl AggregateCore for DDSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "DDSketchAccumulator" + } + + /// Per-window base rotation: drop all bucket counts but keep the + /// relative-accuracy parameter so the next window's bucket deltas + /// index into the same log-bucket layout. `sample_p` is a per-series + /// config constant (not per-window data), so it is intentionally + /// preserved across the rotation — the next window's deltas are sampled + /// at the same rate and must rescale identically. + fn reset_to_empty(&mut self) { + self.inner = DdSketch::new(self.inner.alpha); + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge DDSketchAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + let other_dd = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to DDSketchAccumulator")?; + let merged_inner = DdSketch::merge_refs(&[&self.inner, &other_dd.inner])?; + // sample_p is a per-series config constant, so both operands carry the + // same value in practice. Prefer a sampled factor over the no-sampling + // default so a merge with a freshly-reset (1.0) base keeps the series' + // sampling rate. + let sample_p = if self.sample_p < 1.0 { + self.sample_p + } else { + other_dd.sample_p + }; + Ok(Box::new(Self { + inner: merged_inner, + sample_p, + })) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::DDSketch + } + + fn get_keys(&self) -> Option> { + None + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + _key: &Option, + query_kwargs: &HashMap, + ) -> Result> { + use crate::Statistic; + + match statistic { + Statistic::Quantile => { + // PromQL `histogram_quantile(q, …)` and + // `quantile_over_time(q, …)` both land here with + // `q` in `query_kwargs["quantile"]`. Default to + // 0.99 when the caller didn't provide one + // (defensive — pattern-matched queries in + // `inference_config.yaml` always populate it). + let q: f64 = query_kwargs + .get("quantile") + .and_then(|s| s.parse().ok()) + .unwrap_or(0.99); + if !(0.0..=1.0).contains(&q) { + return Err(format!("DDSketchAccumulator: quantile {q} out of [0,1]").into()); + } + self.inner.quantile(q).ok_or_else(|| { + "DDSketchAccumulator: quantile() returned None (sketch empty?)".into() + }) + } + // Count is derived by summing the bucket store counts — the only + // DataPoint-level scalar that survives the wire-format trim + // (ProjectASAP/sketchlib-go#243 / asap_sketchlib#57). When the edge + // sampled this series (sample_p < 1.0), the stored count is ~p× the + // true count, so rescale by 1/sample_p to recover an unbiased + // estimate. sample_p == 1.0 (unsampled / legacy) makes this a no-op. + Statistic::Count => Ok(self.inner.total_count() as f64 / self.sample_p), + // STRICT policy: the Sum/Min/Max scalars were removed from + // the DDSketch wire format. They are now served by the + // controller-provisioned exact aggregations (an exact `Sum` + // and an exact `MinMax`), NOT estimated from the buckets. + // Surface the unavailable-statistic error so the query path + // routes to those aggregations instead of returning a wrong + // (0 / panicked) value. + Statistic::Sum => Err( + "DDSketchAccumulator: Sum not available from DDSketch wire format \ + (ProjectASAP/sketchlib-go#243); use an exact Sum aggregation" + .into(), + ), + Statistic::Min | Statistic::Max => Err(format!( + "DDSketchAccumulator: {statistic:?} not available from DDSketch wire format \ + (ProjectASAP/sketchlib-go#243); use an exact MinMax aggregation", + ) + .into()), + other => Err(format!( + "DDSketchAccumulator: statistic {other:?} not supported (only Quantile / Count)", + ) + .into()), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // The DataPoint-level METRIC scalars (count/sum/min/max) were dropped + // from `DdSketchState` (ProjectASAP/sketchlib-go#243 / + // asap_sketchlib#57); the proto now carries only + // `alpha`/`store_counts`/`store_offset`. + fn encode_state(alpha: f64, store_counts: Vec, store_offset: i32) -> Vec { + use asap_sketchlib::proto::sketchlib::{sketch_envelope, DdSketchState, SketchEnvelope}; + use prost::Message; + let state = DdSketchState { + alpha, + store_counts, + store_offset, + }; + SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Ddsketch(state)), + ..Default::default() + } + .encode_to_vec() + } + + #[test] + fn test_from_sketchlib_proto_bytes_round_trip() { + let bytes = encode_state(0.01, vec![1, 2, 3, 4], -2); + let acc = DDSketchAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + assert_eq!(acc.inner.alpha, 0.01); + assert_eq!(acc.inner.store_counts, vec![1, 2, 3, 4]); + assert_eq!(acc.inner.store_offset, -2); + // `count` is recovered by summing the bucket store counts. + assert_eq!(acc.inner.total_count(), 10); + } + + #[test] + fn test_from_sketchlib_proto_bytes_rejects_invalid_alpha() { + let bytes = encode_state(0.0, vec![1], 0); + let result = DDSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err()); + assert!(result.unwrap_err().to_string().contains("alpha")); + } + + #[test] + fn test_from_sketchlib_proto_bytes_envelope_wrapped() { + // Mirrors what DataCollector's ddsketchprocessor emits: the + // state wrapped in a `SketchEnvelope{ddsketch: ...}` via + // sketchlib-go's `SerializePortableFO` + `proto.Marshal`. + use asap_sketchlib::proto::sketchlib::{sketch_envelope, DdSketchState, SketchEnvelope}; + use prost::Message; + + let state = DdSketchState { + alpha: 0.01, + store_counts: vec![1, 2, 3, 4], + store_offset: -2, + }; + let env = SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Ddsketch(state)), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + + let acc = DDSketchAccumulator::from_sketchlib_proto_bytes(&bytes) + .expect("envelope-wrapped decode should succeed"); + assert_eq!(acc.inner.alpha, 0.01); + assert_eq!(acc.inner.total_count(), 10); + } + + #[test] + fn test_from_sketchlib_proto_bytes_envelope_wrong_sketch_type() { + use asap_sketchlib::proto::sketchlib::{sketch_envelope, KllState, SketchEnvelope}; + use prost::Message; + + let env = SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Kll(KllState::default())), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + + let result = DDSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err(), "wrong-sketch envelope should error"); + } + + #[test] + fn test_aggregate_core_merge_aligns_buckets() { + let a = DDSketchAccumulator { + inner: DdSketch::from_raw(0.01, vec![1, 1, 1], -1), + sample_p: 1.0, + }; + let b = DDSketchAccumulator { + inner: DdSketch::from_raw(0.01, vec![10, 10, 10], 0), + sample_p: 1.0, + }; + let merged_box = a.merge_with(&b).expect("merge ok"); + let merged = merged_box + .as_any() + .downcast_ref::() + .expect("downcast ok"); + assert_eq!(merged.inner.store_counts, vec![1, 11, 11, 10]); + assert_eq!(merged.inner.store_offset, -1); + assert_eq!(merged.inner.total_count(), 33); + } + + #[test] + fn test_aggregate_core_merge_wrong_type_rejects() { + use crate::accumulators::count_sketch_accumulator::CountSketchAccumulator; + let dd = DDSketchAccumulator::new(0.01); + let cs = CountSketchAccumulator::new(2, 3); + assert!(dd.merge_with(&cs).is_err()); + } + + #[test] + fn test_from_msgpack_bytes_round_trip() { + let original = DdSketch::from_raw(0.01, vec![5, 10, 15, 20], -2); + let bytes = original.to_msgpack().unwrap(); + let acc = DDSketchAccumulator::from_msgpack_bytes(&bytes).expect("decode ok"); + assert_eq!(acc.inner.alpha, 0.01); + assert_eq!(acc.inner.store_counts, vec![5, 10, 15, 20]); + assert_eq!(acc.inner.store_offset, -2); + // `count` is recovered by summing the bucket store counts. + assert_eq!(acc.inner.total_count(), 50); + } + + #[test] + fn test_from_msgpack_bytes_rejects_garbage() { + let result = DDSketchAccumulator::from_msgpack_bytes(b"not valid msgpack"); + assert!(result.is_err()); + } + + #[test] + fn test_apply_proto_delta_bytes_round_trip() { + use asap_sketchlib::proto::sketchlib::{DdSketchBucketDelta, DdSketchDelta as PbDelta}; + use prost::Message; + + let mut acc = DDSketchAccumulator::new(0.01); + acc.inner = DdSketch::from_raw(0.01, vec![1, 2, 3], 0); + + // The wire delta now carries only bucket deltas (tags 2-7 + // reserved); `DdSketchBucketDelta` has just `index` + `d_count`. + let bytes = PbDelta { + buckets: vec![ + DdSketchBucketDelta { + index: 0, + d_count: 10, + }, + DdSketchBucketDelta { + index: 2, + d_count: 20, + }, + ], + } + .encode_to_vec(); + + acc.apply_proto_delta_bytes(&bytes).expect("apply ok"); + assert_eq!(acc.inner.store_counts, vec![11, 2, 23]); + // `count` recomputed from the merged buckets: 11 + 2 + 23 = 36. + assert_eq!(acc.inner.total_count(), 36); + } + + /// A valid protobuf with an inadmissible span must not acknowledge a dropped update. + #[test] + fn test_apply_proto_delta_rejects_span_without_mutating_state() { + use asap_sketchlib::proto::sketchlib::{DdSketchBucketDelta, DdSketchDelta as PbDelta}; + use prost::Message; + let mut acc = DDSketchAccumulator::new(0.01); + acc.inner = DdSketch::from_raw(0.01, vec![1, 2, 3], 0); + let bytes = PbDelta { + buckets: vec![DdSketchBucketDelta { + index: i32::MAX, + d_count: 1, + }], + } + .encode_to_vec(); + assert!(acc.apply_proto_delta_bytes(&bytes).is_err()); + assert_eq!(acc.inner.store_counts, vec![1, 2, 3]); + assert_eq!(acc.inner.store_offset, 0); + } + + #[test] + fn test_apply_proto_delta_bytes_rejects_garbage() { + let mut acc = DDSketchAccumulator::new(0.01); + assert!(acc.apply_proto_delta_bytes(b"not valid proto").is_err()); + } + + // ----- query_statistic STRICT policy ----- + // + // After the DataPoint-level METRIC scalars were dropped from the + // DDSketch wire format (ProjectASAP/sketchlib-go#243 / + // asap_sketchlib#57), DDSketch serves only quantiles and Count. + // Sum/Min/Max move to controller-provisioned exact aggregations and + // MUST surface the unavailable-statistic error (never a panic / 0). + + fn sample_accumulator() -> DDSketchAccumulator { + // Build the in-memory sketch from bucket counts only — no scalars. + DDSketchAccumulator { + inner: DdSketch::from_raw(0.01, vec![1, 2, 3, 4], -2), + sample_p: 1.0, + } + } + + #[test] + fn test_query_statistic_quantile_is_sketch_derived() { + use crate::Statistic; + let acc = sample_accumulator(); + let mut kwargs = HashMap::new(); + kwargs.insert("quantile".to_string(), "0.5".to_string()); + let v = acc + .query_statistic(Statistic::Quantile, &None, &kwargs) + .expect("quantile should be served from the sketch buckets"); + assert!( + v.is_finite() && v > 0.0, + "quantile estimate should be positive finite, got {v}" + ); + } + + #[test] + fn test_query_statistic_count_is_bucket_derived() { + use crate::Statistic; + let acc = sample_accumulator(); + let v = acc + .query_statistic(Statistic::Count, &None, &HashMap::new()) + .expect("count should be derivable from the bucket store"); + // 1 + 2 + 3 + 4 = 10. + assert_eq!(v, 10.0); + } + + #[test] + fn test_query_statistic_sum_min_max_return_unavailable_error() { + use crate::Statistic; + let acc = sample_accumulator(); + for stat in [Statistic::Sum, Statistic::Min, Statistic::Max] { + let result = acc.query_statistic(stat, &None, &HashMap::new()); + assert!( + result.is_err(), + "{stat:?} must return the unavailable-statistic error (not a panic / 0)" + ); + let msg = result.unwrap_err().to_string(); + assert!( + msg.contains("not available"), + "{stat:?} error should explain the statistic is unavailable, got: {msg}" + ); + } + } + + // ----- sample_p count rescale ----- + // + // When the edge sampled a DDSketch (sample_p < 1.0), the stored count is + // ~p× the true count, so Count rescales by 1/p. Quantiles are + // rank-preserving and must NOT be rescaled. + + #[test] + fn test_count_is_rescaled_by_sample_p() { + use crate::Statistic; + let acc = DDSketchAccumulator { + inner: DdSketch::from_raw(0.01, vec![1, 2, 3, 4], -2), + sample_p: 0.1, + }; + let c = acc + .query_statistic(Statistic::Count, &None, &HashMap::new()) + .expect("count ok"); + // Raw bucket sum 10, rescaled by 1/0.1 = 100. + assert!((c - 100.0).abs() < 1e-9, "expected rescaled 100, got {c}"); + } + + #[test] + fn test_quantile_ignores_sample_p() { + use crate::Statistic; + let mut kwargs = HashMap::new(); + kwargs.insert("quantile".to_string(), "0.5".to_string()); + let unsampled = DDSketchAccumulator { + inner: DdSketch::from_raw(0.01, vec![1, 2, 3, 4], -2), + sample_p: 1.0, + }; + let sampled = DDSketchAccumulator { + inner: DdSketch::from_raw(0.01, vec![1, 2, 3, 4], -2), + sample_p: 0.1, + }; + let qu = unsampled + .query_statistic(Statistic::Quantile, &None, &kwargs) + .expect("q ok"); + let qs = sampled + .query_statistic(Statistic::Quantile, &None, &kwargs) + .expect("q ok"); + assert_eq!(qu, qs, "quantile must be sample_p-invariant"); + } + + #[test] + fn test_from_sketchlib_proto_bytes_reads_envelope_sample_p() { + use crate::Statistic; + use asap_sketchlib::proto::sketchlib::{sketch_envelope, DdSketchState, SketchEnvelope}; + use prost::Message; + + let env = SketchEnvelope { + sample_p: 0.25, + sketch_state: Some(sketch_envelope::SketchState::Ddsketch(DdSketchState { + alpha: 0.01, + store_counts: vec![2, 4, 6, 8], + store_offset: -2, + })), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + let acc = DDSketchAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + assert_eq!(acc.sample_p, 0.25); + // Raw 20, rescaled 20 / 0.25 = 80. + let c = acc + .query_statistic(Statistic::Count, &None, &HashMap::new()) + .expect("count ok"); + assert!((c - 80.0).abs() < 1e-9, "expected rescaled 80, got {c}"); + } + + #[test] + fn test_sample_p_normalization() { + // proto3 default (0.0), >=1.0, and non-finite all mean no sampling. + assert_eq!(normalize_sample_p(0.0), 1.0); + assert_eq!(normalize_sample_p(1.0), 1.0); + assert_eq!(normalize_sample_p(1.5), 1.0); + assert_eq!(normalize_sample_p(f64::NAN), 1.0); + assert_eq!(normalize_sample_p(-0.1), 1.0); + assert_eq!(normalize_sample_p(0.5), 0.5); + } + + #[test] + fn test_sample_p_from_envelope_bytes_defaults_to_one() { + use asap_sketchlib::proto::sketchlib::DdSketchState; + use prost::Message; + // Bare DdSketchState bytes (no envelope) → no sampling info → 1.0. + let bare = DdSketchState { + alpha: 0.01, + store_counts: vec![1, 2, 3], + store_offset: 0, + } + .encode_to_vec(); + assert_eq!( + DDSketchAccumulator::sample_p_from_envelope_bytes(&bare), + 1.0 + ); + } + + #[test] + fn test_reset_to_empty_preserves_sample_p() { + let mut acc = DDSketchAccumulator { + inner: DdSketch::from_raw(0.01, vec![1, 2, 3], 0), + sample_p: 0.2, + }; + acc.reset_to_empty(); + assert_eq!(acc.sample_p, 0.2, "window rotation must keep sample_p"); + assert_eq!(acc.inner.total_count(), 0, "buckets cleared"); + } + + #[test] + fn test_merge_prefers_sampled_factor() { + // A sampled base merged with a freshly-reset (1.0) operand keeps the + // series' sampling rate. + let a = DDSketchAccumulator { + inner: DdSketch::from_raw(0.01, vec![1, 1, 1], 0), + sample_p: 0.1, + }; + let b = DDSketchAccumulator { + inner: DdSketch::from_raw(0.01, vec![1, 1, 1], 0), + sample_p: 1.0, + }; + let merged = a.merge_with(&b).expect("merge ok"); + let merged = merged + .as_any() + .downcast_ref::() + .expect("downcast ok"); + assert_eq!(merged.sample_p, 0.1); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/exact_accumulator.rs b/crates/asap-physical-operators/src/accumulators/exact_accumulator.rs new file mode 100644 index 00000000..c466548e --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/exact_accumulator.rs @@ -0,0 +1,326 @@ +//! Exact summary state identified by Planner family, independent of keyed layout. +use super::increase_accumulator::IncreaseAccumulator; +use crate::Statistic; +use crate::{ + AggregateCore, AggregationType, AuxStats, KeyByLabelValues, Measurement, SerializableToSink, +}; +use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; +use serde::{Deserialize, Serialize}; +use std::collections::HashMap; + +type Error = Box; + +#[derive(Debug, Clone, Serialize, Deserialize)] +enum ScalarState { + Sum(f64), + Count(u64), + Min(Option), + Max(Option), + Counter(Option), +} + +/// Both the family and population layout survive persistence. Sharing counter +/// arithmetic never authorizes a Rate state to answer an Increase readout. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ExactAccumulator { + family: SummaryFamilyType, + scalar: ScalarState, + keyed: Option>, +} + +impl ExactAccumulator { + pub fn new(family: SummaryFamilyType, keyed: bool) -> Result { + use ExactKind as K; + use ExactParams as P; + let scalar = match &family { + SummaryFamilyType::ExactAggregate(K::Sum, P::Sum) => ScalarState::Sum(0.0), + SummaryFamilyType::ExactAggregate(K::Count, P::Count) => ScalarState::Count(0), + SummaryFamilyType::ExactAggregate(K::Min, P::Min) => ScalarState::Min(None), + SummaryFamilyType::ExactAggregate(K::Max, P::Max) => ScalarState::Max(None), + SummaryFamilyType::ExactAggregate(K::Rate, P::Rate) + | SummaryFamilyType::ExactAggregate(K::Increase, P::Increase) => { + ScalarState::Counter(None) + } + _ => return Err(format!("unsupported exact Planner family: {family:?}")), + }; + Ok(Self { + family, + scalar, + keyed: keyed.then(HashMap::new), + }) + } + + pub fn family(&self) -> &SummaryFamilyType { + &self.family + } + pub fn is_keyed(&self) -> bool { + self.keyed.is_some() + } + + pub fn update(&mut self, key: Option<&KeyByLabelValues>, value: f64, timestamp: i64) { + let state = match (&mut self.keyed, key) { + (Some(states), Some(key)) => states + .entry(key.clone()) + .or_insert_with(|| self.scalar.clone()), + (None, None) => &mut self.scalar, + _ => panic!("exact update population layout differs from installed DAG"), + }; + match state { + ScalarState::Sum(sum) => *sum += value, + ScalarState::Count(count) => { + *count = count.checked_add(1).expect("exact count overflow") + } + ScalarState::Min(current) => { + *current = Some(current.map_or(value, |old| old.min(value))) + } + ScalarState::Max(current) => { + *current = Some(current.map_or(value, |old| old.max(value))) + } + ScalarState::Counter(current) => match current { + Some(counter) => counter.update(Measurement::new(value), timestamp), + None => { + *current = Some(IncreaseAccumulator::new( + Measurement::new(value), + timestamp, + Measurement::new(value), + timestamp, + )) + } + }, + } + } + + pub fn deserialize_from_bytes(bytes: &[u8]) -> Result { + let state: Self = rmp_serde::from_slice(bytes)?; + let expected = Self::new(state.family.clone(), state.is_keyed())?; + let same_variant = |value: &ScalarState| { + std::mem::discriminant(value) == std::mem::discriminant(&expected.scalar) + }; + if !same_variant(&state.scalar) + || state + .keyed + .as_ref() + .is_some_and(|states| states.values().any(|s| !same_variant(s))) + { + return Err("exact payload differs from declared Planner family".into()); + } + Ok(state) + } + + fn statistic(&self) -> Statistic { + match self.family { + SummaryFamilyType::ExactAggregate(ExactKind::Sum, _) => Statistic::Sum, + SummaryFamilyType::ExactAggregate(ExactKind::Count, _) => Statistic::Count, + SummaryFamilyType::ExactAggregate(ExactKind::Min, _) => Statistic::Min, + SummaryFamilyType::ExactAggregate(ExactKind::Max, _) => Statistic::Max, + SummaryFamilyType::ExactAggregate(ExactKind::Rate, _) => Statistic::Rate, + SummaryFamilyType::ExactAggregate(ExactKind::Increase, _) => Statistic::Increase, + _ => unreachable!("validated exact family"), + } + } +} + +fn merge_scalar(left: &ScalarState, right: &ScalarState) -> Result { + Ok(match (left, right) { + (ScalarState::Sum(a), ScalarState::Sum(b)) => ScalarState::Sum(a + b), + (ScalarState::Count(a), ScalarState::Count(b)) => { + ScalarState::Count(a.checked_add(*b).ok_or("exact count overflow")?) + } + (ScalarState::Min(a), ScalarState::Min(b)) => { + ScalarState::Min(a.iter().chain(b).copied().reduce(f64::min)) + } + (ScalarState::Max(a), ScalarState::Max(b)) => { + ScalarState::Max(a.iter().chain(b).copied().reduce(f64::max)) + } + (ScalarState::Counter(a), ScalarState::Counter(b)) => ScalarState::Counter(match (a, b) { + (Some(a), Some(b)) => Some( + >::merge_accumulators(vec![ + a.clone(), + b.clone(), + ])?, + ), + (a, b) => a.clone().or_else(|| b.clone()), + }), + _ => return Err("exact scalar state families differ".into()), + }) +} + +impl SerializableToSink for ExactAccumulator { + fn serialize_to_json(&self) -> serde_json::Value { + serde_json::json!({"family": self.family, "scalar": self.scalar, "keyed": self.keyed.as_ref().map(|m|m.iter().collect::>())}) + } + fn serialize_to_bytes(&self) -> Vec { + rmp_serde::to_vec_named(self).expect("exact state encoding") + } +} + +impl AggregateCore for ExactAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + fn type_name(&self) -> &'static str { + "PlannerExactAccumulatorV1" + } + fn as_any(&self) -> &dyn std::any::Any { + self + } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + fn merge_with(&self, other: &dyn AggregateCore) -> Result, Error> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("merge requires Planner exact state")?; + if self.family != other.family || self.is_keyed() != other.is_keyed() { + return Err("cannot merge different Planner families or layouts".into()); + } + let mut merged = self.clone(); + if let (Some(target), Some(source)) = (&mut merged.keyed, &other.keyed) { + for (key, state) in source { + let combined = match target.get(key) { + Some(old) => merge_scalar(old, state)?, + None => state.clone(), + }; + target.insert(key.clone(), combined); + } + } else { + merged.scalar = merge_scalar(&self.scalar, &other.scalar)?; + } + Ok(Box::new(merged)) + } + fn get_accumulator_type(&self) -> AggregationType { + match self.statistic() { + Statistic::Sum => AggregationType::Sum, + Statistic::Count => AggregationType::Count, + Statistic::Min => AggregationType::Min, + Statistic::Max => AggregationType::Max, + Statistic::Rate => AggregationType::Rate, + Statistic::Increase => AggregationType::Increase, + _ => unreachable!(), + } + } + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::() + + self.keyed.as_ref().map_or(0, |m| { + m.keys() + .map(|k| { + std::mem::size_of::() + + k.labels.iter().map(String::len).sum::() + }) + .sum::() + }) + } + fn aux_stats(&self) -> AuxStats { + if self.is_keyed() { + return AuxStats::empty(); + } + match self.scalar { + ScalarState::Sum(value) => AuxStats { + sum: Some(value), + ..AuxStats::empty() + }, + ScalarState::Count(value) => AuxStats { + count: Some(value), + ..AuxStats::empty() + }, + ScalarState::Min(value) => AuxStats { + min: value, + ..AuxStats::empty() + }, + ScalarState::Max(value) => AuxStats { + max: value, + ..AuxStats::empty() + }, + ScalarState::Counter(_) => AuxStats::empty(), + } + } + fn get_keys(&self) -> Option> { + self.keyed.as_ref().map(|m| m.keys().cloned().collect()) + } + fn query_statistic( + &self, + statistic: Statistic, + key: &Option, + kwargs: &HashMap, + ) -> Result { + if statistic != self.statistic() { + return Err("readout differs from Planner exact family".into()); + } + let state = match (&self.keyed, key) { + (Some(states), Some(key)) => states.get(key).ok_or("unknown exact population")?, + (None, None) => &self.scalar, + _ => return Err("readout population differs from installed layout".into()), + }; + match state { + ScalarState::Sum(sum) => Ok(*sum), + ScalarState::Count(count) => Ok(*count as f64), + ScalarState::Min(value) | ScalarState::Max(value) => { + value.ok_or_else(|| "empty exact population".into()) + } + ScalarState::Counter(Some(counter)) => { + counter.query_statistic(statistic, &None, kwargs) + } + ScalarState::Counter(None) => Err("empty counter population".into()), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Identity, population isolation, and readout survive the persisted format. + #[test] + fn exact_families_roundtrip_and_reject_cross_family_operations() { + let families = [ + (ExactKind::Sum, ExactParams::Sum, Statistic::Sum, 16.0), + (ExactKind::Count, ExactParams::Count, Statistic::Count, 3.0), + (ExactKind::Min, ExactParams::Min, Statistic::Min, 2.0), + (ExactKind::Max, ExactParams::Max, Statistic::Max, 8.0), + (ExactKind::Rate, ExactParams::Rate, Statistic::Rate, 3.0), + ( + ExactKind::Increase, + ExactParams::Increase, + Statistic::Increase, + 6.0, + ), + ]; + for keyed in [false, true] { + let key = keyed.then(|| KeyByLabelValues::new_with_labels(vec!["a".into()])); + let mut states = Vec::new(); + for (kind, params, stat, value) in &families { + let mut state = ExactAccumulator::new( + SummaryFamilyType::ExactAggregate(kind.clone(), params.clone()), + keyed, + ) + .unwrap(); + for (ts, v) in [(1000, 8.0), (2000, 2.0), (3000, 6.0)] { + state.update(key.as_ref(), v, ts); + } + let restored = + ExactAccumulator::deserialize_from_bytes(&state.serialize_to_bytes()).unwrap(); + assert_eq!(restored.family(), state.family()); + assert_eq!( + restored + .query_statistic(*stat, &key, &HashMap::new()) + .unwrap(), + *value + ); + for (_, _, wrong, _) in &families { + if wrong != stat { + assert!(restored + .query_statistic(*wrong, &key, &HashMap::new()) + .is_err()); + } + } + states.push(restored); + } + for (i, a) in states.iter().enumerate() { + for (j, b) in states.iter().enumerate() { + assert_eq!(a.merge_with(b).is_ok(), i == j); + } + } + } + } +} diff --git a/crates/asap-physical-operators/src/accumulators/hll_sketch_accumulator.rs b/crates/asap-physical-operators/src/accumulators/hll_sketch_accumulator.rs new file mode 100644 index 00000000..d43737a3 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/hll_sketch_accumulator.rs @@ -0,0 +1,788 @@ +//! HLL accumulator — wraps `asap_sketchlib::HllSketch`. +//! +//! Concrete accumulator reached from the modified-OTLP +//! `Metric.data = HLLSketch{…}` hot path (PR C-CountSketch follow-up). +//! Mirrors the CountSketch accumulator's shape: merge via register-wise +//! max on the inner sketch, serialize as MessagePack for the sink, and +//! decode from the sketchlib `HyperLogLogState` proto. +//! +//! Query semantics (cardinality estimation via the three HLL variants' +//! estimators) are intentionally deferred — the wire format carries the +//! registers + variant + HIP accumulators losslessly, so the merge + +//! store round-trip works end-to-end without that richer query surface. + +use crate::accumulators::dd_sketch_accumulator::normalize_sample_p; +use crate::{AggregateCore, AggregationType, KeyByLabelValues, SerializableToSink}; +use asap_sketchlib::{HllSketch, HllVariant, MessagePackCodec}; +use serde_json::Value; +use std::collections::HashMap; + +/// Decode one protobuf base-128 varint (LEB128) from the front of `buf`. +/// Returns `(value, bytes_consumed)`, or `None` if the buffer is truncated +/// or the varint overflows u64. +pub(crate) fn read_uvarint(buf: &[u8]) -> Option<(u64, usize)> { + let mut result: u64 = 0; + let mut shift: u32 = 0; + for (i, &b) in buf.iter().enumerate() { + if shift >= 64 { + return None; + } + result |= u64::from(b & 0x7f) << shift; + if b & 0x80 == 0 { + return Some((result, i + 1)); + } + shift += 7; + } + None +} + +/// Expand sketchlib-go's sparse HLL register encoding +/// (`HLLSparseRegisters.packed`) into the dense `num_registers`-byte array. +/// +/// Layout (sketchlib-go `proto/hll/hll.proto`): varint-packed +/// `(index_delta, value)` pairs in ascending index order; `prev_index` +/// starts at 0, so each register's absolute index is the running sum of the +/// deltas. Mirrors the Go encoder in `sketches/HLL/sparse.go` +/// (`encodeSparseRegisters`). The reconstructed array is byte-identical to +/// the dense `registers` field a high-cardinality producer would have sent. +pub(crate) fn expand_sparse_hll_registers( + packed: &[u8], + num_registers: usize, +) -> Result, Box> { + let mut regs = vec![0u8; num_registers]; + let mut prev: u64 = 0; + let mut pos = 0usize; + while pos < packed.len() { + let (delta, n1) = read_uvarint(&packed[pos..]) + .ok_or("HLLSparseRegisters.packed: truncated index_delta varint")?; + pos += n1; + let (value, n2) = read_uvarint(&packed[pos..]) + .ok_or("HLLSparseRegisters.packed: truncated value varint")?; + pos += n2; + let idx = prev + delta; + let i = usize::try_from(idx) + .map_err(|_| format!("HLLSparseRegisters: index {idx} overflows usize"))?; + if i >= num_registers { + return Err(format!( + "HLLSparseRegisters: register index {i} >= num_registers {num_registers}" + ) + .into()); + } + regs[i] = u8::try_from(value) + .map_err(|_| format!("HLLSparseRegisters: register value {value} > 255"))?; + prev = idx; + } + Ok(regs) +} + +/// HLL accumulator — inner register array + variant metadata. +#[derive(Debug, Clone)] +pub struct HllSketchAccumulator { + pub inner: HllSketch, + /// Edge sampling probability `p ∈ (0,1]` carried on the producer's + /// `SketchEnvelope.sample_p`. HLL uses HASH-THRESHOLD sampling — each + /// DISTINCT key is admitted into the sketch with probability `p`, so the + /// register-derived distinct-count estimate is ~`p`× the true + /// cardinality and a `Cardinality`/`Count` query must rescale by `1/p`. + /// `1.0` (and the proto3 default `0.0`, dual-read as `1.0`) means no + /// sampling, so the rescale is a no-op and the behaviour is identical to + /// before. Mirrors `DDSketchAccumulator::sample_p`; set from the envelope + /// at the `from_sketchlib_proto_bytes` decode site and preserved across + /// `reset_to_empty` and `merge_with`. + /// + /// NOTE: HLL edge sampling is currently force-disabled in the edge + /// (`warm_sketch.go` HLL case always emits `sample_p = 1.0`), so in + /// practice `p = 1.0` today and this is a latent-correctness fix that + /// activates if HLL sampling is ever enabled. + pub sample_p: f64, +} + +impl HllSketchAccumulator { + pub fn new(variant: HllVariant, precision: u32) -> Self { + Self { + inner: HllSketch::new(variant, precision), + sample_p: 1.0, + } + } + + /// Decode from the modified OTLP wire format's + /// `HLLSketchDataPoint.sketch` bytes when + /// `encoding = HLL_SKETCH_ENCODING_MSGPACK`. The bytes are the + /// MessagePack serialization of the cross-language sketch-core + /// `HllSketch` struct — PR I parity entrypoint. + pub fn from_msgpack_bytes(buffer: &[u8]) -> Result> { + Ok(Self { + inner: HllSketch::from_msgpack(buffer) + .map_err(|e| format!("deserialize HllSketch msgpack: {e}"))?, + // The msgpack HllSketch struct carries no envelope/sample_p; the + // msgpack path is parity/test-only and is never edge-sampled. + sample_p: 1.0, + }) + } + + /// Decode from the modified OTLP wire format's + /// `HLLSketchDataPoint.sketch` bytes — the protobuf-encoded + /// `asap_sketchlib::proto::sketchlib::HyperLogLogState` message + /// that DataCollector's `hllprocessor` emits when + /// `encoding = HLL_SKETCH_ENCODING_PROTO`. + pub fn from_sketchlib_proto_bytes(buffer: &[u8]) -> Result> { + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, HllVariant as ProtoVariant, HyperLogLogState, SketchEnvelope, + }; + use prost::Message; + + // DataCollector's hllprocessor wraps the state in a + // `SketchEnvelope{hll: HyperLogLogState}` via sketchlib-go's + // `SerializePortableFO` + `proto.Marshal`. Try envelope first, + // fall back to bare `HyperLogLogState` for callers (e.g. unit + // tests) that encode the state directly. Mirrors the PR #14 + // fix on `CountMinSketchAccumulator::from_sketchlib_proto_bytes`. + // Capture the envelope's `sample_p` alongside the state so a + // Cardinality query can rescale the distinct-count estimate by + // `1/p`. Bare `HyperLogLogState` bytes (no envelope) carry no + // sampling info → `sample_p` 1.0 (no rescale). Mirrors + // `DDSketchAccumulator`. + let (state, sample_p) = match SketchEnvelope::decode(buffer) { + Ok(env) => { + let sp = env.sample_p; + match env.sketch_state { + Some(sketch_envelope::SketchState::Hll(st)) => (st, sp), + Some(other) => { + return Err(format!( + "SketchEnvelope contains non-HLL sketch: {:?}", + std::mem::discriminant(&other) + ) + .into()); + } + None => ( + HyperLogLogState::decode(buffer) + .map_err(|e| format!("decode HyperLogLogState: {e}"))?, + 1.0, + ), + } + } + Err(_) => ( + HyperLogLogState::decode(buffer) + .map_err(|e| format!("decode HyperLogLogState: {e}"))?, + 1.0, + ), + }; + if state.precision == 0 || state.precision > 20 { + return Err(format!( + "HyperLogLogState precision {} out of range (expected 1..=20)", + state.precision + ) + .into()); + } + let expected_len = 1usize << state.precision; + // Register resolution. sketchlib-go emits the SPARSE + // `registers_sparse` (proto tag 7) form below its dense/sparse + // crossover (~6000 non-zero registers — see + // sketchlib-go/sketches/HLL/sparse.go); low-cardinality producers + // (the common case) therefore leave the dense `registers` (tag 3) + // field empty. The proto contract (hll.proto) is: read whichever of + // `registers` / `registers_sparse` is present; if both are empty the + // sketch is all-zero. Reconstruct the dense 2^precision array in all + // three cases so the inner `HllSketch` always gets a full register + // vector. + let dense_registers: Vec = if state.registers.len() == expected_len { + state.registers.clone() + } else if !state.registers.is_empty() { + // A non-empty dense field of the wrong length is a malformed frame. + return Err(format!( + "HyperLogLogState registers has {} bytes, expected 2^precision = {}", + state.registers.len(), + expected_len + ) + .into()); + } else if let Some(sparse) = state.registers_sparse.as_ref() { + expand_sparse_hll_registers(&sparse.packed, expected_len)? + } else { + // Neither representation populated → all-zero register array. + vec![0u8; expected_len] + }; + let proto_variant = ProtoVariant::try_from(state.variant) + .map_err(|_| format!("HyperLogLogState has unknown variant tag {}", state.variant))?; + let variant = match proto_variant { + ProtoVariant::Unspecified => HllVariant::Unspecified, + ProtoVariant::Regular => HllVariant::Regular, + ProtoVariant::ErtlMle => HllVariant::Datafusion, + ProtoVariant::Hip => HllVariant::Hip, + }; + let inner = HllSketch::from_raw( + variant, + state.precision, + dense_registers, + state.hip_kxq0, + state.hip_kxq1, + state.hip_est, + ); + Ok(Self { + inner, + sample_p: normalize_sample_p(sample_p), + }) + } + + /// Apply a proto-encoded `HLLDelta` frame to this accumulator's + /// inner sketch — the decode path for + /// `HLL_SKETCH_ENCODING_PROTO_DELTA` (paper §6.2 B3 / B4). + /// + /// Called against an accumulator that already carries the base + /// sketch state; the caller is the per-series snapshot cache in + /// the ingest path. Bytes are the + /// `asap_sketchlib::proto::sketchlib::HllDelta` message. + pub fn apply_proto_delta_bytes( + &mut self, + buffer: &[u8], + ) -> Result<(), Box> { + // The HLLDelta wire format is a varint-packed (index_delta, value) blob; + // decode + apply (register-wise max) via the shared sketch library so + // the unpacking stays a single source of truth. + self.inner + .apply_delta_bytes(buffer) + .map_err(|e| format!("apply HLLDelta: {e}"))?; + Ok(()) + } +} + +impl SerializableToSink for HllSketchAccumulator { + fn serialize_to_json(&self) -> Value { + serde_json::json!({ + "variant": format!("{:?}", self.inner.variant), + "precision": self.inner.precision, + "register_bytes": self.inner.registers.len(), + "hip_kxq0": self.inner.hip_kxq0, + "hip_kxq1": self.inner.hip_kxq1, + "hip_est": self.inner.hip_est, + }) + } + + fn serialize_to_bytes(&self) -> Vec { + self.inner.to_msgpack().unwrap_or_default() + } +} + +impl AggregateCore for HllSketchAccumulator { + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::().saturating_add(self.inner.registers.capacity()) + } + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "HllSketchAccumulator" + } + + /// Per-window base rotation: zero the registers but keep the variant + /// and precision. Critical for HLL — its register-wise `max` merge + /// has no inverse, so a never-reset base accumulates the all-time-max + /// across windows (`docs/delta-baseline-contract.md` §1.5); rotating + /// to an empty register array makes per-window cardinality correct. + /// `sample_p` is a per-series config constant (not per-window data), so + /// it is intentionally preserved across the rotation — mirrors + /// `DDSketchAccumulator`. + fn reset_to_empty(&mut self) { + self.inner = HllSketch::new(self.inner.variant, self.inner.precision); + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge HllSketchAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + let other_hll = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to HllSketchAccumulator")?; + let merged_inner = HllSketch::merge_refs(&[&self.inner, &other_hll.inner])?; + // Mirror DDSketchAccumulator's merge policy exactly: sample_p is a + // per-series config constant, so both operands carry the same value + // in practice. Prefer a sampled factor over the no-sampling default + // so a merge with a freshly-reset (1.0) base keeps the series' + // sampling rate. + let sample_p = if self.sample_p < 1.0 { + self.sample_p + } else { + other_hll.sample_p + }; + Ok(Box::new(Self { + inner: merged_inner, + sample_p, + })) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::HLL + } + + fn get_keys(&self) -> Option> { + None + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + _key: &Option, + _query_kwargs: &HashMap, + ) -> Result> { + use crate::Statistic; + match statistic { + // HLL's natural answer is unique-cardinality. PromQL's + // `count_over_time(...)` and `count(...)` both surface + // as `Statistic::Count` after pattern matching but + // semantically they mean "how many distinct values + // were observed in this window" when the underlying + // aggregator is HLL — that's the cardinality estimate, + // not a sample-count. Accept both. + Statistic::Cardinality | Statistic::Count => { + // HLL uses hash-threshold sampling — each distinct key is + // admitted with probability `sample_p`, so the register- + // derived distinct-count estimate is ~`p`× the true + // cardinality. Rescale by `1/sample_p` for an unbiased + // estimate. `sample_p == 1.0` (unsampled / legacy / edge + // HLL sampling currently force-disabled) makes this a no-op. + Ok(hll_cardinality_estimate(&self.inner.registers) / self.sample_p) + } + other => Err(format!( + "HllSketchAccumulator: statistic {:?} not supported (only Cardinality / Count)", + other, + ) + .into()), + } + } +} + +/// Standard HyperLogLog cardinality estimate with the canonical +/// `α_m × m² / Σ 2^(-register[i])` formula plus the small-range +/// (linear-counting) and large-range (32-bit space) corrections +/// from the original Flajolet et al. paper. +/// +/// Inlined here rather than added as a method on `asap_sketchlib::HllSketch` +/// because the existing `asap_sketchlib::asap` types only expose merge / +/// serialize today; adding a query method there would force a +/// cross-crate change. +fn hll_cardinality_estimate(registers: &[u8]) -> f64 { + let m = registers.len() as f64; + if m == 0.0 { + return 0.0; + } + let alpha = match registers.len() { + 16 => 0.673, + 32 => 0.697, + 64 => 0.709, + _ => 0.7213 / (1.0 + 1.079 / m), + }; + + let mut sum = 0.0f64; + let mut zero_registers = 0usize; + for &r in registers { + sum += 2f64.powi(-(r as i32)); + if r == 0 { + zero_registers += 1; + } + } + let raw = alpha * m * m / sum; + + // Small-range (linear-counting) correction. + if raw <= 2.5 * m && zero_registers > 0 { + return m * (m / zero_registers as f64).ln(); + } + + // Large-range correction (only meaningful with 32-bit register + // spaces; sketch-core uses up to 64-bit hashes so this branch + // rarely fires in practice — kept for completeness). + let two_pow_32 = 4_294_967_296f64; + if raw > two_pow_32 / 30.0 { + return -two_pow_32 * (1.0 - raw / two_pow_32).ln(); + } + raw +} + +#[cfg(test)] +mod tests { + use super::*; + + fn encode_state( + variant: i32, + precision: u32, + registers: Vec, + hip_kxq0: f64, + hip_kxq1: f64, + hip_est: f64, + ) -> Vec { + use asap_sketchlib::proto::sketchlib::HyperLogLogState; + use prost::Message; + let state = HyperLogLogState { + variant, + precision, + registers, + hip_kxq0, + hip_kxq1, + hip_est, + registers_sparse: None, + }; + state.encode_to_vec() + } + + #[test] + fn test_from_sketchlib_proto_bytes_regular() { + use asap_sketchlib::proto::sketchlib::HllVariant as ProtoVariant; + let bytes = encode_state( + ProtoVariant::Regular as i32, + 2, + vec![1, 2, 3, 4], + 0.0, + 0.0, + 0.0, + ); + let acc = HllSketchAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + assert_eq!(acc.inner.variant, HllVariant::Regular); + assert_eq!(acc.inner.precision, 2); + assert_eq!(acc.inner.registers, vec![1, 2, 3, 4]); + } + + #[test] + fn test_from_sketchlib_proto_bytes_hip_preserves_accumulators() { + use asap_sketchlib::proto::sketchlib::HllVariant as ProtoVariant; + let bytes = encode_state( + ProtoVariant::Hip as i32, + 2, + vec![0, 0, 0, 0], + 1.5, + 2.5, + 42.0, + ); + let acc = HllSketchAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + assert_eq!(acc.inner.variant, HllVariant::Hip); + assert_eq!(acc.inner.hip_kxq0, 1.5); + assert_eq!(acc.inner.hip_kxq1, 2.5); + assert_eq!(acc.inner.hip_est, 42.0); + } + + #[test] + fn test_from_sketchlib_proto_bytes_envelope_wrapped() { + // Mirrors what DataCollector's hllprocessor emits: the state + // wrapped in a `SketchEnvelope{hll: ...}` via sketchlib-go's + // `SerializePortableFO` + `proto.Marshal`. + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, HllVariant as ProtoVariant, HyperLogLogState, SketchEnvelope, + }; + use prost::Message; + + let state = HyperLogLogState { + variant: ProtoVariant::Regular as i32, + precision: 2, + registers: vec![1, 2, 3, 4], + hip_kxq0: 0.0, + hip_kxq1: 0.0, + hip_est: 0.0, + registers_sparse: None, + }; + let env = SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Hll(state)), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + + let acc = HllSketchAccumulator::from_sketchlib_proto_bytes(&bytes) + .expect("envelope-wrapped decode should succeed"); + assert_eq!(acc.inner.variant, HllVariant::Regular); + assert_eq!(acc.inner.registers, vec![1, 2, 3, 4]); + } + + #[test] + fn test_from_sketchlib_proto_bytes_envelope_wrong_sketch_type() { + use asap_sketchlib::proto::sketchlib::{sketch_envelope, KllState, SketchEnvelope}; + use prost::Message; + + let env = SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Kll(KllState::default())), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + + let result = HllSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err(), "wrong-sketch envelope should error"); + } + + #[test] + fn test_from_sketchlib_proto_bytes_register_length_mismatch() { + use asap_sketchlib::proto::sketchlib::HllVariant as ProtoVariant; + // precision=2 → expected 4 registers; supply only 3 + let bytes = encode_state( + ProtoVariant::Regular as i32, + 2, + vec![1, 2, 3], + 0.0, + 0.0, + 0.0, + ); + let result = HllSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err()); + assert!(result.unwrap_err().to_string().contains("registers")); + } + + #[test] + fn test_from_sketchlib_proto_bytes_zero_precision_rejected() { + use asap_sketchlib::proto::sketchlib::HyperLogLogState; + use prost::Message; + let state = HyperLogLogState::default(); + let bytes = state.encode_to_vec(); + let result = HllSketchAccumulator::from_sketchlib_proto_bytes(&bytes); + assert!(result.is_err()); + } + + #[test] + fn test_aggregate_core_merge_matches_register_max() { + let a = HllSketchAccumulator { + inner: HllSketch::from_raw(HllVariant::Regular, 2, vec![1, 5, 3, 7], 0.0, 0.0, 0.0), + sample_p: 1.0, + }; + let b = HllSketchAccumulator { + inner: HllSketch::from_raw(HllVariant::Regular, 2, vec![4, 2, 6, 0], 0.0, 0.0, 0.0), + sample_p: 1.0, + }; + let merged_box = a.merge_with(&b).expect("merge ok"); + let merged = merged_box + .as_any() + .downcast_ref::() + .expect("downcast ok"); + assert_eq!(merged.inner.registers, vec![4, 5, 6, 7]); + } + + #[test] + fn test_aggregate_core_merge_wrong_type_rejects() { + use crate::accumulators::count_sketch_accumulator::CountSketchAccumulator; + let hll = HllSketchAccumulator::new(HllVariant::Regular, 2); + let cs = CountSketchAccumulator::new(2, 3); + assert!(hll.merge_with(&cs).is_err()); + } + + #[test] + fn test_from_msgpack_bytes_round_trip() { + let original = HllSketch::from_raw( + HllVariant::Hip, + 3, + vec![0, 1, 2, 3, 4, 5, 6, 7], + 1.5, + 2.5, + 42.0, + ); + let bytes = original.to_msgpack().unwrap(); + let acc = HllSketchAccumulator::from_msgpack_bytes(&bytes).expect("decode ok"); + assert_eq!(acc.inner.variant, HllVariant::Hip); + assert_eq!(acc.inner.precision, 3); + assert_eq!(acc.inner.registers, vec![0, 1, 2, 3, 4, 5, 6, 7]); + assert_eq!(acc.inner.hip_kxq0, 1.5); + } + + #[test] + fn test_from_msgpack_bytes_rejects_garbage() { + let result = HllSketchAccumulator::from_msgpack_bytes(b"not valid msgpack"); + assert!(result.is_err()); + } + + #[test] + fn test_apply_proto_delta_bytes_round_trip() { + use asap_sketchlib::proto::sketchlib::HllDelta as PbDelta; + use prost::Message; + + let mut acc = HllSketchAccumulator::new(HllVariant::Regular, 2); + acc.inner.registers = vec![1, 5, 3, 7]; + + // Packed (index_delta, value) blob for updates {0:4, 2:6}: + // varint(0),varint(4),varint(2),varint(6). + let delta_bytes = PbDelta { + packed_updates: vec![0, 4, 2, 6], + } + .encode_to_vec(); + + acc.apply_proto_delta_bytes(&delta_bytes).expect("apply ok"); + // Max semantics: reg[0]=max(1,4)=4, reg[2]=max(3,6)=6; others unchanged. + assert_eq!(acc.inner.registers, vec![4, 5, 6, 7]); + } + + #[test] + fn test_apply_proto_delta_bytes_rejects_garbage() { + let mut acc = HllSketchAccumulator::new(HllVariant::Regular, 2); + assert!(acc.apply_proto_delta_bytes(b"not valid proto").is_err()); + } + + // ----- sample_p cardinality rescale ----- + // + // HLL uses hash-threshold sampling: each distinct key is admitted into + // the sketch with probability `p`, so the register-derived cardinality + // estimate is ~p× the true distinct count and must be rescaled by 1/p. + + #[test] + fn test_cardinality_is_rescaled_by_sample_p() { + use crate::Statistic; + // Build two accumulators with identical registers but different + // sample_p. The sampled one (p=0.25) must report ~4× the unsampled + // estimate. Use precision 8 (256 registers) with a spread of + // register values so the estimate is a non-trivial positive number. + let mut registers = vec![0u8; 256]; + for (i, r) in registers.iter_mut().enumerate() { + *r = ((i % 7) + 1) as u8; + } + let unsampled = HllSketchAccumulator { + inner: HllSketch::from_raw(HllVariant::Regular, 8, registers.clone(), 0.0, 0.0, 0.0), + sample_p: 1.0, + }; + let sampled = HllSketchAccumulator { + inner: HllSketch::from_raw(HllVariant::Regular, 8, registers, 0.0, 0.0, 0.0), + sample_p: 0.25, + }; + let raw = unsampled + .query_statistic(Statistic::Cardinality, &None, &HashMap::new()) + .expect("cardinality ok"); + let rescaled = sampled + .query_statistic(Statistic::Cardinality, &None, &HashMap::new()) + .expect("cardinality ok"); + assert!(raw > 0.0, "raw estimate should be positive, got {raw}"); + // Exact algebraic relationship: rescaled == raw / 0.25 == raw * 4. + assert!( + (rescaled - raw * 4.0).abs() < 1e-9, + "expected rescaled ≈ 4×raw ({}), got {rescaled}", + raw * 4.0 + ); + } + + #[test] + fn test_count_statistic_also_rescaled_by_sample_p() { + use crate::Statistic; + // Count maps to the same cardinality estimate for HLL, so it must + // rescale identically. + let registers = vec![3u8; 16]; + let unsampled = HllSketchAccumulator { + inner: HllSketch::from_raw(HllVariant::Regular, 4, registers.clone(), 0.0, 0.0, 0.0), + sample_p: 1.0, + }; + let sampled = HllSketchAccumulator { + inner: HllSketch::from_raw(HllVariant::Regular, 4, registers, 0.0, 0.0, 0.0), + sample_p: 0.25, + }; + let raw = unsampled + .query_statistic(Statistic::Count, &None, &HashMap::new()) + .expect("count ok"); + let rescaled = sampled + .query_statistic(Statistic::Count, &None, &HashMap::new()) + .expect("count ok"); + assert!((rescaled - raw * 4.0).abs() < 1e-9); + } + + #[test] + fn test_sample_p_unset_behaves_as_one() { + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, HllVariant as ProtoVariant, HyperLogLogState, SketchEnvelope, + }; + use prost::Message; + // An envelope with no sample_p set (proto3 default 0.0) must + // normalize to 1.0 (no rescale) — byte-compatible with legacy frames. + let state = HyperLogLogState { + variant: ProtoVariant::Regular as i32, + precision: 4, + registers: vec![2u8; 16], + hip_kxq0: 0.0, + hip_kxq1: 0.0, + hip_est: 0.0, + registers_sparse: None, + }; + let env = SketchEnvelope { + // sample_p left at proto3 default 0.0. + sketch_state: Some(sketch_envelope::SketchState::Hll(state)), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + let acc = HllSketchAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + assert_eq!(acc.sample_p, 1.0, "unset sample_p must normalize to 1.0"); + } + + #[test] + fn test_from_sketchlib_proto_bytes_reads_envelope_sample_p() { + use crate::Statistic; + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, HllVariant as ProtoVariant, HyperLogLogState, SketchEnvelope, + }; + use prost::Message; + + let registers = vec![3u8; 16]; + let state = HyperLogLogState { + variant: ProtoVariant::Regular as i32, + precision: 4, + registers: registers.clone(), + hip_kxq0: 0.0, + hip_kxq1: 0.0, + hip_est: 0.0, + registers_sparse: None, + }; + let env = SketchEnvelope { + sample_p: 0.25, + sketch_state: Some(sketch_envelope::SketchState::Hll(state)), + ..Default::default() + }; + let bytes = env.encode_to_vec(); + let acc = HllSketchAccumulator::from_sketchlib_proto_bytes(&bytes).expect("decode ok"); + assert_eq!(acc.sample_p, 0.25); + + // Compare against the unsampled estimate over the same registers. + let unsampled = HllSketchAccumulator { + inner: HllSketch::from_raw(HllVariant::Regular, 4, registers, 0.0, 0.0, 0.0), + sample_p: 1.0, + }; + let raw = unsampled + .query_statistic(Statistic::Cardinality, &None, &HashMap::new()) + .expect("cardinality ok"); + let rescaled = acc + .query_statistic(Statistic::Cardinality, &None, &HashMap::new()) + .expect("cardinality ok"); + assert!( + (rescaled - raw * 4.0).abs() < 1e-9, + "expected 4×raw rescale" + ); + } + + #[test] + fn test_reset_to_empty_preserves_sample_p() { + let mut acc = HllSketchAccumulator { + inner: HllSketch::from_raw(HllVariant::Regular, 4, vec![3u8; 16], 0.0, 0.0, 0.0), + sample_p: 0.25, + }; + acc.reset_to_empty(); + assert_eq!(acc.sample_p, 0.25, "window rotation must keep sample_p"); + assert_eq!(acc.inner.registers, vec![0u8; 16], "registers cleared"); + } + + #[test] + fn test_merge_prefers_sampled_factor() { + let a = HllSketchAccumulator { + inner: HllSketch::from_raw(HllVariant::Regular, 2, vec![1, 1, 1, 1], 0.0, 0.0, 0.0), + sample_p: 0.25, + }; + let b = HllSketchAccumulator { + inner: HllSketch::from_raw(HllVariant::Regular, 2, vec![1, 1, 1, 1], 0.0, 0.0, 0.0), + sample_p: 1.0, + }; + let merged = a.merge_with(&b).expect("merge ok"); + let merged = merged + .as_any() + .downcast_ref::() + .expect("downcast ok"); + assert_eq!(merged.sample_p, 0.25); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/hydra_kll_accumulator.rs b/crates/asap-physical-operators/src/accumulators/hydra_kll_accumulator.rs new file mode 100644 index 00000000..167bcae3 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/hydra_kll_accumulator.rs @@ -0,0 +1,165 @@ +use crate::{ + AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, + MultipleSubpopulationAggregate, SerializableToSink, +}; +use asap_sketchlib::{HydraKllSketch, MessagePackCodec}; +use base64::{engine::general_purpose, Engine as _}; +use std::collections::HashMap; + +use crate::Statistic; + +/// HydraKLL sketch accumulator — wraps asap_sketchlib::HydraKllSketch. +/// Core struct, update/merge/serde logic live in `asap_sketchlib::sketches`. +/// This file retains QE-specific trait impls and JSON output. +#[derive(Debug, Clone)] +pub struct HydraKllSketchAccumulator { + pub inner: HydraKllSketch, +} + +impl HydraKllSketchAccumulator { + pub fn new(row_num: usize, col_num: usize, k: u16) -> Self { + Self { + inner: HydraKllSketch::new(row_num, col_num, k), + } + } + + pub fn update(&mut self, key: &KeyByLabelValues, value: f64) { + self.inner.update(&key.to_semicolon_str(), value); + } + + pub fn deserialize_from_bytes(_buffer: &[u8]) -> Result> { + Err("deserialize_from_bytes for HydraKllSketchAccumulator not implemented".into()) + } + + pub fn query_key(&self, key: &KeyByLabelValues, quantile: f64) -> f64 { + self.inner.quantile(&key.to_semicolon_str(), quantile) + } +} + +impl SerializableToSink for HydraKllSketchAccumulator { + fn serialize_to_json(&self) -> serde_json::Value { + // Mirror Python implementation: {"sketch": base64_encoded_string} + let sketch_bytes = self.inner.to_msgpack().unwrap_or_default(); + let sketch_b64 = general_purpose::STANDARD.encode(&sketch_bytes); + serde_json::json!({ "sketch": sketch_b64 }) + } + + fn serialize_to_bytes(&self) -> Vec { + self.inner.to_msgpack().unwrap_or_default() + } +} + +impl MergeableAccumulator for HydraKllSketchAccumulator { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + let mut iter = accumulators.into_iter(); + let mut merged = iter.next().unwrap(); + for acc in iter { + merged.inner.merge(&acc.inner)?; + } + Ok(merged) + } +} + +impl AggregateCore for HydraKllSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "HydraKllSketchAccumulator" + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge HydraKllSketchAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + + let hk = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to HydraKllSketchAccumulator")?; + + let merged = Self::merge_accumulators(vec![self.clone(), hk.clone()])?; + Ok(Box::new(merged)) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::HydraKLL + } + + fn approx_memory_bytes(&self) -> usize { + // HydraKLL is a row*col grid of KLL sketches; typical instances + // are on the order of tens of KiB. 32 KiB is a conservative + // per-instance default. + 32 * 1024 + } + + fn get_keys(&self) -> Option> { + None + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + key: &Option, + query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::MultipleSubpopulationAggregate; + let key_val = key + .as_ref() + .ok_or("Key required for HydraKllSketchAccumulator")?; + self.query(statistic, key_val, Some(query_kwargs)) + } +} + +impl MultipleSubpopulationAggregate for HydraKllSketchAccumulator { + fn query( + &self, + statistic: Statistic, + key: &KeyByLabelValues, + query_kwargs: Option<&HashMap>, + ) -> Result> { + match statistic { + Statistic::Quantile => { + let quantile = query_kwargs + .and_then(|kwargs| kwargs.get("quantile")) + .ok_or("Missing quantile parameter for quantile query")? + .parse::() + .map_err(|_| "Invalid quantile parameter format")?; + + if !(0.0..=1.0).contains(&quantile) { + return Err("Quantile must be between 0.0 and 1.0".into()); + } + + Ok(self.query_key(key, quantile)) + } + _ => Err( + format!("Unsupported statistic in HydraKllSketchAccumulator: {statistic:?}").into(), + ), + } + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} diff --git a/crates/asap-physical-operators/src/accumulators/increase_accumulator.rs b/crates/asap-physical-operators/src/accumulators/increase_accumulator.rs new file mode 100644 index 00000000..ec609532 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/increase_accumulator.rs @@ -0,0 +1,742 @@ +use crate::{ + AggregateCore, AggregationType, Measurement, MergeableAccumulator, SerializableToSink, + SingleSubpopulationAggregate, +}; +use serde::{Deserialize, Serialize}; +use serde_json::Value; +use std::collections::HashMap; + +use crate::Statistic; + +const RESET_AWARE_WIRE_MAGIC: &[u8; 8] = b"ASAPINC2"; +const RESET_AWARE_WIRE_EXTENSION_LEN: usize = 8 + 8 + 8; + +/// Accumulator for tracking increases in counter metrics +/// Stores the starting and last seen measurements with timestamps +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct IncreaseAccumulator { + pub starting_measurement: Measurement, + pub starting_timestamp: i64, + pub last_seen_measurement: Measurement, + pub last_seen_timestamp: i64, + /// Sum of monotonic deltas, adding the post-reset value whenever the + /// counter decreases. This is the reset correction Prometheus applies. + #[serde(default)] + pub total_increase: f64, + #[serde(default)] + pub sample_count: u64, +} + +impl IncreaseAccumulator { + /// Return the number of bytes occupied by one accumulator at the start of + /// `buffer`. Old persisted values end after `last_seen_timestamp`; reset- + /// aware values carry a magic-prefixed extension. The magic makes this + /// safe when the buffer also contains the next keyed entry. + pub(crate) fn serialized_len_from_prefix( + buffer: &[u8], + ) -> Result> { + if buffer.len() < 4 { + return Err("Buffer too short for starting measurement length".into()); + } + let starting_len = u32::from_le_bytes(buffer[0..4].try_into()?) as usize; + let last_len_offset = 4usize + .checked_add(starting_len) + .and_then(|offset| offset.checked_add(8)) + .ok_or("IncreaseAccumulator length overflow")?; + if buffer.len() < last_len_offset + 4 { + return Err("Buffer too short for last seen measurement length".into()); + } + let last_len = + u32::from_le_bytes(buffer[last_len_offset..last_len_offset + 4].try_into()?) as usize; + let legacy_len = last_len_offset + .checked_add(4) + .and_then(|offset| offset.checked_add(last_len)) + .and_then(|offset| offset.checked_add(8)) + .ok_or("IncreaseAccumulator length overflow")?; + if buffer.len() < legacy_len { + return Err("Buffer too short for last seen timestamp".into()); + } + let has_extension = buffer.len() >= legacy_len + RESET_AWARE_WIRE_EXTENSION_LEN + && &buffer[legacy_len..legacy_len + RESET_AWARE_WIRE_MAGIC.len()] + == RESET_AWARE_WIRE_MAGIC; + Ok(legacy_len + + if has_extension { + RESET_AWARE_WIRE_EXTENSION_LEN + } else { + 0 + }) + } + + pub fn new( + starting_measurement: Measurement, + starting_timestamp: i64, + last_seen_measurement: Measurement, + last_seen_timestamp: i64, + ) -> Self { + let total_increase = if last_seen_timestamp <= starting_timestamp { + 0.0 + } else if last_seen_measurement.value >= starting_measurement.value { + last_seen_measurement.value - starting_measurement.value + } else { + last_seen_measurement.value + }; + let sample_count = if last_seen_timestamp > starting_timestamp { + 2 + } else { + 1 + }; + Self { + starting_measurement, + starting_timestamp, + last_seen_measurement, + last_seen_timestamp, + total_increase, + sample_count, + } + } + + pub fn update(&mut self, measurement: Measurement, timestamp: i64) { + if timestamp < self.last_seen_timestamp { + return; + } + if timestamp == self.last_seen_timestamp { + return; + } + if measurement.value >= self.last_seen_measurement.value { + self.total_increase += measurement.value - self.last_seen_measurement.value; + } else { + self.total_increase += measurement.value; + } + self.last_seen_measurement = measurement; + self.last_seen_timestamp = timestamp; + self.sample_count = self.sample_count.saturating_add(1); + } + + pub fn deserialize_from_json(data: &Value) -> Result> { + let starting_measurement = + Measurement::deserialize_from_json(&data["starting_measurement"])?; + let starting_timestamp = data["starting_timestamp"] + .as_i64() + .ok_or("Missing or invalid 'starting_timestamp' field")?; + let last_seen_measurement = + Measurement::deserialize_from_json(&data["last_seen_measurement"])?; + let last_seen_timestamp = data["last_seen_timestamp"] + .as_i64() + .ok_or("Missing or invalid 'last_seen_timestamp' field")?; + + let mut accumulator = Self::new( + starting_measurement, + starting_timestamp, + last_seen_measurement, + last_seen_timestamp, + ); + accumulator.total_increase = data["total_increase"] + .as_f64() + .unwrap_or(accumulator.total_increase); + accumulator.sample_count = data["sample_count"] + .as_u64() + .unwrap_or(accumulator.sample_count); + Ok(accumulator) + } + + pub fn deserialize_from_bytes(buffer: &[u8]) -> Result> { + let mut offset = 0; + + // Read starting measurement length and data + if buffer.len() < offset + 4 { + return Err("Buffer too short for starting measurement length".into()); + } + let starting_measurement_length = u32::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + ]) as usize; + offset += 4; + + if buffer.len() < offset + starting_measurement_length { + return Err("Buffer too short for starting measurement".into()); + } + let starting_measurement = Measurement::deserialize_from_bytes( + &buffer[offset..offset + starting_measurement_length], + )?; + offset += starting_measurement_length; + + // Read starting timestamp + if buffer.len() < offset + 8 { + return Err("Buffer too short for starting timestamp".into()); + } + let starting_timestamp = i64::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + buffer[offset + 4], + buffer[offset + 5], + buffer[offset + 6], + buffer[offset + 7], + ]); + offset += 8; + + // Read last seen measurement length and data + if buffer.len() < offset + 4 { + return Err("Buffer too short for last seen measurement length".into()); + } + let last_seen_measurement_length = u32::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + ]) as usize; + offset += 4; + + if buffer.len() < offset + last_seen_measurement_length { + return Err("Buffer too short for last seen measurement".into()); + } + let last_seen_measurement = Measurement::deserialize_from_bytes( + &buffer[offset..offset + last_seen_measurement_length], + )?; + offset += last_seen_measurement_length; + + // Read last seen timestamp + if buffer.len() < offset + 8 { + return Err("Buffer too short for last seen timestamp".into()); + } + let last_seen_timestamp = i64::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + buffer[offset + 4], + buffer[offset + 5], + buffer[offset + 6], + buffer[offset + 7], + ]); + + let mut accumulator = Self::new( + starting_measurement, + starting_timestamp, + last_seen_measurement, + last_seen_timestamp, + ); + offset += 8; + if buffer.len() >= offset + RESET_AWARE_WIRE_EXTENSION_LEN + && &buffer[offset..offset + RESET_AWARE_WIRE_MAGIC.len()] == RESET_AWARE_WIRE_MAGIC + { + offset += RESET_AWARE_WIRE_MAGIC.len(); + accumulator.total_increase = f64::from_le_bytes( + buffer[offset..offset + 8] + .try_into() + .expect("checked total-increase bytes"), + ); + offset += 8; + accumulator.sample_count = u64::from_le_bytes( + buffer[offset..offset + 8] + .try_into() + .expect("checked sample-count bytes"), + ); + } + Ok(accumulator) + } +} + +impl SerializableToSink for IncreaseAccumulator { + fn serialize_to_json(&self) -> Value { + serde_json::json!({ + "starting_measurement": self.starting_measurement.serialize_to_json(), + "starting_timestamp": self.starting_timestamp, + "last_seen_measurement": self.last_seen_measurement.serialize_to_json(), + "last_seen_timestamp": self.last_seen_timestamp, + "total_increase": self.total_increase, + "sample_count": self.sample_count, + }) + } + + fn serialize_to_bytes(&self) -> Vec { + let starting_measurement_bytes = self.starting_measurement.serialize_to_bytes(); + let last_seen_measurement_bytes = self.last_seen_measurement.serialize_to_bytes(); + + let mut buffer = Vec::new(); + + // Starting measurement length and data + buffer.extend_from_slice(&(starting_measurement_bytes.len() as u32).to_le_bytes()); + buffer.extend_from_slice(&starting_measurement_bytes); + + // Starting timestamp + buffer.extend_from_slice(&self.starting_timestamp.to_le_bytes()); + + // Last seen measurement length and data + buffer.extend_from_slice(&(last_seen_measurement_bytes.len() as u32).to_le_bytes()); + buffer.extend_from_slice(&last_seen_measurement_bytes); + + // Last seen timestamp + buffer.extend_from_slice(&self.last_seen_timestamp.to_le_bytes()); + buffer.extend_from_slice(RESET_AWARE_WIRE_MAGIC); + buffer.extend_from_slice(&self.total_increase.to_le_bytes()); + buffer.extend_from_slice(&self.sample_count.to_le_bytes()); + + buffer + } +} + +impl MergeableAccumulator for IncreaseAccumulator { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + + let mut accumulators = accumulators; + accumulators.sort_by_key(|accumulator| accumulator.starting_timestamp); + let mut result = accumulators[0].clone(); + + for acc in &accumulators[1..] { + if acc.starting_timestamp > result.last_seen_timestamp { + result.total_increase += + if acc.starting_measurement.value >= result.last_seen_measurement.value { + acc.starting_measurement.value - result.last_seen_measurement.value + } else { + acc.starting_measurement.value + }; + } + result.total_increase += acc.total_increase; + result.sample_count = result.sample_count.saturating_add(acc.sample_count); + if acc.last_seen_timestamp > result.last_seen_timestamp { + result.last_seen_measurement = acc.last_seen_measurement.clone(); + result.last_seen_timestamp = acc.last_seen_timestamp; + } + } + + Ok(result) + } +} + +impl AggregateCore for IncreaseAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "IncreaseAccumulator" + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + // Check if other is also an IncreaseAccumulator + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge IncreaseAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + + // Downcast to IncreaseAccumulator + let other_increase = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to IncreaseAccumulator")?; + + let (first, second) = if self.starting_timestamp <= other_increase.starting_timestamp { + (self, other_increase) + } else { + (other_increase, self) + }; + let mut merged = first.clone(); + if second.starting_timestamp > merged.last_seen_timestamp { + merged.total_increase += + if second.starting_measurement.value >= merged.last_seen_measurement.value { + second.starting_measurement.value - merged.last_seen_measurement.value + } else { + second.starting_measurement.value + }; + } + merged.total_increase += second.total_increase; + merged.sample_count = merged.sample_count.saturating_add(second.sample_count); + if second.last_seen_timestamp > merged.last_seen_timestamp { + merged.last_seen_measurement = second.last_seen_measurement.clone(); + merged.last_seen_timestamp = second.last_seen_timestamp; + } + + Ok(Box::new(merged)) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::Increase + } + + fn approx_memory_bytes(&self) -> usize { + // Two Measurements + two i64s. Measurements are a few f64 fields. + std::mem::size_of::() + } + + fn get_keys(&self) -> Option> { + None + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + _key: &Option, + query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::SingleSubpopulationAggregate; + self.query( + statistic, + (!query_kwargs.is_empty()).then_some(query_kwargs), + ) + } +} + +impl SingleSubpopulationAggregate for IncreaseAccumulator { + fn query( + &self, + statistic: Statistic, + query_kwargs: Option<&HashMap>, + ) -> Result> { + match statistic { + Statistic::Increase => Ok(self.extrapolated_value(query_kwargs, false)?), + Statistic::Rate => Ok(self.extrapolated_value(query_kwargs, true)?), + // For instant `sum [by (...)] (counter_metric)` Prometheus + // sums the latest cumulative value of each matching series. + // The IncreaseAccumulator already tracks that latest value + // in `last_seen_measurement`, so per-series Sum is just + // that scalar; the engine's outer aggregation groups by the + // `by` labels and adds the per-series totals across keys. + // + // See PR #108 audit conclusion (commit 4359e10) and issue + // ProjectASAP/ASAPCollector#46: pre-fix the ASAP tier ingested + // counters as IncreaseAccumulator and bare `sum by (...) ()` + // capability-missed because this trait did not answer Sum. + Statistic::Sum => Ok(self.last_seen_measurement.value), + _ => Err(format!("Unsupported statistic in IncreaseAccumulator: {statistic:?}").into()), + } + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +impl IncreaseAccumulator { + fn extrapolated_value( + &self, + query_kwargs: Option<&HashMap>, + is_rate: bool, + ) -> Result> { + if self.sample_count < 2 || self.last_seen_timestamp <= self.starting_timestamp { + return Err("at least two ordered counter samples are required".into()); + } + let sampled_interval = (self.last_seen_timestamp - self.starting_timestamp) as f64 / 1000.0; + let Some(kwargs) = query_kwargs else { + return Ok(if is_rate { + self.total_increase / sampled_interval + } else { + self.total_increase + }); + }; + let range_start = kwargs + .get("range_start_ms") + .ok_or("missing range_start_ms")? + .parse::()?; + let range_end = kwargs + .get("range_end_ms") + .ok_or("missing range_end_ms")? + .parse::()?; + if range_end <= range_start { + return Err("invalid counter evaluation range".into()); + } + + let mut duration_to_start = + (self.starting_timestamp.saturating_sub(range_start)) as f64 / 1000.0; + let duration_to_end = (range_end.saturating_sub(self.last_seen_timestamp)) as f64 / 1000.0; + let average_sample_interval = sampled_interval / (self.sample_count - 1) as f64; + let extrapolation_threshold = average_sample_interval * 1.1; + + if self.total_increase > 0.0 && self.starting_measurement.value >= 0.0 { + let duration_to_zero = + sampled_interval * (self.starting_measurement.value / self.total_increase); + duration_to_start = duration_to_start.min(duration_to_zero); + } + let mut extrapolate_to = sampled_interval; + extrapolate_to += if duration_to_start < extrapolation_threshold { + duration_to_start.max(0.0) + } else { + average_sample_interval / 2.0 + }; + extrapolate_to += if duration_to_end < extrapolation_threshold { + duration_to_end.max(0.0) + } else { + average_sample_interval / 2.0 + }; + let mut factor = extrapolate_to / sampled_interval; + if is_rate { + factor /= (range_end - range_start) as f64 / 1000.0; + } + Ok(self.total_increase * factor) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_increase_accumulator_creation() { + let starting_measurement = Measurement::new(10.0); + let last_seen_measurement = Measurement::new(25.0); + let acc = IncreaseAccumulator::new( + starting_measurement.clone(), + 1000, + last_seen_measurement.clone(), + 2000, + ); + + assert_eq!(acc.starting_measurement.value, 10.0); + assert_eq!(acc.starting_timestamp, 1000); + assert_eq!(acc.last_seen_measurement.value, 25.0); + assert_eq!(acc.last_seen_timestamp, 2000); + } + + #[test] + fn test_increase_accumulator_update() { + let starting_measurement = Measurement::new(10.0); + let mut acc = IncreaseAccumulator::new( + starting_measurement.clone(), + 1000, + starting_measurement.clone(), + 1000, + ); + + let new_measurement = Measurement::new(25.0); + acc.update(new_measurement.clone(), 2000); + + assert_eq!(acc.last_seen_measurement.value, 25.0); + assert_eq!(acc.last_seen_timestamp, 2000); + assert_eq!(acc.starting_measurement.value, 10.0); // Should remain unchanged + } + + #[test] + fn test_increase_accumulator_query() { + let starting_measurement = Measurement::new(10.0); + let last_seen_measurement = Measurement::new(25.0); + let acc = IncreaseAccumulator::new( + starting_measurement, + 1000, + last_seen_measurement, + 3000, // 2 second difference + ); + + // Test increase calculation + assert_eq!( + crate::SingleSubpopulationAggregate::query(&acc, Statistic::Increase, None).unwrap(), + 15.0 + ); + + // Test rate calculation (per second) + assert_eq!( + crate::SingleSubpopulationAggregate::query(&acc, Statistic::Rate, None).unwrap(), + 7.5 + ); // 15.0 / 2.0 + + // Statistic::Sum returns the latest cumulative counter value, + // matching Prometheus semantics for instant `sum()`. + // (Issue ProjectASAP/ASAPCollector#46, PR #108 diagnosis.) + assert_eq!( + crate::SingleSubpopulationAggregate::query(&acc, Statistic::Sum, None).unwrap(), + 25.0 + ); + + // Unsupported statistics still error. + assert!(crate::SingleSubpopulationAggregate::query(&acc, Statistic::Min, None).is_err()); + } + + #[test] + fn prometheus_counter_reset_and_boundary_extrapolation() { + let mut acc = IncreaseAccumulator::new( + Measurement::new(10.0), + 10_000, + Measurement::new(10.0), + 10_000, + ); + acc.update(Measurement::new(20.0), 20_000); + acc.update(Measurement::new(3.0), 30_000); + acc.update(Measurement::new(13.0), 50_000); + assert_eq!(acc.total_increase, 23.0); + assert_eq!(acc.sample_count, 4); + + let kwargs = HashMap::from([ + ("range_start_ms".into(), "0".into()), + ("range_end_ms".into(), "60000".into()), + ]); + let increase = + crate::SingleSubpopulationAggregate::query(&acc, Statistic::Increase, Some(&kwargs)) + .unwrap(); + let rate = crate::SingleSubpopulationAggregate::query(&acc, Statistic::Rate, Some(&kwargs)) + .unwrap(); + assert!((increase - 34.5).abs() < 1e-12); + assert!((rate - 0.575).abs() < 1e-12); + } + + #[test] + fn pane_merge_preserves_resets_and_prometheus_extrapolation() { + let mut left = IncreaseAccumulator::new( + Measurement::new(10.0), + 10_000, + Measurement::new(10.0), + 10_000, + ); + left.update(Measurement::new(20.0), 20_000); + let mut right = + IncreaseAccumulator::new(Measurement::new(3.0), 30_000, Measurement::new(3.0), 30_000); + right.update(Measurement::new(13.0), 50_000); + let merged = IncreaseAccumulator::merge_accumulators(vec![right, left]).unwrap(); + assert_eq!(merged.total_increase, 23.0); + assert_eq!(merged.sample_count, 4); + let kwargs = HashMap::from([ + ("range_start_ms".into(), "0".into()), + ("range_end_ms".into(), "60000".into()), + ]); + assert_eq!( + crate::SingleSubpopulationAggregate::query(&merged, Statistic::Increase, Some(&kwargs)) + .unwrap(), + 34.5 + ); + } + + #[test] + fn counter_sds_state_is_constant_size_per_pane() { + let mut acc = IncreaseAccumulator::new(Measurement::new(0.0), 0, Measurement::new(0.0), 0); + let initial = acc.serialize_to_bytes().len(); + for second in 1..=86_400 { + acc.update(Measurement::new(second as f64), second * 1_000); + } + assert_eq!(acc.serialize_to_bytes().len(), initial); + assert_eq!(acc.sample_count, 86_401); + assert_eq!( + acc.approx_memory_bytes(), + std::mem::size_of::() + ); + } + + #[test] + fn test_increase_accumulator_sum_is_latest_cumulative_value() { + // Instant `sum ()` semantics: the per-series summand is + // the latest cumulative counter value. Two series with latest + // values 100 and 50 (started at 10 and 5 respectively) should + // each report Sum = 100 and Sum = 50 — the engine's `sum by` + // outer aggregation does the cross-series total. + let acc_a = + IncreaseAccumulator::new(Measurement::new(10.0), 1000, Measurement::new(100.0), 2000); + let acc_b = + IncreaseAccumulator::new(Measurement::new(5.0), 1000, Measurement::new(50.0), 2000); + assert_eq!( + crate::SingleSubpopulationAggregate::query(&acc_a, Statistic::Sum, None).unwrap(), + 100.0 + ); + assert_eq!( + crate::SingleSubpopulationAggregate::query(&acc_b, Statistic::Sum, None).unwrap(), + 50.0 + ); + } + + #[test] + fn test_increase_accumulator_merge() { + let acc1 = + IncreaseAccumulator::new(Measurement::new(10.0), 1000, Measurement::new(20.0), 2000); + let acc2 = IncreaseAccumulator::new( + Measurement::new(5.0), + 500, // Earlier start + Measurement::new(15.0), + 1500, + ); + let acc3 = IncreaseAccumulator::new( + Measurement::new(20.0), + 2000, + Measurement::new(30.0), + 3000, // Later end + ); + + let merged = + >::merge_accumulators( + vec![acc1, acc2, acc3], + ) + .unwrap(); + + // Should use earliest start and latest end + assert_eq!(merged.starting_measurement.value, 5.0); + assert_eq!(merged.starting_timestamp, 500); + assert_eq!(merged.last_seen_measurement.value, 30.0); + assert_eq!(merged.last_seen_timestamp, 3000); + } + + #[test] + fn test_increase_accumulator_serialization() { + let acc = + IncreaseAccumulator::new(Measurement::new(10.0), 1000, Measurement::new(25.0), 2000); + + // Test JSON serialization + let json = acc.serialize_to_json(); + let deserialized = IncreaseAccumulator::deserialize_from_json(&json).unwrap(); + assert_eq!( + acc.starting_measurement.value, + deserialized.starting_measurement.value + ); + assert_eq!(acc.starting_timestamp, deserialized.starting_timestamp); + assert_eq!( + acc.last_seen_measurement.value, + deserialized.last_seen_measurement.value + ); + assert_eq!(acc.last_seen_timestamp, deserialized.last_seen_timestamp); + + // Test byte serialization + let bytes = acc.serialize_to_bytes(); + let deserialized_bytes = IncreaseAccumulator::deserialize_from_bytes(&bytes).unwrap(); + assert_eq!( + acc.starting_measurement.value, + deserialized_bytes.starting_measurement.value + ); + assert_eq!( + acc.starting_timestamp, + deserialized_bytes.starting_timestamp + ); + assert_eq!( + acc.last_seen_measurement.value, + deserialized_bytes.last_seen_measurement.value + ); + assert_eq!( + acc.last_seen_timestamp, + deserialized_bytes.last_seen_timestamp + ); + assert_eq!(acc.total_increase, deserialized_bytes.total_increase); + assert_eq!(acc.sample_count, deserialized_bytes.sample_count); + + let legacy = &bytes[..bytes.len() - RESET_AWARE_WIRE_EXTENSION_LEN]; + let legacy_value = IncreaseAccumulator::deserialize_from_bytes(legacy).unwrap(); + assert_eq!(legacy_value.total_increase, 15.0); + assert_eq!(legacy_value.sample_count, 2); + } + + #[test] + fn test_trait_object() { + let acc: Box = Box::new(IncreaseAccumulator::new( + Measurement::new(10.0), + 1000, + Measurement::new(25.0), + 2000, + )); + + assert_eq!(acc.type_name(), "IncreaseAccumulator"); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/keyed_counter_state.rs b/crates/asap-physical-operators/src/accumulators/keyed_counter_state.rs new file mode 100644 index 00000000..883460b2 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/keyed_counter_state.rs @@ -0,0 +1,529 @@ +use crate::accumulators::IncreaseAccumulator; +use crate::{ + AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, + MultipleSubpopulationAggregate, SerializableToSink, SingleSubpopulationAggregate, +}; +use serde::{Deserialize, Serialize}; +use serde_json::Value; +use std::collections::HashMap; + +use crate::Statistic; + +/// Accumulator that maintains separate increase accumulators for multiple keys +/// Allows tracking rate/increase for different label combinations +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct KeyedCounterState { + pub increases: HashMap, +} + +impl KeyedCounterState { + pub fn new() -> Self { + Self { + increases: HashMap::new(), + } + } + + pub fn update(&mut self, key: KeyByLabelValues, accumulator: IncreaseAccumulator) { + self.increases.insert(key, accumulator); + } + + pub fn deserialize_from_json(data: &Value) -> Result> { + let mut accumulator = Self::new(); + + if let Some(entries) = data["entries"].as_array() { + for entry in entries { + let key = KeyByLabelValues::deserialize_from_json(&entry["key"])?; + let increase_data = + IncreaseAccumulator::deserialize_from_json(&entry["increase_data"])?; + accumulator.increases.insert(key, increase_data); + } + } + + Ok(accumulator) + } + + pub fn deserialize_from_bytes(buffer: &[u8]) -> Result> { + let mut accumulator = Self::new(); + let mut offset = 0; + + // Read number of entries + if buffer.len() < 4 { + return Err("Buffer too short for entry count".into()); + } + let num_entries = u32::from_le_bytes([buffer[0], buffer[1], buffer[2], buffer[3]]) as usize; + offset += 4; + + for _ in 0..num_entries { + // Read key length and key + if offset + 4 > buffer.len() { + return Err("Buffer too short for key length".into()); + } + let key_length = u32::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + ]) as usize; + offset += 4; + + if offset + key_length > buffer.len() { + return Err("Buffer too short for key data".into()); + } + let key = + KeyByLabelValues::deserialize_from_bytes(&buffer[offset..offset + key_length])?; + offset += key_length; + + // Read IncreaseAccumulator data + if offset >= buffer.len() { + return Err("Buffer too short for increase accumulator data".into()); + } + let consumed_bytes = + IncreaseAccumulator::serialized_len_from_prefix(&buffer[offset..])?; + let increase_data = IncreaseAccumulator::deserialize_from_bytes( + &buffer[offset..offset + consumed_bytes], + )?; + offset += consumed_bytes; + + accumulator.increases.insert(key, increase_data); + } + + Ok(accumulator) + } +} + +impl Default for KeyedCounterState { + fn default() -> Self { + Self::new() + } +} + +impl SerializableToSink for KeyedCounterState { + fn serialize_to_json(&self) -> Value { + let entries: Vec = self + .increases + .iter() + .map(|(key, data)| { + serde_json::json!({ + "key": key.serialize_to_json(), + "increase_data": data.serialize_to_json() + }) + }) + .collect(); + + serde_json::json!({ + "entries": entries + }) + } + + fn serialize_to_bytes(&self) -> Vec { + let mut buffer = Vec::new(); + + // Write number of entries + buffer.extend_from_slice(&(self.increases.len() as u32).to_le_bytes()); + + // Write each key-value pair + for (key, data) in &self.increases { + let key_bytes = key.serialize_to_bytes(); + buffer.extend_from_slice(&(key_bytes.len() as u32).to_le_bytes()); + buffer.extend_from_slice(&key_bytes); + + let data_bytes = data.serialize_to_bytes(); + buffer.extend_from_slice(&data_bytes); + } + + buffer + } +} + +impl AggregateCore for KeyedCounterState { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "KeyedCounterState" + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + // Check if other is also a KeyedCounterState + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge KeyedCounterState with {}", + other.get_accumulator_type() + ) + .into()); + } + + // Downcast to KeyedCounterState + let other_multiple_increase = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to KeyedCounterState")?; + + // Clone self once, then merge each matching counter with the same + // reset-aware, boundary-aware implementation used by the unkeyed path. + let mut merged = self.clone(); + for (key, data) in &other_multiple_increase.increases { + if let Some(existing_data) = merged.increases.get_mut(key) { + *existing_data = IncreaseAccumulator::merge_accumulators(vec![ + existing_data.clone(), + data.clone(), + ])?; + } else { + merged.increases.insert(key.clone(), data.clone()); + } + } + + Ok(Box::new(merged)) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::Increase + } + + fn approx_memory_bytes(&self) -> usize { + // HashMap. IncreaseAccumulator is ~64 B, + // per-entry key/overhead is ~96 B. + const BYTES_PER_ENTRY: usize = 160; + std::mem::size_of::() + self.increases.len() * BYTES_PER_ENTRY + } + + fn get_keys(&self) -> Option> { + Some(self.increases.keys().cloned().collect()) + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + key: &Option, + query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::MultipleSubpopulationAggregate; + let key_val = key.as_ref().ok_or("Key required for KeyedCounterState")?; + self.query(statistic, key_val, Some(query_kwargs)) + } +} + +impl MultipleSubpopulationAggregate for KeyedCounterState { + fn query( + &self, + statistic: Statistic, + key: &KeyByLabelValues, + query_kwargs: Option<&HashMap>, + ) -> Result> { + let data = self + .increases + .get(key) + .ok_or_else(|| format!("Key {key} not found in KeyedCounterState"))?; + + data.query(statistic, query_kwargs) + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +impl MergeableAccumulator for KeyedCounterState { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + + let mut result = KeyedCounterState::new(); + + for accumulator in accumulators { + for (key, data) in accumulator.increases { + if let Some(existing_data) = result.increases.get_mut(&key) { + *existing_data = + IncreaseAccumulator::merge_accumulators(vec![existing_data.clone(), data])?; + } else { + result.increases.insert(key, data); + } + } + } + + Ok(result) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::Measurement; + + fn create_test_increase_accumulator(start_val: f64, end_val: f64) -> IncreaseAccumulator { + IncreaseAccumulator::new( + Measurement::new(start_val), + 1000, + Measurement::new(end_val), + 2000, + ) + } + + fn create_test_increase_accumulator_with_time( + start_val: f64, + start_time: i64, + end_val: f64, + end_time: i64, + ) -> IncreaseAccumulator { + IncreaseAccumulator::new( + Measurement::new(start_val), + start_time, + Measurement::new(end_val), + end_time, + ) + } + + #[test] + fn test_keyed_counter_state_creation() { + let acc = KeyedCounterState::new(); + assert!(acc.increases.is_empty()); + } + + #[test] + fn test_keyed_counter_state_update() { + let mut acc = KeyedCounterState::new(); + + let key1 = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + + let key2 = KeyByLabelValues::new_with_labels(vec!["api".to_string()]); + + let increase1 = create_test_increase_accumulator(10.0, 25.0); + let increase2 = create_test_increase_accumulator(5.0, 15.0); + + acc.update(key1.clone(), increase1); + acc.update(key2.clone(), increase2); + + assert_eq!(acc.increases.len(), 2); + assert!(acc.increases.contains_key(&key1)); + assert!(acc.increases.contains_key(&key2)); + } + + #[test] + fn test_keyed_counter_state_query() { + let mut acc = KeyedCounterState::new(); + + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + + let increase_acc = create_test_increase_accumulator(10.0, 25.0); + acc.update(key.clone(), increase_acc); + + // Test increase query + assert_eq!(acc.query(Statistic::Increase, &key, None).unwrap(), 15.0); + + // Test rate query (15.0 increase over 1 second = 15.0 per second) + assert_eq!(acc.query(Statistic::Rate, &key, None).unwrap(), 15.0); + + // Sum returns the latest cumulative counter value for the + // queried key (per-series Prometheus `sum()` semantics; + // see issue ProjectASAP/ASAPCollector#46 and PR #108 diagnosis). + // The series here was created with last_seen=25.0. + assert_eq!(acc.query(Statistic::Sum, &key, None).unwrap(), 25.0); + + // Unsupported statistic still errors. + assert!(acc.query(Statistic::Min, &key, None).is_err()); + + let unknown_key = KeyByLabelValues::new(); + assert!(acc.query(Statistic::Increase, &unknown_key, None).is_err()); + } + + #[test] + fn test_keyed_counter_state_sum_per_key() { + // `sum by (zone) (counter)` reaches KeyedCounterState + // only when the ASAP-tier ingest groups multiple series under + // a single accumulator (the `Multiple*` variant). In that case + // each per-key Sum should be the series' latest cumulative + // value; the engine's outer `by` aggregation does the cross-key + // grouping. (Issue ProjectASAP/ASAPCollector#46.) + let mut acc = KeyedCounterState::new(); + let east = KeyByLabelValues::new_with_labels(vec!["us-east-1".to_string()]); + let west = KeyByLabelValues::new_with_labels(vec!["us-west-2".to_string()]); + + acc.update( + east.clone(), + IncreaseAccumulator::new(Measurement::new(10.0), 1000, Measurement::new(100.0), 2000), + ); + acc.update( + west.clone(), + IncreaseAccumulator::new(Measurement::new(5.0), 1000, Measurement::new(50.0), 2000), + ); + + assert_eq!(acc.query(Statistic::Sum, &east, None).unwrap(), 100.0); + assert_eq!(acc.query(Statistic::Sum, &west, None).unwrap(), 50.0); + } + + #[test] + fn test_keyed_counter_state_merge() { + let mut acc1 = KeyedCounterState::new(); + let mut acc2 = KeyedCounterState::new(); + + let key1 = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + + let key2 = KeyByLabelValues::new_with_labels(vec!["api".to_string()]); + + // Add different keys to each accumulator + acc1.update(key1.clone(), create_test_increase_accumulator(10.0, 20.0)); + acc2.update(key2.clone(), create_test_increase_accumulator(5.0, 15.0)); + + // Also add overlapping key with different time ranges (later timestamps) + acc2.update( + key1.clone(), + create_test_increase_accumulator_with_time(15.0, 2000, 30.0, 3000), + ); // Later time range + + let merged = KeyedCounterState::merge_accumulators(vec![acc1, acc2]).unwrap(); + + assert_eq!(merged.increases.len(), 2); + assert!(merged.increases.contains_key(&key1)); + assert!(merged.increases.contains_key(&key2)); + + // The merged key1 should have the full range (earliest start to latest end) + let merged_key1 = merged.increases.get(&key1).unwrap(); + assert_eq!(merged_key1.starting_measurement.value, 10.0); // Earlier start + assert_eq!(merged_key1.last_seen_measurement.value, 30.0); // Later end + } + + #[test] + fn test_keyed_counter_state_serialization() { + let mut acc = KeyedCounterState::new(); + + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + let second_key = KeyByLabelValues::new_with_labels(vec!["api".to_string()]); + let mut reset_aware = create_test_increase_accumulator(10.0, 25.0); + reset_aware.update(Measurement::new(3.0), 3000); + acc.update(key.clone(), reset_aware); + acc.update( + second_key.clone(), + create_test_increase_accumulator(4.0, 9.0), + ); + + // Test JSON serialization + let json_value = acc.serialize_to_json(); + let deserialized = KeyedCounterState::deserialize_from_json(&json_value).unwrap(); + + assert_eq!(deserialized.increases.len(), 2); + let deserialized_acc = deserialized.increases.get(&key).unwrap(); + assert_eq!(deserialized_acc.starting_measurement.value, 10.0); + assert_eq!(deserialized_acc.last_seen_measurement.value, 3.0); + assert_eq!(deserialized_acc.total_increase, 18.0); + + // Test binary serialization + let bytes = acc.serialize_to_bytes(); + let deserialized_bytes = KeyedCounterState::deserialize_from_bytes(&bytes).unwrap(); + + assert_eq!(deserialized_bytes.increases.len(), 2); + let deserialized_acc_bytes = deserialized_bytes.increases.get(&key).unwrap(); + assert_eq!(deserialized_acc_bytes.starting_measurement.value, 10.0); + assert_eq!(deserialized_acc_bytes.last_seen_measurement.value, 3.0); + assert_eq!(deserialized_acc_bytes.total_increase, 18.0); + assert_eq!( + deserialized_bytes + .increases + .get(&second_key) + .unwrap() + .last_seen_measurement + .value, + 9.0 + ); + } + + #[test] + fn test_keyed_counter_state_get_keys() { + let mut acc = KeyedCounterState::new(); + + let key1 = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + let key2 = KeyByLabelValues::new_with_labels(vec!["api".to_string()]); + + acc.update(key1.clone(), create_test_increase_accumulator(10.0, 20.0)); + acc.update(key2.clone(), create_test_increase_accumulator(5.0, 15.0)); + + let keys = acc.get_keys().unwrap(); + assert_eq!(keys.len(), 2); + assert!(keys.contains(&key1)); + assert!(keys.contains(&key2)); + } + + #[test] + fn test_trait_object() { + let mut acc = KeyedCounterState::new(); + let key = KeyByLabelValues::new(); + acc.update(key.clone(), create_test_increase_accumulator(10.0, 25.0)); + + let trait_obj: Box = Box::new(acc); + assert_eq!( + trait_obj.query(Statistic::Increase, &key, None).unwrap(), + 15.0 + ); + + let keys = trait_obj.get_keys().unwrap(); + assert_eq!(keys.len(), 1); + } + + // #[test] + // fn test_keyed_counter_state_arroyo_deserialization() { + // // Create test data in Arroyo MessagePack format + // // Format: {key: [starting_value, starting_timestamp, last_seen_value, last_seen_timestamp]} + // let mut test_data = std::collections::HashMap::new(); + // test_data.insert("web;service".to_string(), vec![10.0, 1000.0, 25.0, 2000.0]); + // test_data.insert("api;service".to_string(), vec![5.0, 1500.0, 15.0, 2500.0]); + + // // Serialize to MessagePack + // let arroyo_buffer = rmp_serde::to_vec(&test_data).unwrap(); + + // // Test Arroyo deserialization + // let deserialized_acc = + // KeyedCounterState::deserialize_from_bytes_arroyo(&arroyo_buffer).unwrap(); + + // // Verify the deserialized accumulator has the correct data + // assert_eq!(deserialized_acc.increases.len(), 2); + + // // Check first key (web;service) + // let keys: Vec<_> = deserialized_acc.increases.keys().collect(); + // let key1 = keys + // .iter() + // .find(|k| k.labels.get("label_0").is_some_and(|v| v == "web")) + // .unwrap(); + + // let increase1 = deserialized_acc.increases.get(key1).unwrap(); + // assert_eq!(increase1.starting_measurement.value, 10.0); + // assert_eq!(increase1.starting_timestamp, 1000); + // assert_eq!(increase1.last_seen_measurement.value, 25.0); + // assert_eq!(increase1.last_seen_timestamp, 2000); + + // // Check second key (api;service) + // let key2 = keys + // .iter() + // .find(|k| k.labels.get("label_0").is_some_and(|v| v == "api")) + // .unwrap(); + + // let increase2 = deserialized_acc.increases.get(key2).unwrap(); + // assert_eq!(increase2.starting_measurement.value, 5.0); + // assert_eq!(increase2.starting_timestamp, 1500); + // assert_eq!(increase2.last_seen_measurement.value, 15.0); + // assert_eq!(increase2.last_seen_timestamp, 2500); + + // // Test querying + // assert_eq!( + // deserialized_acc.query(Statistic::Increase, key1).unwrap(), + // 15.0 + // ); // 25.0 - 10.0 + // assert_eq!( + // deserialized_acc.query(Statistic::Increase, key2).unwrap(), + // 10.0 + // ); // 15.0 - 5.0 + // } +} diff --git a/crates/asap-physical-operators/src/accumulators/keyed_max_state.rs b/crates/asap-physical-operators/src/accumulators/keyed_max_state.rs new file mode 100644 index 00000000..304b3a37 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/keyed_max_state.rs @@ -0,0 +1,335 @@ +use crate::{ + AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, + MultipleSubpopulationAggregate, SerializableToSink, +}; +use serde::{Deserialize, Serialize}; +use serde_json::Value; +use std::collections::HashMap; + +use crate::Statistic; + +/// Exact per-key maximum over many populations, mergeable by comparison. +/// +/// The minimum direction is +/// [`KeyedMinState`](super::keyed_min_state::KeyedMinState), +/// a separate type: these used to be one `MultipleMinMaxAccumulator` whose +/// direction lived in a `sub_type` string that every layer above had to carry +/// alongside the family. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct KeyedMaxState { + pub values: HashMap, +} + +impl KeyedMaxState { + pub fn new() -> Self { + Self::default() + } + + pub fn new_with_values(values: HashMap) -> Self { + Self { values } + } + + pub fn update(&mut self, key: KeyByLabelValues, value: f64) { + let current = self.values.entry(key).or_insert(f64::NEG_INFINITY); + if value > *current { + *current = value; + } + } + + pub fn add_value(&mut self, key: KeyByLabelValues, value: f64) { + self.values.insert(key, value); + } + + pub fn deserialize_from_json(data: &Value) -> Result> { + let values_data = data["values"] + .as_object() + .ok_or("Missing or invalid 'values' field")?; + + let mut values = HashMap::new(); + for (key_str, value) in values_data { + let key_json: Value = serde_json::from_str(key_str)?; + let key = KeyByLabelValues::deserialize_from_json(&key_json)?; + let val = value.as_f64().ok_or("Invalid value")?; + values.insert(key, val); + } + + Ok(Self { values }) + } + + pub fn deserialize_from_bytes(buffer: &[u8]) -> Result> { + let mut offset = 0; + + // Read number of entries + if buffer.len() < 4 { + return Err("Buffer too short for entry count".into()); + } + let num_entries = u32::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + ]) as usize; + offset += 4; + + let mut values = HashMap::new(); + + for _ in 0..num_entries { + // Read key length and data + if buffer.len() < offset + 4 { + return Err("Buffer too short for key length".into()); + } + let key_length = u32::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + ]) as usize; + offset += 4; + + if buffer.len() < offset + key_length { + return Err("Buffer too short for key data".into()); + } + let key = + KeyByLabelValues::deserialize_from_bytes(&buffer[offset..offset + key_length])?; + offset += key_length; + + // Read value + if buffer.len() < offset + 8 { + return Err("Buffer too short for value".into()); + } + let value = f64::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + buffer[offset + 4], + buffer[offset + 5], + buffer[offset + 6], + buffer[offset + 7], + ]); + offset += 8; + + values.insert(key, value); + } + + Ok(Self { values }) + } +} + +impl SerializableToSink for KeyedMaxState { + fn serialize_to_json(&self) -> Value { + let mut values_obj = serde_json::Map::new(); + for (key, value) in &self.values { + let key_json = key.serialize_to_json(); + let key_str = serde_json::to_string(&key_json).unwrap(); + values_obj.insert( + key_str, + Value::Number(serde_json::Number::from_f64(*value).unwrap()), + ); + } + + serde_json::json!({ "values": values_obj }) + } + + fn serialize_to_bytes(&self) -> Vec { + let mut buffer = Vec::new(); + + // Write number of entries + buffer.extend_from_slice(&(self.values.len() as u32).to_le_bytes()); + + // Write each key-value pair + for (key, value) in &self.values { + let key_bytes = key.serialize_to_bytes(); + + // Write key length and data + buffer.extend_from_slice(&(key_bytes.len() as u32).to_le_bytes()); + buffer.extend_from_slice(&key_bytes); + + // Write value + buffer.extend_from_slice(&value.to_le_bytes()); + } + + buffer + } +} + +impl AggregateCore for KeyedMaxState { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "KeyedMaxState" + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge KeyedMaxState with {}", + other.get_accumulator_type() + ) + .into()); + } + + let other_multiple = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to KeyedMaxState")?; + + let merged = Self::merge_accumulators(vec![self.clone(), other_multiple.clone()])?; + + Ok(Box::new(merged)) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::Max + } + + fn approx_memory_bytes(&self) -> usize { + const BYTES_PER_ENTRY: usize = 96; + std::mem::size_of::() + self.values.len() * BYTES_PER_ENTRY + } + + fn get_keys(&self) -> Option> { + Some(self.values.keys().cloned().collect()) + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + key: &Option, + query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::MultipleSubpopulationAggregate; + let key_val = key.as_ref().ok_or("Key required for KeyedMaxState")?; + self.query(statistic, key_val, Some(query_kwargs)) + } +} + +impl MultipleSubpopulationAggregate for KeyedMaxState { + fn query( + &self, + statistic: Statistic, + key: &KeyByLabelValues, + _query_kwargs: Option<&HashMap>, + ) -> Result> { + match statistic { + Statistic::Max => self + .values + .get(key) + .copied() + .ok_or_else(|| format!("Key {key} not found in KeyedMaxState").into()), + other => Err(format!("Unsupported statistic in KeyedMaxState: {other:?}").into()), + } + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +impl MergeableAccumulator for KeyedMaxState { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + + let mut result = KeyedMaxState::new(); + + for acc in accumulators { + for (key, value) in acc.values { + result.update(key, value); + } + } + + Ok(result) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn key(value: &str) -> KeyByLabelValues { + KeyByLabelValues::new_with_labels(vec![value.to_string()]) + } + + #[test] + fn keeps_the_largest_per_key() { + let mut acc = KeyedMaxState::new(); + acc.update(key("a"), 10.0); + acc.update(key("a"), 5.0); + acc.update(key("a"), 15.0); + acc.update(key("b"), 7.0); + + assert_eq!(acc.query(Statistic::Max, &key("a"), None).unwrap(), 15.0); + assert_eq!(acc.query(Statistic::Max, &key("b"), None).unwrap(), 7.0); + } + + #[test] + fn refuses_the_opposite_statistic_and_unknown_keys() { + let mut acc = KeyedMaxState::new(); + acc.update(key("a"), 1.0); + assert!(acc.query(Statistic::Min, &key("a"), None).is_err()); + assert!(acc.query(Statistic::Max, &key("missing"), None).is_err()); + } + + #[test] + fn merges_per_key() { + let mut left = KeyedMaxState::new(); + left.update(key("a"), 10.0); + let mut right = KeyedMaxState::new(); + right.update(key("a"), 5.0); + right.update(key("b"), 3.0); + + let merged = + >::merge_accumulators(vec![ + left, right, + ]) + .unwrap(); + + assert_eq!(merged.query(Statistic::Max, &key("a"), None).unwrap(), 10.0); + assert_eq!(merged.query(Statistic::Max, &key("b"), None).unwrap(), 3.0); + } + + #[test] + fn refuses_to_merge_with_the_opposite_direction() { + use super::super::keyed_min_state::KeyedMinState; + let mine = KeyedMaxState::new(); + let theirs = KeyedMinState::new(); + assert!(mine.merge_with(&theirs).is_err()); + } + + #[test] + fn round_trips_through_both_serializations() { + let mut acc = KeyedMaxState::new(); + acc.update(key("a"), 4.0); + + let json = acc.serialize_to_json(); + let from_json = KeyedMaxState::deserialize_from_json(&json).unwrap(); + assert_eq!( + from_json.query(Statistic::Max, &key("a"), None).unwrap(), + 4.0 + ); + + let bytes = acc.serialize_to_bytes(); + let from_bytes = KeyedMaxState::deserialize_from_bytes(&bytes).unwrap(); + assert_eq!( + from_bytes.query(Statistic::Max, &key("a"), None).unwrap(), + 4.0 + ); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/keyed_min_state.rs b/crates/asap-physical-operators/src/accumulators/keyed_min_state.rs new file mode 100644 index 00000000..5c40da24 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/keyed_min_state.rs @@ -0,0 +1,335 @@ +use crate::{ + AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, + MultipleSubpopulationAggregate, SerializableToSink, +}; +use serde::{Deserialize, Serialize}; +use serde_json::Value; +use std::collections::HashMap; + +use crate::Statistic; + +/// Exact per-key minimum over many populations, mergeable by comparison. +/// +/// The maximum direction is +/// [`KeyedMaxState`](super::keyed_max_state::KeyedMaxState), +/// a separate type: these used to be one `MultipleMinMaxAccumulator` whose +/// direction lived in a `sub_type` string that every layer above had to carry +/// alongside the family. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct KeyedMinState { + pub values: HashMap, +} + +impl KeyedMinState { + pub fn new() -> Self { + Self::default() + } + + pub fn new_with_values(values: HashMap) -> Self { + Self { values } + } + + pub fn update(&mut self, key: KeyByLabelValues, value: f64) { + let current = self.values.entry(key).or_insert(f64::INFINITY); + if value < *current { + *current = value; + } + } + + pub fn add_value(&mut self, key: KeyByLabelValues, value: f64) { + self.values.insert(key, value); + } + + pub fn deserialize_from_json(data: &Value) -> Result> { + let values_data = data["values"] + .as_object() + .ok_or("Missing or invalid 'values' field")?; + + let mut values = HashMap::new(); + for (key_str, value) in values_data { + let key_json: Value = serde_json::from_str(key_str)?; + let key = KeyByLabelValues::deserialize_from_json(&key_json)?; + let val = value.as_f64().ok_or("Invalid value")?; + values.insert(key, val); + } + + Ok(Self { values }) + } + + pub fn deserialize_from_bytes(buffer: &[u8]) -> Result> { + let mut offset = 0; + + // Read number of entries + if buffer.len() < 4 { + return Err("Buffer too short for entry count".into()); + } + let num_entries = u32::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + ]) as usize; + offset += 4; + + let mut values = HashMap::new(); + + for _ in 0..num_entries { + // Read key length and data + if buffer.len() < offset + 4 { + return Err("Buffer too short for key length".into()); + } + let key_length = u32::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + ]) as usize; + offset += 4; + + if buffer.len() < offset + key_length { + return Err("Buffer too short for key data".into()); + } + let key = + KeyByLabelValues::deserialize_from_bytes(&buffer[offset..offset + key_length])?; + offset += key_length; + + // Read value + if buffer.len() < offset + 8 { + return Err("Buffer too short for value".into()); + } + let value = f64::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + buffer[offset + 4], + buffer[offset + 5], + buffer[offset + 6], + buffer[offset + 7], + ]); + offset += 8; + + values.insert(key, value); + } + + Ok(Self { values }) + } +} + +impl SerializableToSink for KeyedMinState { + fn serialize_to_json(&self) -> Value { + let mut values_obj = serde_json::Map::new(); + for (key, value) in &self.values { + let key_json = key.serialize_to_json(); + let key_str = serde_json::to_string(&key_json).unwrap(); + values_obj.insert( + key_str, + Value::Number(serde_json::Number::from_f64(*value).unwrap()), + ); + } + + serde_json::json!({ "values": values_obj }) + } + + fn serialize_to_bytes(&self) -> Vec { + let mut buffer = Vec::new(); + + // Write number of entries + buffer.extend_from_slice(&(self.values.len() as u32).to_le_bytes()); + + // Write each key-value pair + for (key, value) in &self.values { + let key_bytes = key.serialize_to_bytes(); + + // Write key length and data + buffer.extend_from_slice(&(key_bytes.len() as u32).to_le_bytes()); + buffer.extend_from_slice(&key_bytes); + + // Write value + buffer.extend_from_slice(&value.to_le_bytes()); + } + + buffer + } +} + +impl AggregateCore for KeyedMinState { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "KeyedMinState" + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge KeyedMinState with {}", + other.get_accumulator_type() + ) + .into()); + } + + let other_multiple = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to KeyedMinState")?; + + let merged = Self::merge_accumulators(vec![self.clone(), other_multiple.clone()])?; + + Ok(Box::new(merged)) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::Min + } + + fn approx_memory_bytes(&self) -> usize { + const BYTES_PER_ENTRY: usize = 96; + std::mem::size_of::() + self.values.len() * BYTES_PER_ENTRY + } + + fn get_keys(&self) -> Option> { + Some(self.values.keys().cloned().collect()) + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + key: &Option, + query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::MultipleSubpopulationAggregate; + let key_val = key.as_ref().ok_or("Key required for KeyedMinState")?; + self.query(statistic, key_val, Some(query_kwargs)) + } +} + +impl MultipleSubpopulationAggregate for KeyedMinState { + fn query( + &self, + statistic: Statistic, + key: &KeyByLabelValues, + _query_kwargs: Option<&HashMap>, + ) -> Result> { + match statistic { + Statistic::Min => self + .values + .get(key) + .copied() + .ok_or_else(|| format!("Key {key} not found in KeyedMinState").into()), + other => Err(format!("Unsupported statistic in KeyedMinState: {other:?}").into()), + } + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +impl MergeableAccumulator for KeyedMinState { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + + let mut result = KeyedMinState::new(); + + for acc in accumulators { + for (key, value) in acc.values { + result.update(key, value); + } + } + + Ok(result) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn key(value: &str) -> KeyByLabelValues { + KeyByLabelValues::new_with_labels(vec![value.to_string()]) + } + + #[test] + fn keeps_the_smallest_per_key() { + let mut acc = KeyedMinState::new(); + acc.update(key("a"), 10.0); + acc.update(key("a"), 5.0); + acc.update(key("a"), 15.0); + acc.update(key("b"), 7.0); + + assert_eq!(acc.query(Statistic::Min, &key("a"), None).unwrap(), 5.0); + assert_eq!(acc.query(Statistic::Min, &key("b"), None).unwrap(), 7.0); + } + + #[test] + fn refuses_the_opposite_statistic_and_unknown_keys() { + let mut acc = KeyedMinState::new(); + acc.update(key("a"), 1.0); + assert!(acc.query(Statistic::Max, &key("a"), None).is_err()); + assert!(acc.query(Statistic::Min, &key("missing"), None).is_err()); + } + + #[test] + fn merges_per_key() { + let mut left = KeyedMinState::new(); + left.update(key("a"), 10.0); + let mut right = KeyedMinState::new(); + right.update(key("a"), 5.0); + right.update(key("b"), 3.0); + + let merged = + >::merge_accumulators(vec![ + left, right, + ]) + .unwrap(); + + assert_eq!(merged.query(Statistic::Min, &key("a"), None).unwrap(), 5.0); + assert_eq!(merged.query(Statistic::Min, &key("b"), None).unwrap(), 3.0); + } + + #[test] + fn refuses_to_merge_with_the_opposite_direction() { + use super::super::keyed_max_state::KeyedMaxState; + let mine = KeyedMinState::new(); + let theirs = KeyedMaxState::new(); + assert!(mine.merge_with(&theirs).is_err()); + } + + #[test] + fn round_trips_through_both_serializations() { + let mut acc = KeyedMinState::new(); + acc.update(key("a"), 4.0); + + let json = acc.serialize_to_json(); + let from_json = KeyedMinState::deserialize_from_json(&json).unwrap(); + assert_eq!( + from_json.query(Statistic::Min, &key("a"), None).unwrap(), + 4.0 + ); + + let bytes = acc.serialize_to_bytes(); + let from_bytes = KeyedMinState::deserialize_from_bytes(&bytes).unwrap(); + assert_eq!( + from_bytes.query(Statistic::Min, &key("a"), None).unwrap(), + 4.0 + ); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/keyed_sum_count_accumulator.rs b/crates/asap-physical-operators/src/accumulators/keyed_sum_count_accumulator.rs new file mode 100644 index 00000000..7bf2b5ce --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/keyed_sum_count_accumulator.rs @@ -0,0 +1,558 @@ +use crate::{ + AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, + MultipleSubpopulationAggregate, SerializableToSink, +}; +use serde::{Deserialize, Serialize}; +use serde_json::Value; +use std::collections::HashMap; + +use crate::Statistic; +use planner_types::post_asap::ExactKind; + +fn sum_family() -> ExactKind { + ExactKind::Sum +} + +/// Accumulator that maintains separate sum values for multiple keys +/// Allows querying sums for specific label combinations +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct KeyedSumCountAccumulator { + #[serde(default = "sum_family")] + pub family: ExactKind, + pub sums: HashMap, + #[serde(default)] + pub counts: HashMap, +} + +impl KeyedSumCountAccumulator { + pub fn new() -> Self { + Self::for_family(ExactKind::Sum) + } + + pub fn for_family(family: ExactKind) -> Self { + assert!(matches!(family, ExactKind::Sum | ExactKind::Count)); + Self { + family, + sums: HashMap::new(), + counts: HashMap::new(), + } + } + + pub fn update(&mut self, key: KeyByLabelValues, value: f64) { + let is_new = !self.sums.contains_key(&key); + *self.sums.entry(key.clone()).or_insert(0.0) += value; + if let Some(count) = self.counts.get(&key).copied() { + if let Some(next) = count.checked_add(1).filter(|next| *next != u64::MAX) { + self.counts.insert(key, next); + } else { + self.counts.remove(&key); + } + } else if is_new { + self.counts.insert(key, 1); + } + } + + pub fn add_sum(&mut self, key: KeyByLabelValues, sum: f64) { + self.counts.remove(&key); + self.sums.insert(key, sum); + } + + pub fn deserialize_from_json(data: &Value) -> Result> { + let sums_data = data["sums"] + .as_object() + .ok_or("Missing or invalid 'sums' field")?; + + let mut sums = HashMap::new(); + for (key_str, value) in sums_data { + let key_json: Value = serde_json::from_str(key_str)?; + let key = KeyByLabelValues::deserialize_from_json(&key_json)?; + let sum = value.as_f64().ok_or("Invalid sum value")?; + sums.insert(key, sum); + } + + let mut counts = HashMap::new(); + if let Some(counts_data) = data.get("counts").and_then(Value::as_object) { + for (key_str, value) in counts_data { + let key_json: Value = serde_json::from_str(key_str)?; + let key = KeyByLabelValues::deserialize_from_json(&key_json)?; + let count = value.as_u64().ok_or("Invalid count value")?; + if !sums.contains_key(&key) { + return Err("Count key missing from sums".into()); + } + counts.insert(key, count); + } + } + let family = match data.get("family").and_then(Value::as_str) { + None | Some("Sum") => ExactKind::Sum, + Some("Count") => ExactKind::Count, + _ => return Err("Invalid keyed additive family".into()), + }; + Ok(Self { + family, + sums, + counts, + }) + } + + pub fn deserialize_from_bytes(buffer: &[u8]) -> Result> { + let mut offset = 0; + + // Read number of entries + if buffer.len() < 4 { + return Err("Buffer too short for entry count".into()); + } + let num_entries = u32::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + ]) as usize; + offset += 4; + + let mut sums = HashMap::new(); + let mut keys = Vec::new(); + + for _ in 0..num_entries { + // Read key length and data + if buffer.len() < offset + 4 { + return Err("Buffer too short for key length".into()); + } + let key_length = u32::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + ]) as usize; + offset += 4; + + if buffer.len() < offset + key_length { + return Err("Buffer too short for key data".into()); + } + let key = + KeyByLabelValues::deserialize_from_bytes(&buffer[offset..offset + key_length])?; + offset += key_length; + + // Read sum value + if buffer.len() < offset + 8 { + return Err("Buffer too short for sum value".into()); + } + let sum = f64::from_le_bytes([ + buffer[offset], + buffer[offset + 1], + buffer[offset + 2], + buffer[offset + 3], + buffer[offset + 4], + buffer[offset + 5], + buffer[offset + 6], + buffer[offset + 7], + ]); + offset += 8; + + keys.push(key.clone()); + sums.insert(key, sum); + } + let remaining = buffer.len() - offset; + let count_bytes = num_entries + .checked_mul(8) + .ok_or("Count section too large")?; + if remaining != 0 && remaining != count_bytes && remaining != count_bytes + 1 { + return Err("Invalid count section length".into()); + } + let mut counts = HashMap::new(); + if count_bytes != 0 && remaining >= count_bytes { + for key in keys { + let count = u64::from_le_bytes(buffer[offset..offset + 8].try_into()?); + offset += 8; + if count != u64::MAX { + counts.insert(key, count); + } + } + } + let family = if remaining == count_bytes + 1 { + match buffer[offset] { + 0 => ExactKind::Sum, + 1 => ExactKind::Count, + _ => return Err("Invalid keyed additive family tag".into()), + } + } else { + ExactKind::Sum + }; + Ok(Self { + family, + sums, + counts, + }) + } +} + +impl Default for KeyedSumCountAccumulator { + fn default() -> Self { + Self::new() + } +} + +impl SerializableToSink for KeyedSumCountAccumulator { + fn serialize_to_json(&self) -> Value { + let mut sums_obj = serde_json::Map::new(); + for (key, sum) in &self.sums { + let key_json = key.serialize_to_json(); + let key_str = serde_json::to_string(&key_json).unwrap(); + sums_obj.insert( + key_str, + Value::Number(serde_json::Number::from_f64(*sum).unwrap()), + ); + } + + let mut counts_obj = serde_json::Map::new(); + for (key, count) in &self.counts { + let key_str = serde_json::to_string(&key.serialize_to_json()).unwrap(); + counts_obj.insert(key_str, Value::from(*count)); + } + + serde_json::json!({ + "family": if self.family == ExactKind::Count { "Count" } else { "Sum" }, + "sums": sums_obj, + "counts": counts_obj + }) + } + + fn serialize_to_bytes(&self) -> Vec { + let mut buffer = Vec::new(); + + // Write number of entries + buffer.extend_from_slice(&(self.sums.len() as u32).to_le_bytes()); + + // Write each key-value pair + let mut ordered_keys = Vec::with_capacity(self.sums.len()); + for (key, sum) in &self.sums { + ordered_keys.push(key); + let key_bytes = key.serialize_to_bytes(); + + // Write key length and data + buffer.extend_from_slice(&(key_bytes.len() as u32).to_le_bytes()); + buffer.extend_from_slice(&key_bytes); + + // Write sum value + buffer.extend_from_slice(&sum.to_le_bytes()); + } + + for key in ordered_keys { + buffer.extend_from_slice( + &self + .counts + .get(key) + .copied() + .unwrap_or(u64::MAX) + .to_le_bytes(), + ); + } + + buffer.push(if self.family == ExactKind::Count { + 1 + } else { + 0 + }); + + buffer + } +} + +impl AggregateCore for KeyedSumCountAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "KeyedSumCountAccumulator" + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + // Check if other is also a KeyedSumCountAccumulator + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge KeyedSumCountAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + + // Downcast to KeyedSumCountAccumulator + let other_multiple_sum = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to KeyedSumCountAccumulator")?; + + // Use the existing merge_accumulators method + let merged = Self::merge_accumulators(vec![self.clone(), other_multiple_sum.clone()])?; + + Ok(Box::new(merged)) + } + + fn get_accumulator_type(&self) -> AggregationType { + if self.family == ExactKind::Count { + AggregationType::Count + } else { + AggregationType::Sum + } + } + + fn approx_memory_bytes(&self) -> usize { + // HashMap. Label strings dominate; use a + // conservative per-entry estimate plus HashMap overhead. + const BYTES_PER_ENTRY: usize = 112; + std::mem::size_of::() + self.sums.len() * BYTES_PER_ENTRY + } + + fn get_keys(&self) -> Option> { + Some(self.sums.keys().cloned().collect()) + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + key: &Option, + query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::MultipleSubpopulationAggregate; + let key_val = key + .as_ref() + .ok_or("Key required for KeyedSumCountAccumulator")?; + self.query(statistic, key_val, Some(query_kwargs)) + } +} + +impl MultipleSubpopulationAggregate for KeyedSumCountAccumulator { + fn query( + &self, + statistic: Statistic, + key: &KeyByLabelValues, + _query_kwargs: Option<&HashMap>, + ) -> Result> { + match (&self.family, statistic) { + (ExactKind::Sum, Statistic::Sum) => self.sums.get(key).copied().ok_or_else(|| { + "Key not found in KeyedSumCountAccumulator" + .to_string() + .into() + }), + (ExactKind::Count, Statistic::Count) => self + .counts + .get(key) + .map(|count| *count as f64) + .ok_or_else(|| { + "Sample count unavailable in KeyedSumCountAccumulator" + .to_string() + .into() + }), + _ => Err( + format!("Unsupported statistic in KeyedSumCountAccumulator: {statistic:?}").into(), + ), + } + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +impl MergeableAccumulator for KeyedSumCountAccumulator { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + + let family = accumulators[0].family.clone(); + if accumulators.iter().any(|acc| acc.family != family) { + return Err("Cannot merge different keyed additive families".into()); + } + let mut result = KeyedSumCountAccumulator::for_family(family); + + for acc in accumulators { + for key in acc.sums.keys() { + match ( + result.counts.get(key).copied(), + acc.counts.get(key).copied(), + ) { + (None, Some(count)) if !result.sums.contains_key(key) => { + result.counts.insert(key.clone(), count); + } + (Some(existing), Some(count)) => { + if let Some(total) = existing.checked_add(count) { + result.counts.insert(key.clone(), total); + } else { + result.counts.remove(key); + } + } + _ => { + result.counts.remove(key); + } + } + } + for (key, sum) in acc.sums { + *result.sums.entry(key).or_insert(0.0) += sum; + } + } + + Ok(result) + } +} + +#[cfg(test)] +mod tests { + use std::vec; + + use super::*; + + #[test] + fn test_keyed_sum_count_accumulator_creation() { + let acc = KeyedSumCountAccumulator::new(); + assert!(acc.sums.is_empty()); + } + + #[test] + fn test_keyed_sum_count_accumulator_update() { + let mut acc = KeyedSumCountAccumulator::new(); + + let key1 = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + + let key2 = KeyByLabelValues::new_with_labels(vec!["api".to_string()]); + + acc.update(key1.clone(), 10.0); + acc.update(key2.clone(), 20.0); + acc.update(key1.clone(), 5.0); // Should add to existing + + assert_eq!(acc.sums.get(&key1), Some(&15.0)); + assert_eq!(acc.sums.get(&key2), Some(&20.0)); + } + + #[test] + fn grouped_count_reads_sample_count_and_survives_merge_and_round_trip() { + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + let mut first = KeyedSumCountAccumulator::for_family(ExactKind::Count); + first.update(key.clone(), 10.0); + first.update(key.clone(), 20.0); + let mut second = KeyedSumCountAccumulator::for_family(ExactKind::Count); + second.update(key.clone(), 7.0); + let merged = KeyedSumCountAccumulator::merge_accumulators(vec![first, second]).unwrap(); + for acc in [ + merged.clone(), + KeyedSumCountAccumulator::deserialize_from_json(&merged.serialize_to_json()).unwrap(), + KeyedSumCountAccumulator::deserialize_from_bytes(&merged.serialize_to_bytes()).unwrap(), + ] { + assert_eq!(acc.family, ExactKind::Count); + assert!(acc.query(Statistic::Sum, &key, None).is_err()); + assert_eq!(acc.query(Statistic::Count, &key, None).unwrap(), 3.0); + } + } + + #[test] + fn keyed_additive_merge_rejects_different_planner_families() { + assert!(KeyedSumCountAccumulator::merge_accumulators(vec![ + KeyedSumCountAccumulator::for_family(ExactKind::Sum), + KeyedSumCountAccumulator::for_family(ExactKind::Count), + ]) + .is_err()); + } + + #[test] + fn test_keyed_sum_count_accumulator_query() { + let mut acc = KeyedSumCountAccumulator::new(); + + let key = KeyByLabelValues::new_with_labels(vec!["service".to_string()]); + + acc.add_sum(key.clone(), 42.0); + + // Test total queries (querying with the specific key) + assert_eq!( + crate::MultipleSubpopulationAggregate::query(&acc, Statistic::Sum, &key, None).unwrap(), + 42.0 + ); + + // Test error cases + assert!( + crate::MultipleSubpopulationAggregate::query(&acc, Statistic::Min, &key, None).is_err() + ); + } + + #[test] + fn test_keyed_sum_count_accumulator_get_keys() { + let mut acc = KeyedSumCountAccumulator::new(); + + let key1 = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + + let key2 = KeyByLabelValues::new_with_labels(vec!["api".to_string()]); + + acc.add_sum(key1.clone(), 10.0); + acc.add_sum(key2.clone(), 20.0); + + let keys = crate::AggregateCore::get_keys(&acc).unwrap(); + assert_eq!(keys.len(), 2); + assert!(keys.contains(&key1)); + assert!(keys.contains(&key2)); + } + + #[test] + fn test_keyed_sum_count_accumulator_merge() { + let mut acc1 = KeyedSumCountAccumulator::new(); + let mut acc2 = KeyedSumCountAccumulator::new(); + + let key1 = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + + let key2 = KeyByLabelValues::new_with_labels(vec!["api".to_string()]); + + acc1.add_sum(key1.clone(), 10.0); + acc1.add_sum(key2.clone(), 20.0); + + acc2.add_sum(key1.clone(), 5.0); // Same key, different accumulator + + let merged = >::merge_accumulators(vec![acc1, acc2]).unwrap(); + + assert_eq!(merged.sums.get(&key1), Some(&15.0)); // Should be merged + assert_eq!(merged.sums.get(&key2), Some(&20.0)); // Should be preserved + } + + #[test] + fn test_keyed_sum_count_accumulator_serialization() { + let mut acc = KeyedSumCountAccumulator::new(); + + let key = KeyByLabelValues::new_with_labels(vec!["service".to_string()]); + + acc.add_sum(key.clone(), 42.5); + + // Test JSON serialization + let json = acc.serialize_to_json(); + let deserialized = KeyedSumCountAccumulator::deserialize_from_json(&json).unwrap(); + assert_eq!(deserialized.sums.get(&key), Some(&42.5)); + + // Test byte serialization + let bytes = acc.serialize_to_bytes(); + let deserialized_bytes = KeyedSumCountAccumulator::deserialize_from_bytes(&bytes).unwrap(); + assert_eq!(deserialized_bytes.sums.get(&key), Some(&42.5)); + } + + #[test] + fn test_trait_object() { + let mut acc = KeyedSumCountAccumulator::new(); + + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + + acc.add_sum(key.clone(), 42.0); + + let trait_obj: Box = Box::new(acc); + + // Test type name through trait object + assert_eq!(trait_obj.type_name(), "KeyedSumCountAccumulator"); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/max_accumulator.rs b/crates/asap-physical-operators/src/accumulators/max_accumulator.rs new file mode 100644 index 00000000..8f6e2af7 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/max_accumulator.rs @@ -0,0 +1,248 @@ +use crate::{ + AggregateCore, AggregationType, AuxStats, MergeableAccumulator, SerializableToSink, + SingleSubpopulationAggregate, +}; +use serde::{Deserialize, Serialize}; +use serde_json::Value; +use std::collections::HashMap; + +use crate::Statistic; + +/// Exact maximum over one population, mergeable by comparison. +/// +/// See [`MinAccumulator`](super::min_accumulator::MinAccumulator) for why the +/// two directions are separate types rather than one accumulator carrying a +/// `sub_type` string. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct MaxAccumulator { + pub value: f64, +} + +impl Default for MaxAccumulator { + fn default() -> Self { + Self::new() + } +} + +impl MaxAccumulator { + pub fn new() -> Self { + Self { + value: f64::NEG_INFINITY, + } + } + + pub fn with_value(value: f64) -> Self { + Self { value } + } + + pub fn update(&mut self, value: f64) { + if value > self.value { + self.value = value; + } + } + + pub fn deserialize_from_json(data: &Value) -> Result> { + let value = data["value"] + .as_f64() + .ok_or("Missing or invalid 'value' field")?; + Ok(Self::with_value(value)) + } + + pub fn deserialize_from_bytes(buffer: &[u8]) -> Result> { + if buffer.len() < 8 { + return Err("Buffer too short".into()); + } + let value = f64::from_le_bytes([ + buffer[0], buffer[1], buffer[2], buffer[3], buffer[4], buffer[5], buffer[6], buffer[7], + ]); + Ok(Self::with_value(value)) + } +} + +impl SerializableToSink for MaxAccumulator { + fn serialize_to_json(&self) -> Value { + serde_json::json!({ "value": self.value }) + } + + fn serialize_to_bytes(&self) -> Vec { + self.value.to_le_bytes().to_vec() + } +} + +impl MergeableAccumulator for MaxAccumulator { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + let mut result = MaxAccumulator::new(); + for acc in accumulators { + result.update(acc.value); + } + Ok(result) + } +} + +impl AggregateCore for MaxAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "MaxAccumulator" + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge MaxAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + let other_max = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to MaxAccumulator")?; + let mut merged = self.clone(); + merged.update(other_max.value); + Ok(Box::new(merged)) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::Max + } + + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::() + } + + fn aux_stats(&self) -> AuxStats { + // The sentinel `f64::NEG_INFINITY` from `new()` is surfaced as-is; the + // query engine already treats it as "no data yet", the same way it + // does for `query_statistic`. + AuxStats { + max: Some(self.value), + ..AuxStats::empty() + } + } + + fn get_keys(&self) -> Option> { + None + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + _key: &Option, + _query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::SingleSubpopulationAggregate; + self.query(statistic, None) + } +} + +impl SingleSubpopulationAggregate for MaxAccumulator { + fn query( + &self, + statistic: Statistic, + query_kwargs: Option<&HashMap>, + ) -> Result> { + if query_kwargs.is_some() { + return Err("MaxAccumulator does not support query parameters".into()); + } + match statistic { + Statistic::Max => Ok(self.value), + other => Err(format!("Unsupported statistic in MaxAccumulator: {other:?}").into()), + } + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn keeps_the_largest_update() { + let mut acc = MaxAccumulator::new(); + acc.update(10.0); + acc.update(5.0); + acc.update(15.0); + + assert_eq!(acc.value, 15.0); + assert_eq!( + crate::SingleSubpopulationAggregate::query(&acc, Statistic::Max, None).unwrap(), + 15.0 + ); + } + + #[test] + fn refuses_to_answer_a_minimum_query() { + let acc = MaxAccumulator::with_value(15.0); + assert!(crate::SingleSubpopulationAggregate::query(&acc, Statistic::Min, None).is_err()); + } + + #[test] + fn merges_by_taking_the_largest() { + let merged = + >::merge_accumulators(vec![ + MaxAccumulator::with_value(10.0), + MaxAccumulator::with_value(5.0), + MaxAccumulator::with_value(15.0), + ]) + .unwrap(); + assert_eq!(merged.value, 15.0); + } + + #[test] + fn refuses_to_merge_with_a_minimum() { + use super::super::min_accumulator::MinAccumulator; + let max = MaxAccumulator::with_value(15.0); + let min = MinAccumulator::with_value(5.0); + assert!(max.merge_with(&min).is_err()); + } + + #[test] + fn round_trips_through_both_serializations() { + let acc = MaxAccumulator::with_value(42.5); + + let json = acc.serialize_to_json(); + assert_eq!( + MaxAccumulator::deserialize_from_json(&json).unwrap().value, + 42.5 + ); + + let bytes = acc.serialize_to_bytes(); + assert_eq!( + MaxAccumulator::deserialize_from_bytes(&bytes) + .unwrap() + .value, + 42.5 + ); + } + + #[test] + fn aux_stats_expose_max_only() { + let aux = MaxAccumulator::with_value(99.0).aux_stats(); + assert_eq!(aux.max, Some(99.0)); + assert_eq!(aux.min, None); + assert_eq!(aux.try_answer(Statistic::Max), Some(99.0)); + assert_eq!(aux.try_answer(Statistic::Min), None); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/min_accumulator.rs b/crates/asap-physical-operators/src/accumulators/min_accumulator.rs new file mode 100644 index 00000000..79343858 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/min_accumulator.rs @@ -0,0 +1,253 @@ +use crate::{ + AggregateCore, AggregationType, AuxStats, MergeableAccumulator, SerializableToSink, + SingleSubpopulationAggregate, +}; +use serde::{Deserialize, Serialize}; +use serde_json::Value; +use std::collections::HashMap; + +use crate::Statistic; + +/// Exact minimum over one population, mergeable by comparison. +/// +/// The sibling [`MaxAccumulator`](super::max_accumulator::MaxAccumulator) is a +/// separate type on purpose: these two used to be one `MinMaxAccumulator` +/// whose direction lived in a `sub_type: String`, which meant every layer +/// above -- the wire `aggregationSubType`, the accumulator factory, the +/// summary catalog -- had to carry the direction alongside the family and +/// could silently answer a `min_over_time` read from maximum state. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct MinAccumulator { + pub value: f64, +} + +impl Default for MinAccumulator { + fn default() -> Self { + Self::new() + } +} + +impl MinAccumulator { + pub fn new() -> Self { + Self { + value: f64::INFINITY, + } + } + + pub fn with_value(value: f64) -> Self { + Self { value } + } + + pub fn update(&mut self, value: f64) { + if value < self.value { + self.value = value; + } + } + + pub fn deserialize_from_json(data: &Value) -> Result> { + let value = data["value"] + .as_f64() + .ok_or("Missing or invalid 'value' field")?; + Ok(Self::with_value(value)) + } + + pub fn deserialize_from_bytes(buffer: &[u8]) -> Result> { + if buffer.len() < 8 { + return Err("Buffer too short".into()); + } + let value = f64::from_le_bytes([ + buffer[0], buffer[1], buffer[2], buffer[3], buffer[4], buffer[5], buffer[6], buffer[7], + ]); + Ok(Self::with_value(value)) + } +} + +impl SerializableToSink for MinAccumulator { + fn serialize_to_json(&self) -> Value { + serde_json::json!({ "value": self.value }) + } + + fn serialize_to_bytes(&self) -> Vec { + self.value.to_le_bytes().to_vec() + } +} + +impl MergeableAccumulator for MinAccumulator { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + if accumulators.is_empty() { + return Err("No accumulators to merge".into()); + } + let mut result = MinAccumulator::new(); + for acc in accumulators { + result.update(acc.value); + } + Ok(result) + } +} + +impl AggregateCore for MinAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "MinAccumulator" + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge MinAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + let other_min = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to MinAccumulator")?; + let mut merged = self.clone(); + merged.update(other_min.value); + Ok(Box::new(merged)) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::Min + } + + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::() + } + + fn aux_stats(&self) -> AuxStats { + // The sentinel `f64::INFINITY` from `new()` is surfaced as-is; the + // query engine already treats it as "no data yet", the same way it + // does for `query_statistic`. + AuxStats { + min: Some(self.value), + ..AuxStats::empty() + } + } + + fn get_keys(&self) -> Option> { + None + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + _key: &Option, + _query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::SingleSubpopulationAggregate; + self.query(statistic, None) + } +} + +impl SingleSubpopulationAggregate for MinAccumulator { + fn query( + &self, + statistic: Statistic, + query_kwargs: Option<&HashMap>, + ) -> Result> { + if query_kwargs.is_some() { + return Err("MinAccumulator does not support query parameters".into()); + } + match statistic { + Statistic::Min => Ok(self.value), + other => Err(format!("Unsupported statistic in MinAccumulator: {other:?}").into()), + } + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn keeps_the_smallest_update() { + let mut acc = MinAccumulator::new(); + acc.update(10.0); + acc.update(5.0); + acc.update(15.0); + + assert_eq!(acc.value, 5.0); + assert_eq!( + crate::SingleSubpopulationAggregate::query(&acc, Statistic::Min, None).unwrap(), + 5.0 + ); + } + + #[test] + fn refuses_to_answer_a_maximum_query() { + let acc = MinAccumulator::with_value(5.0); + assert!(crate::SingleSubpopulationAggregate::query(&acc, Statistic::Max, None).is_err()); + } + + #[test] + fn merges_by_taking_the_smallest() { + let merged = + >::merge_accumulators(vec![ + MinAccumulator::with_value(10.0), + MinAccumulator::with_value(5.0), + MinAccumulator::with_value(15.0), + ]) + .unwrap(); + assert_eq!(merged.value, 5.0); + } + + #[test] + fn refuses_to_merge_with_a_maximum() { + use super::super::max_accumulator::MaxAccumulator; + let min = MinAccumulator::with_value(5.0); + let max = MaxAccumulator::with_value(15.0); + assert!(min.merge_with(&max).is_err()); + } + + #[test] + fn round_trips_through_both_serializations() { + let acc = MinAccumulator::with_value(42.5); + + let json = acc.serialize_to_json(); + assert_eq!( + MinAccumulator::deserialize_from_json(&json).unwrap().value, + 42.5 + ); + + let bytes = acc.serialize_to_bytes(); + assert_eq!( + MinAccumulator::deserialize_from_bytes(&bytes) + .unwrap() + .value, + 42.5 + ); + } + + #[test] + fn aux_stats_expose_min_only() { + let aux = MinAccumulator::with_value(3.5).aux_stats(); + assert_eq!(aux.min, Some(3.5)); + assert_eq!(aux.max, None); + assert_eq!(aux.count, None); + assert_eq!(aux.sum, None); + assert_eq!(aux.try_answer(Statistic::Min), Some(3.5)); + assert_eq!(aux.try_answer(Statistic::Max), None); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/mod.rs b/crates/asap-physical-operators/src/accumulators/mod.rs new file mode 100644 index 00000000..073db6e8 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/mod.rs @@ -0,0 +1,37 @@ +pub mod count_min_sketch_accumulator; +pub mod count_min_sketch_with_heap_accumulator; +pub mod count_sketch_accumulator; +pub mod count_sketch_with_heap_accumulator; +pub mod datasketches_kll_accumulator; +pub mod dd_sketch_accumulator; +pub mod exact_accumulator; +pub mod hll_sketch_accumulator; +pub mod hydra_kll_accumulator; +pub mod increase_accumulator; +pub mod keyed_counter_state; +pub mod keyed_max_state; +pub mod keyed_min_state; +pub mod keyed_sum_count_accumulator; +pub mod max_accumulator; +pub mod min_accumulator; +pub mod sketch_envelope_accumulator; +pub mod sum_accumulator; +pub mod univmon_accumulator; + +pub use count_min_sketch_accumulator::*; +pub use count_min_sketch_with_heap_accumulator::*; +pub use count_sketch_accumulator::*; +pub use count_sketch_with_heap_accumulator::*; +pub use datasketches_kll_accumulator::*; +pub use dd_sketch_accumulator::*; +pub use hll_sketch_accumulator::*; +pub use hydra_kll_accumulator::*; +pub use increase_accumulator::*; +pub use keyed_counter_state::*; +pub use keyed_max_state::*; +pub use keyed_min_state::*; +pub use keyed_sum_count_accumulator::*; +pub use max_accumulator::*; +pub use min_accumulator::*; +pub use sketch_envelope_accumulator::*; +pub use sum_accumulator::*; diff --git a/crates/asap-physical-operators/src/accumulators/sketch_envelope_accumulator.rs b/crates/asap-physical-operators/src/accumulators/sketch_envelope_accumulator.rs new file mode 100644 index 00000000..475d7f51 --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/sketch_envelope_accumulator.rs @@ -0,0 +1,154 @@ +//! SketchEnvelopeAccumulator — wraps a raw SketchEnvelope protobuf payload +//! received via OTLP ingest so it can be stored through the `Store` trait. +//! +//! The accumulator preserves the opaque proto bytes and decodes them lazily +//! (via `SketchEnvelope::decode`) only when merge or query operations need +//! the inner sketch type. + +use crate::{AggregateCore, KeyByLabelValues, SerializableToSink}; +use asap_sketchlib::proto::sketchlib::{sketch_envelope, SketchEnvelope}; +use prost::Message; +use serde_json::Value; +use std::collections::HashMap; + +use crate::AggregationType; +use crate::Statistic; + +/// Accumulator that stores a serialized `SketchEnvelope` protobuf. +/// +/// This is the simplest viable path for OTLP sketch ingest: the OTel Collector +/// has already computed the sketch, so the backend just stores the bytes and +/// serves them back at query time. +#[derive(Debug, Clone)] +pub struct SketchEnvelopeAccumulator { + /// Raw protobuf-encoded `SketchEnvelope`. + pub payload: Vec, + /// Sketch type string cached from decoding (e.g. "CountMin", "KLL"). + pub sketch_type: String, +} + +impl SketchEnvelopeAccumulator { + /// Create from raw protobuf bytes. Decodes the envelope once to cache + /// the sketch type; the full payload is kept for later use. + pub fn from_proto_bytes( + payload: Vec, + ) -> Result> { + let sketch_type = match SketchEnvelope::decode(payload.as_slice()) { + Ok(env) => match env.sketch_state { + Some(sketch_envelope::SketchState::CountMin(_)) => "CountMin".to_string(), + Some(sketch_envelope::SketchState::CountSketch(_)) => "CountSketch".to_string(), + Some(sketch_envelope::SketchState::Kll(_)) => "KLL".to_string(), + Some(sketch_envelope::SketchState::Hll(_)) => "HLL".to_string(), + Some(sketch_envelope::SketchState::Ddsketch(_)) => "DDSketch".to_string(), + Some(sketch_envelope::SketchState::Univmon(_)) => "UnivMon".to_string(), + Some(sketch_envelope::SketchState::Hydra(_)) => "Hydra".to_string(), + Some(sketch_envelope::SketchState::Coco(_)) => "CocoSketch".to_string(), + Some(sketch_envelope::SketchState::Elastic(_)) => "Elastic".to_string(), + None => "Unknown".to_string(), + }, + Err(e) => { + return Err(format!("Failed to decode SketchEnvelope: {}", e).into()); + } + }; + + Ok(Self { + payload, + sketch_type, + }) + } +} + +// --------------------------------------------------------------------------- +// Trait implementations +// --------------------------------------------------------------------------- + +impl SerializableToSink for SketchEnvelopeAccumulator { + fn serialize_to_json(&self) -> Value { + serde_json::json!({ + "type": "SketchEnvelopeAccumulator", + "sketch_type": self.sketch_type, + "payload_bytes": self.payload.len(), + }) + } + + fn serialize_to_bytes(&self) -> Vec { + self.payload.clone() + } +} + +impl AggregateCore for SketchEnvelopeAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "SketchEnvelopeAccumulator" + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge SketchEnvelopeAccumulator with {:?}", + other.get_accumulator_type() + ) + .into()); + } + + // For now, merging opaque envelopes is not supported — each window is + // a self-contained sketch produced by the OTel Collector. Return self + // as-is so the store can still call merge_with without panicking. + Ok(Box::new(self.clone())) + } + + fn get_accumulator_type(&self) -> AggregationType { + // Opaque wrapper — report as the generic multi-subpopulation bucket. + // Direct dispatch is not supported; native sketch query path must + // decode the envelope and delegate to the correct accumulator. + AggregationType::MultipleSubpopulation + } + + fn get_keys(&self) -> Option> { + None + } + + fn query_statistic( + &self, + _statistic: Statistic, + _key: &Option, + _query_kwargs: &HashMap, + ) -> Result> { + Err( + "SketchEnvelopeAccumulator: query_statistic not supported; decode envelope first" + .into(), + ) + } +} + +impl crate::MultipleSubpopulationAggregate for SketchEnvelopeAccumulator { + fn query( + &self, + _statistic: Statistic, + _key: &KeyByLabelValues, + _query_kwargs: Option<&HashMap>, + ) -> Result> { + Err( + "SketchEnvelopeAccumulator: direct query not supported; use native sketch query path" + .into(), + ) + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} diff --git a/crates/asap-physical-operators/src/accumulators/sum_accumulator.rs b/crates/asap-physical-operators/src/accumulators/sum_accumulator.rs new file mode 100644 index 00000000..4e74a45a --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/sum_accumulator.rs @@ -0,0 +1,413 @@ +use crate::{ + AggregateCore, AggregationType, AuxStats, MergeableAccumulator, SerializableToSink, + SingleSubpopulationAggregate, +}; +use serde::{Deserialize, Serialize}; +use serde_json::Value; +use std::collections::HashMap; + +use crate::Statistic; + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct SumAccumulator { + pub sum: f64, + /// None for scalar-only payloads; a sum does not establish a sample count. + #[serde(default)] + pub observation_count: Option, +} + +impl SumAccumulator { + pub fn new() -> Self { + Self { + sum: 0.0, + observation_count: Some(0), + } + } + + pub fn with_sum(sum: f64) -> Self { + Self { + sum, + observation_count: None, + } + } + + pub fn update(&mut self, value: f64) { + self.sum += value; + self.observation_count = self + .observation_count + .and_then(|count| count.checked_add(1)); + } + + pub fn deserialize_from_json(data: &Value) -> Result> { + let sum = data["sum"] + .as_f64() + .ok_or("Missing or invalid 'sum' field")?; + Ok(Self { + sum, + observation_count: data.get("observation_count").and_then(Value::as_u64), + }) + } + + pub fn deserialize_from_bytes(buffer: &[u8]) -> Result> { + match buffer.len() { + // Legacy Python scalar sums carry no sample-count evidence. + 4 => Ok(Self::with_sum(f32::from_le_bytes(buffer.try_into()?) as f64)), + // Counted sums use the same fixed layout as the Collector Sum payload. + 16 => Self::from_sum_bytes(buffer), + len => { + Err(format!("Invalid persisted Sum payload length: {len} (want 4 or 16)").into()) + } + } + } + + /// Decode the fixed Sum payload produced by the first-class Sum + /// AggregationType path (asap-precompute-go's SumWrapper): float64 sum + /// (little-endian) followed by uint64 count (little-endian), 16 bytes. + /// + /// Sum is an aggregation, NOT a sketch, so this deliberately does NOT + /// depend on the sketchlib sketch-envelope proto — the payload is a small + /// self-contained fixed layout. It decodes into the SAME + /// `AggregationType::Sum` accumulator as a plain-OTLP Sum, so the SumAgg + /// envelope and a plain Sum land on one identity (`exact_agg:Sum`) with no + /// new SketchAlgorithm. The supplied observation count is retained for + /// exact sample-count readouts; scalar-only legacy payloads leave it unknown. + pub fn from_sum_bytes(buffer: &[u8]) -> Result> { + if buffer.len() < 16 { + return Err(format!("Sum payload too short: {} bytes (want 16)", buffer.len()).into()); + } + let sum = f64::from_le_bytes(buffer[0..8].try_into().unwrap()); + let count = u64::from_le_bytes(buffer[8..16].try_into().unwrap()); + Ok(Self { + sum, + observation_count: Some(count), + }) + } +} + +impl Default for SumAccumulator { + fn default() -> Self { + Self::new() + } +} + +impl SerializableToSink for SumAccumulator { + fn serialize_to_json(&self) -> Value { + serde_json::json!({ + "sum": self.sum, + "observation_count": self.observation_count + }) + } + + fn serialize_to_bytes(&self) -> Vec { + match self.observation_count { + Some(count) => { + let mut bytes = Vec::with_capacity(16); + bytes.extend_from_slice(&self.sum.to_le_bytes()); + bytes.extend_from_slice(&count.to_le_bytes()); + bytes + } + None => (self.sum as f32).to_le_bytes().to_vec(), + } + } +} + +impl AggregateCore for SumAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn type_name(&self) -> &'static str { + "SumAccumulator" + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + // Check if other is also a SumAccumulator + if other.get_accumulator_type() != self.get_accumulator_type() { + return Err(format!( + "Cannot merge SumAccumulator with {}", + other.get_accumulator_type() + ) + .into()); + } + + // Downcast to SumAccumulator + let other_sum = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to SumAccumulator")?; + + // Use the existing merge_accumulators method + let merged = Self::merge_accumulators(vec![self.clone(), other_sum.clone()])?; + + Ok(Box::new(merged)) + } + + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::Sum + } + + fn approx_memory_bytes(&self) -> usize { + // Single f64 + struct overhead. + std::mem::size_of::() + } + + fn aux_stats(&self) -> AuxStats { + AuxStats { + sum: Some(self.sum), + count: self.observation_count, + ..AuxStats::empty() + } + } + + fn get_keys(&self) -> Option> { + None + } + + fn query_statistic( + &self, + statistic: crate::Statistic, + _key: &Option, + _query_kwargs: &std::collections::HashMap, + ) -> Result> { + use crate::SingleSubpopulationAggregate; + self.query(statistic, None) + } +} + +impl SingleSubpopulationAggregate for SumAccumulator { + fn query( + &self, + statistic: Statistic, + query_kwargs: Option<&HashMap>, + ) -> Result> { + // SumAccumulator doesn't use query_kwargs, assert it's None + if query_kwargs.is_some() { + return Err("SumAccumulator does not support query parameters".into()); + } + + match statistic { + Statistic::Sum => Ok(self.sum), + Statistic::Count => self + .observation_count + .map(|count| count as f64) + .ok_or_else(|| "sample count is unavailable for this Sum payload".into()), + _ => Err(format!("Unsupported statistic in SumAccumulator: {statistic:?}").into()), + } + } + + fn clone_boxed(&self) -> Box { + Box::new(self.clone()) + } +} + +impl MergeableAccumulator for SumAccumulator { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> { + let total_sum = accumulators.iter().map(|acc| acc.sum).sum(); + let observation_count = accumulators + .iter() + .try_fold(0u64, |total, acc| total.checked_add(acc.observation_count?)); + Ok(SumAccumulator { + sum: total_sum, + observation_count, + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Sample counts must survive updates and merges independently of the sum. + #[test] + fn observation_count_survives_merge() { + let mut first = SumAccumulator::new(); + first.update(10.0); + first.update(20.0); + let mut second = SumAccumulator::new(); + second.update(100.0); + let merged = SumAccumulator::merge_accumulators(vec![first, second]).unwrap(); + assert_eq!(merged.sum, 130.0); + assert_eq!(merged.aux_stats().count, Some(3)); + } + + // A legacy scalar sum has no evidence of how many observations produced it. + #[test] + fn legacy_sum_does_not_invent_observation_count() { + let mut raw = SumAccumulator::new(); + raw.update(10.0); + let merged = + SumAccumulator::merge_accumulators(vec![raw, SumAccumulator::with_sum(20.0)]).unwrap(); + assert_eq!(merged.aux_stats().count, None); + } + + // Persistence retains known counts, including zero and the full u64 range. + #[test] + fn counted_sum_binary_round_trip() { + for count in [0, 3, u64::MAX] { + let acc = SumAccumulator { + sum: 1.0000000000001, + observation_count: Some(count), + }; + let bytes = acc.serialize_to_bytes(); + assert_eq!(bytes.len(), 16); + let restored = SumAccumulator::deserialize_from_bytes(&bytes).unwrap(); + assert_eq!(restored.sum, acc.sum); + assert_eq!(restored.observation_count, Some(count)); + } + } + + // Existing scalar-only files remain readable without inventing counts. + #[test] + fn legacy_binary_sum_has_unknown_count() { + let bytes = 42.5f32.to_le_bytes(); + let restored = SumAccumulator::deserialize_from_bytes(&bytes).unwrap(); + assert_eq!(restored.sum, 42.5); + assert_eq!(restored.observation_count, None); + assert_eq!(restored.serialize_to_bytes(), bytes); + } + + // Truncated counted payloads must not silently decode as scalar sums. + #[test] + fn persisted_sum_rejects_invalid_lengths() { + for len in [0, 3, 5, 8, 15, 17] { + assert!(SumAccumulator::deserialize_from_bytes(&vec![0; len]).is_err()); + } + } + + #[test] + fn test_sum_accumulator_creation() { + let acc = SumAccumulator::new(); + assert_eq!(acc.sum, 0.0); + + let acc2 = SumAccumulator::with_sum(42.5); + assert_eq!(acc2.sum, 42.5); + } + + #[test] + fn test_sum_accumulator_update() { + let mut acc = SumAccumulator::new(); + acc.update(10.0); + acc.update(20.0); + assert_eq!(acc.sum, 30.0); + } + + #[test] + fn test_sum_accumulator_query() { + let acc = SumAccumulator::with_sum(42.0); + + assert_eq!( + crate::SingleSubpopulationAggregate::query(&acc, Statistic::Sum, None).unwrap(), + 42.0 + ); + assert!(crate::SingleSubpopulationAggregate::query(&acc, Statistic::Count, None).is_err()); + + assert!(crate::SingleSubpopulationAggregate::query(&acc, Statistic::Min, None).is_err()); + // SumAccumulator is a single subpopulation accumulator, doesn't need key-based queries + assert_eq!( + crate::SingleSubpopulationAggregate::query(&acc, Statistic::Sum, None).unwrap(), + 42.0 + ); + } + + #[test] + fn count_readout_uses_observation_count_not_sum() { + let mut acc = SumAccumulator::new(); + acc.update(10.0); + acc.update(20.0); + assert_eq!( + crate::SingleSubpopulationAggregate::query(&acc, Statistic::Count, None).unwrap(), + 2.0 + ); + } + + #[test] + fn test_sum_accumulator_merge() { + let acc1 = SumAccumulator::with_sum(10.0); + let acc2 = SumAccumulator::with_sum(20.0); + let acc3 = SumAccumulator::with_sum(30.0); + + let merged = + >::merge_accumulators(vec![ + acc1, acc2, acc3, + ]) + .unwrap(); + assert_eq!(merged.sum, 60.0); + } + + #[test] + fn test_sum_accumulator_serialization() { + let acc = SumAccumulator::with_sum(42.5); + + // Test JSON serialization + let json = acc.serialize_to_json(); + let deserialized = SumAccumulator::deserialize_from_json(&json).unwrap(); + assert_eq!(acc.sum, deserialized.sum); + + // Test byte serialization + let bytes = acc.serialize_to_bytes(); + let deserialized_bytes = SumAccumulator::deserialize_from_bytes(&bytes).unwrap(); + assert_eq!(acc.sum, deserialized_bytes.sum); + } + + #[test] + fn test_trait_object() { + let acc: Box = Box::new(SumAccumulator::with_sum(42.0)); + + assert_eq!(acc.type_name(), "SumAccumulator"); + } + + #[test] + fn from_sum_bytes_decodes_go_sum_payload() { + // GOLDEN: the 16-byte payload asap-precompute-go's + // SumWrapper{10,20,30,40}.Snapshot() emits — float64 sum (LE) followed + // by uint64 count (LE), sum=100, count=4. Proves the Rust backend + // decodes the first-class Sum payload the Go agent produces + // (cross-language wire parity, no sketchlib proto dependency). + let go_bytes: &[u8] = &[ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x59, 0x40, // 100.0 f64 LE + 0x04, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, // 4 u64 LE + ]; + let acc = SumAccumulator::from_sum_bytes(go_bytes).expect("decode Go Sum payload"); + assert_eq!(acc.sum, 100.0, "decoded Go SumWrapper payload sum"); + } + + #[test] + fn from_sum_bytes_rejects_short_payload() { + // A short buffer is rejected (the ingest path then skips the point). + assert!(SumAccumulator::from_sum_bytes(&[]).is_err()); + assert!(SumAccumulator::from_sum_bytes(&[0u8; 8]).is_err()); + } + + #[test] + fn aux_stats_exposes_sum_only() { + let acc = SumAccumulator::with_sum(123.5); + let aux = acc.aux_stats(); + assert_eq!(aux.sum, Some(123.5)); + assert_eq!(aux.count, None); + assert_eq!(aux.min, None); + assert_eq!(aux.max, None); + } + + #[test] + fn aux_stats_try_answer_on_sum_statistic() { + use crate::Statistic; + let acc = SumAccumulator::with_sum(42.0); + // Sum statistic is covered by aux without deserialising. + assert_eq!(acc.aux_stats().try_answer(Statistic::Sum), Some(42.0)); + // Count is not tracked by SumAccumulator. + assert_eq!(acc.aux_stats().try_answer(Statistic::Count), None); + } +} diff --git a/crates/asap-physical-operators/src/accumulators/univmon_accumulator.rs b/crates/asap-physical-operators/src/accumulators/univmon_accumulator.rs new file mode 100644 index 00000000..4a52bf1d --- /dev/null +++ b/crates/asap-physical-operators/src/accumulators/univmon_accumulator.rs @@ -0,0 +1,234 @@ +//! One frequency state shared by count, distinct, L2 and entropy readouts. + +use crate::{AggregateCore, AuxStats, KeyByLabelValues, SerializableToSink}; +use crate::{AggregationType, Statistic}; +use asap_sketchlib::{DataInput, UnivMon}; +use serde_json::Value; +use std::collections::HashMap; + +type Error = Box; + +#[derive(Debug, Clone)] +pub struct UnivMonAccumulator { + inner: UnivMon, +} + +impl UnivMonAccumulator { + pub fn new(heap_size: usize, rows: usize, cols: usize, layers: usize) -> Result { + if heap_size == 0 || cols == 0 || !(1..=20).contains(&rows) || !(1..=64).contains(&layers) { + return Err("invalid UnivMon dimensions".into()); + } + rows.checked_mul(cols) + .and_then(|n| n.checked_mul(layers)) + .ok_or("UnivMon dimensions overflow")?; + Ok(Self { + inner: UnivMon::init_univmon(heap_size, rows, cols, layers), + }) + } + + /// Each non-NaN sample is one occurrence. Signed zero has one identity. + pub fn insert_sample(&mut self, value: f64) -> Result<(), Error> { + if value.is_nan() { + return Ok(()); + } + self.inner + .bucket_size + .checked_add(1) + .ok_or("UnivMon count overflow")?; + let bits = if value == 0.0 { 0 } else { value.to_bits() }; + self.inner.insert(&DataInput::U64(bits), 1); + Ok(()) + } + + pub fn from_bytes(bytes: &[u8]) -> Result { + let inner = UnivMon::deserialize_from_bytes(bytes) + .map_err(|e| format!("invalid UnivMon state: {e}"))?; + if !inner.accepts_standard_updates() { + return Err( + "terminal-mode UnivMon state cannot enter the standard-update accumulator".into(), + ); + } + Ok(Self { inner }) + } + + fn compatible(&self, other: &Self) -> bool { + ( + self.inner.heap_size, + self.inner.sketch_row, + self.inner.sketch_col, + self.inner.layer_size, + ) == ( + other.inner.heap_size, + other.inner.sketch_row, + other.inner.sketch_col, + other.inner.layer_size, + ) + } + + pub fn dimensions(&self) -> (usize, usize, usize, usize) { + ( + self.inner.heap_size, + self.inner.sketch_row, + self.inner.sketch_col, + self.inner.layer_size, + ) + } + + pub fn merge_in_place(&mut self, other: &Self) -> Result<(), Error> { + if !self.compatible(other) { + return Err("incompatible UnivMon dimensions".into()); + } + self.inner + .bucket_size + .checked_add(other.inner.bucket_size) + .ok_or("UnivMon count overflow")?; + self.inner.merge(&other.inner); + Ok(()) + } +} + +impl SerializableToSink for UnivMonAccumulator { + fn serialize_to_json(&self) -> Value { + serde_json::json!({"count": self.inner.bucket_size}) + } + + fn serialize_to_bytes(&self) -> Vec { + self.inner + .serialize_to_bytes() + .expect("validated unit-frequency UnivMon state") + } +} + +impl AggregateCore for UnivMonAccumulator { + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::().saturating_add( + self.inner.layer_size.saturating_mul( + self.inner + .sketch_row + .saturating_mul(self.inner.sketch_col) + .saturating_mul(16) + .saturating_add(self.inner.heap_size.saturating_mul(256)), + ), + ) + } + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + fn type_name(&self) -> &'static str { + "UnivMonAccumulator" + } + fn as_any(&self) -> &dyn std::any::Any { + self + } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::UnivMon + } + fn get_keys(&self) -> Option> { + None + } + fn reset_to_empty(&mut self) { + self.inner.free(); + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, Error> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("expected UnivMon state")?; + let mut merged = self.clone(); + merged.merge_in_place(other)?; + Ok(Box::new(merged)) + } + + fn query_statistic( + &self, + statistic: Statistic, + key: &Option, + _: &HashMap, + ) -> Result { + if key.is_some() { + return Err("UnivMon population is selected by the catalog binding".into()); + } + match statistic { + Statistic::Count => Ok(self.inner.calc_l1()), + Statistic::Cardinality => Ok(self.inner.calc_card()), + Statistic::FrequencyL2 => Ok(self.inner.calc_l2()), + Statistic::FrequencyEntropy => Ok(self.inner.calc_entropy()), + _ => Err("unsupported UnivMon readout".into()), + } + } + + fn aux_stats(&self) -> AuxStats { + AuxStats { + count: Some(self.inner.bucket_size as u64), + ..AuxStats::empty() + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn read(state: &dyn AggregateCore, stat: Statistic) -> f64 { + state.query_statistic(stat, &None, &HashMap::new()).unwrap() + } + + /// Duplicate samples affect frequency but not cardinality, including signed zero. + #[test] + fn shared_readouts_survive_serialization() { + let mut state = UnivMonAccumulator::new(32, 5, 1024, 4).unwrap(); + for value in [0.0, -0.0, 2.0, 2.0, f64::NAN] { + state.insert_sample(value).unwrap(); + } + let restored = UnivMonAccumulator::from_bytes(&state.serialize_to_bytes()).unwrap(); + for stat in [ + Statistic::Count, + Statistic::Cardinality, + Statistic::FrequencyL2, + Statistic::FrequencyEntropy, + ] { + assert_eq!(read(&state, stat), read(&restored, stat)); + } + assert_eq!(read(&restored, Statistic::Count), 4.0); + assert!((read(&restored, Statistic::Cardinality) - 2.0).abs() < 0.01); + assert!((read(&restored, Statistic::FrequencyL2) - 8.0f64.sqrt()).abs() < 0.01); + assert!((read(&restored, Statistic::FrequencyEntropy) - 1.0).abs() < 0.01); + } + + /// Terminal-mode serialization is valid sketchlib state but not this accumulator's update domain. + #[test] + fn terminal_state_is_rejected_before_ingestion_or_merge() { + let mut state = UnivMon::init_univmon(4, 3, 16, 2); + state.fast_insert(&DataInput::U64(1), 1); + let bytes = state.serialize_to_bytes().unwrap(); + assert!(UnivMonAccumulator::from_bytes(&bytes).is_err()); + state.free(); + assert!(UnivMonAccumulator::from_bytes(&state.serialize_to_bytes().unwrap()).is_ok()); + } + + /// Pane merge preserves overlapping keys and reset removes the previous window. + #[test] + fn merge_and_reset_preserve_frequency_semantics() { + let mut left = UnivMonAccumulator::new(32, 5, 1024, 4).unwrap(); + let mut right = left.clone(); + for value in [1.0, 2.0] { + left.insert_sample(value).unwrap(); + } + for value in [2.0, 3.0] { + right.insert_sample(value).unwrap(); + } + let merged = left.merge_with(&right).unwrap(); + assert_eq!(read(merged.as_ref(), Statistic::Count), 4.0); + assert!((read(merged.as_ref(), Statistic::Cardinality) - 3.0).abs() < 0.01); + left.reset_to_empty(); + assert_eq!(read(&left, Statistic::Count), 0.0); + assert_eq!(read(&left, Statistic::FrequencyEntropy), 0.0); + assert!(left + .merge_with(&UnivMonAccumulator::new(16, 5, 1024, 4).unwrap()) + .is_err()); + } +} diff --git a/crates/asap-physical-operators/src/aggregation_type.rs b/crates/asap-physical-operators/src/aggregation_type.rs new file mode 100644 index 00000000..647f7604 --- /dev/null +++ b/crates/asap-physical-operators/src/aggregation_type.rs @@ -0,0 +1,215 @@ +//! Shared aggregation vocabulary for configuration and accumulator dispatch. +//! The wire shape combines aggregation type, subtype, and parameters. +//! `AccumulatorSpec` provides a typed representation at conversion boundaries. + +use serde::{Deserialize, Serialize}; +use std::fmt; +use std::str::FromStr; + +/// Concrete aggregation/sketch type used in precompute configs and accumulator dispatch. +/// +/// `Display` outputs the canonical PascalCase name used in YAML/JSON configs. +/// `FromStr` accepts the canonical name plus legacy aliases (e.g. "KLL" → `DatasketchesKLL`). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum AggregationType { + // ---------- single-population (non-keyed) ---------- + Sum, + Count, + Increase, + Rate, + Min, + Max, + DatasketchesKLL, + // ---------- multi-population (keyed) ---------- + HydraKLL, + CountMinSketch, + CountMinSketchWithHeap, + CountSketch, + CountSketchWithHeap, + // ---------- cardinality / set tracking ---------- + HLL, + UnivMon, + DDSketch, + // ---------- legacy config wrapper names ---------- + SingleSubpopulation, + MultipleSubpopulation, +} + +impl AggregationType { + /// Adapt a storage/processor tag to Planner's exact family. Keyed storage + /// changes the payload layout, not the semantic family. + pub fn planner_exact_family(self) -> Option { + use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; + let (kind, params) = match self { + Self::Sum => (ExactKind::Sum, ExactParams::Sum), + Self::Count => (ExactKind::Count, ExactParams::Count), + Self::Increase => (ExactKind::Increase, ExactParams::Increase), + Self::Rate => (ExactKind::Rate, ExactParams::Rate), + Self::Min => (ExactKind::Min, ExactParams::Min), + Self::Max => (ExactKind::Max, ExactParams::Max), + _ => return None, + }; + Some(SummaryFamilyType::ExactAggregate(kind, params)) + } + + pub fn as_str(self) -> &'static str { + match self { + AggregationType::Sum => "Sum", + AggregationType::Count => "Count", + AggregationType::Increase => "Increase", + AggregationType::Rate => "Rate", + AggregationType::Min => "Min", + AggregationType::Max => "Max", + AggregationType::DatasketchesKLL => "DatasketchesKLL", + AggregationType::HydraKLL => "HydraKLL", + AggregationType::CountMinSketch => "CountMinSketch", + AggregationType::CountMinSketchWithHeap => "CountMinSketchWithHeap", + AggregationType::CountSketch => "CountSketch", + AggregationType::CountSketchWithHeap => "CountSketchWithHeap", + AggregationType::HLL => "HLL", + AggregationType::UnivMon => "UnivMon", + AggregationType::DDSketch => "DDSketch", + AggregationType::SingleSubpopulation => "SingleSubpopulation", + AggregationType::MultipleSubpopulation => "MultipleSubpopulation", + } + } + + /// Returns `true` if this type produces keyed (multi-population) accumulators. + pub fn is_keyed(self) -> bool { + matches!( + self, + AggregationType::MultipleSubpopulation + | AggregationType::CountMinSketch + | AggregationType::CountMinSketchWithHeap + | AggregationType::CountSketch + | AggregationType::CountSketchWithHeap + | AggregationType::HydraKLL + ) + } +} + +impl fmt::Display for AggregationType { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(self.as_str()) + } +} + +impl FromStr for AggregationType { + type Err = String; + + fn from_str(s: &str) -> Result { + match s { + // Canonical names + "Sum" => Ok(AggregationType::Sum), + "Count" => Ok(AggregationType::Count), + "Increase" => Ok(AggregationType::Increase), + "Rate" => Ok(AggregationType::Rate), + "Min" => Ok(AggregationType::Min), + "Max" => Ok(AggregationType::Max), + "DatasketchesKLL" => Ok(AggregationType::DatasketchesKLL), + "HydraKLL" => Ok(AggregationType::HydraKLL), + "CountMinSketch" => Ok(AggregationType::CountMinSketch), + "CountMinSketchWithHeap" => Ok(AggregationType::CountMinSketchWithHeap), + "CountSketch" => Ok(AggregationType::CountSketch), + "CountSketchWithHeap" => Ok(AggregationType::CountSketchWithHeap), + "HLL" | "HyperLogLog" => Ok(AggregationType::HLL), + "UnivMon" => Ok(AggregationType::UnivMon), + "DDSketch" | "DdSketch" => Ok(AggregationType::DDSketch), + "SingleSubpopulation" => Ok(AggregationType::SingleSubpopulation), + "MultipleSubpopulation" => Ok(AggregationType::MultipleSubpopulation), + // Legacy accumulator-suffixed aliases + "SumAccumulator" | "SumAggregator" | "sum" => Ok(AggregationType::Sum), + "IncreaseAccumulator" | "IncreaseAggregator" | "increase" => { + Ok(AggregationType::Increase) + } + "MinAccumulator" | "MinAggregator" | "min" => Ok(AggregationType::Min), + "MaxAccumulator" | "MaxAggregator" | "max" => Ok(AggregationType::Max), + "DatasketchesKLLAccumulator" | "KLL" | "kll" | "datasketches_kll" => { + Ok(AggregationType::DatasketchesKLL) + } + "HydraKllSketchAccumulator" | "hydra_kll" => Ok(AggregationType::HydraKLL), + "CountMinSketchAccumulator" | "CMS" | "cms" | "count_min_sketch" => { + Ok(AggregationType::CountMinSketch) + } + "CountMinSketchWithHeapAccumulator" => Ok(AggregationType::CountMinSketchWithHeap), + "CountSketchAccumulator" | "CS" | "cs" | "count_sketch" => { + Ok(AggregationType::CountSketch) + } + "CountSketchWithHeapAccumulator" => Ok(AggregationType::CountSketchWithHeap), + // Retired names. `MinMax` used to be one accumulator whose + // direction rode alongside in `aggregationSubType`; the two + // directions are separate types now, so there is no safe + // direction to guess here -- resolving a min workload as a + // max one is silently wrong, not merely imprecise. + "MinMax" + | "MinMaxAccumulator" + | "MinMaxAggregator" + | "min_max" + | "MultipleMinMax" + | "MultipleMinMaxAccumulator" + | "multiple_min_max" => Err(format!( + "Retired aggregation type: '{s}' -- min and max are separate types now, \ + use 'Min'/'Max'" + )), + _ => Err(format!("Unknown aggregation type: '{s}'")), + } + } +} + +impl Serialize for AggregationType { + fn serialize(&self, serializer: S) -> Result { + serializer.serialize_str(self.as_str()) + } +} + +impl<'de> Deserialize<'de> for AggregationType { + fn deserialize>(deserializer: D) -> Result { + let s = String::deserialize(deserializer)?; + s.parse().map_err(serde::de::Error::custom) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; + + /// Removed layout tags cannot be installed as semantic families. + #[test] + fn rejects_keyed_family_aliases() { + for name in [ + "MultipleSum", + "MultipleIncrease", + "MultipleMin", + "MultipleMax", + ] { + assert!(name.parse::().is_err(), "{name}"); + } + } + + #[test] + fn storage_layout_tags_do_not_create_planner_families() { + for (storage, expected) in [ + (AggregationType::Sum, ExactKind::Sum), + (AggregationType::Count, ExactKind::Count), + (AggregationType::Increase, ExactKind::Increase), + (AggregationType::Rate, ExactKind::Rate), + ] { + let family = storage.planner_exact_family().unwrap(); + assert!( + matches!(family, SummaryFamilyType::ExactAggregate(kind, _) if kind == expected) + ); + } + assert_eq!( + AggregationType::Rate.planner_exact_family(), + Some(SummaryFamilyType::ExactAggregate( + ExactKind::Rate, + ExactParams::Rate + )) + ); + assert_ne!( + AggregationType::Rate.planner_exact_family(), + AggregationType::Increase.planner_exact_family() + ); + } +} diff --git a/crates/asap-physical-operators/src/arithmetic.rs b/crates/asap-physical-operators/src/arithmetic.rs new file mode 100644 index 00000000..bfc50694 --- /dev/null +++ b/crates/asap-physical-operators/src/arithmetic.rs @@ -0,0 +1,19 @@ +//! Float64 arithmetic shared by ASAP execution engines. +//! Preserve IEEE non-finite results; callers own their output policies. + +pub fn evaluate_float64_arithmetic( + operator: &planner_types::pre_asap::ArithmeticOpKind, + left: f64, + right: f64, +) -> f64 { + use planner_types::pre_asap::ArithmeticOpKind::*; + match operator { + Add => left + right, + Sub => left - right, + Mul => left * right, + Div => left / right, + Mod => left % right, + Pow => left.powf(right), + Atan2 => left.atan2(right), + } +} diff --git a/crates/asap-physical-operators/src/capability.rs b/crates/asap-physical-operators/src/capability.rs new file mode 100644 index 00000000..2d047229 --- /dev/null +++ b/crates/asap-physical-operators/src/capability.rs @@ -0,0 +1,129 @@ +//! Allocation-free checks for the concrete summary kernels in this crate. +use planner_types::post_asap::{ + ExactKind, ExactParams, GroupingStrategy, SketchAlgorithm, SketchParams, SummaryFamilyType, + SummaryUpdate, +}; + +/// Check the same contract used by `create_planner_accumulator` before a plan +/// is accepted. Execution timing is deliberately not a kernel property. +pub fn validate_summary_kernel( + family: &SummaryFamilyType, + input: &SummaryUpdate, + grouping: &GroupingStrategy, +) -> Result<(), String> { + if grouping != &GroupingStrategy::PerSubpopulationInstance { + return Err("shared summary grouping has no registered kernel".into()); + } + let keyed = match family { + SummaryFamilyType::ExactAggregate(kind, params) => { + use ExactKind as K; + use ExactParams as P; + if !matches!( + (kind, params), + (K::Sum, P::Sum) + | (K::Count, P::Count) + | (K::Min, P::Min) + | (K::Max, P::Max) + | (K::Rate, P::Rate) + | (K::Increase, P::Increase) + ) { + return Err(format!("unsupported exact kernel {family:?}")); + } + input.item.is_some() + } + SummaryFamilyType::Sketch(kind, layout) => { + if layout != grouping { + return Err("Planner family and operator grouping disagree".into()); + } + use SketchAlgorithm as A; + use SketchParams as P; + match (kind.algorithm(), kind.params()) { + (A::Kll, P::Kll { k }) if (8..=u16::MAX as u32).contains(k) => false, + (A::DDSketch, P::DDSketch { alpha }) + if alpha.is_finite() && *alpha > 0.0 && *alpha < 1.0 => + { + false + } + (A::Hll, P::Hll { precision }) if (4..=18).contains(precision) => false, + (A::Cms, P::Cms { width, depth }) + | (A::CountSketch, P::CountSketch { width, depth }) + if valid_matrix(*width, *depth) => + { + true + } + ( + A::CmsWithHeap, + P::CmsWithHeap { + width, + depth, + heap_size, + }, + ) + | ( + A::CountSketchWithHeap, + P::CountSketchWithHeap { + width, + depth, + heap_size, + }, + ) if valid_matrix(*width, *depth) && *heap_size > 0 => true, + ( + A::UnivMon, + P::UnivMon { + heap_size, + sketch_rows, + sketch_cols, + layers, + }, + ) if *heap_size > 0 + && *sketch_cols > 0 + && (1..=20).contains(sketch_rows) + && (1..=64).contains(layers) + && (*sketch_rows as usize) + .checked_mul(*sketch_cols as usize) + .and_then(|n| n.checked_mul(*layers as usize)) + .is_some() => + { + false + } + _ => { + return Err(format!( + "unsupported kernel or invalid parameters: {kind:?}" + )) + } + } + } + _ => return Err(format!("unsupported summary kernel {family:?}")), + }; + if keyed != input.item.is_some() && !is_unit_sample_frequency(input) { + return Err("Planner item expression does not match kernel layout".into()); + } + Ok(()) +} + +fn valid_matrix(width: u32, depth: u32) -> bool { + // Construction uses the kernel's native row hashing. Packed-wire decoder + // limits describe a different representation and must not reject it here. + width > 0 + && depth > 0 + && (width as usize) + .checked_mul(depth as usize) + .and_then(|n| n.checked_mul(std::mem::size_of::())) + .is_some() +} + +pub(crate) fn is_unit_sample_frequency(update: &planner_types::post_asap::SummaryUpdate) -> bool { + use planner_types::post_asap::{NonNegativeWeightProof, SummaryInputExpr, WeightDomain}; + matches!( + update.item, + Some(SummaryInputExpr::Column( + planner_types::pre_asap::ColumnRef::SampleValue + )) + ) && matches!(update.weight, SummaryInputExpr::Constant(1.0)) + && matches!( + update.weight_domain, + WeightDomain::NonNegative { + proof: NonNegativeWeightProof::UnitCount + } + ) +} diff --git a/crates/asap-physical-operators/src/dag/batch_execution.rs b/crates/asap-physical-operators/src/dag/batch_execution.rs new file mode 100644 index 00000000..a9c20abc --- /dev/null +++ b/crates/asap-physical-operators/src/dag/batch_execution.rs @@ -0,0 +1,200 @@ +//! Execute a bounded in-memory batch through native operators. This is also the +//! bridge for deployments whose boundary values are not yet streaming batches. +use super::{operators::Operator, values::Batch, Error, PhysicalDag, RunContext, SharedValue}; +use futures::{FutureExt, StreamExt}; + +/// Every input is already in memory; the chain contains native operators only. +/// This deliberately does not enter a nested executor when called from a DAG +/// adapter. I/O belongs to source operators in the surrounding execution. +pub fn evaluate_batch( + input: Batch, + operators: Vec, + context: RunContext, +) -> Result>, Error> { + let mut graph = PhysicalDag::default(); + graph.add( + 0, + vec![], + Operator::source(input.schema().clone(), vec![input])?, + )?; + let mut root = 0; + for operator in operators { + graph.add(root + 1, vec![root], operator)?; + root += 1; + } + evaluate_graph(graph, root, context) +} + +/// Bind the ordered in-memory inputs of a native multi-input operator. +pub fn evaluate_inputs( + inputs: Vec, + operator: Operator, + context: RunContext, +) -> Result>, Error> { + let mut graph = PhysicalDag::default(); + let root = inputs.len() as u64; + for (id, input) in inputs.into_iter().enumerate() { + graph.add( + id as u64, + vec![], + Operator::source(input.schema().clone(), vec![input])?, + )?; + } + graph.add(root, (0..root).collect(), operator)?; + evaluate_graph(graph, root, context) +} + +/// Evaluate a native in-memory source, including scalar sources, in the caller's scope. +pub fn evaluate_source( + source: Operator, + context: RunContext, +) -> Result>, Error> { + let mut graph = PhysicalDag::default(); + graph.add(0, vec![], source)?; + evaluate_graph(graph, 0, context) +} + +fn evaluate_graph( + graph: PhysicalDag<'_, Batch, super::values::Schema>, + root: super::NodeId, + context: RunContext, +) -> Result>, Error> { + let mut output = graph.execute(&[root], context)?.remove(0); + let mut batches = Vec::new(); + loop { + match output.next().now_or_never() { + Some(Some(Ok(batch))) => batches.push(batch), + Some(Some(Err(error))) => return Err(error), + Some(None) => return Ok(batches), + // Native operators have no I/O sources here. Pending is the + // shared runtime's cooperative yield after a batch quantum. + None => continue, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::dag::{operators::Expression, values::Value, Limits, Scope}; + use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, + }; + use std::sync::Arc; + + // Engine adapters can run the identical native chain from an outer executor. + #[test] + fn same_native_chain_inside_query_and_ingestion_execution() { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }); + for scope in [ + Scope::Query { + evaluation_time_ms: 20, + revision: 1, + }, + Scope::Ingestion { + window_start_ms: 10, + window_end_ms: 20, + revision: 1, + }, + ] { + let batch = Batch::try_new(schema.clone(), vec![vec![Value::Float64(7.)]]).unwrap(); + let negate = Operator::project( + schema.clone(), + vec![( + "value".into(), + Expression::Negate(Box::new(Expression::Column(0))), + )], + ) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let result = futures::executor::block_on(async { + evaluate_batch(batch, vec![negate], context.clone()) + }) + .unwrap(); + assert!(matches!(result[0].rows()[0][0], Value::Float64(-7.))); + let source = Operator::scalar(Value::Float64(9.), DataType::Float64).unwrap(); + let scalar = evaluate_source(source, context).unwrap(); + assert!(matches!(scalar[0].rows()[0][0], Value::Float64(9.))); + } + } + + // Native sources may cross the runtime's cooperative batch quantum. + #[test] + fn in_memory_source_drives_cooperative_yields() { + let schema = Arc::new(SummarySchema { + fields: vec![], + time_index: None, + }); + let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); + let source = Operator::source(schema, vec![batch; 65]).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + assert_eq!(evaluate_source(source, context).unwrap().len(), 65); + } + + // An adapter-held output must retain its parent's reservation after execution. + #[test] + fn returned_batches_keep_their_resource_reservation() { + let schema = Arc::new(SummarySchema { + fields: vec![], + time_index: None, + }); + let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); + let bytes = batch.bytes(); + let source = Operator::source(schema, vec![batch]).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_bytes: bytes, + max_buffered_batches: 1, + }, + ) + .unwrap(); + let held = evaluate_source(source.clone(), context.clone()).unwrap(); + assert_eq!(context.retained_bytes(), bytes); + assert!(evaluate_source(source.clone(), context.clone()).is_err()); + drop(held); + assert_eq!(context.retained_bytes(), 0); + assert!(evaluate_source(source, context).is_ok()); + } + + // A cancelled surrounding execution also prevents its native computation. + #[test] + fn cancellation_is_not_bypassed_by_in_memory_execution() { + let schema = Arc::new(SummarySchema { + fields: vec![], + time_index: None, + }); + let batch = Batch::try_new(schema, vec![vec![]]).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + context.cancel(); + assert!(matches!( + evaluate_batch(batch, vec![], context), + Err(Error::Cancelled) + )); + } +} diff --git a/crates/asap-physical-operators/src/dag/mod.rs b/crates/asap-physical-operators/src/dag/mod.rs new file mode 100644 index 00000000..1c9ee9e8 --- /dev/null +++ b/crates/asap-physical-operators/src/dag/mod.rs @@ -0,0 +1,517 @@ +//! Independent operator DAG execution. No backend plan or engine types are used. +//! +//! Each run creates one stream per reachable node. Consumers subscribe to that +//! stream independently; retained outputs are released after the last consumer. +use futures::{stream::LocalBoxStream, Stream}; +use std::{ + cell::{Cell, RefCell}, + collections::{BTreeMap, BTreeSet, VecDeque}, + fmt::Debug, + pin::Pin, + rc::Rc, + sync::Arc, + task::{Context, Poll, Waker}, +}; + +pub type NodeId = u64; +pub type OutputStream<'a, V> = LocalBoxStream<'a, Result>; +#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] +pub enum Error { + #[error("invalid DAG: {0}")] + Invalid(String), + #[error("operator failed: {0}")] + Operator(String), + #[error("node {node} ({operation}) failed: {source}")] + AtNode { + node: NodeId, + operation: String, + source: Box, + }, + #[error("execution memory limit exceeded")] + MemoryLimit, + #[error("execution cancelled")] + Cancelled, +} + +/// Scope is part of an execution instance, never mutable state in a reusable plan. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Scope { + Ingestion { + window_start_ms: i64, + window_end_ms: i64, + revision: u64, + }, + Query { + evaluation_time_ms: i64, + revision: u64, + }, +} +#[derive(Clone, Debug)] +pub struct Limits { + pub max_buffered_batches: usize, + pub max_bytes: usize, +} +impl Default for Limits { + fn default() -> Self { + Self { + max_buffered_batches: 8, + max_bytes: 64 * 1024 * 1024, + } + } +} +struct Control { + cancelled: Cell, + bytes: Cell, + peak: Cell, + limits: Limits, + waiters: RefCell>, +} +#[derive(Clone)] +pub struct RunContext { + pub scope: Scope, + control: Rc, +} +impl RunContext { + pub fn new(scope: Scope, limits: Limits) -> Result { + if limits.max_buffered_batches == 0 || limits.max_bytes == 0 { + return Err(Error::Invalid("execution limits must be positive".into())); + } + if matches!(&scope, Scope::Ingestion { window_start_ms, window_end_ms, .. } if window_start_ms > window_end_ms) + { + return Err(Error::Invalid("inverted ingestion window".into())); + } + Ok(Self { + scope, + control: Rc::new(Control { + cancelled: Cell::new(false), + bytes: Cell::new(0), + peak: Cell::new(0), + limits, + waiters: RefCell::new(Vec::new()), + }), + }) + } + pub fn cancel(&self) { + self.control.cancelled.set(true); + for waiter in self.control.waiters.borrow_mut().drain(..) { + waiter.wake(); + } + } + pub fn is_cancelled(&self) -> bool { + self.control.cancelled.get() + } + pub fn retained_bytes(&self) -> usize { + self.control.bytes.get() + } + pub fn peak_bytes(&self) -> usize { + self.control.peak.get() + } + pub fn reserve(&self, bytes: usize) -> Result { + let total = self + .control + .bytes + .get() + .checked_add(bytes) + .ok_or(Error::MemoryLimit)?; + if total > self.control.limits.max_bytes { + return Err(Error::MemoryLimit); + } + self.control.bytes.set(total); + self.control.peak.set(self.control.peak.get().max(total)); + Ok(Reservation { + bytes, + control: Rc::clone(&self.control), + }) + } + fn register(&self, waker: &Waker) { + let mut waiters = self.control.waiters.borrow_mut(); + if !waiters.iter().any(|old| old.will_wake(waker)) { + waiters.push(waker.clone()); + } + } +} +pub struct Reservation { + bytes: usize, + control: Rc, +} +impl Reservation { + /// Adjust an operator-owned allocation without accumulating bookkeeping entries. + pub fn resize(&mut self, bytes: usize) -> Result<(), Error> { + let total = self + .control + .bytes + .get() + .checked_sub(self.bytes) + .and_then(|total| total.checked_add(bytes)) + .ok_or(Error::MemoryLimit)?; + if total > self.control.limits.max_bytes { + return Err(Error::MemoryLimit); + } + self.control.bytes.set(total); + self.control.peak.set(self.control.peak.get().max(total)); + self.bytes = bytes; + Ok(()) + } +} +impl Drop for Reservation { + fn drop(&mut self) { + self.control + .bytes + .set(self.control.bytes.get().saturating_sub(self.bytes)); + } +} + +/// An output owns its memory reservation even after it leaves the DAG's queue. +pub struct SharedValue { + value: Arc, + _reservation: Rc, +} +impl Clone for SharedValue { + fn clone(&self) -> Self { + Self { + value: Arc::clone(&self.value), + _reservation: Rc::clone(&self._reservation), + } + } +} +impl std::ops::Deref for SharedValue { + type Target = V; + fn deref(&self) -> &V { + &self.value + } +} +impl SharedValue { + pub fn value(&self) -> &V { + &self.value + } +} + +/// Operators own computation. The runtime provides already-connected inputs; +/// an operator must not recursively execute another plan node itself. +pub trait PhysicalOperator { + fn name(&self) -> &str; + fn input_schemas(&self) -> Vec; + fn output_schema(&self) -> S; + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error>; + fn output_bytes(&self, value: &V) -> usize; +} +struct Node<'a, V, S> { + inputs: Vec, + operator: Box + 'a>, +} +pub struct PhysicalDag<'a, V, S> { + nodes: BTreeMap>, +} +impl Default for PhysicalDag<'_, V, S> { + fn default() -> Self { + Self { + nodes: BTreeMap::new(), + } + } +} +impl<'a, V: 'a, S: Clone + PartialEq + Debug + 'a> PhysicalDag<'a, V, S> { + pub fn add( + &mut self, + id: NodeId, + inputs: Vec, + operator: impl PhysicalOperator + 'a, + ) -> Result<(), Error> { + self.add_boxed(id, inputs, Box::new(operator)) + } + pub fn add_boxed( + &mut self, + id: NodeId, + inputs: Vec, + operator: Box + 'a>, + ) -> Result<(), Error> { + if self.nodes.contains_key(&id) { + return Err(Error::Invalid(format!("duplicate node {id}"))); + } + self.nodes.insert(id, Node { inputs, operator }); + Ok(()) + } + pub fn validate(&self, roots: &[NodeId]) -> Result<(), Error> { + fn visit( + dag: &PhysicalDag<'_, V, S>, + id: NodeId, + active: &mut BTreeSet, + done: &mut BTreeMap, + ) -> Result { + if let Some(depth) = done.get(&id) { + return Ok(*depth); + } + if active.len() >= 128 { + return Err(Error::Invalid( + "DAG exceeds the supported execution depth of 128".into(), + )); + } + if !active.insert(id) { + return Err(Error::Invalid(format!("cycle at node {id}"))); + } + let node = dag + .nodes + .get(&id) + .ok_or_else(|| Error::Invalid(format!("missing node {id}")))?; + let expected = node.operator.input_schemas(); + if expected.len() != node.inputs.len() { + return Err(Error::Invalid(format!("node {id} input arity mismatch"))); + } + let mut depth = 1; + for (input, schema) in node.inputs.iter().zip(expected) { + depth = depth.max(1 + visit(dag, *input, active, done)?); + let actual = dag.nodes[input].operator.output_schema(); + if actual != schema { + return Err(Error::Invalid(format!( + "node {id} input {input} schema mismatch: {actual:?} vs {schema:?}" + ))); + } + } + if depth > 128 { + return Err(Error::Invalid( + "DAG exceeds the supported execution depth of 128".into(), + )); + } + active.remove(&id); + done.insert(id, depth); + Ok(depth) + } + if roots.is_empty() { + return Err(Error::Invalid("execution needs a root".into())); + } + let mut done = BTreeMap::new(); + for &root in roots { + visit(self, root, &mut BTreeSet::new(), &mut done)?; + } + Ok(()) + } + pub fn execute<'r>( + &'r self, + roots: &[NodeId], + context: RunContext, + ) -> Result>, Error> + where + 'a: 'r, + { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + self.validate(roots)?; + fn build<'r, V: 'r, S: 'r>( + dag: &'r PhysicalDag<'_, V, S>, + id: NodeId, + context: &RunContext, + states: &mut BTreeMap>>>, + ) -> Result>>, Error> { + if let Some(state) = states.get(&id) { + return Ok(Rc::clone(state)); + } + let node = &dag.nodes[&id]; + let mut inputs = Vec::new(); + for &child in &node.inputs { + inputs.push(Input::subscribe(build(dag, child, context, states)?)); + } + let stream = node + .operator + .start(inputs, context.clone()) + .map_err(|source| Error::AtNode { + node: id, + operation: node.operator.name().into(), + source: Box::new(source), + })?; + let op = node.operator.as_ref(); + let state = Rc::new(RefCell::new(Producer { + stream: Some(stream), + node: id, + operation: node.operator.name().into(), + size: Box::new(move |value| op.output_bytes(value)), + context: context.clone(), + queue: VecDeque::new(), + base: 0, + next_reader: 0, + batches_polled: 0, + readers: BTreeMap::new(), + waiters: BTreeMap::new(), + finished: false, + failure: None, + })); + states.insert(id, Rc::clone(&state)); + Ok(state) + } + let mut states = BTreeMap::new(); + roots + .iter() + .map(|&id| build(self, id, &context, &mut states).map(Input::subscribe)) + .collect() + } +} +struct Producer<'a, V> { + node: NodeId, + operation: String, + stream: Option>, + size: Box usize + 'a>, + context: RunContext, + queue: VecDeque>, + base: u64, + next_reader: u64, + batches_polled: usize, + readers: BTreeMap, + waiters: BTreeMap, + finished: bool, + failure: Option, +} +impl Producer<'_, V> { + fn trim(&mut self) { + let minimum = self + .readers + .values() + .copied() + .min() + .unwrap_or(self.base + self.queue.len() as u64); + while self.base < minimum { + self.queue.pop_front(); + self.base += 1; + } + for (_, waker) in std::mem::take(&mut self.waiters) { + waker.wake(); + } + if self.readers.is_empty() { + self.stream = None; + self.queue.clear(); + } + } +} +pub struct Input<'a, V> { + producer: Rc>>, + reader: u64, + done: bool, +} +impl<'a, V> Input<'a, V> { + fn subscribe(producer: Rc>>) -> Self { + let reader = { + let mut state = producer.borrow_mut(); + let id = state.next_reader; + state.next_reader += 1; + let base = state.base; + state.readers.insert(id, base); + id + }; + Self { + producer, + reader, + done: false, + } + } +} +impl Drop for Input<'_, V> { + fn drop(&mut self) { + let mut state = self.producer.borrow_mut(); + state.readers.remove(&self.reader); + state.waiters.remove(&self.reader); + state.trim(); + } +} +impl Stream for Input<'_, V> { + type Item = Result, Error>; + fn poll_next(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + let this = self.get_mut(); + if this.done { + return Poll::Ready(None); + } + let mut state = this.producer.borrow_mut(); + state.context.register(cx.waker()); + if state.context.is_cancelled() { + state.failure = Some(Error::Cancelled); + state.finished = true; + state.stream = None; + state.queue.clear(); + } + let position = state.readers[&this.reader]; + let index = (position - state.base) as usize; + if let Some(value) = state.queue.get(index).cloned() { + state.readers.insert(this.reader, position + 1); + state.trim(); + return Poll::Ready(Some(Ok(value))); + } + if state.finished { + this.done = true; + state.readers.remove(&this.reader); + let failure = state.failure.clone(); + state.trim(); + return Poll::Ready(failure.map(Err)); + } + state.waiters.insert(this.reader, cx.waker().clone()); + if state.queue.len() >= state.context.control.limits.max_buffered_batches { + return Poll::Pending; + } + // Always-ready sources must still give cancellation and other roots a turn. + if state.batches_polled >= 32 { + state.batches_polled = 0; + cx.waker().wake_by_ref(); + return Poll::Pending; + } + let polled = state + .stream + .as_mut() + .expect("unfinished producer") + .as_mut() + .poll_next(cx); + if matches!(&polled, Poll::Ready(Some(Ok(_)))) { + state.batches_polled += 1; + } + match polled { + Poll::Pending => Poll::Pending, + Poll::Ready(Some(Ok(value))) => match state.context.reserve((state.size)(&value)) { + Ok(reservation) => { + let value = SharedValue { + value: Arc::new(value), + _reservation: Rc::new(reservation), + }; + state.queue.push_back(value.clone()); + state.readers.insert(this.reader, position + 1); + state.trim(); + Poll::Ready(Some(Ok(value))) + } + Err(error) => { + state.failure = Some(error.clone()); + state.finished = true; + state.stream = None; + this.done = true; + state.readers.remove(&this.reader); + state.trim(); + Poll::Ready(Some(Err(error))) + } + }, + Poll::Ready(result) => { + let error = result.and_then(Result::err).map(|source| match source { + Error::AtNode { .. } | Error::Cancelled | Error::MemoryLimit => source, + source => Error::AtNode { + node: state.node, + operation: state.operation.clone(), + source: Box::new(source), + }, + }); + state.failure = error.clone(); + state.finished = true; + state.stream = None; + this.done = true; + state.readers.remove(&this.reader); + state.trim(); + Poll::Ready(error.map(Err)) + } + } + } +} + +pub mod operators; +pub mod values; + +#[cfg(test)] +mod tests; + +pub mod planner; + +pub mod batch_execution; diff --git a/crates/asap-physical-operators/src/dag/operators.rs b/crates/asap-physical-operators/src/dag/operators.rs new file mode 100644 index 00000000..eeb1a922 --- /dev/null +++ b/crates/asap-physical-operators/src/dag/operators.rs @@ -0,0 +1,1170 @@ +//! Native DAG operators. Engines bind sources; computation lives here. +use super::{ + values::{group_key, Batch, Schema, Value}, + Error, Input, OutputStream, PhysicalOperator, Reservation, RunContext, +}; +use futures::StreamExt; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema, SummaryUpdate}, + pre_asap::{ArithmeticOpKind, ColumnRef, DataType}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +fn invalid(message: &str) -> Error { + Error::Invalid(message.into()) +} +fn field(schema: &Schema, column: usize) -> Result<&SummaryField, Error> { + schema + .fields + .get(column) + .ok_or_else(|| invalid("column out of range")) +} +fn plain(schema: &Schema, column: usize) -> Result<(&DataType, bool), Error> { + let f = field(schema, column)?; + let SummaryFamilyType::Plain(dtype) = &f.dtype else { + return Err(invalid("plain value required")); + }; + Ok((dtype, f.nullable)) +} +fn schema(fields: Vec) -> Schema { + Arc::new(SummarySchema { + fields, + time_index: None, + }) +} +fn result_field(name: &str, dtype: DataType, nullable: bool) -> SummaryField { + SummaryField { + name: name.into(), + dtype: SummaryFamilyType::Plain(dtype), + nullable, + } +} + +#[derive(Clone, Debug)] +pub enum Expression { + Column(usize), + Literal { + value: Value, + dtype: DataType, + }, + Negate(Box), + Arithmetic { + op: ArithmeticOpKind, + left: Box, + right: Box, + }, + Equal(Box, Box), + Less(Box, Box), + And(Box, Box), + Or(Box, Box), + Not(Box), + IsNull(Box), +} +impl Expression { + fn dtype(&self, input: &Schema) -> Result<(DataType, bool), Error> { + use Expression::*; + match self { + Column(i) => { + let (t, n) = plain(input, *i)?; + Ok((t.clone(), n)) + } + Literal { value, dtype } => { + if value.matches(dtype, true) { + Ok((dtype.clone(), matches!(value, Value::Null))) + } else { + Err(invalid("literal type mismatch")) + } + } + Negate(v) => { + let (t, n) = v.dtype(input)?; + if matches!(t, DataType::Int64 | DataType::Float64) { + Ok((t, n)) + } else { + Err(invalid("numeric negation required")) + } + } + Arithmetic { op, left, right } => { + let (a, n) = left.dtype(input)?; + let (b, m) = right.dtype(input)?; + if a == b + && matches!(a, DataType::Int64 | DataType::Float64) + && !(a == DataType::Int64 && *op == ArithmeticOpKind::Atan2) + { + Ok((a, n || m)) + } else { + Err(invalid("arithmetic requires matching numeric types")) + } + } + Equal(a, b) | Less(a, b) => { + let (a, n) = a.dtype(input)?; + let (b, m) = b.dtype(input)?; + if a == b && ordered(&a) { + Ok((DataType::Bool, n || m)) + } else { + Err(invalid("comparison requires matching ordered types")) + } + } + And(a, b) | Or(a, b) => { + let (a, n) = a.dtype(input)?; + let (b, m) = b.dtype(input)?; + if a == DataType::Bool && b == DataType::Bool { + Ok((DataType::Bool, n || m)) + } else { + Err(invalid("boolean operands required")) + } + } + Not(v) => { + let (t, n) = v.dtype(input)?; + if t == DataType::Bool { + Ok((t, n)) + } else { + Err(invalid("boolean operand required")) + } + } + IsNull(v) => { + v.dtype(input)?; + Ok((DataType::Bool, false)) + } + } + } + fn evaluate(&self, row: &[Value]) -> Result { + use Expression::*; + Ok(match self { + Column(i) => row[*i].clone(), + Literal { value, .. } => value.clone(), + Negate(v) => match v.evaluate(row)? { + Value::Int64(v) => Value::Int64( + v.checked_neg() + .ok_or_else(|| invalid("integer negation overflow"))?, + ), + Value::Float64(v) => Value::Float64(-v), + Value::Null => Value::Null, + _ => return Err(invalid("numeric negation required")), + }, + Arithmetic { op, left, right } => { + numeric(op, left.evaluate(row)?, right.evaluate(row)?)? + } + Equal(a, b) | Less(a, b) => { + let (a, b) = (a.evaluate(row)?, b.evaluate(row)?); + if matches!(a, Value::Null) || matches!(b, Value::Null) { + Value::Null + } else if matches!((&a,&b),(Value::Float64(a),Value::Float64(b)) if a.is_nan() || b.is_nan()) + { + Value::Bool(false) + } else { + let c = a.compare(&b)?; + Value::Bool(if matches!(self, Equal(..)) { + c.is_eq() + } else { + c.is_lt() + }) + } + } + And(a, b) | Or(a, b) => { + let (a, b) = (a.evaluate(row)?, b.evaluate(row)?); + match (a, b, matches!(self, And(..))) { + (Value::Bool(false), _, true) | (_, Value::Bool(false), true) => { + Value::Bool(false) + } + (Value::Bool(true), _, false) | (_, Value::Bool(true), false) => { + Value::Bool(true) + } + (Value::Null, _, _) | (_, Value::Null, _) => Value::Null, + (Value::Bool(a), Value::Bool(b), true) => Value::Bool(a && b), + (Value::Bool(a), Value::Bool(b), false) => Value::Bool(a || b), + _ => return Err(invalid("boolean operands required")), + } + } + Not(v) => match v.evaluate(row)? { + Value::Bool(v) => Value::Bool(!v), + Value::Null => Value::Null, + _ => return Err(invalid("boolean operand required")), + }, + IsNull(v) => Value::Bool(matches!(v.evaluate(row)?, Value::Null)), + }) + } +} +fn ordered(dtype: &DataType) -> bool { + matches!( + dtype, + DataType::Int64 + | DataType::Float64 + | DataType::Utf8 + | DataType::Bool + | DataType::Timestamp + | DataType::Date + ) +} +fn numeric(op: &ArithmeticOpKind, a: Value, b: Value) -> Result { + use ArithmeticOpKind::*; + Ok(match (a, b) { + (Value::Null, _) | (_, Value::Null) => Value::Null, + (Value::Float64(a), Value::Float64(b)) => { + Value::Float64(crate::arithmetic::evaluate_float64_arithmetic(op, a, b)) + } + (Value::Int64(a), Value::Int64(b)) => Value::Int64( + match op { + Add => a.checked_add(b), + Sub => a.checked_sub(b), + Mul => a.checked_mul(b), + Div => a.checked_div(b), + Mod => a.checked_rem(b), + Pow => u32::try_from(b).ok().and_then(|b| a.checked_pow(b)), + Atan2 => None, + } + .ok_or_else(|| invalid("invalid integer arithmetic or overflow"))?, + ), + _ => return Err(invalid("arithmetic type mismatch")), + }) +} +#[derive(Clone, Debug)] +pub struct SortKey { + pub column: usize, + pub descending: bool, + pub nulls_first: bool, +} +#[derive(Clone, Debug)] +pub enum Reduction { + Count, + Sum(usize), + Avg(usize), + Min(usize), + Max(usize), +} +#[derive(Clone)] +enum Kind { + Source(Vec), + Union, + VectorToScalar { + column: usize, + }, + Project(Vec), + Filter(Expression), + Limit { + n: u64, + offset: u64, + groups: Vec, + }, + Sort { + keys: Vec, + groups: Vec, + }, + Aggregate { + groups: Vec, + measures: Vec, + }, + SemiJoin { + keys: Vec<(usize, usize)>, + }, + SummaryBuild { + family: SummaryFamilyType, + value: usize, + time: Option, + groups: Vec, + }, + SummaryMerge { + state: usize, + groups: Vec, + }, + Readout { + state: usize, + statistic: crate::Statistic, + parameters: std::collections::HashMap, + }, +} +/// A bound operation has a fully checked input/output contract before execution. +#[derive(Clone)] +pub struct Operator { + kind: Kind, + inputs: Vec, + output: Schema, +} +impl Operator { + pub fn source(output: Schema, batches: Vec) -> Result { + super::values::validate_schema(&output)?; + if batches.iter().any(|b| b.schema() != &output) { + return Err(invalid("source schema mismatch")); + } + Ok(Self { + kind: Kind::Source(batches), + inputs: vec![], + output, + }) + } + /// Union polls every input fairly, including branches sharing a producer. + pub fn union(input: Schema, arity: usize) -> Result { + if arity == 0 { + return Err(invalid("union needs at least one input")); + } + Ok(Self { + kind: Kind::Union, + inputs: vec![input.clone(); arity], + output: input, + }) + } + pub fn scalar(value: Value, dtype: DataType) -> Result { + let schema = schema(vec![result_field( + "value", + dtype, + matches!(value, Value::Null), + )]); + Self::source( + schema.clone(), + vec![Batch::try_new(schema, vec![vec![value]])?], + ) + } + /// PromQL scalar conversion: zero or multiple elements produce NaN. + pub fn vector_to_scalar(input: Schema, column: usize) -> Result { + if plain(&input, column)? != (&DataType::Float64, false) { + return Err(invalid("scalar conversion requires non-null Float64")); + } + Ok(Self { + kind: Kind::VectorToScalar { column }, + inputs: vec![input], + output: schema(vec![result_field("value", DataType::Float64, false)]), + }) + } + pub fn project(input: Schema, columns: Vec<(String, Expression)>) -> Result { + let fields = columns + .iter() + .map(|(name, e)| { + let (t, n) = e.dtype(&input)?; + Ok(result_field(name, t, n)) + }) + .collect::>()?; + Ok(Self { + kind: Kind::Project(columns.into_iter().map(|(_, e)| e).collect()), + inputs: vec![input], + output: schema(fields), + }) + } + pub fn filter(input: Schema, predicate: Expression) -> Result { + if predicate.dtype(&input)?.0 != DataType::Bool { + return Err(invalid("filter predicate must be boolean")); + } + Ok(Self { + kind: Kind::Filter(predicate), + inputs: vec![input.clone()], + output: input, + }) + } + pub fn limit(input: Schema, n: u64, offset: u64, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + Ok(Self { + kind: Kind::Limit { n, offset, groups }, + inputs: vec![input.clone()], + output: input, + }) + } + pub fn sort(input: Schema, keys: Vec, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + for key in &keys { + if !ordered(plain(&input, key.column)?.0) { + return Err(invalid("unsupported sort type")); + } + } + Ok(Self { + kind: Kind::Sort { keys, groups }, + inputs: vec![input.clone()], + output: input, + }) + } + pub fn aggregate( + input: Schema, + groups: Vec, + measures: Vec<(String, Reduction)>, + ) -> Result { + validate_groups(&input, &groups)?; + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + for (name, reduction) in &measures { + let (t, n) = match reduction { + Reduction::Count => (DataType::Int64, false), + Reduction::Sum(i) | Reduction::Avg(i) => { + let (t, _) = plain(&input, *i)?; + if !matches!(t, DataType::Int64 | DataType::Float64) { + return Err(invalid("numeric aggregate input required")); + } + ( + if matches!(reduction, Reduction::Avg(_)) { + DataType::Float64 + } else { + t.clone() + }, + false, + ) + } + Reduction::Min(i) | Reduction::Max(i) => { + let (t, _) = plain(&input, *i)?; + if !ordered(t) { + return Err(invalid("ordered aggregate input required")); + } + (t.clone(), true) + } + }; + fields.push(result_field(name, t, n)); + } + Ok(Self { + kind: Kind::Aggregate { + groups, + measures: measures.into_iter().map(|(_, r)| r).collect(), + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn semi_join( + left: Schema, + right: Schema, + keys: Vec<(usize, usize)>, + ) -> Result { + if keys.is_empty() { + return Err(invalid("semi-join needs matching keys")); + } + for &(l, r) in &keys { + if plain(&left, l)?.0 != plain(&right, r)?.0 { + return Err(invalid("join key types differ")); + } + } + Ok(Self { + kind: Kind::SemiJoin { keys }, + inputs: vec![left.clone(), right], + output: left, + }) + } + pub fn summary_build( + input: Schema, + family: SummaryFamilyType, + value: usize, + time: Option, + groups: Vec, + ) -> Result { + super::values::validate_family(&family)?; + validate_groups(&input, &groups)?; + if plain(&input, value)? != (&DataType::Float64, false) { + return Err(invalid("summary numeric update requires non-null Float64")); + } + if let Some(time) = time { + if plain(&input, time)? != (&DataType::Timestamp, false) { + return Err(invalid("summary time column must be a timestamp")); + } + } + if time.is_none() + && matches!( + family, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Rate + | planner_types::post_asap::ExactKind::Increase, + _ + ) + ) + { + return Err(invalid("counter summary requires a timestamp column")); + } + crate::capability::validate_summary_kernel( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .map_err(Error::Invalid)?; + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }); + Ok(Self { + kind: Kind::SummaryBuild { + family, + value, + time, + groups, + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn summary_merge(input: Schema, state: usize, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + super::values::validate_family(&field(&input, state)?.dtype)?; + if matches!(field(&input, state)?.dtype, SummaryFamilyType::Plain(_)) { + return Err(invalid("summary state required")); + } + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(input.fields[state].clone()); + Ok(Self { + kind: Kind::SummaryMerge { state, groups }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn readout( + input: Schema, + state: usize, + statistic: crate::Statistic, + parameters: std::collections::HashMap, + ) -> Result { + super::values::validate_family(&field(&input, state)?.dtype)?; + if matches!(field(&input, state)?.dtype, SummaryFamilyType::Plain(_)) { + return Err(invalid("summary state required")); + } + validate_readout(&field(&input, state)?.dtype, statistic, ¶meters)?; + let mut fields = input.fields.clone(); + let result_type = if matches!( + fields[state].dtype, + SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) + ) { + DataType::Int64 + } else { + DataType::Float64 + }; + fields[state] = result_field("value", result_type, false); + Ok(Self { + kind: Kind::Readout { + state, + statistic, + parameters, + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub(crate) fn with_output_schema(mut self, output: Schema) -> Result { + if self.output.fields.len() != output.fields.len() + || self + .output + .fields + .iter() + .zip(&output.fields) + .any(|(actual, declared)| { + actual.dtype != declared.dtype || (actual.nullable && !declared.nullable) + }) + { + return Err(invalid("native output type differs from Planner output")); + } + if output.time_index.is_some_and(|i| { + i >= output.fields.len() + || output.fields[i].dtype != SummaryFamilyType::Plain(DataType::Timestamp) + }) { + return Err(invalid("invalid output time column")); + } + self.output = output; + Ok(self) + } + pub fn schema(&self) -> Schema { + self.output.clone() + } +} +fn validate_groups(input: &Schema, groups: &[usize]) -> Result<(), Error> { + for &i in groups { + plain(input, i)?; + } + if groups + .iter() + .collect::>() + .len() + != groups.len() + { + return Err(invalid("duplicate group columns")); + } + Ok(()) +} +async fn collect_rows( + mut input: Input<'_, Batch>, + context: &RunContext, +) -> Result<(Vec>, Vec), Error> { + let mut rows = Vec::new(); + let mut reservations = Vec::new(); + while let Some(batch) = input.next().await { + let batch = batch?; + reservations.push(context.reserve(batch.bytes())?); + rows.extend(batch.rows().iter().cloned()); + } + Ok((rows, reservations)) +} +impl PhysicalOperator for Operator { + fn name(&self) -> &str { + match self.kind { + Kind::Source(_) => "Source", + Kind::Union => "Union", + Kind::VectorToScalar { .. } => "VectorToScalar", + Kind::Project(_) => "Project", + Kind::Filter(_) => "Filter", + Kind::Limit { .. } => "Limit", + Kind::Sort { .. } => "Sort", + Kind::Aggregate { .. } => "Aggregate", + Kind::SemiJoin { .. } => "SemiJoin", + Kind::SummaryBuild { .. } => "SummaryAgg", + Kind::SummaryMerge { .. } => "SummaryMerge", + Kind::Readout { .. } => "SummaryReadout", + } + } + fn input_schemas(&self) -> Vec { + self.inputs.clone() + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, value: &Batch) -> usize { + value.bytes() + } + fn start<'a>( + &'a self, + mut inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + let output = self.output.clone(); + if let Kind::Source(batches) = &self.kind { + return Ok(futures::stream::iter(batches.iter().cloned().map(Ok)).boxed_local()); + } + if matches!(self.kind, Kind::Union) { + return Ok(futures::stream::select_all(inputs) + .map(|batch| batch.map(|batch| batch.value().clone())) + .boxed_local()); + } + if let Kind::SemiJoin { keys } = &self.kind { + let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; + let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; + return Ok(futures::stream::once(async move { + // Poll both branches together: either may depend on a common producer. + let ((left, _left_memory), (right, _right_memory)) = futures::try_join!( + collect_rows(left, &context), + collect_rows(right, &context) + )?; + let right_cols = keys.iter().map(|(_, r)| *r).collect::>(); + let left_cols = keys.iter().map(|(l, _)| *l).collect::>(); + let members = right + .iter() + .filter(|row| right_cols.iter().all(|&i| !matches!(row[i], Value::Null))) + .map(|r| group_key(r, &right_cols)) + .collect::, _>>()?; + let rows = left + .into_iter() + .filter_map(|r| match group_key(&r, &left_cols) { + Ok(k) + if left_cols.iter().all(|&i| !matches!(r[i], Value::Null)) + && members.contains(&k) => + { + Some(Ok(r)) + } + Ok(_) => None, + Err(e) => Some(Err(e)), + }) + .collect::, _>>()?; + Batch::try_new(output, rows) + }) + .boxed_local()); + } + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &self.kind { + Kind::VectorToScalar { column } => Ok(futures::stream::once(async move { + let mut input = input; + let mut value = f64::NAN; + let mut count = 0usize; + while let Some(batch) = input.next().await { + for row in batch?.rows() { + count = count.saturating_add(1); + if let Value::Float64(v) = row[*column] { + value = v; + } + } + } + Batch::try_new( + output, + vec![vec![Value::Float64(if count == 1 { + value + } else { + f64::NAN + })]], + ) + }) + .boxed_local()), + Kind::Project(expressions) => Ok(input + .map(move |batch| { + let batch = batch?; + let rows = batch + .rows() + .iter() + .map(|r| { + expressions + .iter() + .map(|e| e.evaluate(r)) + .collect::, _>>() + }) + .collect::, _>>()?; + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + Kind::Filter(predicate) => Ok(input + .map(move |batch| { + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + if matches!(predicate.evaluate(row)?, Value::Bool(true)) { + rows.push(row.clone()); + } + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + Kind::Limit { n, offset, groups } => { + let counts = BTreeMap::>, u64>::new(); + Ok(futures::stream::try_unfold( + (input, counts, Vec::::new(), false), + move |(mut input, mut counts, mut memory, done)| { + let output = output.clone(); + let context = context.clone(); + async move { + if done || *n == 0 { + return Ok(None); + } + let Some(batch) = input.next().await else { + return Ok(None); + }; + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + let key = group_key(row, groups)?; + if !counts.contains_key(&key) { + memory.push( + context.reserve( + key.iter() + .map(|part| { + part.len() + std::mem::size_of::>() + }) + .sum::() + + 64, + )?, + ); + } + let count = counts.entry(key).or_default(); + if *count >= *offset && count.saturating_sub(*offset) < *n { + rows.push(row.clone()); + } + *count = count.saturating_add(1); + } + let done = groups.is_empty() + && counts + .get(&vec![]) + .is_some_and(|count| count.saturating_sub(*offset) >= *n); + Ok(Some(( + Batch::try_new(output, rows)?, + (input, counts, memory, done), + ))) + } + }, + ) + .boxed_local()) + } + Kind::SummaryBuild { + family, + value, + time, + groups, + } => Ok(futures::stream::once(async move { + Batch::try_new( + output, + build_summary(input, family, *value, *time, groups, &context).await?, + ) + }) + .boxed_local()), + Kind::Readout { + state, + statistic, + parameters, + } => Ok(input + .map(move |batch| { + let batch = batch?; + let mut rows = batch.rows().to_vec(); + for row in &mut rows { + let Value::Summary { state: summary, .. } = &row[*state] else { + return Err(invalid("summary value required")); + }; + row[*state] = if output.fields[*state].dtype + == SummaryFamilyType::Plain(DataType::Int64) + { + let count = summary.aux_stats().count.ok_or_else(|| { + Error::Operator("exact count state lacks an integer count".into()) + })?; + Value::Int64( + i64::try_from(count).map_err(|_| { + Error::Operator("exact count exceeds Int64".into()) + })?, + ) + } else { + Value::Float64( + summary + .query_statistic(*statistic, &None, parameters) + .map_err(|e| Error::Operator(e.to_string()))?, + ) + }; + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &self.kind { + Kind::Sort { keys, groups } => { + let mut grouped = BTreeMap::>, Vec>>::new(); + for row in rows { + grouped + .entry(group_key(&row, groups)?) + .or_default() + .push(row); + } + let mut result = Vec::new(); + for mut rows in grouped.into_values() { + rows.sort_by(|a, b| compare_rows(a, b, keys)); + result.extend(rows); + } + result + } + Kind::Aggregate { groups, measures } => { + reduce(rows, groups, measures, &self.inputs[0])? + } + Kind::SummaryMerge { state, groups } => merge_summary(rows, *state, groups)?, + _ => return Err(invalid("unexpected blocking operation")), + }; + Batch::try_new(output, result) + }) + .boxed_local()), + } + } +} +fn compare_rows(a: &[Value], b: &[Value], keys: &[SortKey]) -> std::cmp::Ordering { + use std::cmp::Ordering::*; + for key in keys { + let (a, b) = (&a[key.column], &b[key.column]); + let order = match (a, b) { + (Value::Null, Value::Null) => Equal, + (Value::Null, _) => { + if key.nulls_first { + Less + } else { + Greater + } + } + (_, Value::Null) => { + if key.nulls_first { + Greater + } else { + Less + } + } + (Value::Float64(a), Value::Float64(b)) if a.is_nan() || b.is_nan() => { + match (a.is_nan(), b.is_nan()) { + (true, true) => Equal, + (true, false) => Greater, + _ => Less, + } + } + _ => { + let order = a.compare(b).expect("bound ordered types"); + if key.descending { + order.reverse() + } else { + order + } + } + }; + if order != Equal { + return order; + } + } + Equal +} +fn reduce( + rows: Vec>, + groups: &[usize], + measures: &[Reduction], + input: &Schema, +) -> Result>, Error> { + let mut grouped = BTreeMap::>, Vec>>::new(); + if rows.is_empty() && groups.is_empty() { + grouped.insert(vec![], vec![]); + } + for row in rows { + grouped + .entry(group_key(&row, groups)?) + .or_default() + .push(row); + } + grouped + .into_values() + .map(|rows| { + let mut result = groups + .iter() + .map(|&i| rows[0][i].clone()) + .collect::>(); + for measure in measures { + result.push(reduce_one(&rows, measure, input)?); + } + Ok(result) + }) + .collect() +} +fn reduce_one(rows: &[Vec], measure: &Reduction, input: &Schema) -> Result { + let column = match measure { + Reduction::Count => { + return Ok(Value::Int64( + i64::try_from(rows.len()).map_err(|_| invalid("count overflow"))?, + )) + } + Reduction::Sum(i) | Reduction::Avg(i) | Reduction::Min(i) | Reduction::Max(i) => *i, + }; + let values = rows + .iter() + .map(|r| &r[column]) + .filter(|v| !matches!(v, Value::Null)) + .collect::>(); + if matches!(measure, Reduction::Min(_) | Reduction::Max(_)) { + if plain(input, column)?.0 == &DataType::Float64 { + // Match exact-state kernels: ignore NaN when a numeric value exists. + let mut best: Option = None; + for value in values { + let Value::Float64(value) = value else { + return Err(invalid("floating aggregate value required")); + }; + best = Some(best.map_or(*value, |old| { + if matches!(measure, Reduction::Min(_)) { + old.min(*value) + } else { + old.max(*value) + } + })); + } + return Ok(best.map(Value::Float64).unwrap_or(Value::Null)); + } + let mut best: Option<&Value> = None; + for value in values { + if best + .map(|b| value.compare(b)) + .transpose()? + .is_none_or(|order| { + if matches!(measure, Reduction::Min(_)) { + order.is_lt() + } else { + order.is_gt() + } + }) + { + best = Some(value); + } + } + return Ok(best.cloned().unwrap_or(Value::Null)); + } + let count = values.len(); + let dtype = plain(input, column)?.0; + if dtype == &DataType::Int64 { + let sum = values.into_iter().try_fold(0i128, |sum, v| { + let Value::Int64(v) = v else { + return Err(invalid("integer aggregate value required")); + }; + sum.checked_add(i128::from(*v)) + .ok_or_else(|| invalid("integer aggregate overflow")) + })?; + return if matches!(measure, Reduction::Avg(_)) { + Ok(Value::Float64(sum as f64 / count as f64)) + } else { + Ok(Value::Int64( + i64::try_from(sum).map_err(|_| invalid("integer sum overflow"))?, + )) + }; + } + let sum = values + .into_iter() + .map(|v| { + if let Value::Float64(v) = v { + *v + } else { + unreachable!() + } + }) + .sum::(); + Ok(Value::Float64(if matches!(measure, Reduction::Avg(_)) { + sum / count as f64 + } else { + sum + })) +} +async fn build_summary( + mut input: Input<'_, Batch>, + family: &SummaryFamilyType, + value: usize, + time: Option, + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + type State = ( + Vec, + Box, + Reservation, + usize, + Option, + ); + let create = |labels: Vec, key_bytes: usize| -> Result { + let updater = crate::factory::create_planner_accumulator( + family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .map_err(Error::Operator)?; + let overhead = labels.iter().map(Value::bytes).sum::() + key_bytes + 64; + let memory = context.reserve(updater.memory_usage_bytes() + overhead)?; + Ok((labels, updater, memory, overhead, None)) + }; + let mut states = BTreeMap::>, State>::new(); + if groups.is_empty() { + states.insert(vec![], create(vec![], 0)?); + } + let ordered_time = matches!( + family, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Rate + | planner_types::post_asap::ExactKind::Increase, + _ + ) + ); + while let Some(batch) = input.next().await { + let batch = batch?; + for row in batch.rows() { + let key = group_key(row, groups)?; + if !states.contains_key(&key) { + let labels = groups.iter().map(|&i| row[i].clone()).collect(); + let state = create( + labels, + key.iter() + .map(|v| v.len() + std::mem::size_of::>()) + .sum(), + )?; + states.insert(key.clone(), state); + } + let (_, updater, memory, overhead, previous) = + states.get_mut(&key).expect("inserted group"); + let Value::Float64(value) = row[value] else { + return Err(invalid("summary update type")); + }; + let timestamp = if let Some(time) = time { + let Value::Timestamp(time) = row[time] else { + return Err(invalid("summary time type")); + }; + time + } else { + 0 + }; + if ordered_time && previous.is_some_and(|prior| timestamp <= prior) { + return Err(Error::Operator( + "counter samples must have strictly increasing timestamps within each group" + .into(), + )); + } + updater + .validate_single_input(value) + .map_err(Error::Operator)?; + updater.update_single(value, timestamp); + *previous = Some(timestamp); + memory.resize(updater.memory_usage_bytes() + *overhead)?; + } + } + Ok(states + .into_values() + .map(|(mut labels, updater, _memory, _, _)| { + labels.push(Value::Summary { + family: family.clone(), + state: Arc::from(updater.into_accumulator()), + }); + labels + }) + .collect()) +} + +fn merge_summary( + rows: Vec>, + state_column: usize, + groups: &[usize], +) -> Result>, Error> { + type GroupState = (Vec, SummaryFamilyType, Arc); + let mut states: BTreeMap>, GroupState> = BTreeMap::new(); + for row in rows { + let Value::Summary { family, state } = &row[state_column] else { + return Err(invalid("summary state required")); + }; + let key = group_key(&row, groups)?; + if let Some((_, expected, existing)) = states.get_mut(&key) { + if expected != family { + return Err(invalid("incompatible summary family")); + } + *existing = Arc::from( + existing + .merge_with(state.as_ref()) + .map_err(|e| Error::Operator(e.to_string()))?, + ); + } else { + states.insert( + key, + ( + groups.iter().map(|&i| row[i].clone()).collect(), + family.clone(), + state.clone(), + ), + ); + } + } + Ok(states + .into_values() + .map(|(mut keys, family, state)| { + keys.push(Value::Summary { family, state }); + keys + }) + .collect()) +} + +fn validate_readout( + family: &SummaryFamilyType, + statistic: crate::Statistic, + parameters: &std::collections::HashMap, +) -> Result<(), Error> { + use crate::Statistic as S; + use planner_types::post_asap::{ExactKind as E, SketchAlgorithm as A}; + let supported = match family { + SummaryFamilyType::ExactAggregate(kind, _) => matches!( + (kind, statistic), + (E::Sum, S::Sum) + | (E::Count, S::Count) + | (E::Min, S::Min) + | (E::Max, S::Max) + | (E::Rate, S::Rate) + | (E::Increase, S::Increase) + ), + SummaryFamilyType::Sketch(kind, _) => match kind.algorithm() { + A::Kll => statistic == S::Quantile, + A::DDSketch => matches!(statistic, S::Quantile | S::Count), + A::Hll => matches!(statistic, S::Cardinality | S::Count), + _ => false, + }, + _ => false, + }; + if !supported { + return Err(invalid( + "readout is not implemented for this summary family", + )); + } + if statistic == S::Quantile + && !parameters + .get("quantile") + .and_then(|s| s.parse::().ok()) + .is_some_and(|q| (0.0..=1.0).contains(&q)) + { + return Err(invalid("quantile readout requires quantile in [0,1]")); + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/dag/planner.rs b/crates/asap-physical-operators/src/dag/planner.rs new file mode 100644 index 00000000..71aa41a7 --- /dev/null +++ b/crates/asap-physical-operators/src/dag/planner.rs @@ -0,0 +1,558 @@ +//! Bind a post-ASAP DAG to native operators. Sources are explicit execution +//! frontiers supplied by the deployment; unsupported computation is an error. +use super::{ + operators::{Expression, Operator, Reduction, SortKey}, + values::{Batch, Schema, Value}, + Error, NodeId, PhysicalDag, PhysicalOperator, +}; +use planner_types::{ + post_asap::{ + ExactOperation, ExecutableDag, ExecutableDagNode, ExecutableOperatorPayload as Payload, + SketchQuery, SummaryFamilyType, SummaryInputExpr, ValueOperation, + }, + pre_asap::{ + AggIntent, ColumnRef, CompareOpKind, DataType, GroupKeys, QueryExpr, + Reduction as PlannerReduction, ScalarValue, + }, +}; +use std::{ + collections::{BTreeMap, BTreeSet}, + sync::Arc, +}; +fn invalid(message: impl Into) -> Error { + Error::Invalid(message.into()) +} + +/// Source nodes cut the DAG at an installed storage/ingestion frontier. The +/// binding must have exactly the declared schema and no upstream dependencies. +/// A deployment must authorize these frontiers before calling this function. +pub type Source<'a> = Box + 'a>; + +pub fn bind<'a>( + dag: &ExecutableDag, + mut sources: BTreeMap>, + roots: &[NodeId], +) -> Result, Error> { + preflight_depth(dag)?; + dag.validate().map_err(|e| invalid(e.to_string()))?; + let nodes = dag + .nodes + .iter() + .map(|node| (u64::from(node.id.0), node)) + .collect::>(); + let mut dependencies = BTreeMap::>::new(); + // Binary input order is semantic; serialized edge order is not. + let mut edges = dag.edges.iter().collect::>(); + edges.sort_by_key(|edge| { + ( + edge.consumer.0, + match edge.role { + planner_types::post_asap::EdgeRole::Left => 0, + planner_types::post_asap::EdgeRole::Input => 1, + planner_types::post_asap::EdgeRole::Right => 2, + }, + ) + }); + for edge in edges { + dependencies + .entry(u64::from(edge.consumer.0)) + .or_default() + .push(u64::from(edge.producer.0)); + } + if sources.keys().any(|id| !nodes.contains_key(id)) { + return Err(invalid("source binding names an unknown node")); + } + let mut ordered = Vec::new(); + let mut seen = BTreeSet::new(); + let mut pending = roots.iter().map(|&id| (id, false)).collect::>(); + while let Some((id, expanded)) = pending.pop() { + if expanded { + ordered.push(id); + continue; + } + if !seen.insert(id) { + continue; + } + if !nodes.contains_key(&id) { + return Err(invalid(format!("missing root {id}"))); + } + pending.push((id, true)); + if !sources.contains_key(&id) { + for &input in dependencies.get(&id).into_iter().flatten() { + pending.push((input, false)); + } + } + } + let mut graph = PhysicalDag::default(); + let mut auxiliary = u64::MAX; + for id in ordered { + let node = nodes[&id]; + let output = Arc::new(node.output_schema.clone()); + super::values::validate_schema(&output)?; + let (operator, inputs) = if let Some(source) = sources.remove(&id) { + if !source.input_schemas().is_empty() || source.output_schema() != output { + return Err(invalid("frontier is not a source with the declared schema")); + } + ( + Box::new(CheckedSource { source, output }) as Source<'a>, + vec![], + ) + } else { + let mut inputs = dependencies.get(&id).cloned().unwrap_or_default(); + let mut schemas = inputs + .iter() + .map(|id| Arc::new(nodes[id].output_schema.clone())) + .collect::>(); + if matches!(node.payload, Payload::SummaryMerge { .. }) && inputs.len() > 1 { + if schemas.iter().any(|s| s != &schemas[0]) { + return Err(invalid("summary merge inputs have different schemas")); + } + graph.add( + auxiliary, + inputs, + Operator::union(schemas[0].clone(), schemas.len())?, + )?; + inputs = vec![auxiliary]; + auxiliary -= 1; + schemas.truncate(1); + } + let operator = bind_operation(node, &schemas) + .map_err(|error| invalid(format!("node {id}: {error}")))? + .with_output_schema(output)?; + (Box::new(operator) as Source<'a>, inputs) + }; + graph.add_boxed(id, inputs, operator)?; + } + graph.validate(roots)?; + Ok(graph) +} + +fn bind_operation(node: &ExecutableDagNode, inputs: &[Schema]) -> Result { + if let Payload::RelationalJoin { + join_kind: planner_types::pre_asap::JoinKind::Semi, + pred, + .. + } = &node.payload + { + let [left, right] = inputs else { + return Err(invalid("semi-join requires two inputs")); + }; + let mut keys = Vec::new(); + semi_join_keys(&pred.0, left.fields.len(), right.fields.len(), &mut keys)?; + return Operator::semi_join(left.clone(), right.clone(), keys); + } + let [input] = inputs else { + return Err(invalid( + "native Planner binding currently requires a unary operation or an explicit source", + )); + }; + match &node.payload { + Payload::Value { operation, .. } => match operation { + ValueOperation::Project { cols, .. } => Operator::project( + input.clone(), + cols.iter() + .enumerate() + .map(|(i, col)| { + Ok(( + node.output_schema + .fields + .get(i) + .ok_or_else(|| invalid("projection width mismatch"))? + .name + .clone(), + expression(&col.expr)?, + )) + }) + .collect::>()?, + ), + ValueOperation::Filter { pred } => { + Operator::filter(input.clone(), expression(&pred.0)?) + } + ValueOperation::Sort { keys, partition_by } => Operator::sort( + input.clone(), + keys.iter() + .map(|key| { + let QueryExpr::Column(column) = key.expr else { + return Err(invalid( + "sort expression must be projected before sorting", + )); + }; + Ok(SortKey { + column, + descending: !key.ascending, + nulls_first: key.nulls_first, + }) + }) + .collect::>()?, + groups(input, partition_by)?, + ), + ValueOperation::Limit { + n, + offset, + partition_by, + } => Operator::limit( + input.clone(), + *n as u64, + *offset as u64, + groups(input, partition_by)?, + ), + ValueOperation::Exact(ExactOperation::Aggregate { + reduction, + measures, + output_names, + having: None, + }) => { + if measures.len() != output_names.len() { + return Err(invalid("aggregate output names differ from measures")); + } + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid( + "per-entity aggregate requires an explicit entity binding", + )); + }; + let measures = measures + .iter() + .zip(output_names) + .map(|(m, name)| { + let column = |col: Option| { + col.map(Ok) + .unwrap_or_else(|| named_column(input, &ColumnRef::SampleValue)) + }; + let m = match m { + AggIntent::Count { .. } => Reduction::Count, + AggIntent::Sum { col } => Reduction::Sum(column(*col)?), + AggIntent::Avg { col } => Reduction::Avg(column(*col)?), + AggIntent::Min { col } => Reduction::Min(column(*col)?), + AggIntent::Max { col } => Reduction::Max(column(*col)?), + _ => { + return Err(invalid( + "aggregate intent has no native implementation", + )) + } + }; + Ok((name.clone(), m)) + }) + .collect::>()?; + Operator::aggregate(input.clone(), groups(input, keys)?, measures) + } + ValueOperation::FinalizeExactAccumulator => { + let state = summary_column(input)?; + use crate::Statistic as S; + use planner_types::post_asap::ExactKind as E; + let statistic = match &input.fields[state].dtype { + SummaryFamilyType::ExactAggregate(kind, _) => match kind { + E::Sum => S::Sum, + E::Count => S::Count, + E::Min => S::Min, + E::Max => S::Max, + E::Rate => S::Rate, + E::Increase => S::Increase, + _ => return Err(invalid("exact family readout is unsupported")), + }, + _ => return Err(invalid("exact finalization requires exact state")), + }; + Operator::readout(input.clone(), state, statistic, Default::default()) + } + _ => Err(invalid("value operation has no native implementation")), + }, + Payload::SummaryAgg { + family, + input: update, + reduction, + grouping, + } => { + if update.item.is_some() { + return Err(invalid("keyed summary update binding is not implemented")); + } + crate::capability::validate_summary_kernel(family, update, grouping) + .map_err(Error::Invalid)?; + let SummaryInputExpr::Column(column) = &update.weight else { + return Err(invalid( + "summary update expression must be projected to a column", + )); + }; + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid( + "summary construction requires explicit grouping columns", + )); + }; + Operator::summary_build( + input.clone(), + family.clone(), + named_column(input, column)?, + input.time_index, + groups(input, keys)?, + ) + } + Payload::SummaryMerge { .. } => { + let state = summary_column(input)?; + Operator::summary_merge( + input.clone(), + state, + (0..input.fields.len()) + .filter(|&i| i != state && Some(i) != input.time_index) + .collect(), + ) + } + Payload::SummaryEstimate { query } => { + let mut params = std::collections::HashMap::new(); + let statistic = match query { + SketchQuery::Quantile { q } => { + params.insert("quantile".into(), q.to_string()); + crate::Statistic::Quantile + } + SketchQuery::Cardinality => crate::Statistic::Cardinality, + SketchQuery::PointCount { value: None, .. } => crate::Statistic::Count, + _ => return Err(invalid("summary readout is not implemented")), + }; + Operator::readout(input.clone(), summary_column(input)?, statistic, params) + } + _ => Err(invalid( + "physical operation has no native binding; no fallback is installed", + )), + } +} +fn summary_column(input: &Schema) -> Result { + let columns = input + .fields + .iter() + .enumerate() + .filter(|(_, f)| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + .map(|(i, _)| i) + .collect::>(); + match columns.as_slice() { + [column] => Ok(*column), + _ => Err(invalid("one summary state column required")), + } +} +fn named_column(input: &Schema, column: &ColumnRef) -> Result { + let name = match column { + ColumnRef::Named(name) => name.as_str(), + ColumnRef::SampleValue => "value", + _ => { + return Err(invalid( + "summary update requires an unambiguous bound column", + )) + } + }; + let matches = input + .fields + .iter() + .enumerate() + .filter(|(_, field)| field.name == name) + .map(|(i, _)| i) + .collect::>(); + match matches.as_slice() { + [column] => Ok(*column), + _ => Err(invalid("summary update column missing or ambiguous")), + } +} +fn groups(input: &Schema, groups: &GroupKeys) -> Result, Error> { + if groups.is_without() { + return Err(invalid("grouping without requires resolved label columns")); + } + if groups.keys().iter().any(|&i| i >= input.fields.len()) { + return Err(invalid("grouping column out of range")); + } + Ok(groups.keys().to_vec()) +} +fn expression(expr: &QueryExpr) -> Result { + let bind = |e: &QueryExpr| expression(e).map(Box::new); + Ok(match expr { + QueryExpr::Column(i) => Expression::Column(*i), + QueryExpr::Literal(value) => { + let (value, dtype) = match value { + ScalarValue::Int64(v) => (Value::Int64(*v), DataType::Int64), + ScalarValue::Float64(v) => (Value::Float64(*v), DataType::Float64), + ScalarValue::Utf8(v) => (Value::Utf8(v.as_str().into()), DataType::Utf8), + ScalarValue::Boolean(v) => (Value::Bool(*v), DataType::Bool), + ScalarValue::Null => (Value::Null, DataType::Null), + ScalarValue::Interval { + months, + days, + nanos, + } => ( + Value::Interval { + months: *months, + days: *days, + nanos: *nanos, + }, + DataType::Interval, + ), + }; + Expression::Literal { value, dtype } + } + QueryExpr::Arithmetic { op, left, right } => Expression::Arithmetic { + op: op.clone(), + left: bind(left)?, + right: bind(right)?, + }, + QueryExpr::Compare { + left, + op: CompareOpKind::Eq, + right, + } => Expression::Equal(bind(left)?, bind(right)?), + QueryExpr::Compare { + left, + op: CompareOpKind::Lt, + right, + } => Expression::Less(bind(left)?, bind(right)?), + QueryExpr::Not(v) => Expression::Not(bind(v)?), + QueryExpr::IsNull(v) => Expression::IsNull(bind(v)?), + QueryExpr::IsNotNull(v) => Expression::Not(Box::new(Expression::IsNull(bind(v)?))), + QueryExpr::BoolAnd(items) | QueryExpr::BoolOr(items) => { + let and = matches!(expr, QueryExpr::BoolAnd(_)); + let mut result = Expression::Literal { + value: Value::Bool(and), + dtype: DataType::Bool, + }; + for item in items { + result = if and { + Expression::And(Box::new(result), bind(item)?) + } else { + Expression::Or(Box::new(result), bind(item)?) + }; + } + result + } + _ => return Err(invalid("expression has no native implementation")), + }) +} + +// Source adapters may perform I/O, but their actual batches must honor the +// schema accepted by the binder before a downstream expression sees a row. +struct CheckedSource<'a> { + source: Source<'a>, + output: Schema, +} +impl PhysicalOperator for CheckedSource<'_> { + fn name(&self) -> &str { + self.source.name() + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, batch: &Batch) -> usize { + self.source.output_bytes(batch) + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: super::RunContext, + ) -> Result, Error> { + use futures::StreamExt; + Ok(self + .source + .start(inputs, context)? + .map(|batch| { + let batch = batch?; + if batch.schema() != &self.output { + return Err(invalid("source batch differs from its bound schema")); + } + Ok(batch) + }) + .boxed_local()) + } +} + +// Bound recursion before invoking the upstream recursive provenance validator. +fn preflight_depth(dag: &ExecutableDag) -> Result<(), Error> { + let mut remaining = dag + .nodes + .iter() + .map(|node| (node.id, 0usize)) + .collect::>(); + if remaining.len() != dag.nodes.len() { + return Err(invalid("duplicate Planner node")); + } + let mut consumers = BTreeMap::<_, Vec<_>>::new(); + for edge in &dag.edges { + if !remaining.contains_key(&edge.producer) { + return Err(invalid("missing Planner edge producer")); + } + *remaining + .get_mut(&edge.consumer) + .ok_or_else(|| invalid("missing Planner edge consumer"))? += 1; + consumers + .entry(edge.producer) + .or_default() + .push(edge.consumer); + } + let mut ready = remaining + .iter() + .filter(|(_, n)| **n == 0) + .map(|(id, _)| *id) + .collect::>(); + let mut depths = BTreeMap::new(); + let mut visited = 0; + while let Some(id) = ready.pop_front() { + visited += 1; + let depth = *depths.get(&id).unwrap_or(&1usize); + if depth > 128 { + return Err(invalid("DAG exceeds the supported execution depth of 128")); + } + for &consumer in consumers.get(&id).into_iter().flatten() { + let next = depths.entry(consumer).or_insert(1); + *next = (*next).max(depth + 1); + let count = remaining.get_mut(&consumer).expect("validated endpoint"); + *count -= 1; + if *count == 0 { + ready.push_back(consumer); + } + } + } + if visited != dag.nodes.len() { + return Err(invalid("Planner DAG contains a cycle")); + } + Ok(()) +} + +/// Join predicates address the concatenated left/right schema. +fn semi_join_keys( + expr: &QueryExpr, + left: usize, + right: usize, + keys: &mut Vec<(usize, usize)>, +) -> Result<(), Error> { + match expr { + QueryExpr::BoolAnd(parts) => { + for part in parts { + semi_join_keys(part, left, right, keys)?; + } + } + QueryExpr::Compare { + left: a, + op: CompareOpKind::Eq, + right: b, + } => { + let (QueryExpr::Column(a), QueryExpr::Column(b)) = (a.as_ref(), b.as_ref()) else { + return Err(invalid("semi-join requires column equality keys")); + }; + let (a, b) = if a < b { (*a, *b) } else { (*b, *a) }; + if a >= left || b < left || b >= left + right { + return Err(invalid("semi-join key must match left to right")); + } + keys.push((a, b - left)); + } + _ => return Err(invalid("unsupported semi-join predicate")), + } + Ok(()) +} + +/// Resolve equality keys against the Planner join's concatenated input schema. +/// Deployments may use these positions to bind their source columns. +pub fn equijoin_keys( + pred: &planner_types::pre_asap::Predicate, + left: &planner_types::post_asap::SummarySchema, + right: &planner_types::post_asap::SummarySchema, +) -> Result, Error> { + let mut keys = Vec::new(); + semi_join_keys(&pred.0, left.fields.len(), right.fields.len(), &mut keys)?; + if keys.is_empty() { + return Err(invalid("semi-join requires explicit matching keys")); + } + Ok(keys) +} diff --git a/crates/asap-physical-operators/src/dag/tests.rs b/crates/asap-physical-operators/src/dag/tests.rs new file mode 100644 index 00000000..e051fa31 --- /dev/null +++ b/crates/asap-physical-operators/src/dag/tests.rs @@ -0,0 +1,260 @@ +use super::*; +use futures::{executor::block_on, stream, StreamExt}; + +struct Source { + starts: Rc>, + polls: Rc>, + fail: bool, + end: u64, +} +impl PhysicalOperator for Source { + fn name(&self) -> &str { + "CountingSource" + } + fn input_schemas(&self) -> Vec<()> { + vec![] + } + fn output_schema(&self) {} + fn output_bytes(&self, _: &u64) -> usize { + 8 + } + fn start<'a>( + &'a self, + _: Vec>, + _: RunContext, + ) -> Result, Error> { + self.starts.set(self.starts.get() + 1); + Ok(stream::iter(0..self.end) + .map(move |n| { + self.polls.set(self.polls.get() + 1); + if self.fail && n == 1 { + Err(Error::Operator("source failure".into())) + } else { + Ok(n) + } + }) + .boxed_local()) + } +} +struct Identity; +impl PhysicalOperator for Identity { + fn name(&self) -> &str { + "Identity" + } + fn input_schemas(&self) -> Vec<()> { + vec![()] + } + fn output_schema(&self) {} + fn output_bytes(&self, _: &u64) -> usize { + 8 + } + fn start<'a>( + &'a self, + mut inputs: Vec>, + _: RunContext, + ) -> Result, Error> { + Ok(inputs + .remove(0) + .map(|value| value.map(|v| *v)) + .boxed_local()) + } +} +fn context() -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 100, + revision: 1, + }, + Limits { + max_buffered_batches: 1, + max_bytes: 1024, + }, + ) + .unwrap() +} +fn source(fail: bool) -> (Source, Rc>, Rc>) { + let starts = Rc::new(Cell::new(0)); + let polls = Rc::new(Cell::new(0)); + ( + Source { + starts: starts.clone(), + polls: polls.clone(), + fail, + end: 4, + }, + starts, + polls, + ) +} + +// A shared producer runs once, and the slow reader bounds producer progress. +#[test] +fn shared_source_backpressure_and_reader_drop() { + let (source, starts, polls) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let context = context(); + let mut readers = dag.execute(&[0, 0], context.clone()).unwrap(); + let mut slow = readers.pop().unwrap(); + let mut fast = readers.pop().unwrap(); + assert_eq!(starts.get(), 1); + let first = block_on(fast.next()).unwrap().unwrap(); + assert_eq!(*first, 0); + let mut cx = Context::from_waker(futures::task::noop_waker_ref()); + assert!(Pin::new(&mut fast).poll_next(&mut cx).is_pending()); + assert_eq!(polls.get(), 1); + let same = block_on(slow.next()).unwrap().unwrap(); + assert!(Arc::ptr_eq(&first.value, &same.value)); + drop(same); + drop(first); + assert_eq!(context.retained_bytes(), 0); + assert_eq!(*block_on(fast.next()).unwrap().unwrap(), 1); + drop(slow); + assert_eq!(*block_on(fast.next()).unwrap().unwrap(), 2); + assert_eq!(*block_on(fast.next()).unwrap().unwrap(), 3); + assert!(block_on(fast.next()).is_none()); + assert_eq!(polls.get(), 4); + drop(fast); + assert_eq!(context.retained_bytes(), 0); +} + +// Independent branches consume a common node concurrently without duplicate work. +#[test] +fn diamond_and_run_isolation() { + let (source, starts, polls) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + dag.add(1, vec![0], Identity).unwrap(); + dag.add(2, vec![0], Identity).unwrap(); + for _ in 0..2 { + let mut outputs = dag.execute(&[1, 2], context()).unwrap(); + let a = outputs.pop().unwrap(); + let b = outputs.pop().unwrap(); + let (a, b) = + block_on(async { futures::join!(a.collect::>(), b.collect::>()) }); + assert_eq!( + a.iter().map(|v| **v.as_ref().unwrap()).collect::>(), + vec![0, 1, 2, 3] + ); + assert_eq!( + b.iter().map(|v| **v.as_ref().unwrap()).collect::>(), + vec![0, 1, 2, 3] + ); + } + assert_eq!(starts.get(), 2); + assert_eq!(polls.get(), 8); +} + +// Failure reaches every subscriber; cancellation stops further producer work. +#[test] +fn broadcast_error_and_cancel() { + let (source, _, polls) = source(true); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let mut outputs = dag.execute(&[0, 0], context()).unwrap(); + let a = outputs.pop().unwrap(); + let b = outputs.pop().unwrap(); + let (a, b) = block_on(async { futures::join!(a.collect::>(), b.collect::>()) }); + for values in [a, b] { + assert_eq!(values.len(), 2); + assert!(matches!(values[1], Err(Error::AtNode { node: 0, .. }))); + } + assert_eq!(polls.get(), 2); + let run = context(); + let mut output = dag.execute(&[0], run.clone()).unwrap().remove(0); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + assert!(block_on(output.next()).is_none()); + assert_eq!(polls.get(), 2); +} + +// Retaining a consumer output retains its budget lease after queue eviction. +#[test] +fn retained_outputs_count_against_budget() { + let (source, _, _) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let run = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_buffered_batches: 1, + max_bytes: 8, + }, + ) + .unwrap(); + let mut input = dag.execute(&[0], run.clone()).unwrap().remove(0); + let held = block_on(input.next()).unwrap().unwrap(); + assert_eq!(run.retained_bytes(), 8); + assert!(matches!( + block_on(input.next()), + Some(Err(Error::MemoryLimit)) + )); + drop(input); + assert_eq!(run.retained_bytes(), 8); + drop(held); + assert_eq!(run.retained_bytes(), 0); +} + +// Invalid graphs fail before even starting a source. +#[test] +fn invalid_graphs_do_not_start_sources() { + let (source, starts, _) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + dag.add(1, vec![2], Identity).unwrap(); + dag.add(2, vec![1], Identity).unwrap(); + assert!(dag.execute(&[0, 1], context()).is_err()); + assert_eq!(starts.get(), 0); + let mut missing = PhysicalDag::default(); + missing.add(1, vec![9], Identity).unwrap(); + assert!(missing.validate(&[1]).is_err()); + let mut arity = PhysicalDag::default(); + arity.add(1, vec![], Identity).unwrap(); + assert!(arity.validate(&[1]).is_err()); +} + +// An always-ready source must yield so cancellation can be polled on this worker. +#[test] +fn ready_sources_cooperate_with_cancellation() { + let (mut source, _, polls) = source(false); + source.end = 10_000; + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let context = context(); + let mut input = dag.execute(&[0], context.clone()).unwrap().remove(0); + block_on(async { + let drain = async { + while let Some(result) = input.next().await { + if let Err(error) = result { + assert_eq!(error, Error::Cancelled); + return; + } + } + panic!("source completed without yielding"); + }; + let cancel = async { + context.cancel(); + }; + futures::join!(drain, cancel); + }); + assert_eq!(polls.get(), 32); + assert_eq!(context.retained_bytes(), 0); +} + +// Cached shorter paths must not hide an over-deep path through shared nodes. +#[test] +fn depth_limit_covers_shared_paths() { + let (source, _, _) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + for id in 1..129 { + dag.add(id, vec![id - 1], Identity).unwrap(); + } + assert!(dag.validate(&(0..129).collect::>()).is_err()); +} diff --git a/crates/asap-physical-operators/src/dag/values.rs b/crates/asap-physical-operators/src/dag/values.rs new file mode 100644 index 00000000..82b87314 --- /dev/null +++ b/crates/asap-physical-operators/src/dag/values.rs @@ -0,0 +1,327 @@ +//! Runtime values preserve Planner schemas; summary states are typed values too. +use super::Error; +use crate::AggregateCore; +use planner_types::{ + post_asap::{SummaryFamilyType, SummarySchema}, + pre_asap::DataType, +}; +use std::{cmp::Ordering, sync::Arc}; +pub type Schema = Arc; +#[derive(Clone)] +pub enum Value { + Null, + Bool(bool), + Int64(i64), + Float64(f64), + Utf8(Arc), + Timestamp(i64), + Date(i32), + Interval { + months: i32, + days: i32, + nanos: i64, + }, + List(Arc<[Value]>), + Struct(Arc<[Value]>), + Map(Arc<[(Value, Value)]>), + Summary { + family: SummaryFamilyType, + state: Arc, + }, +} +impl std::fmt::Debug for Value { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Summary { family, .. } => f.debug_tuple("Summary").field(family).finish(), + _ => write!(f, "{:?}", self.key()), + } + } +} +impl Value { + pub fn bytes(&self) -> usize { + std::mem::size_of::() + + match self { + Self::Utf8(s) => s.len(), + Self::List(v) | Self::Struct(v) => v.iter().map(Self::bytes).sum(), + Self::Map(v) => v.iter().map(|(k, v)| k.bytes() + v.bytes()).sum(), + Self::Summary { state, .. } => state.approx_memory_bytes(), + _ => 0, + } + } + pub fn matches(&self, dtype: &DataType, nullable: bool) -> bool { + if matches!(self, Self::Null) { + return nullable || matches!(dtype, DataType::Null); + } + match (self, dtype) { + (Self::Bool(_), DataType::Bool) + | (Self::Int64(_), DataType::Int64) + | (Self::Float64(_), DataType::Float64) + | (Self::Utf8(_), DataType::Utf8) + | (Self::Timestamp(_), DataType::Timestamp) + | (Self::Date(_), DataType::Date) + | (Self::Interval { .. }, DataType::Interval) => true, + (Self::List(v), DataType::List { element }) => v + .iter() + .all(|v| v.matches(&element.dtype, element.nullable)), + (Self::Struct(v), DataType::Struct { fields }) => { + v.len() == fields.len() + && v.iter() + .zip(fields) + .all(|(v, f)| v.matches(&f.dtype, f.nullable)) + } + ( + Self::Map(v), + DataType::Map { + key, + value, + value_nullable, + }, + ) => v + .iter() + .all(|(k, v)| k.matches(key, false) && v.matches(value, *value_nullable)), + _ => false, + } + } + /// Stable typed equality key. Zero signs and NaN payloads form one group. + pub fn key(&self) -> Result, Error> { + let mut out = Vec::new(); + macro_rules! number { + ($tag:expr,$v:expr) => {{ + out.push($tag); + out.extend_from_slice(&$v.to_le_bytes()); + }}; + } + match self { + Self::Null => out.push(0), + Self::Bool(v) => out.extend([1, *v as u8]), + Self::Int64(v) => number!(2, v), + Self::Float64(v) => { + let bits = if *v == 0. { + 0 + } else if v.is_nan() { + f64::NAN.to_bits() + } else { + v.to_bits() + }; + number!(3, bits); + } + Self::Utf8(v) => { + out.push(4); + out.extend(v.as_bytes()); + } + Self::Timestamp(v) => number!(5, v), + Self::Date(v) => number!(6, v), + Self::Interval { + months, + days, + nanos, + } => { + number!(7, months); + number!(8, days); + number!(9, nanos); + } + Self::List(v) | Self::Struct(v) => { + out.push(if matches!(self, Self::List(_)) { + 10 + } else { + 11 + }); + for v in v.iter() { + let key = v.key()?; + out.extend((key.len() as u64).to_le_bytes()); + out.extend(key); + } + } + Self::Map(v) => { + out.push(12); + for (k, v) in v.iter() { + for value in [k, v] { + let key = value.key()?; + out.extend((key.len() as u64).to_le_bytes()); + out.extend(key); + } + } + } + Self::Summary { .. } => { + return Err(Error::Invalid( + "summary states cannot be grouping keys".into(), + )) + } + } + Ok(out) + } + pub fn compare(&self, other: &Self) -> Result { + Ok(match (self, other) { + (Self::Null, Self::Null) => Ordering::Equal, + (Self::Int64(a), Self::Int64(b)) | (Self::Timestamp(a), Self::Timestamp(b)) => a.cmp(b), + (Self::Float64(a), Self::Float64(b)) => { + if a == b { + Ordering::Equal + } else { + a.total_cmp(b) + } + } + (Self::Utf8(a), Self::Utf8(b)) => a.cmp(b), + (Self::Bool(a), Self::Bool(b)) => a.cmp(b), + (Self::Date(a), Self::Date(b)) => a.cmp(b), + _ => { + return Err(Error::Operator( + "values do not have a supported common ordering".into(), + )) + } + }) + } +} +#[derive(Clone, Debug)] +pub struct Batch { + schema: Schema, + rows: Vec>, +} +impl Batch { + pub fn try_new(schema: Schema, rows: Vec>) -> Result { + validate_schema(&schema)?; + for row in &rows { + if row.len() != schema.fields.len() { + return Err(Error::Invalid( + "row width differs from Planner schema".into(), + )); + } + for (value, field) in row.iter().zip(&schema.fields) { + let matches = match (&field.dtype, value) { + (SummaryFamilyType::Plain(dtype), value) => { + value.matches(dtype, field.nullable) + } + (expected, Value::Summary { family, state }) => { + expected == family && validate_state(family, state.as_ref()).is_ok() + } + _ => false, + }; + if !matches { + return Err(Error::Invalid(format!( + "value differs from type of {}", + field.name + ))); + } + } + } + Ok(Self { schema, rows }) + } + pub fn schema(&self) -> &Schema { + &self.schema + } + pub fn rows(&self) -> &[Vec] { + &self.rows + } + pub fn bytes(&self) -> usize { + std::mem::size_of::() + + self + .rows + .iter() + .flat_map(|r| r.iter()) + .map(Value::bytes) + .sum::() + } +} +pub(crate) fn group_key(row: &[Value], columns: &[usize]) -> Result>, Error> { + columns + .iter() + .map(|&i| { + row.get(i) + .ok_or_else(|| Error::Invalid("group column out of range".into()))? + .key() + }) + .collect() +} + +pub(crate) fn validate_family(family: &SummaryFamilyType) -> Result<(), Error> { + use planner_types::post_asap::SketchAlgorithm as A; + match family { + SummaryFamilyType::ExactAggregate(..) => {} + SummaryFamilyType::Sketch(kind, _) + if matches!(kind.algorithm(), A::Kll | A::DDSketch | A::Hll) => {} + _ => { + return Err(Error::Invalid( + "summary family has no native DAG state implementation".into(), + )) + } + } + crate::capability::validate_summary_kernel( + family, + &planner_types::post_asap::SummaryUpdate::column( + planner_types::pre_asap::ColumnRef::SampleValue, + ), + &Default::default(), + ) + .map_err(Error::Invalid) +} +fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Result<(), Error> { + use crate::accumulators::{ + datasketches_kll_accumulator::DatasketchesKLLAccumulator, + dd_sketch_accumulator::DDSketchAccumulator, exact_accumulator::ExactAccumulator, + hll_sketch_accumulator::HllSketchAccumulator, + }; + use planner_types::post_asap::SketchParams; + validate_family(family)?; + let valid = match family { + SummaryFamilyType::ExactAggregate(..) => { + state + .as_any() + .downcast_ref::() + .is_some_and(|s| s.family() == family && !s.is_keyed()) + || (matches!( + family, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Sum, + planner_types::post_asap::ExactParams::Sum + ) + ) && state.as_any().is::()) + } + SummaryFamilyType::Sketch(kind, _) => match kind.params() { + SketchParams::Kll { k } => state + .as_any() + .downcast_ref::() + .is_some_and(|s| u32::from(s.inner.k()) == *k), + SketchParams::DDSketch { alpha } => state + .as_any() + .downcast_ref::() + .is_some_and(|s| s.inner.alpha == *alpha && s.sample_p == 1.0), + SketchParams::Hll { precision } => state + .as_any() + .downcast_ref::() + .is_some_and(|s| s.inner.precision == u32::from(*precision) && s.sample_p == 1.0), + _ => false, + }, + _ => false, + }; + if valid { + Ok(()) + } else { + Err(Error::Invalid( + "state payload differs from declared family, parameters or population layout".into(), + )) + } +} + +pub(crate) fn validate_schema(schema: &Schema) -> Result<(), Error> { + if schema.time_index.is_some_and(|index| { + schema + .fields + .get(index) + .is_none_or(|field| field.dtype != SummaryFamilyType::Plain(DataType::Timestamp)) + }) { + return Err(Error::Invalid( + "time index must name a Timestamp column".into(), + )); + } + for field in &schema.fields { + if !matches!(field.dtype, SummaryFamilyType::Plain(_)) { + validate_family(&field.dtype)?; + if field.nullable { + return Err(Error::Invalid( + "nullable summary states are not supported".into(), + )); + } + } + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/factory.rs b/crates/asap-physical-operators/src/factory.rs new file mode 100644 index 00000000..0086c359 --- /dev/null +++ b/crates/asap-physical-operators/src/factory.rs @@ -0,0 +1,1190 @@ +use crate::accumulators::{ + CountMinSketchAccumulator, CountMinSketchWithHeapAccumulator, CountSketchAccumulator, + CountSketchWithHeapAccumulator, DDSketchAccumulator, DatasketchesKLLAccumulator, + HydraKllSketchAccumulator, IncreaseAccumulator, KeyedCounterState, KeyedMaxState, + KeyedMinState, KeyedSumCountAccumulator, MaxAccumulator, MinAccumulator, SumAccumulator, +}; +use crate::{AggregateCore, KeyByLabelValues, Measurement}; +// Production dispatch consumes Planner SummaryAgg payloads directly. The +// config adapter below is compiled only for isolated historical kernel tests. +use crate::accumulators::hll_sketch_accumulator::HllSketchAccumulator; +use crate::accumulators::univmon_accumulator::UnivMonAccumulator; +use planner_types::post_asap::{ExactKind, SketchAlgorithm, SketchParams, SummaryFamilyType}; + +/// Generate the two boilerplate clone-based `AccumulatorUpdater` methods +/// for updaters whose inner `acc` field implements `Clone + AggregateCore`. +/// Not applicable to `IncreaseAccumulatorUpdater` (its `acc` is `Option<_>` +/// with non-trivial `None` handling). +macro_rules! impl_clone_accumulator_methods { + ($acc_field:ident) => { + fn take_accumulator(&mut self) -> Box { + let result = Box::new(self.$acc_field.clone()); + self.reset(); + result + } + + fn snapshot_accumulator(&self) -> Box { + Box::new(self.$acc_field.clone()) + } + + fn into_accumulator(self: Box) -> Box { + // Consume the updater and MOVE the accumulator out — no clone. + // Avoids the expensive `Clone` (a full msgpack serialize/deserialize + // round-trip for sketch accumulators) when a pane is evicted at + // window close. + let this = *self; + Box::new(this.$acc_field) + } + }; +} + +/// Shared update interface for query-time and maintenance-time accumulation. +/// +/// This provides a uniform interface over all accumulator types so that the +/// worker loop doesn't need to know which concrete type it's dealing with. +pub trait AccumulatorUpdater: Send { + /// Validate an immutable maintenance input before an updater can silently + /// discard a value outside its representable domain. + fn validate_single_input(&self, value: f64) -> Result<(), String> { + if value.is_finite() { + Ok(()) + } else { + Err("accumulator input must be finite".into()) + } + } + + /// Feed a single (value, timestamp_ms) pair — for SingleSubpopulation types. + fn update_single(&mut self, value: f64, timestamp_ms: i64); + + /// Feed a keyed (key, value, timestamp_ms) triple — for MultipleSubpopulation types. + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, timestamp_ms: i64); + + /// Extract the final accumulator as a boxed `AggregateCore`. + fn take_accumulator(&mut self) -> Box; + + /// Non-destructive read of the current accumulator state (clone without reset). + /// Used by pane-based sliding windows to read shared panes. + fn snapshot_accumulator(&self) -> Box; + + /// Consume the updater and return its accumulator BY MOVE, avoiding the + /// `Clone` that `take_accumulator`/`snapshot_accumulator` pay (for sketch + /// accumulators that clone is a full msgpack serialize/deserialize + /// round-trip). Used by `merge_panes_for_window` when a pane is evicted at + /// window close. Default falls back to a clone for updaters that can't + /// cheaply move their inner accumulator out. + fn into_accumulator(self: Box) -> Box { + self.snapshot_accumulator() + } + + /// Reset internal state for reuse (avoids re-allocation). + fn reset(&mut self); + + /// Whether this updater is keyed (MultipleSubpopulation). + fn is_keyed(&self) -> bool; + + /// Estimated memory usage in bytes. + fn memory_usage_bytes(&self) -> usize; +} + +// --------------------------------------------------------------------------- +// SumAccumulatorUpdater +// --------------------------------------------------------------------------- + +pub struct SumAccumulatorUpdater { + acc: SumAccumulator, +} + +impl SumAccumulatorUpdater { + pub fn new() -> Self { + Self { + acc: SumAccumulator::new(), + } + } +} + +impl Default for SumAccumulatorUpdater { + fn default() -> Self { + Self::new() + } +} + +impl AccumulatorUpdater for SumAccumulatorUpdater { + fn update_single(&mut self, value: f64, _timestamp_ms: i64) { + self.acc.update(value); + } + + fn update_keyed(&mut self, _key: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = SumAccumulator::new(); + } + + fn is_keyed(&self) -> bool { + false + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + } +} + +// --------------------------------------------------------------------------- +// MinAccumulatorUpdater / MaxAccumulatorUpdater +// --------------------------------------------------------------------------- + +macro_rules! extremum_updater { + ($updater:ident, $acc:ty) => { + #[derive(Default)] + pub struct $updater { + acc: $acc, + } + + impl $updater { + pub fn new() -> Self { + Self::default() + } + } + + impl AccumulatorUpdater for $updater { + fn update_single(&mut self, value: f64, _timestamp_ms: i64) { + self.acc.update(value); + } + + fn update_keyed(&mut self, _key: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = <$acc>::new(); + } + + fn is_keyed(&self) -> bool { + false + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::<$acc>() + } + } + }; +} + +extremum_updater!(MinAccumulatorUpdater, MinAccumulator); +extremum_updater!(MaxAccumulatorUpdater, MaxAccumulator); + +// --------------------------------------------------------------------------- +// IncreaseAccumulatorUpdater +// --------------------------------------------------------------------------- + +pub struct IncreaseAccumulatorUpdater { + acc: Option, +} + +impl IncreaseAccumulatorUpdater { + pub fn new() -> Self { + Self { acc: None } + } +} + +impl Default for IncreaseAccumulatorUpdater { + fn default() -> Self { + Self::new() + } +} + +impl AccumulatorUpdater for IncreaseAccumulatorUpdater { + fn update_single(&mut self, value: f64, timestamp_ms: i64) { + let measurement = Measurement::new(value); + match &mut self.acc { + Some(acc) => acc.update(measurement, timestamp_ms), + None => { + self.acc = Some(IncreaseAccumulator::new( + measurement.clone(), + timestamp_ms, + measurement, + timestamp_ms, + )); + } + } + } + + fn update_keyed(&mut self, _key: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + + // Hand-written: acc is Option<_> with non-trivial None handling. + fn take_accumulator(&mut self) -> Box { + let acc = self.acc.take().unwrap_or_else(|| { + IncreaseAccumulator::new(Measurement::new(0.0), 0, Measurement::new(0.0), 0) + }); + let result = Box::new(acc); + self.reset(); + result + } + + fn snapshot_accumulator(&self) -> Box { + match &self.acc { + Some(acc) => Box::new(acc.clone()), + None => Box::new(IncreaseAccumulator::new( + Measurement::new(0.0), + 0, + Measurement::new(0.0), + 0, + )), + } + } + + fn reset(&mut self) { + self.acc = None; + } + + fn is_keyed(&self) -> bool { + false + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::>() + } +} + +// --------------------------------------------------------------------------- +// KllAccumulatorUpdater +// --------------------------------------------------------------------------- + +pub struct KllAccumulatorUpdater { + acc: DatasketchesKLLAccumulator, + k: u16, +} + +impl KllAccumulatorUpdater { + pub fn new(k: u16) -> Self { + Self { + acc: DatasketchesKLLAccumulator::new(k), + k, + } + } +} + +impl AccumulatorUpdater for KllAccumulatorUpdater { + fn update_single(&mut self, value: f64, _timestamp_ms: i64) { + self.acc.update(value); + } + + fn update_keyed(&mut self, _key: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = DatasketchesKLLAccumulator::new(self.k); + } + + fn is_keyed(&self) -> bool { + false + } + + fn memory_usage_bytes(&self) -> usize { + // KLL sketch size is hard to estimate precisely; use a rough estimate + std::mem::size_of::() + 4096 + } +} + +// --------------------------------------------------------------------------- +// DDSketchAccumulatorUpdater — pendant to KllAccumulatorUpdater +// --------------------------------------------------------------------------- +// +// Drives the agent-aggregated DDSketch path: the worker either +// (a) merges an inbound `DDSketchAccumulator` from the +// modified-OTLP `Data::Ddsketch` ingest (via the worker's +// `merge_with`), or (b) consumes raw values via `update_single` +// when an OTLP scalar datapoint matches an aggregation typed as +// DDSketch. (b) is the less common path but it lets the same +// aggregation slot serve both pre-aggregated agent sketches and +// raw OTLP gauges. +pub struct DDSketchAccumulatorUpdater { + acc: DDSketchAccumulator, + alpha: f64, +} + +impl DDSketchAccumulatorUpdater { + pub fn new(alpha: f64) -> Self { + Self { + acc: DDSketchAccumulator::new(alpha), + alpha, + } + } +} + +impl AccumulatorUpdater for DDSketchAccumulatorUpdater { + fn validate_single_input(&self, value: f64) -> Result<(), String> { + let (minimum, maximum) = + asap_sketchlib::sketches::ddsketch::ddsketch_indexable_bounds(self.alpha); + if value.is_finite() && value > 0.0 && value >= minimum && value <= maximum { + Ok(()) + } else { + Err("DDS maintenance input is outside its positive representable domain".into()) + } + } + + fn update_single(&mut self, value: f64, _timestamp_ms: i64) { + // sketch-core's DdSketch (the inner of DDSketchAccumulator) + // exposes `update(f64)` for single-value ingestion. The + // worker calls this when a raw OTLP datapoint matches an + // aggregation typed as DDSketch — the sketch-merge path + // uses `merge_with` directly. + self.acc.inner.update(value); + } + + fn update_keyed(&mut self, _key: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = DDSketchAccumulator::new(self.alpha); + } + + fn is_keyed(&self) -> bool { + false + } + + fn memory_usage_bytes(&self) -> usize { + // Bucket store is variable; rough estimate matches KLL. + std::mem::size_of::() + 4096 + } +} + +// --------------------------------------------------------------------------- +// KeyedSumCountAccumulatorUpdater +// --------------------------------------------------------------------------- + +pub struct KeyedSumCountAccumulatorUpdater { + acc: KeyedSumCountAccumulator, +} + +impl KeyedSumCountAccumulatorUpdater { + pub fn new() -> Self { + Self::for_family(ExactKind::Sum) + } + + pub fn for_family(family: ExactKind) -> Self { + Self { + acc: KeyedSumCountAccumulator::for_family(family), + } + } +} + +impl Default for KeyedSumCountAccumulatorUpdater { + fn default() -> Self { + Self::new() + } +} + +impl AccumulatorUpdater for KeyedSumCountAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.update(key.clone(), value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = KeyedSumCountAccumulator::for_family(self.acc.family.clone()); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.acc.sums.len() * (std::mem::size_of::() + 16) + } +} + +// --------------------------------------------------------------------------- +// KeyedMinStateUpdater / KeyedMaxStateUpdater +// --------------------------------------------------------------------------- + +macro_rules! multiple_extremum_updater { + ($updater:ident, $acc:ty) => { + #[derive(Default)] + pub struct $updater { + acc: $acc, + } + + impl $updater { + pub fn new() -> Self { + Self::default() + } + } + + impl AccumulatorUpdater for $updater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.update(key.clone(), value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = <$acc>::new(); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::<$acc>() + + self.acc.values.len() * (std::mem::size_of::() + 8) + } + } + }; +} + +multiple_extremum_updater!(KeyedMinStateUpdater, KeyedMinState); +multiple_extremum_updater!(KeyedMaxStateUpdater, KeyedMaxState); + +// --------------------------------------------------------------------------- +// KeyedCounterStateUpdater +// --------------------------------------------------------------------------- + +pub struct KeyedCounterStateUpdater { + acc: KeyedCounterState, +} + +impl KeyedCounterStateUpdater { + pub fn new() -> Self { + Self { + acc: KeyedCounterState::new(), + } + } +} + +impl Default for KeyedCounterStateUpdater { + fn default() -> Self { + Self::new() + } +} + +impl AccumulatorUpdater for KeyedCounterStateUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + let measurement = Measurement::new(value); + match self.acc.increases.entry(key.clone()) { + std::collections::hash_map::Entry::Occupied(mut e) => { + e.get_mut().update(measurement, timestamp_ms); + } + std::collections::hash_map::Entry::Vacant(e) => { + e.insert(IncreaseAccumulator::new( + measurement.clone(), + timestamp_ms, + measurement, + timestamp_ms, + )); + } + } + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = KeyedCounterState::new(); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.acc.increases.len() + * (std::mem::size_of::() + + std::mem::size_of::()) + } +} + +// --------------------------------------------------------------------------- +// CmsAccumulatorUpdater (CountMinSketch) +// --------------------------------------------------------------------------- + +/// Keyed weighted-frequency updater. +/// +/// A raw Prometheus sample represents the observed metric value, so a bare CMS +/// adds `value` for its key. Counting each received sample as one is a distinct +/// event-count operation and requires an explicit typed plan contract; it must +/// not be inferred from the sketch algorithm alone. +pub struct CmsAccumulatorUpdater { + acc: CountMinSketchAccumulator, + row_num: usize, + col_num: usize, +} + +impl CmsAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + acc: CountMinSketchAccumulator::new(row_num, col_num), + row_num, + col_num, + } + } +} + +impl AccumulatorUpdater for CmsAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.inner.update(&key.to_semicolon_str(), value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = CountMinSketchAccumulator::new(self.row_num, self.col_num); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + } +} + +// --------------------------------------------------------------------------- +// CmsHeapAccumulatorUpdater — value-weighted / count-weighted top-k +// --------------------------------------------------------------------------- + +/// What quantity the top-k heap ranks keys by. +/// +/// These are DIFFERENT query semantics and must be chosen explicitly: +/// +/// * [`TopkWeight::Value`] — accumulate **Σ of the datapoint value** per key. +/// This answers "top-k by total " (e.g. "top-k hosts by +/// total CPU"). The heap value is the summed metric value, so the read-side +/// reducer's "sort heap descending by value" yields the correct ranking. +/// +/// * [`TopkWeight::Count`] — accumulate **+1 per event** per key (occurrence +/// frequency), the textbook heavy-hitter / frequency-top-k semantics +/// ("which keys appear most often"). +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum TopkWeight { + /// Σ datapoint value per key (value-weighted top-k). + Value, + /// +1 per event per key (count-weighted / frequency top-k). + Count, +} + +/// Keyed top-k updater backed by a real `CountMinSketchWithHeap` (a CMS +/// matrix PLUS a size-`heap_size` top-k heap). Unlike the heap-LESS +/// `CmsAccumulatorUpdater`, this enumerates top-k keys at read time +/// (`get_topk_keys` / `topk_heap_items`), which is what `topk(...)` queries +/// need. +/// +/// The key is the configured group-by (`aggregated_labels`) value vector — +/// e.g. `host` — formed by `extract_aggregated_key_from_series` in the worker, +/// NOT the hardcoded metric label `item`. The accumulated quantity is selected +/// by [`TopkWeight`]: +/// * `Value` → `inner.update(key, value)` adds the datapoint value (Σ value). +/// * `Count` → `inner.update(key, 1.0)` adds one per event (Σ count). +/// +/// Both `CountMinSketchWithHeap` and `CountSketchWithHeap` raw-input policies +/// route here; the heap is the shared distinguishing payload. +pub struct CmsHeapAccumulatorUpdater { + acc: CountMinSketchWithHeapAccumulator, + row_num: usize, + col_num: usize, + heap_size: usize, + weight: TopkWeight, + weight_scale: f64, +} + +impl CmsHeapAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize, heap_size: usize, weight: TopkWeight) -> Self { + Self::with_weight_scale(row_num, col_num, heap_size, weight, 1.0) + } + + pub fn with_weight_scale( + row_num: usize, + col_num: usize, + heap_size: usize, + weight: TopkWeight, + weight_scale: f64, + ) -> Self { + Self { + acc: CountMinSketchWithHeapAccumulator::new(row_num, col_num, heap_size), + row_num, + col_num, + heap_size, + weight, + weight_scale, + } + } +} + +impl AccumulatorUpdater for CmsHeapAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + // Heap key = the group-by label-value vector (e.g. `host`), joined the + // same way the read-side `get_topk_keys` splits it back apart (`;`). + let weighted = match self.weight { + // Σ value: feed the datapoint value. sketchlib's CMS-heap + // `update(key, w)` adds `w.round()` occurrences of `key`, so the + // heap value accumulates the (rounded) summed metric value. + TopkWeight::Value => value * self.weight_scale, + // Σ count: one occurrence per event, regardless of value. + TopkWeight::Count => 1.0, + }; + self.acc.inner.update(&key.to_semicolon_str(), weighted); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = + CountMinSketchWithHeapAccumulator::new(self.row_num, self.col_num, self.heap_size); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + + self.heap_size * (std::mem::size_of::() + 32) + } +} + +// --------------------------------------------------------------------------- +// CountSketchAccumulatorUpdater (real median-of-signed-rows CountSketch) +// --------------------------------------------------------------------------- + +/// Keyed point-frequency updater backed by a real `asap_sketchlib::CountSketch` +/// (signed rows, median-of-rows estimator) — distinct math from +/// `CmsAccumulatorUpdater`'s CMS (min-of-rows). Closes, on the raw-metric +/// ingest path, the conflation bug where `SketchAlgorithm::CountSketch` silently +/// shared `CmsAccumulatorUpdater` with bare CMS. +/// +/// As with bare CMS, each raw Prometheus sample contributes its `value`. +/// Unit event counting must be selected explicitly by a future typed plan +/// contract rather than being implied by `SketchAlgorithm::CountSketch`. +pub struct CountSketchAccumulatorUpdater { + acc: CountSketchAccumulator, + row_num: usize, + col_num: usize, +} + +impl CountSketchAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + acc: CountSketchAccumulator::new(row_num, col_num), + row_num, + col_num, + } + } +} + +impl AccumulatorUpdater for CountSketchAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.inner.update(&key.to_semicolon_str(), value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = CountSketchAccumulator::new(self.row_num, self.col_num); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + } +} + +// --------------------------------------------------------------------------- +// CountSketchWithHeapAccumulatorUpdater (real CountSketch + top-k heap) +// --------------------------------------------------------------------------- + +/// Keyed top-k updater backed by a real `CountSketchWithHeap` (signed-row +/// CountSketch matrix PLUS a size-`heap_size` top-k heap). Distinct math from +/// `CmsHeapAccumulatorUpdater`'s CMS-with-heap (min-of-rows); shares the same +/// [`TopkWeight`] semantics and heap payload shape. +pub struct CountSketchWithHeapAccumulatorUpdater { + acc: CountSketchWithHeapAccumulator, + row_num: usize, + col_num: usize, + heap_size: usize, + weight: TopkWeight, + weight_scale: f64, +} + +impl CountSketchWithHeapAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize, heap_size: usize, weight: TopkWeight) -> Self { + Self::with_weight_scale(row_num, col_num, heap_size, weight, 1.0) + } + + pub fn with_weight_scale( + row_num: usize, + col_num: usize, + heap_size: usize, + weight: TopkWeight, + weight_scale: f64, + ) -> Self { + Self { + acc: CountSketchWithHeapAccumulator::new(row_num, col_num, heap_size), + row_num, + col_num, + heap_size, + weight, + weight_scale, + } + } +} + +impl AccumulatorUpdater for CountSketchWithHeapAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + let weighted = match self.weight { + TopkWeight::Value => value * self.weight_scale, + TopkWeight::Count => 1.0, + }; + self.acc.inner.update(&key.to_semicolon_str(), weighted); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = CountSketchWithHeapAccumulator::new(self.row_num, self.col_num, self.heap_size); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + + self.heap_size * (std::mem::size_of::() + 32) + } +} + +// --------------------------------------------------------------------------- +// HydraKllAccumulatorUpdater +// --------------------------------------------------------------------------- + +pub struct HydraKllAccumulatorUpdater { + acc: HydraKllSketchAccumulator, + row_num: usize, + col_num: usize, + k: u16, +} + +impl HydraKllAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize, k: u16) -> Self { + Self { + acc: HydraKllSketchAccumulator::new(row_num, col_num, k), + row_num, + col_num, + k, + } + } +} + +impl AccumulatorUpdater for HydraKllAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.update(key, value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = HydraKllSketchAccumulator::new(self.row_num, self.col_num, self.k); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + // Rough estimate: each cell is a KLL sketch + std::mem::size_of::() + self.row_num * self.col_num * 4096 + } +} + +// --------------------------------------------------------------------------- +// Config helpers +// --------------------------------------------------------------------------- + +fn cms_dims(params: &SketchParams) -> (usize, usize) { + match params { + SketchParams::Cms { width, depth } | SketchParams::CountSketch { width, depth } => { + (*depth as usize, *width as usize) + } + other => unreachable!( + "accumulator_spec() paired SketchAlgorithm::Cms/CountSketch with unexpected params: {other:?}" + ), + } +} + +/// Read `(rows = depth, columns = width, heap_size)` out of `SketchParams::CmsWithHeap` +/// or `::CountSketchWithHeap`. +fn cms_heap_dims(params: &SketchParams) -> (usize, usize, usize) { + match params { + SketchParams::CmsWithHeap { + width, + depth, + heap_size, + } + | SketchParams::CountSketchWithHeap { + width, + depth, + heap_size, + } => (*depth as usize, *width as usize, *heap_size as usize), + other => unreachable!( + "accumulator_spec() paired a WithHeap SketchAlgorithm with unexpected params: {other:?}" + ), + } +} + +/// Construct the kernel declared by a Planner SummaryAgg. No backend config +/// tags participate in this dispatch and unsupported payloads are errors. +pub fn create_planner_accumulator( + family: &SummaryFamilyType, + input: &planner_types::post_asap::SummaryUpdate, + grouping: &planner_types::post_asap::GroupingStrategy, +) -> Result, String> { + crate::capability::validate_summary_kernel(family, input, grouping)?; + use planner_types::post_asap::GroupingStrategy; + if grouping != &GroupingStrategy::PerSubpopulationInstance { + return Err("shared summary grouping requires a supported Planner Hydra kernel".into()); + } + if matches!(family, SummaryFamilyType::ExactAggregate(..)) { + return Ok(Box::new(PlannerExactUpdater { + acc: crate::accumulators::exact_accumulator::ExactAccumulator::new( + family.clone(), + input.item.is_some(), + )?, + })); + } + let SummaryFamilyType::Sketch(kind, family_grouping) = family else { + return Err(format!("unsupported Planner summary family {family:?}")); + }; + if family_grouping != grouping { + return Err("Planner family and operator grouping disagree".into()); + } + // Heap counters use fixed-point storage for fractional counter deltas. + // This encodes the selected update; it does not choose another family. + let weight_scale = if matches!( + input.weight, + planner_types::post_asap::SummaryInputExpr::ResetAwareCounterDelta { .. } + ) { + 1_000_000.0 + } else { + 1.0 + }; + let updater: Box = match (kind.algorithm(), kind.params()) { + (SketchAlgorithm::Kll, SketchParams::Kll { k }) => Box::new(KllAccumulatorUpdater::new( + u16::try_from(*k).map_err(|_| "KLL k exceeds runtime bound")?, + )), + (SketchAlgorithm::DDSketch, SketchParams::DDSketch { alpha }) => { + Box::new(DDSketchAccumulatorUpdater::new(*alpha)) + } + (SketchAlgorithm::Cms, params @ SketchParams::Cms { .. }) => { + let (r, c) = cms_dims(params); + Box::new(CmsAccumulatorUpdater::new(r, c)) + } + (SketchAlgorithm::CountSketch, params @ SketchParams::CountSketch { .. }) => { + let (r, c) = cms_dims(params); + Box::new(CountSketchAccumulatorUpdater::new(r, c)) + } + (SketchAlgorithm::CmsWithHeap, params @ SketchParams::CmsWithHeap { .. }) => { + let (r, c, h) = cms_heap_dims(params); + Box::new(CmsHeapAccumulatorUpdater::with_weight_scale( + r, + c, + h, + TopkWeight::Value, + weight_scale, + )) + } + ( + SketchAlgorithm::CountSketchWithHeap, + params @ SketchParams::CountSketchWithHeap { .. }, + ) => { + let (r, c, h) = cms_heap_dims(params); + Box::new(CountSketchWithHeapAccumulatorUpdater::with_weight_scale( + r, + c, + h, + TopkWeight::Value, + weight_scale, + )) + } + (SketchAlgorithm::Hll, SketchParams::Hll { precision }) => Box::new(HllUpdater { + acc: HllSketchAccumulator::new( + asap_sketchlib::HllVariant::Regular, + u32::from(*precision), + ), + }), + ( + SketchAlgorithm::UnivMon, + SketchParams::UnivMon { + heap_size, + sketch_rows, + sketch_cols, + layers, + }, + ) => Box::new(UnivMonUpdater { + acc: UnivMonAccumulator::new( + *heap_size as usize, + *sketch_rows as usize, + *sketch_cols as usize, + *layers as usize, + ) + .map_err(|e| e.to_string())?, + }), + _ => { + return Err(format!( + "unsupported Planner algorithm/parameters: {kind:?}" + )) + } + }; + if updater.is_keyed() != input.item.is_some() + && !crate::capability::is_unit_sample_frequency(input) + { + return Err("Planner item expression does not match the selected kernel layout".into()); + } + Ok(updater) +} + +struct PlannerExactUpdater { + acc: crate::accumulators::exact_accumulator::ExactAccumulator, +} +impl AccumulatorUpdater for PlannerExactUpdater { + fn update_single(&mut self, value: f64, timestamp: i64) { + self.acc.update(None, value, timestamp); + } + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, timestamp: i64) { + self.acc.update(Some(key), value, timestamp); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc = crate::accumulators::exact_accumulator::ExactAccumulator::new( + self.acc.family().clone(), + self.acc.is_keyed(), + ) + .expect("installed exact family"); + } + fn is_keyed(&self) -> bool { + self.acc.is_keyed() + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } +} + +#[cfg(test)] +mod planner_parameter_regression { + use super::*; + use planner_types::post_asap::{SketchKind, SummaryInputExpr, SummaryUpdate}; + + // Planner width is the bucket count; depth is the independent hash-row count. + #[test] + fn planner_sketch_dimensions_are_not_transposed() { + for (algorithm, params) in [ + ( + SketchAlgorithm::Cms, + SketchParams::Cms { + width: 128, + depth: 3, + }, + ), + ( + SketchAlgorithm::CountSketch, + SketchParams::CountSketch { + width: 128, + depth: 3, + }, + ), + ( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 128, + depth: 3, + heap_size: 8, + }, + ), + ( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 128, + depth: 3, + heap_size: 8, + }, + ), + ] { + let family = SummaryFamilyType::Sketch( + SketchKind::new(algorithm.clone(), params), + Default::default(), + ); + let update = SummaryUpdate { + item: Some(SummaryInputExpr::Column( + planner_types::pre_asap::ColumnRef::Named("host".into()), + )), + weight: SummaryInputExpr::Constant(1.0), + weight_domain: Default::default(), + }; + let state = create_planner_accumulator(&family, &update, &Default::default()) + .unwrap() + .snapshot_accumulator(); + let dims = match algorithm { + SketchAlgorithm::Cms => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows(), s.inner.cols()) + } + SketchAlgorithm::CountSketch => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows, s.inner.cols) + } + SketchAlgorithm::CmsWithHeap => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows(), s.inner.cols()) + } + SketchAlgorithm::CountSketchWithHeap => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows(), s.inner.cols()) + } + _ => unreachable!(), + }; + assert_eq!(dims, (3, 128), "{algorithm:?}"); + } + } +} + +struct UnivMonUpdater { + acc: UnivMonAccumulator, +} + +struct HllUpdater { + acc: HllSketchAccumulator, +} + +impl AccumulatorUpdater for HllUpdater { + fn is_keyed(&self) -> bool { + false + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } + fn update_single(&mut self, value: f64, _: i64) { + if !value.is_nan() { + let bits = if value == 0.0 { 0 } else { value.to_bits() }; + self.acc.inner.update(&bits.to_le_bytes()); + } + } + fn update_keyed(&mut self, _: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc.reset_to_empty(); + } +} + +impl AccumulatorUpdater for UnivMonUpdater { + fn is_keyed(&self) -> bool { + false + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } + fn update_single(&mut self, value: f64, _: i64) { + self.acc + .insert_sample(value) + .expect("UnivMon sample counter overflow"); + } + fn update_keyed(&mut self, _: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc.reset_to_empty(); + } +} diff --git a/crates/asap-physical-operators/src/key_by_label_values.rs b/crates/asap-physical-operators/src/key_by_label_values.rs new file mode 100644 index 00000000..34bc8489 --- /dev/null +++ b/crates/asap-physical-operators/src/key_by_label_values.rs @@ -0,0 +1,164 @@ +use serde::{Deserialize, Serialize}; +// use std::collections::HashMap; +use std::hash::{Hash, Hasher}; + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct KeyByLabelValues { + // pub labels: HashMap, + pub labels: Vec, +} + +impl KeyByLabelValues { + pub fn new() -> Self { + Self { labels: Vec::new() } + } + + pub fn new_with_labels(labels: Vec) -> Self { + Self { labels } + } + + pub fn insert(&mut self, value: String) { + self.labels.push(value); + } + + pub fn get(&self, index: usize) -> Option<&String> { + self.labels.get(index) + } + + pub fn serialize_to_json(&self) -> serde_json::Value { + serde_json::to_value(&self.labels).unwrap_or(serde_json::Value::Null) + } + + pub fn deserialize_from_json(data: &serde_json::Value) -> Result { + let labels: Vec = serde_json::from_value(data.clone())?; + Ok(Self { labels }) + } + + pub fn serialize_to_bytes(&self) -> Vec { + bincode::serialize(&self.labels).unwrap_or_default() + } + + pub fn deserialize_from_bytes(buffer: &[u8]) -> Result> { + let labels: Vec = bincode::deserialize(buffer)?; + Ok(Self { labels }) + } + + /// Encode labels as a semicolon-joined string — the canonical key format used + /// for all sketch hashing (CountMinSketch, HydraKLL, SetAggregator, DeltaSet). + pub fn to_semicolon_str(&self) -> String { + self.labels.join(";") + } + + #[cfg(test)] + /// Decode a semicolon-joined string back into a KeyByLabelValues. + pub fn from_semicolon_str(s: &str) -> Self { + Self { + labels: s.split(';').map(|s| s.to_string()).collect(), + } + } + + pub fn is_empty(&self) -> bool { + self.labels.is_empty() + } + + pub fn len(&self) -> usize { + self.labels.len() + } +} + +impl Hash for KeyByLabelValues { + fn hash(&self, state: &mut H) { + // Create a sorted vector of key-value pairs for consistent hashing + let mut sorted_pairs: Vec<_> = self.labels.iter().collect(); + sorted_pairs.sort(); + + for value in sorted_pairs { + value.hash(state); + } + } +} + +impl Default for KeyByLabelValues { + fn default() -> Self { + Self::new() + } +} + +impl std::fmt::Display for KeyByLabelValues { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{{")?; + let mut first = true; + for value in &self.labels { + if !first { + write!(f, ", ")?; + } + write!(f, "{value}")?; + first = false; + } + write!(f, "}}") + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_key_by_label_values() { + let mut key = KeyByLabelValues::new(); + key.insert("localhost:8080".to_string()); + key.insert("prometheus".to_string()); + + assert_eq!(key.len(), 2); + assert_eq!(key.get(0), Some(&"localhost:8080".to_string())); + assert_eq!(key.get(1), Some(&"prometheus".to_string())); + } + + #[test] + fn test_serialization() { + let mut key = KeyByLabelValues::new(); + key.insert("test".to_string()); + + let json = key.serialize_to_json(); + let deserialized = KeyByLabelValues::deserialize_from_json(&json).unwrap(); + assert_eq!(key, deserialized); + } + + #[test] + fn test_byte_serialization() { + let mut key = KeyByLabelValues::new(); + key.insert("test".to_string()); + + let bytes = key.serialize_to_bytes(); + let deserialized = KeyByLabelValues::deserialize_from_bytes(&bytes).unwrap(); + assert_eq!(key, deserialized); + } + + #[test] + fn test_semicolon_roundtrip() { + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string(), "prod".to_string()]); + assert_eq!(key.to_semicolon_str(), "web;prod"); + let roundtripped = KeyByLabelValues::from_semicolon_str("web;prod"); + assert_eq!(roundtripped, key); + } + + #[test] + fn test_hash_consistency() { + let mut key1 = KeyByLabelValues::new(); + key1.insert("a".to_string()); + key1.insert("b".to_string()); + + let mut key2 = KeyByLabelValues::new(); + key2.insert("b".to_string()); + key2.insert("a".to_string()); + + // Should hash to the same value regardless of insertion order + let mut hasher1 = std::collections::hash_map::DefaultHasher::new(); + let mut hasher2 = std::collections::hash_map::DefaultHasher::new(); + + key1.hash(&mut hasher1); + key2.hash(&mut hasher2); + + assert_eq!(hasher1.finish(), hasher2.finish()); + } +} diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs new file mode 100644 index 00000000..22456e3d --- /dev/null +++ b/crates/asap-physical-operators/src/lib.rs @@ -0,0 +1,23 @@ +#![doc = include_str!("../README.md")] + +pub mod accumulators; +pub mod key_by_label_values; +pub mod measurement; +pub mod traits; + +mod aggregation_type; +mod statistic; +pub use aggregation_type::AggregationType; +pub use key_by_label_values::KeyByLabelValues; +pub use measurement::Measurement; +pub use statistic::Statistic; +pub use traits::*; + +pub mod arithmetic; +pub mod capability; +pub mod factory; + +/// The exact Planner contract used by these kernels. +pub use planner_types as planner; + +pub mod dag; diff --git a/crates/asap-physical-operators/src/measurement.rs b/crates/asap-physical-operators/src/measurement.rs new file mode 100644 index 00000000..0fe1abc0 --- /dev/null +++ b/crates/asap-physical-operators/src/measurement.rs @@ -0,0 +1,94 @@ +use serde::{Deserialize, Serialize}; +use std::ops::Add; + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct Measurement { + pub value: f64, +} + +impl Measurement { + pub fn new(value: f64) -> Self { + Self { value } + } + + pub fn serialize_to_bytes(&self) -> Vec { + self.value.to_le_bytes().to_vec() + } + + pub fn serialize_to_json(&self) -> serde_json::Value { + serde_json::json!({ + "value": self.value + }) + } + + pub fn deserialize_from_json(data: &serde_json::Value) -> Result { + let value = data["value"].as_f64().ok_or_else(|| { + serde_json::Error::io(std::io::Error::new( + std::io::ErrorKind::InvalidData, + "Missing or invalid 'value' field", + )) + })?; + Ok(Self::new(value)) + } + + pub fn deserialize_from_bytes(buffer: &[u8]) -> Result> { + if buffer.len() < 8 { + return Err("Buffer too short for f64".into()); + } + let value = f64::from_le_bytes([ + buffer[0], buffer[1], buffer[2], buffer[3], buffer[4], buffer[5], buffer[6], buffer[7], + ]); + Ok(Self::new(value)) + } +} + +impl Add for Measurement { + type Output = Measurement; + + fn add(self, other: Measurement) -> Measurement { + Measurement::new(self.value + other.value) + } +} + +impl Add for &Measurement { + type Output = Measurement; + + fn add(self, other: &Measurement) -> Measurement { + Measurement::new(self.value + other.value) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_measurement_creation() { + let measurement = Measurement::new(42.5); + assert_eq!(measurement.value, 42.5); + } + + #[test] + fn test_measurement_addition() { + let m1 = Measurement::new(10.0); + let m2 = Measurement::new(20.0); + let result = m1 + m2; + assert_eq!(result.value, 30.0); + } + + #[test] + fn test_serialization() { + let measurement = Measurement::new(42.5); + let json = measurement.serialize_to_json(); + let deserialized = Measurement::deserialize_from_json(&json).unwrap(); + assert_eq!(measurement, deserialized); + } + + #[test] + fn test_byte_serialization() { + let measurement = Measurement::new(42.5); + let bytes = measurement.serialize_to_bytes(); + let deserialized = Measurement::deserialize_from_bytes(&bytes).unwrap(); + assert_eq!(measurement, deserialized); + } +} diff --git a/crates/asap-physical-operators/src/statistic.rs b/crates/asap-physical-operators/src/statistic.rs new file mode 100644 index 00000000..7053308d --- /dev/null +++ b/crates/asap-physical-operators/src/statistic.rs @@ -0,0 +1,67 @@ +use std::{fmt, str::FromStr}; +use tracing::debug; +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] +pub enum Statistic { + Count, + Sum, + Cardinality, + FrequencyL2, + FrequencyEntropy, + Increase, + Rate, + Min, + Max, + Quantile, + Topk, +} + +impl fmt::Display for Statistic { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + debug!("Formatting Statistic: {:?}", self); + match self { + Statistic::Count => write!(f, "count"), + Statistic::Sum => write!(f, "sum"), + Statistic::Cardinality => write!(f, "cardinality"), + Statistic::FrequencyL2 => write!(f, "frequency_l2"), + Statistic::FrequencyEntropy => write!(f, "frequency_entropy"), + Statistic::Increase => write!(f, "increase"), + Statistic::Rate => write!(f, "rate"), + Statistic::Min => write!(f, "min"), + Statistic::Max => write!(f, "max"), + Statistic::Quantile => write!(f, "quantile"), + Statistic::Topk => write!(f, "topk"), + } + } +} + +#[allow(clippy::should_implement_trait)] +impl Statistic { + pub fn from_str(s: &str) -> Option { + debug!("Parsing Statistic from string: {}", s); + match s.to_lowercase().as_str() { + "count" => Some(Statistic::Count), + "sum" => Some(Statistic::Sum), + "cardinality" => Some(Statistic::Cardinality), + "frequency_l2" => Some(Statistic::FrequencyL2), + "frequency_entropy" => Some(Statistic::FrequencyEntropy), + "increase" => Some(Statistic::Increase), + "rate" => Some(Statistic::Rate), + "min" => Some(Statistic::Min), + "max" => Some(Statistic::Max), + "quantile" => Some(Statistic::Quantile), + "topk" => Some(Statistic::Topk), + _ => None, + } + } +} + +impl FromStr for Statistic { + type Err = (); + + /// Parse a statistic from a string (case-insensitive). + /// Use `s.parse::()` or `Statistic::from_str(s)`. + fn from_str(s: &str) -> Result { + debug!("FromStr trait parsing Statistic: {}", s); + Statistic::from_str(s).ok_or(()) + } +} diff --git a/crates/asap-physical-operators/src/traits.rs b/crates/asap-physical-operators/src/traits.rs new file mode 100644 index 00000000..ae05ca06 --- /dev/null +++ b/crates/asap-physical-operators/src/traits.rs @@ -0,0 +1,357 @@ +use crate::KeyByLabelValues; +use std::collections::HashMap; + +use crate::AggregationType; +use crate::Statistic; + +use serde_json::Value; + +/// Trait for objects that can be serialized to different formats +pub trait SerializableToSink { + fn serialize_to_json(&self) -> Value; + fn serialize_to_bytes(&self) -> Vec; +} + +/// Core trait for all aggregates containing shared functionality +/// This trait provides common operations like serialization, cloning, and type identification +pub trait AggregateCore: SerializableToSink + Send + Sync { + /// Clone this accumulator into a boxed trait object + fn clone_boxed_core(&self) -> Box; + + /// Get the type name of this accumulator + fn type_name(&self) -> &'static str; + + /// Downcast to Any for type checking + fn as_any(&self) -> &dyn std::any::Any; + + /// Mutable downcast to Any. Used by ingest paths that need to + /// mutate a boxed accumulator in place — e.g. the PROTO_DELTA + /// delta-merge applier in `drivers::ingest::otel::apply_modified_otlp_delta_bytes`. + fn as_any_mut(&mut self) -> &mut dyn std::any::Any; + + /// Merge this accumulator with another accumulator of the same type + /// Returns a new merged accumulator, leaving the original unchanged + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box>; + + /// Get the accumulator type identifier for merge compatibility checking + fn get_accumulator_type(&self) -> AggregationType; + + /// Get all keys stored in this accumulator + fn get_keys(&self) -> Option>; + + /// Dispatch a statistic query without downcasting. + /// + /// Replaces the 12-arm `match get_accumulator_type()` in the engine. + /// Single-subpopulation types ignore `key`; multiple-subpopulation types + /// require it and return `Err` when it is `None`. + /// Special cases (DeltaSetAggregator, SetAggregator) fall back to a + /// cardinality value when `key` is `None`. + fn query_statistic( + &self, + statistic: Statistic, + key: &Option, + query_kwargs: &HashMap, + ) -> Result>; + + /// Approximate in-memory byte footprint of this accumulator. + /// + /// Used by the `SketchStore` persistence layer to drive its + /// memory-pressure trigger. Not required to be exact — the flusher + /// only needs rough proportionality. The default is a conservative + /// 4 KiB constant; concrete types should override it with a + /// type-aware estimate (e.g. KLL: `k * 8` plus overhead). + /// + /// Implementors must not call `serialize_to_bytes` here — this is + /// on the insert hot path. + fn approx_memory_bytes(&self) -> usize { + 4096 + } + + /// Typed auxiliary statistics — `count`, `sum`, `min`, `max` — + /// exposed as first-class scalars alongside the sketch payload. + /// + /// The overwhelming majority of production queries + /// (`count_over_time`, `sum_over_time`, `min_over_time`, + /// `max_over_time`, and the additive aggregations built on + /// them) only need these scalars. Returning them directly here + /// lets callers avoid deserialising the full sketch bytes. + /// + /// Returning fields as `None` means the accumulator doesn't + /// track that statistic exactly (e.g. a pure HLL doesn't carry + /// sum/min/max). Callers then fall back to the sketch's + /// `query_statistic` method. + /// + /// This is the phase-1 piece of the sketch DB design + /// (docs/design_docs/summary-storage.md). + fn aux_stats(&self) -> AuxStats { + AuxStats::empty() + } + + /// Reset the sketch state to empty **in place**, preserving its + /// shape / configuration (dimensions, relative accuracy, register + /// width, …) so a subsequent delta-apply lands on a clean, + /// same-shape base. + /// + /// Used by the OTLP ingest path's per-window base rotation: when a + /// delta frame opens a new tumbling window for a series, the cached + /// base is reset here before the new window's delta is applied, so + /// the reconstructed state reflects that window only rather than an + /// all-time accumulation across windows (see + /// `docs/delta-baseline-contract.md` §3). + /// + /// The default is a no-op: only the delta-capable, additive families + /// (DDSketch, CMS, CountSketch, HLL) ever reach the rotation path and + /// override this. KLL never deltas, and the non-sketch accumulators + /// are never cached as a delta base. + fn reset_to_empty(&mut self) {} +} + +/// Four typed auxiliary scalars tracked alongside every sketch entry: +/// `count`, `sum`, `min`, `max`. Exposed so the query engine can +/// serve Count / Sum / Min / Max statistics without touching sketch +/// bytes. +/// +/// Each field is `Option<…>` because not every accumulator tracks +/// every stat (e.g. HLL has cardinality but no meaningful +/// sum / min / max; DeltaSetAggregator tracks set transitions, not +/// numeric aggregates). +#[derive(Debug, Default, Clone, Copy, PartialEq)] +pub struct AuxStats { + pub count: Option, + pub sum: Option, + pub min: Option, + pub max: Option, +} + +impl AuxStats { + pub const fn empty() -> Self { + Self { + count: None, + sum: None, + min: None, + max: None, + } + } + + /// Attempt to fulfil a `Statistic` purely from the typed aux + /// columns, without needing to deserialise the sketch. Returns + /// `None` if the requested statistic isn't covered by aux + /// (e.g. Quantile, Cardinality, TopK) or if the corresponding + /// aux field is `None`. + pub fn try_answer(&self, statistic: Statistic) -> Option { + match statistic { + Statistic::Count => self.count.map(|c| c as f64), + Statistic::Sum => self.sum, + Statistic::Min => self.min, + Statistic::Max => self.max, + // Increase / Rate need two samples; aux columns carry + // window totals, so one entry's aux is insufficient. + // Cardinality / Quantile / Topk are sketch-native and + // must go through query_statistic. + _ => None, + } + } + + /// Merge two aux stats the way the corresponding sketch merge + /// would. Count / sum add, min / max take the extremum. When + /// either side is `None` the result is the other side (so a + /// window that only has partial aux still contributes). + pub fn merge(self, other: Self) -> Self { + fn add_opt_u(a: Option, b: Option) -> Option { + match (a, b) { + (Some(x), Some(y)) => Some(x.saturating_add(y)), + (x, None) => x, + (None, y) => y, + } + } + fn add_opt_f(a: Option, b: Option) -> Option { + match (a, b) { + (Some(x), Some(y)) => Some(x + y), + (x, None) => x, + (None, y) => y, + } + } + fn min_opt(a: Option, b: Option) -> Option { + match (a, b) { + (Some(x), Some(y)) => Some(x.min(y)), + (x, None) => x, + (None, y) => y, + } + } + fn max_opt(a: Option, b: Option) -> Option { + match (a, b) { + (Some(x), Some(y)) => Some(x.max(y)), + (x, None) => x, + (None, y) => y, + } + } + Self { + count: add_opt_u(self.count, other.count), + sum: add_opt_f(self.sum, other.sum), + min: min_opt(self.min, other.min), + max: max_opt(self.max, other.max), + } + } +} + +/// Trait for accumulators that support a single subpopulation +/// These accumulators store a single aggregate value (e.g., Sum, Increase) +pub trait SingleSubpopulationAggregate: AggregateCore { + /// Query the accumulator for a specific statistic + fn query( + &self, + statistic: Statistic, + query_kwargs: Option<&HashMap>, + ) -> Result>; + + /// Clone this accumulator into a boxed trait object + fn clone_boxed(&self) -> Box; +} + +/// Trait for accumulators that support multiple subpopulations identified by keys +/// These accumulators store separate values for different label combinations +pub trait MultipleSubpopulationAggregate: AggregateCore { + /// Query the accumulator for a specific statistic and key + fn query( + &self, + statistic: Statistic, + key: &KeyByLabelValues, + query_kwargs: Option<&HashMap>, + ) -> Result>; + + /// Clone this accumulator into a boxed trait object + fn clone_boxed(&self) -> Box; +} + +/// Trait for merging multiple accumulators of the same type +pub trait MergeableAccumulator { + fn merge_accumulators( + accumulators: Vec, + ) -> Result> + where + T: Sized; +} + +// Implement Clone for the new trait objects +impl Clone for Box { + fn clone(&self) -> Self { + self.clone_boxed_core() + } +} + +impl Clone for Box { + fn clone(&self) -> Self { + self.clone_boxed() + } +} + +impl Clone for Box { + fn clone(&self) -> Self { + self.clone_boxed() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn aux_stats_empty_answers_nothing() { + let e = AuxStats::empty(); + assert_eq!(e.try_answer(Statistic::Count), None); + assert_eq!(e.try_answer(Statistic::Sum), None); + assert_eq!(e.try_answer(Statistic::Min), None); + assert_eq!(e.try_answer(Statistic::Max), None); + } + + #[test] + fn aux_stats_try_answer_covers_typed_stats() { + let a = AuxStats { + count: Some(7), + sum: Some(42.0), + min: Some(1.5), + max: Some(9.25), + }; + assert_eq!(a.try_answer(Statistic::Count), Some(7.0)); + assert_eq!(a.try_answer(Statistic::Sum), Some(42.0)); + assert_eq!(a.try_answer(Statistic::Min), Some(1.5)); + assert_eq!(a.try_answer(Statistic::Max), Some(9.25)); + } + + #[test] + fn aux_stats_try_answer_skips_sketch_native_stats() { + let a = AuxStats { + count: Some(100), + sum: Some(500.0), + min: Some(1.0), + max: Some(10.0), + }; + assert_eq!(a.try_answer(Statistic::Quantile), None); + assert_eq!(a.try_answer(Statistic::Cardinality), None); + assert_eq!(a.try_answer(Statistic::Topk), None); + assert_eq!(a.try_answer(Statistic::Increase), None); + assert_eq!(a.try_answer(Statistic::Rate), None); + } + + #[test] + fn aux_stats_merge_adds_count_and_sum_takes_extrema() { + let a = AuxStats { + count: Some(10), + sum: Some(50.0), + min: Some(1.0), + max: Some(9.0), + }; + let b = AuxStats { + count: Some(5), + sum: Some(20.0), + min: Some(0.5), + max: Some(12.0), + }; + let merged = a.merge(b); + assert_eq!(merged.count, Some(15)); + assert_eq!(merged.sum, Some(70.0)); + assert_eq!(merged.min, Some(0.5)); + assert_eq!(merged.max, Some(12.0)); + } + + #[test] + fn aux_stats_merge_handles_partial_sides() { + // HLL-like (count only) merged with Sum-only side. + let hll_like = AuxStats { + count: Some(100), + ..AuxStats::empty() + }; + let sum_like = AuxStats { + sum: Some(500.0), + ..AuxStats::empty() + }; + let merged = hll_like.merge(sum_like); + assert_eq!(merged.count, Some(100)); + assert_eq!(merged.sum, Some(500.0)); + assert_eq!(merged.min, None); + assert_eq!(merged.max, None); + } + + #[test] + fn aux_stats_merge_is_empty_plus_empty() { + let merged = AuxStats::empty().merge(AuxStats::empty()); + assert_eq!(merged, AuxStats::empty()); + } + + #[test] + fn aux_stats_count_saturates_on_overflow() { + let a = AuxStats { + count: Some(u64::MAX - 1), + ..AuxStats::empty() + }; + let b = AuxStats { + count: Some(100), + ..AuxStats::empty() + }; + let merged = a.merge(b); + assert_eq!(merged.count, Some(u64::MAX)); + } +} diff --git a/crates/asap-physical-operators/tests/deployment.rs b/crates/asap-physical-operators/tests/deployment.rs new file mode 100644 index 00000000..ffdd3aea --- /dev/null +++ b/crates/asap-physical-operators/tests/deployment.rs @@ -0,0 +1,96 @@ +//! Exercise the public library without a backend server, store, or scheduler. +use asap_physical_operators::planner::{ + post_asap::{ + GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryFamilyType, + SummaryUpdate, + }, + pre_asap::ColumnRef, +}; +use asap_physical_operators::{factory::create_planner_accumulator, AggregateCore, Statistic}; +use std::collections::HashMap; + +fn family(k: u32) -> SummaryFamilyType { + SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), + GroupingStrategy::PerSubpopulationInstance, + ) +} +fn build(values: &[f64]) -> Box { + let mut operator = create_planner_accumulator( + &family(512), + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .unwrap(); + for (at, value) in values.iter().enumerate() { + operator.validate_single_input(*value).unwrap(); + operator.update_single(*value, at as i64); + } + operator.into_accumulator() +} +fn read(state: &dyn AggregateCore) -> f64 { + state + .query_statistic( + Statistic::Quantile, + &None, + &HashMap::from([("quantile".into(), "0.5".into())]), + ) + .unwrap() +} + +// The same kernels work when every build is query-time, when only a prefix +// was precomputed, and when all state was precomputed before the readout. +#[test] +fn raw_partial_and_fully_precomputed_use_the_same_kernels() { + let raw: Vec = (0..128).map(f64::from).collect(); + let raw_only = build(&raw); + let stored_prefix = build(&raw[..64]); + let query_time_suffix = build(&raw[64..]); + let partial = stored_prefix.merge_with(&*query_time_suffix).unwrap(); + let stored_complete = build(&raw); + assert_eq!(read(&*raw_only), read(&*partial)); + assert_eq!(read(&*partial), read(&*stored_complete)); + assert!((read(&*raw_only) - 64.0).abs() <= 1.0); +} + +// A compiler must reject invalid physical parameters before starting execution. +#[test] +fn invalid_kll_parameters_are_rejected_at_binding() { + let result = create_planner_accumulator( + &family(0), + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ); + assert!(result.is_err()); +} + +// Native CountSketch supports the confidence-sized depth used by the backend; +// a packed-wire column-bit budget must not be imposed on this constructor. +#[test] +fn native_count_sketch_dimensions_are_not_packed_wire_dimensions() { + use asap_physical_operators::planner::post_asap::SummaryInputExpr; + use asap_physical_operators::KeyByLabelValues; + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 1200, + depth: 55, + heap_size: 3, + }, + ), + Default::default(), + ); + let mut update = SummaryUpdate::column(ColumnRef::SampleValue); + update.item = Some(SummaryInputExpr::Column(ColumnRef::Named("host".into()))); + let mut operator = create_planner_accumulator(&family, &update, &Default::default()).unwrap(); + let key = KeyByLabelValues::new_with_labels(vec!["a".into()]); + operator.update_keyed(&key, 7.0, 1000); + let state = operator.into_accumulator(); + assert_eq!( + state + .query_statistic(Statistic::Sum, &Some(key), &Default::default()) + .unwrap(), + 7.0 + ); +} diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs new file mode 100644 index 00000000..8729a1d9 --- /dev/null +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -0,0 +1,814 @@ +//! Acceptance tests use the library directly, without either backend engine. +use asap_physical_operators::{ + dag::{ + operators::{Expression, Operator, Reduction, SortKey}, + values::{Batch, Schema, Value}, + Limits, PhysicalDag, RunContext, Scope, + }, + Statistic, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{ExactKind, ExactParams, SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, +}; +use std::sync::Arc; +fn schema(fields: &[(&str, DataType, bool)]) -> Schema { + Arc::new(SummarySchema { + fields: fields + .iter() + .map(|(name, dtype, nullable)| SummaryField { + name: (*name).into(), + dtype: SummaryFamilyType::Plain(dtype.clone()), + nullable: *nullable, + }) + .collect(), + time_index: None, + }) +} +fn run(dag: &PhysicalDag<'_, Batch, Schema>, root: u64, scope: Scope) -> Vec> { + let context = RunContext::new( + scope, + Limits { + max_buffered_batches: 1, + ..Limits::default() + }, + ) + .unwrap(); + block_on(async { + let mut stream = dag.execute(&[root], context.clone()).unwrap().remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend(batch.unwrap().rows().iter().cloned()); + } + assert_eq!(context.retained_bytes(), 0); + rows + }) +} +fn query() -> Scope { + Scope::Query { + evaluation_time_ms: 1000, + revision: 2, + } +} +fn floats(rows: &[Vec], column: usize) -> Vec { + rows.iter() + .map(|r| { + if let Value::Float64(v) = r[column] { + v + } else { + panic!("not Float64") + } + }) + .collect() +} + +// Sort followed by partitioned Limit implements ranking independently per group. +#[test] +fn grouped_sort_limit_across_batches() { + let schema = schema(&[ + ("group", DataType::Int64, false), + ("score", DataType::Float64, false), + ]); + let batches = [ + vec![(1, 1.), (2, 4.), (1, 9.)], + vec![(2, 8.), (1, 5.), (2, 2.)], + ] + .into_iter() + .map(|rows| { + Batch::try_new( + schema.clone(), + rows.into_iter() + .map(|(g, v)| vec![Value::Int64(g), Value::Float64(v)]) + .collect(), + ) + .unwrap() + }) + .collect(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(schema.clone(), batches).unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::sort( + schema.clone(), + vec![SortKey { + column: 1, + descending: true, + nulls_first: false, + }], + vec![0], + ) + .unwrap(), + ) + .unwrap(); + dag.add(2, vec![1], Operator::limit(schema, 1, 1, vec![0]).unwrap()) + .unwrap(); + assert_eq!(floats(&run(&dag, 2, query()), 1), vec![5., 4.]); +} + +// The same computation runs in either engine scope with fresh per-run state. +#[test] +fn summary_construction_merge_and_readout_at_both_phases() { + let schema = schema(&[("v", DataType::Float64, false)]); + let batches = (1..=20) + .map(|v| Batch::try_new(schema.clone(), vec![vec![Value::Float64(v as f64)]]).unwrap()) + .collect(); + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let build = Operator::summary_build(schema.clone(), family, 0, None, vec![]).unwrap(); + let state = build.schema(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(schema, batches).unwrap()) + .unwrap(); + dag.add(1, vec![0], build).unwrap(); + dag.add(2, vec![1, 1], Operator::union(state.clone(), 2).unwrap()) + .unwrap(); + dag.add( + 3, + vec![2], + Operator::summary_merge(state.clone(), 0, vec![]).unwrap(), + ) + .unwrap(); + dag.add( + 4, + vec![3], + Operator::readout(state, 0, Statistic::Sum, Default::default()).unwrap(), + ) + .unwrap(); + for scope in [ + query(), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1000, + revision: 2, + }, + ] { + assert_eq!(floats(&run(&dag, 4, scope), 0), vec![420.]); + } +} + +// A semi-join can consume two branches of one producer with a one-batch buffer. +#[test] +fn diamond_semijoin_preserves_left_values_and_multiplicity() { + let schema = schema(&[("key", DataType::Int64, false)]); + let batches = [1, 2, 2, 3] + .into_iter() + .map(|v| Batch::try_new(schema.clone(), vec![vec![Value::Int64(v)]]).unwrap()) + .collect(); + let filter = Operator::filter( + schema.clone(), + Expression::Equal( + Box::new(Expression::Column(0)), + Box::new(Expression::Literal { + value: Value::Int64(2), + dtype: DataType::Int64, + }), + ), + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(schema.clone(), batches).unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], filter).unwrap(); + dag.add( + 2, + vec![0, 1], + Operator::semi_join(schema.clone(), schema, vec![(0, 0)]).unwrap(), + ) + .unwrap(); + let rows = run(&dag, 2, query()); + assert_eq!(rows.len(), 2); + assert!(rows.iter().all(|r| matches!(r[0], Value::Int64(2)))); +} + +// Integer aggregation must not silently lose precision through Float64. +#[test] +fn exact_integer_and_empty_extrema() { + let schema = schema(&[("v", DataType::Int64, false)]); + let aggregate = Operator::aggregate( + schema.clone(), + vec![], + vec![("sum".into(), Reduction::Sum(0))], + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + let value = 9_007_199_254_740_993; + dag.add( + 0, + vec![], + Operator::source( + schema.clone(), + vec![Batch::try_new( + schema.clone(), + vec![vec![Value::Int64(value)], vec![Value::Int64(2)]], + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], aggregate).unwrap(); + assert!(matches!(run(&dag,1,query())[0][0],Value::Int64(v) if v==value+2)); + let mut empty = PhysicalDag::default(); + empty + .add(0, vec![], Operator::source(schema.clone(), vec![]).unwrap()) + .unwrap(); + empty + .add( + 1, + vec![0], + Operator::aggregate(schema, vec![], vec![("min".into(), Reduction::Min(0))]).unwrap(), + ) + .unwrap(); + assert!(matches!(run(&empty, 1, query())[0][0], Value::Null)); +} + +// Plain value operators are library implementations, including NaN comparison. +#[test] +fn scalar_negation_and_vector_conversion() { + let scalar = Operator::scalar(Value::Float64(7.), DataType::Float64).unwrap(); + let project = Operator::project( + scalar.schema(), + vec![( + "v".into(), + Expression::Negate(Box::new(Expression::Column(0))), + )], + ) + .unwrap(); + let convert = Operator::vector_to_scalar(project.schema(), 0).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], scalar).unwrap(); + dag.add(1, vec![0], project).unwrap(); + dag.add(2, vec![1], convert).unwrap(); + assert_eq!(floats(&run(&dag, 2, query()), 0), vec![-7.]); + let scalar = Operator::scalar(Value::Float64(f64::NAN), DataType::Float64).unwrap(); + let predicate = Expression::Equal( + Box::new(Expression::Column(0)), + Box::new(Expression::Column(0)), + ); + let filter = Operator::filter(scalar.schema(), predicate).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], scalar).unwrap(); + dag.add(1, vec![0], filter).unwrap(); + assert!(run(&dag, 1, query()).is_empty()); +} + +// Invalid operations fail at binding rather than becoming external fallbacks. +#[test] +fn binding_rejects_unsupported_operations() { + let schema = schema(&[("v", DataType::Float64, false)]); + assert!(Operator::summary_build( + schema.clone(), + SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), + 0, + None, + vec![] + ) + .is_err()); + let sum = Operator::summary_build( + schema.clone(), + SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + 0, + None, + vec![], + ) + .unwrap(); + assert!(Operator::readout(sum.schema(), 0, Statistic::Quantile, Default::default()).is_err()); + assert!(Operator::filter(schema, Expression::Column(0)).is_err()); +} + +// KLL is one family example: precomputation changes input sources, not operators. +#[test] +fn kll_raw_partial_and_precomputed_are_native_dags() { + use planner_types::post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("value", DataType::Float64, false)]); + let family = SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), + GroupingStrategy::PerSubpopulationInstance, + ); + let build = Operator::summary_build(input.clone(), family, 0, None, vec![]).unwrap(); + let state = build.schema(); + let build_range = |start: u32, end: u32| { + let mut dag = PhysicalDag::default(); + let batch = Batch::try_new( + input.clone(), + (start..end) + .map(|v| vec![Value::Float64(f64::from(v))]) + .collect(), + ) + .unwrap(); + dag.add( + 0, + vec![], + Operator::source(input.clone(), vec![batch]).unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], build.clone()).unwrap(); + run( + &dag, + 1, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1000, + revision: 1, + }, + ) + }; + let prefix = build_range(0, 64); + let complete = build_range(0, 128); + let query_plan = |stored: Option>>, raw_start: Option| { + let mut dag = PhysicalDag::default(); + let mut states = vec![]; + if let Some(rows) = stored { + dag.add( + 0, + vec![], + Operator::source( + state.clone(), + vec![Batch::try_new(state.clone(), rows).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + states.push(0); + } + if let Some(start) = raw_start { + dag.add( + 1, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new( + input.clone(), + (start..128) + .map(|v| vec![Value::Float64(f64::from(v))]) + .collect(), + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add(2, vec![1], build.clone()).unwrap(); + states.push(2); + } + dag.add( + 3, + states.clone(), + Operator::union(state.clone(), states.len()).unwrap(), + ) + .unwrap(); + dag.add( + 4, + vec![3], + Operator::summary_merge(state.clone(), 0, vec![]).unwrap(), + ) + .unwrap(); + dag.add( + 5, + vec![4], + Operator::readout( + state.clone(), + 0, + Statistic::Quantile, + std::collections::HashMap::from([("quantile".into(), "0.5".into())]), + ) + .unwrap(), + ) + .unwrap(); + floats(&run(&dag, 5, query()), 0)[0] + }; + let raw = query_plan(None, Some(0)); + let partial = query_plan(Some(prefix), Some(64)); + let full = query_plan(Some(complete), None); + assert_eq!(raw, partial); + assert_eq!(partial, full); + assert!((raw - 64.).abs() <= 1.); +} + +// Restored state must retain its family; a mislabeled state is rejected. +#[test] +fn restored_exact_state_and_family_validation() { + use asap_physical_operators::{ + accumulators::exact_accumulator::ExactAccumulator, SerializableToSink, + }; + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); + acc.update(None, 7., 0); + let acc = ExactAccumulator::deserialize_from_bytes(&acc.serialize_to_bytes()).unwrap(); + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }], + time_index: None, + }); + let value = Value::Summary { + family: family.clone(), + state: Arc::new(acc), + }; + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + schema.clone(), + vec![Batch::try_new(schema.clone(), vec![vec![value]]).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::readout(schema.clone(), 0, Statistic::Sum, Default::default()).unwrap(), + ) + .unwrap(); + assert_eq!(floats(&run(&dag, 1, query()), 0), vec![7.]); + let wrong = ExactAccumulator::new( + SummaryFamilyType::ExactAggregate(ExactKind::Max, ExactParams::Max), + false, + ) + .unwrap(); + assert!(Batch::try_new( + schema, + vec![vec![Value::Summary { + family, + state: Arc::new(wrong) + }]] + ) + .is_err()); +} + +// Planner binding rejects unknown computation instead of accepting a fallback. +#[test] +fn bind_post_asap_before_execution() { + use asap_physical_operators::dag::planner::bind; + use planner_types::{ + post_asap::{ + EdgeRole, ExecutableDag, ExecutableDagEdge, ExecutableDagNode, + ExecutableOperatorPayload, ExecutionDataState, GroupingEdgeCompatibility, + PostAsapNodeId, ValueOperation, WindowEdgeCompatibility, + }, + pre_asap::{ArithmeticOpKind, ProjectItem, QueryExpr, ScalarValue}, + }; + use std::{collections::BTreeMap, rc::Rc}; + let schema = schema(&[("value", DataType::Float64, false)]); + let node = |id, payload| ExecutableDagNode { + id: PostAsapNodeId(id), + payload, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*schema).clone(), + guarantee: None, + }; + let mut dag = ExecutableDag { + nodes: vec![ + node( + 0, + ExecutableOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(1.), + }, + ), + node( + 1, + ExecutableOperatorPayload::Value { + operation: ValueOperation::Project { + cols: vec![ProjectItem { + alias: None, + expr: QueryExpr::Arithmetic { + op: ArithmeticOpKind::Add, + left: Rc::new(QueryExpr::Column(0)), + right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.))), + }, + }], + qualifier: None, + }, + }, + ), + ], + edges: vec![ExecutableDagEdge { + producer: PostAsapNodeId(0), + consumer: PostAsapNodeId(1), + role: EdgeRole::Input, + intermediate_schema: (*schema).clone(), + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }], + root: PostAsapNodeId(1), + }; + let sources = || -> BTreeMap> { + BTreeMap::from([( + 0, + Box::new( + Operator::source( + schema.clone(), + vec![Batch::try_new(schema.clone(), vec![vec![Value::Float64(1.)]]).unwrap()], + ) + .unwrap(), + ) as asap_physical_operators::dag::planner::Source<'static>, + )]) + }; + let native = bind(&dag, sources(), &[1]).unwrap(); + assert_eq!(floats(&run(&native, 1, query()), 0), vec![3.]); + assert!(bind(&dag, BTreeMap::new(), &[1]).is_err()); + dag.nodes[1].payload = ExecutableOperatorPayload::Value { + operation: ValueOperation::Extension { + name: "unknown".into(), + }, + }; + assert!(bind(&dag, sources(), &[1]).is_err()); +} + +// A completed empty population has an exact zero count, with integer output. +#[test] +fn empty_exact_count_is_an_integer_state_readout() { + let input = schema(&[("value", DataType::Float64, false)]); + let build = Operator::summary_build( + input.clone(), + SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count), + 0, + None, + vec![], + ) + .unwrap(); + let read = Operator::readout(build.schema(), 0, Statistic::Count, Default::default()).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(input, vec![]).unwrap()) + .unwrap(); + dag.add(1, vec![0], build).unwrap(); + dag.add(2, vec![1], read).unwrap(); + assert!(matches!(run(&dag, 2, query())[0][0], Value::Int64(0))); +} + +// A deployment source cannot pass a different row shape to bound expressions. +#[test] +fn source_batches_must_match_the_bound_schema() { + use asap_physical_operators::dag::{self, PhysicalOperator}; + use planner_types::{ + post_asap::{ + ExecutableDag, ExecutableDagNode, ExecutableOperatorPayload, ExecutionDataState, + PostAsapNodeId, + }, + pre_asap::QueryExpr, + }; + use std::{cell::Cell, collections::BTreeMap, rc::Rc}; + struct WrongSource { + schema: Schema, + starts: Rc>, + } + impl PhysicalOperator for WrongSource { + fn name(&self) -> &str { + "ExternalSource" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.schema.clone() + } + fn output_bytes(&self, value: &Batch) -> usize { + value.bytes() + } + fn start<'a>( + &'a self, + _: Vec>, + _: RunContext, + ) -> Result, dag::Error> { + self.starts.set(self.starts.get() + 1); + Ok( + futures::stream::once(async { Batch::try_new(schema(&[]), vec![vec![]]) }) + .boxed_local(), + ) + } + } + let expected = schema(&[("value", DataType::Float64, false)]); + let starts = Rc::new(Cell::new(0)); + let plan = ExecutableDag { + nodes: vec![ExecutableDagNode { + id: PostAsapNodeId(0), + payload: ExecutableOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(1.), + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*expected).clone(), + guarantee: None, + }], + edges: vec![], + root: PostAsapNodeId(0), + }; + let source = Box::new(WrongSource { + schema: expected, + starts: starts.clone(), + }) as dag::planner::Source<'static>; + let native = dag::planner::bind(&plan, BTreeMap::from([(0, source)]), &[0]).unwrap(); + assert_eq!(starts.get(), 0); + let context = RunContext::new(query(), Limits::default()).unwrap(); + let mut output = native.execute(&[0], context).unwrap().remove(0); + assert!(matches!( + block_on(output.next()), + Some(Err(dag::Error::AtNode { node: 0, .. })) + )); + assert_eq!(starts.get(), 1); +} + +// Float extrema have the same NaN behavior as the exact summary kernels. +#[test] +fn extrema_preserve_numeric_values_in_the_presence_of_nan() { + let input = schema(&[("v", DataType::Float64, false)]); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new( + input.clone(), + vec![vec![Value::Float64(-f64::NAN)], vec![Value::Float64(5.)]], + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::aggregate( + input, + vec![], + vec![ + ("min".into(), Reduction::Min(0)), + ("max".into(), Reduction::Max(0)), + ], + ) + .unwrap(), + ) + .unwrap(); + let rows = run(&dag, 1, query()); + assert_eq!(floats(&rows, 0), vec![5.]); + assert_eq!(floats(&rows, 1), vec![5.]); +} + +// Planner wire nodes, including grouping and edge roles, are executable at either phase. +#[test] +fn planner_semijoin_sort_limit_contract_at_both_phases() { + use asap_physical_operators::dag::planner::{bind, Source}; + use planner_types::{ + post_asap::*, + pre_asap::{CompareOpKind, GroupKeys, JoinKind, Predicate, QueryExpr, SortKey}, + }; + use std::{collections::BTreeMap, rc::Rc}; + let rows_schema = schema(&[ + ("group", DataType::Utf8, false), + ("key", DataType::Utf8, false), + ("score", DataType::Float64, false), + ]); + let keys_schema = schema(&[("key", DataType::Utf8, false)]); + let node = |id, payload, schema: &Schema| ExecutableDagNode { + id: PostAsapNodeId(id), + payload, + output_schema: (**schema).clone(), + output_state: ExecutionDataState::QUERY_ROWS, + guarantee: None, + }; + let edge = |producer, consumer, role, schema: &Schema| ExecutableDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role, + intermediate_schema: (**schema).clone(), + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }; + let groups = GroupKeys::by(vec![0]); + let dag = ExecutableDag { + nodes: vec![ + node( + 0, + ExecutableOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(0.), + }, + &rows_schema, + ), + node( + 1, + ExecutableOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(0.), + }, + &keys_schema, + ), + node( + 2, + ExecutableOperatorPayload::RelationalJoin { + join_kind: JoinKind::Semi, + pruning: None, + pred: Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(1)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(3)), + })), + }, + &rows_schema, + ), + node( + 3, + ExecutableOperatorPayload::Value { + operation: ValueOperation::Sort { + keys: vec![SortKey { + expr: QueryExpr::Column(2), + ascending: false, + nulls_first: false, + }], + partition_by: groups.clone(), + }, + }, + &rows_schema, + ), + node( + 4, + ExecutableOperatorPayload::Value { + operation: ValueOperation::Limit { + n: 1, + offset: 0, + partition_by: groups, + }, + }, + &rows_schema, + ), + ], + // Deliberately put Right before Left: list order must not swap inputs. + edges: vec![ + edge(1, 2, EdgeRole::Right, &keys_schema), + edge(0, 2, EdgeRole::Left, &rows_schema), + edge(2, 3, EdgeRole::Input, &rows_schema), + edge(3, 4, EdgeRole::Input, &rows_schema), + ], + root: PostAsapNodeId(4), + }; + let text = |v: &str| Value::Utf8(v.into()); + for (phase, scope) in [ + (ExecutionTiming::QueryTime, query()), + ( + ExecutionTiming::IngestionTime, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1000, + revision: 2, + }, + ), + ] { + let dag = dag + .with_execution_phases(&dag.nodes.iter().map(|node| (node.id, phase)).collect()) + .unwrap(); + let sources: BTreeMap> = BTreeMap::from([ + ( + 0, + Box::new( + Operator::source( + rows_schema.clone(), + vec![Batch::try_new( + rows_schema.clone(), + vec![ + vec![text("a"), text("x"), Value::Float64(8.)], + vec![text("a"), text("y"), Value::Float64(9.)], + vec![text("b"), text("x"), Value::Float64(2.)], + vec![text("b"), text("z"), Value::Float64(99.)], + ], + ) + .unwrap()], + ) + .unwrap(), + ) as Source<'static>, + ), + ( + 1, + Box::new( + Operator::source( + keys_schema.clone(), + vec![Batch::try_new( + keys_schema.clone(), + vec![vec![text("x")], vec![text("y")]], + ) + .unwrap()], + ) + .unwrap(), + ) as Source<'static>, + ), + ]); + let native = bind(&dag, sources, &[4]).unwrap(); + let mut scores = floats(&run(&native, 4, scope), 2); + scores.sort_by(f64::total_cmp); + assert_eq!(scores, vec![2., 9.]); + } +} diff --git a/crates/asap_sketch_codec/Cargo.toml b/crates/asap_sketch_codec/Cargo.toml new file mode 100644 index 00000000..9ce14d38 --- /dev/null +++ b/crates/asap_sketch_codec/Cargo.toml @@ -0,0 +1,8 @@ +[package] +name = "asap_sketch_codec" +version = "0.1.0" +edition = "2021" + +[dependencies] +asap_sketchlib = { git = "https://github.com/ProjectASAP/asap_sketchlib", rev = "026cd18c7b8c23ae6c46d4d683151ba562b8cd3a" } +prost = "0.13" diff --git a/crates/asap_sketch_codec/src/lib.rs b/crates/asap_sketch_codec/src/lib.rs new file mode 100644 index 00000000..279efe16 --- /dev/null +++ b/crates/asap_sketch_codec/src/lib.rs @@ -0,0 +1,84 @@ +//! Runtime-independent decoding of the sketchlib protobuf envelope. + +use asap_sketchlib::proto::sketchlib::{ + sketch_envelope::SketchState, DdSketchState, KllState, SketchEnvelope, +}; +use asap_sketchlib::DdSketch; +use prost::Message; + +pub fn envelope_state(bytes: &[u8]) -> Result, String> { + SketchEnvelope::decode(bytes) + .map(|envelope| envelope.sketch_state) + .map_err(|error| format!("decode SketchEnvelope: {error}")) +} + +pub fn ddsketch_state(bytes: &[u8]) -> Result<(DdSketchState, f64), String> { + let envelope = + SketchEnvelope::decode(bytes).map_err(|error| format!("decode SketchEnvelope: {error}"))?; + match envelope.sketch_state { + Some(SketchState::Ddsketch(state)) => Ok((state, envelope.sample_p)), + _ => Err("SketchEnvelope contains no DDSketch state".into()), + } +} + +pub fn reconstruct_ddsketch(bytes: &[u8]) -> Result<(DdSketch, f64), String> { + let (state, sample_p) = ddsketch_state(bytes)?; + if !state.alpha.is_finite() || !(0.0..1.0).contains(&state.alpha) || state.alpha == 0.0 { + return Err("DDSketch alpha must be finite and between zero and one".into()); + } + Ok(( + DdSketch::from_raw(state.alpha, state.store_counts, state.store_offset), + sample_p, + )) +} + +pub fn kll_state(bytes: &[u8]) -> Result { + let envelope = + SketchEnvelope::decode(bytes).map_err(|error| format!("decode SketchEnvelope: {error}"))?; + match envelope.sketch_state { + Some(SketchState::Kll(state)) => Ok(state), + _ => Err("SketchEnvelope contains no KLL state".into()), + } +} + +pub fn encode_ddsketch(sketch: &DdSketch) -> Vec { + let envelope = SketchEnvelope { + format_version: 1, + producer: None, + hash_spec: None, + sample_p: 0.0, + sketch_state: Some(SketchState::Ddsketch(DdSketchState { + alpha: sketch.wire_alpha(), + store_counts: sketch.store_counts.clone(), + store_offset: sketch.store_offset, + })), + }; + envelope.encode_to_vec() +} + +pub fn encode_kll(sketch: &asap_sketchlib::sketches::kll::KLL) -> Vec { + use asap_sketchlib::proto::sketchlib::CoinState; + let (state, bit_cache, remaining_bits) = sketch.wire_coin(); + SketchEnvelope { + format_version: 1, + producer: None, + hash_spec: None, + sample_p: 0.0, + sketch_state: Some(SketchState::Kll(KllState { + k: sketch.wire_k(), + m: sketch.wire_m(), + num_levels: sketch.wire_num_levels(), + levels: sketch.wire_levels(), + items: sketch.wire_items(), + coin: Some(CoinState { + state, + bit_cache, + remaining_bits, + }), + offset: 0.0, + value_scale: 0, + residuals: Vec::new(), + })), + } + .encode_to_vec() +} diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md new file mode 100644 index 00000000..609311c0 --- /dev/null +++ b/docs/design_docs/physical-operators.md @@ -0,0 +1,85 @@ +# Shared physical operators and DAG execution + +## Decision + +ASAP owns an independent physical operator library and DAG runtime. Precompute +and query engines bind inputs and consume outputs from the same library. The +operator defines computation; the engine supplies ingestion time or query time, +window boundaries, storage access and publication. There is no second execution +algorithm selected by phase. + +The library lives in the ASAPPlanner workspace as `asap-physical-operators`, +alongside `asap-types` and logical-to-physical lowering. It depends on those local +IR types and never on ASAPQuery-backend. New IR nodes and their implementations +can be reviewed and tested in one Planner PR. Private implementation tests live +inside their owning crate; integration tests exercise exported physical DAGs. +Its native operators can execute independently of either backend engine. +Engine integration must use these operators for computation, rather than merely +using the shared scheduler around a second implementation. + +## Execution contract + +An immutable plan describes typed nodes and dependency edges. Each execution +creates its own operator state. One producer may have multiple consumers; the +producer executes once in that run and sends the same outputs to all consumers. +Separate runs, query evaluation times and ingestion windows do not share mutable +state. Request-local caching of intermediate results is scoped to execution. + +The runtime validates dependencies, schemas, arity and cycles before sources +start. Each consumer advances independently. Bounded queues apply backpressure; +dropping one consumer does not cancel other consumers. Whole-run cancellation +wakes readers and releases queued work as streams are polled or dropped. + +Execution runs on the caller's worker without an internal thread pool. Active +streams are worker-local. Deployments poll all consumers concurrently. The byte +budget accounts for retained outputs and native operator state, including outputs +held after queue eviction. It is not an RSS limit: source-owned data, temporary +allocation peaks and allocator overhead remain outside that estimate. Blocking +operators currently have no spill implementation. + +## Operator coverage + +Native operations include scalar sources, typed Project and Filter, arithmetic +and boolean expressions, exact grouped aggregation, semi-join, grouped Sort and +Limit, Union, vector-to-scalar conversion, and summary construction, merge and +readout. Grouped TopK composes Sort and Limit within each group; candidate +completeness is an earlier pruning obligation. + +Values retain Planner types and nullability. Native summary batches currently +support exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch and HLL. Other available +low-level kernels do not imply native batch bindings. Unsupported expressions, +state families and parameters must be rejected during binding, without an +implicit external fallback. The backend retains source, storage, publication and protocol adapters. +Computation must bind to Planner operations without a second backend operator +vocabulary. Binding rejects unsupported Planner nodes before starting sources. + +Deployments provide explicit storage or ingestion source frontiers. A supplied +batch source is not a backend raw Scan implementation. Local backend raw Scan +is deferred; a raw-only library test does not establish that deployment capability. + +## DataFusion reuse vs independent implementation + +| Decision dimension | Reuse DataFusion | Independent ASAP implementation | +| --- | --- | --- | +| General computation | Reuse mature Arrow operators and expression execution | Implement and test the supported Planner vocabulary explicitly | +| Shared DAG producer | Shared plan references need an explicit execution-sharing and buffering policy | One producer and independent consumer cursors are part of the runtime contract | +| Summary lifecycle | Add custom summary state operators to the framework | Summary construction, merge and readout are native capabilities | +| Engine reuse | Adapt both engines to DataFusion's execution model | Both engines bind the same ASAP interfaces | +| Engineering cost | Less generic operator work; integration and semantic adaptation remain | More operator, typing, scheduling and resource-accounting responsibility | + +DataFusion is a design reference, not this library's execution dependency. This +choice does not claim that DataFusion cannot express shared dependencies. ASAP +chooses direct ownership of execution sharing and summary-state semantics across +both engines. Mathematical sketch kernels remain reusable implementation details. + +## Acceptance + +Independent tests must execute shared-producer diamonds without duplicated work +or deadlock, exercise slow and dropped consumers, propagate cancellation and +errors, retain memory accounting, and isolate separate executions. Operator tests +must cover types, nulls, grouped limits, state compatibility and unsupported +bindings. The same summary pipeline must run at ingestion time and query time. + +Backend integration adds deployment acceptance for source binding, window and revision +scope, durable publication and query output adaptation. External exact forwarding +does not count as evidence that a local operator was implemented. From 7f986bf8650812bc6bab00a0a28f2b4b11987e21 Mon Sep 17 00:00:00 2001 From: zz_y Date: Wed, 23 Sep 2026 20:56:28 +0000 Subject: [PATCH 02/90] style: align shared operator code with Planner lint policy --- .../src/dag/planner.rs | 4 +- crates/asap-physical-operators/src/factory.rs | 102 +++++++++--------- 2 files changed, 53 insertions(+), 53 deletions(-) diff --git a/crates/asap-physical-operators/src/dag/planner.rs b/crates/asap-physical-operators/src/dag/planner.rs index 71aa41a7..49e15e55 100644 --- a/crates/asap-physical-operators/src/dag/planner.rs +++ b/crates/asap-physical-operators/src/dag/planner.rs @@ -103,7 +103,7 @@ pub fn bind<'a>( .iter() .map(|id| Arc::new(nodes[id].output_schema.clone())) .collect::>(); - if matches!(node.payload, Payload::SummaryMerge { .. }) && inputs.len() > 1 { + if matches!(node.payload, Payload::SummaryMerge) && inputs.len() > 1 { if schemas.iter().any(|s| s != &schemas[0]) { return Err(invalid("summary merge inputs have different schemas")); } @@ -284,7 +284,7 @@ fn bind_operation(node: &ExecutableDagNode, inputs: &[Schema]) -> Result { + Payload::SummaryMerge => { let state = summary_column(input)?; Operator::summary_merge( input.clone(), diff --git a/crates/asap-physical-operators/src/factory.rs b/crates/asap-physical-operators/src/factory.rs index 0086c359..2a44e0f1 100644 --- a/crates/asap-physical-operators/src/factory.rs +++ b/crates/asap-physical-operators/src/factory.rs @@ -1048,6 +1048,57 @@ impl AccumulatorUpdater for PlannerExactUpdater { } } +struct UnivMonUpdater { + acc: UnivMonAccumulator, +} + +struct HllUpdater { + acc: HllSketchAccumulator, +} + +impl AccumulatorUpdater for HllUpdater { + fn is_keyed(&self) -> bool { + false + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } + fn update_single(&mut self, value: f64, _: i64) { + if !value.is_nan() { + let bits = if value == 0.0 { 0 } else { value.to_bits() }; + self.acc.inner.update(&bits.to_le_bytes()); + } + } + fn update_keyed(&mut self, _: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc.reset_to_empty(); + } +} + +impl AccumulatorUpdater for UnivMonUpdater { + fn is_keyed(&self) -> bool { + false + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } + fn update_single(&mut self, value: f64, _: i64) { + self.acc + .insert_sample(value) + .expect("UnivMon sample counter overflow"); + } + fn update_keyed(&mut self, _: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc.reset_to_empty(); + } +} + #[cfg(test)] mod planner_parameter_regression { use super::*; @@ -1137,54 +1188,3 @@ mod planner_parameter_regression { } } } - -struct UnivMonUpdater { - acc: UnivMonAccumulator, -} - -struct HllUpdater { - acc: HllSketchAccumulator, -} - -impl AccumulatorUpdater for HllUpdater { - fn is_keyed(&self) -> bool { - false - } - fn memory_usage_bytes(&self) -> usize { - self.acc.approx_memory_bytes() - } - fn update_single(&mut self, value: f64, _: i64) { - if !value.is_nan() { - let bits = if value == 0.0 { 0 } else { value.to_bits() }; - self.acc.inner.update(&bits.to_le_bytes()); - } - } - fn update_keyed(&mut self, _: &KeyByLabelValues, value: f64, timestamp_ms: i64) { - self.update_single(value, timestamp_ms); - } - impl_clone_accumulator_methods!(acc); - fn reset(&mut self) { - self.acc.reset_to_empty(); - } -} - -impl AccumulatorUpdater for UnivMonUpdater { - fn is_keyed(&self) -> bool { - false - } - fn memory_usage_bytes(&self) -> usize { - self.acc.approx_memory_bytes() - } - fn update_single(&mut self, value: f64, _: i64) { - self.acc - .insert_sample(value) - .expect("UnivMon sample counter overflow"); - } - fn update_keyed(&mut self, _: &KeyByLabelValues, value: f64, timestamp_ms: i64) { - self.update_single(value, timestamp_ms); - } - impl_clone_accumulator_methods!(acc); - fn reset(&mut self) { - self.acc.reset_to_empty(); - } -} From 3b999105dddf2c4f4e1aeb7d3ccad7a5e9757c62 Mon Sep 17 00:00:00 2001 From: zz_y Date: Wed, 23 Sep 2026 21:03:21 +0000 Subject: [PATCH 03/90] docs: compare Arrow batches with native summary state formats --- docs/design_docs/physical-operators.md | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index 609311c0..ac1b28f6 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -17,6 +17,28 @@ Its native operators can execute independently of either backend engine. Engine integration must use these operators for computation, rather than merely using the shared scheduler around a second implementation. +## Workspace organization + +[DataFusion's physical-plan crate](https://github.com/apache/datafusion/tree/main/datafusion/physical-plan) +is the ownership reference: it owns the execution-plan interface, concrete +operators, streams, metrics and operator tests within the same repository as +planning. ASAP follows that repository boundary, while retaining its own DAG +execution model. + +| Responsibility | ASAP owner | +| --- | --- | +| post-ASAP nodes, schemas, parameters and execution phase | `asap-types` | +| Logical-to-physical lowering and candidate correctness | `asap-aware-mapping` | +| Physical operator implementations, input/output validation and streams | `asap-physical-operators` | +| Shared-producer scheduling, cancellation and resource accounting | `asap-physical-operators` | +| Summary state encoding | `asap_sketch_codec` | +| Sources, durable stores, publication and serving protocols | Deployment repositories | + +The IR crate does not depend on execution. The physical operator crate depends +on the local IR crate. Operator unit tests can exercise private implementation +details; Planner integration tests check that emitted DAGs bind and execute. +Runtime values preserve the IR schema instead of redefining its type semantics. + ## Execution contract An immutable plan describes typed nodes and dependency edges. Each execution @@ -64,6 +86,7 @@ is deferred; a raw-only library test does not establish that deployment capabili | General computation | Reuse mature Arrow operators and expression execution | Implement and test the supported Planner vocabulary explicitly | | Shared DAG producer | Shared plan references need an explicit execution-sharing and buffering policy | One producer and independent consumer cursors are part of the runtime contract | | Summary lifecycle | Add custom summary state operators to the framework | Summary construction, merge and readout are native capabilities | +| In-memory representation | Operators exchange Arrow RecordBatch values. Custom summary state may not map naturally to Arrow and can require an explicit encoding, wrapper or conversion, with associated integration and potential copying costs | Native values can carry ASAP-defined summary state directly, without requiring every state format to fit Arrow; the library must still define and validate state types, ownership and compatibility | | Engine reuse | Adapt both engines to DataFusion's execution model | Both engines bind the same ASAP interfaces | | Engineering cost | Less generic operator work; integration and semantic adaptation remain | More operator, typing, scheduling and resource-accounting responsibility | From 9536cca5e4137bc86ad1af95b8d26e954455c535 Mon Sep 17 00:00:00 2001 From: zz_y Date: Wed, 23 Sep 2026 21:12:00 +0000 Subject: [PATCH 04/90] feat: execute Planner expressions and relational joins in native DAGs --- .../src/dag/expressions.rs | 431 ++++++++++++++++++ crates/asap-physical-operators/src/dag/mod.rs | 2 + .../src/dag/operators.rs | 109 ++++- .../src/dag/planner.rs | 119 ++--- .../tests/physical_dag.rs | 151 ++++++ 5 files changed, 734 insertions(+), 78 deletions(-) create mode 100644 crates/asap-physical-operators/src/dag/expressions.rs diff --git a/crates/asap-physical-operators/src/dag/expressions.rs b/crates/asap-physical-operators/src/dag/expressions.rs new file mode 100644 index 00000000..740dd323 --- /dev/null +++ b/crates/asap-physical-operators/src/dag/expressions.rs @@ -0,0 +1,431 @@ +//! Planner scalar expressions evaluated over native typed rows. +use super::{ + values::{Schema, Value}, + Error, +}; +use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind, DataType, QueryExpr, ScalarValue}; +use std::{cmp::Ordering, sync::Arc}; + +pub(super) fn evaluate( + expr: &QueryExpr, + row: &[Value], + schema: &planner_types::pre_asap::Schema, +) -> Result { + match expr { + QueryExpr::Column(index) => row.get(*index).cloned().ok_or(Error::Invalid(format!( + "column {index} outside row width {}", + row.len() + ))), + QueryExpr::Literal(value) => Ok(match value { + ScalarValue::Interval { + months, + days, + nanos, + } => Value::Interval { + months: *months, + days: *days, + nanos: *nanos, + }, + ScalarValue::Int64(value) => Value::Int64(*value), + ScalarValue::Float64(value) => Value::Float64(*value), + ScalarValue::Utf8(value) => Value::Utf8(value.clone().into()), + ScalarValue::Boolean(value) => Value::Bool(*value), + ScalarValue::Null => Value::Null, + }), + QueryExpr::Compare { left, op, right } => { + let left = evaluate(left, row, schema)?; + let right = evaluate(right, row, schema)?; + compare(op, left, right) + } + QueryExpr::Arithmetic { op, left, right } => arithmetic( + op, + evaluate(left, row, schema)?, + evaluate(right, row, schema)?, + ), + QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + let and = matches!(expr, QueryExpr::BoolAnd(_)); + let mut null = false; + for part in parts { + match evaluate(part, row, schema)? { + Value::Bool(value) if value != and => return Ok(Value::Bool(value)), + Value::Bool(_) => {} + Value::Null => null = true, + _ => return Err(Error::Invalid("boolean predicate required".into())), + } + } + Ok(if null { Value::Null } else { Value::Bool(and) }) + } + QueryExpr::Not(value) => match evaluate(value, row, schema)? { + Value::Bool(value) => Ok(Value::Bool(!value)), + Value::Null => Ok(Value::Null), + _ => Err(Error::Invalid("boolean predicate required".into())), + }, + QueryExpr::IsNull(value) => Ok(Value::Bool(matches!( + evaluate(value, row, schema)?, + Value::Null + ))), + QueryExpr::IsNotNull(value) => Ok(Value::Bool(!matches!( + evaluate(value, row, schema)?, + Value::Null + ))), + QueryExpr::FunctionCall { name, args } => { + use planner_types::pre_asap::scalar_signature::MapScalarFunction; + if name.eq_ignore_ascii_case("asap_struct_field") { + expr.scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))?; + let DataType::Struct { fields } = args[0] + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))? + .0 + else { + unreachable!() + }; + let offset = match &args[1] { + QueryExpr::Literal(ScalarValue::Int64(index)) => { + usize::try_from(index - 1).ok() + } + QueryExpr::Literal(ScalarValue::Utf8(name)) => { + fields.iter().position(|field| &field.name == name) + } + _ => None, + } + .ok_or_else(|| Error::Invalid("struct field selector".into()))?; + let Value::Struct(values) = evaluate(&args[0], row, schema)? else { + return Err(Error::Invalid("struct field input".into())); + }; + return values + .get(offset) + .cloned() + .ok_or_else(|| Error::Invalid("struct field value".into())); + } + if name.eq_ignore_ascii_case("asap_element_access") { + let (output_type, _) = expr + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))?; + if let DataType::List { element } = args[0] + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))? + .0 + { + let Value::List(values) = evaluate(&args[0], row, schema)? else { + return Err(Error::Invalid("array access input".into())); + }; + let index = match evaluate(&args[1], row, schema)? { + Value::Null => return Ok(Value::Null), + Value::Int64(index) => index, + _ => return Err(Error::Invalid("array access index".into())), + }; + let offset = if index > 0 { + usize::try_from(index - 1).ok() + } else if index < 0 { + usize::try_from(index.unsigned_abs()) + .ok() + .and_then(|distance| values.len().checked_sub(distance)) + } else { + None + }; + return match offset.and_then(|offset| values.get(offset)) { + Some(value) => Ok(value.clone()), + None => default_collection_element(&output_type, element.nullable), + }; + } + } + let function = (if name.eq_ignore_ascii_case("asap_element_access") { + Some(MapScalarFunction::Access) + } else { + MapScalarFunction::from_name(name) + }) + .ok_or_else(|| Error::Invalid(format!("scalar function {name}")))?; + expr.scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))?; + let values = args + .iter() + .map(|arg| evaluate(arg, row, schema)) + .collect::, _>>()?; + match function { + MapScalarFunction::Construct => { + let mut values = values.into_iter(); + let mut entries = Vec::new(); + while let Some(key) = values.next() { + if !matches!(key, Value::Int64(_) | Value::Utf8(_) | Value::Bool(_)) { + return Err(Error::Invalid("map key value type".into())); + } + entries.push(( + key, + values + .next() + .ok_or_else(|| Error::Invalid("odd map argument count".into()))?, + )); + } + Ok(Value::Map(entries.into())) + } + MapScalarFunction::Concat => { + let mut entries = Vec::new(); + for value in values { + let Value::Map(next) = value else { + return Err(Error::Invalid("map concat argument".into())); + }; + entries.extend(next.iter().cloned()); + } + Ok(Value::Map(entries.into())) + } + MapScalarFunction::Access => { + let [Value::Map(entries), key] = values.as_slice() else { + return Err(Error::Invalid("map access arguments".into())); + }; + if matches!(key, Value::Null) { + return Ok(Value::Null); + } + if !matches!(key, Value::Int64(_) | Value::Utf8(_) | Value::Bool(_)) { + return Err(Error::Invalid("map lookup key type".into())); + } + if let Some((_, value)) = entries + .iter() + .find(|(candidate, _)| cell_cmp(candidate, key) == Some(Ordering::Equal)) + { + return Ok(value.clone()); + } + let ( + DataType::Map { + value, + value_nullable, + .. + }, + _, + ) = args[0] + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))? + else { + unreachable!() + }; + default_collection_element(&value, value_nullable) + } + } + } + other => Err(Error::Invalid(format!("scalar expression {other:?}"))), + } +} + +fn default_collection_element(dtype: &DataType, nullable: bool) -> Result { + if nullable { + return Ok(Value::Null); + } + Ok(match dtype { + DataType::Interval | DataType::Date => { + return Err(Error::Invalid("temporal value transport".into())) + } + DataType::Null => Value::Null, + DataType::Int64 => Value::Int64(0), + DataType::Float64 => Value::Float64(0.0), + DataType::Utf8 => Value::Utf8("".into()), + DataType::Bool => Value::Bool(false), + DataType::Map { .. } => Value::Map(Arc::from([])), + DataType::List { .. } => Value::List(Arc::from([])), + DataType::Struct { fields } => Value::Struct( + fields + .iter() + .map(|field| default_collection_element(&field.dtype, field.nullable)) + .collect::, _>>()? + .into(), + ), + _ => { + return Err(Error::Invalid( + "collection missing-element default type".into(), + )) + } + }) +} + +fn compare(op: &CompareOpKind, left: Value, right: Value) -> Result { + if matches!(left, Value::Null) || matches!(right, Value::Null) { + return Ok(Value::Null); + } + let ordering = cell_cmp(&left, &right) + .ok_or_else(|| Error::Invalid("comparison of incompatible values".into()))?; + let value = match op { + CompareOpKind::Eq => ordering == Ordering::Equal, + CompareOpKind::Ne => ordering != Ordering::Equal, + CompareOpKind::Lt => ordering == Ordering::Less, + CompareOpKind::Le => ordering != Ordering::Greater, + CompareOpKind::Gt => ordering == Ordering::Greater, + CompareOpKind::Ge => ordering != Ordering::Less, + _ => return Err(Error::Invalid(format!("comparison {op:?}"))), + }; + Ok(Value::Bool(value)) +} + +fn arithmetic(op: &ArithmeticOpKind, left: Value, right: Value) -> Result { + let (left, right) = match (left, right) { + (Value::Int64(a), Value::Float64(b)) => (Value::Float64(a as f64), Value::Float64(b)), + (Value::Float64(a), Value::Int64(b)) => (Value::Float64(a), Value::Float64(b as f64)), + pair => pair, + }; + super::operators::numeric(op, left, right) +} + +fn integer_float_cmp(integer: i64, float: f64) -> Option { + if float.is_nan() { + return None; + } + // These bounds are powers of two, exactly representable as Float64. + if float >= 9_223_372_036_854_775_808.0 { + return Some(Ordering::Less); + } + if float < -9_223_372_036_854_775_808.0 { + return Some(Ordering::Greater); + } + let integral = float as i64; + match integer.cmp(&integral) { + Ordering::Equal => 0.0_f64.partial_cmp(&float.fract()), + other => Some(other), + } +} + +fn cell_cmp(left: &Value, right: &Value) -> Option { + match (left, right) { + (Value::Int64(left), Value::Int64(right)) => Some(left.cmp(right)), + (Value::Float64(left), Value::Float64(right)) => left.partial_cmp(right), + (Value::Int64(left), Value::Float64(right)) => integer_float_cmp(*left, *right), + (Value::Float64(left), Value::Int64(right)) => { + integer_float_cmp(*right, *left).map(Ordering::reverse) + } + (Value::Utf8(left), Value::Utf8(right)) => Some(left.cmp(right)), + (Value::Bool(left), Value::Bool(right)) => Some(left.cmp(right)), + (Value::Timestamp(left), Value::Timestamp(right)) => Some(left.cmp(right)), + (Value::Map(left), Value::Map(right)) => { + for ((left_key, left_value), (right_key, right_value)) in left.iter().zip(right.iter()) + { + let order = cell_cmp(left_key, right_key)?; + if order != Ordering::Equal { + return Some(order); + } + let order = match (left_value, right_value) { + (Value::Null, Value::Null) => Ordering::Equal, + (Value::Null, _) => Ordering::Greater, + (_, Value::Null) => Ordering::Less, + _ => cell_cmp(left_value, right_value)?, + }; + if order != Ordering::Equal { + return Some(order); + } + } + Some(left.len().cmp(&right.len())) + } + _ => None, + } +} + +#[derive(Clone, Debug)] +pub struct CompiledExpression { + expression: QueryExpr, + schema: planner_types::pre_asap::Schema, + output: (DataType, bool), +} +impl CompiledExpression { + pub fn compile(expression: &QueryExpr, input: &Schema) -> Result { + let schema = input + .fields + .iter() + .map(|field| { + let planner_types::post_asap::SummaryFamilyType::Plain(dtype) = &field.dtype else { + return Err(Error::Invalid( + "scalar expression cannot consume opaque summary state".into(), + )); + }; + Ok(planner_types::pre_asap::Column::new( + field.name.clone(), + dtype.clone(), + field.nullable, + )) + }) + .collect::, Error>>()?; + let schema = planner_types::pre_asap::Schema::new(schema); + validate(expression, &schema)?; + let output = expression + .scalar_type(&schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + Ok(Self { + expression: expression.clone(), + schema, + output, + }) + } + pub(super) fn dtype(&self) -> (DataType, bool) { + self.output.clone() + } + /// Evaluate a row under the same typed schema used when binding the expression. + pub fn evaluate(&self, row: &[Value]) -> Result { + if row.len() != self.schema.columns.len() + || row + .iter() + .zip(&self.schema.columns) + .any(|(value, column)| !value.matches(&column.dtype, column.nullable)) + { + return Err(Error::Invalid( + "expression input differs from its bound schema".into(), + )); + } + evaluate(&self.expression, row, &self.schema) + } +} +fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Result<(), Error> { + let invalid = || Error::Invalid(format!("unsupported scalar expression: {expr:?}")); + expr.scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + match expr { + QueryExpr::Column(_) | QueryExpr::Literal(_) => Ok(()), + QueryExpr::Arithmetic { left, right, .. } => { + for value in [left, right] { + validate(value, schema)?; + if !matches!( + value + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0, + DataType::Int64 | DataType::Float64 | DataType::Null + ) { + return Err(invalid()); + } + } + Ok(()) + } + QueryExpr::Compare { left, right, op } => { + if !matches!( + op, + CompareOpKind::Eq + | CompareOpKind::Ne + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge + ) { + return Err(invalid()); + } + validate(left, schema)?; + validate(right, schema) + } + QueryExpr::FunctionCall { name, args } => { + if name != "asap_struct_field" + && name != "asap_element_access" + && planner_types::pre_asap::scalar_signature::MapScalarFunction::from_name(name) + .is_none() + { + return Err(invalid()); + } + for arg in args { + validate(arg, schema)?; + } + Ok(()) + } + QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + for part in parts { + validate(part, schema)?; + } + Ok(()) + } + QueryExpr::Not(value) | QueryExpr::IsNull(value) | QueryExpr::IsNotNull(value) => { + validate(value, schema) + } + _ => Err(invalid()), + } +} diff --git a/crates/asap-physical-operators/src/dag/mod.rs b/crates/asap-physical-operators/src/dag/mod.rs index 1c9ee9e8..5f19f31e 100644 --- a/crates/asap-physical-operators/src/dag/mod.rs +++ b/crates/asap-physical-operators/src/dag/mod.rs @@ -515,3 +515,5 @@ mod tests; pub mod planner; pub mod batch_execution; + +pub mod expressions; diff --git a/crates/asap-physical-operators/src/dag/operators.rs b/crates/asap-physical-operators/src/dag/operators.rs index eeb1a922..ad6c4901 100644 --- a/crates/asap-physical-operators/src/dag/operators.rs +++ b/crates/asap-physical-operators/src/dag/operators.rs @@ -42,6 +42,7 @@ fn result_field(name: &str, dtype: DataType, nullable: bool) -> SummaryField { #[derive(Clone, Debug)] pub enum Expression { + Planner(Box), Column(usize), Literal { value: Value, @@ -61,9 +62,13 @@ pub enum Expression { IsNull(Box), } impl Expression { + pub fn planner(expression: super::expressions::CompiledExpression) -> Self { + Self::Planner(Box::new(expression)) + } fn dtype(&self, input: &Schema) -> Result<(DataType, bool), Error> { use Expression::*; match self { + Planner(expression) => Ok(expression.dtype()), Column(i) => { let (t, n) = plain(input, *i)?; Ok((t.clone(), n)) @@ -130,6 +135,7 @@ impl Expression { fn evaluate(&self, row: &[Value]) -> Result { use Expression::*; Ok(match self { + Planner(expression) => expression.evaluate(row)?, Column(i) => row[*i].clone(), Literal { value, .. } => value.clone(), Negate(v) => match v.evaluate(row)? { @@ -195,7 +201,7 @@ fn ordered(dtype: &DataType) -> bool { | DataType::Date ) } -fn numeric(op: &ArithmeticOpKind, a: Value, b: Value) -> Result { +pub(super) fn numeric(op: &ArithmeticOpKind, a: Value, b: Value) -> Result { use ArithmeticOpKind::*; Ok(match (a, b) { (Value::Null, _) | (_, Value::Null) => Value::Null, @@ -256,6 +262,10 @@ enum Kind { SemiJoin { keys: Vec<(usize, usize)>, }, + Join { + kind: planner_types::pre_asap::JoinKind, + predicate: super::expressions::CompiledExpression, + }, SummaryBuild { family: SummaryFamilyType, value: usize, @@ -434,6 +444,43 @@ impl Operator { output: left, }) } + pub fn relational_join( + left: Schema, + right: Schema, + kind: planner_types::pre_asap::JoinKind, + predicate: &planner_types::pre_asap::Predicate, + output: Schema, + ) -> Result { + use planner_types::pre_asap::JoinKind; + let mut joined = left.fields.clone(); + joined.extend(right.fields.clone()); + let predicate = + super::expressions::CompiledExpression::compile(&predicate.0, &schema(joined.clone()))?; + if predicate.dtype().0 != DataType::Bool { + return Err(invalid("join predicate must be boolean")); + } + let fields = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + left.fields.clone() + } else { + for field in &mut joined[..left.fields.len()] { + if matches!(kind, JoinKind::Right | JoinKind::Full) { + field.nullable = true; + } + } + for field in &mut joined[left.fields.len()..] { + if matches!(kind, JoinKind::Left | JoinKind::Full) { + field.nullable = true; + } + } + joined + }; + Self { + kind: Kind::Join { kind, predicate }, + inputs: vec![left, right], + output: schema(fields), + } + .with_output_schema(output) + } pub fn summary_build( input: Schema, family: SummaryFamilyType, @@ -602,6 +649,7 @@ impl PhysicalOperator for Operator { Kind::Sort { .. } => "Sort", Kind::Aggregate { .. } => "Aggregate", Kind::SemiJoin { .. } => "SemiJoin", + Kind::Join { .. } => "RelationalJoin", Kind::SummaryBuild { .. } => "SummaryAgg", Kind::SummaryMerge { .. } => "SummaryMerge", Kind::Readout { .. } => "SummaryReadout", @@ -630,6 +678,65 @@ impl PhysicalOperator for Operator { .map(|batch| batch.map(|batch| batch.value().clone())) .boxed_local()); } + if let Kind::Join { kind, predicate } = &self.kind { + let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; + let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; + return Ok(futures::stream::once(async move { + use planner_types::pre_asap::JoinKind; + let ((left, _left_memory), (right, _right_memory)) = futures::try_join!( + collect_rows(left, &context), + collect_rows(right, &context) + )?; + let mut result = Vec::new(); + let mut right_matched = vec![false; right.len()]; + for left_row in &left { + let mut matched = false; + for (i, right_row) in right.iter().enumerate() { + let mut joined = left_row.clone(); + joined.extend(right_row.iter().cloned()); + if *kind == JoinKind::Cross + || matches!(predicate.evaluate(&joined)?, Value::Bool(true)) + { + matched = true; + right_matched[i] = true; + match kind { + JoinKind::Semi => { + result.push(left_row.clone()); + break; + } + JoinKind::Anti => break, + _ => result.push(joined), + } + } + } + if !matched { + match kind { + JoinKind::Left | JoinKind::Full => { + let mut joined = left_row.clone(); + joined.resize( + joined.len() + self.inputs[1].fields.len(), + Value::Null, + ); + result.push(joined); + } + JoinKind::Anti => result.push(left_row.clone()), + _ => {} + } + } + } + if matches!(kind, JoinKind::Right | JoinKind::Full) { + for (matched, row) in right_matched.into_iter().zip(right) { + if !matched { + let mut joined = vec![Value::Null; self.inputs[0].fields.len()]; + joined.extend(row); + result.push(joined); + } + } + } + Batch::try_new(output, result) + }) + .boxed_local()); + } if let Kind::SemiJoin { keys } = &self.kind { let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; diff --git a/crates/asap-physical-operators/src/dag/planner.rs b/crates/asap-physical-operators/src/dag/planner.rs index 49e15e55..f8e0286b 100644 --- a/crates/asap-physical-operators/src/dag/planner.rs +++ b/crates/asap-physical-operators/src/dag/planner.rs @@ -2,7 +2,7 @@ //! frontiers supplied by the deployment; unsupported computation is an error. use super::{ operators::{Expression, Operator, Reduction, SortKey}, - values::{Batch, Schema, Value}, + values::{Batch, Schema}, Error, NodeId, PhysicalDag, PhysicalOperator, }; use planner_types::{ @@ -11,8 +11,7 @@ use planner_types::{ SketchQuery, SummaryFamilyType, SummaryInputExpr, ValueOperation, }, pre_asap::{ - AggIntent, ColumnRef, CompareOpKind, DataType, GroupKeys, QueryExpr, - Reduction as PlannerReduction, ScalarValue, + AggIntent, ColumnRef, CompareOpKind, GroupKeys, QueryExpr, Reduction as PlannerReduction, }, }; use std::{ @@ -116,9 +115,8 @@ pub fn bind<'a>( auxiliary -= 1; schemas.truncate(1); } - let operator = bind_operation(node, &schemas) - .map_err(|error| invalid(format!("node {id}: {error}")))? - .with_output_schema(output)?; + let operator = bind_node(node, &schemas) + .map_err(|error| invalid(format!("node {id}: {error}")))?; (Box::new(operator) as Source<'a>, inputs) }; graph.add_boxed(id, inputs, operator)?; @@ -127,19 +125,45 @@ pub fn bind<'a>( Ok(graph) } +/// Bind a Planner node against the schemas supplied by its deployment edges. +/// This is the same checked path used by complete DAG binding. +pub fn bind_node(node: &ExecutableDagNode, inputs: &[Schema]) -> Result { + for schema in inputs { + super::values::validate_schema(schema)?; + } + bind_operation(node, inputs)?.with_output_schema(Arc::new(node.output_schema.clone())) +} + fn bind_operation(node: &ExecutableDagNode, inputs: &[Schema]) -> Result { if let Payload::RelationalJoin { - join_kind: planner_types::pre_asap::JoinKind::Semi, + join_kind, pred, - .. + pruning, } = &node.payload { + use planner_types::{post_asap::CandidateCompleteness, pre_asap::JoinKind}; + if pruning.is_some() && *join_kind != JoinKind::Semi { + return Err(invalid("pruning certificate requires a semi-join")); + } + if matches!(pruning,Some(CandidateCompleteness::Certified { guarantee }) if guarantee.has_unknown() || guarantee.metric != planner_types::post_asap::ErrorMetric::TopKMembership) + { + return Err(invalid("invalid pruning certificate")); + } let [left, right] = inputs else { - return Err(invalid("semi-join requires two inputs")); + return Err(invalid("join requires two inputs")); }; - let mut keys = Vec::new(); - semi_join_keys(&pred.0, left.fields.len(), right.fields.len(), &mut keys)?; - return Operator::semi_join(left.clone(), right.clone(), keys); + if *join_kind == JoinKind::Semi { + if let Ok(keys) = equijoin_keys(pred, left, right) { + return Operator::semi_join(left.clone(), right.clone(), keys); + } + } + return Operator::relational_join( + left.clone(), + right.clone(), + join_kind.clone(), + pred, + Arc::new(node.output_schema.clone()), + ); } let [input] = inputs else { return Err(invalid( @@ -160,13 +184,13 @@ fn bind_operation(node: &ExecutableDagNode, inputs: &[Schema]) -> Result>()?, ), ValueOperation::Filter { pred } => { - Operator::filter(input.clone(), expression(&pred.0)?) + Operator::filter(input.clone(), expression(&pred.0, input)?) } ValueOperation::Sort { keys, partition_by } => Operator::sort( input.clone(), @@ -356,71 +380,12 @@ fn groups(input: &Schema, groups: &GroupKeys) -> Result, Error> { } Ok(groups.keys().to_vec()) } -fn expression(expr: &QueryExpr) -> Result { - let bind = |e: &QueryExpr| expression(e).map(Box::new); - Ok(match expr { - QueryExpr::Column(i) => Expression::Column(*i), - QueryExpr::Literal(value) => { - let (value, dtype) = match value { - ScalarValue::Int64(v) => (Value::Int64(*v), DataType::Int64), - ScalarValue::Float64(v) => (Value::Float64(*v), DataType::Float64), - ScalarValue::Utf8(v) => (Value::Utf8(v.as_str().into()), DataType::Utf8), - ScalarValue::Boolean(v) => (Value::Bool(*v), DataType::Bool), - ScalarValue::Null => (Value::Null, DataType::Null), - ScalarValue::Interval { - months, - days, - nanos, - } => ( - Value::Interval { - months: *months, - days: *days, - nanos: *nanos, - }, - DataType::Interval, - ), - }; - Expression::Literal { value, dtype } - } - QueryExpr::Arithmetic { op, left, right } => Expression::Arithmetic { - op: op.clone(), - left: bind(left)?, - right: bind(right)?, - }, - QueryExpr::Compare { - left, - op: CompareOpKind::Eq, - right, - } => Expression::Equal(bind(left)?, bind(right)?), - QueryExpr::Compare { - left, - op: CompareOpKind::Lt, - right, - } => Expression::Less(bind(left)?, bind(right)?), - QueryExpr::Not(v) => Expression::Not(bind(v)?), - QueryExpr::IsNull(v) => Expression::IsNull(bind(v)?), - QueryExpr::IsNotNull(v) => Expression::Not(Box::new(Expression::IsNull(bind(v)?))), - QueryExpr::BoolAnd(items) | QueryExpr::BoolOr(items) => { - let and = matches!(expr, QueryExpr::BoolAnd(_)); - let mut result = Expression::Literal { - value: Value::Bool(and), - dtype: DataType::Bool, - }; - for item in items { - result = if and { - Expression::And(Box::new(result), bind(item)?) - } else { - Expression::Or(Box::new(result), bind(item)?) - }; - } - result - } - _ => return Err(invalid("expression has no native implementation")), - }) +fn expression(expr: &QueryExpr, input: &Schema) -> Result { + Ok(Expression::planner( + super::expressions::CompiledExpression::compile(expr, input)?, + )) } -// Source adapters may perform I/O, but their actual batches must honor the -// schema accepted by the binder before a downstream expression sees a row. struct CheckedSource<'a> { source: Source<'a>, output: Schema, diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 8729a1d9..73d8b5d2 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -812,3 +812,154 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { assert_eq!(scores, vec![2., 9.]); } } + +// Planner scalar signatures, collection access and null predicates share native execution. +#[test] +fn planner_expressions_preserve_collection_and_nullable_types() { + use asap_physical_operators::dag::expressions::CompiledExpression; + use planner_types::pre_asap::{CompareOpKind, QueryExpr, ScalarValue}; + use std::rc::Rc; + let input_schema = schema(&[( + "items", + DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Int64), + value_nullable: false, + }, + false, + )]); + let access = QueryExpr::FunctionCall { + name: "asap_element_access".into(), + args: vec![ + QueryExpr::Column(0), + QueryExpr::Literal(ScalarValue::Utf8("count".into())), + ], + }; + let project = Operator::project( + input_schema.clone(), + vec![( + "count".into(), + Expression::planner(CompiledExpression::compile(&access, &input_schema).unwrap()), + )], + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + input_schema.clone(), + vec![Batch::try_new( + input_schema.clone(), + vec![ + vec![Value::Map( + vec![(Value::Utf8("count".into()), Value::Int64(7))].into(), + )], + vec![Value::Map(Arc::from([]))], + ], + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + let projected = project.schema(); + dag.add(1, vec![0], project).unwrap(); + let predicate = QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Ge, + right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + }; + dag.add( + 2, + vec![1], + Operator::filter( + projected.clone(), + Expression::planner(CompiledExpression::compile(&predicate, &projected).unwrap()), + ) + .unwrap(), + ) + .unwrap(); + let rows = run(&dag, 2, query()); + assert!(matches!(rows.as_slice(),[row] if matches!(row.as_slice(),[Value::Int64(7)]))); + let unknown = QueryExpr::FunctionCall { + name: "unregistered_function".into(), + args: vec![QueryExpr::Column(0)], + }; + assert!(CompiledExpression::compile(&unknown, &input_schema).is_err()); +} + +// Outer, semi and anti joins share Planner predicates and preserve SQL null behavior. +#[test] +fn native_relational_join_kinds_preserve_unmatched_rows() { + use planner_types::pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}; + use std::rc::Rc; + let input = schema(&[("key", DataType::Int64, true)]); + let predicate = Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(1)), + })); + for (kind, count) in [ + (JoinKind::Inner, 1), + (JoinKind::Left, 3), + (JoinKind::Right, 3), + (JoinKind::Full, 5), + (JoinKind::Semi, 1), + (JoinKind::Anti, 2), + (JoinKind::Cross, 9), + ] { + let output = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + input.clone() + } else { + schema(&[ + ("left", DataType::Int64, true), + ("right", DataType::Int64, true), + ]) + }; + let mut dag = PhysicalDag::default(); + for (id, rows) in [ + ( + 0, + vec![ + vec![Value::Int64(1)], + vec![Value::Int64(2)], + vec![Value::Null], + ], + ), + ( + 1, + vec![ + vec![Value::Int64(2)], + vec![Value::Int64(3)], + vec![Value::Null], + ], + ), + ] { + dag.add( + id, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new(input.clone(), rows).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + } + dag.add( + 2, + vec![0, 1], + Operator::relational_join( + input.clone(), + input.clone(), + kind.clone(), + &predicate, + output, + ) + .unwrap(), + ) + .unwrap(); + assert_eq!(run(&dag, 2, query()).len(), count, "{kind:?}"); + } +} From 92e5f2e512e040345ac4e96a8ee63c33c715b671 Mon Sep 17 00:00:00 2001 From: zz_y Date: Wed, 23 Sep 2026 22:48:32 +0000 Subject: [PATCH 05/90] feat: share window computations and typed binary execution --- .../asap-physical-operators/src/arithmetic.rs | 44 +++ .../src/dag/expressions.rs | 51 +++- crates/asap-physical-operators/src/dag/mod.rs | 2 + .../src/dag/operators.rs | 172 +++++++++++- .../src/dag/temporal.rs | 264 ++++++++++++++++++ .../asap-physical-operators/src/dag/values.rs | 23 ++ docs/design_docs/physical-operators.md | 10 +- 7 files changed, 557 insertions(+), 9 deletions(-) create mode 100644 crates/asap-physical-operators/src/dag/temporal.rs diff --git a/crates/asap-physical-operators/src/arithmetic.rs b/crates/asap-physical-operators/src/arithmetic.rs index bfc50694..5de95606 100644 --- a/crates/asap-physical-operators/src/arithmetic.rs +++ b/crates/asap-physical-operators/src/arithmetic.rs @@ -17,3 +17,47 @@ pub fn evaluate_float64_arithmetic( Atan2 => left.atan2(right), } } + +/// Execute the Planner binary contract after a deployment has resolved matching rows. +pub fn evaluate_binary( + operator: &planner_types::post_asap::BinaryOperator, + left: f64, + right: f64, +) -> Result { + use crate::dag::{values::Value, Error}; + use planner_types::pre_asap::{ArithmeticOpKind, BinaryOpKind, CompareOpKind}; + let invalid = + || Error::Invalid("unsupported binary operation or invalid checked-division domain".into()); + if operator.vector_match.is_some() { + return Err(invalid()); + } + if operator.checked_relative_division || operator.checked_finite_division { + if operator.kind != BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + || !left.is_finite() + || !right.is_finite() + || right == 0. + { + return Err(invalid()); + } + let value = left / right; + if !value.is_finite() || (operator.checked_relative_division && !value.is_normal()) { + return Err(invalid()); + } + return Ok(Value::Float64(value)); + } + Ok(match operator.kind { + BinaryOpKind::Arithmetic(ref op) => { + Value::Float64(evaluate_float64_arithmetic(op, left, right)) + } + BinaryOpKind::Compare(ref op) => Value::Bool(match op { + CompareOpKind::Eq => left == right, + CompareOpKind::Ne => left != right, + CompareOpKind::Lt => left < right, + CompareOpKind::Le => left <= right, + CompareOpKind::Gt => left > right, + CompareOpKind::Ge => left >= right, + _ => return Err(invalid()), + }), + _ => return Err(invalid()), + }) +} diff --git a/crates/asap-physical-operators/src/dag/expressions.rs b/crates/asap-physical-operators/src/dag/expressions.rs index 740dd323..571a9d42 100644 --- a/crates/asap-physical-operators/src/dag/expressions.rs +++ b/crates/asap-physical-operators/src/dag/expressions.rs @@ -420,12 +420,59 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { for part in parts { validate(part, schema)?; + if !matches!( + part.scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0, + DataType::Bool | DataType::Null + ) { + return Err(invalid()); + } } Ok(()) } - QueryExpr::Not(value) | QueryExpr::IsNull(value) | QueryExpr::IsNotNull(value) => { - validate(value, schema) + QueryExpr::Not(value) => { + validate(value, schema)?; + if !matches!( + value + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0, + DataType::Bool | DataType::Null + ) { + return Err(invalid()); + } + Ok(()) } + QueryExpr::IsNull(value) | QueryExpr::IsNotNull(value) => validate(value, schema), _ => Err(invalid()), } } + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn mixed_comparison_preserves_integer_precision_and_boundaries() { + assert_eq!( + integer_float_cmp(9_007_199_254_740_993, 9_007_199_254_740_992.0), + Some(Ordering::Greater) + ); + assert_eq!( + integer_float_cmp(i64::MAX, 9_223_372_036_854_775_808.0), + Some(Ordering::Less) + ); + assert_eq!( + integer_float_cmp(i64::MIN, -9_223_372_036_854_775_808.0), + Some(Ordering::Equal) + ); + assert_eq!(integer_float_cmp(-1, -1.5), Some(Ordering::Greater)); + assert_eq!(integer_float_cmp(1, 1.5), Some(Ordering::Less)); + assert_eq!(integer_float_cmp(0, f64::INFINITY), Some(Ordering::Less)); + assert_eq!( + integer_float_cmp(0, f64::NEG_INFINITY), + Some(Ordering::Greater) + ); + assert_eq!(integer_float_cmp(0, f64::NAN), None); + } +} diff --git a/crates/asap-physical-operators/src/dag/mod.rs b/crates/asap-physical-operators/src/dag/mod.rs index 5f19f31e..c00d5417 100644 --- a/crates/asap-physical-operators/src/dag/mod.rs +++ b/crates/asap-physical-operators/src/dag/mod.rs @@ -517,3 +517,5 @@ pub mod planner; pub mod batch_execution; pub mod expressions; + +mod temporal; diff --git a/crates/asap-physical-operators/src/dag/operators.rs b/crates/asap-physical-operators/src/dag/operators.rs index ad6c4901..1057e99e 100644 --- a/crates/asap-physical-operators/src/dag/operators.rs +++ b/crates/asap-physical-operators/src/dag/operators.rs @@ -42,6 +42,11 @@ fn result_field(name: &str, dtype: DataType, nullable: bool) -> SummaryField { #[derive(Clone, Debug)] pub enum Expression { + Binary { + operator: planner_types::post_asap::BinaryOperator, + left: Box, + right: Box, + }, Planner(Box), Column(usize), Literal { @@ -68,6 +73,38 @@ impl Expression { fn dtype(&self, input: &Schema) -> Result<(DataType, bool), Error> { use Expression::*; match self { + Binary { + operator, + left, + right, + } => { + use planner_types::pre_asap::{BinaryOpKind, CompareOpKind}; + let (a, n) = left.dtype(input)?; + let (b, m) = right.dtype(input)?; + if a != DataType::Float64 || b != a || operator.vector_match.is_some() { + return Err(invalid( + "binary expression requires resolved Float64 operands", + )); + } + if (operator.checked_relative_division || operator.checked_finite_division) + && operator.kind != BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + { + return Err(invalid("checked division contract on non-division")); + } + let dtype = match operator.kind { + BinaryOpKind::Arithmetic(_) => DataType::Float64, + BinaryOpKind::Compare( + CompareOpKind::Eq + | CompareOpKind::Ne + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge, + ) => DataType::Bool, + _ => return Err(invalid("unsupported binary operation")), + }; + Ok((dtype, n || m)) + } Planner(expression) => Ok(expression.dtype()), Column(i) => { let (t, n) = plain(input, *i)?; @@ -135,6 +172,21 @@ impl Expression { fn evaluate(&self, row: &[Value]) -> Result { use Expression::*; Ok(match self { + Binary { + operator, + left, + right, + } => { + let (a, b) = (left.evaluate(row)?, right.evaluate(row)?); + if matches!(a, Value::Null) || matches!(b, Value::Null) { + Value::Null + } else { + let (Value::Float64(a), Value::Float64(b)) = (a, b) else { + return Err(invalid("binary value schema mismatch")); + }; + crate::arithmetic::evaluate_binary(operator, a, b)? + } + } Planner(expression) => expression.evaluate(row)?, Column(i) => row[*i].clone(), Literal { value, .. } => value.clone(), @@ -191,9 +243,13 @@ impl Expression { } } fn ordered(dtype: &DataType) -> bool { + if let DataType::Map { key, value, .. } = dtype { + return ordered(key) && ordered(value); + } matches!( dtype, - DataType::Int64 + DataType::Null + | DataType::Int64 | DataType::Float64 | DataType::Utf8 | DataType::Bool @@ -255,6 +311,13 @@ enum Kind { keys: Vec, groups: Vec, }, + Window { + intent: Box>, + coordinate: usize, + value: usize, + groups: Vec, + window: Option<(i64, i64)>, + }, Aggregate { groups: Vec, measures: Vec, @@ -264,7 +327,7 @@ enum Kind { }, Join { kind: planner_types::pre_asap::JoinKind, - predicate: super::expressions::CompiledExpression, + predicate: Box, }, SummaryBuild { family: SummaryFamilyType, @@ -407,11 +470,11 @@ impl Operator { ) } Reduction::Min(i) | Reduction::Max(i) => { - let (t, _) = plain(&input, *i)?; + let (t, nullable) = plain(&input, *i)?; if !ordered(t) { return Err(invalid("ordered aggregate input required")); } - (t.clone(), true) + (t.clone(), nullable || groups.is_empty()) } }; fields.push(result_field(name, t, n)); @@ -425,6 +488,74 @@ impl Operator { output: schema(fields), }) } + /// Bind a Planner temporal or histogram intent to explicit columns and window. + pub fn window( + input: Schema, + intent: planner_types::pre_asap::AggIntent, + coordinate: usize, + value: usize, + groups: Vec, + window: Option<(i64, i64)>, + ) -> Result { + use planner_types::pre_asap::AggIntent; + validate_groups(&input, &groups)?; + let histogram = matches!(intent, AggIntent::HistogramQuantile { .. }); + if !matches!( + intent, + AggIntent::Rate + | AggIntent::Increase + | AggIntent::Count { .. } + | AggIntent::Sum { col: None } + | AggIntent::Avg { col: None } + | AggIntent::Min { col: None } + | AggIntent::Max { col: None } + | AggIntent::HistogramQuantile { .. } + ) { + return Err(invalid( + "unsupported temporal intent or unresolved value column", + )); + } + let coordinate_type = if histogram { + DataType::Float64 + } else { + DataType::Timestamp + }; + if plain(&input, coordinate)? != (&coordinate_type, false) + || plain(&input, value)? != (&DataType::Float64, false) + { + return Err(invalid("window coordinate/value schema mismatch")); + } + if (!histogram && !matches!(window, Some((start, end)) if start < end)) + || (histogram && window.is_some()) + { + return Err(invalid("invalid temporal window")); + } + let mut fields = groups + .iter() + .map(|i| input.fields[*i].clone()) + .collect::>(); + fields.push(result_field( + "value", + if matches!(intent, AggIntent::Count { .. }) { + DataType::Int64 + } else { + DataType::Float64 + }, + false, + )); + Ok(Self { + kind: Kind::Window { + intent: Box::new(intent), + coordinate, + value, + groups, + window, + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn semi_join( left: Schema, right: Schema, @@ -475,7 +606,10 @@ impl Operator { joined }; Self { - kind: Kind::Join { kind, predicate }, + kind: Kind::Join { + kind, + predicate: Box::new(predicate), + }, inputs: vec![left, right], output: schema(fields), } @@ -648,6 +782,7 @@ impl PhysicalOperator for Operator { Kind::Limit { .. } => "Limit", Kind::Sort { .. } => "Sort", Kind::Aggregate { .. } => "Aggregate", + Kind::Window { .. } => "WindowAggregate", Kind::SemiJoin { .. } => "SemiJoin", Kind::Join { .. } => "RelationalJoin", Kind::SummaryBuild { .. } => "SummaryAgg", @@ -923,6 +1058,13 @@ impl PhysicalOperator for Operator { Kind::Sort { keys, groups } => { let mut grouped = BTreeMap::>, Vec>>::new(); for row in rows { + for key in keys { + if matches!(row[key.column], Value::Map(_)) + && nested_nan(&row[key.column]) + { + return Err(invalid("NaN in collection sort key")); + } + } grouped .entry(group_key(&row, groups)?) .or_default() @@ -935,6 +1077,15 @@ impl PhysicalOperator for Operator { } result } + Kind::Window { + intent, + coordinate, + value, + groups, + window, + } => { + super::temporal::reduce(rows, intent, groups, *coordinate, *value, *window)? + } Kind::Aggregate { groups, measures } => { reduce(rows, groups, measures, &self.inputs[0])? } @@ -1275,3 +1426,14 @@ fn validate_readout( } Ok(()) } + +fn nested_nan(value: &Value) -> bool { + match value { + Value::Float64(value) => value.is_nan(), + Value::Map(values) => values + .iter() + .any(|(key, value)| nested_nan(key) || nested_nan(value)), + Value::List(values) | Value::Struct(values) => values.iter().any(nested_nan), + _ => false, + } +} diff --git a/crates/asap-physical-operators/src/dag/temporal.rs b/crates/asap-physical-operators/src/dag/temporal.rs new file mode 100644 index 00000000..37f97ada --- /dev/null +++ b/crates/asap-physical-operators/src/dag/temporal.rs @@ -0,0 +1,264 @@ +//! Windowed computations use Planner intents; deployments supply the input window. +use super::{ + values::{group_key, Value}, + Error, +}; +use planner_types::pre_asap::{AggIntent, ColumnRef}; +use std::collections::BTreeMap; + +pub(super) fn reduce( + rows: Vec>, + intent: &AggIntent, + groups: &[usize], + coordinate: usize, + value: usize, + window: Option<(i64, i64)>, +) -> Result>, Error> { + let mut grouped = + BTreeMap::>, (Vec, Vec<(f64, f64)>, Vec<(i64, f64)>)>::new(); + for row in rows { + let key = group_key(&row, groups)?; + let entry = grouped.entry(key).or_insert_with(|| { + ( + groups.iter().map(|i| row[*i].clone()).collect(), + vec![], + vec![], + ) + }); + let Value::Float64(v) = row[value] else { + return Err(Error::Invalid("window value must be Float64".into())); + }; + match row[coordinate] { + Value::Timestamp(t) => entry.2.push((t, v)), + Value::Float64(bound) => entry.1.push((bound, v)), + _ => return Err(Error::Invalid("invalid window coordinate".into())), + } + } + let mut output = Vec::new(); + for (_, (mut keys, buckets, mut points)) in grouped { + let result = if let AggIntent::HistogramQuantile { q } = intent { + Some(Value::Float64(bucket_quantile(*q, buckets))) + } else { + points.sort_by_key(|p| p.0); + let (start, end) = + window.ok_or_else(|| Error::Invalid("missing temporal window".into()))?; + if points.iter().any(|p| p.0 < start || p.0 > end) + || points.windows(2).any(|p| p[0].0 == p[1].0) + { + return Err(Error::Invalid( + "duplicate or out-of-window timestamp".into(), + )); + } + match intent { + AggIntent::Rate => rate(&points, start, end).map(Value::Float64), + AggIntent::Increase => rate(&points, start, end) + .map(|v| Value::Float64(v * (end as f64 - start as f64) / 1000.)), + AggIntent::Count { .. } => Some(Value::Int64( + i64::try_from(points.len()) + .map_err(|_| Error::Invalid("count overflow".into()))?, + )), + AggIntent::Sum { .. } => Some(Value::Float64(points.iter().map(|p| p.1).sum())), + AggIntent::Avg { .. } => Some(Value::Float64( + points.iter().map(|p| p.1).sum::() / points.len() as f64, + )), + AggIntent::Min { .. } => { + Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { + if a.is_nan() || p.1 < a { + p.1 + } else { + a + } + }))) + } + AggIntent::Max { .. } => { + Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { + if a.is_nan() || p.1 > a { + p.1 + } else { + a + } + }))) + } + _ => return Err(Error::Invalid("unsupported temporal intent".into())), + } + }; + if let Some(result) = result { + keys.push(result); + output.push(keys); + } + } + Ok(output) +} + +fn rate(points: &[(i64, f64)], start: i64, end: i64) -> Option { + if points.len() < 2 { + return None; + } + let (first_t, first) = points[0]; + let (last_t, last) = *points.last()?; + let span = (last_t as f64 - first_t as f64) / 1000.; + if span <= 0. { + return None; + } + let mut delta = last - first; + for pair in points.windows(2) { + if pair[1].1 < pair[0].1 { + delta += pair[0].1; + } + } + let average = span / (points.len() - 1) as f64; + let mut to_start = (first_t as f64 - start as f64) / 1000.; + let mut to_end = (end as f64 - last_t as f64) / 1000.; + if to_start >= average * 1.1 { + to_start = average / 2.; + } + // Apply the zero bound after the sparse-window half-interval cap. + if delta > 0. && first >= 0. { + to_start = to_start.min(span * first / delta); + } + if to_end >= average * 1.1 { + to_end = average / 2.; + } + Some(delta * (span + to_start + to_end) / span / ((end as f64 - start as f64) / 1000.)) +} + +fn bucket_quantile(q: f64, mut b: Vec<(f64, f64)>) -> f64 { + if q.is_nan() { + return f64::NAN; + } + if q < 0. { + return f64::NEG_INFINITY; + } + if q > 1. { + return f64::INFINITY; + } + b.retain(|p| !p.0.is_nan()); + b.sort_by(|a, b| a.0.total_cmp(&b.0)); + let mut buckets: Vec<(f64, f64)> = Vec::new(); + for p in b { + if let Some(last) = buckets.last_mut() { + if last.0 == p.0 { + last.1 += p.1; + continue; + } + } + buckets.push(p); + } + if buckets.len() < 2 || buckets.last().unwrap().0 != f64::INFINITY { + return f64::NAN; + } + let mut prev = buckets[0].1; + for p in buckets.iter_mut().skip(1) { + if p.1 < prev || (p.1 - prev).abs() <= 1e-12 * (p.1.abs() + prev.abs()) { + p.1 = prev; + } + prev = p.1; + } + let count = buckets.last().unwrap().1; + if count == 0. { + return f64::NAN; + } + let rank = q * count; + let idx = buckets[..buckets.len() - 1].partition_point(|p| p.1 < rank); + if idx == buckets.len() - 1 { + return buckets[idx - 1].0; + } + if idx == 0 && buckets[0].0 <= 0. { + return buckets[0].0; + } + let (start, base) = if idx == 0 { (0., 0.) } else { buckets[idx - 1] }; + let (end, upper) = buckets[idx]; + start + (end - start) * (rank - base) / (upper - base) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::dag::{ + batch_execution::evaluate_batch, operators::Operator, values::Batch, Limits, RunContext, + Scope, + }; + use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, + types::AccuracyTarget, + }; + use std::sync::Arc; + + // The same window operator must give the same answer in either engine phase. + #[test] + fn temporal_windows_execute_in_both_phases_and_count_is_integer() { + let schema = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "time".into(), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), + nullable: false, + }, + SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }, + ], + time_index: Some(0), + }); + for scope in [ + Scope::Query { + evaluation_time_ms: 2000, + revision: 1, + }, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }, + ] { + for (intent, expected) in [ + (AggIntent::Rate, Value::Float64(2.)), + (AggIntent::Increase, Value::Float64(4.)), + ( + AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }, + Value::Int64(3), + ), + ] { + let batch = Batch::try_new( + schema.clone(), + vec![ + vec![Value::Timestamp(0), Value::Float64(2.)], + vec![Value::Timestamp(1000), Value::Float64(4.)], + vec![Value::Timestamp(2000), Value::Float64(2.)], + ], + ) + .unwrap(); + let operator = + Operator::window(schema.clone(), intent, 0, 1, vec![], Some((0, 2000))) + .unwrap(); + let result = evaluate_batch( + batch, + vec![operator], + RunContext::new(scope.clone(), Limits::default()).unwrap(), + ) + .unwrap(); + assert_eq!( + format!("{:?}", result[0].rows()[0][0]), + format!("{expected:?}") + ); + } + } + assert!(Operator::window(schema, AggIntent::Rate, 0, 1, vec![], Some((1, 1))).is_err()); + } + + // Histogram interpolation requires an infinite terminal bucket and coalesces duplicates. + #[test] + fn histogram_boundaries_and_duplicate_buckets() { + assert_eq!( + bucket_quantile(0.5, vec![(1., 1.), (1., 1.), (2., 4.), (f64::INFINITY, 4.)]), + 1. + ); + assert!(bucket_quantile(0.5, vec![(1., 2.), (2., 4.)]).is_nan()); + assert_eq!(bucket_quantile(-0.1, vec![]), f64::NEG_INFINITY); + } +} diff --git a/crates/asap-physical-operators/src/dag/values.rs b/crates/asap-physical-operators/src/dag/values.rs index 82b87314..74c49b95 100644 --- a/crates/asap-physical-operators/src/dag/values.rs +++ b/crates/asap-physical-operators/src/dag/values.rs @@ -164,6 +164,29 @@ impl Value { (Self::Utf8(a), Self::Utf8(b)) => a.cmp(b), (Self::Bool(a), Self::Bool(b)) => a.cmp(b), (Self::Date(a), Self::Date(b)) => a.cmp(b), + (Self::Map(left), Self::Map(right)) => { + let mut result = Ordering::Equal; + for ((lk, lv), (rk, rv)) in left.iter().zip(right.iter()) { + result = lk.compare(rk)?; + if result != Ordering::Equal { + break; + } + result = match (lv, rv) { + (Self::Null, Self::Null) => Ordering::Equal, + (Self::Null, _) => Ordering::Greater, + (_, Self::Null) => Ordering::Less, + _ => lv.compare(rv)?, + }; + if result != Ordering::Equal { + break; + } + } + if result == Ordering::Equal { + left.len().cmp(&right.len()) + } else { + result + } + } _ => { return Err(Error::Operator( "values do not have a supported common ordering".into(), diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index ac1b28f6..b1b55b97 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -62,9 +62,15 @@ operators currently have no spill implementation. ## Operator coverage Native operations include scalar sources, typed Project and Filter, arithmetic -and boolean expressions, exact grouped aggregation, semi-join, grouped Sort and +and boolean expressions, exact grouped aggregation, relational joins (including semi-join), grouped Sort and Limit, Union, vector-to-scalar conversion, and summary construction, merge and -readout. Grouped TopK composes Sort and Limit within each group; candidate +readout. Window operators consume Planner aggregate intents for Rate, Increase, +Sum, Avg, Min, Max, Count and histogram quantiles. Deployments supply window +boundaries and bound columns; the computation is identical in either phase. +Count outputs Int64. Binary expressions use Planner arithmetic/comparison kinds +and enforce its checked-division domains. + +Grouped TopK composes Sort and Limit within each group; candidate completeness is an earlier pruning obligation. Values retain Planner types and nullability. Native summary batches currently From cb8bede158d792449d5ace3dd1bddf4d73ad02f3 Mon Sep 17 00:00:00 2001 From: zz_y Date: Wed, 23 Sep 2026 22:51:45 +0000 Subject: [PATCH 06/90] refactor: own stored-summary decoding and readout in shared library --- crates/asap-physical-operators/src/lib.rs | 2 + .../src/stored_state/decoders.rs | 366 +++++ .../src/stored_state/delta_apply.rs | 1256 +++++++++++++++++ .../src/stored_state/mod.rs | 19 + .../src/stored_state/readout.rs | 114 ++ 5 files changed, 1757 insertions(+) create mode 100644 crates/asap-physical-operators/src/stored_state/decoders.rs create mode 100644 crates/asap-physical-operators/src/stored_state/delta_apply.rs create mode 100644 crates/asap-physical-operators/src/stored_state/mod.rs create mode 100644 crates/asap-physical-operators/src/stored_state/readout.rs diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs index 22456e3d..3896eb7a 100644 --- a/crates/asap-physical-operators/src/lib.rs +++ b/crates/asap-physical-operators/src/lib.rs @@ -21,3 +21,5 @@ pub mod factory; pub use planner_types as planner; pub mod dag; + +pub mod stored_state; diff --git a/crates/asap-physical-operators/src/stored_state/decoders.rs b/crates/asap-physical-operators/src/stored_state/decoders.rs new file mode 100644 index 00000000..99bde575 --- /dev/null +++ b/crates/asap-physical-operators/src/stored_state/decoders.rs @@ -0,0 +1,366 @@ +//! Shared sketch state reconstruction and decoding. +use asap_sketchlib::CountMinSketch; +use asap_sketchlib::CountMinSketchDelta; +use asap_sketchlib::CountMinSketchWithHeap; +use asap_sketchlib::CountSketch; +use asap_sketchlib::CountSketchDelta; +use asap_sketchlib::CountSketchWithHeap; +use asap_sketchlib::CsHeapItem; +use asap_sketchlib::MessagePackCodec; + +use crate::accumulators::count_min_sketch_with_heap_accumulator::CountMinSketchWithHeapAccumulator; + +/// Decode a `CountMinSketch` from the modified-OTLP wire bytes. +/// MSGPACK path round-trips `CountMinSketch::deserialize_msgpack`; +/// PROTO path decodes a `SketchEnvelope{count_min: CountMinState}` +/// (or bare `CountMinState`) and re-projects to a flat matrix. Mirrors +/// `precompute_operators::count_min_sketch_accumulator::from_sketchlib_proto_bytes`. +pub fn decode_cms_from_proto(buffer: &[u8]) -> Result { + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, CountMinState, CounterType, SketchEnvelope, + }; + use prost::Message; + + let state = match SketchEnvelope::decode(buffer) { + Ok(env) => match env.sketch_state { + Some(sketch_envelope::SketchState::CountMin(st)) => st, + Some(_) => return Err("SketchEnvelope contains non-CountMin sketch".to_string()), + None => { + CountMinState::decode(buffer).map_err(|e| format!("decode CountMinState: {e}"))? + } + }, + Err(_) => { + CountMinState::decode(buffer).map_err(|e| format!("decode CountMinState: {e}"))? + } + }; + let rows = state.rows as usize; + let cols = state.cols as usize; + if rows == 0 || cols == 0 { + return Err(format!( + "CountMinState has zero dims (rows={rows}, cols={cols})" + )); + } + let expected_len = rows * cols; + let counter_type = CounterType::try_from(state.counter_type) + .map_err(|_| format!("CountMinState unknown counter_type {}", state.counter_type))?; + let flat: Vec = match counter_type { + CounterType::Int32 | CounterType::Int64 => { + if state.counts_int.len() != expected_len { + return Err(format!( + "CountMinState counts_int has {} entries, expected {}", + state.counts_int.len(), + expected_len + )); + } + state.counts_int.iter().map(|&v| v as f64).collect() + } + CounterType::Float64 => { + if state.counts_float.len() != expected_len { + return Err(format!( + "CountMinState counts_float has {} entries, expected {}", + state.counts_float.len(), + expected_len + )); + } + state.counts_float.clone() + } + other => { + return Err(format!( + "CountMinState counter_type {other:?} not yet supported in reducer" + )); + } + }; + let mut matrix = Vec::with_capacity(rows); + for r in 0..rows { + let start = r * cols; + matrix.push(flat[start..start + cols].to_vec()); + } + Ok(CountMinSketch::from_legacy_matrix(matrix, rows, cols)) +} + +/// Decode a `CountMinSketch` from msgpack bytes (sketch-core wire +/// format). Mirrors +/// `CountMinSketchAccumulator::from_msgpack_bytes`. +pub fn decode_cms_from_msgpack(buffer: &[u8]) -> Result { + CountMinSketch::from_msgpack(buffer) + .map_err(|e| format!("deserialize CountMinSketch msgpack: {e}")) +} + +/// Decode a `CountSketch` from the modified-OTLP proto wire bytes. +/// Mirrors +/// `precompute_operators::count_sketch_accumulator::from_sketchlib_proto_bytes`. +pub fn decode_cs_from_proto(buffer: &[u8]) -> Result { + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, CountSketchState, CounterType, SketchEnvelope, + }; + use prost::Message; + + let state = match SketchEnvelope::decode(buffer) { + Ok(env) => match env.sketch_state { + Some(sketch_envelope::SketchState::CountSketch(st)) => st, + Some(_) => return Err("SketchEnvelope contains non-CountSketch sketch".to_string()), + None => CountSketchState::decode(buffer) + .map_err(|e| format!("decode CountSketchState: {e}"))?, + }, + Err(_) => { + CountSketchState::decode(buffer).map_err(|e| format!("decode CountSketchState: {e}"))? + } + }; + let rows = state.rows as usize; + let cols = state.cols as usize; + if rows == 0 || cols == 0 { + return Err(format!( + "CountSketchState has zero dims (rows={rows}, cols={cols})" + )); + } + let expected_len = rows * cols; + let counter_type = CounterType::try_from(state.counter_type).map_err(|_| { + format!( + "CountSketchState unknown counter_type {}", + state.counter_type + ) + })?; + let flat: Vec = match counter_type { + CounterType::Int32 | CounterType::Int64 => { + if state.counts_int.len() != expected_len { + return Err(format!( + "CountSketchState counts_int has {} entries, expected {}", + state.counts_int.len(), + expected_len + )); + } + state.counts_int.iter().map(|&v| v as f64).collect() + } + CounterType::Float64 => { + if state.counts_float.len() != expected_len { + return Err(format!( + "CountSketchState counts_float has {} entries, expected {}", + state.counts_float.len(), + expected_len + )); + } + state.counts_float.clone() + } + other => { + return Err(format!( + "CountSketchState counter_type {other:?} not yet supported in reducer" + )); + } + }; + let mut matrix = Vec::with_capacity(rows); + for r in 0..rows { + let start = r * cols; + matrix.push(flat[start..start + cols].to_vec()); + } + Ok(CountSketch::from_legacy_matrix(matrix, rows, cols)) +} + +/// Decode a `CountSketch` from msgpack bytes (sketch-core wire format). +pub fn decode_cs_from_msgpack(buffer: &[u8]) -> Result { + CountSketch::from_msgpack(buffer).map_err(|e| format!("deserialize CountSketch msgpack: {e}")) +} + +/// Decode a `CountMinSketchWithHeap` from msgpack bytes — the OTLP +/// `CountMinSketch` wire bytes when the gateway/precompute layer +/// marked the sid as CmsWithHeap (heap embedded in the +/// `CountMinSketchWithHeapSerialized` outer wrapper). Delegates to +/// `asap_sketchlib::CountMinSketchWithHeap::deserialize_msgpack`. +pub fn decode_cms_with_heap_from_msgpack(buffer: &[u8]) -> Result { + CountMinSketchWithHeap::from_msgpack(buffer) + .map_err(|e| format!("deserialize CountMinSketchWithHeap msgpack: {e}")) +} + +/// Decode a `CountSketchWithHeap` (median-estimator, Count Sketch family) +/// from msgpack bytes. Distinct wire type from `CountMinSketchWithHeap` +/// (min-estimator, Count-Min family) even though both are heap-bearing +/// frequency sketches — see `asap_sketchlib::CountSketchWithHeap`. +/// Delegates to `asap_sketchlib::CountSketchWithHeap::from_msgpack`. +pub fn decode_cs_with_heap_from_msgpack(buffer: &[u8]) -> Result { + CountSketchWithHeap::from_msgpack(buffer) + .map_err(|e| format!("deserialize CountSketchWithHeap msgpack: {e}")) +} + +// --------------------------------------------------------------------------- +// Delta decoders. Under the per-window-reset (PWR) contract +// (`asap-precompute-go/window.go`: a delta is that window's own state +// applied onto a freshly-reset per-series sketch), each stored *Delta +// frame reconstructs into the FULL window state when applied onto an +// EMPTY base of the frame's declared dimensions. The reducer's +// `FrequencyEstimate` / `FrequencyTopk` paths are per-window evaluations, +// so "empty + apply(this window's delta)" yields exactly the window's +// matrix/heap — no cross-window stitching needed (mirrors how the ingest +// accumulators reset_to_empty per window before applying). +// +// The proto path reuses the PUBLIC `asap_sketchlib::{CountSketch, +// CountMinSketch}::apply_delta`; the proto `*Delta` message is decoded via +// `asap_sketchlib::proto::sketchlib::{CountSketchDelta, CountMinDelta}`, +// exactly as `precompute_operators::{count_sketch, +// count_min_sketch}_accumulator::apply_proto_delta_bytes` does. +// --------------------------------------------------------------------------- + +/// Decode a `CountMinSketch` PROTO_DELTA frame into a FULL sketch by +/// applying the sparse cell delta onto an empty base of the frame's +/// declared dimensions. Mirrors +/// `precompute_operators::count_min_sketch_accumulator::apply_proto_delta_bytes`. +pub fn decode_cms_from_proto_delta(buffer: &[u8]) -> Result { + use asap_sketchlib::proto::sketchlib::CountMinDelta as PbDelta; + use prost::Message; + + let pb = PbDelta::decode(buffer).map_err(|e| format!("decode CountMinDelta: {e}"))?; + if pb.cell_rows.len() != pb.cell_cols.len() || pb.cell_rows.len() != pb.d_counts.len() { + return Err(format!( + "CountMinDelta packed-array length mismatch: cell_rows={}, cell_cols={}, d_counts={}", + pb.cell_rows.len(), + pb.cell_cols.len(), + pb.d_counts.len() + )); + } + let rows = pb.rows as usize; + let cols = pb.cols as usize; + if rows == 0 || cols == 0 { + return Err(format!( + "CountMinDelta has zero dims (rows={rows}, cols={cols})" + )); + } + let cells = pb + .cell_rows + .iter() + .zip(pb.cell_cols.iter()) + .zip(pb.d_counts.iter()) + .map(|((r, c), dc)| (*r, *c, *dc)) + .collect(); + // hh_keys is parsed off the wire by the precompute accumulator but + // intentionally dropped (the vendored Go proto bindings don't yet + // populate it); match that to keep behavior identical. + let delta = CountMinSketchDelta { + rows: pb.rows, + cols: pb.cols, + cells, + l1: pb.l1, + l2: pb.l2, + hh_keys: Vec::new(), + }; + let mut cms = CountMinSketch::from_legacy_matrix(vec![vec![0.0; cols]; rows], rows, cols); + cms.apply_delta(&delta) + .map_err(|e| format!("apply CountMinDelta onto empty base: {e}"))?; + Ok(cms) +} + +/// Decode a `CountSketch` PROTO_DELTA frame into a FULL sketch by applying +/// the sparse cell delta onto an empty base of the frame's declared +/// dimensions. Mirrors +/// `precompute_operators::count_sketch_accumulator::apply_proto_delta_bytes`. +pub fn decode_cs_from_proto_delta(buffer: &[u8]) -> Result { + use asap_sketchlib::proto::sketchlib::CountSketchDelta as PbDelta; + use prost::Message; + + let pb = PbDelta::decode(buffer).map_err(|e| format!("decode CountSketchDelta: {e}"))?; + if pb.cell_rows.len() != pb.cell_cols.len() || pb.cell_rows.len() != pb.d_counts.len() { + return Err(format!( + "CountSketchDelta packed-array length mismatch: cell_rows={}, cell_cols={}, d_counts={}", + pb.cell_rows.len(), + pb.cell_cols.len(), + pb.d_counts.len() + )); + } + let rows = pb.rows as usize; + let cols = pb.cols as usize; + if rows == 0 || cols == 0 { + return Err(format!( + "CountSketchDelta has zero dims (rows={rows}, cols={cols})" + )); + } + let cells = pb + .cell_rows + .iter() + .zip(pb.cell_cols.iter()) + .zip(pb.d_counts.iter()) + .map(|((r, c), dc)| (*r, *c, *dc)) + .collect(); + let delta = CountSketchDelta { + rows: pb.rows, + cols: pb.cols, + cells, + l2: pb.l2, + hh_keys: Vec::new(), + }; + let mut cs = CountSketch::from_legacy_matrix(vec![vec![0.0; cols]; rows], rows, cols); + cs.apply_delta(&delta) + .map_err(|e| format!("apply CountSketchDelta onto empty base: {e}"))?; + Ok(cs) +} + +/// Decode a heap-bearing CountSketch MSGPACK_DELTA frame into a FULL +/// `CountMinSketchWithHeap` by applying the sparse matrix delta + full +/// heap onto an empty base of the frame's declared dimensions. This +/// REUSES the ingest-side delta-heap apply logic +/// (`CountMinSketchWithHeapAccumulator::from_msgpack_heap_delta_bytes` → +/// `apply_msgpack_heap_delta_bytes`), which decodes the frame generically +/// with `rmp_serde` — no `asap_sketchlib` delta API is added. +pub fn decode_cms_with_heap_from_msgpack_delta( + buffer: &[u8], +) -> Result { + let acc = CountMinSketchWithHeapAccumulator::from_msgpack_heap_delta_bytes(buffer) + .map_err(|e| format!("reconstruct CountMinSketchWithHeap from delta: {e}"))?; + Ok(acc.inner) +} + +/// Decode a heap-bearing CountSketch (median-estimator) MSGPACK_DELTA frame +/// into a FULL `asap_sketchlib::CountSketchWithHeap` by applying the sparse +/// matrix delta + full heap onto an empty base of the frame's declared +/// dimensions. Same DELTA-HEAP wire shape as the CmsWithHeap delta frame +/// (see `HeapDeltaWire`/`MatrixDeltaWire` in +/// `count_min_sketch_with_heap_accumulator.rs`), decoded here directly +/// with `rmp_serde` since there is no CountSketchWithHeap ingest +/// accumulator to delegate to. No `asap_sketchlib` delta API needed — the +/// public `from_legacy_matrix` rebuilds both the matrix and heap. +pub fn decode_cs_with_heap_from_msgpack_delta( + buffer: &[u8], +) -> Result { + #[derive(serde::Deserialize)] + struct HeapDeltaWire { + is_delta: bool, + matrix_delta: MatrixDeltaWire, + topk_heap: Vec<(String, f64)>, + heap_size: u64, + } + #[derive(serde::Deserialize)] + struct MatrixDeltaWire { + rows: u32, + cols: u32, + cells: Vec<(u32, u32, i64)>, + } + + let wire: HeapDeltaWire = rmp_serde::from_slice(buffer) + .map_err(|e| format!("decode CountSketchWithHeap delta msgpack: {e}"))?; + if !wire.is_delta { + return Err("CountSketchWithHeap delta frame has is_delta=false".to_string()); + } + let rows = wire.matrix_delta.rows as usize; + let cols = wire.matrix_delta.cols as usize; + if rows == 0 || cols == 0 { + return Err(format!( + "CountSketchWithHeap delta frame has zero dims (rows={rows}, cols={cols})" + )); + } + let mut matrix = vec![vec![0.0; cols]; rows]; + for (r, c, dc) in &wire.matrix_delta.cells { + let (r, c) = (*r as usize, *c as usize); + if r >= rows || c >= cols { + continue; + } + matrix[r][c] += *dc as f64; + } + let heap: Vec = wire + .topk_heap + .into_iter() + .map(|(key, value)| CsHeapItem { key, value }) + .collect(); + Ok(CountSketchWithHeap::from_legacy_matrix( + matrix, + heap, + rows, + cols, + wire.heap_size as usize, + )) +} diff --git a/crates/asap-physical-operators/src/stored_state/delta_apply.rs b/crates/asap-physical-operators/src/stored_state/delta_apply.rs new file mode 100644 index 00000000..dd9ec4d3 --- /dev/null +++ b/crates/asap-physical-operators/src/stored_state/delta_apply.rs @@ -0,0 +1,1256 @@ +//! Shared sketch state reconstruction and decoding. +use asap_sketchlib::CountMinSketch; +use asap_sketchlib::CountMinSketchWithHeap; +use asap_sketchlib::CountSketch; +use asap_sketchlib::CountSketchWithHeap; +use asap_sketchlib::DdSketch; +use asap_sketchlib::HllSketch; +use asap_sketchlib::HllVariant; +use asap_sketchlib::KllSketch; +use asap_sketchlib::MessagePackCodec; + +use super::decoders::{ + decode_cms_from_msgpack, decode_cms_from_proto, decode_cms_from_proto_delta, + decode_cms_with_heap_from_msgpack, decode_cms_with_heap_from_msgpack_delta, + decode_cs_from_msgpack, decode_cs_from_proto, decode_cs_from_proto_delta, + decode_cs_with_heap_from_msgpack, decode_cs_with_heap_from_msgpack_delta, +}; +use super::{SketchEncoding, SketchSampleState}; + +/// Which sketch family a candidate is, and the parameters needed to +/// *bootstrap an empty state* — required by the per-window-reset (PWR) +/// delta model where a window's FIRST frame is a delta-from-empty (no +/// carry-in Full). Most families' deltas embed their own params in the +/// wire fragment (decoded independently, then merged in — see +/// `SummaryState::apply_delta_bytes`); HLL register deltas and DD's +/// bucket-index deltas are applied onto a pre-sized structure instead, +/// so those two need the params known up front to allocate it. +#[derive(Debug, Clone, Copy)] +pub enum DeltaSketchKind { + UnivMon { + heap_size: u32, + sketch_rows: u32, + sketch_cols: u32, + layers: u8, + }, + DDSketch { + alpha: f64, + }, + Hll { + precision: u32, + }, + Kll { + k: u32, + }, + Cms { + rows: usize, + cols: usize, + }, + CountSketch { + rows: usize, + cols: usize, + }, + /// `CmsWithHeap` wraps `asap_sketchlib::CountMinSketchWithHeap` + /// (min-over-rows estimator) and `CountSketchWithHeap` wraps the + /// distinct `asap_sketchlib::CountSketchWithHeap` (median-of-signed-rows + /// estimator) -- different algorithms that happen to share a storage + /// shape. Kept as two variants (not one shared `Heap`) so + /// `merge_same_family` rejects merging one into the other the same + /// way it already rejects e.g. merging a `Cms` into a `Kll`; now the + /// type system enforces it too, since the two variants hold different + /// Rust types. + CmsWithHeap { + rows: usize, + cols: usize, + heap_size: usize, + }, + CountSketchWithHeap { + rows: usize, + cols: usize, + heap_size: usize, + }, +} + +impl DeltaSketchKind { + /// Construct an EMPTY state for this kind, used to seed a new window + /// when its first frame is a delta-from-empty (PWR). A delta applied + /// onto this empty base reconstructs exactly that window's state + /// (delta-from-empty ⊕ empty = window state). + fn bootstrap_empty(&self) -> SummaryState { + match self { + Self::UnivMon { + heap_size, + sketch_rows, + sketch_cols, + layers, + } => SummaryState::UnivMon( + crate::accumulators::univmon_accumulator::UnivMonAccumulator::new( + *heap_size as usize, + *sketch_rows as usize, + *sketch_cols as usize, + *layers as usize, + ) + .expect("validated UnivMon catalog dimensions"), + ), + DeltaSketchKind::DDSketch { alpha } => SummaryState::Dd(DdSketch::new(*alpha)), + DeltaSketchKind::Kll { k } => SummaryState::Kll(KllSketch::new(*k as u16)), + DeltaSketchKind::Hll { precision } => { + SummaryState::Hll(HllSketch::new(HllVariant::Regular, *precision)) + } + DeltaSketchKind::Cms { rows, cols } => { + SummaryState::Cms(CountMinSketch::new(*rows, *cols)) + } + DeltaSketchKind::CountSketch { rows, cols } => { + SummaryState::CountSketch(CountSketch::new(*rows, *cols)) + } + DeltaSketchKind::CmsWithHeap { + rows, + cols, + heap_size, + } => SummaryState::CmsWithHeap(CountMinSketchWithHeap::new(*rows, *cols, *heap_size)), + DeltaSketchKind::CountSketchWithHeap { + rows, + cols, + heap_size, + } => SummaryState::CountSketchWithHeap(CountSketchWithHeap::new( + *rows, *cols, *heap_size, + )), + } + } +} + +/// Try to decode a "full" sketch from the bytes (used by both +/// per-window and cumulative modes when the encoding is `*Full`). +fn decode_full( + kind: &DeltaSketchKind, + bytes: &[u8], + encoding: SketchEncoding, +) -> Result { + match (kind, encoding) { + ( + DeltaSketchKind::UnivMon { + heap_size, + sketch_rows, + sketch_cols, + layers, + }, + SketchEncoding::MsgpackFull, + ) => { + let state = + crate::accumulators::univmon_accumulator::UnivMonAccumulator::from_bytes(bytes) + .map_err(|e| e.to_string())?; + if state.dimensions() + != ( + *heap_size as usize, + *sketch_rows as usize, + *sketch_cols as usize, + *layers as usize, + ) + { + return Err("UnivMon payload dimensions differ from installed catalog".into()); + } + Ok(SummaryState::UnivMon(state)) + } + (DeltaSketchKind::DDSketch { .. }, SketchEncoding::ProtoFull) => { + let sk = dd_from_proto(bytes)?; + Ok(SummaryState::Dd(sk)) + } + (DeltaSketchKind::DDSketch { .. }, SketchEncoding::MsgpackFull) => { + let sk = DdSketch::from_msgpack(bytes) + .map_err(|e| format!("deserialize DDSketch msgpack: {e}"))?; + Ok(SummaryState::Dd(sk)) + } + (DeltaSketchKind::Hll { .. }, SketchEncoding::ProtoFull) => { + let sk = hll_from_proto(bytes)?; + Ok(SummaryState::Hll(sk)) + } + (DeltaSketchKind::Hll { .. }, SketchEncoding::MsgpackFull) => { + let sk = HllSketch::from_msgpack(bytes) + .map_err(|e| format!("deserialize HllSketch msgpack: {e}"))?; + Ok(SummaryState::Hll(sk)) + } + (DeltaSketchKind::Kll { .. }, SketchEncoding::ProtoFull) => { + let sk = kll_from_proto(bytes)?; + Ok(SummaryState::Kll(sk)) + } + (DeltaSketchKind::Kll { .. }, SketchEncoding::MsgpackFull) => { + let sk = KllSketch::from_msgpack(bytes) + .map_err(|e| format!("deserialize KllSketch msgpack: {e}"))?; + Ok(SummaryState::Kll(sk)) + } + (DeltaSketchKind::Cms { .. }, SketchEncoding::ProtoFull) => { + Ok(SummaryState::Cms(decode_cms_from_proto(bytes)?)) + } + (DeltaSketchKind::Cms { .. }, SketchEncoding::MsgpackFull) => { + Ok(SummaryState::Cms(decode_cms_from_msgpack(bytes)?)) + } + (DeltaSketchKind::CountSketch { .. }, SketchEncoding::ProtoFull) => { + Ok(SummaryState::CountSketch(decode_cs_from_proto(bytes)?)) + } + (DeltaSketchKind::CountSketch { .. }, SketchEncoding::MsgpackFull) => { + Ok(SummaryState::CountSketch(decode_cs_from_msgpack(bytes)?)) + } + // The heap-bearing wire format is msgpack-only in this + // deployment; `decode_cms_with_heap_from_msgpack` is the same + // "Full" decoder the reducer's existing per-frame dispatch falls + // through to for any non-MsgpackDelta encoding. + ( + DeltaSketchKind::CmsWithHeap { .. }, + SketchEncoding::ProtoFull | SketchEncoding::MsgpackFull, + ) => Ok(SummaryState::CmsWithHeap( + decode_cms_with_heap_from_msgpack(bytes)?, + )), + ( + DeltaSketchKind::CountSketchWithHeap { .. }, + SketchEncoding::ProtoFull | SketchEncoding::MsgpackFull, + ) => Ok(SummaryState::CountSketchWithHeap( + decode_cs_with_heap_from_msgpack(bytes)?, + )), + (_, e) => Err(format!("decode_full called with non-Full encoding {e:?}")), + } +} + +/// The reconstructed state one candidate sid contributes — either +/// folded across a window (or several) via delta application, or merged +/// in from another sid's own reconstruction. +pub enum SummaryState { + UnivMon(crate::accumulators::univmon_accumulator::UnivMonAccumulator), + Dd(DdSketch), + Hll(HllSketch), + Kll(KllSketch), + Cms(CountMinSketch), + CountSketch(CountSketch), + /// See `DeltaSketchKind::CmsWithHeap`/`CountSketchWithHeap` for why + /// these are two variants holding two different sketchlib types. + CmsWithHeap(CountMinSketchWithHeap), + CountSketchWithHeap(CountSketchWithHeap), +} + +impl SummaryState { + /// Apply a delta-encoded payload from a window sample. For DD / KLL, + /// the delta is interpreted as a "mergeable fragment" decoded + /// through the same full-state decoder and merged into the + /// rolling state. For HLL, the wire delta is a sparse register + /// update applied via the sketch's `apply_delta`. + /// + /// On encoding mismatch (e.g. trying to apply an HllDelta to a + /// DDSketch rolling state) returns Err. + pub fn apply_delta_bytes( + &mut self, + bytes: &[u8], + encoding: SketchEncoding, + ) -> Result<(), String> { + if !matches!( + encoding, + SketchEncoding::ProtoDelta | SketchEncoding::MsgpackDelta + ) { + return Err(format!( + "apply_delta_bytes called with non-Delta encoding {encoding:?}" + )); + } + match self { + SummaryState::UnivMon(_) => Err("UnivMon requires full pane snapshots".into()), + SummaryState::Dd(sk) => { + match encoding { + // PROTO_DELTA: dispatch on the payload SHAPE, mirroring the + // supported DDSketch frame decoder, which tries the + // full-envelope decode first, then falls back to + // the bucket-delta proto. Two wire shapes can arrive on the + // ProtoDelta channel: + // + // 1. `SketchEnvelope{DdSketchState}` — a full-state + // fragment, mergeable via `DdSketch::merge`. (The edge + // sends this when `compute_delta_against` hits the + // empty-current / undecodable-prior fallback and ships + // a full snapshot tagged as a delta.) + // 2. `DDSketchDelta { buckets: [{index, d_count}] }` — a + // bucket-index delta proto, applied additively. This is + // the COMMON delta_transmission frame the edge emits + // under per-window-reset (`compute_delta(&empty)`). + // + // Before this fix the reducer decoded ONLY shape (1) via + // `decode_full`. A real shape-(2) frame failed with a wire- + // type mismatch on field 1 (delta field 1 = repeated + // submessage; state field 1 = `double alpha`) → the whole + // `quantile_over_time` returned `No result` for every + // delta_transmission DDSketch stream. We wrap the rolling + // `DdSketch` in a transient accumulator so the bucket-delta + // apply lands on `sk` in place. + SketchEncoding::ProtoDelta => { + // Shape (1): full envelope fragment → merge. Try this + // first (cheap decode attempt; a bucket-delta proto + // fails it on the field-1 wire-type mismatch). + if let Ok(SummaryState::Dd(other)) = decode_full( + &DeltaSketchKind::DDSketch { alpha: 0.0 }, + bytes, + SketchEncoding::ProtoFull, + ) { + sk.merge(&other) + .map_err(|e| format!("merge DDSketch delta envelope: {e}"))?; + return Ok(()); + } + // Shape (2): bucket-delta proto → additive apply via the + // SAME decoder the ingest delta path uses. + use crate::accumulators::dd_sketch_accumulator::DDSketchAccumulator; + let mut acc = DDSketchAccumulator { + inner: std::mem::replace(sk, DdSketch::new(sk.alpha)), + sample_p: 1.0, + }; + let res = acc.apply_proto_delta_bytes(bytes); + *sk = acc.inner; + res.map_err(|e| format!("apply DDSketch proto bucket-delta: {e}"))?; + Ok(()) + } + // MSGPACK_DELTA: a serialized full-sketch fragment, mergeable + // via the full-state decoder. Kept for completeness — the + // edge wires PROTO_DELTA for DDSketch today. + SketchEncoding::MsgpackDelta => { + let other = match decode_full( + &DeltaSketchKind::DDSketch { alpha: 0.0 }, + bytes, + SketchEncoding::MsgpackFull, + ) { + Ok(SummaryState::Dd(s)) => s, + Ok(_) => { + return Err( + "decode_full(DDSketch) returned non-DDSketch state".to_string() + ) + } + Err(e) => return Err(e), + }; + sk.merge(&other) + .map_err(|e| format!("merge DDSketch delta: {e}"))?; + Ok(()) + } + _ => unreachable!(), + } + } + SummaryState::Hll(sk) => { + // HLL has a true sparse register delta in the proto + // wire format. Use the same path the precompute + // accumulator uses (`apply_proto_delta_bytes`-style). + if encoding == SketchEncoding::ProtoDelta { + apply_hll_proto_delta(sk, bytes) + } else { + // MsgpackDelta for HLL isn't a sparse encoding; + // it's a serialized HllSketch fragment, mergeable + // via `HllSketch::merge`. + let other = HllSketch::from_msgpack(bytes) + .map_err(|e| format!("deserialize HllSketch (delta-as-msgpack): {e}"))?; + sk.merge(&other) + .map_err(|e| format!("merge HLL delta: {e}"))?; + Ok(()) + } + } + SummaryState::Kll(sk) => { + let full_enc = match encoding { + SketchEncoding::ProtoDelta => SketchEncoding::ProtoFull, + SketchEncoding::MsgpackDelta => SketchEncoding::MsgpackFull, + _ => unreachable!(), + }; + let other = match decode_full(&DeltaSketchKind::Kll { k: 0 }, bytes, full_enc) { + Ok(SummaryState::Kll(s)) => s, + Ok(_) => return Err("decode_full(Kll) returned non-Kll state".to_string()), + Err(e) => return Err(e), + }; + sk.merge(&other) + .map_err(|e| format!("merge KLL delta: {e}"))?; + Ok(()) + } + // CMS/CountSketch/Heap have no true sparse in-place delta + // (unlike DD's bucket-index proto or HLL's register proto, + // above) — every delta frame already decodes into a + // complete, standalone state on its own (the PWR wire + // contract resets to empty at the source), so applying one + // is always "decode independently, then merge". + SummaryState::Cms(sk) => { + if encoding != SketchEncoding::ProtoDelta { + return Err( + "CountMin (heap-less) MSGPACK_DELTA is not a valid producer encoding \ + (msgpack-delta is the heap-bearing form)" + .to_string(), + ); + } + let other = decode_cms_from_proto_delta(bytes)?; + sk.merge(&other) + .map_err(|e| format!("merge CountMinSketch delta: {e}")) + } + SummaryState::CountSketch(sk) => { + if encoding != SketchEncoding::ProtoDelta { + return Err( + "CountSketch (heap-less) MSGPACK_DELTA is not a valid producer encoding \ + (msgpack-delta is the heap-bearing form)" + .to_string(), + ); + } + let other = decode_cs_from_proto_delta(bytes)?; + sk.merge(&other) + .map_err(|e| format!("merge CountSketch delta: {e}")) + } + SummaryState::CmsWithHeap(sk) => { + // Matches the existing per-frame reducer dispatch: only + // MsgpackDelta gets true delta treatment; ProtoDelta (not + // produced for this family in this deployment) falls + // through to the full-msgpack decoder, same as `decode_full`. + let other = if encoding == SketchEncoding::MsgpackDelta { + decode_cms_with_heap_from_msgpack_delta(bytes)? + } else { + decode_cms_with_heap_from_msgpack(bytes)? + }; + sk.merge(&other) + .map_err(|e| format!("merge CmsWithHeap delta: {e}")) + } + SummaryState::CountSketchWithHeap(sk) => { + let other = if encoding == SketchEncoding::MsgpackDelta { + decode_cs_with_heap_from_msgpack_delta(bytes)? + } else { + decode_cs_with_heap_from_msgpack(bytes)? + }; + sk.merge(&other) + .map_err(|e| format!("merge CountSketchWithHeap delta: {e}")) + } + } + } + + pub fn quantile(&self, q: f64) -> f64 { + match self { + SummaryState::Dd(sk) => sk.quantile(q).unwrap_or(0.0), + SummaryState::Kll(sk) => sk.quantile(q), + _ => 0.0, + } + } + + pub fn cardinality(&self) -> f64 { + match self { + SummaryState::Hll(sk) => sk.estimate(), + _ => 0.0, + } + } + + /// The bucket TOTAL — sum of row 0 of the underlying matrix. What a + /// bare `count_over_time`/`sum by (item) (rate(...))`-shaped query + /// (no specific item key) reads out. `0.0` for non-Frequency-family + /// states. + pub fn total(&self) -> f64 { + let matrix = match self { + SummaryState::Cms(c) => c.sketch(), + SummaryState::CountSketch(c) => c.sketch().clone(), + SummaryState::CmsWithHeap(h) => h.sketch_matrix(), + SummaryState::CountSketchWithHeap(h) => h.sketch_matrix(), + _ => return 0.0, + }; + matrix + .first() + .map(|row| row.iter().copied().sum::()) + .unwrap_or(0.0) + } + + /// Per-key point estimate — `count(metric{item="x"})`-shaped queries. + /// Unlike [`Self::topk_items`], no heap is needed: all four Frequency + /// variants (heap-bearing or not) already carry a keyed `estimate` + /// over their matrix. `None` for the quantile/cardinality states, + /// which have no item universe at all. + pub fn estimate(&self, key: &str) -> Option { + match self { + SummaryState::Cms(c) => Some(c.estimate(key)), + SummaryState::CountSketch(c) => Some(c.estimate(key)), + SummaryState::CmsWithHeap(h) => Some(h.estimate(key)), + SummaryState::CountSketchWithHeap(h) => Some(h.estimate(key)), + _ => None, + } + } + + /// Top-k `(key, value)` pairs from the heap, descending by value. + /// `None` for anything other than a heap-bearing state — the + /// heap-less Frequency states (`Cms`/`CountSketch`) carry no item + /// universe to enumerate, and the quantile/cardinality states have + /// no heap at all. + pub fn topk_items(&self) -> Option> { + match self { + SummaryState::CmsWithHeap(h) => Some( + h.topk_heap_items() + .into_iter() + .map(|item| (item.key, item.value)) + .collect(), + ), + SummaryState::CountSketchWithHeap(h) => Some( + h.topk_heap_items() + .into_iter() + .map(|item| (item.key, item.value)) + .collect(), + ), + _ => None, + } + } + + /// Merge `other` into `self` in place — both must be the same sketch + /// family. Used to combine several sids' reconstructed states + /// (`cumulative_summary_state`/`per_window_summary_states`) into one + /// cross-sid answer. `CmsWithHeap`/`CountSketchWithHeap` fall through + /// to the catch-all mismatch arm below like any other mixed pair — + /// and since the two variants now hold distinct sketchlib types + /// (`CountMinSketchWithHeap` vs `CountSketchWithHeap`), there is no + /// arm that could accidentally match them together — see their doc + /// on `DeltaSketchKind`. + pub fn merge_same_family(&mut self, other: &SummaryState) -> Result<(), String> { + match (self, other) { + (SummaryState::UnivMon(a), SummaryState::UnivMon(b)) => { + a.merge_in_place(b).map_err(|e| e.to_string()) + } + (SummaryState::Dd(a), SummaryState::Dd(b)) => { + a.merge(b).map_err(|e| format!("merge DDSketch: {e}")) + } + (SummaryState::Hll(a), SummaryState::Hll(b)) => { + a.merge(b).map_err(|e| format!("merge HLL: {e}")) + } + (SummaryState::Kll(a), SummaryState::Kll(b)) => { + a.merge(b).map_err(|e| format!("merge KLL: {e}")) + } + (SummaryState::Cms(a), SummaryState::Cms(b)) => { + a.merge(b).map_err(|e| format!("merge CountMinSketch: {e}")) + } + (SummaryState::CountSketch(a), SummaryState::CountSketch(b)) => { + a.merge(b).map_err(|e| format!("merge CountSketch: {e}")) + } + (SummaryState::CmsWithHeap(a), SummaryState::CmsWithHeap(b)) => { + a.merge(b).map_err(|e| format!("merge CmsWithHeap: {e}")) + } + (SummaryState::CountSketchWithHeap(a), SummaryState::CountSketchWithHeap(b)) => a + .merge(b) + .map_err(|e| format!("merge CountSketchWithHeap: {e}")), + (a, _) => Err(format!( + "SummaryState family mismatch in merge_same_family (self is {})", + a.family_name() + )), + } + } + + /// Diagnostic family name for error messages — not used for dispatch. + fn family_name(&self) -> &'static str { + match self { + SummaryState::UnivMon(_) => "UnivMon", + SummaryState::Dd(_) => "DDSketch", + SummaryState::Hll(_) => "Hll", + SummaryState::Kll(_) => "Kll", + SummaryState::Cms(_) => "Cms", + SummaryState::CountSketch(_) => "CountSketch", + SummaryState::CmsWithHeap(_) => "CmsWithHeap", + SummaryState::CountSketchWithHeap(_) => "CountSketchWithHeap", + } + } +} + +/// Fold every in-range window's frames for ONE series into a single +/// merged `SummaryState` (cumulative over `[t0, t1]`), returning `None` +/// if no Full frame ever landed (every sample was a leading delta). The +/// per-sid building block for a cross-sid answer: reconstruct each +/// candidate sid's state this way, then merge them (`merge_same_family`) +/// before reading out a quantile/cardinality over the combined data. +pub fn cumulative_summary_state( + samples: &[(i64, &SketchSampleState)], + kind: DeltaSketchKind, +) -> Result, String> { + let mut rolling: Option = None; + for (_window_end, state) in samples { + match state.encoding { + SketchEncoding::ProtoFull | SketchEncoding::MsgpackFull => { + let new_state = decode_full(&kind, &state.bytes, state.encoding)?; + rolling = Some(match rolling.take() { + None => new_state, + Some(mut prev) => { + prev.merge_same_family(&new_state)?; + prev + } + }); + } + SketchEncoding::ProtoDelta | SketchEncoding::MsgpackDelta => { + if rolling.is_none() { + rolling = Some(kind.bootstrap_empty()); + } + if let Some(rs) = rolling.as_mut() { + rs.apply_delta_bytes(&state.bytes, state.encoding)?; + } + } + } + } + Ok(rolling) +} + +#[cfg(test)] +/// Walk a sorted-by-window-end slice of samples in time order and +/// produce ONE per-window scalar `(window_end_ms, scalar)`. +/// +/// ## Per-window-reset (PWR) delta model +/// +/// The edge emits frames grouped by window (all frames of one window +/// share the same `window_end` key; the key changes across windows). +/// The edge RESETS its snapshot base at each window boundary, so each +/// window's state is built *from empty*: +/// +/// * Within a window, frames accumulate to the window total. The first +/// frame may be a `Full` (window 1, or a periodic re-snapshot) or a +/// `Delta`-from-empty (windows 2+ under PWR); subsequent frames are +/// `Delta` INCREMENTS applied onto the window's running base. +/// * Across windows, the base MUST reset — a new `window_end` discards +/// the previous window's rolling state and starts from empty. Never +/// carry one window's state into the next (that would inflate via +/// cross-window accumulation). +/// +/// Concretely this fixes two bugs in the old "single rolling Option that +/// only ever resets on a Full" walk: +/// 1. A query range whose Full lives only in window 1 (or out of +/// range) left windows 2+ as deltas with `rolling=None`, all +/// skipped → empty result. +/// 2. A window 2+ delta applied onto window 1's leftover rolling state +/// → cross-window inflation. +/// +/// For a `Delta` that is the window's FIRST frame (the PWR delta-from- +/// empty case), we bootstrap an EMPTY rolling state of `kind` and apply +/// the delta onto it (delta-from-empty ⊕ empty = that window's state). +/// +/// The delta-OFF path (exactly one `Full` per window) still produces one +/// correct value per window: the window opens with a Full, has no +/// further frames, and emits that Full's scalar. +/// +/// `eval` reads a scalar from the rolling state (`quantile(q)` / +/// `cardinality()`). `skipped` counts frames that could not contribute +/// (a delta we genuinely couldn't bootstrap from — should be rare). +/// +/// Returns `Ok(per_window_samples, skipped)`. +pub fn per_window_evaluate( + samples: &[(i64, &SketchSampleState)], + kind: DeltaSketchKind, + eval: E, +) -> Result<(Vec<(i64, f64)>, usize), String> +where + E: Fn(&SummaryState) -> f64, +{ + let (states, skipped) = per_window_summary_states(samples, kind)?; + Ok(( + states.into_iter().map(|(w, rs)| (w, eval(&rs))).collect(), + skipped, + )) +} + +/// Walk a sorted-by-window-end slice of samples in time order and +/// reconstruct ONE sid's per-window `SummaryState` (same per-window-reset +/// walk as [`per_window_evaluate`], generalized to return the +/// reconstructed state itself instead of an already-evaluated scalar). +/// The per-sid building block for cross-sid per-window merging (unlike +/// [`cumulative_summary_state`], which folds a whole `[t0, t1]` range +/// into one answer, this keeps each window separate so a caller can +/// merge same-window states across several sids before evaluating -- +/// needed for a matrix/range-query answer, where each output point is +/// itself a cross-sid merge for that one window). +/// +/// Returns `Ok((per_window_states, skipped))`. +pub fn per_window_summary_states( + samples: &[(i64, &SketchSampleState)], + kind: DeltaSketchKind, +) -> Result<(Vec<(i64, SummaryState)>, usize), String> { + let mut out: Vec<(i64, SummaryState)> = Vec::new(); + let mut skipped = 0usize; + + // Rolling state for the CURRENT window only. Reset to None whenever + // `window_end` changes (a new window establishes its own base from + // empty). `cur_end` tracks which window `rolling` belongs to. + let mut rolling: Option = None; + let mut cur_end: Option = None; + + for (window_end, state) in samples { + // Window boundary: flush the previous window's final accumulated + // state, then reset the base so this window starts from empty. + if cur_end != Some(*window_end) { + if let (Some(prev_end), Some(rs)) = (cur_end, rolling.take()) { + out.push((prev_end, rs)); + } + cur_end = Some(*window_end); + } + + match state.encoding { + SketchEncoding::ProtoFull | SketchEncoding::MsgpackFull => { + // A Full (re)sets this window's base. + rolling = Some(decode_full(&kind, &state.bytes, state.encoding)?); + } + SketchEncoding::ProtoDelta | SketchEncoding::MsgpackDelta => { + // Apply onto this window's running base. If this is the + // window's first frame (PWR delta-from-empty), bootstrap + // an empty base and apply onto it. + if rolling.is_none() { + rolling = Some(kind.bootstrap_empty()); + } + match rolling.as_mut() { + Some(rs) => rs.apply_delta_bytes(&state.bytes, state.encoding)?, + None => skipped += 1, + } + } + } + } + + // Flush the final window. + if let (Some(prev_end), Some(rs)) = (cur_end, rolling.take()) { + out.push((prev_end, rs)); + } + + Ok((out, skipped)) +} + +// --------------------------------------------------------------------------- +// Proto-envelope decoders — P2-4: ONE decoder per family. +// +// These delegate to the precompute-side accumulators' +// `from_sketchlib_proto_bytes`, which are the single source of truth for +// the modified-OTLP proto wire format (envelope unwrapping, alpha/k/ +// precision validation, and — critically for HLL — SPARSE +// `registers_sparse` expansion). Folding the warm read path onto the +// same decoder the ingest path uses means the sparse-register fix (and +// any future format change) can never drift between the two copies again +// — the bug class P2-3 / P2-4 closed. We extract the accumulator's +// public `inner` sketch for the rolling-state merge. +// --------------------------------------------------------------------------- + +fn dd_from_proto(buffer: &[u8]) -> Result { + use crate::accumulators::dd_sketch_accumulator::DDSketchAccumulator; + DDSketchAccumulator::from_sketchlib_proto_bytes(buffer) + .map(|acc| acc.inner) + .map_err(|e| e.to_string()) +} + +fn kll_from_proto(buffer: &[u8]) -> Result { + use crate::accumulators::datasketches_kll_accumulator::DatasketchesKLLAccumulator; + DatasketchesKLLAccumulator::from_sketchlib_proto_bytes(buffer) + .map(|acc| acc.inner) + .map_err(|e| e.to_string()) +} + +fn hll_from_proto(buffer: &[u8]) -> Result { + use crate::accumulators::hll_sketch_accumulator::HllSketchAccumulator; + HllSketchAccumulator::from_sketchlib_proto_bytes(buffer) + .map(|acc| acc.inner) + .map_err(|e| e.to_string()) +} + +/// Apply a proto-encoded `HllDelta` frame onto the HLL register vector — the +/// delta is a varint-packed (index_delta, value) blob; decode + apply +/// (register-wise max) via the shared sketch library so the unpacking stays a +/// single source of truth. +fn apply_hll_proto_delta(sk: &mut HllSketch, buffer: &[u8]) -> Result<(), String> { + sk.apply_delta_bytes(buffer) + .map_err(|e| format!("apply HLLDelta: {e}"))?; + Ok(()) +} + +#[cfg(test)] +mod tests { + //! P2-3 / P2-4 regression tests for the consolidated single-decoder + //! path. These exercise the family proto decoders that now delegate + //! to the precompute accumulators (the single source of truth), so a + //! divergence between the warm read path and the ingest path — + //! notably the SPARSE-register HLL handling the deleted dead decoder + //! got wrong — fails the build. + use super::*; + use asap_sketchlib::HllVariant; + + fn encode_dd(sk: &DdSketch) -> Vec { + use asap_sketchlib::proto::sketchlib::{sketch_envelope, DdSketchState, SketchEnvelope}; + use prost::Message; + let state = DdSketchState { + alpha: sk.alpha, + store_counts: sk.store_counts.clone(), + store_offset: sk.store_offset, + }; + SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Ddsketch(state)), + ..Default::default() + } + .encode_to_vec() + } + + fn encode_kll(k: u16, items: &[f64]) -> Vec { + use asap_sketchlib::proto::sketchlib::{sketch_envelope, KllState, SketchEnvelope}; + use prost::Message; + let state = KllState { + k: k as u32, + items: items.to_vec(), + levels: vec![], + num_levels: 0, + ..Default::default() + }; + SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Kll(state)), + ..Default::default() + } + .encode_to_vec() + } + + fn encode_hll_dense(sk: &HllSketch) -> Vec { + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, HllVariant as ProtoVariant, HyperLogLogState, SketchEnvelope, + }; + use prost::Message; + let state = HyperLogLogState { + variant: ProtoVariant::Regular as i32, + precision: sk.precision, + registers: sk.registers.clone(), + hip_kxq0: sk.hip_kxq0, + hip_kxq1: sk.hip_kxq1, + hip_est: sk.hip_est, + registers_sparse: None, + }; + SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Hll(state)), + ..Default::default() + } + .encode_to_vec() + } + + /// Build a SPARSE HLL proto frame: dense `registers` left empty, + /// `registers_sparse.packed` = varint (index_delta, value) pairs. + /// This is exactly the wire form a low-cardinality producer emits + /// (sketchlib-go below its dense/sparse crossover) — the frame the + /// DELETED `HllSketch_from_sketchlib_proto_bytes` hard-rejected with + /// "registers has 0 bytes". + fn encode_hll_sparse(precision: u32, nonzero: &[(u64, u8)]) -> Vec { + use asap_sketchlib::proto::sketchlib::{ + sketch_envelope, HllSparseRegisters, HllVariant as ProtoVariant, HyperLogLogState, + SketchEnvelope, + }; + use prost::Message; + // Varint-pack (index_delta, value), ascending index order. + let mut packed: Vec = Vec::new(); + let mut prev: u64 = 0; + let put_uvarint = |buf: &mut Vec, mut v: u64| loop { + let b = (v & 0x7f) as u8; + v >>= 7; + if v != 0 { + buf.push(b | 0x80); + } else { + buf.push(b); + break; + } + }; + let mut sorted = nonzero.to_vec(); + sorted.sort_by_key(|(i, _)| *i); + for (idx, val) in &sorted { + put_uvarint(&mut packed, idx - prev); + put_uvarint(&mut packed, *val as u64); + prev = *idx; + } + let state = HyperLogLogState { + variant: ProtoVariant::Regular as i32, + precision, + registers: Vec::new(), // dense field empty → sparse path + hip_kxq0: 0.0, + hip_kxq1: 0.0, + hip_est: 0.0, + // `num_registers` is informational — the decoder expands + // against `expected_len` from precision, not this field. + registers_sparse: Some(HllSparseRegisters { + num_registers: 1u32 << precision, + packed, + }), + }; + SketchEnvelope { + sketch_state: Some(sketch_envelope::SketchState::Hll(state)), + ..Default::default() + } + .encode_to_vec() + } + + #[test] + fn hll_from_proto_accepts_sparse_frame() { + // The consolidated decoder must accept the sparse wire form (the + // deleted dead decoder rejected it). Build a sparse frame setting + // a handful of registers, decode it, and confirm those register + // slots came back set in the dense array. + let precision = 12u32; + let nonzero = [(3u64, 5u8), (100, 2), (4000, 7)]; + let bytes = encode_hll_sparse(precision, &nonzero); + let sk = hll_from_proto(&bytes).expect("sparse HLL frame must decode (P2-3 regression)"); + assert_eq!(sk.registers.len(), 1usize << precision); + for (idx, val) in nonzero { + assert_eq!( + sk.registers[idx as usize], val, + "sparse register {idx} expanded to wrong value" + ); + } + } + + #[test] + fn hll_from_proto_matches_accumulator_decoder() { + // P2-4: the warm read path and the ingest accumulator must decode + // the SAME bytes to the SAME sketch (one source of truth). + use crate::accumulators::hll_sketch_accumulator::HllSketchAccumulator; + let mut sk = HllSketch::new(HllVariant::Regular, 12); + for i in 0..500u64 { + sk.update(format!("item-{i}").as_bytes()); + } + let bytes = encode_hll_dense(&sk); + let via_delta = hll_from_proto(&bytes).expect("delta_apply hll decode"); + let via_acc = HllSketchAccumulator::from_sketchlib_proto_bytes(&bytes) + .expect("accumulator hll decode") + .inner; + assert_eq!( + via_delta.registers, via_acc.registers, + "delta_apply and accumulator must produce identical HLL registers" + ); + assert!((via_delta.estimate() - via_acc.estimate()).abs() < 1e-9); + } + + #[test] + fn dd_from_proto_matches_accumulator_decoder() { + use crate::accumulators::dd_sketch_accumulator::DDSketchAccumulator; + let mut sk = DdSketch::new(0.01); + for v in [1.0, 2.0, 5.0, 5.0, 9.0, 42.0] { + sk.update(v); + } + let bytes = encode_dd(&sk); + let via_delta = dd_from_proto(&bytes).expect("delta_apply dd decode"); + let via_acc = DDSketchAccumulator::from_sketchlib_proto_bytes(&bytes) + .expect("accumulator dd decode") + .inner; + // Same quantile answers from the same bytes through both paths. + assert_eq!(via_delta.quantile(0.5), via_acc.quantile(0.5)); + assert_eq!(via_delta.quantile(0.99), via_acc.quantile(0.99)); + } + + #[test] + fn kll_from_proto_matches_accumulator_decoder() { + use crate::accumulators::datasketches_kll_accumulator::DatasketchesKLLAccumulator; + let items: Vec = (0..200).map(|i| i as f64).collect(); + let bytes = encode_kll(256, &items); + let via_delta = kll_from_proto(&bytes).expect("delta_apply kll decode"); + let via_acc = DatasketchesKLLAccumulator::from_sketchlib_proto_bytes(&bytes) + .expect("accumulator kll decode") + .inner; + assert_eq!(via_delta.quantile(0.5), via_acc.quantile(0.5)); + } + + // ----------------------------------------------------------------- + // Per-window-reset (PWR) delta-apply regression tests. + // + // The edge resets its snapshot base at every window boundary, so a + // window's first frame is either a Full (window 1 / re-snapshot) or + // a Delta-from-empty (windows 2+). The query-side walk must: + // * reset the rolling base when `window_end` changes, + // * bootstrap an empty base for a window's leading Delta, + // * emit ONE value per window (the window's final accumulated + // state), never per-frame and never cross-window-accumulated. + // ----------------------------------------------------------------- + + fn full(bytes: Vec) -> SketchSampleState { + SketchSampleState { + bytes, + encoding: SketchEncoding::ProtoFull, + } + } + fn delta(bytes: Vec) -> SketchSampleState { + SketchSampleState { + bytes, + encoding: SketchEncoding::ProtoDelta, + } + } + + fn dd_over(alpha: f64, vals: &[f64]) -> DdSketch { + let mut sk = DdSketch::new(alpha); + for &v in vals { + sk.update(v); + } + sk + } + + /// PWR across 3 windows: window 1 is `[Full]`, windows 2 & 3 are + /// `[Delta-from-empty]` (NO Full carry-in). Each window must + /// reconstruct its OWN distribution's median — not empty (the old + /// "skip delta with no base" bug) and not cross-window-inflated. + #[test] + fn pwr_ddsketch_three_windows_delta_from_empty() { + let alpha = 0.01; + let w1 = dd_over(alpha, &[1.0, 2.0, 3.0, 4.0, 5.0]); + let w2 = dd_over(alpha, &[10.0, 20.0, 30.0, 40.0, 50.0]); + let w3 = dd_over(alpha, &[100.0, 200.0, 300.0, 400.0, 500.0]); + + // window 1 ships a Full; windows 2+ ship a delta-from-empty. + let s1 = full(encode_dd(&w1)); + let s2 = delta(encode_dd(&w2)); + let s3 = delta(encode_dd(&w3)); + let samples = vec![(1000_i64, &s1), (2000, &s2), (3000, &s3)]; + + let kind = DeltaSketchKind::DDSketch { alpha }; + let (out, skipped) = + per_window_evaluate(&samples, kind, |rs| rs.quantile(0.5)).expect("pwr eval"); + assert_eq!(skipped, 0, "PWR must not skip delta-from-empty frames"); + assert_eq!(out.len(), 3, "one value per window"); + + // Each window's median ≈ that window's own distribution median, + // independent of the others (no carry-in inflation). + let truth = [ + w1.quantile(0.5).unwrap(), + w2.quantile(0.5).unwrap(), + w3.quantile(0.5).unwrap(), + ]; + for (i, (w_end, est)) in out.iter().enumerate() { + assert_eq!(*w_end, (i as i64 + 1) * 1000); + let rel = (est - truth[i]).abs() / truth[i].max(1e-9); + assert!( + rel < 0.05, + "window {i}: est={est} truth={} rel={rel}", + truth[i] + ); + } + // Cross-window-inflation guard: window 2's median must NOT have + // absorbed window 1 (would pull it well below 30). + assert!( + out[1].1 > 20.0, + "window 2 median {} suggests cross-window accumulation", + out[1].1 + ); + } + + /// Sub-window producer: a SINGLE window carries multiple frames + /// `[Full, Delta, Delta]`, where each later delta is an increment + /// since the previous emit in that window. The walk must COLLAPSE + /// them to ONE value = the window's running total, not emit three. + #[test] + fn pwr_ddsketch_subwindow_frames_collapse_to_window_total() { + let alpha = 0.01; + // Three sub-window increments that together cover 1..=15. + let a = dd_over(alpha, &[1.0, 2.0, 3.0, 4.0, 5.0]); + let b = dd_over(alpha, &[6.0, 7.0, 8.0, 9.0, 10.0]); + let c = dd_over(alpha, &[11.0, 12.0, 13.0, 14.0, 15.0]); + let s_a = full(encode_dd(&a)); + let s_b = delta(encode_dd(&b)); + let s_c = delta(encode_dd(&c)); + // All three share the same window_end (one window, sub-window frames). + let samples = vec![(5000_i64, &s_a), (5000, &s_b), (5000, &s_c)]; + + let kind = DeltaSketchKind::DDSketch { alpha }; + let (out, skipped) = + per_window_evaluate(&samples, kind, |rs| rs.quantile(0.5)).expect("subwindow eval"); + assert_eq!(skipped, 0); + assert_eq!(out.len(), 1, "sub-window frames collapse to ONE value"); + assert_eq!(out[0].0, 5000); + + let truth = dd_over(alpha, &(1..=15).map(|v| v as f64).collect::>()) + .quantile(0.5) + .unwrap(); + let rel = (out[0].1 - truth).abs() / truth.max(1e-9); + assert!(rel < 0.05, "window total est={} truth={truth}", out[0].1); + } + + /// Same sub-window collapse, but the window's FIRST frame is a + /// Delta-from-empty (PWR window 2+ with sub-window frames): + /// `[Delta-from-empty, Delta, Delta]`. + #[test] + fn pwr_ddsketch_subwindow_first_frame_delta_from_empty() { + let alpha = 0.01; + let a = dd_over(alpha, &[1.0, 2.0, 3.0, 4.0, 5.0]); + let b = dd_over(alpha, &[6.0, 7.0, 8.0, 9.0, 10.0]); + let c = dd_over(alpha, &[11.0, 12.0, 13.0, 14.0, 15.0]); + let s_a = delta(encode_dd(&a)); // first frame is delta-from-empty + let s_b = delta(encode_dd(&b)); + let s_c = delta(encode_dd(&c)); + let samples = vec![(9000_i64, &s_a), (9000, &s_b), (9000, &s_c)]; + + let kind = DeltaSketchKind::DDSketch { alpha }; + let (out, skipped) = + per_window_evaluate(&samples, kind, |rs| rs.quantile(0.5)).expect("eval"); + assert_eq!(skipped, 0); + assert_eq!(out.len(), 1); + let truth = dd_over(alpha, &(1..=15).map(|v| v as f64).collect::>()) + .quantile(0.5) + .unwrap(); + let rel = (out[0].1 - truth).abs() / truth.max(1e-9); + assert!(rel < 0.05, "est={} truth={truth}", out[0].1); + } + + /// PWR for HLL across 3 windows, each a Delta-from-empty (sparse + /// register delta). Bootstrapping an EMPTY HLL of the right precision + /// is required (register deltas index into a pre-sized array). Each + /// window's cardinality must reflect its OWN item set. + #[test] + fn pwr_hll_three_windows_delta_from_empty() { + let precision = 12u32; + // Build per-window HLLs, then encode each as a register-delta + // against an EMPTY sketch (= that window's full register state, + // the PWR delta-from-empty wire form). + let empty = HllSketch::new(HllVariant::Regular, precision); + let mut frames = Vec::new(); + let truths = [200usize, 800, 1500]; + for (w, &n) in truths.iter().enumerate() { + let mut sk = HllSketch::new(HllVariant::Regular, precision); + let base = (w as u64) * 100_000; // disjoint item sets per window + for i in 0..n as u64 { + sk.update(format!("u-{}", base + i).as_bytes()); + } + let bytes = sk.compute_delta(&empty, 0); + frames.push((((w as u64) + 1) * 1000, delta(bytes))); + } + let samples: Vec<(i64, &SketchSampleState)> = + frames.iter().map(|(t, s)| (*t as i64, s)).collect(); + + let kind = DeltaSketchKind::Hll { precision }; + let (out, skipped) = + per_window_evaluate(&samples, kind, |rs| rs.cardinality()).expect("hll pwr eval"); + assert_eq!(skipped, 0, "HLL delta-from-empty must bootstrap, not skip"); + assert_eq!(out.len(), 3); + for (i, (_w_end, est)) in out.iter().enumerate() { + let n = truths[i] as f64; + let rel = (est - n).abs() / n; + assert!( + rel < 0.15, + "window {i}: HLL est={est} truth={n} rel={rel} (each window independent)" + ); + } + } + + /// `CmsWithHeap` (min-over-rows estimator, `CountMinSketchWithHeap`) + /// and `CountSketchWithHeap` (median-of-signed-rows estimator, the + /// distinct `CountSketchWithHeap` type) are different sketch + /// algorithms that merely happen to share a storage shape — merging + /// one into the other must be rejected as a family mismatch, the + /// same as merging a `Cms` into a `Kll` would be. Since the two + /// `SummaryState` variants now hold genuinely different Rust types, + /// this is also enforced at compile time — there is no arm in + /// `merge_same_family` that type-checks a mixed pair together. + #[test] + fn cms_with_heap_and_count_sketch_with_heap_are_not_the_same_family() { + use asap_sketchlib::{CountMinSketchWithHeap, CountSketchWithHeap, MessagePackCodec}; + + let mut cms_heap = CountMinSketchWithHeap::new(4, 256, 10); + cms_heap.update("a", 1.0); + let mut cs_heap = CountSketchWithHeap::new(4, 256, 10); + cs_heap.update("b", 1.0); + + let mut a = SummaryState::CmsWithHeap( + CountMinSketchWithHeap::from_msgpack(&cms_heap.to_msgpack().unwrap()).unwrap(), + ); + let b = SummaryState::CountSketchWithHeap( + CountSketchWithHeap::from_msgpack(&cs_heap.to_msgpack().unwrap()).unwrap(), + ); + + match a.merge_same_family(&b) { + Err(msg) => assert!( + msg.contains("family mismatch"), + "expected a family-mismatch error, got: {msg}" + ), + Ok(()) => panic!( + "CmsWithHeap must not merge with CountSketchWithHeap -- \ + different algorithms sharing only a storage shape" + ), + } + } + + fn encode_delta_heap( + rows: u32, + cols: u32, + cells: &[(u32, u32, i64)], + heap: &[(&str, f64)], + heap_size: u64, + ) -> Vec { + #[derive(serde::Serialize)] + struct W<'a>( + bool, + (u32, u32, &'a [(u32, u32, i64)]), + Vec<(String, f64)>, + u64, + ); + let heap_owned: Vec<(String, f64)> = + heap.iter().map(|(k, v)| (k.to_string(), *v)).collect(); + let w = W(true, (rows, cols, cells), heap_owned, heap_size); + rmp_serde::to_vec(&w).expect("encode delta-heap") + } + + /// `SummaryState::CountSketchWithHeap` must decode both FULL and + /// DELTA-HEAP msgpack frames through the genuine + /// `asap_sketchlib::CountSketchWithHeap` (median-of-signed-rows + /// estimator) rather than the CMS-family `CountMinSketchWithHeap` + /// (min-over-rows estimator) it used to alias — the bug this split + /// fixed. Built via real `update()` calls (not a hand-crafted matrix) + /// so the sign-hashed row semantics are genuinely exercised, then + /// checks both decode paths reproduce the same matrix and the same + /// `estimate()` as the in-memory sketch they were encoded from. + #[test] + fn count_sketch_with_heap_full_and_delta_decode_via_new_asap_sketchlib_type() { + use asap_sketchlib::{CountSketchWithHeap, MessagePackCodec}; + + let mut built = CountSketchWithHeap::new(4, 64, 10); + for _ in 0..50 { + built.update("k", 1.0); + } + let expected_matrix = built.sketch_matrix(); + let expected_estimate = built.estimate("k"); + + // FULL path. + let full_bytes = built.to_msgpack().expect("encode full CountSketchWithHeap"); + let full_state = decode_full( + &DeltaSketchKind::CountSketchWithHeap { + rows: 4, + cols: 64, + heap_size: 10, + }, + &full_bytes, + SketchEncoding::MsgpackFull, + ) + .expect("decode_full CountSketchWithHeap"); + match full_state { + SummaryState::CountSketchWithHeap(inner) => { + assert_eq!(inner.sketch_matrix(), expected_matrix); + assert_eq!(inner.estimate("k"), expected_estimate); + } + other => panic!( + "expected CountSketchWithHeap state, got {}", + other.family_name() + ), + } + + // DELTA-HEAP path: same cells + heap against an empty base (PWR + // contract), encoded the way the Go producer does. + let cells: Vec<(u32, u32, i64)> = expected_matrix + .iter() + .enumerate() + .flat_map(|(r, row)| { + row.iter().enumerate().filter_map(move |(c, v)| { + if *v != 0.0 { + Some((r as u32, c as u32, *v as i64)) + } else { + None + } + }) + }) + .collect(); + let heap_pairs: Vec<(String, f64)> = built + .topk_heap_items() + .into_iter() + .map(|item| (item.key, item.value)) + .collect(); + assert!(!heap_pairs.is_empty(), "expected \"k\" in the top-k heap"); + let heap_refs: Vec<(&str, f64)> = + heap_pairs.iter().map(|(k, v)| (k.as_str(), *v)).collect(); + let delta_bytes = encode_delta_heap(4, 64, &cells, &heap_refs, 10); + + let mut rolling = DeltaSketchKind::CountSketchWithHeap { + rows: 4, + cols: 64, + heap_size: 10, + } + .bootstrap_empty(); + rolling + .apply_delta_bytes(&delta_bytes, SketchEncoding::MsgpackDelta) + .expect("apply CountSketchWithHeap delta"); + match rolling { + SummaryState::CountSketchWithHeap(inner) => { + assert_eq!( + inner.sketch_matrix(), + expected_matrix, + "delta path must reconstruct the identical matrix" + ); + assert_eq!(inner.estimate("k"), expected_estimate); + } + other => panic!( + "expected CountSketchWithHeap state, got {}", + other.family_name() + ), + } + } +} diff --git a/crates/asap-physical-operators/src/stored_state/mod.rs b/crates/asap-physical-operators/src/stored_state/mod.rs new file mode 100644 index 00000000..0a3d0e7b --- /dev/null +++ b/crates/asap-physical-operators/src/stored_state/mod.rs @@ -0,0 +1,19 @@ +//! Portable stored-summary payloads and reconstruction, independent of storage engines. +pub mod decoders; +pub mod delta_apply; +pub mod readout; + +#[derive(Debug, Clone)] +pub struct SketchSampleState { + pub bytes: Vec, + /// Wire-encoding hint from the OTLP DataPoint's `encoding` field. + pub encoding: SketchEncoding, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum SketchEncoding { + ProtoFull, + ProtoDelta, + MsgpackFull, + MsgpackDelta, +} diff --git a/crates/asap-physical-operators/src/stored_state/readout.rs b/crates/asap-physical-operators/src/stored_state/readout.rs new file mode 100644 index 00000000..6d99fce4 --- /dev/null +++ b/crates/asap-physical-operators/src/stored_state/readout.rs @@ -0,0 +1,114 @@ +//! Planner-declared readouts over reconstructed summary states. +use super::delta_apply::SummaryState; +use planner_types::{post_asap::SketchQuery, pre_asap::ColumnRef}; +#[derive(Debug, thiserror::Error)] +pub enum Error { + #[error("{0}")] + Unsupported(&'static str), +} +pub fn sketch_query_value(rs: &SummaryState, query: &SketchQuery) -> Result { + if let SummaryState::UnivMon(state) = rs { + use crate::AggregateCore; + let statistic = match query { + SketchQuery::Cardinality => crate::Statistic::Cardinality, + SketchQuery::FrequencyL2 => crate::Statistic::FrequencyL2, + SketchQuery::FrequencyEntropy => crate::Statistic::FrequencyEntropy, + SketchQuery::PointCount { + key: ColumnRef::SampleValue, + value: None, + } => crate::Statistic::Count, + _ => return Err(Error::Unsupported("unsupported UnivMon readout")), + }; + return state + .query_statistic(statistic, &None, &Default::default()) + .map_err(|_| Error::Unsupported("UnivMon readout failed")); + } + match query { + SketchQuery::FrequencyL2 | SketchQuery::FrequencyEntropy => Err(Error::Unsupported( + "frequency moment readout requires UnivMon", + )), + SketchQuery::Quantile { q } => match rs { + // Typed PromQL/continuous-percentile readout uses interpolation; + // portable DDS `quantile` deliberately retains lower-rank parity. + SummaryState::Dd(sketch) => sketch.quantile_interpolated(*q).ok_or(Error::Unsupported( + "DDS interpolated quantile is unavailable", + )), + _ => Ok(rs.quantile(*q)), + }, + SketchQuery::Cardinality => Ok(rs.cardinality()), + // `key: ColumnRef::SampleValue, value: None` means "no specific + // item" -- the bare bucket total. `key: Named(_), value: Some(v)` + // is a per-item point lookup (e.g. `count(cms_metric{item="x"})`) + // -- `value` is where the filter's actual value lives (see + // `planner_types::post_asap::SketchQuery::PointCount`'s doc for why `readout` + // can't resolve it itself). Any other combination (e.g. a `Named` + // key with no value, or `SampleValue` with a value) is a shape + // this executor doesn't expect to see and reports rather than + // silently misreading. + SketchQuery::PointCount { + key: ColumnRef::SampleValue, + value: None, + } => Ok(rs.total()), + SketchQuery::PointCount { + key: ColumnRef::Named(_) | ColumnRef::Qualified { .. }, + value: Some(v), + } => rs.estimate(v).ok_or(Error::Unsupported( + "PointCount by key requires a Frequency-family sketch (Cms/CountSketch/..WithHeap)", + )), + SketchQuery::PointCount { .. } => Err(Error::Unsupported( + "unrecognized PointCount shape (key/value combination not expected)", + )), + // Both readout callers branch on `TopK` before ever calling this + // function (see `readout_cumulative`/`readout_per_window`), so + // this arm is unreachable in practice; kept for match + // exhaustiveness (`SketchQuery` has no `#[non_exhaustive]`) and to + // fail loudly rather than panic if that invariant is ever broken. + SketchQuery::TopK { .. } => Err(Error::Unsupported( + "TopK must be read out via topk_ranked, not sketch_query_value", + )), + } +} + +/// Rank a merged `SummaryState`'s top-k heap items descending by value and +/// cap at the requested `k`. The sort is load-bearing, not defensive +/// polish: `SummaryState::topk_items` reads back a bounded min-heap's +/// backing array as-is (`HHHeap::heap()`, asap_sketchlib) -- it does NOT +/// actually guarantee order despite its own doc wording. Errors for a +/// heap-less family (`Dd`/`Hll`/`Kll`/`Cms`/`CountSketch` -- no item +/// universe to rank), not for an empty heap (a heap-bearing family that +/// simply never received any updates yields `Ok(vec![])`, not an error). +pub fn topk_ranked(rs: &SummaryState, k: usize) -> Result, Error> { + let mut items = rs.topk_items().ok_or(Error::Unsupported( + "TopK requires a heap-bearing family (CmsWithHeap/CountSketchWithHeap) -- \ + this state's family carries no item universe to rank", + ))?; + items.sort_by(|a, b| { + b.1.partial_cmp(&a.1) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.0.cmp(&b.0)) // deterministic tie-break for equal counts + }); + items.truncate(k); + Ok(items) +} + +/// Merge already selected exact panes and finalize using the shared accumulator contract. +pub fn exact_readout( + states: impl IntoIterator>, + statistic: crate::Statistic, + key: &Option, + parameters: &std::collections::HashMap, +) -> Result { + let mut states = states.into_iter(); + let mut merged = states + .next() + .ok_or_else(|| "empty exact state input".to_string())? + .clone_boxed_core(); + for state in states { + merged = merged + .merge_with(state.as_ref()) + .map_err(|e| e.to_string())?; + } + merged + .query_statistic(statistic, key, parameters) + .map_err(|e| e.to_string()) +} From 06495200007662551ec05e6b3f976b50184e8e98 Mon Sep 17 00:00:00 2001 From: zz_y Date: Wed, 23 Sep 2026 22:57:40 +0000 Subject: [PATCH 07/90] docs: clarify shared stored-state computation ownership --- docs/design_docs/physical-operators.md | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index b1b55b97..9fa0aa6c 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -74,8 +74,10 @@ Grouped TopK composes Sort and Limit within each group; candidate completeness is an earlier pruning obligation. Values retain Planner types and nullability. Native summary batches currently -support exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch and HLL. Other available -low-level kernels do not imply native batch bindings. Unsupported expressions, +support exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch and HLL. Stored-summary decoding, delta reconstruction, exact finalization and +family-specific SketchQuery readout also live in this library. Deployment code +selects compatible panes and supplies source batches. Stored-state kernels do +not imply native batch bindings for every family. Unsupported expressions, state families and parameters must be rejected during binding, without an implicit external fallback. The backend retains source, storage, publication and protocol adapters. Computation must bind to Planner operations without a second backend operator From 5197bfe1291258a80a1e3b4733a912aafc7fd6a7 Mon Sep 17 00:00:00 2001 From: zz_y Date: Wed, 23 Sep 2026 23:04:29 +0000 Subject: [PATCH 08/90] fix: finalize exact state before relational value consumers --- .../src/dag/expressions.rs | 31 ++++++++++++++++++- 1 file changed, 30 insertions(+), 1 deletion(-) diff --git a/crates/asap-physical-operators/src/dag/expressions.rs b/crates/asap-physical-operators/src/dag/expressions.rs index 571a9d42..0a6227cf 100644 --- a/crates/asap-physical-operators/src/dag/expressions.rs +++ b/crates/asap-physical-operators/src/dag/expressions.rs @@ -402,7 +402,36 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul return Err(invalid()); } validate(left, schema)?; - validate(right, schema) + validate(right, schema)?; + let (a, _) = left + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + let (b, _) = right + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + fn comparable(dtype: &DataType) -> bool { + match dtype { + DataType::Null + | DataType::Int64 + | DataType::Float64 + | DataType::Utf8 + | DataType::Bool + | DataType::Timestamp => true, + DataType::Map { key, value, .. } => comparable(key) && comparable(value), + _ => false, + } + } + let numeric = |dtype: &DataType| matches!(dtype, DataType::Int64 | DataType::Float64); + if !comparable(&a) + || !comparable(&b) + || (a != b + && !matches!(a, DataType::Null) + && !matches!(b, DataType::Null) + && !(numeric(&a) && numeric(&b))) + { + return Err(invalid()); + } + Ok(()) } QueryExpr::FunctionCall { name, args } => { if name != "asap_struct_field" From 31461483f319f7f7e05ce0d1b32a50c1ff06f67b Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 02:55:29 +0000 Subject: [PATCH 09/90] feat: bind raw scans through shared data source connectors --- crates/asap-physical-operators/src/dag/mod.rs | 2 + .../src/dag/planner.rs | 32 ++ .../asap-physical-operators/src/dag/scan.rs | 212 +++++++++++ .../asap-physical-operators/tests/raw_scan.rs | 328 ++++++++++++++++++ docs/design_docs/physical-operators.md | 72 +++- 5 files changed, 641 insertions(+), 5 deletions(-) create mode 100644 crates/asap-physical-operators/src/dag/scan.rs create mode 100644 crates/asap-physical-operators/tests/raw_scan.rs diff --git a/crates/asap-physical-operators/src/dag/mod.rs b/crates/asap-physical-operators/src/dag/mod.rs index c00d5417..c4ad17ea 100644 --- a/crates/asap-physical-operators/src/dag/mod.rs +++ b/crates/asap-physical-operators/src/dag/mod.rs @@ -519,3 +519,5 @@ pub mod batch_execution; pub mod expressions; mod temporal; + +pub mod scan; diff --git a/crates/asap-physical-operators/src/dag/planner.rs b/crates/asap-physical-operators/src/dag/planner.rs index f8e0286b..7b801c32 100644 --- a/crates/asap-physical-operators/src/dag/planner.rs +++ b/crates/asap-physical-operators/src/dag/planner.rs @@ -28,9 +28,29 @@ fn invalid(message: impl Into) -> Error { pub type Source<'a> = Box + 'a>; pub fn bind<'a>( + dag: &ExecutableDag, + sources: BTreeMap>, + roots: &[NodeId], +) -> Result, Error> { + bind_internal(dag, sources, roots, None) +} + +/// Bind raw Planner Scan leaves through registered connectors. Other retained +/// pre-ASAP expressions remain unsupported; they are not executed externally. +pub fn bind_with_data_sources<'a>( + dag: &ExecutableDag, + sources: BTreeMap>, + roots: &[NodeId], + data_sources: &super::scan::DataSources, +) -> Result, Error> { + bind_internal(dag, sources, roots, Some(data_sources)) +} + +fn bind_internal<'a>( dag: &ExecutableDag, mut sources: BTreeMap>, roots: &[NodeId], + data_sources: Option<&super::scan::DataSources>, ) -> Result, Error> { preflight_depth(dag)?; dag.validate().map_err(|e| invalid(e.to_string()))?; @@ -96,6 +116,18 @@ pub fn bind<'a>( Box::new(CheckedSource { source, output }) as Source<'a>, vec![], ) + } else if let ( + Some(registry), + Payload::Fallback { + expression: expression @ QueryExpr::Scan { .. }, + }, + ) = (data_sources, &node.payload) + { + let scan = registry.bind(expression)?; + if scan.output_schema() != output { + return Err(invalid("Scan output differs from post-ASAP schema")); + } + (Box::new(scan) as Source<'a>, vec![]) } else { let mut inputs = dependencies.get(&id).cloned().unwrap_or_default(); let mut schemas = inputs diff --git a/crates/asap-physical-operators/src/dag/scan.rs b/crates/asap-physical-operators/src/dag/scan.rs new file mode 100644 index 00000000..4c225389 --- /dev/null +++ b/crates/asap-physical-operators/src/dag/scan.rs @@ -0,0 +1,212 @@ +//! Raw data access. Connectors provide rows; Scan owns Planner predicate semantics. +use super::{ + expressions::CompiledExpression, + values::{Batch, Schema, Value}, + Error, Input, OutputStream, PhysicalOperator, RunContext, +}; +use futures::{stream, StreamExt}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{DataType, QueryExpr, Source}, +}; +use std::sync::Arc; + +/// A bound data source. Metadata must be stable for the lifetime of the binding. +/// Each scan opens an independent cursor. Connectors return raw, unfiltered rows +/// and must honor cancellation and bound their own I/O buffers. Dropping a cursor +/// must release its resources. A connector error is never an empty successful scan. +pub trait RawSource { + fn schema(&self) -> Schema; + fn scan(&self, context: RunContext) -> Result, Error>; +} + +/// Explicit source identities; no implicit network discovery or fallback. +#[derive(Default)] +pub struct DataSources { + sources: Vec<(Source, Arc)>, +} +impl DataSources { + pub fn register(&mut self, identity: Source, source: Arc) -> Result<(), Error> { + if self.sources.iter().any(|(key, _)| key == &identity) { + return Err(Error::Invalid("duplicate data source".into())); + } + super::values::validate_schema(&source.schema())?; + self.sources.push((identity, source)); + Ok(()) + } + pub fn bind(&self, expression: &QueryExpr) -> Result { + let QueryExpr::Scan { + source, + predicates, + schema, + } = expression + else { + return Err(Error::Invalid( + "raw Scan requires a Planner Scan leaf".into(), + )); + }; + let output = Arc::new(SummarySchema { + fields: schema + .columns + .iter() + .map(|column| SummaryField { + name: column.name.clone(), + dtype: SummaryFamilyType::Plain(column.dtype.clone()), + nullable: column.nullable, + }) + .collect(), + time_index: schema.time_index, + }); + super::values::validate_schema(&output)?; + let reader = self + .sources + .iter() + .find(|(key, _)| key == source) + .map(|(_, reader)| reader.clone()) + .ok_or_else(|| Error::Invalid(format!("unbound raw source: {source:?}")))?; + if reader.schema() != output { + return Err(Error::Invalid( + "raw source differs from Planner Scan schema".into(), + )); + } + let predicates = predicates + .iter() + .map(|predicate| { + let predicate = CompiledExpression::compile(&predicate.0, &output)?; + if predicate.dtype().0 != DataType::Bool { + return Err(Error::Invalid("Scan predicate must be boolean".into())); + } + Ok(predicate) + }) + .collect::, Error>>()?; + Ok(Scan { + reader, + output, + predicates, + }) + } +} + +pub struct Scan { + reader: Arc, + output: Schema, + predicates: Vec, +} +impl PhysicalOperator for Scan { + fn name(&self) -> &str { + "Scan" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, batch: &Batch) -> usize { + batch.bytes() + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + if !inputs.is_empty() { + return Err(Error::Invalid("Scan cannot have inputs".into())); + } + if context.is_cancelled() { + return Err(Error::Cancelled); + } + // Opening is lazy: validation and construction of a run perform no I/O. + let opening = context.clone(); + let stream = stream::once(async move { + if opening.is_cancelled() { + return Err(Error::Cancelled); + } + self.reader.scan(opening) + }); + use futures::TryStreamExt; + Ok(stream + .try_flatten() + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + if batch.schema() != &self.output { + return Err(Error::Invalid( + "connector returned a different Scan schema".into(), + )); + } + if self.predicates.is_empty() { + return Ok(batch); + } + let _workspace = + context.reserve(batch.bytes().checked_mul(2).ok_or(Error::MemoryLimit)?)?; + let mut rows = Vec::new(); + for row in batch.rows() { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let mut keep = true; + for predicate in &self.predicates { + match predicate.evaluate(row)? { + Value::Bool(true) => {} + Value::Bool(false) | Value::Null => { + keep = false; + break; + } + _ => { + return Err(Error::Invalid("Scan predicate is not boolean".into())) + } + } + } + if keep { + rows.push(row.clone()); + } + } + Batch::try_new(self.output.clone(), rows) + }) + .boxed_local()) + } +} + +/// Immutable in-memory raw data. The connector owns the resident input; each +/// cursor clones only the next requested batch, not the entire data set. +pub struct MemorySource { + schema: Schema, + batches: Vec, +} +impl MemorySource { + pub fn new(schema: Schema, batches: Vec) -> Result { + super::values::validate_schema(&schema)?; + if schema + .fields + .iter() + .any(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + { + return Err(Error::Invalid( + "raw source cannot contain summary states".into(), + )); + } + if batches.iter().any(|batch| batch.schema() != &schema) { + return Err(Error::Invalid("memory source batch schema mismatch".into())); + } + Ok(Self { schema, batches }) + } +} +impl RawSource for MemorySource { + fn schema(&self) -> Schema { + self.schema.clone() + } + fn scan(&self, context: RunContext) -> Result, Error> { + Ok(stream::iter(self.batches.iter()) + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let _allocation = context.reserve(batch.bytes())?; + Ok(batch.clone()) + }) + .boxed_local()) + } +} diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs new file mode 100644 index 00000000..5390bec2 --- /dev/null +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -0,0 +1,328 @@ +//! Scan acceptance uses the public connector contract and Planner physical DAGs. +use asap_physical_operators::dag::{ + planner::bind_with_data_sources, + scan::{DataSources, MemorySource, RawSource}, + values::{Batch, Schema, Value}, + Error, Limits, OutputStream, RunContext, Scope, +}; +use futures::{executor::block_on, stream, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{Column, DataType, GroupKeys, Predicate, QueryExpr, Source}, +}; +use std::{ + collections::BTreeMap, + rc::Rc, + sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, + }, +}; + +fn fixture() -> (QueryExpr, Schema, Vec) { + let schema = + planner_types::pre_asap::Schema::new(vec![Column::new("value", DataType::Int64, true)]); + let output = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: true, + }], + time_index: None, + }); + let scan = QueryExpr::Scan { + source: Source::Table { + table_ref: "numbers".into(), + }, + predicates: vec![Predicate(Rc::new(QueryExpr::IsNotNull(Rc::new( + QueryExpr::Column(0), + ))))], + schema, + }; + let batches = vec![ + Batch::try_new( + output.clone(), + vec![vec![Value::Int64(3)], vec![Value::Null]], + ) + .unwrap(), + Batch::try_new( + output.clone(), + vec![vec![Value::Int64(9)], vec![Value::Int64(2)]], + ) + .unwrap(), + ]; + (scan, output, batches) +} +fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> ExecutableDag { + let node = |id, payload| ExecutableDagNode { + id: PostAsapNodeId(id), + payload, + output_state: state, + output_schema: (**schema).clone(), + guarantee: None, + }; + let edge = |producer, consumer| ExecutableDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role: EdgeRole::Input, + intermediate_schema: (**schema).clone(), + data_state: state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }; + ExecutableDag { + nodes: vec![ + node(0, ExecutableOperatorPayload::Fallback { expression: scan }), + node( + 1, + ExecutableOperatorPayload::Value { + operation: ValueOperation::Sort { + keys: vec![planner_types::pre_asap::SortKey { + expr: QueryExpr::Column(0), + ascending: false, + nulls_first: false, + }], + partition_by: GroupKeys::by(vec![]), + }, + }, + ), + node( + 2, + ExecutableOperatorPayload::Value { + operation: ValueOperation::Limit { + n: 2, + offset: 0, + partition_by: GroupKeys::by(vec![]), + }, + }, + ), + ], + edges: vec![edge(0, 1), edge(1, 2)], + root: PostAsapNodeId(2), + } +} +fn registry(source: Arc) -> DataSources { + let mut r = DataSources::default(); + r.register( + Source::Table { + table_ref: "numbers".into(), + }, + source, + ) + .unwrap(); + r +} +fn context() -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits::default(), + ) + .unwrap() +} + +// Raw-only execution filters nulls and ranks across batches at either phase. +#[test] +fn raw_scan_to_sort_limit_at_both_phases() { + let (scan, schema, batches) = fixture(); + let sources = registry(Arc::new( + MemorySource::new(schema.clone(), batches).unwrap(), + )); + for state in [ + ExecutionDataState::QUERY_ROWS, + ExecutionDataState::INGESTION_ROWS, + ] { + let dag = plan(scan.clone(), &schema, state); + let bound = bind_with_data_sources(&dag, BTreeMap::new(), &[2], &sources).unwrap(); + let ctx = if state == ExecutionDataState::QUERY_ROWS { + context() + } else { + RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1, + revision: 1, + }, + Limits::default(), + ) + .unwrap() + }; + let rows = block_on(async { + let mut output = bound.execute(&[2], ctx.clone()).unwrap().remove(0); + let mut rows = vec![]; + while let Some(batch) = output.next().await { + rows.extend(batch.unwrap().rows().iter().cloned()); + } + rows + }); + assert!( + matches!(rows.as_slice(), [a,b] if matches!(a.as_slice(), [Value::Int64(9)]) && matches!(b.as_slice(), [Value::Int64(3)])) + ); + assert_eq!(ctx.retained_bytes(), 0); + } +} +struct CountingSource { + schema: Schema, + opened: Arc, + fail: bool, +} +impl RawSource for CountingSource { + fn schema(&self) -> Schema { + self.schema.clone() + } + fn scan(&self, _: RunContext) -> Result, Error> { + self.opened.fetch_add(1, Ordering::SeqCst); + if self.fail { + return Err(Error::Operator("reader failed".into())); + } + Ok(stream::iter(vec![Batch::try_new( + self.schema.clone(), + vec![vec![Value::Int64(7)]], + )]) + .boxed_local()) + } +} +// Binding and cancellation do not perform I/O; fan-out opens one cursor per run. +#[test] +fn lazy_open_shared_producer_and_cancellation() { + let (scan, schema, _) = fixture(); + let opened = Arc::new(AtomicUsize::new(0)); + let sources = registry(Arc::new(CountingSource { + schema: schema.clone(), + opened: opened.clone(), + fail: false, + })); + let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let bound = bind_with_data_sources(&plan, BTreeMap::new(), &[0, 2], &sources).unwrap(); + let ctx = context(); + let streams = bound.execute(&[0, 2], ctx.clone()).unwrap(); + assert_eq!(opened.load(Ordering::SeqCst), 0); + ctx.cancel(); + drop(streams); + assert_eq!(opened.load(Ordering::SeqCst), 0); + for _ in 0..2 { + block_on(async { + let streams = bound.execute(&[0, 2], context()).unwrap(); + let all = + futures::future::join_all(streams.into_iter().map(|s| s.collect::>())).await; + assert!(all.iter().flatten().all(Result::is_ok)); + }); + } + assert_eq!(opened.load(Ordering::SeqCst), 2); +} +// Unavailable sources and unsupported predicates fail before opening any cursor. +#[test] +fn binding_errors_and_reader_errors_are_not_empty_results() { + let (mut scan, schema, _) = fixture(); + assert!(DataSources::default().bind(&scan).is_err()); + let opened = Arc::new(AtomicUsize::new(0)); + let sources = registry(Arc::new(CountingSource { + schema: schema.clone(), + opened: opened.clone(), + fail: true, + })); + if let QueryExpr::Scan { predicates, .. } = &mut scan { + predicates.push(Predicate(Rc::new(QueryExpr::Column(0)))); + } + assert!(sources.bind(&scan).is_err()); + assert_eq!(opened.load(Ordering::SeqCst), 0); + let (scan, _, _) = fixture(); + let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let bound = bind_with_data_sources(&plan, BTreeMap::new(), &[2], &sources).unwrap(); + block_on(async { + let mut stream = bound.execute(&[2], context()).unwrap().remove(0); + assert!(stream.next().await.unwrap().is_err()); + }); +} + +// Schema drift cannot enter the DAG, and connector batches obey execution limits. +#[test] +fn schema_drift_and_memory_limits_fail_the_scan() { + struct Drift { + expected: Schema, + batch: Batch, + } + impl RawSource for Drift { + fn schema(&self) -> Schema { + self.expected.clone() + } + fn scan(&self, _: RunContext) -> Result, Error> { + Ok(stream::once(async { Ok(self.batch.clone()) }).boxed_local()) + } + } + let (scan, schema, batches) = fixture(); + let mut different = (*schema).clone(); + different.fields[0].name = "wrong".into(); + let bad = Batch::try_new(Arc::new(different), vec![vec![Value::Int64(1)]]).unwrap(); + let sources = registry(Arc::new(Drift { + expected: schema.clone(), + batch: bad, + })); + let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + block_on(async { + let mut s = graph.execute(&[0], context()).unwrap().remove(0); + assert!(s.next().await.unwrap().is_err()); + }); + let sources = registry(Arc::new(MemorySource::new(schema, batches).unwrap())); + let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + let ctx = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits { + max_bytes: 1, + max_buffered_batches: 1, + }, + ) + .unwrap(); + block_on(async { + let mut s = graph.execute(&[0], ctx.clone()).unwrap().remove(0); + assert!(s.next().await.unwrap().is_err()); + }); + assert_eq!(ctx.retained_bytes(), 0); +} + +// An empty table is a valid empty scan; nullable comparisons retain only TRUE. +#[test] +fn empty_sources_and_three_valued_predicates() { + use planner_types::pre_asap::{CompareOpKind, ScalarValue}; + let (mut scan, schema, batches) = fixture(); + if let QueryExpr::Scan { + predicates, source, .. + } = &mut scan + { + *source = Source::TimeSeries { + metric: "samples".into(), + }; + *predicates = vec![Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Gt, + right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(2))), + }))]; + } + for (batches, expected) in [(vec![], 0), (batches, 2)] { + let mut sources = DataSources::default(); + sources + .register( + Source::TimeSeries { + metric: "samples".into(), + }, + Arc::new(MemorySource::new(schema.clone(), batches).unwrap()), + ) + .unwrap(); + let plan = plan(scan.clone(), &schema, ExecutionDataState::QUERY_ROWS); + let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + block_on(async { + let mut s = graph.execute(&[0], context()).unwrap().remove(0); + let mut count = 0; + while let Some(b) = s.next().await { + count += b.unwrap().rows().len(); + } + assert_eq!(count, expected); + }); + } +} diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index 9fa0aa6c..59e47e2b 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -17,6 +17,67 @@ Its native operators can execute independently of either backend engine. Engine integration must use these operators for computation, rather than merely using the shared scheduler around a second implementation. +## Engine and storage architecture + +```mermaid +flowchart TB + subgraph Engine[ASAP Query Engine or Precompute Engine] + Planner[ASAPPlanner] --> Plan[Physical DAG] + Plan --> Runtime[ASAP Runtime] + Runtime --> Operators[Physical Operator Library] + end + Operators --> API[Data Source / Storage API] + API --> Connectors[Connectors / Adapters] + Connectors --> Systems[Storage / Data Systems] +``` + +Planner describes source identities, schemas and computation. The runtime +schedules operators and owns shared execution, backpressure, cancellation and +resource accounting. Scan reads raw rows through the data-source interface; +Filter, Project, joins, aggregation and summary construction execute inside ASAP. +The same interface serves ingestion time and query time. + +Connectors own system-specific access and decoding. Prometheus, S3, Parquet, +Kafka and Iceberg are possible integrations, not built-in dependencies or claims +of implemented support. A file format adapter and a storage transport can be +composed; they need not each be a separate query engine. Asking an external +system to execute the complete query remains external execution, not raw Scan. + +The shared API exchanges typed native batches. Ordinary columns preserve +Planner types; other operator edges can carry ASAP summary states without an +Arrow representation. Raw Scan accepts ordinary rows only. Reading stored +summary state remains a distinct storage operation, with its own compatibility +and coverage requirements. + +### Raw Scan contract + +The library provides Scan, a registry keyed by Planner table/time-series source +identity, and an immutable in-memory connector. A deployment registers its +connectors before binding the physical DAG. Binding resolves metadata and +validates schemas and predicates without opening a reader. Execution lazily +opens one cursor per reachable Scan per run, even when multiple consumers share +that node. Each run gets a fresh cursor; dropping it releases connector resources. + +Scan evaluates Planner leaf predicates itself with three-valued boolean logic: +only TRUE retains a row. Predicate pushdown is not assumed. Projection, time-range +selection and aggregation remain explicit downstream operations; Scan does not +silently interpret query evaluation time as a lookback or replay Kafka offsets. +Connectors receive the execution context for cancellation and resource control; +a deployment must bind any required snapshot/offset and bound its I/O buffers. +Reader errors and schema drift fail execution, rather than becoming empty results. + +The current post-ASAP IR retains raw Scan as a leaf expression in its `Fallback` +payload. The source-aware binder recognizes only that Scan expression and runs +it locally; other retained expressions are still rejected. This does not invoke +an external fallback or introduce another operator vocabulary. Explicit stored +frontiers can still cut a DAG at a precomputed result. + +Acceptance includes a raw-only Scan → Sort → Limit DAG at both phases, shared +consumers, independent runs, cancellation before opening, schema drift, reader +errors, null predicates, empty inputs and memory limits. This establishes the +library path. A backend must register a real reader before it can serve raw-only +plans; a Prometheus reader and other external connectors are not implemented here. + ## Workspace organization [DataFusion's physical-plan crate](https://github.com/apache/datafusion/tree/main/datafusion/physical-plan) @@ -32,7 +93,8 @@ execution model. | Physical operator implementations, input/output validation and streams | `asap-physical-operators` | | Shared-producer scheduling, cancellation and resource accounting | `asap-physical-operators` | | Summary state encoding | `asap_sketch_codec` | -| Sources, durable stores, publication and serving protocols | Deployment repositories | +| Scan and data-source interface; reference memory connector | `asap-physical-operators` | +| External connectors, durable stores, publication and serving protocols | Deployment repositories | The IR crate does not depend on execution. The physical operator crate depends on the local IR crate. Operator unit tests can exercise private implementation @@ -61,7 +123,7 @@ operators currently have no spill implementation. ## Operator coverage -Native operations include scalar sources, typed Project and Filter, arithmetic +Native operations include raw Scan and scalar sources, typed Project and Filter, arithmetic and boolean expressions, exact grouped aggregation, relational joins (including semi-join), grouped Sort and Limit, Union, vector-to-scalar conversion, and summary construction, merge and readout. Window operators consume Planner aggregate intents for Rate, Increase, @@ -83,9 +145,9 @@ implicit external fallback. The backend retains source, storage, publication and Computation must bind to Planner operations without a second backend operator vocabulary. Binding rejects unsupported Planner nodes before starting sources. -Deployments provide explicit storage or ingestion source frontiers. A supplied -batch source is not a backend raw Scan implementation. Local backend raw Scan -is deferred; a raw-only library test does not establish that deployment capability. +Deployments provide explicit storage or ingestion source frontiers and may bind +raw Scan through the shared data-source interface. The memory connector proves +the library contract; backend raw-data access still requires a deployment connector. ## DataFusion reuse vs independent implementation From cc04eb0fdadda8aa3ec656d069e413bdf65588e7 Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 16:05:22 +0000 Subject: [PATCH 10/90] feat: execute weighted CMS summaries in the shared DAG runtime --- Cargo.lock | 2 + crates/asap-physical-operators/Cargo.toml | 2 + .../src/dag/operators.rs | 210 ++++++++++- .../src/dag/planner.rs | 52 ++- .../asap-physical-operators/src/dag/values.rs | 46 ++- crates/asap-physical-operators/src/factory.rs | 73 ++-- crates/asap-physical-operators/src/lib.rs | 2 +- .../src/stored_state/decoders.rs | 2 +- .../src/stored_state/delta_apply.rs | 24 +- .../count_min_sketch_accumulator.rs | 4 +- .../count_min_sketch_with_heap_accumulator.rs | 0 .../count_sketch_accumulator.rs | 4 +- .../count_sketch_with_heap_accumulator.rs | 2 +- .../datasketches_kll_accumulator.rs | 2 +- .../dd_sketch_accumulator.rs | 2 +- .../exact_accumulator.rs | 0 .../hll_sketch_accumulator.rs | 4 +- .../hydra_kll_accumulator.rs | 0 .../increase_accumulator.rs | 0 .../keyed_counter_state.rs | 2 +- .../keyed_max_state.rs | 0 .../keyed_min_state.rs | 0 .../keyed_sum_count_accumulator.rs | 0 .../max_accumulator.rs | 0 .../min_accumulator.rs | 0 .../mod.rs | 2 + .../sketch_envelope_accumulator.rs | 0 .../sum_accumulator.rs | 0 .../univmon_accumulator.rs | 0 .../src/summary_operators/weighted_cms.rs | 331 ++++++++++++++++++ .../tests/physical_dag.rs | 111 +++++- .../tests/weighted_topk_binding.rs | 251 +++++++++++++ docs/design_docs/physical-operators.md | 20 +- 33 files changed, 1065 insertions(+), 83 deletions(-) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/count_min_sketch_accumulator.rs (99%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/count_min_sketch_with_heap_accumulator.rs (100%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/count_sketch_accumulator.rs (99%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/count_sketch_with_heap_accumulator.rs (99%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/datasketches_kll_accumulator.rs (99%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/dd_sketch_accumulator.rs (99%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/exact_accumulator.rs (100%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/hll_sketch_accumulator.rs (99%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/hydra_kll_accumulator.rs (100%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/increase_accumulator.rs (100%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/keyed_counter_state.rs (99%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/keyed_max_state.rs (100%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/keyed_min_state.rs (100%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/keyed_sum_count_accumulator.rs (100%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/max_accumulator.rs (100%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/min_accumulator.rs (100%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/mod.rs (98%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/sketch_envelope_accumulator.rs (100%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/sum_accumulator.rs (100%) rename crates/asap-physical-operators/src/{accumulators => summary_operators}/univmon_accumulator.rs (100%) create mode 100644 crates/asap-physical-operators/src/summary_operators/weighted_cms.rs create mode 100644 crates/asap-physical-operators/tests/weighted_topk_binding.rs diff --git a/Cargo.lock b/Cargo.lock index 6cd77629..5a54ee8a 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -378,6 +378,8 @@ dependencies = [ name = "asap-physical-operators" version = "0.1.0" dependencies = [ + "asap-aware-mapping", + "asap-frontend-promql", "asap-types", "asap_sketch_codec", "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=026cd18c7b8c23ae6c46d4d683151ba562b8cd3a)", diff --git a/crates/asap-physical-operators/Cargo.toml b/crates/asap-physical-operators/Cargo.toml index 8fe7eb31..e2314c77 100644 --- a/crates/asap-physical-operators/Cargo.toml +++ b/crates/asap-physical-operators/Cargo.toml @@ -24,3 +24,5 @@ extra_debugging = [] [dev-dependencies] hex = "0.4" +asap-aware-mapping = { path = "../asap-aware-mapping" } +asap-frontend-promql = { path = "../frontend-promql" } diff --git a/crates/asap-physical-operators/src/dag/operators.rs b/crates/asap-physical-operators/src/dag/operators.rs index 1057e99e..38e2f4d3 100644 --- a/crates/asap-physical-operators/src/dag/operators.rs +++ b/crates/asap-physical-operators/src/dag/operators.rs @@ -335,6 +335,16 @@ enum Kind { time: Option, groups: Vec, }, + KeyedSummaryBuild { + family: SummaryFamilyType, + value: usize, + items: Vec, + groups: Vec, + }, + KeyedReadout { + state: usize, + k: usize, + }, SummaryMerge { state: usize, groups: Vec, @@ -670,6 +680,97 @@ impl Operator { output: schema(fields), }) } + pub fn keyed_summary_build( + input: Schema, + family: SummaryFamilyType, + value: usize, + items: Vec, + groups: Vec, + ) -> Result { + use planner_types::post_asap::{SketchAlgorithm, SketchParams}; + super::values::validate_family(&family)?; + let SummaryFamilyType::Sketch(kind, _) = &family else { + return Err(invalid("keyed sketch required")); + }; + if kind.algorithm() != &SketchAlgorithm::CmsWithHeap + || !matches!(kind.params(), SketchParams::CmsWithHeap { .. }) + { + return Err(invalid( + "Float64 weighted keyed construction currently supports CMS with heap", + )); + } + validate_groups(&input, &groups)?; + if items.is_empty() || plain(&input, value)? != (&DataType::Float64, false) { + return Err(invalid( + "keyed summary requires identities and non-null Float64 weights", + )); + } + for &item in &items { + if !matches!( + plain(&input, item)?.0, + DataType::Utf8 + | DataType::Int64 + | DataType::Float64 + | DataType::Bool + | DataType::Null + ) { + return Err(invalid("unsupported keyed summary identity type")); + } + } + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }); + Ok(Self { + kind: Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn keyed_readout( + input: Schema, + state: usize, + k: usize, + output: Schema, + ) -> Result { + use planner_types::post_asap::{SketchAlgorithm, SketchParams}; + let SummaryFamilyType::Sketch(kind, _) = &field(&input, state)?.dtype else { + return Err(invalid("keyed readout requires summary state")); + }; + let SketchParams::CmsWithHeap { heap_size, .. } = kind.params() else { + return Err(invalid("unsupported keyed readout family")); + }; + if kind.algorithm() != &SketchAlgorithm::CmsWithHeap + || k > *heap_size as usize + || output.fields.len() <= input.fields.len() + { + return Err(invalid("invalid keyed readout shape or capacity")); + } + if state + 1 != input.fields.len() + || output.fields[..state] != input.fields[..state] + || output.fields.last().unwrap().dtype != SummaryFamilyType::Plain(DataType::Float64) + { + return Err(invalid( + "keyed readout must preserve partitions and return a Float64 score", + )); + } + super::values::validate_schema(&output)?; + Ok(Self { + kind: Kind::KeyedReadout { state, k }, + inputs: vec![input], + output, + }) + } pub fn summary_merge(input: Schema, state: usize, groups: Vec) -> Result { validate_groups(&input, &groups)?; super::values::validate_family(&field(&input, state)?.dtype)?; @@ -785,7 +886,8 @@ impl PhysicalOperator for Operator { Kind::Window { .. } => "WindowAggregate", Kind::SemiJoin { .. } => "SemiJoin", Kind::Join { .. } => "RelationalJoin", - Kind::SummaryBuild { .. } => "SummaryAgg", + Kind::SummaryBuild { .. } | Kind::KeyedSummaryBuild { .. } => "SummaryAgg", + Kind::KeyedReadout { .. } => "SummaryEstimate", Kind::SummaryMerge { .. } => "SummaryMerge", Kind::Readout { .. } => "SummaryReadout", } @@ -1018,6 +1120,39 @@ impl PhysicalOperator for Operator { ) }) .boxed_local()), + Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + } => Ok(futures::stream::once(async move { + Batch::try_new( + output, + build_keyed_summary(input, family, *value, items, groups, &context).await?, + ) + }) + .boxed_local()), + Kind::KeyedReadout { state, k } => Ok(input + .map(move |batch| { + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + let Value::Summary { state: summary, .. } = &row[*state] else { + return Err(invalid("summary value required")); + }; + let summary = summary + .as_any() + .downcast_ref::() + .ok_or_else(|| invalid("weighted CMS typed state required"))?; + for items in summary.rows(*k) { + let mut values = row[..*state].to_vec(); + values.extend(items); + rows.push(values); + } + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), Kind::Readout { state, statistic, @@ -1254,6 +1389,79 @@ fn reduce_one(rows: &[Vec], measure: &Reduction, input: &Schema) -> Resul sum })) } +async fn build_keyed_summary( + mut input: Input<'_, Batch>, + family: &SummaryFamilyType, + value: usize, + items: &[usize], + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + use crate::{summary_operators::weighted_cms::WeightedCms, AggregateCore}; + use planner_types::post_asap::SketchParams; + let SummaryFamilyType::Sketch(kind, _) = family else { + unreachable!() + }; + let SketchParams::CmsWithHeap { + width, + depth, + heap_size, + } = kind.params() + else { + unreachable!() + }; + let mut states = BTreeMap::>, (Vec, WeightedCms, Reservation, usize)>::new(); + while let Some(batch) = input.next().await { + let batch = batch?; + for row in batch.rows() { + if context.is_cancelled() { + return Err(Error::Operator("execution cancelled".into())); + } + let key = group_key(row, groups)?; + if !states.contains_key(&key) { + let labels = groups.iter().map(|&i| row[i].clone()).collect::>(); + let overhead = labels.iter().map(Value::bytes).sum::() + + key.iter().map(|v| v.len() + 24).sum::() + + 128; + let bytes = (*width as usize) + .checked_mul(*depth as usize) + .and_then(|n| n.checked_mul(8)) + .and_then(|n| n.checked_add(overhead)) + .ok_or_else(|| invalid("weighted CMS memory size overflow"))?; + let reservation = context.reserve(bytes)?; + states.insert( + key.clone(), + ( + labels, + WeightedCms::new(*width as usize, *depth as usize, *heap_size as usize)?, + reservation, + overhead, + ), + ); + } + let (_, summary, reservation, overhead) = states.get_mut(&key).unwrap(); + let Value::Float64(weight) = row[value] else { + return Err(invalid("weighted CMS weight type")); + }; + summary.update( + &items.iter().map(|&i| row[i].clone()).collect::>(), + weight, + )?; + reservation.resize(summary.approx_memory_bytes() + *overhead)?; + } + } + Ok(states + .into_values() + .map(|(mut labels, summary, _, _)| { + labels.push(Value::Summary { + family: family.clone(), + state: Arc::new(summary), + }); + labels + }) + .collect()) +} + async fn build_summary( mut input: Input<'_, Batch>, family: &SummaryFamilyType, diff --git a/crates/asap-physical-operators/src/dag/planner.rs b/crates/asap-physical-operators/src/dag/planner.rs index 7b801c32..04143122 100644 --- a/crates/asap-physical-operators/src/dag/planner.rs +++ b/crates/asap-physical-operators/src/dag/planner.rs @@ -317,8 +317,48 @@ fn bind_operation(node: &ExecutableDagNode, inputs: &[Schema]) -> Result { - if update.item.is_some() { - return Err(invalid("keyed summary update binding is not implemented")); + if let Some(item) = &update.item { + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid("keyed summary requires explicit partitions")); + }; + let SummaryInputExpr::Column(weight) = &update.weight else { + return Err(invalid( + "keyed summary weight must be a finalized value column", + )); + }; + if !matches!( + update.weight_domain, + planner_types::post_asap::WeightDomain::NonNegative { .. } + ) { + return Err(invalid("CMS requires a nonnegative weight contract")); + } + fn columns( + expr: &SummaryInputExpr, + input: &Schema, + result: &mut Vec, + ) -> Result<(), Error> { + match expr { + SummaryInputExpr::Column(column) => { + result.push(named_column(input, column)?) + } + SummaryInputExpr::Tuple(items) => { + for item in items { + columns(item, input, result)?; + } + } + _ => return Err(invalid("keyed summary needs explicit item columns")), + } + Ok(()) + } + let mut items = Vec::new(); + columns(item, input, &mut items)?; + return Operator::keyed_summary_build( + input.clone(), + family.clone(), + named_column(input, weight)?, + items, + groups(input, keys)?, + ); } crate::capability::validate_summary_kernel(family, update, grouping) .map_err(Error::Invalid)?; @@ -351,6 +391,14 @@ fn bind_operation(node: &ExecutableDagNode, inputs: &[Schema]) -> Result { + if let SketchQuery::TopK { k } = query { + return Operator::keyed_readout( + input.clone(), + summary_column(input)?, + *k, + Arc::new(node.output_schema.clone()), + ); + } let mut params = std::collections::HashMap::new(); let statistic = match query { SketchQuery::Quantile { q } => { diff --git a/crates/asap-physical-operators/src/dag/values.rs b/crates/asap-physical-operators/src/dag/values.rs index 74c49b95..9e403084 100644 --- a/crates/asap-physical-operators/src/dag/values.rs +++ b/crates/asap-physical-operators/src/dag/values.rs @@ -258,6 +258,27 @@ pub(crate) fn group_key(row: &[Value], columns: &[usize]) -> Result> pub(crate) fn validate_family(family: &SummaryFamilyType) -> Result<(), Error> { use planner_types::post_asap::SketchAlgorithm as A; + if let SummaryFamilyType::Sketch(kind, grouping) = family { + if let planner_types::post_asap::SketchParams::CmsWithHeap { + width, + depth, + heap_size, + } = kind.params() + { + return if kind.algorithm() == &A::CmsWithHeap + && *width > 0 + && *depth > 0 + && *heap_size > 0 + && grouping == &Default::default() + { + Ok(()) + } else { + Err(Error::Invalid( + "invalid weighted CMS family or grouping strategy".into(), + )) + }; + } + } match family { SummaryFamilyType::ExactAggregate(..) => {} SummaryFamilyType::Sketch(kind, _) @@ -278,7 +299,7 @@ pub(crate) fn validate_family(family: &SummaryFamilyType) -> Result<(), Error> { .map_err(Error::Invalid) } fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Result<(), Error> { - use crate::accumulators::{ + use crate::summary_operators::{ datasketches_kll_accumulator::DatasketchesKLLAccumulator, dd_sketch_accumulator::DDSketchAccumulator, exact_accumulator::ExactAccumulator, hll_sketch_accumulator::HllSketchAccumulator, @@ -286,6 +307,25 @@ fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Resu use planner_types::post_asap::SketchParams; validate_family(family)?; let valid = match family { + SummaryFamilyType::Sketch(kind, _) + if matches!(kind.params(), SketchParams::CmsWithHeap { .. }) => + { + let SketchParams::CmsWithHeap { + width, + depth, + heap_size, + } = kind.params() + else { + unreachable!() + }; + state + .as_any() + .downcast_ref::() + .is_some_and(|state| { + state.shape() == (*width as usize, *depth as usize, *heap_size as usize) + }) + } + SummaryFamilyType::ExactAggregate(..) => { state .as_any() @@ -297,7 +337,9 @@ fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Resu planner_types::post_asap::ExactKind::Sum, planner_types::post_asap::ExactParams::Sum ) - ) && state.as_any().is::()) + ) && state + .as_any() + .is::()) } SummaryFamilyType::Sketch(kind, _) => match kind.params() { SketchParams::Kll { k } => state diff --git a/crates/asap-physical-operators/src/factory.rs b/crates/asap-physical-operators/src/factory.rs index 2a44e0f1..30f74971 100644 --- a/crates/asap-physical-operators/src/factory.rs +++ b/crates/asap-physical-operators/src/factory.rs @@ -1,4 +1,4 @@ -use crate::accumulators::{ +use crate::summary_operators::{ CountMinSketchAccumulator, CountMinSketchWithHeapAccumulator, CountSketchAccumulator, CountSketchWithHeapAccumulator, DDSketchAccumulator, DatasketchesKLLAccumulator, HydraKllSketchAccumulator, IncreaseAccumulator, KeyedCounterState, KeyedMaxState, @@ -7,8 +7,8 @@ use crate::accumulators::{ use crate::{AggregateCore, KeyByLabelValues, Measurement}; // Production dispatch consumes Planner SummaryAgg payloads directly. The // config adapter below is compiled only for isolated historical kernel tests. -use crate::accumulators::hll_sketch_accumulator::HllSketchAccumulator; -use crate::accumulators::univmon_accumulator::UnivMonAccumulator; +use crate::summary_operators::hll_sketch_accumulator::HllSketchAccumulator; +use crate::summary_operators::univmon_accumulator::UnivMonAccumulator; use planner_types::post_asap::{ExactKind, SketchAlgorithm, SketchParams, SummaryFamilyType}; /// Generate the two boilerplate clone-based `AccumulatorUpdater` methods @@ -630,28 +630,16 @@ pub struct CmsHeapAccumulatorUpdater { col_num: usize, heap_size: usize, weight: TopkWeight, - weight_scale: f64, } impl CmsHeapAccumulatorUpdater { pub fn new(row_num: usize, col_num: usize, heap_size: usize, weight: TopkWeight) -> Self { - Self::with_weight_scale(row_num, col_num, heap_size, weight, 1.0) - } - - pub fn with_weight_scale( - row_num: usize, - col_num: usize, - heap_size: usize, - weight: TopkWeight, - weight_scale: f64, - ) -> Self { Self { acc: CountMinSketchWithHeapAccumulator::new(row_num, col_num, heap_size), row_num, col_num, heap_size, weight, - weight_scale, } } } @@ -671,7 +659,7 @@ impl AccumulatorUpdater for CmsHeapAccumulatorUpdater { // Σ value: feed the datapoint value. sketchlib's CMS-heap // `update(key, w)` adds `w.round()` occurrences of `key`, so the // heap value accumulates the (rounded) summed metric value. - TopkWeight::Value => value * self.weight_scale, + TopkWeight::Value => value, // Σ count: one occurrence per event, regardless of value. TopkWeight::Count => 1.0, }; @@ -767,28 +755,16 @@ pub struct CountSketchWithHeapAccumulatorUpdater { col_num: usize, heap_size: usize, weight: TopkWeight, - weight_scale: f64, } impl CountSketchWithHeapAccumulatorUpdater { pub fn new(row_num: usize, col_num: usize, heap_size: usize, weight: TopkWeight) -> Self { - Self::with_weight_scale(row_num, col_num, heap_size, weight, 1.0) - } - - pub fn with_weight_scale( - row_num: usize, - col_num: usize, - heap_size: usize, - weight: TopkWeight, - weight_scale: f64, - ) -> Self { Self { acc: CountSketchWithHeapAccumulator::new(row_num, col_num, heap_size), row_num, col_num, heap_size, weight, - weight_scale, } } } @@ -803,7 +779,7 @@ impl AccumulatorUpdater for CountSketchWithHeapAccumulatorUpdater { fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { let weighted = match self.weight { - TopkWeight::Value => value * self.weight_scale, + TopkWeight::Value => value, TopkWeight::Count => 1.0, }; self.acc.inner.update(&key.to_semicolon_str(), weighted); @@ -918,6 +894,18 @@ pub fn create_planner_accumulator( input: &planner_types::post_asap::SummaryUpdate, grouping: &planner_types::post_asap::GroupingStrategy, ) -> Result, String> { + if input.item.is_some() + && matches!( + input.weight_domain, + planner_types::post_asap::WeightDomain::NonNegative { + proof: + planner_types::post_asap::NonNegativeWeightProof::ResetAwareCounterDerivative + } + ) + { + return Err("window-weighted summaries require typed DAG binding; integer heap updaters cannot consume rates".into()); + } + crate::capability::validate_summary_kernel(family, input, grouping)?; use planner_types::post_asap::GroupingStrategy; if grouping != &GroupingStrategy::PerSubpopulationInstance { @@ -925,7 +913,7 @@ pub fn create_planner_accumulator( } if matches!(family, SummaryFamilyType::ExactAggregate(..)) { return Ok(Box::new(PlannerExactUpdater { - acc: crate::accumulators::exact_accumulator::ExactAccumulator::new( + acc: crate::summary_operators::exact_accumulator::ExactAccumulator::new( family.clone(), input.item.is_some(), )?, @@ -937,16 +925,6 @@ pub fn create_planner_accumulator( if family_grouping != grouping { return Err("Planner family and operator grouping disagree".into()); } - // Heap counters use fixed-point storage for fractional counter deltas. - // This encodes the selected update; it does not choose another family. - let weight_scale = if matches!( - input.weight, - planner_types::post_asap::SummaryInputExpr::ResetAwareCounterDelta { .. } - ) { - 1_000_000.0 - } else { - 1.0 - }; let updater: Box = match (kind.algorithm(), kind.params()) { (SketchAlgorithm::Kll, SketchParams::Kll { k }) => Box::new(KllAccumulatorUpdater::new( u16::try_from(*k).map_err(|_| "KLL k exceeds runtime bound")?, @@ -964,25 +942,18 @@ pub fn create_planner_accumulator( } (SketchAlgorithm::CmsWithHeap, params @ SketchParams::CmsWithHeap { .. }) => { let (r, c, h) = cms_heap_dims(params); - Box::new(CmsHeapAccumulatorUpdater::with_weight_scale( - r, - c, - h, - TopkWeight::Value, - weight_scale, - )) + Box::new(CmsHeapAccumulatorUpdater::new(r, c, h, TopkWeight::Value)) } ( SketchAlgorithm::CountSketchWithHeap, params @ SketchParams::CountSketchWithHeap { .. }, ) => { let (r, c, h) = cms_heap_dims(params); - Box::new(CountSketchWithHeapAccumulatorUpdater::with_weight_scale( + Box::new(CountSketchWithHeapAccumulatorUpdater::new( r, c, h, TopkWeight::Value, - weight_scale, )) } (SketchAlgorithm::Hll, SketchParams::Hll { precision }) => Box::new(HllUpdater { @@ -1023,7 +994,7 @@ pub fn create_planner_accumulator( } struct PlannerExactUpdater { - acc: crate::accumulators::exact_accumulator::ExactAccumulator, + acc: crate::summary_operators::exact_accumulator::ExactAccumulator, } impl AccumulatorUpdater for PlannerExactUpdater { fn update_single(&mut self, value: f64, timestamp: i64) { @@ -1034,7 +1005,7 @@ impl AccumulatorUpdater for PlannerExactUpdater { } impl_clone_accumulator_methods!(acc); fn reset(&mut self) { - self.acc = crate::accumulators::exact_accumulator::ExactAccumulator::new( + self.acc = crate::summary_operators::exact_accumulator::ExactAccumulator::new( self.acc.family().clone(), self.acc.is_keyed(), ) diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs index 3896eb7a..3cf29efa 100644 --- a/crates/asap-physical-operators/src/lib.rs +++ b/crates/asap-physical-operators/src/lib.rs @@ -1,8 +1,8 @@ #![doc = include_str!("../README.md")] -pub mod accumulators; pub mod key_by_label_values; pub mod measurement; +pub mod summary_operators; pub mod traits; mod aggregation_type; diff --git a/crates/asap-physical-operators/src/stored_state/decoders.rs b/crates/asap-physical-operators/src/stored_state/decoders.rs index 99bde575..3706c9c3 100644 --- a/crates/asap-physical-operators/src/stored_state/decoders.rs +++ b/crates/asap-physical-operators/src/stored_state/decoders.rs @@ -8,7 +8,7 @@ use asap_sketchlib::CountSketchWithHeap; use asap_sketchlib::CsHeapItem; use asap_sketchlib::MessagePackCodec; -use crate::accumulators::count_min_sketch_with_heap_accumulator::CountMinSketchWithHeapAccumulator; +use crate::summary_operators::count_min_sketch_with_heap_accumulator::CountMinSketchWithHeapAccumulator; /// Decode a `CountMinSketch` from the modified-OTLP wire bytes. /// MSGPACK path round-trips `CountMinSketch::deserialize_msgpack`; diff --git a/crates/asap-physical-operators/src/stored_state/delta_apply.rs b/crates/asap-physical-operators/src/stored_state/delta_apply.rs index dd9ec4d3..03467ab3 100644 --- a/crates/asap-physical-operators/src/stored_state/delta_apply.rs +++ b/crates/asap-physical-operators/src/stored_state/delta_apply.rs @@ -84,7 +84,7 @@ impl DeltaSketchKind { sketch_cols, layers, } => SummaryState::UnivMon( - crate::accumulators::univmon_accumulator::UnivMonAccumulator::new( + crate::summary_operators::univmon_accumulator::UnivMonAccumulator::new( *heap_size as usize, *sketch_rows as usize, *sketch_cols as usize, @@ -137,8 +137,10 @@ fn decode_full( SketchEncoding::MsgpackFull, ) => { let state = - crate::accumulators::univmon_accumulator::UnivMonAccumulator::from_bytes(bytes) - .map_err(|e| e.to_string())?; + crate::summary_operators::univmon_accumulator::UnivMonAccumulator::from_bytes( + bytes, + ) + .map_err(|e| e.to_string())?; if state.dimensions() != ( *heap_size as usize, @@ -214,7 +216,7 @@ fn decode_full( /// folded across a window (or several) via delta application, or merged /// in from another sid's own reconstruction. pub enum SummaryState { - UnivMon(crate::accumulators::univmon_accumulator::UnivMonAccumulator), + UnivMon(crate::summary_operators::univmon_accumulator::UnivMonAccumulator), Dd(DdSketch), Hll(HllSketch), Kll(KllSketch), @@ -291,7 +293,7 @@ impl SummaryState { } // Shape (2): bucket-delta proto → additive apply via the // SAME decoder the ingest delta path uses. - use crate::accumulators::dd_sketch_accumulator::DDSketchAccumulator; + use crate::summary_operators::dd_sketch_accumulator::DDSketchAccumulator; let mut acc = DDSketchAccumulator { inner: std::mem::replace(sk, DdSketch::new(sk.alpha)), sample_p: 1.0, @@ -710,21 +712,21 @@ pub fn per_window_summary_states( // --------------------------------------------------------------------------- fn dd_from_proto(buffer: &[u8]) -> Result { - use crate::accumulators::dd_sketch_accumulator::DDSketchAccumulator; + use crate::summary_operators::dd_sketch_accumulator::DDSketchAccumulator; DDSketchAccumulator::from_sketchlib_proto_bytes(buffer) .map(|acc| acc.inner) .map_err(|e| e.to_string()) } fn kll_from_proto(buffer: &[u8]) -> Result { - use crate::accumulators::datasketches_kll_accumulator::DatasketchesKLLAccumulator; + use crate::summary_operators::datasketches_kll_accumulator::DatasketchesKLLAccumulator; DatasketchesKLLAccumulator::from_sketchlib_proto_bytes(buffer) .map(|acc| acc.inner) .map_err(|e| e.to_string()) } fn hll_from_proto(buffer: &[u8]) -> Result { - use crate::accumulators::hll_sketch_accumulator::HllSketchAccumulator; + use crate::summary_operators::hll_sketch_accumulator::HllSketchAccumulator; HllSketchAccumulator::from_sketchlib_proto_bytes(buffer) .map(|acc| acc.inner) .map_err(|e| e.to_string()) @@ -880,7 +882,7 @@ mod tests { fn hll_from_proto_matches_accumulator_decoder() { // P2-4: the warm read path and the ingest accumulator must decode // the SAME bytes to the SAME sketch (one source of truth). - use crate::accumulators::hll_sketch_accumulator::HllSketchAccumulator; + use crate::summary_operators::hll_sketch_accumulator::HllSketchAccumulator; let mut sk = HllSketch::new(HllVariant::Regular, 12); for i in 0..500u64 { sk.update(format!("item-{i}").as_bytes()); @@ -899,7 +901,7 @@ mod tests { #[test] fn dd_from_proto_matches_accumulator_decoder() { - use crate::accumulators::dd_sketch_accumulator::DDSketchAccumulator; + use crate::summary_operators::dd_sketch_accumulator::DDSketchAccumulator; let mut sk = DdSketch::new(0.01); for v in [1.0, 2.0, 5.0, 5.0, 9.0, 42.0] { sk.update(v); @@ -916,7 +918,7 @@ mod tests { #[test] fn kll_from_proto_matches_accumulator_decoder() { - use crate::accumulators::datasketches_kll_accumulator::DatasketchesKLLAccumulator; + use crate::summary_operators::datasketches_kll_accumulator::DatasketchesKLLAccumulator; let items: Vec = (0..200).map(|i| i as f64).collect(); let bytes = encode_kll(256, &items); let via_delta = kll_from_proto(&bytes).expect("delta_apply kll decode"); diff --git a/crates/asap-physical-operators/src/accumulators/count_min_sketch_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/count_min_sketch_accumulator.rs similarity index 99% rename from crates/asap-physical-operators/src/accumulators/count_min_sketch_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/count_min_sketch_accumulator.rs index e9d7fd64..fe63e3f3 100644 --- a/crates/asap-physical-operators/src/accumulators/count_min_sketch_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/count_min_sketch_accumulator.rs @@ -1,4 +1,4 @@ -use crate::accumulators::dd_sketch_accumulator::normalize_sample_p; +use crate::summary_operators::dd_sketch_accumulator::normalize_sample_p; use crate::{ AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, MultipleSubpopulationAggregate, SerializableToSink, @@ -810,7 +810,7 @@ mod tests { let boxed_accs: Vec> = vec![Box::new(cms1), Box::new(cms2)]; assert!(CountMinSketchAccumulator::merge_multiple(&boxed_accs).is_err()); - use crate::accumulators::sum_accumulator::SumAccumulator; + use crate::summary_operators::sum_accumulator::SumAccumulator; let cms = CountMinSketchAccumulator::new(2, 3); let sum = SumAccumulator::new(); let mixed_accs: Vec> = vec![Box::new(cms), Box::new(sum)]; diff --git a/crates/asap-physical-operators/src/accumulators/count_min_sketch_with_heap_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/count_min_sketch_with_heap_accumulator.rs similarity index 100% rename from crates/asap-physical-operators/src/accumulators/count_min_sketch_with_heap_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/count_min_sketch_with_heap_accumulator.rs diff --git a/crates/asap-physical-operators/src/accumulators/count_sketch_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/count_sketch_accumulator.rs similarity index 99% rename from crates/asap-physical-operators/src/accumulators/count_sketch_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/count_sketch_accumulator.rs index fc74a018..78cdb61b 100644 --- a/crates/asap-physical-operators/src/accumulators/count_sketch_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/count_sketch_accumulator.rs @@ -96,7 +96,7 @@ impl CountSketchAccumulator { // ingest caller skips the data point) instead of building a // degenerate or huge matrix. Shares the CMS validator since the // CountSketch matrix uses the same packed-hash column layout. - crate::accumulators::count_min_sketch_accumulator::validate_sketch_dims( + crate::summary_operators::count_min_sketch_accumulator::validate_sketch_dims( "CountSketchState", rows, cols, @@ -554,7 +554,7 @@ mod tests { #[test] fn test_aggregate_core_merge_wrong_type_rejects() { - use crate::accumulators::count_min_sketch_accumulator::CountMinSketchAccumulator; + use crate::summary_operators::count_min_sketch_accumulator::CountMinSketchAccumulator; let cs = CountSketchAccumulator::new(2, 3); let cms = CountMinSketchAccumulator::new(2, 3); let result = cs.merge_with(&cms); diff --git a/crates/asap-physical-operators/src/accumulators/count_sketch_with_heap_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/count_sketch_with_heap_accumulator.rs similarity index 99% rename from crates/asap-physical-operators/src/accumulators/count_sketch_with_heap_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/count_sketch_with_heap_accumulator.rs index 59dd3df1..2a079347 100644 --- a/crates/asap-physical-operators/src/accumulators/count_sketch_with_heap_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/count_sketch_with_heap_accumulator.rs @@ -561,7 +561,7 @@ mod tests { /// min-over-rows divergence at the sketch-math level). #[test] fn test_rejects_merge_with_cms_family_accumulator() { - use crate::accumulators::count_min_sketch_with_heap_accumulator::CountMinSketchWithHeapAccumulator; + use crate::summary_operators::count_min_sketch_with_heap_accumulator::CountMinSketchWithHeapAccumulator; let cs = CountSketchWithHeapAccumulator::new(4, 64, 10); let cms = CountMinSketchWithHeapAccumulator::new(4, 64, 10); diff --git a/crates/asap-physical-operators/src/accumulators/datasketches_kll_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/datasketches_kll_accumulator.rs similarity index 99% rename from crates/asap-physical-operators/src/accumulators/datasketches_kll_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/datasketches_kll_accumulator.rs index 4ccfd3d0..4124ddf9 100644 --- a/crates/asap-physical-operators/src/accumulators/datasketches_kll_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/datasketches_kll_accumulator.rs @@ -537,7 +537,7 @@ mod tests { let boxed_accs: Vec> = vec![Box::new(kll1), Box::new(kll2)]; assert!(DatasketchesKLLAccumulator::merge_multiple(&boxed_accs).is_err()); - use crate::accumulators::sum_accumulator::SumAccumulator; + use crate::summary_operators::sum_accumulator::SumAccumulator; let kll = DatasketchesKLLAccumulator::new(200); let sum = SumAccumulator::new(); let mixed_accs: Vec> = vec![Box::new(kll), Box::new(sum)]; diff --git a/crates/asap-physical-operators/src/accumulators/dd_sketch_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/dd_sketch_accumulator.rs similarity index 99% rename from crates/asap-physical-operators/src/accumulators/dd_sketch_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/dd_sketch_accumulator.rs index 5015b50b..1e926d68 100644 --- a/crates/asap-physical-operators/src/accumulators/dd_sketch_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/dd_sketch_accumulator.rs @@ -398,7 +398,7 @@ mod tests { #[test] fn test_aggregate_core_merge_wrong_type_rejects() { - use crate::accumulators::count_sketch_accumulator::CountSketchAccumulator; + use crate::summary_operators::count_sketch_accumulator::CountSketchAccumulator; let dd = DDSketchAccumulator::new(0.01); let cs = CountSketchAccumulator::new(2, 3); assert!(dd.merge_with(&cs).is_err()); diff --git a/crates/asap-physical-operators/src/accumulators/exact_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/exact_accumulator.rs similarity index 100% rename from crates/asap-physical-operators/src/accumulators/exact_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/exact_accumulator.rs diff --git a/crates/asap-physical-operators/src/accumulators/hll_sketch_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/hll_sketch_accumulator.rs similarity index 99% rename from crates/asap-physical-operators/src/accumulators/hll_sketch_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/hll_sketch_accumulator.rs index d43737a3..0d8b714d 100644 --- a/crates/asap-physical-operators/src/accumulators/hll_sketch_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/hll_sketch_accumulator.rs @@ -11,7 +11,7 @@ //! registers + variant + HIP accumulators losslessly, so the merge + //! store round-trip works end-to-end without that richer query surface. -use crate::accumulators::dd_sketch_accumulator::normalize_sample_p; +use crate::summary_operators::dd_sketch_accumulator::normalize_sample_p; use crate::{AggregateCore, AggregationType, KeyByLabelValues, SerializableToSink}; use asap_sketchlib::{HllSketch, HllVariant, MessagePackCodec}; use serde_json::Value; @@ -567,7 +567,7 @@ mod tests { #[test] fn test_aggregate_core_merge_wrong_type_rejects() { - use crate::accumulators::count_sketch_accumulator::CountSketchAccumulator; + use crate::summary_operators::count_sketch_accumulator::CountSketchAccumulator; let hll = HllSketchAccumulator::new(HllVariant::Regular, 2); let cs = CountSketchAccumulator::new(2, 3); assert!(hll.merge_with(&cs).is_err()); diff --git a/crates/asap-physical-operators/src/accumulators/hydra_kll_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/hydra_kll_accumulator.rs similarity index 100% rename from crates/asap-physical-operators/src/accumulators/hydra_kll_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/hydra_kll_accumulator.rs diff --git a/crates/asap-physical-operators/src/accumulators/increase_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/increase_accumulator.rs similarity index 100% rename from crates/asap-physical-operators/src/accumulators/increase_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/increase_accumulator.rs diff --git a/crates/asap-physical-operators/src/accumulators/keyed_counter_state.rs b/crates/asap-physical-operators/src/summary_operators/keyed_counter_state.rs similarity index 99% rename from crates/asap-physical-operators/src/accumulators/keyed_counter_state.rs rename to crates/asap-physical-operators/src/summary_operators/keyed_counter_state.rs index 883460b2..9411088f 100644 --- a/crates/asap-physical-operators/src/accumulators/keyed_counter_state.rs +++ b/crates/asap-physical-operators/src/summary_operators/keyed_counter_state.rs @@ -1,4 +1,4 @@ -use crate::accumulators::IncreaseAccumulator; +use crate::summary_operators::IncreaseAccumulator; use crate::{ AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, MultipleSubpopulationAggregate, SerializableToSink, SingleSubpopulationAggregate, diff --git a/crates/asap-physical-operators/src/accumulators/keyed_max_state.rs b/crates/asap-physical-operators/src/summary_operators/keyed_max_state.rs similarity index 100% rename from crates/asap-physical-operators/src/accumulators/keyed_max_state.rs rename to crates/asap-physical-operators/src/summary_operators/keyed_max_state.rs diff --git a/crates/asap-physical-operators/src/accumulators/keyed_min_state.rs b/crates/asap-physical-operators/src/summary_operators/keyed_min_state.rs similarity index 100% rename from crates/asap-physical-operators/src/accumulators/keyed_min_state.rs rename to crates/asap-physical-operators/src/summary_operators/keyed_min_state.rs diff --git a/crates/asap-physical-operators/src/accumulators/keyed_sum_count_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/keyed_sum_count_accumulator.rs similarity index 100% rename from crates/asap-physical-operators/src/accumulators/keyed_sum_count_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/keyed_sum_count_accumulator.rs diff --git a/crates/asap-physical-operators/src/accumulators/max_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/max_accumulator.rs similarity index 100% rename from crates/asap-physical-operators/src/accumulators/max_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/max_accumulator.rs diff --git a/crates/asap-physical-operators/src/accumulators/min_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/min_accumulator.rs similarity index 100% rename from crates/asap-physical-operators/src/accumulators/min_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/min_accumulator.rs diff --git a/crates/asap-physical-operators/src/accumulators/mod.rs b/crates/asap-physical-operators/src/summary_operators/mod.rs similarity index 98% rename from crates/asap-physical-operators/src/accumulators/mod.rs rename to crates/asap-physical-operators/src/summary_operators/mod.rs index 073db6e8..731d1eb9 100644 --- a/crates/asap-physical-operators/src/accumulators/mod.rs +++ b/crates/asap-physical-operators/src/summary_operators/mod.rs @@ -35,3 +35,5 @@ pub use max_accumulator::*; pub use min_accumulator::*; pub use sketch_envelope_accumulator::*; pub use sum_accumulator::*; + +pub mod weighted_cms; diff --git a/crates/asap-physical-operators/src/accumulators/sketch_envelope_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/sketch_envelope_accumulator.rs similarity index 100% rename from crates/asap-physical-operators/src/accumulators/sketch_envelope_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/sketch_envelope_accumulator.rs diff --git a/crates/asap-physical-operators/src/accumulators/sum_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/sum_accumulator.rs similarity index 100% rename from crates/asap-physical-operators/src/accumulators/sum_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/sum_accumulator.rs diff --git a/crates/asap-physical-operators/src/accumulators/univmon_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/univmon_accumulator.rs similarity index 100% rename from crates/asap-physical-operators/src/accumulators/univmon_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/univmon_accumulator.rs diff --git a/crates/asap-physical-operators/src/summary_operators/weighted_cms.rs b/crates/asap-physical-operators/src/summary_operators/weighted_cms.rs new file mode 100644 index 00000000..ac365b9c --- /dev/null +++ b/crates/asap-physical-operators/src/summary_operators/weighted_cms.rs @@ -0,0 +1,331 @@ +//! Float64 weighted CMS state with typed candidate identities. Each instance +//! represents one partition at one evaluation scope; updates never round rates +//! to integer counts. Candidate membership still requires Planner evidence. +use crate::dag::{values::Value, Error}; +use crate::{AggregateCore, AggregationType, KeyByLabelValues, SerializableToSink, Statistic}; +use serde::{Deserialize, Serialize}; +use std::{ + cmp::Ordering, + collections::{BinaryHeap, HashMap}, + sync::Arc, +}; + +#[derive(Clone, Debug, Serialize, Deserialize)] +enum Identity { + Null, + Bool(bool), + Int64(i64), + Float64(f64), + Utf8(String), +} +impl Identity { + fn from_value(value: &Value) -> Result { + Ok(match value { + Value::Null => Self::Null, + Value::Bool(v) => Self::Bool(*v), + Value::Int64(v) => Self::Int64(*v), + Value::Float64(v) if v.is_finite() => Self::Float64(if *v == 0.0 { 0.0 } else { *v }), + Value::Utf8(v) => Self::Utf8(v.to_string()), + _ => return Err(Error::Invalid("unsupported weighted CMS identity".into())), + }) + } + fn value(&self) -> Value { + match self { + Self::Null => Value::Null, + Self::Bool(v) => Value::Bool(*v), + Self::Int64(v) => Value::Int64(*v), + Self::Float64(v) => Value::Float64(*v), + Self::Utf8(v) => Value::Utf8(Arc::from(v.as_str())), + } + } +} +#[derive(Clone, Debug, Serialize, Deserialize)] +struct Candidate { + identity: Vec, + key: Vec, + score: f64, +} +impl PartialEq for Candidate { + fn eq(&self, other: &Self) -> bool { + self.cmp(other) == Ordering::Equal + } +} +impl Eq for Candidate {} +impl PartialOrd for Candidate { + fn partial_cmp(&self, other: &Self) -> Option { + Some(self.cmp(other)) + } +} +impl Ord for Candidate { + fn cmp(&self, other: &Self) -> Ordering { + // The weakest candidate is the root of this bounded min-heap. + other + .score + .total_cmp(&self.score) + .then_with(|| other.key.cmp(&self.key)) + } +} +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct WeightedCms { + width: usize, + depth: usize, + capacity: usize, + cells: Vec, + candidates: BinaryHeap, +} +impl WeightedCms { + pub(crate) fn shape(&self) -> (usize, usize, usize) { + (self.width, self.depth, self.capacity) + } + pub fn new(width: usize, depth: usize, capacity: usize) -> Result { + let len = width + .checked_mul(depth) + .filter(|_| width > 0 && depth > 0 && capacity > 0) + .ok_or_else(|| Error::Invalid("invalid weighted CMS dimensions".into()))?; + let mut cells = Vec::new(); + cells + .try_reserve_exact(len) + .map_err(|_| Error::Invalid("weighted CMS allocation failed".into()))?; + cells.resize(len, 0.0); + Ok(Self { + width, + depth, + capacity, + cells, + candidates: BinaryHeap::new(), + }) + } + /// Decode only this kernel's versioned Float64 representation. Integer CMS + /// wire frames are different representations and are not accepted here. + pub fn from_bytes(bytes: &[u8]) -> Result { + use bincode::Options; + let bytes = bytes + .strip_prefix(b"ASAP-WCMS-1\0") + .ok_or_else(|| Error::Invalid("weighted CMS format/version mismatch".into()))?; + let mut state: Self = bincode::DefaultOptions::new() + .with_fixint_encoding() + .with_limit(bytes.len() as u64) + .reject_trailing_bytes() + .deserialize(bytes) + .map_err(|e| Error::Invalid(e.to_string()))?; + if state.width == 0 + || state.depth == 0 + || state.capacity == 0 + || state.width.checked_mul(state.depth) != Some(state.cells.len()) + || state.cells.iter().any(|v| !v.is_finite() || *v < 0.0) + || state.candidates.len() > state.capacity + { + return Err(Error::Invalid("invalid weighted CMS state".into())); + } + for candidate in &state.candidates { + if candidate + .identity + .iter() + .any(|v| matches!(v, Identity::Float64(n) if !n.is_finite())) + || bincode::serialize(&candidate.identity) + .map_err(|e| Error::Invalid(e.to_string()))? + != candidate.key + { + return Err(Error::Invalid("invalid weighted CMS identity".into())); + } + } + state.retain(state.candidates.iter().cloned().collect()); + Ok(state) + } + fn indexes(&self, key: &[u8]) -> impl Iterator + '_ { + let key = key.to_vec(); + (0..self.depth).map(move |row| { + row * self.width + + (xxhash_rust::xxh64::xxh64(&key, row as u64) % self.width as u64) as usize + }) + } + fn estimate(&self, key: &[u8]) -> f64 { + self.indexes(key) + .map(|i| self.cells[i]) + .fold(f64::INFINITY, f64::min) + } + fn retain(&mut self, mut candidates: Vec) { + candidates.sort_by(|a, b| a.key.cmp(&b.key)); + candidates.dedup_by(|a, b| a.key == b.key); + self.candidates.clear(); + for mut candidate in candidates { + candidate.score = self.estimate(&candidate.key); + self.candidates.push(candidate); + if self.candidates.len() > self.capacity { + self.candidates.pop(); + } + } + } + pub fn update(&mut self, values: &[Value], weight: f64) -> Result<(), Error> { + if !weight.is_finite() || weight < 0.0 { + return Err(Error::Operator( + "weighted CMS requires finite nonnegative rates".into(), + )); + } + let identity = values + .iter() + .map(Identity::from_value) + .collect::, _>>()?; + let key = bincode::serialize(&identity).map_err(|e| Error::Operator(e.to_string()))?; + let indexes = self.indexes(&key).collect::>(); + if indexes + .iter() + .any(|&i| !(self.cells[i] + weight).is_finite()) + { + return Err(Error::Operator("weighted CMS sum overflow".into())); + } + for i in indexes { + self.cells[i] += weight; + } + let mut candidates = self.candidates.iter().cloned().collect::>(); + candidates.push(Candidate { + identity, + key, + score: 0.0, + }); + self.retain(candidates); + Ok(()) + } + pub fn rows(&self, n: usize) -> Vec> { + let mut candidates = self.candidates.iter().collect::>(); + candidates.sort_by(|a, b| b.score.total_cmp(&a.score).then_with(|| a.key.cmp(&b.key))); + candidates + .into_iter() + .take(n) + .map(|c| { + let mut row = c.identity.iter().map(Identity::value).collect::>(); + row.push(Value::Float64(c.score)); + row + }) + .collect() + } +} +impl SerializableToSink for WeightedCms { + fn serialize_to_json(&self) -> serde_json::Value { + serde_json::to_value(self).expect("finite validated CMS state") + } + fn serialize_to_bytes(&self) -> Vec { + let mut bytes = b"ASAP-WCMS-1\0".to_vec(); + bytes.extend(bincode::serialize(self).expect("serializable CMS state")); + bytes + } +} +impl AggregateCore for WeightedCms { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + fn type_name(&self) -> &'static str { + "WeightedCms" + } + fn as_any(&self) -> &dyn std::any::Any { + self + } + fn as_any_mut(&mut self) -> &mut dyn std::any::Any { + self + } + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("weighted CMS state type mismatch")?; + if (self.width, self.depth, self.capacity) != (other.width, other.depth, other.capacity) { + return Err("weighted CMS shape mismatch".into()); + } + let mut result = self.clone(); + for (value, rhs) in result.cells.iter_mut().zip(&other.cells) { + *value += rhs; + if !value.is_finite() { + return Err("weighted CMS merge overflow".into()); + } + } + result.retain( + self.candidates + .iter() + .chain(&other.candidates) + .cloned() + .collect(), + ); + Ok(Box::new(result)) + } + fn get_accumulator_type(&self) -> AggregationType { + AggregationType::CountMinSketchWithHeap + } + fn get_keys(&self) -> Option> { + None + } + fn query_statistic( + &self, + _: Statistic, + _: &Option, + _: &HashMap, + ) -> Result> { + Err("weighted CMS uses typed row readout".into()) + } + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::() + + self.cells.capacity() * 8 + + self + .candidates + .iter() + .map(|c| { + std::mem::size_of::() + + c.key.capacity() + + c.identity + .iter() + .map(|v| { + std::mem::size_of::() + + if let Identity::Utf8(s) = v { + s.capacity() + } else { + 0 + } + }) + .sum::() + }) + .sum::() + } +} + +#[cfg(test)] +mod tests { + use super::*; + // Invalid rates must not mutate state; typed keys cannot collide by formatting. + #[test] + fn fractional_updates_typed_identities_and_invalid_weights() { + let mut state = WeightedCms::new(4096, 5, 8).unwrap(); + state.update(&[Value::Int64(1)], 0.125).unwrap(); + state.update(&[Value::Int64(1)], 0.125).unwrap(); + state.update(&[Value::Utf8("1".into())], 0.5).unwrap(); + state.update(&[Value::Null], 0.75).unwrap(); + let before = state.serialize_to_bytes(); + let decoded = WeightedCms::from_bytes(&before).unwrap(); + assert_eq!(decoded.rows(8).len(), 3); + assert!(WeightedCms::from_bytes(b"old integer state").is_err()); + for weight in [-1.0, f64::INFINITY, f64::NAN] { + assert!(state.update(&[Value::Null], weight).is_err()); + assert_eq!(state.serialize_to_bytes(), before); + } + let rows = state.rows(8); + assert_eq!(rows.len(), 3); + assert!(matches!(rows[0][0], Value::Null)); + assert!(matches!(rows[1][0], Value::Utf8(_))); + assert!(matches!(rows[2][1], Value::Float64(0.25))); + } + // Merge uses the same Float64 state representation and rejects other shapes. + #[test] + fn compatible_merge_preserves_fractional_weights() { + let mut left = WeightedCms::new(4096, 5, 8).unwrap(); + let mut right = left.clone(); + left.update(&[Value::Int64(7)], 0.125).unwrap(); + right.update(&[Value::Int64(7)], 0.25).unwrap(); + let merged = left.merge_with(&right).unwrap(); + let merged = merged.as_any().downcast_ref::().unwrap(); + assert!(matches!(merged.rows(1)[0][1], Value::Float64(0.375))); + assert!(left + .merge_with(&WeightedCms::new(32, 5, 8).unwrap()) + .is_err()); + } +} diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 73d8b5d2..63228402 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -399,7 +399,7 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { #[test] fn restored_exact_state_and_family_validation() { use asap_physical_operators::{ - accumulators::exact_accumulator::ExactAccumulator, SerializableToSink, + summary_operators::exact_accumulator::ExactAccumulator, SerializableToSink, }; let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); @@ -963,3 +963,112 @@ fn native_relational_join_kinds_preserve_unmatched_rows() { assert_eq!(run(&dag, 2, query()).len(), count, "{kind:?}"); } } + +// Per-series fractional rates feed one independent CMS per job, in either scope. +#[test] +fn weighted_rate_topk_preserves_partitions_fractional_scores_and_evaluation_scope() { + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let raw = schema(&[ + ("service", DataType::Utf8, false), + ("job", DataType::Utf8, false), + ("instance", DataType::Int64, false), + ("t", DataType::Timestamp, false), + ("value", DataType::Float64, false), + ]); + let mut rows = Vec::new(); + // Multiple instances of auth accumulate. Batch has a very different scale. + for (service, job, instance, rate) in [ + ("auth", "api", 1, 0.125), + ("auth", "api", 2, 0.25), + ("checkout", "api", 1, 0.3125), + ("search", "api", 1, 0.0625), + ("ingest", "batch", 1, 100.0), + ("export", "batch", 1, 80.0), + ("cleanup", "batch", 1, 20.0), + ] { + for (t, value) in [(0, 0.0), (30_000, rate * 30.0), (60_000, rate * 60.0)] { + rows.push(vec![ + Value::Utf8(service.into()), + Value::Utf8(job.into()), + Value::Int64(instance), + Value::Timestamp(t), + Value::Float64(value), + ]); + } + } + let rates = Operator::window( + raw.clone(), + planner_types::pre_asap::AggIntent::Rate, + 3, + 4, + vec![0, 1, 2], + Some((0, 60_000)), + ) + .unwrap(); + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 4096, + depth: 5, + heap_size: 8, + }, + ), + Default::default(), + ); + let build = Operator::keyed_summary_build(rates.schema(), family, 3, vec![0], vec![1]).unwrap(); + let output = schema(&[ + ("job", DataType::Utf8, false), + ("service", DataType::Utf8, false), + ("score", DataType::Float64, false), + ]); + let readout = Operator::keyed_readout(build.schema(), 1, 8, output.clone()).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(raw.clone(), vec![Batch::try_new(raw, rows).unwrap()]).unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], rates).unwrap(); + dag.add(2, vec![1], build).unwrap(); + dag.add(3, vec![2], readout).unwrap(); + dag.add( + 4, + vec![3], + Operator::sort( + output.clone(), + vec![SortKey { + column: 2, + descending: true, + nulls_first: false, + }], + vec![0], + ) + .unwrap(), + ) + .unwrap(); + dag.add(5, vec![4], Operator::limit(output, 2, 0, vec![0]).unwrap()) + .unwrap(); + for scope in [ + query(), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 60_000, + revision: 2, + }, + query(), + ] { + let result = run(&dag, 5, scope); + assert_eq!(result.len(), 4); + assert_eq!(floats(&result, 2), vec![0.375, 0.3125, 100.0, 80.0]); + let services = result + .iter() + .map(|row| match &row[1] { + Value::Utf8(v) => v.as_ref(), + _ => panic!("service"), + }) + .collect::>(); + assert_eq!(services, vec!["auth", "checkout", "ingest", "export"]); + } +} diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs new file mode 100644 index 00000000..8843ce8a --- /dev/null +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -0,0 +1,251 @@ +//! Planner output binds directly to the shared runtime at a declared rate-value frontier. +use asap_aware_mapping::{ + accuracy::{ + AccuracyEvidenceProvider, DefaultAccuracyModel, EqualSplitAllocator, PropagationStats, + }, + cost_model::DefaultCostModel, + Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG, +}; +use asap_physical_operators::dag::{ + operators::Operator, + planner::{bind, Source}, + values::{Batch, Value}, + Limits, RunContext, Scope, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{DataType, QueryExpr}, + types::AccuracyTarget, +}; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; +struct Evidence; +impl AccuracyEvidenceProvider for Evidence { + fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + Some(1000) + } + fn propagation_stats( + &self, + op: &CompositionOperator, + _: &SummaryFamilyType, + _: Option<&SketchQuery>, + ) -> PropagationStats { + if matches!(op, CompositionOperator::TopKSelection) { + PropagationStats { + topk_selected_lower_bound: Some(101.), + topk_excluded_upper_bound: Some(100.), + topk_interval_failure_probability: Some(0.001), + ..Default::default() + } + } else { + Default::default() + } + } +} +// The evidence here exercises binding; it is not inferred from the sample data. +#[test] +fn planner_weighted_topk_binds_at_either_deployment_phase() { + assert_weighted_binding(&Evidence); +} + +// Binding validates representation, while deployment owns evidence acceptance. +#[test] +fn physical_binding_does_not_impose_an_accuracy_acceptance_policy() { + assert_weighted_binding(&asap_aware_mapping::accuracy::NoAccuracyEvidence); +} + +fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider) { + let root = Rc::new( + lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(), + ); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + evidence, + ); + let plan = strategy + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .find_map(|candidate| match candidate.replacement { + Replacement::Summary(node) if candidate.rationale.contains("CmsWithHeap") => Some(node), + _ => None, + }) + .unwrap(); + let dag = compile_executable_dag(&plan).unwrap(); + let build=dag.nodes.iter().find(|node|matches!(&node.payload,ExecutableOperatorPayload::SummaryAgg{family:SummaryFamilyType::Sketch(kind,_),..}if kind.algorithm()==&SketchAlgorithm::CmsWithHeap)).unwrap(); + let rate_id = dag + .edges + .iter() + .find(|edge| edge.consumer == build.id) + .unwrap() + .producer; + let rates = Arc::new( + dag.nodes + .iter() + .find(|node| node.id == rate_id) + .unwrap() + .output_schema + .clone(), + ); + let rows = [ + ("auth", "api", 0.125), + ("auth", "api", 0.25), + ("checkout", "api", 0.3125), + ("search", "api", 0.0625), + ("ingest", "batch", 100.), + ("export", "batch", 80.), + ("cleanup", "batch", 20.), + ] + .into_iter() + .map(|(service, job, value)| { + rates + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "job" => Value::Utf8(job.into()), + "value" => Value::Float64(value), + _ => match field.dtype { + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), + _ => panic!("unexpected rate column {field:?}"), + }, + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(rates.clone(), rows).unwrap(); + for (phase, scope) in [ + ( + ExecutionTiming::IngestionTime, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 60_000, + revision: 1, + }, + ), + ( + ExecutionTiming::QueryTime, + Scope::Query { + evaluation_time_ms: 60_000, + revision: 1, + }, + ), + ] { + let placed = dag + .with_execution_phases(&dag.nodes.iter().map(|node| (node.id, phase)).collect()) + .unwrap(); + let source = Box::new(Operator::source(rates.clone(), vec![batch.clone()]).unwrap()) + as Source<'static>; + let graph = bind( + &placed, + BTreeMap::from([(rate_id.0 as u64, source)]), + &[dag.root.0 as u64], + ) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let output = block_on(async { + let mut output = Vec::new(); + let mut stream = graph + .execute(&[dag.root.0 as u64], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + output.extend(batch.unwrap().rows().iter().cloned()); + } + output + }); + assert_eq!(output.len(), 4); + let mut scores = output + .iter() + .map(|row| { + row.iter() + .find_map(|v| { + if let Value::Float64(v) = v { + Some(*v) + } else { + None + } + }) + .unwrap() + }) + .collect::>(); + scores.sort_by(f64::total_cmp); + assert_eq!(scores, vec![0.3125, 0.375, 80., 100.]); + } +} + +use asap_frontend_promql::lower_promql_workload; +use planner_types::workload::{ + AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence as WorkloadEvidence, + PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, + TimeSelection, +}; +pub fn lower_promql( + query: &str, + accuracy: AccuracyTarget, +) -> Result { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: WorkloadEvidence { + value: Some(DurationMs(1_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let mut lowered = lower_promql_workload(&workload, 0)?; + Ok(lowered.remove(0)) +} + +// The old untyped heap updater must not silently round a Planner rate update. +#[test] +fn rate_updates_cannot_enter_integer_heap_factory() { + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 272, + depth: 5, + heap_size: 100, + }, + ), + Default::default(), + ); + let input = SummaryUpdate { + item: Some(SummaryInputExpr::Column( + planner_types::pre_asap::ColumnRef::Named("service".into()), + )), + weight: SummaryInputExpr::Column(planner_types::pre_asap::ColumnRef::SampleValue), + weight_domain: WeightDomain::NonNegative { + proof: NonNegativeWeightProof::ResetAwareCounterDerivative, + }, + }; + assert!( + asap_physical_operators::factory::create_planner_accumulator( + &family, + &input, + &Default::default() + ) + .is_err() + ); +} diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index 59e47e2b..dbb50405 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -132,11 +132,25 @@ boundaries and bound columns; the computation is identical in either phase. Count outputs Int64. Binary expressions use Planner arithmetic/comparison kinds and enforce its checked-division domains. -Grouped TopK composes Sort and Limit within each group; candidate -completeness is an earlier pruning obligation. +Grouped TopK composes Sort and Limit within each group. A weighted summary can +consume per-series rates directly: each job has its own CMS and candidate heap, +with service as the item and rate as the weight. Sum accumulation inside the +summary replaces the exact grouped-sum materialization. Typed readout returns +candidate identities and estimated scores; a semi-join is not required for this +realization. Both score error and membership require accuracy guarantees. + +The `summary_operators` module owns typed summary kernels. Native weighted CMS +uses Float64 counters and preserves typed item identities, including numeric and +NULL keys. It does not use the integer-count codec or fixed-point counter-delta +updates. The DAG binder supports column weights and explicit column/tuple item +identities for this CMS path; unsupported families or identities are rejected. +Summary construction and readout run in either ingestion or query scope, as +chosen by deployment. Each run constructs independent partition state; deployment +must supply one complete evaluation window, or an equivalent maintained snapshot. +The candidate capacity is independent of the grouped Limit's output count. Values retain Planner types and nullability. Native summary batches currently -support exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch and HLL. Stored-summary decoding, delta reconstruction, exact finalization and +support exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS with a candidate heap. Stored-summary decoding, delta reconstruction, exact finalization and family-specific SketchQuery readout also live in this library. Deployment code selects compatible panes and supplies source batches. Stored-state kernels do not imply native batch bindings for every family. Unsupported expressions, From deb1d03edd36194a2f722269875507a11c26239a Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 16:06:48 +0000 Subject: [PATCH 11/90] refactor: clarify physical execution contracts and DataFusion tradeoffs --- crates/asap-physical-operators/README.md | 44 +- .../src/{dag/planner.rs => binding/mod.rs} | 25 +- .../asap-physical-operators/src/capability.rs | 99 +- crates/asap-physical-operators/src/dag/mod.rs | 529 +----- .../src/dag/operators.rs | 1647 ----------------- crates/asap-physical-operators/src/error.rs | 18 + .../src/{ => expressions}/arithmetic.rs | 4 +- .../src/expressions/mod.rs | 252 +++ .../expressions.rs => expressions/planner.rs} | 6 +- crates/asap-physical-operators/src/lib.rs | 20 +- .../src/operators/aggregate/mod.rs | 293 +++ .../{dag => operators/aggregate}/temporal.rs | 65 +- .../src/operators/common.rs | 75 + .../src/operators/filter.rs | 39 + .../src/operators/joins/mod.rs | 178 ++ .../src/operators/limit.rs | 69 + .../src/operators/mod.rs | 207 +++ .../src/operators/projection.rs | 47 + .../src/operators/sort.rs | 172 ++ .../src/operators/source.rs | 88 + .../src/operators/summary/mod.rs | 538 ++++++ .../asap-physical-operators/src/plan/mod.rs | 156 ++ .../src/plan/properties.rs | 32 + .../src/{dag => runtime}/batch_execution.rs | 12 +- .../src/runtime/context.rs | 133 ++ .../src/runtime/cooperative.rs | 40 + .../src/runtime/mod.rs | 263 +++ .../src/{dag => runtime}/tests.rs | 2 + .../src/sources/memory.rs | 44 + .../src/{dag/scan.rs => sources/mod.rs} | 63 +- .../src/{ => summary_operators}/factory.rs | 0 .../src/summary_operators/mod.rs | 3 + .../src/{ => summary_operators}/traits.rs | 0 .../src/summary_operators/weighted_cms.rs | 2 +- .../src/{dag => }/values.rs | 62 +- .../tests/blocking_resources.rs | 181 ++ .../tests/plan_properties.rs | 155 ++ .../asap-physical-operators/tests/raw_scan.rs | 3 + .../datafusion-execution-comparison.md | 295 +++ docs/design_docs/physical-operators.md | 126 +- 40 files changed, 3671 insertions(+), 2316 deletions(-) rename crates/asap-physical-operators/src/{dag/planner.rs => binding/mod.rs} (97%) delete mode 100644 crates/asap-physical-operators/src/dag/operators.rs create mode 100644 crates/asap-physical-operators/src/error.rs rename crates/asap-physical-operators/src/{ => expressions}/arithmetic.rs (95%) create mode 100644 crates/asap-physical-operators/src/expressions/mod.rs rename crates/asap-physical-operators/src/{dag/expressions.rs => expressions/planner.rs} (99%) create mode 100644 crates/asap-physical-operators/src/operators/aggregate/mod.rs rename crates/asap-physical-operators/src/{dag => operators/aggregate}/temporal.rs (82%) create mode 100644 crates/asap-physical-operators/src/operators/common.rs create mode 100644 crates/asap-physical-operators/src/operators/filter.rs create mode 100644 crates/asap-physical-operators/src/operators/joins/mod.rs create mode 100644 crates/asap-physical-operators/src/operators/limit.rs create mode 100644 crates/asap-physical-operators/src/operators/mod.rs create mode 100644 crates/asap-physical-operators/src/operators/projection.rs create mode 100644 crates/asap-physical-operators/src/operators/sort.rs create mode 100644 crates/asap-physical-operators/src/operators/source.rs create mode 100644 crates/asap-physical-operators/src/operators/summary/mod.rs create mode 100644 crates/asap-physical-operators/src/plan/mod.rs create mode 100644 crates/asap-physical-operators/src/plan/properties.rs rename crates/asap-physical-operators/src/{dag => runtime}/batch_execution.rs (96%) create mode 100644 crates/asap-physical-operators/src/runtime/context.rs create mode 100644 crates/asap-physical-operators/src/runtime/cooperative.rs create mode 100644 crates/asap-physical-operators/src/runtime/mod.rs rename crates/asap-physical-operators/src/{dag => runtime}/tests.rs (99%) create mode 100644 crates/asap-physical-operators/src/sources/memory.rs rename crates/asap-physical-operators/src/{dag/scan.rs => sources/mod.rs} (79%) rename crates/asap-physical-operators/src/{ => summary_operators}/factory.rs (100%) rename crates/asap-physical-operators/src/{ => summary_operators}/traits.rs (100%) rename crates/asap-physical-operators/src/{dag => }/values.rs (89%) create mode 100644 crates/asap-physical-operators/tests/blocking_resources.rs create mode 100644 crates/asap-physical-operators/tests/plan_properties.rs create mode 100644 docs/design_docs/datafusion-execution-comparison.md diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index 8ff91704..cb9b04d9 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -5,14 +5,14 @@ query time execution. The library requires neither backend engine, a server, a storage implementation, Arrow nor DataFusion. DataFusion informed the design; it is not the execution framework. -`dag::PhysicalDag` binds typed operator inputs to node IDs. Each execution starts +`plan::PhysicalDag` binds typed operator inputs to node IDs. Each execution starts one producer per reachable node, shares output batches among its consumers, and bounds buffering. Dropping one consumer does not cancel other consumers. A `RunContext` carries query or ingestion scope, cancellation and byte accounting. Executions use the caller's worker and worker-local streams, with no internal thread pool. Poll multiple root streams concurrently when they share inputs. -`dag::operators::Operator` implements native batch sources, scalar values, +`operators::Operator` implements native batch sources, scalar values, projection, filtering, grouped exact aggregation, semi-join, grouped Sort and Limit, vector-to-scalar conversion, Union, and summary construction/merge/readout. Sort followed by Limit implements grouped ranking; no dedicated TopK physical @@ -20,10 +20,12 @@ operator is needed. Summary construction updates state batch by batch. End of input means the supplied query range or ingestion window is complete. ```rust -use asap_physical_operators::dag::{ - operators::{Expression, Operator}, +use asap_physical_operators::{ + expressions::Expression, + operators::Operator, values::Value, - Limits, PhysicalDag, RunContext, Scope, + plan::PhysicalDag, + runtime::{Limits, RunContext, Scope}, }; use asap_physical_operators::planner::pre_asap::DataType; use futures::{executor::block_on, StreamExt}; @@ -45,7 +47,7 @@ assert!(matches!(batch.rows()[0][0], Value::Int64(-7))); # Ok::<(), asap_physical_operators::dag::Error>(()) ``` -`dag::planner::bind` accepts a post-ASAP DAG and explicit source bindings for +`binding::bind` accepts a post-ASAP DAG and explicit source bindings for installed ingestion/storage frontiers. It rejects unsupported operations and schema mismatches before starting a source. Implement `PhysicalOperator` for a deployment source, including asynchronous I/O; computation operators remain in @@ -56,7 +58,7 @@ operations; it does not interpret an unknown node as external fallback. Plain values preserve Planner scalar/collection types and nullability. Numeric arithmetic uses matching Int64 or Float64 inputs; integer overflow is an error. Boolean predicates use three-valued logic. Native summary states currently cover -exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch and HLL. Binding checks family, +exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS with a candidate heap. Binding checks family, parameters and readout compatibility; source batches also validate state payloads. Existing accumulator algorithms are reused as kernels behind these operators. @@ -67,3 +69,31 @@ library has no ASAPQuery-backend dependency. Backend raw Scan remains a separate deployment capability. See [the design](../../docs/design_docs/physical-operators.md). + +## Module boundaries + +- `plan`: immutable graph, operator interface, schemas and execution properties. +- `runtime`: per-run streams, shared producers, memory reservations and cancellation. +- `expressions`: scalar evaluation; typed builders and the Planner expression adapter. +- `operators`: projection, filter, joins, aggregate/window, sort, limit and summary implementations. +- `sources`: raw-source interface, Scan and the memory connector. +- `binding`: Planner executable DAG binding and installed source frontiers. +- `summary_operators`: mathematical summary kernels, update adapters and accumulator traits. +- `stored_state`: persisted-state decoding, delta reconstruction and readout. +- `capability`: explicit kernel and native-batch/readout validation. + +The old `dag`, `accumulators`, `factory`, `traits` and `arithmetic` paths remain re-exports for deployment +source compatibility. They contain no alternative execution implementations. + +A source must declare `Boundedness::Bounded` to feed a blocking operator. +The default for a custom raw source is `Unknown`; query or ingestion scope alone +does not promise that its cursor ends. `PhysicalDag::properties` validates these +requirements before any source starts and returns boundedness and emission mode +for every reachable node. The memory connector declares finite input. Custom +physical sources expose the same facts through `PhysicalOperator::properties`. + +Blocking operators reserve estimated workspace and yield cooperatively during +row processing and sort merges. Cancellation releases reservations when the +stream is polled or dropped. Individual scalar evaluations, bounded sort chunks +and sketch kernel calls are synchronous; this is not preemptive execution. +There is no spill or partitioned parallel execution in this implementation. diff --git a/crates/asap-physical-operators/src/dag/planner.rs b/crates/asap-physical-operators/src/binding/mod.rs similarity index 97% rename from crates/asap-physical-operators/src/dag/planner.rs rename to crates/asap-physical-operators/src/binding/mod.rs index 04143122..09a2df0c 100644 --- a/crates/asap-physical-operators/src/dag/planner.rs +++ b/crates/asap-physical-operators/src/binding/mod.rs @@ -1,9 +1,10 @@ //! Bind a post-ASAP DAG to native operators. Sources are explicit execution //! frontiers supplied by the deployment; unsupported computation is an error. -use super::{ +use crate::{ operators::{Expression, Operator, Reduction, SortKey}, + plan::{NodeId, PhysicalDag, PhysicalOperator}, values::{Batch, Schema}, - Error, NodeId, PhysicalDag, PhysicalOperator, + Error, }; use planner_types::{ post_asap::{ @@ -41,7 +42,7 @@ pub fn bind_with_data_sources<'a>( dag: &ExecutableDag, sources: BTreeMap>, roots: &[NodeId], - data_sources: &super::scan::DataSources, + data_sources: &crate::sources::DataSources, ) -> Result, Error> { bind_internal(dag, sources, roots, Some(data_sources)) } @@ -50,7 +51,7 @@ fn bind_internal<'a>( dag: &ExecutableDag, mut sources: BTreeMap>, roots: &[NodeId], - data_sources: Option<&super::scan::DataSources>, + data_sources: Option<&crate::sources::DataSources>, ) -> Result, Error> { preflight_depth(dag)?; dag.validate().map_err(|e| invalid(e.to_string()))?; @@ -107,7 +108,7 @@ fn bind_internal<'a>( for id in ordered { let node = nodes[&id]; let output = Arc::new(node.output_schema.clone()); - super::values::validate_schema(&output)?; + crate::values::validate_schema(&output)?; let (operator, inputs) = if let Some(source) = sources.remove(&id) { if !source.input_schemas().is_empty() || source.output_schema() != output { return Err(invalid("frontier is not a source with the declared schema")); @@ -161,7 +162,7 @@ fn bind_internal<'a>( /// This is the same checked path used by complete DAG binding. pub fn bind_node(node: &ExecutableDagNode, inputs: &[Schema]) -> Result { for schema in inputs { - super::values::validate_schema(schema)?; + crate::values::validate_schema(schema)?; } bind_operation(node, inputs)?.with_output_schema(Arc::new(node.output_schema.clone())) } @@ -462,7 +463,7 @@ fn groups(input: &Schema, groups: &GroupKeys) -> Result, Error> { } fn expression(expr: &QueryExpr, input: &Schema) -> Result { Ok(Expression::planner( - super::expressions::CompiledExpression::compile(expr, input)?, + crate::expressions::CompiledExpression::compile(expr, input)?, )) } @@ -471,6 +472,10 @@ struct CheckedSource<'a> { output: Schema, } impl PhysicalOperator for CheckedSource<'_> { + fn properties(&self, inputs: &[crate::plan::PlanProperties]) -> crate::plan::PlanProperties { + self.source.properties(inputs) + } + fn name(&self) -> &str { self.source.name() } @@ -485,9 +490,9 @@ impl PhysicalOperator for CheckedSource<'_> { } fn start<'a>( &'a self, - inputs: Vec>, - context: super::RunContext, - ) -> Result, Error> { + inputs: Vec>, + context: crate::runtime::RunContext, + ) -> Result, Error> { use futures::StreamExt; Ok(self .source diff --git a/crates/asap-physical-operators/src/capability.rs b/crates/asap-physical-operators/src/capability.rs index 2d047229..65b98716 100644 --- a/crates/asap-physical-operators/src/capability.rs +++ b/crates/asap-physical-operators/src/capability.rs @@ -1,4 +1,15 @@ -//! Allocation-free checks for the concrete summary kernels in this crate. +//! Capability boundaries, checked without constructing accumulator state. +//! +//! `validate_summary_kernel` checks update kernels, including families without a +//! native batch representation. `validate_native_family` and +//! `validate_native_readout` check native state and scalar readout support. +//! Keyed weighted-CMS readouts are checked by `Operator::keyed_readout`. +//! A successful kernel check alone does not mean an executable DAG will bind. +//! +//! Persisted state uses `stored_state` decoding and readout contracts; support +//! there does not imply a native build/merge operator. Full plan acceptance is +//! owned by `binding`, which also validates schemas, expressions and inputs. +use crate::Error; use planner_types::post_asap::{ ExactKind, ExactParams, GroupingStrategy, SketchAlgorithm, SketchParams, SummaryFamilyType, SummaryUpdate, @@ -127,3 +138,89 @@ pub(crate) fn is_unit_sample_frequency(update: &planner_types::post_asap::Summar } ) } + +pub fn validate_native_family(family: &SummaryFamilyType) -> Result<(), Error> { + use planner_types::post_asap::SketchAlgorithm as A; + if let SummaryFamilyType::Sketch(kind, grouping) = family { + if let planner_types::post_asap::SketchParams::CmsWithHeap { + width, + depth, + heap_size, + } = kind.params() + { + return if kind.algorithm() == &A::CmsWithHeap + && valid_matrix(*width, *depth) + && *heap_size > 0 + && grouping == &Default::default() + { + Ok(()) + } else { + Err(Error::Invalid( + "invalid weighted CMS family or grouping strategy".into(), + )) + }; + } + } + match family { + SummaryFamilyType::ExactAggregate(..) => {} + SummaryFamilyType::Sketch(kind, _) + if matches!(kind.algorithm(), A::Kll | A::DDSketch | A::Hll) => {} + _ => { + return Err(Error::Invalid( + "summary family has no native DAG state implementation".into(), + )) + } + } + crate::capability::validate_summary_kernel( + family, + &planner_types::post_asap::SummaryUpdate::column( + planner_types::pre_asap::ColumnRef::SampleValue, + ), + &Default::default(), + ) + .map_err(Error::Invalid) +} + +pub fn validate_native_readout( + family: &SummaryFamilyType, + statistic: crate::Statistic, + parameters: &std::collections::HashMap, +) -> Result<(), Error> { + validate_native_family(family)?; + use crate::Statistic as S; + use planner_types::post_asap::{ExactKind as E, SketchAlgorithm as A}; + let supported = match family { + SummaryFamilyType::ExactAggregate(kind, _) => matches!( + (kind, statistic), + (E::Sum, S::Sum) + | (E::Count, S::Count) + | (E::Min, S::Min) + | (E::Max, S::Max) + | (E::Rate, S::Rate) + | (E::Increase, S::Increase) + ), + SummaryFamilyType::Sketch(kind, _) => match kind.algorithm() { + A::Kll => statistic == S::Quantile, + A::DDSketch => matches!(statistic, S::Quantile | S::Count), + A::Hll => matches!(statistic, S::Cardinality | S::Count), + _ => false, + }, + _ => false, + }; + if !supported { + return Err(Error::Invalid( + "readout is not implemented for this summary family".into(), + )); + } + if statistic == S::Quantile + && !parameters + .get("quantile") + .and_then(|s| s.parse::().ok()) + .is_some_and(|q| (0.0..=1.0).contains(&q)) + { + return Err(Error::Invalid( + "quantile readout requires quantile in [0,1]".into(), + )); + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/dag/mod.rs b/crates/asap-physical-operators/src/dag/mod.rs index c4ad17ea..a3ea3a7e 100644 --- a/crates/asap-physical-operators/src/dag/mod.rs +++ b/crates/asap-physical-operators/src/dag/mod.rs @@ -1,523 +1,8 @@ -//! Independent operator DAG execution. No backend plan or engine types are used. -//! -//! Each run creates one stream per reachable node. Consumers subscribe to that -//! stream independently; retained outputs are released after the last consumer. -use futures::{stream::LocalBoxStream, Stream}; -use std::{ - cell::{Cell, RefCell}, - collections::{BTreeMap, BTreeSet, VecDeque}, - fmt::Debug, - pin::Pin, - rc::Rc, - sync::Arc, - task::{Context, Poll, Waker}, +//! Compatibility imports. New code should use plan, runtime, operators, binding and sources directly. +pub use crate::plan::{NodeId, PhysicalDag, PhysicalOperator}; +pub use crate::runtime::batch_execution; +pub use crate::runtime::{ + Input, Limits, OutputStream, Reservation, RunContext, Scope, SharedValue, }; - -pub type NodeId = u64; -pub type OutputStream<'a, V> = LocalBoxStream<'a, Result>; -#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] -pub enum Error { - #[error("invalid DAG: {0}")] - Invalid(String), - #[error("operator failed: {0}")] - Operator(String), - #[error("node {node} ({operation}) failed: {source}")] - AtNode { - node: NodeId, - operation: String, - source: Box, - }, - #[error("execution memory limit exceeded")] - MemoryLimit, - #[error("execution cancelled")] - Cancelled, -} - -/// Scope is part of an execution instance, never mutable state in a reusable plan. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum Scope { - Ingestion { - window_start_ms: i64, - window_end_ms: i64, - revision: u64, - }, - Query { - evaluation_time_ms: i64, - revision: u64, - }, -} -#[derive(Clone, Debug)] -pub struct Limits { - pub max_buffered_batches: usize, - pub max_bytes: usize, -} -impl Default for Limits { - fn default() -> Self { - Self { - max_buffered_batches: 8, - max_bytes: 64 * 1024 * 1024, - } - } -} -struct Control { - cancelled: Cell, - bytes: Cell, - peak: Cell, - limits: Limits, - waiters: RefCell>, -} -#[derive(Clone)] -pub struct RunContext { - pub scope: Scope, - control: Rc, -} -impl RunContext { - pub fn new(scope: Scope, limits: Limits) -> Result { - if limits.max_buffered_batches == 0 || limits.max_bytes == 0 { - return Err(Error::Invalid("execution limits must be positive".into())); - } - if matches!(&scope, Scope::Ingestion { window_start_ms, window_end_ms, .. } if window_start_ms > window_end_ms) - { - return Err(Error::Invalid("inverted ingestion window".into())); - } - Ok(Self { - scope, - control: Rc::new(Control { - cancelled: Cell::new(false), - bytes: Cell::new(0), - peak: Cell::new(0), - limits, - waiters: RefCell::new(Vec::new()), - }), - }) - } - pub fn cancel(&self) { - self.control.cancelled.set(true); - for waiter in self.control.waiters.borrow_mut().drain(..) { - waiter.wake(); - } - } - pub fn is_cancelled(&self) -> bool { - self.control.cancelled.get() - } - pub fn retained_bytes(&self) -> usize { - self.control.bytes.get() - } - pub fn peak_bytes(&self) -> usize { - self.control.peak.get() - } - pub fn reserve(&self, bytes: usize) -> Result { - let total = self - .control - .bytes - .get() - .checked_add(bytes) - .ok_or(Error::MemoryLimit)?; - if total > self.control.limits.max_bytes { - return Err(Error::MemoryLimit); - } - self.control.bytes.set(total); - self.control.peak.set(self.control.peak.get().max(total)); - Ok(Reservation { - bytes, - control: Rc::clone(&self.control), - }) - } - fn register(&self, waker: &Waker) { - let mut waiters = self.control.waiters.borrow_mut(); - if !waiters.iter().any(|old| old.will_wake(waker)) { - waiters.push(waker.clone()); - } - } -} -pub struct Reservation { - bytes: usize, - control: Rc, -} -impl Reservation { - /// Adjust an operator-owned allocation without accumulating bookkeeping entries. - pub fn resize(&mut self, bytes: usize) -> Result<(), Error> { - let total = self - .control - .bytes - .get() - .checked_sub(self.bytes) - .and_then(|total| total.checked_add(bytes)) - .ok_or(Error::MemoryLimit)?; - if total > self.control.limits.max_bytes { - return Err(Error::MemoryLimit); - } - self.control.bytes.set(total); - self.control.peak.set(self.control.peak.get().max(total)); - self.bytes = bytes; - Ok(()) - } -} -impl Drop for Reservation { - fn drop(&mut self) { - self.control - .bytes - .set(self.control.bytes.get().saturating_sub(self.bytes)); - } -} - -/// An output owns its memory reservation even after it leaves the DAG's queue. -pub struct SharedValue { - value: Arc, - _reservation: Rc, -} -impl Clone for SharedValue { - fn clone(&self) -> Self { - Self { - value: Arc::clone(&self.value), - _reservation: Rc::clone(&self._reservation), - } - } -} -impl std::ops::Deref for SharedValue { - type Target = V; - fn deref(&self) -> &V { - &self.value - } -} -impl SharedValue { - pub fn value(&self) -> &V { - &self.value - } -} - -/// Operators own computation. The runtime provides already-connected inputs; -/// an operator must not recursively execute another plan node itself. -pub trait PhysicalOperator { - fn name(&self) -> &str; - fn input_schemas(&self) -> Vec; - fn output_schema(&self) -> S; - fn start<'a>( - &'a self, - inputs: Vec>, - context: RunContext, - ) -> Result, Error>; - fn output_bytes(&self, value: &V) -> usize; -} -struct Node<'a, V, S> { - inputs: Vec, - operator: Box + 'a>, -} -pub struct PhysicalDag<'a, V, S> { - nodes: BTreeMap>, -} -impl Default for PhysicalDag<'_, V, S> { - fn default() -> Self { - Self { - nodes: BTreeMap::new(), - } - } -} -impl<'a, V: 'a, S: Clone + PartialEq + Debug + 'a> PhysicalDag<'a, V, S> { - pub fn add( - &mut self, - id: NodeId, - inputs: Vec, - operator: impl PhysicalOperator + 'a, - ) -> Result<(), Error> { - self.add_boxed(id, inputs, Box::new(operator)) - } - pub fn add_boxed( - &mut self, - id: NodeId, - inputs: Vec, - operator: Box + 'a>, - ) -> Result<(), Error> { - if self.nodes.contains_key(&id) { - return Err(Error::Invalid(format!("duplicate node {id}"))); - } - self.nodes.insert(id, Node { inputs, operator }); - Ok(()) - } - pub fn validate(&self, roots: &[NodeId]) -> Result<(), Error> { - fn visit( - dag: &PhysicalDag<'_, V, S>, - id: NodeId, - active: &mut BTreeSet, - done: &mut BTreeMap, - ) -> Result { - if let Some(depth) = done.get(&id) { - return Ok(*depth); - } - if active.len() >= 128 { - return Err(Error::Invalid( - "DAG exceeds the supported execution depth of 128".into(), - )); - } - if !active.insert(id) { - return Err(Error::Invalid(format!("cycle at node {id}"))); - } - let node = dag - .nodes - .get(&id) - .ok_or_else(|| Error::Invalid(format!("missing node {id}")))?; - let expected = node.operator.input_schemas(); - if expected.len() != node.inputs.len() { - return Err(Error::Invalid(format!("node {id} input arity mismatch"))); - } - let mut depth = 1; - for (input, schema) in node.inputs.iter().zip(expected) { - depth = depth.max(1 + visit(dag, *input, active, done)?); - let actual = dag.nodes[input].operator.output_schema(); - if actual != schema { - return Err(Error::Invalid(format!( - "node {id} input {input} schema mismatch: {actual:?} vs {schema:?}" - ))); - } - } - if depth > 128 { - return Err(Error::Invalid( - "DAG exceeds the supported execution depth of 128".into(), - )); - } - active.remove(&id); - done.insert(id, depth); - Ok(depth) - } - if roots.is_empty() { - return Err(Error::Invalid("execution needs a root".into())); - } - let mut done = BTreeMap::new(); - for &root in roots { - visit(self, root, &mut BTreeSet::new(), &mut done)?; - } - Ok(()) - } - pub fn execute<'r>( - &'r self, - roots: &[NodeId], - context: RunContext, - ) -> Result>, Error> - where - 'a: 'r, - { - if context.is_cancelled() { - return Err(Error::Cancelled); - } - self.validate(roots)?; - fn build<'r, V: 'r, S: 'r>( - dag: &'r PhysicalDag<'_, V, S>, - id: NodeId, - context: &RunContext, - states: &mut BTreeMap>>>, - ) -> Result>>, Error> { - if let Some(state) = states.get(&id) { - return Ok(Rc::clone(state)); - } - let node = &dag.nodes[&id]; - let mut inputs = Vec::new(); - for &child in &node.inputs { - inputs.push(Input::subscribe(build(dag, child, context, states)?)); - } - let stream = node - .operator - .start(inputs, context.clone()) - .map_err(|source| Error::AtNode { - node: id, - operation: node.operator.name().into(), - source: Box::new(source), - })?; - let op = node.operator.as_ref(); - let state = Rc::new(RefCell::new(Producer { - stream: Some(stream), - node: id, - operation: node.operator.name().into(), - size: Box::new(move |value| op.output_bytes(value)), - context: context.clone(), - queue: VecDeque::new(), - base: 0, - next_reader: 0, - batches_polled: 0, - readers: BTreeMap::new(), - waiters: BTreeMap::new(), - finished: false, - failure: None, - })); - states.insert(id, Rc::clone(&state)); - Ok(state) - } - let mut states = BTreeMap::new(); - roots - .iter() - .map(|&id| build(self, id, &context, &mut states).map(Input::subscribe)) - .collect() - } -} -struct Producer<'a, V> { - node: NodeId, - operation: String, - stream: Option>, - size: Box usize + 'a>, - context: RunContext, - queue: VecDeque>, - base: u64, - next_reader: u64, - batches_polled: usize, - readers: BTreeMap, - waiters: BTreeMap, - finished: bool, - failure: Option, -} -impl Producer<'_, V> { - fn trim(&mut self) { - let minimum = self - .readers - .values() - .copied() - .min() - .unwrap_or(self.base + self.queue.len() as u64); - while self.base < minimum { - self.queue.pop_front(); - self.base += 1; - } - for (_, waker) in std::mem::take(&mut self.waiters) { - waker.wake(); - } - if self.readers.is_empty() { - self.stream = None; - self.queue.clear(); - } - } -} -pub struct Input<'a, V> { - producer: Rc>>, - reader: u64, - done: bool, -} -impl<'a, V> Input<'a, V> { - fn subscribe(producer: Rc>>) -> Self { - let reader = { - let mut state = producer.borrow_mut(); - let id = state.next_reader; - state.next_reader += 1; - let base = state.base; - state.readers.insert(id, base); - id - }; - Self { - producer, - reader, - done: false, - } - } -} -impl Drop for Input<'_, V> { - fn drop(&mut self) { - let mut state = self.producer.borrow_mut(); - state.readers.remove(&self.reader); - state.waiters.remove(&self.reader); - state.trim(); - } -} -impl Stream for Input<'_, V> { - type Item = Result, Error>; - fn poll_next(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { - let this = self.get_mut(); - if this.done { - return Poll::Ready(None); - } - let mut state = this.producer.borrow_mut(); - state.context.register(cx.waker()); - if state.context.is_cancelled() { - state.failure = Some(Error::Cancelled); - state.finished = true; - state.stream = None; - state.queue.clear(); - } - let position = state.readers[&this.reader]; - let index = (position - state.base) as usize; - if let Some(value) = state.queue.get(index).cloned() { - state.readers.insert(this.reader, position + 1); - state.trim(); - return Poll::Ready(Some(Ok(value))); - } - if state.finished { - this.done = true; - state.readers.remove(&this.reader); - let failure = state.failure.clone(); - state.trim(); - return Poll::Ready(failure.map(Err)); - } - state.waiters.insert(this.reader, cx.waker().clone()); - if state.queue.len() >= state.context.control.limits.max_buffered_batches { - return Poll::Pending; - } - // Always-ready sources must still give cancellation and other roots a turn. - if state.batches_polled >= 32 { - state.batches_polled = 0; - cx.waker().wake_by_ref(); - return Poll::Pending; - } - let polled = state - .stream - .as_mut() - .expect("unfinished producer") - .as_mut() - .poll_next(cx); - if matches!(&polled, Poll::Ready(Some(Ok(_)))) { - state.batches_polled += 1; - } - match polled { - Poll::Pending => Poll::Pending, - Poll::Ready(Some(Ok(value))) => match state.context.reserve((state.size)(&value)) { - Ok(reservation) => { - let value = SharedValue { - value: Arc::new(value), - _reservation: Rc::new(reservation), - }; - state.queue.push_back(value.clone()); - state.readers.insert(this.reader, position + 1); - state.trim(); - Poll::Ready(Some(Ok(value))) - } - Err(error) => { - state.failure = Some(error.clone()); - state.finished = true; - state.stream = None; - this.done = true; - state.readers.remove(&this.reader); - state.trim(); - Poll::Ready(Some(Err(error))) - } - }, - Poll::Ready(result) => { - let error = result.and_then(Result::err).map(|source| match source { - Error::AtNode { .. } | Error::Cancelled | Error::MemoryLimit => source, - source => Error::AtNode { - node: state.node, - operation: state.operation.clone(), - source: Box::new(source), - }, - }); - state.failure = error.clone(); - state.finished = true; - state.stream = None; - this.done = true; - state.readers.remove(&this.reader); - state.trim(); - Poll::Ready(error.map(Err)) - } - } - } -} - -pub mod operators; -pub mod values; - -#[cfg(test)] -mod tests; - -pub mod planner; - -pub mod batch_execution; - -pub mod expressions; - -mod temporal; - -pub mod scan; +pub use crate::Error; +pub use crate::{binding as planner, expressions, operators, sources as scan, values}; diff --git a/crates/asap-physical-operators/src/dag/operators.rs b/crates/asap-physical-operators/src/dag/operators.rs deleted file mode 100644 index 38e2f4d3..00000000 --- a/crates/asap-physical-operators/src/dag/operators.rs +++ /dev/null @@ -1,1647 +0,0 @@ -//! Native DAG operators. Engines bind sources; computation lives here. -use super::{ - values::{group_key, Batch, Schema, Value}, - Error, Input, OutputStream, PhysicalOperator, Reservation, RunContext, -}; -use futures::StreamExt; -use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema, SummaryUpdate}, - pre_asap::{ArithmeticOpKind, ColumnRef, DataType}, -}; -use std::{collections::BTreeMap, sync::Arc}; - -fn invalid(message: &str) -> Error { - Error::Invalid(message.into()) -} -fn field(schema: &Schema, column: usize) -> Result<&SummaryField, Error> { - schema - .fields - .get(column) - .ok_or_else(|| invalid("column out of range")) -} -fn plain(schema: &Schema, column: usize) -> Result<(&DataType, bool), Error> { - let f = field(schema, column)?; - let SummaryFamilyType::Plain(dtype) = &f.dtype else { - return Err(invalid("plain value required")); - }; - Ok((dtype, f.nullable)) -} -fn schema(fields: Vec) -> Schema { - Arc::new(SummarySchema { - fields, - time_index: None, - }) -} -fn result_field(name: &str, dtype: DataType, nullable: bool) -> SummaryField { - SummaryField { - name: name.into(), - dtype: SummaryFamilyType::Plain(dtype), - nullable, - } -} - -#[derive(Clone, Debug)] -pub enum Expression { - Binary { - operator: planner_types::post_asap::BinaryOperator, - left: Box, - right: Box, - }, - Planner(Box), - Column(usize), - Literal { - value: Value, - dtype: DataType, - }, - Negate(Box), - Arithmetic { - op: ArithmeticOpKind, - left: Box, - right: Box, - }, - Equal(Box, Box), - Less(Box, Box), - And(Box, Box), - Or(Box, Box), - Not(Box), - IsNull(Box), -} -impl Expression { - pub fn planner(expression: super::expressions::CompiledExpression) -> Self { - Self::Planner(Box::new(expression)) - } - fn dtype(&self, input: &Schema) -> Result<(DataType, bool), Error> { - use Expression::*; - match self { - Binary { - operator, - left, - right, - } => { - use planner_types::pre_asap::{BinaryOpKind, CompareOpKind}; - let (a, n) = left.dtype(input)?; - let (b, m) = right.dtype(input)?; - if a != DataType::Float64 || b != a || operator.vector_match.is_some() { - return Err(invalid( - "binary expression requires resolved Float64 operands", - )); - } - if (operator.checked_relative_division || operator.checked_finite_division) - && operator.kind != BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) - { - return Err(invalid("checked division contract on non-division")); - } - let dtype = match operator.kind { - BinaryOpKind::Arithmetic(_) => DataType::Float64, - BinaryOpKind::Compare( - CompareOpKind::Eq - | CompareOpKind::Ne - | CompareOpKind::Lt - | CompareOpKind::Le - | CompareOpKind::Gt - | CompareOpKind::Ge, - ) => DataType::Bool, - _ => return Err(invalid("unsupported binary operation")), - }; - Ok((dtype, n || m)) - } - Planner(expression) => Ok(expression.dtype()), - Column(i) => { - let (t, n) = plain(input, *i)?; - Ok((t.clone(), n)) - } - Literal { value, dtype } => { - if value.matches(dtype, true) { - Ok((dtype.clone(), matches!(value, Value::Null))) - } else { - Err(invalid("literal type mismatch")) - } - } - Negate(v) => { - let (t, n) = v.dtype(input)?; - if matches!(t, DataType::Int64 | DataType::Float64) { - Ok((t, n)) - } else { - Err(invalid("numeric negation required")) - } - } - Arithmetic { op, left, right } => { - let (a, n) = left.dtype(input)?; - let (b, m) = right.dtype(input)?; - if a == b - && matches!(a, DataType::Int64 | DataType::Float64) - && !(a == DataType::Int64 && *op == ArithmeticOpKind::Atan2) - { - Ok((a, n || m)) - } else { - Err(invalid("arithmetic requires matching numeric types")) - } - } - Equal(a, b) | Less(a, b) => { - let (a, n) = a.dtype(input)?; - let (b, m) = b.dtype(input)?; - if a == b && ordered(&a) { - Ok((DataType::Bool, n || m)) - } else { - Err(invalid("comparison requires matching ordered types")) - } - } - And(a, b) | Or(a, b) => { - let (a, n) = a.dtype(input)?; - let (b, m) = b.dtype(input)?; - if a == DataType::Bool && b == DataType::Bool { - Ok((DataType::Bool, n || m)) - } else { - Err(invalid("boolean operands required")) - } - } - Not(v) => { - let (t, n) = v.dtype(input)?; - if t == DataType::Bool { - Ok((t, n)) - } else { - Err(invalid("boolean operand required")) - } - } - IsNull(v) => { - v.dtype(input)?; - Ok((DataType::Bool, false)) - } - } - } - fn evaluate(&self, row: &[Value]) -> Result { - use Expression::*; - Ok(match self { - Binary { - operator, - left, - right, - } => { - let (a, b) = (left.evaluate(row)?, right.evaluate(row)?); - if matches!(a, Value::Null) || matches!(b, Value::Null) { - Value::Null - } else { - let (Value::Float64(a), Value::Float64(b)) = (a, b) else { - return Err(invalid("binary value schema mismatch")); - }; - crate::arithmetic::evaluate_binary(operator, a, b)? - } - } - Planner(expression) => expression.evaluate(row)?, - Column(i) => row[*i].clone(), - Literal { value, .. } => value.clone(), - Negate(v) => match v.evaluate(row)? { - Value::Int64(v) => Value::Int64( - v.checked_neg() - .ok_or_else(|| invalid("integer negation overflow"))?, - ), - Value::Float64(v) => Value::Float64(-v), - Value::Null => Value::Null, - _ => return Err(invalid("numeric negation required")), - }, - Arithmetic { op, left, right } => { - numeric(op, left.evaluate(row)?, right.evaluate(row)?)? - } - Equal(a, b) | Less(a, b) => { - let (a, b) = (a.evaluate(row)?, b.evaluate(row)?); - if matches!(a, Value::Null) || matches!(b, Value::Null) { - Value::Null - } else if matches!((&a,&b),(Value::Float64(a),Value::Float64(b)) if a.is_nan() || b.is_nan()) - { - Value::Bool(false) - } else { - let c = a.compare(&b)?; - Value::Bool(if matches!(self, Equal(..)) { - c.is_eq() - } else { - c.is_lt() - }) - } - } - And(a, b) | Or(a, b) => { - let (a, b) = (a.evaluate(row)?, b.evaluate(row)?); - match (a, b, matches!(self, And(..))) { - (Value::Bool(false), _, true) | (_, Value::Bool(false), true) => { - Value::Bool(false) - } - (Value::Bool(true), _, false) | (_, Value::Bool(true), false) => { - Value::Bool(true) - } - (Value::Null, _, _) | (_, Value::Null, _) => Value::Null, - (Value::Bool(a), Value::Bool(b), true) => Value::Bool(a && b), - (Value::Bool(a), Value::Bool(b), false) => Value::Bool(a || b), - _ => return Err(invalid("boolean operands required")), - } - } - Not(v) => match v.evaluate(row)? { - Value::Bool(v) => Value::Bool(!v), - Value::Null => Value::Null, - _ => return Err(invalid("boolean operand required")), - }, - IsNull(v) => Value::Bool(matches!(v.evaluate(row)?, Value::Null)), - }) - } -} -fn ordered(dtype: &DataType) -> bool { - if let DataType::Map { key, value, .. } = dtype { - return ordered(key) && ordered(value); - } - matches!( - dtype, - DataType::Null - | DataType::Int64 - | DataType::Float64 - | DataType::Utf8 - | DataType::Bool - | DataType::Timestamp - | DataType::Date - ) -} -pub(super) fn numeric(op: &ArithmeticOpKind, a: Value, b: Value) -> Result { - use ArithmeticOpKind::*; - Ok(match (a, b) { - (Value::Null, _) | (_, Value::Null) => Value::Null, - (Value::Float64(a), Value::Float64(b)) => { - Value::Float64(crate::arithmetic::evaluate_float64_arithmetic(op, a, b)) - } - (Value::Int64(a), Value::Int64(b)) => Value::Int64( - match op { - Add => a.checked_add(b), - Sub => a.checked_sub(b), - Mul => a.checked_mul(b), - Div => a.checked_div(b), - Mod => a.checked_rem(b), - Pow => u32::try_from(b).ok().and_then(|b| a.checked_pow(b)), - Atan2 => None, - } - .ok_or_else(|| invalid("invalid integer arithmetic or overflow"))?, - ), - _ => return Err(invalid("arithmetic type mismatch")), - }) -} -#[derive(Clone, Debug)] -pub struct SortKey { - pub column: usize, - pub descending: bool, - pub nulls_first: bool, -} -#[derive(Clone, Debug)] -pub enum Reduction { - Count, - Sum(usize), - Avg(usize), - Min(usize), - Max(usize), -} -#[derive(Clone)] -enum Kind { - Source(Vec), - Union, - VectorToScalar { - column: usize, - }, - Project(Vec), - Filter(Expression), - Limit { - n: u64, - offset: u64, - groups: Vec, - }, - Sort { - keys: Vec, - groups: Vec, - }, - Window { - intent: Box>, - coordinate: usize, - value: usize, - groups: Vec, - window: Option<(i64, i64)>, - }, - Aggregate { - groups: Vec, - measures: Vec, - }, - SemiJoin { - keys: Vec<(usize, usize)>, - }, - Join { - kind: planner_types::pre_asap::JoinKind, - predicate: Box, - }, - SummaryBuild { - family: SummaryFamilyType, - value: usize, - time: Option, - groups: Vec, - }, - KeyedSummaryBuild { - family: SummaryFamilyType, - value: usize, - items: Vec, - groups: Vec, - }, - KeyedReadout { - state: usize, - k: usize, - }, - SummaryMerge { - state: usize, - groups: Vec, - }, - Readout { - state: usize, - statistic: crate::Statistic, - parameters: std::collections::HashMap, - }, -} -/// A bound operation has a fully checked input/output contract before execution. -#[derive(Clone)] -pub struct Operator { - kind: Kind, - inputs: Vec, - output: Schema, -} -impl Operator { - pub fn source(output: Schema, batches: Vec) -> Result { - super::values::validate_schema(&output)?; - if batches.iter().any(|b| b.schema() != &output) { - return Err(invalid("source schema mismatch")); - } - Ok(Self { - kind: Kind::Source(batches), - inputs: vec![], - output, - }) - } - /// Union polls every input fairly, including branches sharing a producer. - pub fn union(input: Schema, arity: usize) -> Result { - if arity == 0 { - return Err(invalid("union needs at least one input")); - } - Ok(Self { - kind: Kind::Union, - inputs: vec![input.clone(); arity], - output: input, - }) - } - pub fn scalar(value: Value, dtype: DataType) -> Result { - let schema = schema(vec![result_field( - "value", - dtype, - matches!(value, Value::Null), - )]); - Self::source( - schema.clone(), - vec![Batch::try_new(schema, vec![vec![value]])?], - ) - } - /// PromQL scalar conversion: zero or multiple elements produce NaN. - pub fn vector_to_scalar(input: Schema, column: usize) -> Result { - if plain(&input, column)? != (&DataType::Float64, false) { - return Err(invalid("scalar conversion requires non-null Float64")); - } - Ok(Self { - kind: Kind::VectorToScalar { column }, - inputs: vec![input], - output: schema(vec![result_field("value", DataType::Float64, false)]), - }) - } - pub fn project(input: Schema, columns: Vec<(String, Expression)>) -> Result { - let fields = columns - .iter() - .map(|(name, e)| { - let (t, n) = e.dtype(&input)?; - Ok(result_field(name, t, n)) - }) - .collect::>()?; - Ok(Self { - kind: Kind::Project(columns.into_iter().map(|(_, e)| e).collect()), - inputs: vec![input], - output: schema(fields), - }) - } - pub fn filter(input: Schema, predicate: Expression) -> Result { - if predicate.dtype(&input)?.0 != DataType::Bool { - return Err(invalid("filter predicate must be boolean")); - } - Ok(Self { - kind: Kind::Filter(predicate), - inputs: vec![input.clone()], - output: input, - }) - } - pub fn limit(input: Schema, n: u64, offset: u64, groups: Vec) -> Result { - validate_groups(&input, &groups)?; - Ok(Self { - kind: Kind::Limit { n, offset, groups }, - inputs: vec![input.clone()], - output: input, - }) - } - pub fn sort(input: Schema, keys: Vec, groups: Vec) -> Result { - validate_groups(&input, &groups)?; - for key in &keys { - if !ordered(plain(&input, key.column)?.0) { - return Err(invalid("unsupported sort type")); - } - } - Ok(Self { - kind: Kind::Sort { keys, groups }, - inputs: vec![input.clone()], - output: input, - }) - } - pub fn aggregate( - input: Schema, - groups: Vec, - measures: Vec<(String, Reduction)>, - ) -> Result { - validate_groups(&input, &groups)?; - let mut fields = groups - .iter() - .map(|&i| input.fields[i].clone()) - .collect::>(); - for (name, reduction) in &measures { - let (t, n) = match reduction { - Reduction::Count => (DataType::Int64, false), - Reduction::Sum(i) | Reduction::Avg(i) => { - let (t, _) = plain(&input, *i)?; - if !matches!(t, DataType::Int64 | DataType::Float64) { - return Err(invalid("numeric aggregate input required")); - } - ( - if matches!(reduction, Reduction::Avg(_)) { - DataType::Float64 - } else { - t.clone() - }, - false, - ) - } - Reduction::Min(i) | Reduction::Max(i) => { - let (t, nullable) = plain(&input, *i)?; - if !ordered(t) { - return Err(invalid("ordered aggregate input required")); - } - (t.clone(), nullable || groups.is_empty()) - } - }; - fields.push(result_field(name, t, n)); - } - Ok(Self { - kind: Kind::Aggregate { - groups, - measures: measures.into_iter().map(|(_, r)| r).collect(), - }, - inputs: vec![input], - output: schema(fields), - }) - } - /// Bind a Planner temporal or histogram intent to explicit columns and window. - pub fn window( - input: Schema, - intent: planner_types::pre_asap::AggIntent, - coordinate: usize, - value: usize, - groups: Vec, - window: Option<(i64, i64)>, - ) -> Result { - use planner_types::pre_asap::AggIntent; - validate_groups(&input, &groups)?; - let histogram = matches!(intent, AggIntent::HistogramQuantile { .. }); - if !matches!( - intent, - AggIntent::Rate - | AggIntent::Increase - | AggIntent::Count { .. } - | AggIntent::Sum { col: None } - | AggIntent::Avg { col: None } - | AggIntent::Min { col: None } - | AggIntent::Max { col: None } - | AggIntent::HistogramQuantile { .. } - ) { - return Err(invalid( - "unsupported temporal intent or unresolved value column", - )); - } - let coordinate_type = if histogram { - DataType::Float64 - } else { - DataType::Timestamp - }; - if plain(&input, coordinate)? != (&coordinate_type, false) - || plain(&input, value)? != (&DataType::Float64, false) - { - return Err(invalid("window coordinate/value schema mismatch")); - } - if (!histogram && !matches!(window, Some((start, end)) if start < end)) - || (histogram && window.is_some()) - { - return Err(invalid("invalid temporal window")); - } - let mut fields = groups - .iter() - .map(|i| input.fields[*i].clone()) - .collect::>(); - fields.push(result_field( - "value", - if matches!(intent, AggIntent::Count { .. }) { - DataType::Int64 - } else { - DataType::Float64 - }, - false, - )); - Ok(Self { - kind: Kind::Window { - intent: Box::new(intent), - coordinate, - value, - groups, - window, - }, - inputs: vec![input], - output: schema(fields), - }) - } - - pub fn semi_join( - left: Schema, - right: Schema, - keys: Vec<(usize, usize)>, - ) -> Result { - if keys.is_empty() { - return Err(invalid("semi-join needs matching keys")); - } - for &(l, r) in &keys { - if plain(&left, l)?.0 != plain(&right, r)?.0 { - return Err(invalid("join key types differ")); - } - } - Ok(Self { - kind: Kind::SemiJoin { keys }, - inputs: vec![left.clone(), right], - output: left, - }) - } - pub fn relational_join( - left: Schema, - right: Schema, - kind: planner_types::pre_asap::JoinKind, - predicate: &planner_types::pre_asap::Predicate, - output: Schema, - ) -> Result { - use planner_types::pre_asap::JoinKind; - let mut joined = left.fields.clone(); - joined.extend(right.fields.clone()); - let predicate = - super::expressions::CompiledExpression::compile(&predicate.0, &schema(joined.clone()))?; - if predicate.dtype().0 != DataType::Bool { - return Err(invalid("join predicate must be boolean")); - } - let fields = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { - left.fields.clone() - } else { - for field in &mut joined[..left.fields.len()] { - if matches!(kind, JoinKind::Right | JoinKind::Full) { - field.nullable = true; - } - } - for field in &mut joined[left.fields.len()..] { - if matches!(kind, JoinKind::Left | JoinKind::Full) { - field.nullable = true; - } - } - joined - }; - Self { - kind: Kind::Join { - kind, - predicate: Box::new(predicate), - }, - inputs: vec![left, right], - output: schema(fields), - } - .with_output_schema(output) - } - pub fn summary_build( - input: Schema, - family: SummaryFamilyType, - value: usize, - time: Option, - groups: Vec, - ) -> Result { - super::values::validate_family(&family)?; - validate_groups(&input, &groups)?; - if plain(&input, value)? != (&DataType::Float64, false) { - return Err(invalid("summary numeric update requires non-null Float64")); - } - if let Some(time) = time { - if plain(&input, time)? != (&DataType::Timestamp, false) { - return Err(invalid("summary time column must be a timestamp")); - } - } - if time.is_none() - && matches!( - family, - SummaryFamilyType::ExactAggregate( - planner_types::post_asap::ExactKind::Rate - | planner_types::post_asap::ExactKind::Increase, - _ - ) - ) - { - return Err(invalid("counter summary requires a timestamp column")); - } - crate::capability::validate_summary_kernel( - &family, - &SummaryUpdate::column(ColumnRef::SampleValue), - &Default::default(), - ) - .map_err(Error::Invalid)?; - let mut fields = groups - .iter() - .map(|&i| input.fields[i].clone()) - .collect::>(); - fields.push(SummaryField { - name: "state".into(), - dtype: family.clone(), - nullable: false, - }); - Ok(Self { - kind: Kind::SummaryBuild { - family, - value, - time, - groups, - }, - inputs: vec![input], - output: schema(fields), - }) - } - pub fn keyed_summary_build( - input: Schema, - family: SummaryFamilyType, - value: usize, - items: Vec, - groups: Vec, - ) -> Result { - use planner_types::post_asap::{SketchAlgorithm, SketchParams}; - super::values::validate_family(&family)?; - let SummaryFamilyType::Sketch(kind, _) = &family else { - return Err(invalid("keyed sketch required")); - }; - if kind.algorithm() != &SketchAlgorithm::CmsWithHeap - || !matches!(kind.params(), SketchParams::CmsWithHeap { .. }) - { - return Err(invalid( - "Float64 weighted keyed construction currently supports CMS with heap", - )); - } - validate_groups(&input, &groups)?; - if items.is_empty() || plain(&input, value)? != (&DataType::Float64, false) { - return Err(invalid( - "keyed summary requires identities and non-null Float64 weights", - )); - } - for &item in &items { - if !matches!( - plain(&input, item)?.0, - DataType::Utf8 - | DataType::Int64 - | DataType::Float64 - | DataType::Bool - | DataType::Null - ) { - return Err(invalid("unsupported keyed summary identity type")); - } - } - let mut fields = groups - .iter() - .map(|&i| input.fields[i].clone()) - .collect::>(); - fields.push(SummaryField { - name: "state".into(), - dtype: family.clone(), - nullable: false, - }); - Ok(Self { - kind: Kind::KeyedSummaryBuild { - family, - value, - items, - groups, - }, - inputs: vec![input], - output: schema(fields), - }) - } - pub fn keyed_readout( - input: Schema, - state: usize, - k: usize, - output: Schema, - ) -> Result { - use planner_types::post_asap::{SketchAlgorithm, SketchParams}; - let SummaryFamilyType::Sketch(kind, _) = &field(&input, state)?.dtype else { - return Err(invalid("keyed readout requires summary state")); - }; - let SketchParams::CmsWithHeap { heap_size, .. } = kind.params() else { - return Err(invalid("unsupported keyed readout family")); - }; - if kind.algorithm() != &SketchAlgorithm::CmsWithHeap - || k > *heap_size as usize - || output.fields.len() <= input.fields.len() - { - return Err(invalid("invalid keyed readout shape or capacity")); - } - if state + 1 != input.fields.len() - || output.fields[..state] != input.fields[..state] - || output.fields.last().unwrap().dtype != SummaryFamilyType::Plain(DataType::Float64) - { - return Err(invalid( - "keyed readout must preserve partitions and return a Float64 score", - )); - } - super::values::validate_schema(&output)?; - Ok(Self { - kind: Kind::KeyedReadout { state, k }, - inputs: vec![input], - output, - }) - } - pub fn summary_merge(input: Schema, state: usize, groups: Vec) -> Result { - validate_groups(&input, &groups)?; - super::values::validate_family(&field(&input, state)?.dtype)?; - if matches!(field(&input, state)?.dtype, SummaryFamilyType::Plain(_)) { - return Err(invalid("summary state required")); - } - let mut fields = groups - .iter() - .map(|&i| input.fields[i].clone()) - .collect::>(); - fields.push(input.fields[state].clone()); - Ok(Self { - kind: Kind::SummaryMerge { state, groups }, - inputs: vec![input], - output: schema(fields), - }) - } - pub fn readout( - input: Schema, - state: usize, - statistic: crate::Statistic, - parameters: std::collections::HashMap, - ) -> Result { - super::values::validate_family(&field(&input, state)?.dtype)?; - if matches!(field(&input, state)?.dtype, SummaryFamilyType::Plain(_)) { - return Err(invalid("summary state required")); - } - validate_readout(&field(&input, state)?.dtype, statistic, ¶meters)?; - let mut fields = input.fields.clone(); - let result_type = if matches!( - fields[state].dtype, - SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) - ) { - DataType::Int64 - } else { - DataType::Float64 - }; - fields[state] = result_field("value", result_type, false); - Ok(Self { - kind: Kind::Readout { - state, - statistic, - parameters, - }, - inputs: vec![input], - output: schema(fields), - }) - } - pub(crate) fn with_output_schema(mut self, output: Schema) -> Result { - if self.output.fields.len() != output.fields.len() - || self - .output - .fields - .iter() - .zip(&output.fields) - .any(|(actual, declared)| { - actual.dtype != declared.dtype || (actual.nullable && !declared.nullable) - }) - { - return Err(invalid("native output type differs from Planner output")); - } - if output.time_index.is_some_and(|i| { - i >= output.fields.len() - || output.fields[i].dtype != SummaryFamilyType::Plain(DataType::Timestamp) - }) { - return Err(invalid("invalid output time column")); - } - self.output = output; - Ok(self) - } - pub fn schema(&self) -> Schema { - self.output.clone() - } -} -fn validate_groups(input: &Schema, groups: &[usize]) -> Result<(), Error> { - for &i in groups { - plain(input, i)?; - } - if groups - .iter() - .collect::>() - .len() - != groups.len() - { - return Err(invalid("duplicate group columns")); - } - Ok(()) -} -async fn collect_rows( - mut input: Input<'_, Batch>, - context: &RunContext, -) -> Result<(Vec>, Vec), Error> { - let mut rows = Vec::new(); - let mut reservations = Vec::new(); - while let Some(batch) = input.next().await { - let batch = batch?; - reservations.push(context.reserve(batch.bytes())?); - rows.extend(batch.rows().iter().cloned()); - } - Ok((rows, reservations)) -} -impl PhysicalOperator for Operator { - fn name(&self) -> &str { - match self.kind { - Kind::Source(_) => "Source", - Kind::Union => "Union", - Kind::VectorToScalar { .. } => "VectorToScalar", - Kind::Project(_) => "Project", - Kind::Filter(_) => "Filter", - Kind::Limit { .. } => "Limit", - Kind::Sort { .. } => "Sort", - Kind::Aggregate { .. } => "Aggregate", - Kind::Window { .. } => "WindowAggregate", - Kind::SemiJoin { .. } => "SemiJoin", - Kind::Join { .. } => "RelationalJoin", - Kind::SummaryBuild { .. } | Kind::KeyedSummaryBuild { .. } => "SummaryAgg", - Kind::KeyedReadout { .. } => "SummaryEstimate", - Kind::SummaryMerge { .. } => "SummaryMerge", - Kind::Readout { .. } => "SummaryReadout", - } - } - fn input_schemas(&self) -> Vec { - self.inputs.clone() - } - fn output_schema(&self) -> Schema { - self.output.clone() - } - fn output_bytes(&self, value: &Batch) -> usize { - value.bytes() - } - fn start<'a>( - &'a self, - mut inputs: Vec>, - context: RunContext, - ) -> Result, Error> { - let output = self.output.clone(); - if let Kind::Source(batches) = &self.kind { - return Ok(futures::stream::iter(batches.iter().cloned().map(Ok)).boxed_local()); - } - if matches!(self.kind, Kind::Union) { - return Ok(futures::stream::select_all(inputs) - .map(|batch| batch.map(|batch| batch.value().clone())) - .boxed_local()); - } - if let Kind::Join { kind, predicate } = &self.kind { - let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; - let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; - return Ok(futures::stream::once(async move { - use planner_types::pre_asap::JoinKind; - let ((left, _left_memory), (right, _right_memory)) = futures::try_join!( - collect_rows(left, &context), - collect_rows(right, &context) - )?; - let mut result = Vec::new(); - let mut right_matched = vec![false; right.len()]; - for left_row in &left { - let mut matched = false; - for (i, right_row) in right.iter().enumerate() { - let mut joined = left_row.clone(); - joined.extend(right_row.iter().cloned()); - if *kind == JoinKind::Cross - || matches!(predicate.evaluate(&joined)?, Value::Bool(true)) - { - matched = true; - right_matched[i] = true; - match kind { - JoinKind::Semi => { - result.push(left_row.clone()); - break; - } - JoinKind::Anti => break, - _ => result.push(joined), - } - } - } - if !matched { - match kind { - JoinKind::Left | JoinKind::Full => { - let mut joined = left_row.clone(); - joined.resize( - joined.len() + self.inputs[1].fields.len(), - Value::Null, - ); - result.push(joined); - } - JoinKind::Anti => result.push(left_row.clone()), - _ => {} - } - } - } - if matches!(kind, JoinKind::Right | JoinKind::Full) { - for (matched, row) in right_matched.into_iter().zip(right) { - if !matched { - let mut joined = vec![Value::Null; self.inputs[0].fields.len()]; - joined.extend(row); - result.push(joined); - } - } - } - Batch::try_new(output, result) - }) - .boxed_local()); - } - if let Kind::SemiJoin { keys } = &self.kind { - let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; - let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; - return Ok(futures::stream::once(async move { - // Poll both branches together: either may depend on a common producer. - let ((left, _left_memory), (right, _right_memory)) = futures::try_join!( - collect_rows(left, &context), - collect_rows(right, &context) - )?; - let right_cols = keys.iter().map(|(_, r)| *r).collect::>(); - let left_cols = keys.iter().map(|(l, _)| *l).collect::>(); - let members = right - .iter() - .filter(|row| right_cols.iter().all(|&i| !matches!(row[i], Value::Null))) - .map(|r| group_key(r, &right_cols)) - .collect::, _>>()?; - let rows = left - .into_iter() - .filter_map(|r| match group_key(&r, &left_cols) { - Ok(k) - if left_cols.iter().all(|&i| !matches!(r[i], Value::Null)) - && members.contains(&k) => - { - Some(Ok(r)) - } - Ok(_) => None, - Err(e) => Some(Err(e)), - }) - .collect::, _>>()?; - Batch::try_new(output, rows) - }) - .boxed_local()); - } - let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; - match &self.kind { - Kind::VectorToScalar { column } => Ok(futures::stream::once(async move { - let mut input = input; - let mut value = f64::NAN; - let mut count = 0usize; - while let Some(batch) = input.next().await { - for row in batch?.rows() { - count = count.saturating_add(1); - if let Value::Float64(v) = row[*column] { - value = v; - } - } - } - Batch::try_new( - output, - vec![vec![Value::Float64(if count == 1 { - value - } else { - f64::NAN - })]], - ) - }) - .boxed_local()), - Kind::Project(expressions) => Ok(input - .map(move |batch| { - let batch = batch?; - let rows = batch - .rows() - .iter() - .map(|r| { - expressions - .iter() - .map(|e| e.evaluate(r)) - .collect::, _>>() - }) - .collect::, _>>()?; - Batch::try_new(output.clone(), rows) - }) - .boxed_local()), - Kind::Filter(predicate) => Ok(input - .map(move |batch| { - let batch = batch?; - let mut rows = Vec::new(); - for row in batch.rows() { - if matches!(predicate.evaluate(row)?, Value::Bool(true)) { - rows.push(row.clone()); - } - } - Batch::try_new(output.clone(), rows) - }) - .boxed_local()), - Kind::Limit { n, offset, groups } => { - let counts = BTreeMap::>, u64>::new(); - Ok(futures::stream::try_unfold( - (input, counts, Vec::::new(), false), - move |(mut input, mut counts, mut memory, done)| { - let output = output.clone(); - let context = context.clone(); - async move { - if done || *n == 0 { - return Ok(None); - } - let Some(batch) = input.next().await else { - return Ok(None); - }; - let batch = batch?; - let mut rows = Vec::new(); - for row in batch.rows() { - let key = group_key(row, groups)?; - if !counts.contains_key(&key) { - memory.push( - context.reserve( - key.iter() - .map(|part| { - part.len() + std::mem::size_of::>() - }) - .sum::() - + 64, - )?, - ); - } - let count = counts.entry(key).or_default(); - if *count >= *offset && count.saturating_sub(*offset) < *n { - rows.push(row.clone()); - } - *count = count.saturating_add(1); - } - let done = groups.is_empty() - && counts - .get(&vec![]) - .is_some_and(|count| count.saturating_sub(*offset) >= *n); - Ok(Some(( - Batch::try_new(output, rows)?, - (input, counts, memory, done), - ))) - } - }, - ) - .boxed_local()) - } - Kind::SummaryBuild { - family, - value, - time, - groups, - } => Ok(futures::stream::once(async move { - Batch::try_new( - output, - build_summary(input, family, *value, *time, groups, &context).await?, - ) - }) - .boxed_local()), - Kind::KeyedSummaryBuild { - family, - value, - items, - groups, - } => Ok(futures::stream::once(async move { - Batch::try_new( - output, - build_keyed_summary(input, family, *value, items, groups, &context).await?, - ) - }) - .boxed_local()), - Kind::KeyedReadout { state, k } => Ok(input - .map(move |batch| { - let batch = batch?; - let mut rows = Vec::new(); - for row in batch.rows() { - let Value::Summary { state: summary, .. } = &row[*state] else { - return Err(invalid("summary value required")); - }; - let summary = summary - .as_any() - .downcast_ref::() - .ok_or_else(|| invalid("weighted CMS typed state required"))?; - for items in summary.rows(*k) { - let mut values = row[..*state].to_vec(); - values.extend(items); - rows.push(values); - } - } - Batch::try_new(output.clone(), rows) - }) - .boxed_local()), - Kind::Readout { - state, - statistic, - parameters, - } => Ok(input - .map(move |batch| { - let batch = batch?; - let mut rows = batch.rows().to_vec(); - for row in &mut rows { - let Value::Summary { state: summary, .. } = &row[*state] else { - return Err(invalid("summary value required")); - }; - row[*state] = if output.fields[*state].dtype - == SummaryFamilyType::Plain(DataType::Int64) - { - let count = summary.aux_stats().count.ok_or_else(|| { - Error::Operator("exact count state lacks an integer count".into()) - })?; - Value::Int64( - i64::try_from(count).map_err(|_| { - Error::Operator("exact count exceeds Int64".into()) - })?, - ) - } else { - Value::Float64( - summary - .query_statistic(*statistic, &None, parameters) - .map_err(|e| Error::Operator(e.to_string()))?, - ) - }; - } - Batch::try_new(output.clone(), rows) - }) - .boxed_local()), - _ => Ok(futures::stream::once(async move { - let (rows, _memory) = collect_rows(input, &context).await?; - let result = match &self.kind { - Kind::Sort { keys, groups } => { - let mut grouped = BTreeMap::>, Vec>>::new(); - for row in rows { - for key in keys { - if matches!(row[key.column], Value::Map(_)) - && nested_nan(&row[key.column]) - { - return Err(invalid("NaN in collection sort key")); - } - } - grouped - .entry(group_key(&row, groups)?) - .or_default() - .push(row); - } - let mut result = Vec::new(); - for mut rows in grouped.into_values() { - rows.sort_by(|a, b| compare_rows(a, b, keys)); - result.extend(rows); - } - result - } - Kind::Window { - intent, - coordinate, - value, - groups, - window, - } => { - super::temporal::reduce(rows, intent, groups, *coordinate, *value, *window)? - } - Kind::Aggregate { groups, measures } => { - reduce(rows, groups, measures, &self.inputs[0])? - } - Kind::SummaryMerge { state, groups } => merge_summary(rows, *state, groups)?, - _ => return Err(invalid("unexpected blocking operation")), - }; - Batch::try_new(output, result) - }) - .boxed_local()), - } - } -} -fn compare_rows(a: &[Value], b: &[Value], keys: &[SortKey]) -> std::cmp::Ordering { - use std::cmp::Ordering::*; - for key in keys { - let (a, b) = (&a[key.column], &b[key.column]); - let order = match (a, b) { - (Value::Null, Value::Null) => Equal, - (Value::Null, _) => { - if key.nulls_first { - Less - } else { - Greater - } - } - (_, Value::Null) => { - if key.nulls_first { - Greater - } else { - Less - } - } - (Value::Float64(a), Value::Float64(b)) if a.is_nan() || b.is_nan() => { - match (a.is_nan(), b.is_nan()) { - (true, true) => Equal, - (true, false) => Greater, - _ => Less, - } - } - _ => { - let order = a.compare(b).expect("bound ordered types"); - if key.descending { - order.reverse() - } else { - order - } - } - }; - if order != Equal { - return order; - } - } - Equal -} -fn reduce( - rows: Vec>, - groups: &[usize], - measures: &[Reduction], - input: &Schema, -) -> Result>, Error> { - let mut grouped = BTreeMap::>, Vec>>::new(); - if rows.is_empty() && groups.is_empty() { - grouped.insert(vec![], vec![]); - } - for row in rows { - grouped - .entry(group_key(&row, groups)?) - .or_default() - .push(row); - } - grouped - .into_values() - .map(|rows| { - let mut result = groups - .iter() - .map(|&i| rows[0][i].clone()) - .collect::>(); - for measure in measures { - result.push(reduce_one(&rows, measure, input)?); - } - Ok(result) - }) - .collect() -} -fn reduce_one(rows: &[Vec], measure: &Reduction, input: &Schema) -> Result { - let column = match measure { - Reduction::Count => { - return Ok(Value::Int64( - i64::try_from(rows.len()).map_err(|_| invalid("count overflow"))?, - )) - } - Reduction::Sum(i) | Reduction::Avg(i) | Reduction::Min(i) | Reduction::Max(i) => *i, - }; - let values = rows - .iter() - .map(|r| &r[column]) - .filter(|v| !matches!(v, Value::Null)) - .collect::>(); - if matches!(measure, Reduction::Min(_) | Reduction::Max(_)) { - if plain(input, column)?.0 == &DataType::Float64 { - // Match exact-state kernels: ignore NaN when a numeric value exists. - let mut best: Option = None; - for value in values { - let Value::Float64(value) = value else { - return Err(invalid("floating aggregate value required")); - }; - best = Some(best.map_or(*value, |old| { - if matches!(measure, Reduction::Min(_)) { - old.min(*value) - } else { - old.max(*value) - } - })); - } - return Ok(best.map(Value::Float64).unwrap_or(Value::Null)); - } - let mut best: Option<&Value> = None; - for value in values { - if best - .map(|b| value.compare(b)) - .transpose()? - .is_none_or(|order| { - if matches!(measure, Reduction::Min(_)) { - order.is_lt() - } else { - order.is_gt() - } - }) - { - best = Some(value); - } - } - return Ok(best.cloned().unwrap_or(Value::Null)); - } - let count = values.len(); - let dtype = plain(input, column)?.0; - if dtype == &DataType::Int64 { - let sum = values.into_iter().try_fold(0i128, |sum, v| { - let Value::Int64(v) = v else { - return Err(invalid("integer aggregate value required")); - }; - sum.checked_add(i128::from(*v)) - .ok_or_else(|| invalid("integer aggregate overflow")) - })?; - return if matches!(measure, Reduction::Avg(_)) { - Ok(Value::Float64(sum as f64 / count as f64)) - } else { - Ok(Value::Int64( - i64::try_from(sum).map_err(|_| invalid("integer sum overflow"))?, - )) - }; - } - let sum = values - .into_iter() - .map(|v| { - if let Value::Float64(v) = v { - *v - } else { - unreachable!() - } - }) - .sum::(); - Ok(Value::Float64(if matches!(measure, Reduction::Avg(_)) { - sum / count as f64 - } else { - sum - })) -} -async fn build_keyed_summary( - mut input: Input<'_, Batch>, - family: &SummaryFamilyType, - value: usize, - items: &[usize], - groups: &[usize], - context: &RunContext, -) -> Result>, Error> { - use crate::{summary_operators::weighted_cms::WeightedCms, AggregateCore}; - use planner_types::post_asap::SketchParams; - let SummaryFamilyType::Sketch(kind, _) = family else { - unreachable!() - }; - let SketchParams::CmsWithHeap { - width, - depth, - heap_size, - } = kind.params() - else { - unreachable!() - }; - let mut states = BTreeMap::>, (Vec, WeightedCms, Reservation, usize)>::new(); - while let Some(batch) = input.next().await { - let batch = batch?; - for row in batch.rows() { - if context.is_cancelled() { - return Err(Error::Operator("execution cancelled".into())); - } - let key = group_key(row, groups)?; - if !states.contains_key(&key) { - let labels = groups.iter().map(|&i| row[i].clone()).collect::>(); - let overhead = labels.iter().map(Value::bytes).sum::() - + key.iter().map(|v| v.len() + 24).sum::() - + 128; - let bytes = (*width as usize) - .checked_mul(*depth as usize) - .and_then(|n| n.checked_mul(8)) - .and_then(|n| n.checked_add(overhead)) - .ok_or_else(|| invalid("weighted CMS memory size overflow"))?; - let reservation = context.reserve(bytes)?; - states.insert( - key.clone(), - ( - labels, - WeightedCms::new(*width as usize, *depth as usize, *heap_size as usize)?, - reservation, - overhead, - ), - ); - } - let (_, summary, reservation, overhead) = states.get_mut(&key).unwrap(); - let Value::Float64(weight) = row[value] else { - return Err(invalid("weighted CMS weight type")); - }; - summary.update( - &items.iter().map(|&i| row[i].clone()).collect::>(), - weight, - )?; - reservation.resize(summary.approx_memory_bytes() + *overhead)?; - } - } - Ok(states - .into_values() - .map(|(mut labels, summary, _, _)| { - labels.push(Value::Summary { - family: family.clone(), - state: Arc::new(summary), - }); - labels - }) - .collect()) -} - -async fn build_summary( - mut input: Input<'_, Batch>, - family: &SummaryFamilyType, - value: usize, - time: Option, - groups: &[usize], - context: &RunContext, -) -> Result>, Error> { - type State = ( - Vec, - Box, - Reservation, - usize, - Option, - ); - let create = |labels: Vec, key_bytes: usize| -> Result { - let updater = crate::factory::create_planner_accumulator( - family, - &SummaryUpdate::column(ColumnRef::SampleValue), - &Default::default(), - ) - .map_err(Error::Operator)?; - let overhead = labels.iter().map(Value::bytes).sum::() + key_bytes + 64; - let memory = context.reserve(updater.memory_usage_bytes() + overhead)?; - Ok((labels, updater, memory, overhead, None)) - }; - let mut states = BTreeMap::>, State>::new(); - if groups.is_empty() { - states.insert(vec![], create(vec![], 0)?); - } - let ordered_time = matches!( - family, - SummaryFamilyType::ExactAggregate( - planner_types::post_asap::ExactKind::Rate - | planner_types::post_asap::ExactKind::Increase, - _ - ) - ); - while let Some(batch) = input.next().await { - let batch = batch?; - for row in batch.rows() { - let key = group_key(row, groups)?; - if !states.contains_key(&key) { - let labels = groups.iter().map(|&i| row[i].clone()).collect(); - let state = create( - labels, - key.iter() - .map(|v| v.len() + std::mem::size_of::>()) - .sum(), - )?; - states.insert(key.clone(), state); - } - let (_, updater, memory, overhead, previous) = - states.get_mut(&key).expect("inserted group"); - let Value::Float64(value) = row[value] else { - return Err(invalid("summary update type")); - }; - let timestamp = if let Some(time) = time { - let Value::Timestamp(time) = row[time] else { - return Err(invalid("summary time type")); - }; - time - } else { - 0 - }; - if ordered_time && previous.is_some_and(|prior| timestamp <= prior) { - return Err(Error::Operator( - "counter samples must have strictly increasing timestamps within each group" - .into(), - )); - } - updater - .validate_single_input(value) - .map_err(Error::Operator)?; - updater.update_single(value, timestamp); - *previous = Some(timestamp); - memory.resize(updater.memory_usage_bytes() + *overhead)?; - } - } - Ok(states - .into_values() - .map(|(mut labels, updater, _memory, _, _)| { - labels.push(Value::Summary { - family: family.clone(), - state: Arc::from(updater.into_accumulator()), - }); - labels - }) - .collect()) -} - -fn merge_summary( - rows: Vec>, - state_column: usize, - groups: &[usize], -) -> Result>, Error> { - type GroupState = (Vec, SummaryFamilyType, Arc); - let mut states: BTreeMap>, GroupState> = BTreeMap::new(); - for row in rows { - let Value::Summary { family, state } = &row[state_column] else { - return Err(invalid("summary state required")); - }; - let key = group_key(&row, groups)?; - if let Some((_, expected, existing)) = states.get_mut(&key) { - if expected != family { - return Err(invalid("incompatible summary family")); - } - *existing = Arc::from( - existing - .merge_with(state.as_ref()) - .map_err(|e| Error::Operator(e.to_string()))?, - ); - } else { - states.insert( - key, - ( - groups.iter().map(|&i| row[i].clone()).collect(), - family.clone(), - state.clone(), - ), - ); - } - } - Ok(states - .into_values() - .map(|(mut keys, family, state)| { - keys.push(Value::Summary { family, state }); - keys - }) - .collect()) -} - -fn validate_readout( - family: &SummaryFamilyType, - statistic: crate::Statistic, - parameters: &std::collections::HashMap, -) -> Result<(), Error> { - use crate::Statistic as S; - use planner_types::post_asap::{ExactKind as E, SketchAlgorithm as A}; - let supported = match family { - SummaryFamilyType::ExactAggregate(kind, _) => matches!( - (kind, statistic), - (E::Sum, S::Sum) - | (E::Count, S::Count) - | (E::Min, S::Min) - | (E::Max, S::Max) - | (E::Rate, S::Rate) - | (E::Increase, S::Increase) - ), - SummaryFamilyType::Sketch(kind, _) => match kind.algorithm() { - A::Kll => statistic == S::Quantile, - A::DDSketch => matches!(statistic, S::Quantile | S::Count), - A::Hll => matches!(statistic, S::Cardinality | S::Count), - _ => false, - }, - _ => false, - }; - if !supported { - return Err(invalid( - "readout is not implemented for this summary family", - )); - } - if statistic == S::Quantile - && !parameters - .get("quantile") - .and_then(|s| s.parse::().ok()) - .is_some_and(|q| (0.0..=1.0).contains(&q)) - { - return Err(invalid("quantile readout requires quantile in [0,1]")); - } - Ok(()) -} - -fn nested_nan(value: &Value) -> bool { - match value { - Value::Float64(value) => value.is_nan(), - Value::Map(values) => values - .iter() - .any(|(key, value)| nested_nan(key) || nested_nan(value)), - Value::List(values) | Value::Struct(values) => values.iter().any(nested_nan), - _ => false, - } -} diff --git a/crates/asap-physical-operators/src/error.rs b/crates/asap-physical-operators/src/error.rs new file mode 100644 index 00000000..afee33d8 --- /dev/null +++ b/crates/asap-physical-operators/src/error.rs @@ -0,0 +1,18 @@ +use crate::plan::NodeId; +#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] +pub enum Error { + #[error("invalid DAG: {0}")] + Invalid(String), + #[error("operator failed: {0}")] + Operator(String), + #[error("node {node} ({operation}) failed: {source}")] + AtNode { + node: NodeId, + operation: String, + source: Box, + }, + #[error("execution memory limit exceeded")] + MemoryLimit, + #[error("execution cancelled")] + Cancelled, +} diff --git a/crates/asap-physical-operators/src/arithmetic.rs b/crates/asap-physical-operators/src/expressions/arithmetic.rs similarity index 95% rename from crates/asap-physical-operators/src/arithmetic.rs rename to crates/asap-physical-operators/src/expressions/arithmetic.rs index 5de95606..30e277d4 100644 --- a/crates/asap-physical-operators/src/arithmetic.rs +++ b/crates/asap-physical-operators/src/expressions/arithmetic.rs @@ -23,8 +23,8 @@ pub fn evaluate_binary( operator: &planner_types::post_asap::BinaryOperator, left: f64, right: f64, -) -> Result { - use crate::dag::{values::Value, Error}; +) -> Result { + use crate::{values::Value, Error}; use planner_types::pre_asap::{ArithmeticOpKind, BinaryOpKind, CompareOpKind}; let invalid = || Error::Invalid("unsupported binary operation or invalid checked-division domain".into()); diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs new file mode 100644 index 00000000..2ed99a36 --- /dev/null +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -0,0 +1,252 @@ +//! Scalar semantics and typed expression binding. Planner expressions enter through CompiledExpression. +use crate::{ + values::{plain, Schema, Value}, + Error, +}; +use planner_types::pre_asap::{ArithmeticOpKind, DataType}; +pub mod arithmetic; +mod planner; +pub use planner::CompiledExpression; +#[derive(Clone, Debug)] +pub enum Expression { + Binary { + operator: planner_types::post_asap::BinaryOperator, + left: Box, + right: Box, + }, + Planner(Box), + Column(usize), + Literal { + value: Value, + dtype: DataType, + }, + Negate(Box), + Arithmetic { + op: ArithmeticOpKind, + left: Box, + right: Box, + }, + Equal(Box, Box), + Less(Box, Box), + And(Box, Box), + Or(Box, Box), + Not(Box), + IsNull(Box), +} +impl Expression { + pub fn planner(expression: crate::expressions::CompiledExpression) -> Self { + Self::Planner(Box::new(expression)) + } + pub(crate) fn dtype(&self, input: &Schema) -> Result<(DataType, bool), Error> { + use Expression::*; + match self { + Binary { + operator, + left, + right, + } => { + use planner_types::pre_asap::{BinaryOpKind, CompareOpKind}; + let (a, n) = left.dtype(input)?; + let (b, m) = right.dtype(input)?; + if a != DataType::Float64 || b != a || operator.vector_match.is_some() { + return Err(invalid( + "binary expression requires resolved Float64 operands", + )); + } + if (operator.checked_relative_division || operator.checked_finite_division) + && operator.kind != BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + { + return Err(invalid("checked division contract on non-division")); + } + let dtype = match operator.kind { + BinaryOpKind::Arithmetic(_) => DataType::Float64, + BinaryOpKind::Compare( + CompareOpKind::Eq + | CompareOpKind::Ne + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge, + ) => DataType::Bool, + _ => return Err(invalid("unsupported binary operation")), + }; + Ok((dtype, n || m)) + } + Planner(expression) => Ok(expression.dtype()), + Column(i) => { + let (t, n) = plain(input, *i)?; + Ok((t.clone(), n)) + } + Literal { value, dtype } => { + if value.matches(dtype, true) { + Ok((dtype.clone(), matches!(value, Value::Null))) + } else { + Err(invalid("literal type mismatch")) + } + } + Negate(v) => { + let (t, n) = v.dtype(input)?; + if matches!(t, DataType::Int64 | DataType::Float64) { + Ok((t, n)) + } else { + Err(invalid("numeric negation required")) + } + } + Arithmetic { op, left, right } => { + let (a, n) = left.dtype(input)?; + let (b, m) = right.dtype(input)?; + if a == b + && matches!(a, DataType::Int64 | DataType::Float64) + && !(a == DataType::Int64 && *op == ArithmeticOpKind::Atan2) + { + Ok((a, n || m)) + } else { + Err(invalid("arithmetic requires matching numeric types")) + } + } + Equal(a, b) | Less(a, b) => { + let (a, n) = a.dtype(input)?; + let (b, m) = b.dtype(input)?; + if a == b && ordered(&a) { + Ok((DataType::Bool, n || m)) + } else { + Err(invalid("comparison requires matching ordered types")) + } + } + And(a, b) | Or(a, b) => { + let (a, n) = a.dtype(input)?; + let (b, m) = b.dtype(input)?; + if a == DataType::Bool && b == DataType::Bool { + Ok((DataType::Bool, n || m)) + } else { + Err(invalid("boolean operands required")) + } + } + Not(v) => { + let (t, n) = v.dtype(input)?; + if t == DataType::Bool { + Ok((t, n)) + } else { + Err(invalid("boolean operand required")) + } + } + IsNull(v) => { + v.dtype(input)?; + Ok((DataType::Bool, false)) + } + } + } + pub(crate) fn evaluate(&self, row: &[Value]) -> Result { + use Expression::*; + Ok(match self { + Binary { + operator, + left, + right, + } => { + let (a, b) = (left.evaluate(row)?, right.evaluate(row)?); + if matches!(a, Value::Null) || matches!(b, Value::Null) { + Value::Null + } else { + let (Value::Float64(a), Value::Float64(b)) = (a, b) else { + return Err(invalid("binary value schema mismatch")); + }; + arithmetic::evaluate_binary(operator, a, b)? + } + } + Planner(expression) => expression.evaluate(row)?, + Column(i) => row[*i].clone(), + Literal { value, .. } => value.clone(), + Negate(v) => match v.evaluate(row)? { + Value::Int64(v) => Value::Int64( + v.checked_neg() + .ok_or_else(|| invalid("integer negation overflow"))?, + ), + Value::Float64(v) => Value::Float64(-v), + Value::Null => Value::Null, + _ => return Err(invalid("numeric negation required")), + }, + Arithmetic { op, left, right } => { + numeric(op, left.evaluate(row)?, right.evaluate(row)?)? + } + Equal(a, b) | Less(a, b) => { + let (a, b) = (a.evaluate(row)?, b.evaluate(row)?); + if matches!(a, Value::Null) || matches!(b, Value::Null) { + Value::Null + } else if matches!((&a,&b),(Value::Float64(a),Value::Float64(b)) if a.is_nan() || b.is_nan()) + { + Value::Bool(false) + } else { + let c = a.compare(&b)?; + Value::Bool(if matches!(self, Equal(..)) { + c.is_eq() + } else { + c.is_lt() + }) + } + } + And(a, b) | Or(a, b) => { + let (a, b) = (a.evaluate(row)?, b.evaluate(row)?); + match (a, b, matches!(self, And(..))) { + (Value::Bool(false), _, true) | (_, Value::Bool(false), true) => { + Value::Bool(false) + } + (Value::Bool(true), _, false) | (_, Value::Bool(true), false) => { + Value::Bool(true) + } + (Value::Null, _, _) | (_, Value::Null, _) => Value::Null, + (Value::Bool(a), Value::Bool(b), true) => Value::Bool(a && b), + (Value::Bool(a), Value::Bool(b), false) => Value::Bool(a || b), + _ => return Err(invalid("boolean operands required")), + } + } + Not(v) => match v.evaluate(row)? { + Value::Bool(v) => Value::Bool(!v), + Value::Null => Value::Null, + _ => return Err(invalid("boolean operand required")), + }, + IsNull(v) => Value::Bool(matches!(v.evaluate(row)?, Value::Null)), + }) + } +} +pub(crate) fn ordered(dtype: &DataType) -> bool { + if let DataType::Map { key, value, .. } = dtype { + return ordered(key) && ordered(value); + } + matches!( + dtype, + DataType::Null + | DataType::Int64 + | DataType::Float64 + | DataType::Utf8 + | DataType::Bool + | DataType::Timestamp + | DataType::Date + ) +} +pub(crate) fn numeric(op: &ArithmeticOpKind, a: Value, b: Value) -> Result { + use ArithmeticOpKind::*; + Ok(match (a, b) { + (Value::Null, _) | (_, Value::Null) => Value::Null, + (Value::Float64(a), Value::Float64(b)) => { + Value::Float64(arithmetic::evaluate_float64_arithmetic(op, a, b)) + } + (Value::Int64(a), Value::Int64(b)) => Value::Int64( + match op { + Add => a.checked_add(b), + Sub => a.checked_sub(b), + Mul => a.checked_mul(b), + Div => a.checked_div(b), + Mod => a.checked_rem(b), + Pow => u32::try_from(b).ok().and_then(|b| a.checked_pow(b)), + Atan2 => None, + } + .ok_or_else(|| invalid("invalid integer arithmetic or overflow"))?, + ), + _ => return Err(invalid("arithmetic type mismatch")), + }) +} + +fn invalid(message: &str) -> Error { + Error::Invalid(message.into()) +} diff --git a/crates/asap-physical-operators/src/dag/expressions.rs b/crates/asap-physical-operators/src/expressions/planner.rs similarity index 99% rename from crates/asap-physical-operators/src/dag/expressions.rs rename to crates/asap-physical-operators/src/expressions/planner.rs index 0a6227cf..39b9df5d 100644 --- a/crates/asap-physical-operators/src/dag/expressions.rs +++ b/crates/asap-physical-operators/src/expressions/planner.rs @@ -1,5 +1,5 @@ //! Planner scalar expressions evaluated over native typed rows. -use super::{ +use crate::{ values::{Schema, Value}, Error, }; @@ -260,7 +260,7 @@ fn arithmetic(op: &ArithmeticOpKind, left: Value, right: Value) -> Result (Value::Float64(a), Value::Float64(b as f64)), pair => pair, }; - super::operators::numeric(op, left, right) + super::numeric(op, left, right) } fn integer_float_cmp(integer: i64, float: f64) -> Option { @@ -350,7 +350,7 @@ impl CompiledExpression { output, }) } - pub(super) fn dtype(&self) -> (DataType, bool) { + pub(crate) fn dtype(&self) -> (DataType, bool) { self.output.clone() } /// Evaluate a row under the same typed schema used when binding the expression. diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs index 3cf29efa..4c644bd7 100644 --- a/crates/asap-physical-operators/src/lib.rs +++ b/crates/asap-physical-operators/src/lib.rs @@ -1,9 +1,11 @@ #![doc = include_str!("../README.md")] +pub mod summary_operators; +/// Compatibility alias for existing deployments. +pub use summary_operators as accumulators; pub mod key_by_label_values; pub mod measurement; -pub mod summary_operators; -pub mod traits; +pub use summary_operators::traits; mod aggregation_type; mod statistic; @@ -13,9 +15,9 @@ pub use measurement::Measurement; pub use statistic::Statistic; pub use traits::*; -pub mod arithmetic; +pub use expressions::arithmetic; pub mod capability; -pub mod factory; +pub use summary_operators::factory; /// The exact Planner contract used by these kernels. pub use planner_types as planner; @@ -23,3 +25,13 @@ pub use planner_types as planner; pub mod dag; pub mod stored_state; + +mod error; +pub use error::Error; +pub mod binding; +pub mod expressions; +pub mod operators; +pub mod plan; +pub mod runtime; +pub mod sources; +pub mod values; diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs new file mode 100644 index 00000000..0270942c --- /dev/null +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -0,0 +1,293 @@ +use super::*; +impl Operator { + pub fn aggregate( + input: Schema, + groups: Vec, + measures: Vec<(String, Reduction)>, + ) -> Result { + validate_groups(&input, &groups)?; + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + for (name, reduction) in &measures { + let (t, n) = match reduction { + Reduction::Count => (DataType::Int64, false), + Reduction::Sum(i) | Reduction::Avg(i) => { + let (t, _) = plain(&input, *i)?; + if !matches!(t, DataType::Int64 | DataType::Float64) { + return Err(invalid("numeric aggregate input required")); + } + ( + if matches!(reduction, Reduction::Avg(_)) { + DataType::Float64 + } else { + t.clone() + }, + false, + ) + } + Reduction::Min(i) | Reduction::Max(i) => { + let (t, nullable) = plain(&input, *i)?; + if !ordered(t) { + return Err(invalid("ordered aggregate input required")); + } + (t.clone(), nullable || groups.is_empty()) + } + }; + fields.push(result_field(name, t, n)); + } + Ok(Self { + kind: Kind::Aggregate { + groups, + measures: measures.into_iter().map(|(_, r)| r).collect(), + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn window( + input: Schema, + intent: planner_types::pre_asap::AggIntent, + coordinate: usize, + value: usize, + groups: Vec, + window: Option<(i64, i64)>, + ) -> Result { + use planner_types::pre_asap::AggIntent; + validate_groups(&input, &groups)?; + let histogram = matches!(intent, AggIntent::HistogramQuantile { .. }); + if !matches!( + intent, + AggIntent::Rate + | AggIntent::Increase + | AggIntent::Count { .. } + | AggIntent::Sum { col: None } + | AggIntent::Avg { col: None } + | AggIntent::Min { col: None } + | AggIntent::Max { col: None } + | AggIntent::HistogramQuantile { .. } + ) { + return Err(invalid( + "unsupported temporal intent or unresolved value column", + )); + } + let coordinate_type = if histogram { + DataType::Float64 + } else { + DataType::Timestamp + }; + if plain(&input, coordinate)? != (&coordinate_type, false) + || plain(&input, value)? != (&DataType::Float64, false) + { + return Err(invalid("window coordinate/value schema mismatch")); + } + if (!histogram && !matches!(window, Some((start, end)) if start < end)) + || (histogram && window.is_some()) + { + return Err(invalid("invalid temporal window")); + } + let mut fields = groups + .iter() + .map(|i| input.fields[*i].clone()) + .collect::>(); + fields.push(result_field( + "value", + if matches!(intent, AggIntent::Count { .. }) { + DataType::Int64 + } else { + DataType::Float64 + }, + false, + )); + Ok(Self { + kind: Kind::Window { + intent: Box::new(intent), + coordinate, + value, + groups, + window, + }, + inputs: vec![input], + output: schema(fields), + }) + } +} +#[derive(Clone, Debug)] +pub enum Reduction { + Count, + Sum(usize), + Avg(usize), + Min(usize), + Max(usize), +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &operator.kind { + Kind::Window { + intent, + coordinate, + value, + groups, + window, + } => { + crate::operators::aggregate::temporal::reduce( + rows, + intent, + groups, + *coordinate, + *value, + *window, + &context, + ) + .await? + } + Kind::Aggregate { groups, measures } => { + reduce(rows, groups, measures, &operator.inputs[0], &context).await? + } + _ => unreachable!(), + }; + Batch::try_new(output, result) + }) + .boxed_local()) +} + +mod temporal; +async fn reduce( + rows: Vec>, + groups: &[usize], + measures: &[Reduction], + input: &Schema, + context: &RunContext, +) -> Result>, Error> { + let mut work = Cooperative::new(context); + let mut workspace = Workspace::new(context)?; + let mut grouped = BTreeMap::>, Vec>>::new(); + if rows.is_empty() && groups.is_empty() { + grouped.insert(vec![], vec![]); + } + for row in rows { + work.checkpoint().await?; + let key = group_key(&row, groups)?; + workspace.grow(std::mem::size_of::>())?; + if !grouped.contains_key(&key) { + workspace.grow(key_bytes(&key))?; + } + grouped.entry(key).or_default().push(row); + } + let mut output = Vec::new(); + for rows in grouped.into_values() { + work.checkpoint().await?; + let mut result = groups + .iter() + .map(|&i| rows[0][i].clone()) + .collect::>(); + for measure in measures { + result.push(reduce_one(&rows, measure, input, &mut work).await?); + } + workspace.grow(row_bytes(&result))?; + output.push(result); + } + Ok(output) +} + +async fn reduce_one( + rows: &[Vec], + measure: &Reduction, + input: &Schema, + work: &mut Cooperative, +) -> Result { + let column = match measure { + Reduction::Count => { + return Ok(Value::Int64( + i64::try_from(rows.len()).map_err(|_| invalid("count overflow"))?, + )) + } + Reduction::Sum(i) | Reduction::Avg(i) | Reduction::Min(i) | Reduction::Max(i) => *i, + }; + let values = rows + .iter() + .map(|r| &r[column]) + .filter(|v| !matches!(v, Value::Null)); + if matches!(measure, Reduction::Min(_) | Reduction::Max(_)) { + if plain(input, column)?.0 == &DataType::Float64 { + // Match exact-state kernels: ignore NaN when a numeric value exists. + let mut best: Option = None; + for value in values { + work.checkpoint().await?; + let Value::Float64(value) = value else { + return Err(invalid("floating aggregate value required")); + }; + best = Some(best.map_or(*value, |old| { + if matches!(measure, Reduction::Min(_)) { + old.min(*value) + } else { + old.max(*value) + } + })); + } + return Ok(best.map(Value::Float64).unwrap_or(Value::Null)); + } + let mut best: Option<&Value> = None; + for value in values { + work.checkpoint().await?; + if best + .map(|b| value.compare(b)) + .transpose()? + .is_none_or(|order| { + if matches!(measure, Reduction::Min(_)) { + order.is_lt() + } else { + order.is_gt() + } + }) + { + best = Some(value); + } + } + return Ok(best.cloned().unwrap_or(Value::Null)); + } + let mut count = 0usize; + let dtype = plain(input, column)?.0; + if dtype == &DataType::Int64 { + let mut sum = 0i128; + for v in values { + work.checkpoint().await?; + let Value::Int64(v) = v else { + return Err(invalid("integer aggregate value required")); + }; + sum = sum + .checked_add(i128::from(*v)) + .ok_or_else(|| invalid("integer aggregate overflow"))?; + count += 1; + } + return if matches!(measure, Reduction::Avg(_)) { + Ok(Value::Float64(sum as f64 / count as f64)) + } else { + Ok(Value::Int64( + i64::try_from(sum).map_err(|_| invalid("integer sum overflow"))?, + )) + }; + } + let mut sum = -0.0; + for v in values { + work.checkpoint().await?; + let Value::Float64(v) = v else { + return Err(invalid("floating aggregate value required")); + }; + sum += v; + count += 1; + } + Ok(Value::Float64(if matches!(measure, Reduction::Avg(_)) { + sum / count as f64 + } else { + sum + })) +} diff --git a/crates/asap-physical-operators/src/dag/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs similarity index 82% rename from crates/asap-physical-operators/src/dag/temporal.rs rename to crates/asap-physical-operators/src/operators/aggregate/temporal.rs index 37f97ada..fb74965a 100644 --- a/crates/asap-physical-operators/src/dag/temporal.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -1,23 +1,38 @@ //! Windowed computations use Planner intents; deployments supply the input window. -use super::{ +use crate::{ + operators::{ + common::{key_bytes, row_bytes, Workspace}, + sort::cooperative_sort, + }, + runtime::{Cooperative, RunContext}, +}; +use crate::{ values::{group_key, Value}, Error, }; use planner_types::pre_asap::{AggIntent, ColumnRef}; use std::collections::BTreeMap; -pub(super) fn reduce( +pub(super) async fn reduce( rows: Vec>, intent: &AggIntent, groups: &[usize], coordinate: usize, value: usize, window: Option<(i64, i64)>, + context: &RunContext, ) -> Result>, Error> { + let mut work = Cooperative::new(context); + let mut workspace = Workspace::new(context)?; let mut grouped = BTreeMap::>, (Vec, Vec<(f64, f64)>, Vec<(i64, f64)>)>::new(); for row in rows { + work.checkpoint().await?; let key = group_key(&row, groups)?; + workspace.grow(32)?; + if !grouped.contains_key(&key) { + workspace.grow(key_bytes(&key) + row_bytes(&row))?; + } let entry = grouped.entry(key).or_insert_with(|| { ( groups.iter().map(|i| row[*i].clone()).collect(), @@ -35,11 +50,12 @@ pub(super) fn reduce( } } let mut output = Vec::new(); - for (_, (mut keys, buckets, mut points)) in grouped { + for (_, (mut keys, buckets, points)) in grouped { + work.checkpoint().await?; let result = if let AggIntent::HistogramQuantile { q } = intent { - Some(Value::Float64(bucket_quantile(*q, buckets))) + Some(Value::Float64(bucket_quantile(*q, buckets, context).await?)) } else { - points.sort_by_key(|p| p.0); + let points = cooperative_sort(points, |a, b| a.0.cmp(&b.0), context).await?; let (start, end) = window.ok_or_else(|| Error::Invalid("missing temporal window".into()))?; if points.iter().any(|p| p.0 < start || p.0 > end) @@ -122,20 +138,27 @@ fn rate(points: &[(i64, f64)], start: i64, end: i64) -> Option { Some(delta * (span + to_start + to_end) / span / ((end as f64 - start as f64) / 1000.)) } -fn bucket_quantile(q: f64, mut b: Vec<(f64, f64)>) -> f64 { +async fn bucket_quantile( + q: f64, + mut b: Vec<(f64, f64)>, + context: &RunContext, +) -> Result { + let mut work = Cooperative::new(context); + let _scratch = context.reserve(b.len().checked_mul(16).ok_or(Error::MemoryLimit)?)?; if q.is_nan() { - return f64::NAN; + return Ok(f64::NAN); } if q < 0. { - return f64::NEG_INFINITY; + return Ok(f64::NEG_INFINITY); } if q > 1. { - return f64::INFINITY; + return Ok(f64::INFINITY); } b.retain(|p| !p.0.is_nan()); - b.sort_by(|a, b| a.0.total_cmp(&b.0)); + b = cooperative_sort(b, |a, b| a.0.total_cmp(&b.0), context).await?; let mut buckets: Vec<(f64, f64)> = Vec::new(); for p in b { + work.checkpoint().await?; if let Some(last) = buckets.last_mut() { if last.0 == p.0 { last.1 += p.1; @@ -145,10 +168,11 @@ fn bucket_quantile(q: f64, mut b: Vec<(f64, f64)>) -> f64 { buckets.push(p); } if buckets.len() < 2 || buckets.last().unwrap().0 != f64::INFINITY { - return f64::NAN; + return Ok(f64::NAN); } let mut prev = buckets[0].1; for p in buckets.iter_mut().skip(1) { + work.checkpoint().await?; if p.1 < prev || (p.1 - prev).abs() <= 1e-12 * (p.1.abs() + prev.abs()) { p.1 = prev; } @@ -156,19 +180,19 @@ fn bucket_quantile(q: f64, mut b: Vec<(f64, f64)>) -> f64 { } let count = buckets.last().unwrap().1; if count == 0. { - return f64::NAN; + return Ok(f64::NAN); } let rank = q * count; let idx = buckets[..buckets.len() - 1].partition_point(|p| p.1 < rank); if idx == buckets.len() - 1 { - return buckets[idx - 1].0; + return Ok(buckets[idx - 1].0); } if idx == 0 && buckets[0].0 <= 0. { - return buckets[0].0; + return Ok(buckets[0].0); } let (start, base) = if idx == 0 { (0., 0.) } else { buckets[idx - 1] }; let (end, upper) = buckets[idx]; - start + (end - start) * (rank - base) / (upper - base) + Ok(start + (end - start) * (rank - base) / (upper - base)) } #[cfg(test)] @@ -254,6 +278,17 @@ mod tests { // Histogram interpolation requires an infinite terminal bucket and coalesces duplicates. #[test] fn histogram_boundaries_and_duplicate_buckets() { + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + let bucket_quantile = |q, buckets| { + futures::executor::block_on(super::bucket_quantile(q, buckets, &context)).unwrap() + }; assert_eq!( bucket_quantile(0.5, vec![(1., 1.), (1., 1.), (2., 4.), (f64::INFINITY, 4.)]), 1. diff --git a/crates/asap-physical-operators/src/operators/common.rs b/crates/asap-physical-operators/src/operators/common.rs new file mode 100644 index 00000000..9f1709da --- /dev/null +++ b/crates/asap-physical-operators/src/operators/common.rs @@ -0,0 +1,75 @@ +use super::*; +pub(super) fn invalid(message: &str) -> Error { + Error::Invalid(message.into()) +} +pub(super) fn schema(fields: Vec) -> Schema { + Arc::new(SummarySchema { + fields, + time_index: None, + }) +} +pub(super) fn result_field(name: &str, dtype: DataType, nullable: bool) -> SummaryField { + SummaryField { + name: name.into(), + dtype: SummaryFamilyType::Plain(dtype), + nullable, + } +} + +pub(super) fn validate_groups(input: &Schema, groups: &[usize]) -> Result<(), Error> { + for &i in groups { + plain(input, i)?; + } + if groups + .iter() + .collect::>() + .len() + != groups.len() + { + return Err(invalid("duplicate group columns")); + } + Ok(()) +} +pub(super) async fn collect_rows( + mut input: Input<'_, Batch>, + context: &RunContext, +) -> Result<(Vec>, Vec), Error> { + let mut rows = Vec::new(); + let mut work = Cooperative::new(context); + let mut reservations = Vec::new(); + while let Some(batch) = input.next().await { + let batch = batch?; + reservations.push(context.reserve(batch.bytes())?); + for row in batch.rows() { + work.checkpoint().await?; + rows.push(row.clone()); + } + } + Ok((rows, reservations)) +} +/// Estimates retained workspace before growing collections. It is not an RSS limit. +pub(super) struct Workspace { + reservation: Reservation, + bytes: usize, +} +impl Workspace { + pub(super) fn new(context: &RunContext) -> Result { + Ok(Self { + reservation: context.reserve(0)?, + bytes: 0, + }) + } + pub(super) fn grow(&mut self, bytes: usize) -> Result<(), Error> { + self.bytes = self.bytes.checked_add(bytes).ok_or(Error::MemoryLimit)?; + self.reservation.resize(self.bytes) + } +} +pub(super) fn row_bytes(row: &[Value]) -> usize { + std::mem::size_of::>() + row.iter().map(Value::bytes).sum::() +} +pub(super) fn key_bytes(key: &[Vec]) -> usize { + 64 + key + .iter() + .map(|part| std::mem::size_of::>() + part.len()) + .sum::() +} diff --git a/crates/asap-physical-operators/src/operators/filter.rs b/crates/asap-physical-operators/src/operators/filter.rs new file mode 100644 index 00000000..8862aacc --- /dev/null +++ b/crates/asap-physical-operators/src/operators/filter.rs @@ -0,0 +1,39 @@ +use super::*; +impl Operator { + pub fn filter(input: Schema, predicate: Expression) -> Result { + if predicate.dtype(&input)?.0 != DataType::Bool { + return Err(invalid("filter predicate must be boolean")); + } + Ok(Self { + kind: Kind::Filter(predicate), + inputs: vec![input.clone()], + output: input, + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::Filter(predicate) => Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + if matches!(predicate.evaluate(row)?, Value::Bool(true)) { + rows.push(row.clone()); + } + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/joins/mod.rs b/crates/asap-physical-operators/src/operators/joins/mod.rs new file mode 100644 index 00000000..6d10bb4f --- /dev/null +++ b/crates/asap-physical-operators/src/operators/joins/mod.rs @@ -0,0 +1,178 @@ +use super::*; +impl Operator { + pub fn semi_join( + left: Schema, + right: Schema, + keys: Vec<(usize, usize)>, + ) -> Result { + if keys.is_empty() { + return Err(invalid("semi-join needs matching keys")); + } + for &(l, r) in &keys { + if plain(&left, l)?.0 != plain(&right, r)?.0 { + return Err(invalid("join key types differ")); + } + } + Ok(Self { + kind: Kind::SemiJoin { keys }, + inputs: vec![left.clone(), right], + output: left, + }) + } + pub fn relational_join( + left: Schema, + right: Schema, + kind: planner_types::pre_asap::JoinKind, + predicate: &planner_types::pre_asap::Predicate, + output: Schema, + ) -> Result { + use planner_types::pre_asap::JoinKind; + let mut joined = left.fields.clone(); + joined.extend(right.fields.clone()); + let predicate = + crate::expressions::CompiledExpression::compile(&predicate.0, &schema(joined.clone()))?; + if predicate.dtype().0 != DataType::Bool { + return Err(invalid("join predicate must be boolean")); + } + let fields = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + left.fields.clone() + } else { + for field in &mut joined[..left.fields.len()] { + if matches!(kind, JoinKind::Right | JoinKind::Full) { + field.nullable = true; + } + } + for field in &mut joined[left.fields.len()..] { + if matches!(kind, JoinKind::Left | JoinKind::Full) { + field.nullable = true; + } + } + joined + }; + Self { + kind: Kind::Join { + kind, + predicate: Box::new(predicate), + }, + inputs: vec![left, right], + output: schema(fields), + } + .with_output_schema(output) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + if let Kind::Join { kind, predicate } = &operator.kind { + let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; + let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; + return Ok(futures::stream::once(async move { + use planner_types::pre_asap::JoinKind; + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + let mut workspace = Workspace::new(&context)?; + let mut work = Cooperative::new(&context); + workspace.grow(right.len())?; + let mut result = Vec::new(); + let mut right_matched = vec![false; right.len()]; + for left_row in &left { + work.checkpoint().await?; + let mut matched = false; + for (i, right_row) in right.iter().enumerate() { + work.checkpoint().await?; + let mut joined = left_row.clone(); + joined.extend(right_row.iter().cloned()); + if *kind == JoinKind::Cross + || matches!(predicate.evaluate(&joined)?, Value::Bool(true)) + { + matched = true; + right_matched[i] = true; + match kind { + JoinKind::Semi => { + workspace.grow(row_bytes(left_row))?; + result.push(left_row.clone()); + break; + } + JoinKind::Anti => break, + _ => { + workspace.grow(row_bytes(&joined))?; + result.push(joined); + } + } + } + } + if !matched { + match kind { + JoinKind::Left | JoinKind::Full => { + let mut joined = left_row.clone(); + joined.resize( + joined.len() + operator.inputs[1].fields.len(), + Value::Null, + ); + workspace.grow(row_bytes(&joined))?; + result.push(joined); + } + JoinKind::Anti => { + workspace.grow(row_bytes(left_row))?; + result.push(left_row.clone()); + } + _ => {} + } + } + } + if matches!(kind, JoinKind::Right | JoinKind::Full) { + for (matched, row) in right_matched.into_iter().zip(right) { + work.checkpoint().await?; + if !matched { + let mut joined = vec![Value::Null; operator.inputs[0].fields.len()]; + joined.extend(row); + workspace.grow(row_bytes(&joined))?; + result.push(joined); + } + } + } + Batch::try_new(output, result) + }) + .boxed_local()); + } + if let Kind::SemiJoin { keys } = &operator.kind { + let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; + let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; + return Ok(futures::stream::once(async move { + // Poll both branches together: either may depend on a common producer. + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + let right_cols = keys.iter().map(|(_, r)| *r).collect::>(); + let left_cols = keys.iter().map(|(l, _)| *l).collect::>(); + let mut members = std::collections::BTreeSet::new(); + let mut workspace = Workspace::new(&context)?; + let mut work = Cooperative::new(&context); + for row in &right { + work.checkpoint().await?; + if right_cols.iter().all(|&i| !matches!(row[i], Value::Null)) { + let key = group_key(row, &right_cols)?; + if !members.contains(&key) { + workspace.grow(key_bytes(&key))?; + members.insert(key); + } + } + } + let mut rows = Vec::new(); + for row in left { + work.checkpoint().await?; + if left_cols.iter().all(|&i| !matches!(row[i], Value::Null)) + && members.contains(&group_key(&row, &left_cols)?) + { + workspace.grow(std::mem::size_of::>())?; + rows.push(row); + } + } + Batch::try_new(output, rows) + }) + .boxed_local()); + } + unreachable!() +} diff --git a/crates/asap-physical-operators/src/operators/limit.rs b/crates/asap-physical-operators/src/operators/limit.rs new file mode 100644 index 00000000..5f5e601f --- /dev/null +++ b/crates/asap-physical-operators/src/operators/limit.rs @@ -0,0 +1,69 @@ +use super::*; +impl Operator { + pub fn limit(input: Schema, n: u64, offset: u64, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + Ok(Self { + kind: Kind::Limit { n, offset, groups }, + inputs: vec![input.clone()], + output: input, + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::Limit { n, offset, groups } => { + let counts = BTreeMap::>, u64>::new(); + Ok(futures::stream::try_unfold( + (input, counts, Vec::::new(), false), + move |(mut input, mut counts, mut memory, done)| { + let output = output.clone(); + let context = context.clone(); + async move { + if done || *n == 0 { + return Ok(None); + } + let Some(batch) = input.next().await else { + return Ok(None); + }; + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + let key = group_key(row, groups)?; + if !counts.contains_key(&key) { + memory.push( + context.reserve( + key.iter() + .map(|part| part.len() + std::mem::size_of::>()) + .sum::() + + 64, + )?, + ); + } + let count = counts.entry(key).or_default(); + if *count >= *offset && count.saturating_sub(*offset) < *n { + rows.push(row.clone()); + } + *count = count.saturating_add(1); + } + let done = groups.is_empty() + && counts + .get(&vec![]) + .is_some_and(|count| count.saturating_sub(*offset) >= *n); + Ok(Some(( + Batch::try_new(output, rows)?, + (input, counts, memory, done), + ))) + } + }, + ) + .boxed_local()) + } + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs new file mode 100644 index 00000000..4fee793e --- /dev/null +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -0,0 +1,207 @@ +//! Native physical operators. Each module owns its constructors and execution. +use crate::plan::{Boundedness, Emission, PhysicalOperator, PlanProperties}; +use crate::{ + runtime::{Cooperative, Input, OutputStream, Reservation, RunContext}, + values::{field, group_key, plain, Batch, Schema, Value}, + Error, +}; +use futures::StreamExt; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema, SummaryUpdate}, + pre_asap::{ColumnRef, DataType}, +}; +use std::{collections::BTreeMap, sync::Arc}; +pub(crate) mod common; +use crate::expressions::ordered; +pub use crate::expressions::Expression; +use common::*; +mod aggregate; +mod filter; +mod joins; +mod limit; +mod projection; +mod sort; +mod source; +mod summary; +pub use aggregate::Reduction; +pub use sort::SortKey; +#[derive(Clone)] +enum Kind { + Source(Vec), + Union, + VectorToScalar { + column: usize, + }, + Project(Vec), + Filter(Expression), + Limit { + n: u64, + offset: u64, + groups: Vec, + }, + Sort { + keys: Vec, + groups: Vec, + }, + Window { + intent: Box>, + coordinate: usize, + value: usize, + groups: Vec, + window: Option<(i64, i64)>, + }, + Aggregate { + groups: Vec, + measures: Vec, + }, + SemiJoin { + keys: Vec<(usize, usize)>, + }, + Join { + kind: planner_types::pre_asap::JoinKind, + predicate: Box, + }, + SummaryBuild { + family: SummaryFamilyType, + value: usize, + time: Option, + groups: Vec, + }, + KeyedSummaryBuild { + family: SummaryFamilyType, + value: usize, + items: Vec, + groups: Vec, + }, + KeyedReadout { + state: usize, + k: usize, + }, + SummaryMerge { + state: usize, + groups: Vec, + }, + Readout { + state: usize, + statistic: crate::Statistic, + parameters: std::collections::HashMap, + }, +} +/// A bound operation has a fully checked input/output contract before execution. +#[derive(Clone)] +pub struct Operator { + kind: Kind, + inputs: Vec, + output: Schema, +} +impl Operator { + pub(crate) fn with_output_schema(mut self, output: Schema) -> Result { + if self.output.fields.len() != output.fields.len() + || self + .output + .fields + .iter() + .zip(&output.fields) + .any(|(actual, declared)| { + actual.dtype != declared.dtype || (actual.nullable && !declared.nullable) + }) + { + return Err(invalid("native output type differs from Planner output")); + } + if output.time_index.is_some_and(|i| { + i >= output.fields.len() + || output.fields[i].dtype != SummaryFamilyType::Plain(DataType::Timestamp) + }) { + return Err(invalid("invalid output time column")); + } + self.output = output; + Ok(self) + } + pub fn schema(&self) -> Schema { + self.output.clone() + } +} +impl PhysicalOperator for Operator { + fn requires_bounded_input(&self) -> bool { + matches!( + self.kind, + Kind::Sort { .. } + | Kind::Aggregate { .. } + | Kind::Window { .. } + | Kind::Join { .. } + | Kind::SemiJoin { .. } + | Kind::SummaryBuild { .. } + | Kind::KeyedSummaryBuild { .. } + | Kind::SummaryMerge { .. } + | Kind::VectorToScalar { .. } + ) + } + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + let boundedness = match &self.kind { + Kind::Source(_) => Boundedness::Bounded, + Kind::Limit { groups, .. } if groups.is_empty() => Boundedness::Bounded, + _ => Boundedness::from_inputs(inputs), + }; + PlanProperties { + boundedness, + emission: if self.requires_bounded_input() { + Emission::AfterInput + } else { + Emission::Incremental + }, + } + } + + fn name(&self) -> &str { + match self.kind { + Kind::Source(_) => "Source", + Kind::Union => "Union", + Kind::VectorToScalar { .. } => "VectorToScalar", + Kind::Project(_) => "Project", + Kind::Filter(_) => "Filter", + Kind::Limit { .. } => "Limit", + Kind::Sort { .. } => "Sort", + Kind::Aggregate { .. } => "Aggregate", + Kind::Window { .. } => "WindowAggregate", + Kind::SemiJoin { .. } => "SemiJoin", + Kind::Join { .. } => "RelationalJoin", + Kind::SummaryBuild { .. } | Kind::KeyedSummaryBuild { .. } => "SummaryAgg", + Kind::KeyedReadout { .. } => "SummaryEstimate", + Kind::SummaryMerge { .. } => "SummaryMerge", + Kind::Readout { .. } => "SummaryReadout", + } + } + fn input_schemas(&self) -> Vec { + self.inputs.clone() + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, value: &Batch) -> usize { + value.bytes() + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + match self.kind { + Kind::Source(_) | Kind::Union | Kind::VectorToScalar { .. } => { + source::execute(self, inputs, context) + } + Kind::Project(_) => projection::execute(self, inputs, context), + Kind::Filter(_) => filter::execute(self, inputs, context), + Kind::Limit { .. } => limit::execute(self, inputs, context), + Kind::Sort { .. } => sort::execute(self, inputs, context), + Kind::Window { .. } | Kind::Aggregate { .. } => { + aggregate::execute(self, inputs, context) + } + Kind::Join { .. } | Kind::SemiJoin { .. } => joins::execute(self, inputs, context), + Kind::SummaryMerge { .. } => summary::execute_merge(self, inputs, context), + Kind::SummaryBuild { .. } + | Kind::Readout { .. } + | Kind::KeyedSummaryBuild { .. } + | Kind::KeyedReadout { .. } => summary::execute(self, inputs, context), + } + } +} diff --git a/crates/asap-physical-operators/src/operators/projection.rs b/crates/asap-physical-operators/src/operators/projection.rs new file mode 100644 index 00000000..9337f7ab --- /dev/null +++ b/crates/asap-physical-operators/src/operators/projection.rs @@ -0,0 +1,47 @@ +use super::*; +impl Operator { + pub fn project(input: Schema, columns: Vec<(String, Expression)>) -> Result { + let fields = columns + .iter() + .map(|(name, e)| { + let (t, n) = e.dtype(&input)?; + Ok(result_field(name, t, n)) + }) + .collect::>()?; + Ok(Self { + kind: Kind::Project(columns.into_iter().map(|(_, e)| e).collect()), + inputs: vec![input], + output: schema(fields), + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::Project(expressions) => Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + let rows = batch + .rows() + .iter() + .map(|r| { + expressions + .iter() + .map(|e| e.evaluate(r)) + .collect::, _>>() + }) + .collect::, _>>()?; + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/sort.rs b/crates/asap-physical-operators/src/operators/sort.rs new file mode 100644 index 00000000..be9c1d7d --- /dev/null +++ b/crates/asap-physical-operators/src/operators/sort.rs @@ -0,0 +1,172 @@ +use super::*; +impl Operator { + pub fn sort(input: Schema, keys: Vec, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + for key in &keys { + if !ordered(plain(&input, key.column)?.0) { + return Err(invalid("unsupported sort type")); + } + } + Ok(Self { + kind: Kind::Sort { keys, groups }, + inputs: vec![input.clone()], + output: input, + }) + } +} +#[derive(Clone, Debug)] +pub struct SortKey { + pub column: usize, + pub descending: bool, + pub nulls_first: bool, +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &operator.kind { + Kind::Sort { keys, groups } => { + let mut grouped = BTreeMap::>, Vec>>::new(); + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + for row in rows { + work.checkpoint().await?; + for key in keys { + if matches!(row[key.column], Value::Map(_)) && nested_nan(&row[key.column]) + { + return Err(invalid("NaN in collection sort key")); + } + } + let key = group_key(&row, groups)?; + workspace.grow(std::mem::size_of::>())?; + if !grouped.contains_key(&key) { + workspace.grow(key_bytes(&key))?; + } + grouped.entry(key).or_default().push(row); + } + let mut result = Vec::new(); + for rows in grouped.into_values() { + let rows = + cooperative_sort(rows, |a, b| compare_rows(a, b, keys), &context).await?; + result.extend(rows); + } + result + } + _ => unreachable!(), + }; + Batch::try_new(output, result) + }) + .boxed_local()) +} + +fn compare_rows(a: &[Value], b: &[Value], keys: &[SortKey]) -> std::cmp::Ordering { + use std::cmp::Ordering::*; + for key in keys { + let (a, b) = (&a[key.column], &b[key.column]); + let order = match (a, b) { + (Value::Null, Value::Null) => Equal, + (Value::Null, _) => { + if key.nulls_first { + Less + } else { + Greater + } + } + (_, Value::Null) => { + if key.nulls_first { + Greater + } else { + Less + } + } + (Value::Float64(a), Value::Float64(b)) if a.is_nan() || b.is_nan() => { + match (a.is_nan(), b.is_nan()) { + (true, true) => Equal, + (true, false) => Greater, + _ => Less, + } + } + _ => { + let order = a.compare(b).expect("bound ordered types"); + if key.descending { + order.reverse() + } else { + order + } + } + }; + if order != Equal { + return order; + } + } + Equal +} +fn nested_nan(value: &Value) -> bool { + match value { + Value::Float64(value) => value.is_nan(), + Value::Map(values) => values + .iter() + .any(|(key, value)| nested_nan(key) || nested_nan(value)), + Value::List(values) | Value::Struct(values) => values.iter().any(nested_nan), + _ => false, + } +} +/// Stable in-memory merge sort with bounded synchronous chunks. Scratch storage +/// is reserved before allocation; comparisons yield between merge steps. +pub(super) async fn cooperative_sort( + rows: Vec, + compare: impl Fn(&T, &T) -> std::cmp::Ordering, + context: &RunContext, +) -> Result, Error> { + use std::collections::VecDeque; + let bytes = rows + .len() + .checked_mul(std::mem::size_of::() + std::mem::size_of::>()) + .and_then(|n| n.checked_mul(3)) + .ok_or(Error::MemoryLimit)?; + let _scratch = context.reserve(bytes)?; + let mut work = Cooperative::new(context); + let mut rows = rows.into_iter(); + let mut runs = VecDeque::new(); + loop { + work.checkpoint().await?; + let mut chunk = rows.by_ref().take(256).collect::>(); + if chunk.is_empty() { + break; + } + chunk.sort_by(&compare); + runs.push_back(VecDeque::from(chunk)); + } + // Merge adjacent runs in rounds to preserve ties in original input order. + while runs.len() > 1 { + let mut next = VecDeque::new(); + while let Some(mut left) = runs.pop_front() { + let Some(mut right) = runs.pop_front() else { + next.push_back(left); + break; + }; + let mut merged = VecDeque::with_capacity(left.len() + right.len()); + while !left.is_empty() || !right.is_empty() { + work.checkpoint().await?; + let take_left = match (left.front(), right.front()) { + (Some(a), Some(b)) => !compare(a, b).is_gt(), + (Some(_), None) => true, + _ => false, + }; + merged.push_back(if take_left { + left.pop_front().unwrap() + } else { + right.pop_front().unwrap() + }); + } + next.push_back(merged); + } + runs = next; + } + Ok(runs.pop_front().unwrap_or_default().into()) +} diff --git a/crates/asap-physical-operators/src/operators/source.rs b/crates/asap-physical-operators/src/operators/source.rs new file mode 100644 index 00000000..060a5381 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/source.rs @@ -0,0 +1,88 @@ +use super::*; +impl Operator { + pub fn source(output: Schema, batches: Vec) -> Result { + crate::values::validate_schema(&output)?; + if batches.iter().any(|b| b.schema() != &output) { + return Err(invalid("source schema mismatch")); + } + Ok(Self { + kind: Kind::Source(batches), + inputs: vec![], + output, + }) + } + pub fn scalar(value: Value, dtype: DataType) -> Result { + let schema = schema(vec![result_field( + "value", + dtype, + matches!(value, Value::Null), + )]); + Self::source( + schema.clone(), + vec![Batch::try_new(schema, vec![vec![value]])?], + ) + } + pub fn vector_to_scalar(input: Schema, column: usize) -> Result { + if plain(&input, column)? != (&DataType::Float64, false) { + return Err(invalid("scalar conversion requires non-null Float64")); + } + Ok(Self { + kind: Kind::VectorToScalar { column }, + inputs: vec![input], + output: schema(vec![result_field("value", DataType::Float64, false)]), + }) + } + pub fn union(input: Schema, arity: usize) -> Result { + if arity == 0 { + return Err(invalid("union needs at least one input")); + } + Ok(Self { + kind: Kind::Union, + inputs: vec![input.clone(); arity], + output: input, + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + if let Kind::Source(batches) = &operator.kind { + return Ok(futures::stream::iter(batches.iter().cloned().map(Ok)).boxed_local()); + } + if matches!(operator.kind, Kind::Union) { + return Ok(futures::stream::select_all(inputs) + .map(|batch| batch.map(|batch| batch.value().clone())) + .boxed_local()); + } + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::VectorToScalar { column } => Ok(futures::stream::once(async move { + let mut input = input; + let mut work = Cooperative::new(&context); + let mut value = f64::NAN; + let mut count = 0usize; + while let Some(batch) = input.next().await { + for row in batch?.rows() { + work.checkpoint().await?; + count = count.saturating_add(1); + if let Value::Float64(v) = row[*column] { + value = v; + } + } + } + Batch::try_new( + output, + vec![vec![Value::Float64(if count == 1 { + value + } else { + f64::NAN + })]], + ) + }) + .boxed_local()), + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs new file mode 100644 index 00000000..2024e2af --- /dev/null +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -0,0 +1,538 @@ +use super::*; +impl Operator { + pub fn keyed_summary_build( + input: Schema, + family: SummaryFamilyType, + value: usize, + items: Vec, + groups: Vec, + ) -> Result { + use planner_types::post_asap::{SketchAlgorithm, SketchParams}; + crate::values::validate_family(&family)?; + let SummaryFamilyType::Sketch(kind, _) = &family else { + return Err(invalid("keyed sketch required")); + }; + if kind.algorithm() != &SketchAlgorithm::CmsWithHeap + || !matches!(kind.params(), SketchParams::CmsWithHeap { .. }) + { + return Err(invalid( + "Float64 weighted keyed construction currently supports CMS with heap", + )); + } + validate_groups(&input, &groups)?; + if items.is_empty() || plain(&input, value)? != (&DataType::Float64, false) { + return Err(invalid( + "keyed summary requires identities and non-null Float64 weights", + )); + } + for &item in &items { + if !matches!( + plain(&input, item)?.0, + DataType::Utf8 + | DataType::Int64 + | DataType::Float64 + | DataType::Bool + | DataType::Null + ) { + return Err(invalid("unsupported keyed summary identity type")); + } + } + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }); + Ok(Self { + kind: Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn keyed_readout( + input: Schema, + state: usize, + k: usize, + output: Schema, + ) -> Result { + use planner_types::post_asap::{SketchAlgorithm, SketchParams}; + crate::capability::validate_native_family(&field(&input, state)?.dtype)?; + let SummaryFamilyType::Sketch(kind, _) = &field(&input, state)?.dtype else { + return Err(invalid("keyed readout requires summary state")); + }; + let SketchParams::CmsWithHeap { heap_size, .. } = kind.params() else { + return Err(invalid("unsupported keyed readout family")); + }; + if kind.algorithm() != &SketchAlgorithm::CmsWithHeap + || k > *heap_size as usize + || output.fields.len() <= input.fields.len() + { + return Err(invalid("invalid keyed readout shape or capacity")); + } + if state + 1 != input.fields.len() + || output.fields[..state] != input.fields[..state] + || output.fields.last().unwrap().dtype != SummaryFamilyType::Plain(DataType::Float64) + { + return Err(invalid( + "keyed readout must preserve partitions and return a Float64 score", + )); + } + crate::values::validate_schema(&output)?; + Ok(Self { + kind: Kind::KeyedReadout { state, k }, + inputs: vec![input], + output, + }) + } + + pub fn summary_build( + input: Schema, + family: SummaryFamilyType, + value: usize, + time: Option, + groups: Vec, + ) -> Result { + crate::values::validate_family(&family)?; + validate_groups(&input, &groups)?; + if plain(&input, value)? != (&DataType::Float64, false) { + return Err(invalid("summary numeric update requires non-null Float64")); + } + if let Some(time) = time { + if plain(&input, time)? != (&DataType::Timestamp, false) { + return Err(invalid("summary time column must be a timestamp")); + } + } + if time.is_none() + && matches!( + family, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Rate + | planner_types::post_asap::ExactKind::Increase, + _ + ) + ) + { + return Err(invalid("counter summary requires a timestamp column")); + } + crate::capability::validate_summary_kernel( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .map_err(Error::Invalid)?; + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }); + Ok(Self { + kind: Kind::SummaryBuild { + family, + value, + time, + groups, + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn summary_merge(input: Schema, state: usize, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + crate::values::validate_family(&field(&input, state)?.dtype)?; + if matches!(field(&input, state)?.dtype, SummaryFamilyType::Plain(_)) { + return Err(invalid("summary state required")); + } + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(input.fields[state].clone()); + Ok(Self { + kind: Kind::SummaryMerge { state, groups }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn readout( + input: Schema, + state: usize, + statistic: crate::Statistic, + parameters: std::collections::HashMap, + ) -> Result { + crate::values::validate_family(&field(&input, state)?.dtype)?; + if matches!(field(&input, state)?.dtype, SummaryFamilyType::Plain(_)) { + return Err(invalid("summary state required")); + } + crate::capability::validate_native_readout( + &field(&input, state)?.dtype, + statistic, + ¶meters, + )?; + let mut fields = input.fields.clone(); + let result_type = if matches!( + fields[state].dtype, + SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) + ) { + DataType::Int64 + } else { + DataType::Float64 + }; + fields[state] = result_field("value", result_type, false); + Ok(Self { + kind: Kind::Readout { + state, + statistic, + parameters, + }, + inputs: vec![input], + output: schema(fields), + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::SummaryBuild { + family, + value, + time, + groups, + } => Ok(futures::stream::once(async move { + Batch::try_new( + output, + build_summary(input, family, *value, *time, groups, &context).await?, + ) + }) + .boxed_local()), + Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + } => Ok(futures::stream::once(async move { + Batch::try_new( + output, + build_keyed_summary(input, family, *value, items, groups, &context).await?, + ) + }) + .boxed_local()), + Kind::KeyedReadout { state, k } => Ok(input + .map(move |batch| { + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + let Value::Summary { state: summary, .. } = &row[*state] else { + return Err(invalid("summary value required")); + }; + let summary = summary + .as_any() + .downcast_ref::() + .ok_or_else(|| invalid("weighted CMS typed state required"))?; + for items in summary.rows(*k) { + let mut values = row[..*state].to_vec(); + values.extend(items); + rows.push(values); + } + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + Kind::Readout { + state, + statistic, + parameters, + } => Ok(input + .map(move |batch| { + let batch = batch?; + let mut rows = batch.rows().to_vec(); + for row in &mut rows { + let Value::Summary { state: summary, .. } = &row[*state] else { + return Err(invalid("summary value required")); + }; + row[*state] = if output.fields[*state].dtype + == SummaryFamilyType::Plain(DataType::Int64) + { + let count = summary.aux_stats().count.ok_or_else(|| { + Error::Operator("exact count state lacks an integer count".into()) + })?; + Value::Int64( + i64::try_from(count) + .map_err(|_| Error::Operator("exact count exceeds Int64".into()))?, + ) + } else { + Value::Float64( + summary + .query_statistic(*statistic, &None, parameters) + .map_err(|e| Error::Operator(e.to_string()))?, + ) + }; + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => unreachable!(), + } +} +pub(super) fn execute_merge<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &operator.kind { + Kind::SummaryMerge { state, groups } => { + merge_summary(rows, *state, groups, &context).await? + } + _ => unreachable!(), + }; + Batch::try_new(output, result) + }) + .boxed_local()) +} + +async fn build_summary( + mut input: Input<'_, Batch>, + family: &SummaryFamilyType, + value: usize, + time: Option, + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + type State = ( + Vec, + Box, + Reservation, + usize, + Option, + ); + let create = |labels: Vec, key_bytes: usize| -> Result { + let updater = crate::factory::create_planner_accumulator( + family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .map_err(Error::Operator)?; + let overhead = labels.iter().map(Value::bytes).sum::() + key_bytes + 64; + let memory = context.reserve(updater.memory_usage_bytes() + overhead)?; + Ok((labels, updater, memory, overhead, None)) + }; + let mut work = Cooperative::new(context); + let mut states = BTreeMap::>, State>::new(); + if groups.is_empty() { + states.insert(vec![], create(vec![], 0)?); + } + let ordered_time = matches!( + family, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Rate + | planner_types::post_asap::ExactKind::Increase, + _ + ) + ); + while let Some(batch) = input.next().await { + let batch = batch?; + for row in batch.rows() { + work.checkpoint().await?; + let key = group_key(row, groups)?; + if !states.contains_key(&key) { + let labels = groups.iter().map(|&i| row[i].clone()).collect(); + let state = create( + labels, + key.iter() + .map(|v| v.len() + std::mem::size_of::>()) + .sum(), + )?; + states.insert(key.clone(), state); + } + let (_, updater, memory, overhead, previous) = + states.get_mut(&key).expect("inserted group"); + let Value::Float64(value) = row[value] else { + return Err(invalid("summary update type")); + }; + let timestamp = if let Some(time) = time { + let Value::Timestamp(time) = row[time] else { + return Err(invalid("summary time type")); + }; + time + } else { + 0 + }; + if ordered_time && previous.is_some_and(|prior| timestamp <= prior) { + return Err(Error::Operator( + "counter samples must have strictly increasing timestamps within each group" + .into(), + )); + } + updater + .validate_single_input(value) + .map_err(Error::Operator)?; + updater.update_single(value, timestamp); + *previous = Some(timestamp); + memory.resize(updater.memory_usage_bytes() + *overhead)?; + } + } + Ok(states + .into_values() + .map(|(mut labels, updater, _memory, _, _)| { + labels.push(Value::Summary { + family: family.clone(), + state: Arc::from(updater.into_accumulator()), + }); + labels + }) + .collect()) +} +async fn merge_summary( + rows: Vec>, + state_column: usize, + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + type GroupState = (Vec, SummaryFamilyType, Arc); + let mut states: BTreeMap>, GroupState> = BTreeMap::new(); + let mut work = Cooperative::new(context); + let mut memory = context.reserve(0)?; + let mut retained = 0usize; + for row in rows { + work.checkpoint().await?; + let Value::Summary { family, state } = &row[state_column] else { + return Err(invalid("summary state required")); + }; + let key = group_key(&row, groups)?; + if let Some((_, expected, existing)) = states.get_mut(&key) { + if expected != family { + return Err(invalid("incompatible summary family")); + } + let old_bytes = existing.approx_memory_bytes(); + // Reserve an estimate for the replacement while both input states remain live. + memory.resize( + retained + .checked_add(old_bytes) + .and_then(|n| n.checked_add(state.approx_memory_bytes())) + .ok_or(Error::MemoryLimit)?, + )?; + *existing = Arc::from( + existing + .merge_with(state.as_ref()) + .map_err(|e| Error::Operator(e.to_string()))?, + ); + retained = retained + .checked_sub(old_bytes) + .and_then(|n| n.checked_add(existing.approx_memory_bytes())) + .ok_or(Error::MemoryLimit)?; + memory.resize(retained)?; + } else { + retained = retained + .checked_add(key_bytes(&key) + row_bytes(&row)) + .ok_or(Error::MemoryLimit)?; + memory.resize(retained)?; + states.insert( + key, + ( + groups.iter().map(|&i| row[i].clone()).collect(), + family.clone(), + state.clone(), + ), + ); + } + } + Ok(states + .into_values() + .map(|(mut keys, family, state)| { + keys.push(Value::Summary { family, state }); + keys + }) + .collect()) +} + +async fn build_keyed_summary( + mut input: Input<'_, Batch>, + family: &SummaryFamilyType, + value: usize, + items: &[usize], + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + use crate::{summary_operators::weighted_cms::WeightedCms, AggregateCore}; + use planner_types::post_asap::SketchParams; + let SummaryFamilyType::Sketch(kind, _) = family else { + unreachable!() + }; + let SketchParams::CmsWithHeap { + width, + depth, + heap_size, + } = kind.params() + else { + unreachable!() + }; + let mut work = Cooperative::new(context); + let mut states = BTreeMap::>, (Vec, WeightedCms, Reservation, usize)>::new(); + while let Some(batch) = input.next().await { + let batch = batch?; + for row in batch.rows() { + work.checkpoint().await?; + let key = group_key(row, groups)?; + if !states.contains_key(&key) { + let labels = groups.iter().map(|&i| row[i].clone()).collect::>(); + let overhead = labels.iter().map(Value::bytes).sum::() + + key.iter().map(|v| v.len() + 24).sum::() + + 128; + let bytes = (*width as usize) + .checked_mul(*depth as usize) + .and_then(|n| n.checked_mul(8)) + .and_then(|n| n.checked_add(overhead)) + .ok_or_else(|| invalid("weighted CMS memory size overflow"))?; + let reservation = context.reserve(bytes)?; + states.insert( + key.clone(), + ( + labels, + WeightedCms::new(*width as usize, *depth as usize, *heap_size as usize)?, + reservation, + overhead, + ), + ); + } + let (_, summary, reservation, overhead) = states.get_mut(&key).unwrap(); + let Value::Float64(weight) = row[value] else { + return Err(invalid("weighted CMS weight type")); + }; + summary.update( + &items.iter().map(|&i| row[i].clone()).collect::>(), + weight, + )?; + reservation.resize(summary.approx_memory_bytes() + *overhead)?; + } + } + Ok(states + .into_values() + .map(|(mut labels, summary, _, _)| { + labels.push(Value::Summary { + family: family.clone(), + state: Arc::new(summary), + }); + labels + }) + .collect()) +} diff --git a/crates/asap-physical-operators/src/plan/mod.rs b/crates/asap-physical-operators/src/plan/mod.rs new file mode 100644 index 00000000..ffc6ebd9 --- /dev/null +++ b/crates/asap-physical-operators/src/plan/mod.rs @@ -0,0 +1,156 @@ +//! Immutable physical graph, operator contracts and pre-execution validation. +use crate::{ + runtime::{Input, OutputStream, RunContext}, + Error, +}; +use std::{ + collections::{BTreeMap, BTreeSet}, + fmt::Debug, +}; +pub type NodeId = u64; +mod properties; +pub use properties::{Boundedness, Emission, PlanProperties}; +/// Operators own computation. The runtime provides already-connected inputs; +/// an operator must not recursively execute another plan node itself. +pub trait PhysicalOperator { + fn name(&self) -> &str; + /// Source implementations must explicitly declare finite input before feeding blocking operators. + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + PlanProperties { + boundedness: Boundedness::from_inputs(inputs), + emission: Emission::Unknown, + } + } + fn requires_bounded_input(&self) -> bool { + false + } + + fn input_schemas(&self) -> Vec; + fn output_schema(&self) -> S; + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error>; + fn output_bytes(&self, value: &V) -> usize; +} +pub(crate) struct Node<'a, V, S> { + pub(crate) inputs: Vec, + pub(crate) operator: Box + 'a>, +} +pub struct PhysicalDag<'a, V, S> { + pub(crate) nodes: BTreeMap>, +} +impl Default for PhysicalDag<'_, V, S> { + fn default() -> Self { + Self { + nodes: BTreeMap::new(), + } + } +} +impl<'a, V: 'a, S: Clone + PartialEq + Debug + 'a> PhysicalDag<'a, V, S> { + pub fn add( + &mut self, + id: NodeId, + inputs: Vec, + operator: impl PhysicalOperator + 'a, + ) -> Result<(), Error> { + self.add_boxed(id, inputs, Box::new(operator)) + } + pub fn add_boxed( + &mut self, + id: NodeId, + inputs: Vec, + operator: Box + 'a>, + ) -> Result<(), Error> { + if self.nodes.contains_key(&id) { + return Err(Error::Invalid(format!("duplicate node {id}"))); + } + self.nodes.insert(id, Node { inputs, operator }); + Ok(()) + } + pub fn validate(&self, roots: &[NodeId]) -> Result<(), Error> { + self.properties(roots).map(|_| ()) + } + /// Derive properties while checking topology and schemas, before starting sources. + pub fn properties(&self, roots: &[NodeId]) -> Result, Error> { + fn visit( + dag: &PhysicalDag<'_, V, S>, + id: NodeId, + active: &mut BTreeSet, + done: &mut BTreeMap, + ) -> Result { + if let Some((depth, _)) = done.get(&id) { + return Ok(*depth); + } + if active.len() >= 128 { + return Err(Error::Invalid( + "DAG exceeds the supported execution depth of 128".into(), + )); + } + if !active.insert(id) { + return Err(Error::Invalid(format!("cycle at node {id}"))); + } + let node = dag + .nodes + .get(&id) + .ok_or_else(|| Error::Invalid(format!("missing node {id}")))?; + let expected = node.operator.input_schemas(); + if expected.len() != node.inputs.len() { + return Err(Error::Invalid(format!("node {id} input arity mismatch"))); + } + let mut depth = 1; + let mut input_properties = Vec::new(); + for (input, schema) in node.inputs.iter().zip(expected) { + depth = depth.max(1 + visit(dag, *input, active, done)?); + input_properties.push(done[input].1); + let actual = dag.nodes[input].operator.output_schema(); + if actual != schema { + return Err(Error::Invalid(format!( + "node {id} input {input} schema mismatch: {actual:?} vs {schema:?}" + ))); + } + } + if depth > 128 { + return Err(Error::Invalid( + "DAG exceeds the supported execution depth of 128".into(), + )); + } + if node.operator.requires_bounded_input() + && input_properties + .iter() + .any(|p| p.boundedness != Boundedness::Bounded) + { + return Err(Error::Invalid(format!( + "node {id} ({}) requires bounded inputs", + node.operator.name() + ))); + } + let properties = node.operator.properties(&input_properties); + active.remove(&id); + done.insert(id, (depth, properties)); + Ok(depth) + } + if roots.is_empty() { + return Err(Error::Invalid("execution needs a root".into())); + } + let mut done = BTreeMap::new(); + for &root in roots { + visit(self, root, &mut BTreeSet::new(), &mut done)?; + } + Ok(done + .into_iter() + .map(|(id, (_, properties))| (id, properties)) + .collect()) + } + pub fn execute<'r>( + &'r self, + roots: &[NodeId], + context: RunContext, + ) -> Result>, Error> + where + 'a: 'r, + { + crate::runtime::execute(self, roots, context) + } +} diff --git a/crates/asap-physical-operators/src/plan/properties.rs b/crates/asap-physical-operators/src/plan/properties.rs new file mode 100644 index 00000000..de1c7474 --- /dev/null +++ b/crates/asap-physical-operators/src/plan/properties.rs @@ -0,0 +1,32 @@ +//! Execution facts used to reject operators that cannot finish on their inputs. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Boundedness { + /// The source or operator promises a finite result for this run. + Bounded, + Unbounded, + /// No finite-input guarantee has been supplied. + Unknown, +} +impl Boundedness { + pub fn from_inputs(inputs: &[PlanProperties]) -> Self { + if inputs.iter().any(|p| p.boundedness == Self::Unbounded) { + Self::Unbounded + } else if inputs.is_empty() || inputs.iter().any(|p| p.boundedness == Self::Unknown) { + Self::Unknown + } else { + Self::Bounded + } + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Emission { + Incremental, + /// Produces its result only after all inputs end, even if accumulation is incremental. + AfterInput, + Unknown, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct PlanProperties { + pub boundedness: Boundedness, + pub emission: Emission, +} diff --git a/crates/asap-physical-operators/src/dag/batch_execution.rs b/crates/asap-physical-operators/src/runtime/batch_execution.rs similarity index 96% rename from crates/asap-physical-operators/src/dag/batch_execution.rs rename to crates/asap-physical-operators/src/runtime/batch_execution.rs index a9c20abc..86809f2b 100644 --- a/crates/asap-physical-operators/src/dag/batch_execution.rs +++ b/crates/asap-physical-operators/src/runtime/batch_execution.rs @@ -1,6 +1,12 @@ //! Execute a bounded in-memory batch through native operators. This is also the //! bridge for deployments whose boundary values are not yet streaming batches. -use super::{operators::Operator, values::Batch, Error, PhysicalDag, RunContext, SharedValue}; +use crate::{ + operators::Operator, + plan::PhysicalDag, + runtime::{RunContext, SharedValue}, + values::Batch, + Error, +}; use futures::{FutureExt, StreamExt}; /// Every input is already in memory; the chain contains native operators only. @@ -55,8 +61,8 @@ pub fn evaluate_source( } fn evaluate_graph( - graph: PhysicalDag<'_, Batch, super::values::Schema>, - root: super::NodeId, + graph: PhysicalDag<'_, Batch, crate::values::Schema>, + root: crate::plan::NodeId, context: RunContext, ) -> Result>, Error> { let mut output = graph.execute(&[root], context)?.remove(0); diff --git a/crates/asap-physical-operators/src/runtime/context.rs b/crates/asap-physical-operators/src/runtime/context.rs new file mode 100644 index 00000000..c145a457 --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/context.rs @@ -0,0 +1,133 @@ +use crate::Error; +use std::{ + cell::{Cell, RefCell}, + rc::Rc, + task::Waker, +}; +/// Scope is part of an execution instance, never mutable state in a reusable plan. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Scope { + Ingestion { + window_start_ms: i64, + window_end_ms: i64, + revision: u64, + }, + Query { + evaluation_time_ms: i64, + revision: u64, + }, +} +#[derive(Clone, Debug)] +pub struct Limits { + pub max_buffered_batches: usize, + pub max_bytes: usize, +} +impl Default for Limits { + fn default() -> Self { + Self { + max_buffered_batches: 8, + max_bytes: 64 * 1024 * 1024, + } + } +} +pub(super) struct Control { + cancelled: Cell, + bytes: Cell, + peak: Cell, + pub(super) limits: Limits, + waiters: RefCell>, +} +#[derive(Clone)] +pub struct RunContext { + pub scope: Scope, + pub(super) control: Rc, +} +impl RunContext { + pub fn new(scope: Scope, limits: Limits) -> Result { + if limits.max_buffered_batches == 0 || limits.max_bytes == 0 { + return Err(Error::Invalid("execution limits must be positive".into())); + } + if matches!(&scope, Scope::Ingestion { window_start_ms, window_end_ms, .. } if window_start_ms > window_end_ms) + { + return Err(Error::Invalid("inverted ingestion window".into())); + } + Ok(Self { + scope, + control: Rc::new(Control { + cancelled: Cell::new(false), + bytes: Cell::new(0), + peak: Cell::new(0), + limits, + waiters: RefCell::new(Vec::new()), + }), + }) + } + pub fn cancel(&self) { + self.control.cancelled.set(true); + for waiter in self.control.waiters.borrow_mut().drain(..) { + waiter.wake(); + } + } + pub fn is_cancelled(&self) -> bool { + self.control.cancelled.get() + } + pub fn retained_bytes(&self) -> usize { + self.control.bytes.get() + } + pub fn peak_bytes(&self) -> usize { + self.control.peak.get() + } + pub fn reserve(&self, bytes: usize) -> Result { + let total = self + .control + .bytes + .get() + .checked_add(bytes) + .ok_or(Error::MemoryLimit)?; + if total > self.control.limits.max_bytes { + return Err(Error::MemoryLimit); + } + self.control.bytes.set(total); + self.control.peak.set(self.control.peak.get().max(total)); + Ok(Reservation { + bytes, + control: Rc::clone(&self.control), + }) + } + pub(super) fn register(&self, waker: &Waker) { + let mut waiters = self.control.waiters.borrow_mut(); + if !waiters.iter().any(|old| old.will_wake(waker)) { + waiters.push(waker.clone()); + } + } +} +pub struct Reservation { + bytes: usize, + pub(super) control: Rc, +} +impl Reservation { + /// Adjust an operator-owned allocation without accumulating bookkeeping entries. + pub fn resize(&mut self, bytes: usize) -> Result<(), Error> { + let total = self + .control + .bytes + .get() + .checked_sub(self.bytes) + .and_then(|total| total.checked_add(bytes)) + .ok_or(Error::MemoryLimit)?; + if total > self.control.limits.max_bytes { + return Err(Error::MemoryLimit); + } + self.control.bytes.set(total); + self.control.peak.set(self.control.peak.get().max(total)); + self.bytes = bytes; + Ok(()) + } +} +impl Drop for Reservation { + fn drop(&mut self) { + self.control + .bytes + .set(self.control.bytes.get().saturating_sub(self.bytes)); + } +} diff --git a/crates/asap-physical-operators/src/runtime/cooperative.rs b/crates/asap-physical-operators/src/runtime/cooperative.rs new file mode 100644 index 00000000..fae869d6 --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/cooperative.rs @@ -0,0 +1,40 @@ +//! Worker-local CPU loops yield so other consumers and cancellation can progress. +use super::RunContext; +use crate::Error; +use std::task::Poll; + +pub(crate) struct Cooperative { + context: RunContext, + remaining: usize, +} +impl Cooperative { + pub(crate) fn new(context: &RunContext) -> Self { + Self { + context: context.clone(), + remaining: 1024, + } + } + pub(crate) async fn checkpoint(&mut self) -> Result<(), Error> { + if self.context.is_cancelled() { + return Err(Error::Cancelled); + } + self.remaining -= 1; + if self.remaining == 0 { + self.remaining = 1024; + let mut yielded = false; + futures::future::poll_fn(|cx| { + if self.context.is_cancelled() { + return Poll::Ready(Err(Error::Cancelled)); + } + if yielded { + return Poll::Ready(Ok(())); + } + yielded = true; + cx.waker().wake_by_ref(); + Poll::Pending + }) + .await?; + } + Ok(()) + } +} diff --git a/crates/asap-physical-operators/src/runtime/mod.rs b/crates/asap-physical-operators/src/runtime/mod.rs new file mode 100644 index 00000000..cd0721e8 --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/mod.rs @@ -0,0 +1,263 @@ +//! Per-run producer sharing, streams, backpressure and resource ownership. +use crate::{ + plan::{NodeId, PhysicalDag}, + Error, +}; +use futures::{stream::LocalBoxStream, Stream}; +use std::{ + cell::RefCell, + collections::{BTreeMap, VecDeque}, + fmt::Debug, + pin::Pin, + rc::Rc, + sync::Arc, + task::{Context, Poll, Waker}, +}; +mod context; +pub use context::{Limits, Reservation, RunContext, Scope}; +pub type OutputStream<'a, V> = LocalBoxStream<'a, Result>; +/// An output owns its memory reservation even after it leaves the DAG's queue. +pub struct SharedValue { + value: Arc, + _reservation: Rc, +} +impl Clone for SharedValue { + fn clone(&self) -> Self { + Self { + value: Arc::clone(&self.value), + _reservation: Rc::clone(&self._reservation), + } + } +} +impl std::ops::Deref for SharedValue { + type Target = V; + fn deref(&self) -> &V { + &self.value + } +} +impl SharedValue { + pub fn value(&self) -> &V { + &self.value + } +} + +pub(crate) fn execute<'r, V: 'r, S: Clone + PartialEq + Debug + 'r>( + dag: &'r PhysicalDag<'_, V, S>, + roots: &[NodeId], + context: RunContext, +) -> Result>, Error> { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + dag.validate(roots)?; + fn build<'r, V: 'r, S: 'r>( + dag: &'r PhysicalDag<'_, V, S>, + id: NodeId, + context: &RunContext, + states: &mut BTreeMap>>>, + ) -> Result>>, Error> { + if let Some(state) = states.get(&id) { + return Ok(Rc::clone(state)); + } + let node = &dag.nodes[&id]; + let mut inputs = Vec::new(); + for &child in &node.inputs { + inputs.push(Input::subscribe(build(dag, child, context, states)?)); + } + let stream = node + .operator + .start(inputs, context.clone()) + .map_err(|source| Error::AtNode { + node: id, + operation: node.operator.name().into(), + source: Box::new(source), + })?; + let op = node.operator.as_ref(); + let state = Rc::new(RefCell::new(Producer { + stream: Some(stream), + node: id, + operation: node.operator.name().into(), + size: Box::new(move |value| op.output_bytes(value)), + context: context.clone(), + queue: VecDeque::new(), + base: 0, + next_reader: 0, + batches_polled: 0, + readers: BTreeMap::new(), + waiters: BTreeMap::new(), + finished: false, + failure: None, + })); + states.insert(id, Rc::clone(&state)); + Ok(state) + } + let mut states = BTreeMap::new(); + roots + .iter() + .map(|&id| build(dag, id, &context, &mut states).map(Input::subscribe)) + .collect() +} +struct Producer<'a, V> { + node: NodeId, + operation: String, + stream: Option>, + size: Box usize + 'a>, + context: RunContext, + queue: VecDeque>, + base: u64, + next_reader: u64, + batches_polled: usize, + readers: BTreeMap, + waiters: BTreeMap, + finished: bool, + failure: Option, +} +impl Producer<'_, V> { + fn trim(&mut self) { + let minimum = self + .readers + .values() + .copied() + .min() + .unwrap_or(self.base + self.queue.len() as u64); + while self.base < minimum { + self.queue.pop_front(); + self.base += 1; + } + for (_, waker) in std::mem::take(&mut self.waiters) { + waker.wake(); + } + if self.readers.is_empty() { + self.stream = None; + self.queue.clear(); + } + } +} +pub struct Input<'a, V> { + producer: Rc>>, + reader: u64, + done: bool, +} +impl<'a, V> Input<'a, V> { + fn subscribe(producer: Rc>>) -> Self { + let reader = { + let mut state = producer.borrow_mut(); + let id = state.next_reader; + state.next_reader += 1; + let base = state.base; + state.readers.insert(id, base); + id + }; + Self { + producer, + reader, + done: false, + } + } +} +impl Drop for Input<'_, V> { + fn drop(&mut self) { + let mut state = self.producer.borrow_mut(); + state.readers.remove(&self.reader); + state.waiters.remove(&self.reader); + state.trim(); + } +} +impl Stream for Input<'_, V> { + type Item = Result, Error>; + fn poll_next(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + let this = self.get_mut(); + if this.done { + return Poll::Ready(None); + } + let mut state = this.producer.borrow_mut(); + state.context.register(cx.waker()); + if state.context.is_cancelled() { + state.failure = Some(Error::Cancelled); + state.finished = true; + state.stream = None; + state.queue.clear(); + } + let position = state.readers[&this.reader]; + let index = (position - state.base) as usize; + if let Some(value) = state.queue.get(index).cloned() { + state.readers.insert(this.reader, position + 1); + state.trim(); + return Poll::Ready(Some(Ok(value))); + } + if state.finished { + this.done = true; + state.readers.remove(&this.reader); + let failure = state.failure.clone(); + state.trim(); + return Poll::Ready(failure.map(Err)); + } + state.waiters.insert(this.reader, cx.waker().clone()); + if state.queue.len() >= state.context.control.limits.max_buffered_batches { + return Poll::Pending; + } + // Always-ready sources must still give cancellation and other roots a turn. + if state.batches_polled >= 32 { + state.batches_polled = 0; + cx.waker().wake_by_ref(); + return Poll::Pending; + } + let polled = state + .stream + .as_mut() + .expect("unfinished producer") + .as_mut() + .poll_next(cx); + if matches!(&polled, Poll::Ready(Some(Ok(_)))) { + state.batches_polled += 1; + } + match polled { + Poll::Pending => Poll::Pending, + Poll::Ready(Some(Ok(value))) => match state.context.reserve((state.size)(&value)) { + Ok(reservation) => { + let value = SharedValue { + value: Arc::new(value), + _reservation: Rc::new(reservation), + }; + state.queue.push_back(value.clone()); + state.readers.insert(this.reader, position + 1); + state.trim(); + Poll::Ready(Some(Ok(value))) + } + Err(error) => { + state.failure = Some(error.clone()); + state.finished = true; + state.stream = None; + this.done = true; + state.readers.remove(&this.reader); + state.trim(); + Poll::Ready(Some(Err(error))) + } + }, + Poll::Ready(result) => { + let error = result.and_then(Result::err).map(|source| match source { + Error::AtNode { .. } | Error::Cancelled | Error::MemoryLimit => source, + source => Error::AtNode { + node: state.node, + operation: state.operation.clone(), + source: Box::new(source), + }, + }); + state.failure = error.clone(); + state.finished = true; + state.stream = None; + this.done = true; + state.readers.remove(&this.reader); + state.trim(); + Poll::Ready(error.map(Err)) + } + } + } +} + +pub mod batch_execution; +#[cfg(test)] +mod tests; + +mod cooperative; +pub(crate) use cooperative::Cooperative; diff --git a/crates/asap-physical-operators/src/dag/tests.rs b/crates/asap-physical-operators/src/runtime/tests.rs similarity index 99% rename from crates/asap-physical-operators/src/dag/tests.rs rename to crates/asap-physical-operators/src/runtime/tests.rs index e051fa31..f683041c 100644 --- a/crates/asap-physical-operators/src/dag/tests.rs +++ b/crates/asap-physical-operators/src/runtime/tests.rs @@ -1,5 +1,7 @@ use super::*; +use crate::plan::PhysicalOperator; use futures::{executor::block_on, stream, StreamExt}; +use std::cell::Cell; struct Source { starts: Rc>, diff --git a/crates/asap-physical-operators/src/sources/memory.rs b/crates/asap-physical-operators/src/sources/memory.rs new file mode 100644 index 00000000..7f77c54f --- /dev/null +++ b/crates/asap-physical-operators/src/sources/memory.rs @@ -0,0 +1,44 @@ +use super::*; +/// Immutable in-memory raw data. The connector owns the resident input; each +/// cursor clones only the next requested batch, not the entire data set. +pub struct MemorySource { + schema: Schema, + batches: Vec, +} +impl MemorySource { + pub fn new(schema: Schema, batches: Vec) -> Result { + crate::values::validate_schema(&schema)?; + if schema + .fields + .iter() + .any(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + { + return Err(Error::Invalid( + "raw source cannot contain summary states".into(), + )); + } + if batches.iter().any(|batch| batch.schema() != &schema) { + return Err(Error::Invalid("memory source batch schema mismatch".into())); + } + Ok(Self { schema, batches }) + } +} +impl RawSource for MemorySource { + fn boundedness(&self) -> crate::plan::Boundedness { + crate::plan::Boundedness::Bounded + } + fn schema(&self) -> Schema { + self.schema.clone() + } + fn scan(&self, context: RunContext) -> Result, Error> { + Ok(stream::iter(self.batches.iter()) + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let _allocation = context.reserve(batch.bytes())?; + Ok(batch.clone()) + }) + .boxed_local()) + } +} diff --git a/crates/asap-physical-operators/src/dag/scan.rs b/crates/asap-physical-operators/src/sources/mod.rs similarity index 79% rename from crates/asap-physical-operators/src/dag/scan.rs rename to crates/asap-physical-operators/src/sources/mod.rs index 4c225389..4da778b0 100644 --- a/crates/asap-physical-operators/src/dag/scan.rs +++ b/crates/asap-physical-operators/src/sources/mod.rs @@ -1,8 +1,10 @@ //! Raw data access. Connectors provide rows; Scan owns Planner predicate semantics. -use super::{ +use crate::{ expressions::CompiledExpression, + plan::PhysicalOperator, + runtime::{Input, OutputStream, RunContext}, values::{Batch, Schema, Value}, - Error, Input, OutputStream, PhysicalOperator, RunContext, + Error, }; use futures::{stream, StreamExt}; use planner_types::{ @@ -17,6 +19,10 @@ use std::sync::Arc; /// must release its resources. A connector error is never an empty successful scan. pub trait RawSource { fn schema(&self) -> Schema; + /// Declare a finite snapshot/window explicitly; execution scope alone does not bound a cursor. + fn boundedness(&self) -> crate::plan::Boundedness { + crate::plan::Boundedness::Unknown + } fn scan(&self, context: RunContext) -> Result, Error>; } @@ -30,7 +36,7 @@ impl DataSources { if self.sources.iter().any(|(key, _)| key == &identity) { return Err(Error::Invalid("duplicate data source".into())); } - super::values::validate_schema(&source.schema())?; + crate::values::validate_schema(&source.schema())?; self.sources.push((identity, source)); Ok(()) } @@ -57,7 +63,7 @@ impl DataSources { .collect(), time_index: schema.time_index, }); - super::values::validate_schema(&output)?; + crate::values::validate_schema(&output)?; let reader = self .sources .iter() @@ -93,6 +99,13 @@ pub struct Scan { predicates: Vec, } impl PhysicalOperator for Scan { + fn properties(&self, _: &[crate::plan::PlanProperties]) -> crate::plan::PlanProperties { + crate::plan::PlanProperties { + boundedness: self.reader.boundedness(), + emission: crate::plan::Emission::Incremental, + } + } + fn name(&self) -> &str { "Scan" } @@ -170,43 +183,5 @@ impl PhysicalOperator for Scan { } } -/// Immutable in-memory raw data. The connector owns the resident input; each -/// cursor clones only the next requested batch, not the entire data set. -pub struct MemorySource { - schema: Schema, - batches: Vec, -} -impl MemorySource { - pub fn new(schema: Schema, batches: Vec) -> Result { - super::values::validate_schema(&schema)?; - if schema - .fields - .iter() - .any(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) - { - return Err(Error::Invalid( - "raw source cannot contain summary states".into(), - )); - } - if batches.iter().any(|batch| batch.schema() != &schema) { - return Err(Error::Invalid("memory source batch schema mismatch".into())); - } - Ok(Self { schema, batches }) - } -} -impl RawSource for MemorySource { - fn schema(&self) -> Schema { - self.schema.clone() - } - fn scan(&self, context: RunContext) -> Result, Error> { - Ok(stream::iter(self.batches.iter()) - .map(move |batch| { - if context.is_cancelled() { - return Err(Error::Cancelled); - } - let _allocation = context.reserve(batch.bytes())?; - Ok(batch.clone()) - }) - .boxed_local()) - } -} +mod memory; +pub use memory::MemorySource; diff --git a/crates/asap-physical-operators/src/factory.rs b/crates/asap-physical-operators/src/summary_operators/factory.rs similarity index 100% rename from crates/asap-physical-operators/src/factory.rs rename to crates/asap-physical-operators/src/summary_operators/factory.rs diff --git a/crates/asap-physical-operators/src/summary_operators/mod.rs b/crates/asap-physical-operators/src/summary_operators/mod.rs index 731d1eb9..904f421f 100644 --- a/crates/asap-physical-operators/src/summary_operators/mod.rs +++ b/crates/asap-physical-operators/src/summary_operators/mod.rs @@ -37,3 +37,6 @@ pub use sketch_envelope_accumulator::*; pub use sum_accumulator::*; pub mod weighted_cms; + +pub mod factory; +pub mod traits; diff --git a/crates/asap-physical-operators/src/traits.rs b/crates/asap-physical-operators/src/summary_operators/traits.rs similarity index 100% rename from crates/asap-physical-operators/src/traits.rs rename to crates/asap-physical-operators/src/summary_operators/traits.rs diff --git a/crates/asap-physical-operators/src/summary_operators/weighted_cms.rs b/crates/asap-physical-operators/src/summary_operators/weighted_cms.rs index ac365b9c..626a13e7 100644 --- a/crates/asap-physical-operators/src/summary_operators/weighted_cms.rs +++ b/crates/asap-physical-operators/src/summary_operators/weighted_cms.rs @@ -1,7 +1,7 @@ //! Float64 weighted CMS state with typed candidate identities. Each instance //! represents one partition at one evaluation scope; updates never round rates //! to integer counts. Candidate membership still requires Planner evidence. -use crate::dag::{values::Value, Error}; +use crate::{values::Value, Error}; use crate::{AggregateCore, AggregationType, KeyByLabelValues, SerializableToSink, Statistic}; use serde::{Deserialize, Serialize}; use std::{ diff --git a/crates/asap-physical-operators/src/dag/values.rs b/crates/asap-physical-operators/src/values.rs similarity index 89% rename from crates/asap-physical-operators/src/dag/values.rs rename to crates/asap-physical-operators/src/values.rs index 9e403084..b3b84b0c 100644 --- a/crates/asap-physical-operators/src/dag/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -1,8 +1,8 @@ //! Runtime values preserve Planner schemas; summary states are typed values too. -use super::Error; use crate::AggregateCore; +use crate::Error; use planner_types::{ - post_asap::{SummaryFamilyType, SummarySchema}, + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, pre_asap::DataType, }; use std::{cmp::Ordering, sync::Arc}; @@ -256,48 +256,8 @@ pub(crate) fn group_key(row: &[Value], columns: &[usize]) -> Result> .collect() } -pub(crate) fn validate_family(family: &SummaryFamilyType) -> Result<(), Error> { - use planner_types::post_asap::SketchAlgorithm as A; - if let SummaryFamilyType::Sketch(kind, grouping) = family { - if let planner_types::post_asap::SketchParams::CmsWithHeap { - width, - depth, - heap_size, - } = kind.params() - { - return if kind.algorithm() == &A::CmsWithHeap - && *width > 0 - && *depth > 0 - && *heap_size > 0 - && grouping == &Default::default() - { - Ok(()) - } else { - Err(Error::Invalid( - "invalid weighted CMS family or grouping strategy".into(), - )) - }; - } - } - match family { - SummaryFamilyType::ExactAggregate(..) => {} - SummaryFamilyType::Sketch(kind, _) - if matches!(kind.algorithm(), A::Kll | A::DDSketch | A::Hll) => {} - _ => { - return Err(Error::Invalid( - "summary family has no native DAG state implementation".into(), - )) - } - } - crate::capability::validate_summary_kernel( - family, - &planner_types::post_asap::SummaryUpdate::column( - planner_types::pre_asap::ColumnRef::SampleValue, - ), - &Default::default(), - ) - .map_err(Error::Invalid) -} +pub(crate) use crate::capability::validate_native_family as validate_family; + fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Result<(), Error> { use crate::summary_operators::{ datasketches_kll_accumulator::DatasketchesKLLAccumulator, @@ -390,3 +350,17 @@ pub(crate) fn validate_schema(schema: &Schema) -> Result<(), Error> { } Ok(()) } + +pub(crate) fn field(schema: &Schema, column: usize) -> Result<&SummaryField, Error> { + schema + .fields + .get(column) + .ok_or_else(|| Error::Invalid("column out of range".into())) +} +pub(crate) fn plain(schema: &Schema, column: usize) -> Result<(&DataType, bool), Error> { + let f = field(schema, column)?; + let SummaryFamilyType::Plain(dtype) = &f.dtype else { + return Err(Error::Invalid("plain value required".into())); + }; + Ok((dtype, f.nullable)) +} diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/asap-physical-operators/tests/blocking_resources.rs new file mode 100644 index 00000000..6cdc3979 --- /dev/null +++ b/crates/asap-physical-operators/tests/blocking_resources.rs @@ -0,0 +1,181 @@ +//! Blocking operators enforce resources before returning their first batch. +use asap_physical_operators::dag::{ + operators::Operator, + values::{Batch, Schema, Value}, + Error, Limits, PhysicalDag, PhysicalOperator, RunContext, Scope, +}; +use futures::{executor::block_on, FutureExt, StreamExt}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{DataType, JoinKind, Predicate, QueryExpr, ScalarValue}, +}; +use std::sync::Arc; + +fn schema(width: usize) -> Schema { + Arc::new(SummarySchema { + fields: (0..width) + .map(|i| SummaryField { + name: format!("v{i}"), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: false, + }) + .collect(), + time_index: None, + }) +} +fn context(max_bytes: usize) -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_bytes, + ..Limits::default() + }, + ) + .unwrap() +} +fn source(n: usize) -> PhysicalDag<'static, Batch, Schema> { + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + schema(1), + vec![Batch::try_new(schema(1), vec![vec![Value::Int64(1)]; n]).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag +} +fn cross_join() -> Operator { + Operator::relational_join( + schema(1), + schema(1), + JoinKind::Cross, + &Predicate(std::rc::Rc::new(QueryExpr::Literal(ScalarValue::Boolean( + true, + )))), + schema(2), + ) + .unwrap() +} + +// Even callers starting an operator directly cannot bypass its workspace budget. +#[test] +fn join_reserves_result_growth_before_returning_output() { + let sources = source(64); + let run = context(32 * 1024); + let inputs = sources.execute(&[0, 0], run.clone()).unwrap(); + let join = cross_join(); + let mut output = join.start(inputs, run.clone()).unwrap(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::MemoryLimit)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); +} + +// A single large input batch must not monopolize the worker during a join. +#[test] +fn join_yields_during_computation_and_observes_cancellation() { + let sources = source(64); + let run = context(16 * 1024 * 1024); + let inputs = sources.execute(&[0, 0], run.clone()).unwrap(); + let join = cross_join(); + let mut output = join.start(inputs, run.clone()).unwrap(); + assert!( + output.next().now_or_never().is_none(), + "join should yield before producing all 4096 rows" + ); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); +} + +// Sorting and grouping yield even for one large batch. +#[test] +fn blocking_reductions_yield_and_release_memory_on_cancellation() { + use asap_physical_operators::{ + operators::{Reduction, SortKey}, + plan::PhysicalOperator, + }; + let operators = vec![ + Operator::sort( + schema(1), + vec![SortKey { + column: 0, + descending: false, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + Operator::aggregate(schema(1), vec![], vec![("sum".into(), Reduction::Sum(0))]).unwrap(), + ]; + for operator in operators { + let sources = source(768); + let run = context(16 * 1024 * 1024); + let inputs = sources.execute(&[0], run.clone()).unwrap(); + let mut output = operator.start(inputs, run.clone()).unwrap(); + assert!(output.next().now_or_never().is_none()); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); + } +} + +// Merge-sort rounds preserve input order for tied keys across chunk boundaries. +#[test] +fn cooperative_sort_preserves_ties_across_chunks() { + use asap_physical_operators::operators::SortKey; + let batch = Batch::try_new( + schema(2), + (0..1025) + .rev() + .map(|i| vec![Value::Int64(i % 3), Value::Int64(i)]) + .collect(), + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(schema(2), vec![batch]).unwrap()) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::sort( + schema(2), + vec![SortKey { + column: 0, + descending: false, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + ) + .unwrap(); + let mut output = dag + .execute(&[1], context(16 * 1024 * 1024)) + .unwrap() + .remove(0); + let batch = block_on(output.next()).unwrap().unwrap(); + let expected = (0..3) + .flat_map(|key| (0..1025).rev().filter(move |i| i % 3 == key)) + .collect::>(); + for (row, expected) in batch.rows().iter().zip(expected) { + assert!(matches!(row[1], Value::Int64(i) if i == expected)); + } + assert_eq!(batch.rows().len(), 1025); +} diff --git a/crates/asap-physical-operators/tests/plan_properties.rs b/crates/asap-physical-operators/tests/plan_properties.rs new file mode 100644 index 00000000..f43ab23e --- /dev/null +++ b/crates/asap-physical-operators/tests/plan_properties.rs @@ -0,0 +1,155 @@ +//! Finite-input contracts are validated before source execution. +use asap_physical_operators::{ + operators::{Operator, SortKey}, + plan::{Boundedness, Emission, PhysicalDag}, + runtime::{Limits, OutputStream, RunContext, Scope}, + sources::{DataSources, RawSource}, + values::{Batch, Schema}, + Error, +}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{Column, DataType, QueryExpr, Schema as LogicalSchema, Source}, +}; +use std::sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, +}; +struct DeclaredSource { + schema: Schema, + boundedness: Boundedness, + opens: Arc, +} +impl RawSource for DeclaredSource { + fn schema(&self) -> Schema { + self.schema.clone() + } + fn boundedness(&self) -> Boundedness { + self.boundedness + } + fn scan(&self, _: RunContext) -> Result, Error> { + self.opens.fetch_add(1, Ordering::SeqCst); + Ok(Box::pin(futures::stream::empty())) + } +} +// A blocking parent must reject unknown and unbounded Scan inputs without opening a reader. +#[test] +fn blocking_inputs_require_an_explicit_finite_source() { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "v".into(), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: false, + }], + time_index: None, + }); + for boundedness in [ + Boundedness::Unknown, + Boundedness::Unbounded, + Boundedness::Bounded, + ] { + let opens = Arc::new(AtomicUsize::new(0)); + let mut registry = DataSources::default(); + let identity = Source::Table { + table_ref: "t".into(), + }; + registry + .register( + identity.clone(), + Arc::new(DeclaredSource { + schema: schema.clone(), + boundedness, + opens: opens.clone(), + }), + ) + .unwrap(); + let scan = registry + .bind(&QueryExpr::Scan { + source: identity, + schema: LogicalSchema::new(vec![Column::new("v", DataType::Int64, false)]), + predicates: vec![], + }) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], scan).unwrap(); + dag.add( + 1, + vec![0], + Operator::sort( + schema.clone(), + vec![SortKey { + column: 0, + descending: false, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + ) + .unwrap(); + let run = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + if boundedness == Boundedness::Bounded { + let properties = dag.properties(&[1]).unwrap(); + assert_eq!(properties[&1].emission, Emission::AfterInput); + assert_eq!(properties[&1].boundedness, Boundedness::Bounded); + assert!(dag.execute(&[1], run).is_ok()); + } else { + assert!( + matches!(dag.execute(&[1], run), Err(Error::Invalid(message)) if message.contains("requires bounded inputs")) + ); + } + assert_eq!(opens.load(Ordering::SeqCst), 0); + } +} + +// Kernel support must not be mistaken for executable native state/readout support. +#[test] +fn summary_capability_levels_are_distinct() { + use asap_physical_operators::{ + capability::{validate_native_family, validate_native_readout, validate_summary_kernel}, + Statistic, + }; + use planner_types::{ + post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate}, + pre_asap::ColumnRef, + }; + let grouping = GroupingStrategy::default(); + let cms = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::Cms, + SketchParams::Cms { + width: 64, + depth: 4, + }, + ), + grouping.clone(), + ); + let update = SummaryUpdate { + item: Some(planner_types::post_asap::SummaryInputExpr::Column( + ColumnRef::Named("host".into()), + )), + weight: planner_types::post_asap::SummaryInputExpr::Constant(1.0), + weight_domain: Default::default(), + }; + assert!(validate_summary_kernel(&cms, &update, &grouping).is_ok()); + assert!(validate_native_family(&cms).is_err()); + let kll = SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 128 }), + grouping, + ); + assert!(validate_native_family(&kll).is_ok()); + assert!(validate_native_readout(&kll, Statistic::Quantile, &Default::default()).is_err()); + assert!(validate_native_readout( + &kll, + Statistic::Quantile, + &[("quantile".into(), "0.5".into())].into() + ) + .is_ok()); +} diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs index 5390bec2..5aa9e10b 100644 --- a/crates/asap-physical-operators/tests/raw_scan.rs +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -169,6 +169,9 @@ struct CountingSource { fail: bool, } impl RawSource for CountingSource { + fn boundedness(&self) -> asap_physical_operators::plan::Boundedness { + asap_physical_operators::plan::Boundedness::Bounded + } fn schema(&self) -> Schema { self.schema.clone() } diff --git a/docs/design_docs/datafusion-execution-comparison.md b/docs/design_docs/datafusion-execution-comparison.md new file mode 100644 index 00000000..236ac75f --- /dev/null +++ b/docs/design_docs/datafusion-execution-comparison.md @@ -0,0 +1,295 @@ +# DataFusion and native ASAP execution: architecture comparison + +Audience: designers and maintainers of the Planner and execution engines. + +This survey evaluates the execution choice in PR #462. It does not replace +ASAP's summary planning semantics or introduce a DataFusion dependency. Sources +were inspected on 2026-09-24 at DataFusion commit +[`e2ca7f3`](https://github.com/apache/datafusion/tree/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38). +The native baseline is PR #462 at `d782c4e`; its subsequent module/resource +refactor improves contracts but does not add partitioned execution or spill. +Performance statements below are hypotheses unless described as implementation +facts. No comparative benchmark has been run. + +## Findings that affect the decision + +1. DataFusion's physical operator boundary uses Arrow RecordBatch streams. + Internal sketch state need not be an Arrow array or be serialized per update. +2. DataFusion does not provide arbitrary common-subplan fan-out merely by sharing + an `Arc`. That is different from being unable to implement it: + custom physical operators and explicit shared execution state are available. +3. Both logical and physical extension points are supported. A basic summary + operator need not require a DataFusion fork. Integration with optimization, + state transport and execution lifecycle is the substantial work. +4. A smaller native runtime is not evidence of lower end-to-end overhead. + Representation, batch size, state crossings and algorithm choice may dominate. +5. Borrowing module boundaries is inexpensive. Porting DataFusion algorithms to + another batch/runtime contract creates an ongoing integration and maintenance + obligation; it does not retain upstream improvements automatically. + +## Runtime costs: compare the actual execution paths + +DataFusion is an embedded Rust library. Its normal physical interface starts a +stream for an output partition; an ordinary projection starts its child stream +and wraps it. It does not create a network boundary or a separately scheduled +worker for every operator. Repartition and explicit buffering introduce tasks, +channels and buffering where the plan calls for them. See +[ProjectionExec](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/projection.rs), +[RepartitionExec](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/repartition/mod.rs), +and [BufferExec](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/buffer.rs). + +A useful decomposition, rather than a fixed “DataFusion overhead” percentage, is: + +```text +elapsed work ≈ planning + binding + execution setup + + batch/stream bookkeeping + representation conversion + + scalar/aggregate kernels + exchange + spill/I/O +``` + +| Cost | DataFusion implementation | Native #462 implementation | +| --- | --- | --- | +| Plan setup | Schema/property derivation, optimizer passes, physical construction | Graph/binding validation, per-node stream and consumer setup | +| Stream dispatch | Boxed streams and dynamic plan/expression interfaces | Local boxed streams and dynamic PhysicalOperator calls | +| Sharing | Arc ownership; explicit shared state where an operator implements it | Rc/RefCell producer state, reader maps, queues, wakers and output reservations | +| Ordinary data | Column arrays, null bitmaps, batched kernels; some operations allocate new arrays | Vec>, per-value enum dispatch, row vectors and clones; repeated row/schema checks in some paths | +| Summary data | Native accumulator while computing; Arrow-compatible state at standard operator boundaries | Native accumulator/state objects can cross edges behind Arc without encoding | +| Parallelism | Partition/exchange tasks, synchronization and data movement when selected | Worker-local execution; comparable partition parallelism is not implemented | +| Large blocking work | Specialized algorithms and spill-capable operators | In-memory joins/sorts/reductions with budget failure rather than spill | + +A RecordBatch clone shares its array references; it does not inherently copy all +column buffers. Creating arrays from row-oriented input, filtering/taking values, +and serializing opaque state can still allocate or copy. Conversely, native +fan-out shares an output Batch, but downstream operators can clone its row +vectors. Neither representation makes every operation zero-copy. See the +[Arrow RecordBatch contract](https://arrow.apache.org/rust/arrow/array/struct.RecordBatch.html) +and [array/buffer model](https://arrow.apache.org/rust/arrow_array/index.html). + +For a tiny precomputed-state readout, binding, allocation and decoding might +exceed kernel time; native execution could have an advantage. For large scans, +joins or high-cardinality grouping, vectorized kernels and a better algorithm +can outweigh framework bookkeeping. These are workload hypotheses, not measured +results. Compare equal thread counts first, then compare each engine's usable +parallelism separately. More CPU consumption from parallel execution can coexist +with lower latency. + +Both runtimes need cooperative cancellation: a long synchronous kernel can +prevent its worker from observing cancellation. DataFusion explicitly documents +this and supplies cooperative wrappers/optimizer support; embedding it does not +make a custom KLL operator preemptible. ASAP's new checkpoints solve the same +class of problem. See [DataFusion cooperative scheduling](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/coop.rs). + +## Arrow is a boundary contract, not a required sketch implementation + +If ASAP reuses DataFusion's standard ExecutionPlan and existing operators, their +interchange remains `SendableRecordBatchStream`. Replacing its item with an +arbitrary ASAP Batch would require adapters or a separate/forked execution +contract. Merely implementing a logical extension does not change that boundary. +See [ExecutionPlan](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/execution_plan.rs). + +There are several practical state representations: + +| Representation | Where the native sketch lives | Consequence | +| --- | --- | --- | +| Custom UDAF accumulator | Rust accumulator object, updated from Arrow arrays | Native update algorithm; encode partial state only when exporting it for merge/spill or final state output | +| Binary/LargeBinary state column | Serialized sketch payload in a standard Arrow column | Portable through generic batch transport; encode/decode cost at state-consuming boundaries | +| Struct/List state columns | Sketch components represented by supported Arrow types | May expose buffers without a monolithic encoding; requires a stable representation and reconstruction rules | +| Run-local handle column | Native state in a registry, integer/binary handle in the batch | Avoids payload serialization locally; registry lifetime, memory, retries and transport require custom handling | +| Fused custom physical operator | Native state remains internal until ordinary outputs are produced | Avoids intermediate state transport; generic optimizers cannot operate inside the fused region | + +DataFusion's `Accumulator` has `update_batch`, `state`, `merge_batch`, `evaluate` +and `size`. Its partial state can differ from its final value and contain several +fields. This is a natural fit for build/merge/finalize sketches, provided ASAP +implements parameter compatibility and the appropriate state schema. It does +not imply that an accumulator serializes itself for every input update. See +[Accumulator](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/expr-common/src/accumulator.rs). + +An Arrow extension type annotates a supported storage type with additional +semantics; it is not a universal container for a Rust trait object. Implementing +a custom Array trait object likewise does not ensure that generic take, filter, +IPC or spill code understands it. A binary extension type for “KLL version X, +parameters Y” is more interoperable than disguising an in-process pointer as a +portable value. Handles may work inside a controlled execution island, but must +not silently escape through spill, distributed exchange or persisted results. +See [Arrow extension types](https://arrow.apache.org/docs/format/Intro.html#extension-types). + +Family, parameters, encoding version, grouping layout, source/window coverage +and ownership must be validated whichever representation is selected. Ordinary +transport of bytes does not prove that two sketches can legally merge. + +## One producer and multiple consumers + +There are three separate mechanisms: + +- Detect equivalent expressions or subplans. +- Represent shared identity in the plan. +- Execute a producer once and deliver its results to independent consumers. + +DataFusion's expression CSE addresses repeated expressions; it is not a general +shared-subplan execution guarantee. Holding the same Arc from two parents is +also insufficient: ordinary parent execution can start its child independently. +See [expression CSE](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/optimizer/src/common_subexpr_eliminate.rs). + +However, “DataFusion cannot represent one producer, multiple consumers” is too +strong. Its custom ExecutionPlan implementations can coordinate shared state. +Upstream already has specialized sharing, for example one-shot scalar-subquery +execution and shared results. Repartition also coordinates producer/output +partition state, although partition distribution is not general broadcast. +These are evidence that custom coordination is possible, not an off-the-shelf +replacement for ASAP's DAG runtime. See +[ScalarSubqueryExec](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/scalar_subquery.rs). + +The motivating KLL case has a simpler alternative when the consumers are +compatible readouts of the same population/window: + +```text +raw rows → build/merge one KLL → estimate_many([0.5, 0.9, 0.99]) + → ordinary columns q50, q90, q99 +``` + +This can be a custom aggregate returning a Struct, or a build-state aggregate +followed by a multi-readout operator. It can avoid both general DAG broadcast and +repeated decoding. Three independent quantile aggregate calls do not by +themselves guarantee one KLL: an ASAP rule must explicitly select the common +state. Different filters, windows or downstream pipelines may prevent this +fusion and still require true fan-out. + +A general DataFusion integration would need an explicit run-scoped shared +producer/subscription operator or materialization service. Its contract must +cover producer identity per partition/run, consumer registration, bounded queues +or replay, slow/dropped consumers, terminal errors, cancellation, memory and plan +reuse. A new run must not accidentally reuse stale mutable state. Cross-query +reuse additionally requires cache coverage/revision invalidation; it is a +separate feature in both engines. + +A particularly important acceptance test is a diamond with asymmetric polling. +If one branch waits for the other to finish before polling, a shared producer +can fill that dormant branch's bounded queue and deadlock. The design must poll +branches appropriately, materialize/spill, or otherwise resolve the dependency. +This obligation applies to an ASAP adapter too; bounded broadcast alone does +not solve it. BufferExec's background queue is not automatically multicast. + +## Logical versus physical extension effort + +The difference is semantic scope, not simply adding an enum case. + +| Layer | Extension work | What is not automatic | +| --- | --- | --- | +| Logical node | UserDefinedLogicalNode, schema, children, expressions, reconstruction, equality/hash and explain | Correct approximation semantics, coverage, sharing identity and rewrite legality | +| Logical-to-physical lowering | Register an ExtensionPlanner mapping the node to existing or custom physical operators | Choosing a legal summary implementation and preserving ASAP guarantees | +| Physical operator | ExecutionPlan properties, child replacement, partition execution, stream, metrics and memory behavior | Efficient merging, spillable state, consumer sharing and durable maintenance | +| UDAF route | Native accumulator, state fields, update/merge/finalize, memory size; optionally specialized grouped accumulation | Arbitrary summary subtraction/join or a shared multi-consumer DAG | + +Logical extensions default to conservative predicate pushdown and expose hooks +for required columns and reconstruction. Approximation-sensitive rules must be +explicit: filtering before a summary can change its population, and removing a +grouping/identity column can invalidate it. See +[UserDefinedLogicalNode](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/expr/src/logical_plan/extension.rs). + +The physical planner already invokes registered extension planners and checks +their output schema. A conventional SummaryBuild or SummaryReadout can therefore +be implemented outside upstream DataFusion. Basic physical extension is feasible; +claiming transparent partitioning, spill, shared execution and maintenance +semantics is the larger engineering task. See +[extension planning](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/core/src/physical_planner.rs). + +For aggregate-shaped work, an AggregateUDF can reuse the existing aggregate +operator rather than introducing a custom physical node. High group cardinality +may justify implementing GroupsAccumulator to avoid a generic per-group adapter. +Reusing a partial/final aggregate pipeline still requires testing the sketch's +merge guarantees, memory reporting and exported state. See +[aggregate expression support](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-expr/src/aggregate.rs). + +ASAP would retain its accuracy/cost model, state-family rules, coverage and +revision metadata, and deployment publication/window lifecycle. DataFusion does +not infer these from an opaque extension. Optimizations must also preserve +correlations introduced when several estimates use the same randomized state; +shared estimates must not silently acquire independence assumptions. + +## How much optimization is actually reused? + +Reusing standard relational nodes gives the widest access to existing expression, +projection/filter, join, aggregation, ordering and partition optimizations. A +large opaque extension hides internal opportunities unless it supplies properties +or is lowered into supported nodes. False ordering/distribution declarations +risk correctness, while conservative declarations can reduce optimization. + +Keeping ASAP's IR and lowering directly to DataFusion physical operators is a +valid alternative to replacing the logical IR. It can reuse kernels and selected +physical optimization, but does not automatically run DataFusion's logical +optimizations over ASAP-specific nodes. Likewise, reusing logical IR alone does +not supply DataFusion physical execution to a separate native runtime. + +| Option | Reused capability | Main obligation | +| --- | --- | --- | +| Native ASAP execution | Existing summary model and explicit DAG fan-out | Own relational kernels, partitioning, spill, metrics and optimizer/runtime contracts | +| DataFusion backend with extensions | Standard relational operators, physical infrastructure; logical optimizations where mapped | Arrow boundaries, custom summary semantics and shared execution adapter | +| ASAP scheduler around coarse DataFusion subplans | DataFusion relational execution inside islands; native summary edges outside | Control conversion boundaries, resource budgets, cancellation and parallelism across both layers | +| DataFusion outer plan with fused native summary regions | DataFusion surrounding relational computation; native state inside regions | Keep opaque regions coarse enough to avoid repeated conversion but expose useful properties | + +A hybrid is an option to benchmark, not automatically the best of both worlds. +Wrapping every tiny native operation in a separate DataFusion execution adds +repeated setup and conversions. Coarse subplans amortize those boundaries, but +two independent budgets or schedulers must not oversubscribe memory or threads. + +## Borrowing organization versus maintaining copied algorithms + +Adopting modules such as plan, runtime, expressions, joins, aggregate, sources +and spill is a useful ownership decision independent of execution framework. +It does not require adopting DataFusion IR or Arrow, and #462 does this now. + +Porting a hash join or external sort is much more than copying its main loop. +Those implementations rely on array kernels, expressions, row encodings, +distribution/ordering facts, memory reservations, async streams, spill formats +and tests. Retaining Arrow can reduce the port surface; replacing the data model +increases it. A maintained fork must also track upstream correctness fixes and +behavioral changes. See the concrete dependency surface in +[grouped aggregation](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs). + +A native engine can intentionally support less. That can be a sound decision +when workloads remain bounded and summary-heavy, but missing large-query +capabilities are a scope tradeoff, not evidence that their overhead has been +eliminated at equivalent functionality. + +## Experiments needed before making a performance claim + +Use the same sketch implementation, parameters, input values, grouping, windows +and accuracy contract. First measure already-bound execution; measure planning +and cold setup separately. Report both single-worker and independently tuned +parallel results. Do not compare a parallel hash join with a single-worker nested +loop and label the difference “runtime overhead.” + +| Workload | Question isolated | +| --- | --- | +| Existing KLL → 1/3/10 quantiles | Readout setup, state cloning, decoding, fusion and consumer overhead | +| Raw values → KLL → several quantiles | Build cost versus state export/transport; verify one build | +| One producer → two asymmetric branches | Buffer growth, progress, dropped consumer and cancellation behavior | +| Many tiny panes and high group cardinality | Allocation, per-group adapters, state bytes and setup amortization | +| Raw Scan → Filter/Project → grouped aggregate | Native rows versus Arrow conversion and columnar computation | +| Join and Sort/Limit under memory pressure | Algorithm choice, workspace, graceful failure versus spill | + +Record wall latency (including p50/p95 for repeated small queries), CPU time, +allocations, peak RSS, charged memory, serialized bytes, encode/decode counts, +producer starts, task/partition counts and cancellation latency. Distinguish +first-run initialization from warm execution. Verify results and summary +compatibility before interpreting timing. Acceptance also includes repeated +executions, multiple roots, reordered consumers and shared-error propagation. + +The KLL prototype should compare native fan-out, a DataFusion UDAF producing +multiple estimates, DataFusion state output plus readout, and only then a custom +shared-producer adapter if independent branches are necessary. This distinguishes +an unavoidable domain cost from a cost introduced by a particular adapter. + +## Position for PR #462 + +Proceed with the native module/runtime cleanup as scoped, while keeping the +execution-backend decision evidence-based. The current justification is direct +ownership of native summary-state edges and shared execution across both engines, +with explicitly limited generic execution capabilities. It is not established +that DataFusion is slower, cannot carry sketches, or cannot support fan-out. + +If broad relational workloads, partition scaling and spill become near-term +requirements, a DataFusion execution backend or coarse hybrid deserves a serious +prototype before porting those subsystems. If bounded summary DAGs dominate and +measured conversion/state-transport costs are material, the native path has a +stronger workload-specific case. The proposed experiments are the decision gate; +this survey alone does not establish a performance winner. diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index dbb50405..237c4958 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -101,6 +101,48 @@ on the local IR crate. Operator unit tests can exercise private implementation details; Planner integration tests check that emitted DAGs bind and execute. Runtime values preserve the IR schema instead of redefining its type semantics. +### Module ownership + +```text +src/ + plan/ PhysicalDag, PhysicalOperator, properties and validation + runtime/ streams, shared producers, context, memory and cancellation + expressions/ scalar evaluation and Planner expression adaptation + operators/ + projection.rs + filter.rs + joins/ + aggregate/ ordinary and temporal reductions + sort.rs + limit.rs + summary/ build, merge and readout + source.rs literal/batch sources, union and scalar conversion + sources/ raw-source API, Scan and memory connector + binding/ Planner executable DAG to physical operators + summary_operators/ mathematical summary kernels, factory and traits + stored_state/ decoding, delta application and persisted-state readout + capability.rs kernel and native operator support checks + values.rs typed rows and state payload validation +``` + +The graph owns topology and static checks; the runtime owns each execution's +producer state. Operator modules own both construction checks and computation. +The `Operator` enum dispatch remains a small internal routing point. This does +not change `SummaryExpr`, Planner semantics or shared-producer identity. + +`Expression` is the typed native builder; `CompiledExpression` validates and +adapts Planner scalar expressions. Both are owned by `expressions`, with shared +numeric execution. Planner-specific coercions and checked PromQL division remain +explicit at their respective binding boundaries. No expression evaluator lives +inside the projection or filter implementation. + +Deployment engines normally use `binding`, `plan`, `runtime` and `sources`. +`summary_operators` exposes update kernels for pane maintenance; `stored_state` +serves deployments reconstructing persisted panes. Kernel traits include state +serialization because persistence consumes those states, but neither the graph +nor its scheduler depends on serialization. Existing `dag`, `accumulators`, `factory`, `traits` and +`arithmetic` import paths are thin compatibility re-exports. + ## Execution contract An immutable plan describes typed nodes and dependency edges. Each execution @@ -119,7 +161,31 @@ streams are worker-local. Deployments poll all consumers concurrently. The byte budget accounts for retained outputs and native operator state, including outputs held after queue eviction. It is not an RSS limit: source-owned data, temporary allocation peaks and allocator overhead remain outside that estimate. Blocking -operators currently have no spill implementation. +operators currently have no spill implementation. Join results and membership +sets, grouping workspace, sort scratch space and merged summary-state estimates +are charged while retained. Long row loops and sort merge steps yield to the +caller, so cancellation and other consumers can progress within a single batch. +Individual kernel calls and scalar evaluations remain synchronous; memory +estimates are not allocator-exact peak bounds. + +### Finite input and emission + +`PhysicalOperator::properties` reports output boundedness and emission mode. +Unknown source boundedness is conservative: it cannot satisfy a finite-input +requirement. `PhysicalDag::properties` derives these facts together with topology +and schema validation before `start` is called on any source. + +Sort, ordinary aggregate, temporal reductions, both joins, scalar/keyed summary build and summary merge +and vector-to-scalar require bounded inputs and emit after input ends. Summary +build updates incrementally but still finalizes at end-of-input. Projection, +filter, limit, union and readout emit incrementally. A global Limit bounds its +output cardinality; a grouped Limit inherits input boundedness because new +groups may continue arriving. Neither declaration promises a time deadline. + +`RawSource::boundedness` defaults to Unknown. Connectors must explicitly promise +that a snapshot or window ends; merely receiving a query/ingestion `Scope` is +insufficient. The memory connector declares Bounded. Installed physical source +frontiers preserve their supplied properties through the checked binding wrapper. ## Operator coverage @@ -163,21 +229,55 @@ Deployments provide explicit storage or ingestion source frontiers and may bind raw Scan through the shared data-source interface. The memory connector proves the library contract; backend raw-data access still requires a deployment connector. +### Capability levels + +| Level | Acceptance contract | Scope | +| --- | --- | --- | +| Update kernel | `capability::validate_summary_kernel` | Family, parameters, grouping and item/update layout; used by the accumulator factory | +| Native state edge | `capability::validate_native_family` | Exact accumulators, KLL, DDSketch, HLL and Float64 weighted CMS with compatible parameters | +| Scalar native readout | `capability::validate_native_readout` | Supported native state plus statistic/readout arguments | +| Keyed native readout | `Operator::keyed_readout` | Weighted CMS family, heap capacity, typed identity/score schema and preserved partition columns | +| Complete physical plan | `binding::bind` / `bind_with_data_sources` | Node support, expressions, schemas, source frontiers and bounded input requirements | +| Persisted state | `stored_state` decoders and readout functions | Stored format and family-specific reconstruction/readout support | + +For example, a valid CMS update kernel does not imply a native CMS batch edge. +Stored-state support also does not register a native operator. Consumers must +use the contract for the path they intend to execute rather than treating kernel +availability as whole-plan acceptance. + +Partitioned parallel execution, disk spill, cost-based algorithm selection, +physical ordering/distribution properties and per-operator Explain/Analyze +metrics remain future extensions. This change establishes finite-input and +emission contracts without claiming those additional DataFusion capabilities. + ## DataFusion reuse vs independent implementation -| Decision dimension | Reuse DataFusion | Independent ASAP implementation | +The [execution comparison survey](datafusion-execution-comparison.md) examines +runtime costs, Arrow/sketch representation, producer sharing, logical/physical +extension points, optimization reuse, native maintenance cost and a benchmark +plan against pinned upstream sources. + +| Decision dimension | DataFusion backend | Native ASAP execution | | --- | --- | --- | -| General computation | Reuse mature Arrow operators and expression execution | Implement and test the supported Planner vocabulary explicitly | -| Shared DAG producer | Shared plan references need an explicit execution-sharing and buffering policy | One producer and independent consumer cursors are part of the runtime contract | -| Summary lifecycle | Add custom summary state operators to the framework | Summary construction, merge and readout are native capabilities | -| In-memory representation | Operators exchange Arrow RecordBatch values. Custom summary state may not map naturally to Arrow and can require an explicit encoding, wrapper or conversion, with associated integration and potential copying costs | Native values can carry ASAP-defined summary state directly, without requiring every state format to fit Arrow; the library must still define and validate state types, ownership and compatibility | -| Engine reuse | Adapt both engines to DataFusion's execution model | Both engines bind the same ASAP interfaces | -| Engineering cost | Less generic operator work; integration and semantic adaptation remain | More operator, typing, scheduling and resource-accounting responsibility | - -DataFusion is a design reference, not this library's execution dependency. This -choice does not claim that DataFusion cannot express shared dependencies. ASAP -chooses direct ownership of execution sharing and summary-state semantics across -both engines. Mathematical sketch kernels remain reusable implementation details. +| Runtime overhead | Partition streams; ordinary operators do not each require a separate task. Conversion, state transport and exchanges depend on the chosen integration | Local streams and native state edges; producer queues, per-row values, validation and copies still have costs | +| Summary representation | Internal accumulators may remain native Rust objects; standard physical edges use Arrow batches, with binary/structured state or custom adapters | Native summary objects can cross edges directly | +| Shared execution | Shared plan references do not automatically share results; fusion, materialization or custom run-scoped producer coordination can implement reuse | One producer per node per run, with independent consumer cursors | +| Extension effort | Both logical and physical extension APIs exist; conventional custom nodes need not require an upstream fork | Direct control of both interfaces and implementations | +| Generic computation | Mature relational kernels, partitioning and spill infrastructure, subject to correct custom properties and state contracts | Supported vocabulary implemented locally; partitioned execution and spill remain deferred | +| Maintenance cost | Adapter, semantic and upstream-version integration | Ownership of operators, scheduler, resource contracts and future generic execution features | + +The native decision in this PR is scoped to direct ownership of summary-state +edges and shared DAG execution. It does not establish that DataFusion is slower, +that Arrow requires serializing every sketch update, or that DataFusion cannot +implement fan-out. Multiple quantiles of one KLL can often be fused into one +readout; independent downstream branches may still require general sharing. + +Borrowing DataFusion's module boundaries is useful regardless of backend choice. +Porting its algorithms also means adapting their array, expression, memory, +partition and spill dependencies and maintaining those adaptations. The survey +specifies the measurements needed to compare full DataFusion, native execution +and coarse hybrid subplans without confusing algorithm improvements with runtime +overhead. No comparative benchmark is claimed here. ## Acceptance From e83a40cf6e42f3652309e9fdd2876214bd8b255e Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 16:09:09 +0000 Subject: [PATCH 12/90] test: cover integrated weighted-summary cancellation and clarify survey version scope --- .../tests/blocking_resources.rs | 59 +++++++++++++++++++ .../datafusion-execution-comparison.md | 9 +++ 2 files changed, 68 insertions(+) diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/asap-physical-operators/tests/blocking_resources.rs index 6cdc3979..19381175 100644 --- a/crates/asap-physical-operators/tests/blocking_resources.rs +++ b/crates/asap-physical-operators/tests/blocking_resources.rs @@ -179,3 +179,62 @@ fn cooperative_sort_preserves_ties_across_chunks() { } assert_eq!(batch.rows().len(), 1025); } + +// The integrated weighted-summary path obeys the same cooperative cancellation contract. +#[test] +fn weighted_summary_build_yields_within_a_batch() { + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let input = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "item".into(), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: false, + }, + SummaryField { + name: "weight".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }, + ], + time_index: None, + }); + let mut sources = PhysicalDag::default(); + let batch = Batch::try_new( + input.clone(), + (0..1500) + .map(|i| vec![Value::Int64(i % 8), Value::Float64(0.25)]) + .collect(), + ) + .unwrap(); + sources + .add( + 0, + vec![], + Operator::source(input.clone(), vec![batch]).unwrap(), + ) + .unwrap(); + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 64, + depth: 3, + heap_size: 8, + }, + ), + Default::default(), + ); + let operator = Operator::keyed_summary_build(input, family, 1, vec![0], vec![]).unwrap(); + let run = context(16 * 1024 * 1024); + let inputs = sources.execute(&[0], run.clone()).unwrap(); + let mut output = operator.start(inputs, run.clone()).unwrap(); + assert!(output.next().now_or_never().is_none()); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); +} diff --git a/docs/design_docs/datafusion-execution-comparison.md b/docs/design_docs/datafusion-execution-comparison.md index 236ac75f..f1eb55a9 100644 --- a/docs/design_docs/datafusion-execution-comparison.md +++ b/docs/design_docs/datafusion-execution-comparison.md @@ -11,6 +11,15 @@ refactor improves contracts but does not add partitioned execution or spill. Performance statements below are hypotheses unless described as implementation facts. No comparative benchmark has been run. +The [SQL frontend](../../crates/frontend-sql/Cargo.toml) already depends on +DataFusion 43 for parsing/planning; the physical-operator crate does not depend +on DataFusion. Reusing the existing frontend dependency and selecting a newer +execution backend are different choices. This survey follows the requested +upstream main, so an implementation must select a supported release and verify +its exact APIs rather than assume main's interfaces exist in version 43. +Build/binary-size effects also depend on which crates a deployment already links; +they should be measured separately from per-query runtime costs. + ## Findings that affect the decision 1. DataFusion's physical operator boundary uses Arrow RecordBatch streams. From 44bd40c1df29fc7568b80635989a145a3ce1b58f Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 16:21:22 +0000 Subject: [PATCH 13/90] feat: align weighted CountSketch DAG execution with Planner candidates --- .../src/binding/mod.rs | 10 +- .../asap-physical-operators/src/capability.rs | 18 +- .../src/summary_operators/mod.rs | 3 +- ...{weighted_cms.rs => weighted_frequency.rs} | 215 ++++++++++++++---- crates/asap-physical-operators/src/values.rs | 19 +- .../tests/physical_dag.rs | 29 ++- .../tests/weighted_topk_binding.rs | 24 +- docs/design_docs/physical-operators.md | 21 +- 8 files changed, 246 insertions(+), 93 deletions(-) rename crates/asap-physical-operators/src/summary_operators/{weighted_cms.rs => weighted_frequency.rs} (53%) diff --git a/crates/asap-physical-operators/src/binding/mod.rs b/crates/asap-physical-operators/src/binding/mod.rs index 09a2df0c..03dad551 100644 --- a/crates/asap-physical-operators/src/binding/mod.rs +++ b/crates/asap-physical-operators/src/binding/mod.rs @@ -327,10 +327,12 @@ fn bind_operation(node: &ExecutableDagNode, inputs: &[Schema]) -> Result Result<(), Error> { use planner_types::post_asap::SketchAlgorithm as A; if let SummaryFamilyType::Sketch(kind, grouping) = family { - if let planner_types::post_asap::SketchParams::CmsWithHeap { - width, - depth, - heap_size, - } = kind.params() - { - return if kind.algorithm() == &A::CmsWithHeap - && valid_matrix(*width, *depth) - && *heap_size > 0 - && grouping == &Default::default() - { + if matches!(kind.algorithm(), A::CmsWithHeap | A::CountSketchWithHeap) { + let (_, width, depth, _) = crate::summary_operators::weighted_frequency::WeightedFrequency::configuration(kind)?; + return if valid_matrix(width as u32, depth as u32) && grouping == &Default::default() { Ok(()) } else { - Err(Error::Invalid( - "invalid weighted CMS family or grouping strategy".into(), - )) + Err(Error::Invalid("invalid weighted frequency dimensions or grouping strategy".into())) }; } } diff --git a/crates/asap-physical-operators/src/summary_operators/mod.rs b/crates/asap-physical-operators/src/summary_operators/mod.rs index 904f421f..1dc0d247 100644 --- a/crates/asap-physical-operators/src/summary_operators/mod.rs +++ b/crates/asap-physical-operators/src/summary_operators/mod.rs @@ -36,7 +36,6 @@ pub use min_accumulator::*; pub use sketch_envelope_accumulator::*; pub use sum_accumulator::*; -pub mod weighted_cms; - +pub mod weighted_frequency; pub mod factory; pub mod traits; diff --git a/crates/asap-physical-operators/src/summary_operators/weighted_cms.rs b/crates/asap-physical-operators/src/summary_operators/weighted_frequency.rs similarity index 53% rename from crates/asap-physical-operators/src/summary_operators/weighted_cms.rs rename to crates/asap-physical-operators/src/summary_operators/weighted_frequency.rs index 626a13e7..d78dc116 100644 --- a/crates/asap-physical-operators/src/summary_operators/weighted_cms.rs +++ b/crates/asap-physical-operators/src/summary_operators/weighted_frequency.rs @@ -1,4 +1,4 @@ -//! Float64 weighted CMS state with typed candidate identities. Each instance +//! Float64 weighted CMS and CountSketch state with typed candidate identities. Each instance //! represents one partition at one evaluation scope; updates never round rates //! to integer counts. Candidate membership still requires Planner evidence. use crate::{values::Value, Error}; @@ -26,7 +26,11 @@ impl Identity { Value::Int64(v) => Self::Int64(*v), Value::Float64(v) if v.is_finite() => Self::Float64(if *v == 0.0 { 0.0 } else { *v }), Value::Utf8(v) => Self::Utf8(v.to_string()), - _ => return Err(Error::Invalid("unsupported weighted CMS identity".into())), + _ => { + return Err(Error::Invalid( + "unsupported weighted frequency identity".into(), + )) + } }) } fn value(&self) -> Value { @@ -65,29 +69,84 @@ impl Ord for Candidate { .then_with(|| other.key.cmp(&self.key)) } } +#[derive(Copy, Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +pub enum FrequencyAlgorithm { + Cms, + CountSketch, +} #[derive(Clone, Debug, Serialize, Deserialize)] -pub struct WeightedCms { +pub struct WeightedFrequency { + algorithm: FrequencyAlgorithm, width: usize, depth: usize, capacity: usize, cells: Vec, candidates: BinaryHeap, } -impl WeightedCms { +impl WeightedFrequency { + pub(crate) fn configuration( + kind: &planner_types::post_asap::SketchKind, + ) -> Result<(FrequencyAlgorithm, usize, usize, usize), Error> { + use planner_types::post_asap::{SketchAlgorithm as A, SketchParams as P}; + let (algorithm, width, depth, capacity) = match (kind.algorithm(), kind.params()) { + ( + A::CmsWithHeap, + P::CmsWithHeap { + width, + depth, + heap_size, + }, + ) => (FrequencyAlgorithm::Cms, *width, *depth, *heap_size), + ( + A::CountSketchWithHeap, + P::CountSketchWithHeap { + width, + depth, + heap_size, + }, + ) if depth % 2 == 1 => (FrequencyAlgorithm::CountSketch, *width, *depth, *heap_size), + _ => { + return Err(Error::Invalid( + "unsupported weighted frequency family or depth".into(), + )) + } + }; + if width == 0 || depth == 0 || capacity == 0 { + return Err(Error::Invalid( + "invalid weighted frequency dimensions".into(), + )); + } + Ok((algorithm, width as usize, depth as usize, capacity as usize)) + } + + pub(crate) fn algorithm(&self) -> FrequencyAlgorithm { + self.algorithm + } pub(crate) fn shape(&self) -> (usize, usize, usize) { (self.width, self.depth, self.capacity) } - pub fn new(width: usize, depth: usize, capacity: usize) -> Result { + pub fn new( + algorithm: FrequencyAlgorithm, + width: usize, + depth: usize, + capacity: usize, + ) -> Result { let len = width .checked_mul(depth) - .filter(|_| width > 0 && depth > 0 && capacity > 0) - .ok_or_else(|| Error::Invalid("invalid weighted CMS dimensions".into()))?; + .filter(|_| { + width > 0 + && depth > 0 + && capacity > 0 + && (algorithm != FrequencyAlgorithm::CountSketch || depth % 2 == 1) + }) + .ok_or_else(|| Error::Invalid("invalid weighted frequency dimensions".into()))?; let mut cells = Vec::new(); cells .try_reserve_exact(len) - .map_err(|_| Error::Invalid("weighted CMS allocation failed".into()))?; + .map_err(|_| Error::Invalid("weighted frequency allocation failed".into()))?; cells.resize(len, 0.0); Ok(Self { + algorithm, width, depth, capacity, @@ -100,8 +159,8 @@ impl WeightedCms { pub fn from_bytes(bytes: &[u8]) -> Result { use bincode::Options; let bytes = bytes - .strip_prefix(b"ASAP-WCMS-1\0") - .ok_or_else(|| Error::Invalid("weighted CMS format/version mismatch".into()))?; + .strip_prefix(b"ASAP-WFREQ-1\0") + .ok_or_else(|| Error::Invalid("weighted frequency format/version mismatch".into()))?; let mut state: Self = bincode::DefaultOptions::new() .with_fixint_encoding() .with_limit(bytes.len() as u64) @@ -111,11 +170,15 @@ impl WeightedCms { if state.width == 0 || state.depth == 0 || state.capacity == 0 + || (state.algorithm == FrequencyAlgorithm::CountSketch && state.depth.is_multiple_of(2)) || state.width.checked_mul(state.depth) != Some(state.cells.len()) - || state.cells.iter().any(|v| !v.is_finite() || *v < 0.0) + || state + .cells + .iter() + .any(|v| !v.is_finite() || (state.algorithm == FrequencyAlgorithm::Cms && *v < 0.0)) || state.candidates.len() > state.capacity { - return Err(Error::Invalid("invalid weighted CMS state".into())); + return Err(Error::Invalid("invalid weighted frequency state".into())); } for candidate in &state.candidates { if candidate @@ -126,23 +189,38 @@ impl WeightedCms { .map_err(|e| Error::Invalid(e.to_string()))? != candidate.key { - return Err(Error::Invalid("invalid weighted CMS identity".into())); + return Err(Error::Invalid("invalid weighted frequency identity".into())); } } state.retain(state.candidates.iter().cloned().collect()); Ok(state) } - fn indexes(&self, key: &[u8]) -> impl Iterator + '_ { - let key = key.to_vec(); + fn indexes<'a>(&'a self, key: &'a [u8]) -> impl Iterator + 'a { (0..self.depth).map(move |row| { - row * self.width - + (xxhash_rust::xxh64::xxh64(&key, row as u64) % self.width as u64) as usize + let bucket = + (xxhash_rust::xxh64::xxh64(key, 2 * row as u64) % self.width as u64) as usize; + let sign = if self.algorithm == FrequencyAlgorithm::CountSketch + && xxhash_rust::xxh64::xxh64(key, 2 * row as u64 + 1) & 1 != 0 + { + -1.0 + } else { + 1.0 + }; + (row * self.width + bucket, sign) }) } fn estimate(&self, key: &[u8]) -> f64 { - self.indexes(key) - .map(|i| self.cells[i]) - .fold(f64::INFINITY, f64::min) + let mut estimates = self + .indexes(key) + .map(|(i, sign)| self.cells[i] * sign) + .collect::>(); + match self.algorithm { + FrequencyAlgorithm::Cms => estimates.into_iter().fold(f64::INFINITY, f64::min), + FrequencyAlgorithm::CountSketch => { + let middle = estimates.len() / 2; + *estimates.select_nth_unstable_by(middle, f64::total_cmp).1 + } + } } fn retain(&mut self, mut candidates: Vec) { candidates.sort_by(|a, b| a.key.cmp(&b.key)); @@ -157,9 +235,9 @@ impl WeightedCms { } } pub fn update(&mut self, values: &[Value], weight: f64) -> Result<(), Error> { - if !weight.is_finite() || weight < 0.0 { + if !weight.is_finite() || (self.algorithm == FrequencyAlgorithm::Cms && weight < 0.0) { return Err(Error::Operator( - "weighted CMS requires finite nonnegative rates".into(), + "weighted frequency requires finite weights; CMS additionally requires nonnegative weights".into(), )); } let identity = values @@ -170,12 +248,12 @@ impl WeightedCms { let indexes = self.indexes(&key).collect::>(); if indexes .iter() - .any(|&i| !(self.cells[i] + weight).is_finite()) + .any(|&(i, sign)| !(self.cells[i] + sign * weight).is_finite()) { - return Err(Error::Operator("weighted CMS sum overflow".into())); + return Err(Error::Operator("weighted frequency sum overflow".into())); } - for i in indexes { - self.cells[i] += weight; + for (i, sign) in indexes { + self.cells[i] += sign * weight; } let mut candidates = self.candidates.iter().cloned().collect::>(); candidates.push(Candidate { @@ -200,22 +278,22 @@ impl WeightedCms { .collect() } } -impl SerializableToSink for WeightedCms { +impl SerializableToSink for WeightedFrequency { fn serialize_to_json(&self) -> serde_json::Value { - serde_json::to_value(self).expect("finite validated CMS state") + serde_json::to_value(self).expect("finite validated frequency state") } fn serialize_to_bytes(&self) -> Vec { - let mut bytes = b"ASAP-WCMS-1\0".to_vec(); - bytes.extend(bincode::serialize(self).expect("serializable CMS state")); + let mut bytes = b"ASAP-WFREQ-1\0".to_vec(); + bytes.extend(bincode::serialize(self).expect("serializable frequency state")); bytes } } -impl AggregateCore for WeightedCms { +impl AggregateCore for WeightedFrequency { fn clone_boxed_core(&self) -> Box { Box::new(self.clone()) } fn type_name(&self) -> &'static str { - "WeightedCms" + "WeightedFrequency" } fn as_any(&self) -> &dyn std::any::Any { self @@ -230,15 +308,15 @@ impl AggregateCore for WeightedCms { let other = other .as_any() .downcast_ref::() - .ok_or("weighted CMS state type mismatch")?; - if (self.width, self.depth, self.capacity) != (other.width, other.depth, other.capacity) { - return Err("weighted CMS shape mismatch".into()); + .ok_or("weighted frequency state type mismatch")?; + if self.algorithm != other.algorithm || self.shape() != other.shape() { + return Err("weighted frequency shape mismatch".into()); } let mut result = self.clone(); for (value, rhs) in result.cells.iter_mut().zip(&other.cells) { *value += rhs; if !value.is_finite() { - return Err("weighted CMS merge overflow".into()); + return Err("weighted frequency merge overflow".into()); } } result.retain( @@ -251,7 +329,10 @@ impl AggregateCore for WeightedCms { Ok(Box::new(result)) } fn get_accumulator_type(&self) -> AggregationType { - AggregationType::CountMinSketchWithHeap + match self.algorithm { + FrequencyAlgorithm::Cms => AggregationType::CountMinSketchWithHeap, + FrequencyAlgorithm::CountSketch => AggregationType::CountSketchWithHeap, + } } fn get_keys(&self) -> Option> { None @@ -262,7 +343,7 @@ impl AggregateCore for WeightedCms { _: &Option, _: &HashMap, ) -> Result> { - Err("weighted CMS uses typed row readout".into()) + Err("weighted frequency uses typed row readout".into()) } fn approx_memory_bytes(&self) -> usize { std::mem::size_of::() @@ -292,18 +373,64 @@ impl AggregateCore for WeightedCms { #[cfg(test)] mod tests { use super::*; + // Signed fractional updates and merges retain numeric ranking, not magnitude ranking. + #[test] + fn count_sketch_signed_updates_roundtrip_and_merge() { + let mut left = WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 4096, 5, 8).unwrap(); + left.update(&[Value::Int64(1)], -10.5).unwrap(); + left.update(&[Value::Null], 0.125).unwrap(); + left.update(&[Value::Null], -0.0625).unwrap(); + let mut right = + WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 4096, 5, 8).unwrap(); + right.update(&[Value::Null], 0.25).unwrap(); + let merged = left.merge_with(&right).unwrap(); + let merged = merged.as_any().downcast_ref::().unwrap(); + let decoded = WeightedFrequency::from_bytes(&merged.serialize_to_bytes()).unwrap(); + let rows = decoded.rows(2); + assert!(matches!(rows[0][0], Value::Null)); + assert!(matches!(rows[0][1], Value::Float64(0.3125))); + assert!(matches!(rows[1][1], Value::Float64(-10.5))); + assert!(left + .merge_with(&WeightedFrequency::new(FrequencyAlgorithm::Cms, 4096, 5, 8).unwrap()) + .is_err()); + let before = left.serialize_to_bytes(); + for weight in [f64::NAN, f64::INFINITY, f64::NEG_INFINITY] { + assert!(left.update(&[Value::Null], weight).is_err()); + assert_eq!(left.serialize_to_bytes(), before); + } + } + + // CountSketch uses sign-corrected median: one corrupted row cannot dominate five rows. + #[test] + fn count_sketch_median_and_odd_depth_contract() { + assert!(WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 8, 2, 2).is_err()); + let mut state = WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 16, 5, 8).unwrap(); + state.update(&[Value::Int64(4)], -0.375).unwrap(); + let key = state.candidates.peek().unwrap().key.clone(); + let indexes = state.indexes(&key).collect::>(); + for &(index, sign) in &indexes { + assert_eq!(state.cells[index] * sign, -0.375); + } + state.cells[indexes[0].0] += 1000.0; + assert_eq!(state.estimate(&key), -0.375); + let mut invalid = state.clone(); + invalid.depth = 4; + invalid.cells.truncate(64); + assert!(WeightedFrequency::from_bytes(&invalid.serialize_to_bytes()).is_err()); + } + // Invalid rates must not mutate state; typed keys cannot collide by formatting. #[test] fn fractional_updates_typed_identities_and_invalid_weights() { - let mut state = WeightedCms::new(4096, 5, 8).unwrap(); + let mut state = WeightedFrequency::new(FrequencyAlgorithm::Cms, 4096, 5, 8).unwrap(); state.update(&[Value::Int64(1)], 0.125).unwrap(); state.update(&[Value::Int64(1)], 0.125).unwrap(); state.update(&[Value::Utf8("1".into())], 0.5).unwrap(); state.update(&[Value::Null], 0.75).unwrap(); let before = state.serialize_to_bytes(); - let decoded = WeightedCms::from_bytes(&before).unwrap(); + let decoded = WeightedFrequency::from_bytes(&before).unwrap(); assert_eq!(decoded.rows(8).len(), 3); - assert!(WeightedCms::from_bytes(b"old integer state").is_err()); + assert!(WeightedFrequency::from_bytes(b"old integer state").is_err()); for weight in [-1.0, f64::INFINITY, f64::NAN] { assert!(state.update(&[Value::Null], weight).is_err()); assert_eq!(state.serialize_to_bytes(), before); @@ -317,15 +444,15 @@ mod tests { // Merge uses the same Float64 state representation and rejects other shapes. #[test] fn compatible_merge_preserves_fractional_weights() { - let mut left = WeightedCms::new(4096, 5, 8).unwrap(); + let mut left = WeightedFrequency::new(FrequencyAlgorithm::Cms, 4096, 5, 8).unwrap(); let mut right = left.clone(); left.update(&[Value::Int64(7)], 0.125).unwrap(); right.update(&[Value::Int64(7)], 0.25).unwrap(); let merged = left.merge_with(&right).unwrap(); - let merged = merged.as_any().downcast_ref::().unwrap(); + let merged = merged.as_any().downcast_ref::().unwrap(); assert!(matches!(merged.rows(1)[0][1], Value::Float64(0.375))); assert!(left - .merge_with(&WeightedCms::new(32, 5, 8).unwrap()) + .merge_with(&WeightedFrequency::new(FrequencyAlgorithm::Cms, 32, 5, 8).unwrap()) .is_err()); } } diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs index b3b84b0c..ea019209 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -268,21 +268,18 @@ fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Resu validate_family(family)?; let valid = match family { SummaryFamilyType::Sketch(kind, _) - if matches!(kind.params(), SketchParams::CmsWithHeap { .. }) => + if matches!( + kind.params(), + SketchParams::CmsWithHeap { .. } | SketchParams::CountSketchWithHeap { .. } + ) => { - let SketchParams::CmsWithHeap { - width, - depth, - heap_size, - } = kind.params() - else { - unreachable!() - }; + use crate::summary_operators::weighted_frequency::WeightedFrequency; + let (algorithm, width, depth, capacity) = WeightedFrequency::configuration(kind)?; state .as_any() - .downcast_ref::() + .downcast_ref::() .is_some_and(|state| { - state.shape() == (*width as usize, *depth as usize, *heap_size as usize) + state.algorithm() == algorithm && state.shape() == (width, depth, capacity) }) } diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 63228402..b6ad21d7 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -964,9 +964,14 @@ fn native_relational_join_kinds_preserve_unmatched_rows() { } } -// Per-series fractional rates feed one independent CMS per job, in either scope. +// Per-series fractional rates feed either weighted frequency family per job, in either scope. #[test] fn weighted_rate_topk_preserves_partitions_fractional_scores_and_evaluation_scope() { + for count_sketch in [false, true] { + assert_weighted_rate_topk(count_sketch); + } +} +fn assert_weighted_rate_topk(count_sketch: bool) { use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; let raw = schema(&[ ("service", DataType::Utf8, false), @@ -1007,11 +1012,23 @@ fn weighted_rate_topk_preserves_partitions_fractional_scores_and_evaluation_scop .unwrap(); let family = SummaryFamilyType::Sketch( SketchKind::new( - SketchAlgorithm::CmsWithHeap, - SketchParams::CmsWithHeap { - width: 4096, - depth: 5, - heap_size: 8, + if count_sketch { + SketchAlgorithm::CountSketchWithHeap + } else { + SketchAlgorithm::CmsWithHeap + }, + if count_sketch { + SketchParams::CountSketchWithHeap { + width: 4096, + depth: 5, + heap_size: 8, + } + } else { + SketchParams::CmsWithHeap { + width: 4096, + depth: 5, + heap_size: 8, + } }, ), Default::default(), diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index 8843ce8a..7761161d 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -45,20 +45,28 @@ impl AccuracyEvidenceProvider for Evidence { // The evidence here exercises binding; it is not inferred from the sample data. #[test] fn planner_weighted_topk_binds_at_either_deployment_phase() { - assert_weighted_binding(&Evidence); + assert_weighted_binding(&Evidence, SketchAlgorithm::CmsWithHeap); + assert_weighted_binding(&Evidence, SketchAlgorithm::CountSketchWithHeap); } // Binding validates representation, while deployment owns evidence acceptance. #[test] fn physical_binding_does_not_impose_an_accuracy_acceptance_policy() { - assert_weighted_binding(&asap_aware_mapping::accuracy::NoAccuracyEvidence); + assert_weighted_binding( + &asap_aware_mapping::accuracy::NoAccuracyEvidence, + SketchAlgorithm::CmsWithHeap, + ); + assert_weighted_binding( + &asap_aware_mapping::accuracy::NoAccuracyEvidence, + SketchAlgorithm::CountSketchWithHeap, + ); } -fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider) { +fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: SketchAlgorithm) { let root = Rc::new( lower_promql( "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.01), + AccuracyTarget::Epsilon(0.1), ) .unwrap(), ); @@ -72,12 +80,16 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider) { .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CmsWithHeap") => Some(node), + Replacement::Summary(node) + if candidate.rationale.contains(&format!("{algorithm:?}")) => + { + Some(node) + } _ => None, }) .unwrap(); let dag = compile_executable_dag(&plan).unwrap(); - let build=dag.nodes.iter().find(|node|matches!(&node.payload,ExecutableOperatorPayload::SummaryAgg{family:SummaryFamilyType::Sketch(kind,_),..}if kind.algorithm()==&SketchAlgorithm::CmsWithHeap)).unwrap(); + let build=dag.nodes.iter().find(|node|matches!(&node.payload,ExecutableOperatorPayload::SummaryAgg{family:SummaryFamilyType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); let rate_id = dag .edges .iter() diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index 237c4958..ef1478b6 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -199,24 +199,33 @@ Count outputs Int64. Binary expressions use Planner arithmetic/comparison kinds and enforce its checked-division domains. Grouped TopK composes Sort and Limit within each group. A weighted summary can -consume per-series rates directly: each job has its own CMS and candidate heap, +consume per-series rates directly: each job has its own CMS or CountSketch with a candidate heap, with service as the item and rate as the weight. Sum accumulation inside the summary replaces the exact grouped-sum materialization. Typed readout returns candidate identities and estimated scores; a semi-join is not required for this realization. Both score error and membership require accuracy guarantees. -The `summary_operators` module owns typed summary kernels. Native weighted CMS -uses Float64 counters and preserves typed item identities, including numeric and -NULL keys. It does not use the integer-count codec or fixed-point counter-delta +The `summary_operators` module owns typed summary kernels. Native weighted CMS and CountSketch +use Float64 counters and preserves typed item identities, including numeric and +NULL keys. Neither uses the integer-count codec or fixed-point counter-delta updates. The DAG binder supports column weights and explicit column/tuple item -identities for this CMS path; unsupported families or identities are rejected. +identities for both families; unsupported families or identities are rejected. +CMS accepts nonnegative weights and estimates each score using the minimum row +counter. CountSketch accepts signed weights, uses separate bucket/sign hash seeds, +and takes the median of sign-corrected estimates across a positive odd number of +rows. Candidate heaps rank estimated scores, not absolute magnitudes. A bounded +heap alone does not establish candidate completeness, including after signed +updates or merges. Missing accuracy evidence remains a candidate requirement; +physical binding does not impose deployment's accuracy acceptance policy. +Versioned Float64 states carry their algorithm and dimensions; cross-family or +incompatible-shape merges are rejected, without integer-state compatibility decoding. Summary construction and readout run in either ingestion or query scope, as chosen by deployment. Each run constructs independent partition state; deployment must supply one complete evaluation window, or an equivalent maintained snapshot. The candidate capacity is independent of the grouped Limit's output count. Values retain Planner types and nullability. Native summary batches currently -support exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS with a candidate heap. Stored-summary decoding, delta reconstruction, exact finalization and +support exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS and CountSketch with candidate heaps. Stored-summary decoding, delta reconstruction, exact finalization and family-specific SketchQuery readout also live in this library. Deployment code selects compatible panes and supplies source batches. Stored-state kernels do not imply native batch bindings for every family. Unsupported expressions, From ad60b7ade2049461eb48097bc694641e13335b05 Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 16:23:34 +0000 Subject: [PATCH 14/90] fix: integrate weighted frequency operators with current execution modules --- crates/asap-physical-operators/README.md | 2 +- .../asap-physical-operators/src/capability.rs | 11 ++-- .../src/operators/summary/mod.rs | 53 ++++++------------- .../src/summary_operators/mod.rs | 2 +- crates/asap-physical-operators/src/values.rs | 52 ++++++++++++++++++ docs/design_docs/physical-operators.md | 2 +- 6 files changed, 80 insertions(+), 42 deletions(-) diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index cb9b04d9..a526ee7c 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -58,7 +58,7 @@ operations; it does not interpret an unknown node as external fallback. Plain values preserve Planner scalar/collection types and nullability. Numeric arithmetic uses matching Int64 or Float64 inputs; integer overflow is an error. Boolean predicates use three-valued logic. Native summary states currently cover -exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS with a candidate heap. Binding checks family, +exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS and CountSketch with candidate heaps. Binding checks family, parameters and readout compatibility; source batches also validate state payloads. Existing accumulator algorithms are reused as kernels behind these operators. diff --git a/crates/asap-physical-operators/src/capability.rs b/crates/asap-physical-operators/src/capability.rs index 2bb671a1..eb6459e6 100644 --- a/crates/asap-physical-operators/src/capability.rs +++ b/crates/asap-physical-operators/src/capability.rs @@ -3,7 +3,7 @@ //! `validate_summary_kernel` checks update kernels, including families without a //! native batch representation. `validate_native_family` and //! `validate_native_readout` check native state and scalar readout support. -//! Keyed weighted-CMS readouts are checked by `Operator::keyed_readout`. +//! Keyed weighted-frequency readouts are checked by `Operator::keyed_readout`. //! A successful kernel check alone does not mean an executable DAG will bind. //! //! Persisted state uses `stored_state` decoding and readout contracts; support @@ -143,11 +143,16 @@ pub fn validate_native_family(family: &SummaryFamilyType) -> Result<(), Error> { use planner_types::post_asap::SketchAlgorithm as A; if let SummaryFamilyType::Sketch(kind, grouping) = family { if matches!(kind.algorithm(), A::CmsWithHeap | A::CountSketchWithHeap) { - let (_, width, depth, _) = crate::summary_operators::weighted_frequency::WeightedFrequency::configuration(kind)?; + let (_, width, depth, _) = + crate::summary_operators::weighted_frequency::WeightedFrequency::configuration( + kind, + )?; return if valid_matrix(width as u32, depth as u32) && grouping == &Default::default() { Ok(()) } else { - Err(Error::Invalid("invalid weighted frequency dimensions or grouping strategy".into())) + Err(Error::Invalid( + "invalid weighted frequency dimensions or grouping strategy".into(), + )) }; } } diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs index 2024e2af..7af3ab0e 100644 --- a/crates/asap-physical-operators/src/operators/summary/mod.rs +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -7,18 +7,12 @@ impl Operator { items: Vec, groups: Vec, ) -> Result { - use planner_types::post_asap::{SketchAlgorithm, SketchParams}; + use crate::summary_operators::weighted_frequency::WeightedFrequency; crate::values::validate_family(&family)?; let SummaryFamilyType::Sketch(kind, _) = &family else { return Err(invalid("keyed sketch required")); }; - if kind.algorithm() != &SketchAlgorithm::CmsWithHeap - || !matches!(kind.params(), SketchParams::CmsWithHeap { .. }) - { - return Err(invalid( - "Float64 weighted keyed construction currently supports CMS with heap", - )); - } + WeightedFrequency::configuration(kind)?; validate_groups(&input, &groups)?; if items.is_empty() || plain(&input, value)? != (&DataType::Float64, false) { return Err(invalid( @@ -63,18 +57,13 @@ impl Operator { k: usize, output: Schema, ) -> Result { - use planner_types::post_asap::{SketchAlgorithm, SketchParams}; - crate::capability::validate_native_family(&field(&input, state)?.dtype)?; + use crate::summary_operators::weighted_frequency::WeightedFrequency; + crate::values::validate_family(&field(&input, state)?.dtype)?; let SummaryFamilyType::Sketch(kind, _) = &field(&input, state)?.dtype else { return Err(invalid("keyed readout requires summary state")); }; - let SketchParams::CmsWithHeap { heap_size, .. } = kind.params() else { - return Err(invalid("unsupported keyed readout family")); - }; - if kind.algorithm() != &SketchAlgorithm::CmsWithHeap - || k > *heap_size as usize - || output.fields.len() <= input.fields.len() - { + let (_, _, _, capacity) = WeightedFrequency::configuration(kind)?; + if k > capacity || output.fields.len() <= input.fields.len() { return Err(invalid("invalid keyed readout shape or capacity")); } if state + 1 != input.fields.len() @@ -92,7 +81,6 @@ impl Operator { output, }) } - pub fn summary_build( input: Schema, family: SummaryFamilyType, @@ -243,8 +231,8 @@ pub(super) fn execute<'a>( }; let summary = summary .as_any() - .downcast_ref::() - .ok_or_else(|| invalid("weighted CMS typed state required"))?; + .downcast_ref::() + .ok_or_else(|| invalid("weighted frequency typed state required"))?; for items in summary.rows(*k) { let mut values = row[..*state].to_vec(); values.extend(items); @@ -473,21 +461,14 @@ async fn build_keyed_summary( groups: &[usize], context: &RunContext, ) -> Result>, Error> { - use crate::{summary_operators::weighted_cms::WeightedCms, AggregateCore}; - use planner_types::post_asap::SketchParams; + use crate::{summary_operators::weighted_frequency::WeightedFrequency, AggregateCore}; let SummaryFamilyType::Sketch(kind, _) = family else { unreachable!() }; - let SketchParams::CmsWithHeap { - width, - depth, - heap_size, - } = kind.params() - else { - unreachable!() - }; + let (algorithm, width, depth, capacity) = WeightedFrequency::configuration(kind)?; let mut work = Cooperative::new(context); - let mut states = BTreeMap::>, (Vec, WeightedCms, Reservation, usize)>::new(); + let mut states = + BTreeMap::>, (Vec, WeightedFrequency, Reservation, usize)>::new(); while let Some(batch) = input.next().await { let batch = batch?; for row in batch.rows() { @@ -498,17 +479,17 @@ async fn build_keyed_summary( let overhead = labels.iter().map(Value::bytes).sum::() + key.iter().map(|v| v.len() + 24).sum::() + 128; - let bytes = (*width as usize) - .checked_mul(*depth as usize) + let bytes = width + .checked_mul(depth) .and_then(|n| n.checked_mul(8)) .and_then(|n| n.checked_add(overhead)) - .ok_or_else(|| invalid("weighted CMS memory size overflow"))?; + .ok_or_else(|| invalid("weighted frequency memory size overflow"))?; let reservation = context.reserve(bytes)?; states.insert( key.clone(), ( labels, - WeightedCms::new(*width as usize, *depth as usize, *heap_size as usize)?, + WeightedFrequency::new(algorithm, width, depth, capacity)?, reservation, overhead, ), @@ -516,7 +497,7 @@ async fn build_keyed_summary( } let (_, summary, reservation, overhead) = states.get_mut(&key).unwrap(); let Value::Float64(weight) = row[value] else { - return Err(invalid("weighted CMS weight type")); + return Err(invalid("weighted frequency weight type")); }; summary.update( &items.iter().map(|&i| row[i].clone()).collect::>(), diff --git a/crates/asap-physical-operators/src/summary_operators/mod.rs b/crates/asap-physical-operators/src/summary_operators/mod.rs index 1dc0d247..a462382a 100644 --- a/crates/asap-physical-operators/src/summary_operators/mod.rs +++ b/crates/asap-physical-operators/src/summary_operators/mod.rs @@ -36,6 +36,6 @@ pub use min_accumulator::*; pub use sketch_envelope_accumulator::*; pub use sum_accumulator::*; -pub mod weighted_frequency; pub mod factory; pub mod traits; +pub mod weighted_frequency; diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs index ea019209..eb4dfc5a 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -361,3 +361,55 @@ pub(crate) fn plain(schema: &Schema, column: usize) -> Result<(&DataType, bool), }; Ok((dtype, f.nullable)) } + +#[cfg(test)] +mod weighted_state_tests { + use super::*; + use crate::summary_operators::weighted_frequency::{FrequencyAlgorithm, WeightedFrequency}; + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + + // A state cannot acquire a different family or shape merely by relabeling its batch. + #[test] + fn weighted_state_family_and_shape_must_match() { + let cms = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 32, + depth: 5, + heap_size: 8, + }, + ), + Default::default(), + ); + let cs = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 32, + depth: 5, + heap_size: 8, + }, + ), + Default::default(), + ); + let state = WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 32, 5, 8).unwrap(); + assert!(validate_state(&cs, &state).is_ok()); + assert!(validate_state(&cms, &state).is_err()); + let wrong_shape = + WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 64, 5, 8).unwrap(); + assert!(validate_state(&cs, &wrong_shape).is_err()); + let even_depth = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 32, + depth: 4, + heap_size: 8, + }, + ), + Default::default(), + ); + assert!(validate_family(&even_depth).is_err()); + } +} diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index ef1478b6..78b4020f 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -243,7 +243,7 @@ the library contract; backend raw-data access still requires a deployment connec | Level | Acceptance contract | Scope | | --- | --- | --- | | Update kernel | `capability::validate_summary_kernel` | Family, parameters, grouping and item/update layout; used by the accumulator factory | -| Native state edge | `capability::validate_native_family` | Exact accumulators, KLL, DDSketch, HLL and Float64 weighted CMS with compatible parameters | +| Native state edge | `capability::validate_native_family` | Exact accumulators, KLL, DDSketch, HLL and Float64 weighted CMS and CountSketch with compatible parameters | | Scalar native readout | `capability::validate_native_readout` | Supported native state plus statistic/readout arguments | | Keyed native readout | `Operator::keyed_readout` | Weighted CMS family, heap capacity, typed identity/score schema and preserved partition columns | | Complete physical plan | `binding::bind` / `bind_with_data_sources` | Node support, expressions, schemas, source frontiers and bounded input requirements | From bc30464479de23f9678ba5bed3ec8671c6d14e22 Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 16:33:18 +0000 Subject: [PATCH 15/90] refactor: remove accumulator suffix from summary operator modules --- .../src/stored_state/decoders.rs | 14 ++--- .../src/stored_state/delta_apply.rs | 23 ++++--- ...tch_accumulator.rs => count_min_sketch.rs} | 4 +- ...lator.rs => count_min_sketch_with_heap.rs} | 0 ..._sketch_accumulator.rs => count_sketch.rs} | 4 +- ...cumulator.rs => count_sketch_with_heap.rs} | 6 +- ...kll_accumulator.rs => datasketches_kll.rs} | 2 +- ...{dd_sketch_accumulator.rs => dd_sketch.rs} | 2 +- .../{exact_accumulator.rs => exact.rs} | 2 +- .../src/summary_operators/factory.rs | 10 ++-- ...ll_sketch_accumulator.rs => hll_sketch.rs} | 4 +- ...{hydra_kll_accumulator.rs => hydra_kll.rs} | 0 .../{increase_accumulator.rs => increase.rs} | 0 ...ount_accumulator.rs => keyed_sum_count.rs} | 0 .../{max_accumulator.rs => max.rs} | 4 +- .../{min_accumulator.rs => min.rs} | 4 +- .../src/summary_operators/mod.rs | 60 +++++++++---------- ...lope_accumulator.rs => sketch_envelope.rs} | 0 .../{sum_accumulator.rs => sum.rs} | 0 .../{univmon_accumulator.rs => univmon.rs} | 0 crates/asap-physical-operators/src/values.rs | 5 +- .../tests/physical_dag.rs | 4 +- docs/design_docs/physical-operators.md | 4 +- 23 files changed, 74 insertions(+), 78 deletions(-) rename crates/asap-physical-operators/src/summary_operators/{count_min_sketch_accumulator.rs => count_min_sketch.rs} (99%) rename crates/asap-physical-operators/src/summary_operators/{count_min_sketch_with_heap_accumulator.rs => count_min_sketch_with_heap.rs} (100%) rename crates/asap-physical-operators/src/summary_operators/{count_sketch_accumulator.rs => count_sketch.rs} (99%) rename crates/asap-physical-operators/src/summary_operators/{count_sketch_with_heap_accumulator.rs => count_sketch_with_heap.rs} (99%) rename crates/asap-physical-operators/src/summary_operators/{datasketches_kll_accumulator.rs => datasketches_kll.rs} (99%) rename crates/asap-physical-operators/src/summary_operators/{dd_sketch_accumulator.rs => dd_sketch.rs} (99%) rename crates/asap-physical-operators/src/summary_operators/{exact_accumulator.rs => exact.rs} (99%) rename crates/asap-physical-operators/src/summary_operators/{hll_sketch_accumulator.rs => hll_sketch.rs} (99%) rename crates/asap-physical-operators/src/summary_operators/{hydra_kll_accumulator.rs => hydra_kll.rs} (100%) rename crates/asap-physical-operators/src/summary_operators/{increase_accumulator.rs => increase.rs} (100%) rename crates/asap-physical-operators/src/summary_operators/{keyed_sum_count_accumulator.rs => keyed_sum_count.rs} (100%) rename crates/asap-physical-operators/src/summary_operators/{max_accumulator.rs => max.rs} (98%) rename crates/asap-physical-operators/src/summary_operators/{min_accumulator.rs => min.rs} (98%) rename crates/asap-physical-operators/src/summary_operators/{sketch_envelope_accumulator.rs => sketch_envelope.rs} (100%) rename crates/asap-physical-operators/src/summary_operators/{sum_accumulator.rs => sum.rs} (100%) rename crates/asap-physical-operators/src/summary_operators/{univmon_accumulator.rs => univmon.rs} (100%) diff --git a/crates/asap-physical-operators/src/stored_state/decoders.rs b/crates/asap-physical-operators/src/stored_state/decoders.rs index 3706c9c3..8741d537 100644 --- a/crates/asap-physical-operators/src/stored_state/decoders.rs +++ b/crates/asap-physical-operators/src/stored_state/decoders.rs @@ -8,13 +8,13 @@ use asap_sketchlib::CountSketchWithHeap; use asap_sketchlib::CsHeapItem; use asap_sketchlib::MessagePackCodec; -use crate::summary_operators::count_min_sketch_with_heap_accumulator::CountMinSketchWithHeapAccumulator; +use crate::summary_operators::count_min_sketch_with_heap::CountMinSketchWithHeapAccumulator; /// Decode a `CountMinSketch` from the modified-OTLP wire bytes. /// MSGPACK path round-trips `CountMinSketch::deserialize_msgpack`; /// PROTO path decodes a `SketchEnvelope{count_min: CountMinState}` /// (or bare `CountMinState`) and re-projects to a flat matrix. Mirrors -/// `precompute_operators::count_min_sketch_accumulator::from_sketchlib_proto_bytes`. +/// `precompute_operators::count_min_sketch::from_sketchlib_proto_bytes`. pub fn decode_cms_from_proto(buffer: &[u8]) -> Result { use asap_sketchlib::proto::sketchlib::{ sketch_envelope, CountMinState, CounterType, SketchEnvelope, @@ -88,7 +88,7 @@ pub fn decode_cms_from_msgpack(buffer: &[u8]) -> Result /// Decode a `CountSketch` from the modified-OTLP proto wire bytes. /// Mirrors -/// `precompute_operators::count_sketch_accumulator::from_sketchlib_proto_bytes`. +/// `precompute_operators::count_sketch::from_sketchlib_proto_bytes`. pub fn decode_cs_from_proto(buffer: &[u8]) -> Result { use asap_sketchlib::proto::sketchlib::{ sketch_envelope, CountSketchState, CounterType, SketchEnvelope, @@ -195,13 +195,13 @@ pub fn decode_cs_with_heap_from_msgpack(buffer: &[u8]) -> Result Result { use asap_sketchlib::proto::sketchlib::CountMinDelta as PbDelta; use prost::Message; @@ -249,7 +249,7 @@ pub fn decode_cms_from_proto_delta(buffer: &[u8]) -> Result Result { use asap_sketchlib::proto::sketchlib::CountSketchDelta as PbDelta; use prost::Message; @@ -310,7 +310,7 @@ pub fn decode_cms_with_heap_from_msgpack_delta( /// matrix delta + full heap onto an empty base of the frame's declared /// dimensions. Same DELTA-HEAP wire shape as the CmsWithHeap delta frame /// (see `HeapDeltaWire`/`MatrixDeltaWire` in -/// `count_min_sketch_with_heap_accumulator.rs`), decoded here directly +/// `count_min_sketch_with_heap.rs`), decoded here directly /// with `rmp_serde` since there is no CountSketchWithHeap ingest /// accumulator to delegate to. No `asap_sketchlib` delta API needed — the /// public `from_legacy_matrix` rebuilds both the matrix and heap. diff --git a/crates/asap-physical-operators/src/stored_state/delta_apply.rs b/crates/asap-physical-operators/src/stored_state/delta_apply.rs index 03467ab3..a5c4a836 100644 --- a/crates/asap-physical-operators/src/stored_state/delta_apply.rs +++ b/crates/asap-physical-operators/src/stored_state/delta_apply.rs @@ -84,7 +84,7 @@ impl DeltaSketchKind { sketch_cols, layers, } => SummaryState::UnivMon( - crate::summary_operators::univmon_accumulator::UnivMonAccumulator::new( + crate::summary_operators::univmon::UnivMonAccumulator::new( *heap_size as usize, *sketch_rows as usize, *sketch_cols as usize, @@ -136,10 +136,7 @@ fn decode_full( }, SketchEncoding::MsgpackFull, ) => { - let state = - crate::summary_operators::univmon_accumulator::UnivMonAccumulator::from_bytes( - bytes, - ) + let state = crate::summary_operators::univmon::UnivMonAccumulator::from_bytes(bytes) .map_err(|e| e.to_string())?; if state.dimensions() != ( @@ -216,7 +213,7 @@ fn decode_full( /// folded across a window (or several) via delta application, or merged /// in from another sid's own reconstruction. pub enum SummaryState { - UnivMon(crate::summary_operators::univmon_accumulator::UnivMonAccumulator), + UnivMon(crate::summary_operators::univmon::UnivMonAccumulator), Dd(DdSketch), Hll(HllSketch), Kll(KllSketch), @@ -293,7 +290,7 @@ impl SummaryState { } // Shape (2): bucket-delta proto → additive apply via the // SAME decoder the ingest delta path uses. - use crate::summary_operators::dd_sketch_accumulator::DDSketchAccumulator; + use crate::summary_operators::dd_sketch::DDSketchAccumulator; let mut acc = DDSketchAccumulator { inner: std::mem::replace(sk, DdSketch::new(sk.alpha)), sample_p: 1.0, @@ -712,21 +709,21 @@ pub fn per_window_summary_states( // --------------------------------------------------------------------------- fn dd_from_proto(buffer: &[u8]) -> Result { - use crate::summary_operators::dd_sketch_accumulator::DDSketchAccumulator; + use crate::summary_operators::dd_sketch::DDSketchAccumulator; DDSketchAccumulator::from_sketchlib_proto_bytes(buffer) .map(|acc| acc.inner) .map_err(|e| e.to_string()) } fn kll_from_proto(buffer: &[u8]) -> Result { - use crate::summary_operators::datasketches_kll_accumulator::DatasketchesKLLAccumulator; + use crate::summary_operators::datasketches_kll::DatasketchesKLLAccumulator; DatasketchesKLLAccumulator::from_sketchlib_proto_bytes(buffer) .map(|acc| acc.inner) .map_err(|e| e.to_string()) } fn hll_from_proto(buffer: &[u8]) -> Result { - use crate::summary_operators::hll_sketch_accumulator::HllSketchAccumulator; + use crate::summary_operators::hll_sketch::HllSketchAccumulator; HllSketchAccumulator::from_sketchlib_proto_bytes(buffer) .map(|acc| acc.inner) .map_err(|e| e.to_string()) @@ -882,7 +879,7 @@ mod tests { fn hll_from_proto_matches_accumulator_decoder() { // P2-4: the warm read path and the ingest accumulator must decode // the SAME bytes to the SAME sketch (one source of truth). - use crate::summary_operators::hll_sketch_accumulator::HllSketchAccumulator; + use crate::summary_operators::hll_sketch::HllSketchAccumulator; let mut sk = HllSketch::new(HllVariant::Regular, 12); for i in 0..500u64 { sk.update(format!("item-{i}").as_bytes()); @@ -901,7 +898,7 @@ mod tests { #[test] fn dd_from_proto_matches_accumulator_decoder() { - use crate::summary_operators::dd_sketch_accumulator::DDSketchAccumulator; + use crate::summary_operators::dd_sketch::DDSketchAccumulator; let mut sk = DdSketch::new(0.01); for v in [1.0, 2.0, 5.0, 5.0, 9.0, 42.0] { sk.update(v); @@ -918,7 +915,7 @@ mod tests { #[test] fn kll_from_proto_matches_accumulator_decoder() { - use crate::summary_operators::datasketches_kll_accumulator::DatasketchesKLLAccumulator; + use crate::summary_operators::datasketches_kll::DatasketchesKLLAccumulator; let items: Vec = (0..200).map(|i| i as f64).collect(); let bytes = encode_kll(256, &items); let via_delta = kll_from_proto(&bytes).expect("delta_apply kll decode"); diff --git a/crates/asap-physical-operators/src/summary_operators/count_min_sketch_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/count_min_sketch.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/count_min_sketch_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/count_min_sketch.rs index fe63e3f3..26a6ae96 100644 --- a/crates/asap-physical-operators/src/summary_operators/count_min_sketch_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/count_min_sketch.rs @@ -1,4 +1,4 @@ -use crate::summary_operators::dd_sketch_accumulator::normalize_sample_p; +use crate::summary_operators::dd_sketch::normalize_sample_p; use crate::{ AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, MultipleSubpopulationAggregate, SerializableToSink, @@ -810,7 +810,7 @@ mod tests { let boxed_accs: Vec> = vec![Box::new(cms1), Box::new(cms2)]; assert!(CountMinSketchAccumulator::merge_multiple(&boxed_accs).is_err()); - use crate::summary_operators::sum_accumulator::SumAccumulator; + use crate::summary_operators::sum::SumAccumulator; let cms = CountMinSketchAccumulator::new(2, 3); let sum = SumAccumulator::new(); let mixed_accs: Vec> = vec![Box::new(cms), Box::new(sum)]; diff --git a/crates/asap-physical-operators/src/summary_operators/count_min_sketch_with_heap_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/count_min_sketch_with_heap.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/count_min_sketch_with_heap_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/count_min_sketch_with_heap.rs diff --git a/crates/asap-physical-operators/src/summary_operators/count_sketch_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/count_sketch.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/count_sketch_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/count_sketch.rs index 78cdb61b..9cfb4b24 100644 --- a/crates/asap-physical-operators/src/summary_operators/count_sketch_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/count_sketch.rs @@ -96,7 +96,7 @@ impl CountSketchAccumulator { // ingest caller skips the data point) instead of building a // degenerate or huge matrix. Shares the CMS validator since the // CountSketch matrix uses the same packed-hash column layout. - crate::summary_operators::count_min_sketch_accumulator::validate_sketch_dims( + crate::summary_operators::count_min_sketch::validate_sketch_dims( "CountSketchState", rows, cols, @@ -554,7 +554,7 @@ mod tests { #[test] fn test_aggregate_core_merge_wrong_type_rejects() { - use crate::summary_operators::count_min_sketch_accumulator::CountMinSketchAccumulator; + use crate::summary_operators::count_min_sketch::CountMinSketchAccumulator; let cs = CountSketchAccumulator::new(2, 3); let cms = CountMinSketchAccumulator::new(2, 3); let result = cs.merge_with(&cms); diff --git a/crates/asap-physical-operators/src/summary_operators/count_sketch_with_heap_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/count_sketch_with_heap.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/count_sketch_with_heap_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/count_sketch_with_heap.rs index 2a079347..afb5e58b 100644 --- a/crates/asap-physical-operators/src/summary_operators/count_sketch_with_heap_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/count_sketch_with_heap.rs @@ -1,7 +1,7 @@ //! Count Sketch with Heap accumulator — wraps //! `asap_sketchlib::CountSketchWithHeap`. //! -//! Port of `count_min_sketch_with_heap_accumulator.rs` for the distinct +//! Port of `count_min_sketch_with_heap.rs` for the distinct //! `CountSketchWithHeap` (median-of-signed-rows estimator) rather than //! `CountMinSketchWithHeap` (min-over-rows estimator). The two are //! different sketch algorithms that happen to share a storage shape and @@ -24,7 +24,7 @@ use std::collections::HashMap; use crate::Statistic; /// Local serde view of the DELTA-HEAP wire frame (encoding `MSGPACK_DELTA`). -/// Identical shape to `count_min_sketch_with_heap_accumulator.rs`'s +/// Identical shape to `count_min_sketch_with_heap.rs`'s /// `HeapDeltaWire`/`MatrixDeltaWire` -- the wire frame is generic (sparse /// cell deltas + a full heap), not CMS-specific. See that file's doc for /// the exact rmp_serde positional layout. @@ -561,7 +561,7 @@ mod tests { /// min-over-rows divergence at the sketch-math level). #[test] fn test_rejects_merge_with_cms_family_accumulator() { - use crate::summary_operators::count_min_sketch_with_heap_accumulator::CountMinSketchWithHeapAccumulator; + use crate::summary_operators::count_min_sketch_with_heap::CountMinSketchWithHeapAccumulator; let cs = CountSketchWithHeapAccumulator::new(4, 64, 10); let cms = CountMinSketchWithHeapAccumulator::new(4, 64, 10); diff --git a/crates/asap-physical-operators/src/summary_operators/datasketches_kll_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/datasketches_kll.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/datasketches_kll_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/datasketches_kll.rs index 4124ddf9..c031ddb1 100644 --- a/crates/asap-physical-operators/src/summary_operators/datasketches_kll_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/datasketches_kll.rs @@ -537,7 +537,7 @@ mod tests { let boxed_accs: Vec> = vec![Box::new(kll1), Box::new(kll2)]; assert!(DatasketchesKLLAccumulator::merge_multiple(&boxed_accs).is_err()); - use crate::summary_operators::sum_accumulator::SumAccumulator; + use crate::summary_operators::sum::SumAccumulator; let kll = DatasketchesKLLAccumulator::new(200); let sum = SumAccumulator::new(); let mixed_accs: Vec> = vec![Box::new(kll), Box::new(sum)]; diff --git a/crates/asap-physical-operators/src/summary_operators/dd_sketch_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/dd_sketch.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/dd_sketch_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/dd_sketch.rs index 1e926d68..3d55787b 100644 --- a/crates/asap-physical-operators/src/summary_operators/dd_sketch_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/dd_sketch.rs @@ -398,7 +398,7 @@ mod tests { #[test] fn test_aggregate_core_merge_wrong_type_rejects() { - use crate::summary_operators::count_sketch_accumulator::CountSketchAccumulator; + use crate::summary_operators::count_sketch::CountSketchAccumulator; let dd = DDSketchAccumulator::new(0.01); let cs = CountSketchAccumulator::new(2, 3); assert!(dd.merge_with(&cs).is_err()); diff --git a/crates/asap-physical-operators/src/summary_operators/exact_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/exact.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/exact_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/exact.rs index c466548e..058cb5b7 100644 --- a/crates/asap-physical-operators/src/summary_operators/exact_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/exact.rs @@ -1,5 +1,5 @@ //! Exact summary state identified by Planner family, independent of keyed layout. -use super::increase_accumulator::IncreaseAccumulator; +use super::increase::IncreaseAccumulator; use crate::Statistic; use crate::{ AggregateCore, AggregationType, AuxStats, KeyByLabelValues, Measurement, SerializableToSink, diff --git a/crates/asap-physical-operators/src/summary_operators/factory.rs b/crates/asap-physical-operators/src/summary_operators/factory.rs index 30f74971..8ea40b85 100644 --- a/crates/asap-physical-operators/src/summary_operators/factory.rs +++ b/crates/asap-physical-operators/src/summary_operators/factory.rs @@ -7,8 +7,8 @@ use crate::summary_operators::{ use crate::{AggregateCore, KeyByLabelValues, Measurement}; // Production dispatch consumes Planner SummaryAgg payloads directly. The // config adapter below is compiled only for isolated historical kernel tests. -use crate::summary_operators::hll_sketch_accumulator::HllSketchAccumulator; -use crate::summary_operators::univmon_accumulator::UnivMonAccumulator; +use crate::summary_operators::hll_sketch::HllSketchAccumulator; +use crate::summary_operators::univmon::UnivMonAccumulator; use planner_types::post_asap::{ExactKind, SketchAlgorithm, SketchParams, SummaryFamilyType}; /// Generate the two boilerplate clone-based `AccumulatorUpdater` methods @@ -913,7 +913,7 @@ pub fn create_planner_accumulator( } if matches!(family, SummaryFamilyType::ExactAggregate(..)) { return Ok(Box::new(PlannerExactUpdater { - acc: crate::summary_operators::exact_accumulator::ExactAccumulator::new( + acc: crate::summary_operators::exact::ExactAccumulator::new( family.clone(), input.item.is_some(), )?, @@ -994,7 +994,7 @@ pub fn create_planner_accumulator( } struct PlannerExactUpdater { - acc: crate::summary_operators::exact_accumulator::ExactAccumulator, + acc: crate::summary_operators::exact::ExactAccumulator, } impl AccumulatorUpdater for PlannerExactUpdater { fn update_single(&mut self, value: f64, timestamp: i64) { @@ -1005,7 +1005,7 @@ impl AccumulatorUpdater for PlannerExactUpdater { } impl_clone_accumulator_methods!(acc); fn reset(&mut self) { - self.acc = crate::summary_operators::exact_accumulator::ExactAccumulator::new( + self.acc = crate::summary_operators::exact::ExactAccumulator::new( self.acc.family().clone(), self.acc.is_keyed(), ) diff --git a/crates/asap-physical-operators/src/summary_operators/hll_sketch_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/hll_sketch.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/hll_sketch_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/hll_sketch.rs index 0d8b714d..efe15baf 100644 --- a/crates/asap-physical-operators/src/summary_operators/hll_sketch_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/hll_sketch.rs @@ -11,7 +11,7 @@ //! registers + variant + HIP accumulators losslessly, so the merge + //! store round-trip works end-to-end without that richer query surface. -use crate::summary_operators::dd_sketch_accumulator::normalize_sample_p; +use crate::summary_operators::dd_sketch::normalize_sample_p; use crate::{AggregateCore, AggregationType, KeyByLabelValues, SerializableToSink}; use asap_sketchlib::{HllSketch, HllVariant, MessagePackCodec}; use serde_json::Value; @@ -567,7 +567,7 @@ mod tests { #[test] fn test_aggregate_core_merge_wrong_type_rejects() { - use crate::summary_operators::count_sketch_accumulator::CountSketchAccumulator; + use crate::summary_operators::count_sketch::CountSketchAccumulator; let hll = HllSketchAccumulator::new(HllVariant::Regular, 2); let cs = CountSketchAccumulator::new(2, 3); assert!(hll.merge_with(&cs).is_err()); diff --git a/crates/asap-physical-operators/src/summary_operators/hydra_kll_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/hydra_kll.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/hydra_kll_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/hydra_kll.rs diff --git a/crates/asap-physical-operators/src/summary_operators/increase_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/increase.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/increase_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/increase.rs diff --git a/crates/asap-physical-operators/src/summary_operators/keyed_sum_count_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/keyed_sum_count.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/keyed_sum_count_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/keyed_sum_count.rs diff --git a/crates/asap-physical-operators/src/summary_operators/max_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/max.rs similarity index 98% rename from crates/asap-physical-operators/src/summary_operators/max_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/max.rs index 8f6e2af7..0c954708 100644 --- a/crates/asap-physical-operators/src/summary_operators/max_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/max.rs @@ -10,7 +10,7 @@ use crate::Statistic; /// Exact maximum over one population, mergeable by comparison. /// -/// See [`MinAccumulator`](super::min_accumulator::MinAccumulator) for why the +/// See [`MinAccumulator`](super::min::MinAccumulator) for why the /// two directions are separate types rather than one accumulator carrying a /// `sub_type` string. #[derive(Debug, Clone, Serialize, Deserialize)] @@ -212,7 +212,7 @@ mod tests { #[test] fn refuses_to_merge_with_a_minimum() { - use super::super::min_accumulator::MinAccumulator; + use super::super::min::MinAccumulator; let max = MaxAccumulator::with_value(15.0); let min = MinAccumulator::with_value(5.0); assert!(max.merge_with(&min).is_err()); diff --git a/crates/asap-physical-operators/src/summary_operators/min_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/min.rs similarity index 98% rename from crates/asap-physical-operators/src/summary_operators/min_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/min.rs index 79343858..ff2ad548 100644 --- a/crates/asap-physical-operators/src/summary_operators/min_accumulator.rs +++ b/crates/asap-physical-operators/src/summary_operators/min.rs @@ -10,7 +10,7 @@ use crate::Statistic; /// Exact minimum over one population, mergeable by comparison. /// -/// The sibling [`MaxAccumulator`](super::max_accumulator::MaxAccumulator) is a +/// The sibling [`MaxAccumulator`](super::max::MaxAccumulator) is a /// separate type on purpose: these two used to be one `MinMaxAccumulator` /// whose direction lived in a `sub_type: String`, which meant every layer /// above -- the wire `aggregationSubType`, the accumulator factory, the @@ -215,7 +215,7 @@ mod tests { #[test] fn refuses_to_merge_with_a_maximum() { - use super::super::max_accumulator::MaxAccumulator; + use super::super::max::MaxAccumulator; let min = MinAccumulator::with_value(5.0); let max = MaxAccumulator::with_value(15.0); assert!(min.merge_with(&max).is_err()); diff --git a/crates/asap-physical-operators/src/summary_operators/mod.rs b/crates/asap-physical-operators/src/summary_operators/mod.rs index a462382a..2e35ed6f 100644 --- a/crates/asap-physical-operators/src/summary_operators/mod.rs +++ b/crates/asap-physical-operators/src/summary_operators/mod.rs @@ -1,40 +1,40 @@ -pub mod count_min_sketch_accumulator; -pub mod count_min_sketch_with_heap_accumulator; -pub mod count_sketch_accumulator; -pub mod count_sketch_with_heap_accumulator; -pub mod datasketches_kll_accumulator; -pub mod dd_sketch_accumulator; -pub mod exact_accumulator; -pub mod hll_sketch_accumulator; -pub mod hydra_kll_accumulator; -pub mod increase_accumulator; +pub mod count_min_sketch; +pub mod count_min_sketch_with_heap; +pub mod count_sketch; +pub mod count_sketch_with_heap; +pub mod datasketches_kll; +pub mod dd_sketch; +pub mod exact; +pub mod hll_sketch; +pub mod hydra_kll; +pub mod increase; pub mod keyed_counter_state; pub mod keyed_max_state; pub mod keyed_min_state; -pub mod keyed_sum_count_accumulator; -pub mod max_accumulator; -pub mod min_accumulator; -pub mod sketch_envelope_accumulator; -pub mod sum_accumulator; -pub mod univmon_accumulator; +pub mod keyed_sum_count; +pub mod max; +pub mod min; +pub mod sketch_envelope; +pub mod sum; +pub mod univmon; -pub use count_min_sketch_accumulator::*; -pub use count_min_sketch_with_heap_accumulator::*; -pub use count_sketch_accumulator::*; -pub use count_sketch_with_heap_accumulator::*; -pub use datasketches_kll_accumulator::*; -pub use dd_sketch_accumulator::*; -pub use hll_sketch_accumulator::*; -pub use hydra_kll_accumulator::*; -pub use increase_accumulator::*; +pub use count_min_sketch::*; +pub use count_min_sketch_with_heap::*; +pub use count_sketch::*; +pub use count_sketch_with_heap::*; +pub use datasketches_kll::*; +pub use dd_sketch::*; +pub use hll_sketch::*; +pub use hydra_kll::*; +pub use increase::*; pub use keyed_counter_state::*; pub use keyed_max_state::*; pub use keyed_min_state::*; -pub use keyed_sum_count_accumulator::*; -pub use max_accumulator::*; -pub use min_accumulator::*; -pub use sketch_envelope_accumulator::*; -pub use sum_accumulator::*; +pub use keyed_sum_count::*; +pub use max::*; +pub use min::*; +pub use sketch_envelope::*; +pub use sum::*; pub mod factory; pub mod traits; diff --git a/crates/asap-physical-operators/src/summary_operators/sketch_envelope_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/sketch_envelope.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/sketch_envelope_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/sketch_envelope.rs diff --git a/crates/asap-physical-operators/src/summary_operators/sum_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/sum.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/sum_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/sum.rs diff --git a/crates/asap-physical-operators/src/summary_operators/univmon_accumulator.rs b/crates/asap-physical-operators/src/summary_operators/univmon.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/univmon_accumulator.rs rename to crates/asap-physical-operators/src/summary_operators/univmon.rs diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs index eb4dfc5a..cccfb7ad 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -260,9 +260,8 @@ pub(crate) use crate::capability::validate_native_family as validate_family; fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Result<(), Error> { use crate::summary_operators::{ - datasketches_kll_accumulator::DatasketchesKLLAccumulator, - dd_sketch_accumulator::DDSketchAccumulator, exact_accumulator::ExactAccumulator, - hll_sketch_accumulator::HllSketchAccumulator, + datasketches_kll::DatasketchesKLLAccumulator, dd_sketch::DDSketchAccumulator, + exact::ExactAccumulator, hll_sketch::HllSketchAccumulator, }; use planner_types::post_asap::SketchParams; validate_family(family)?; diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index b6ad21d7..79aae966 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -398,9 +398,7 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { // Restored state must retain its family; a mislabeled state is rejected. #[test] fn restored_exact_state_and_family_validation() { - use asap_physical_operators::{ - summary_operators::exact_accumulator::ExactAccumulator, SerializableToSink, - }; + use asap_physical_operators::{summary_operators::exact::ExactAccumulator, SerializableToSink}; let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); acc.update(None, 7., 0); diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index 78b4020f..ca678fc2 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -205,7 +205,9 @@ summary replaces the exact grouped-sum materialization. Typed readout returns candidate identities and estimated scores; a semi-join is not required for this realization. Both score error and membership require accuracy guarantees. -The `summary_operators` module owns typed summary kernels. Native weighted CMS and CountSketch +The `summary_operators` module owns typed summary kernels. Its modules use operation +names, such as `count_min_sketch`, `exact` and `sum`, without an +`_accumulator` filename suffix or old-path aliases. Native weighted CMS and CountSketch use Float64 counters and preserves typed item identities, including numeric and NULL keys. Neither uses the integer-count codec or fixed-point counter-delta updates. The DAG binder supports column weights and explicit column/tuple item From bfb4b0736de10aaa683efe16292e7990330963c4 Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 16:40:25 +0000 Subject: [PATCH 16/90] docs: condense physical execution design and choose native implementation --- .../datafusion-execution-comparison.md | 381 ++++------------- docs/design_docs/physical-operators.md | 397 +++++------------- 2 files changed, 198 insertions(+), 580 deletions(-) diff --git a/docs/design_docs/datafusion-execution-comparison.md b/docs/design_docs/datafusion-execution-comparison.md index f1eb55a9..b388fa16 100644 --- a/docs/design_docs/datafusion-execution-comparison.md +++ b/docs/design_docs/datafusion-execution-comparison.md @@ -1,304 +1,89 @@ -# DataFusion and native ASAP execution: architecture comparison +# DataFusion versus native ASAP execution -Audience: designers and maintainers of the Planner and execution engines. +## Decision for #462 -This survey evaluates the execution choice in PR #462. It does not replace -ASAP's summary planning semantics or introduce a DataFusion dependency. Sources -were inspected on 2026-09-24 at DataFusion commit -[`e2ca7f3`](https://github.com/apache/datafusion/tree/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38). -The native baseline is PR #462 at `d782c4e`; its subsequent module/resource -refactor improves contracts but does not add partitioned execution or spill. -Performance statements below are hypotheses unless described as implementation -facts. No comparative benchmark has been run. +Implement and maintain all ASAP physical operators, summary operators and shared +DAG execution locally, following DataFusion's separation of plan contracts, +runtime, expressions and concrete operators. A DataFusion backend or hybrid +runtime is outside this PR's chosen direction. -The [SQL frontend](../../crates/frontend-sql/Cargo.toml) already depends on -DataFusion 43 for parsing/planning; the physical-operator crate does not depend -on DataFusion. Reusing the existing frontend dependency and selecting a newer -execution backend are different choices. This survey follows the requested -upstream main, so an implementation must select a supported release and verify -its exact APIs rather than assume main's interfaces exist in version 43. -Build/binary-size effects also depend on which crates a deployment already links; -they should be measured separately from per-query runtime costs. +This gives ASAP direct ownership of native summary-state edges and shared +execution across precompute and query engines. It also makes ASAP responsible +for operator correctness, resource control and future parallelism/spill. +It is an architectural choice, not a measured performance advantage. -## Findings that affect the decision +## Key differences -1. DataFusion's physical operator boundary uses Arrow RecordBatch streams. - Internal sketch state need not be an Arrow array or be serialized per update. -2. DataFusion does not provide arbitrary common-subplan fan-out merely by sharing - an `Arc`. That is different from being unable to implement it: - custom physical operators and explicit shared execution state are available. -3. Both logical and physical extension points are supported. A basic summary - operator need not require a DataFusion fork. Integration with optimization, - state transport and execution lifecycle is the substantial work. -4. A smaller native runtime is not evidence of lower end-to-end overhead. - Representation, batch size, state crossings and algorithm choice may dominate. -5. Borrowing module boundaries is inexpensive. Porting DataFusion algorithms to - another batch/runtime contract creates an ongoing integration and maintenance - obligation; it does not retain upstream improvements automatically. - -## Runtime costs: compare the actual execution paths - -DataFusion is an embedded Rust library. Its normal physical interface starts a -stream for an output partition; an ordinary projection starts its child stream -and wraps it. It does not create a network boundary or a separately scheduled -worker for every operator. Repartition and explicit buffering introduce tasks, -channels and buffering where the plan calls for them. See -[ProjectionExec](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/projection.rs), -[RepartitionExec](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/repartition/mod.rs), -and [BufferExec](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/buffer.rs). - -A useful decomposition, rather than a fixed “DataFusion overhead” percentage, is: - -```text -elapsed work ≈ planning + binding + execution setup - + batch/stream bookkeeping + representation conversion - + scalar/aggregate kernels + exchange + spill/I/O -``` - -| Cost | DataFusion implementation | Native #462 implementation | -| --- | --- | --- | -| Plan setup | Schema/property derivation, optimizer passes, physical construction | Graph/binding validation, per-node stream and consumer setup | -| Stream dispatch | Boxed streams and dynamic plan/expression interfaces | Local boxed streams and dynamic PhysicalOperator calls | -| Sharing | Arc ownership; explicit shared state where an operator implements it | Rc/RefCell producer state, reader maps, queues, wakers and output reservations | -| Ordinary data | Column arrays, null bitmaps, batched kernels; some operations allocate new arrays | Vec>, per-value enum dispatch, row vectors and clones; repeated row/schema checks in some paths | -| Summary data | Native accumulator while computing; Arrow-compatible state at standard operator boundaries | Native accumulator/state objects can cross edges behind Arc without encoding | -| Parallelism | Partition/exchange tasks, synchronization and data movement when selected | Worker-local execution; comparable partition parallelism is not implemented | -| Large blocking work | Specialized algorithms and spill-capable operators | In-memory joins/sorts/reductions with budget failure rather than spill | - -A RecordBatch clone shares its array references; it does not inherently copy all -column buffers. Creating arrays from row-oriented input, filtering/taking values, -and serializing opaque state can still allocate or copy. Conversely, native -fan-out shares an output Batch, but downstream operators can clone its row -vectors. Neither representation makes every operation zero-copy. See the -[Arrow RecordBatch contract](https://arrow.apache.org/rust/arrow/array/struct.RecordBatch.html) -and [array/buffer model](https://arrow.apache.org/rust/arrow_array/index.html). - -For a tiny precomputed-state readout, binding, allocation and decoding might -exceed kernel time; native execution could have an advantage. For large scans, -joins or high-cardinality grouping, vectorized kernels and a better algorithm -can outweigh framework bookkeeping. These are workload hypotheses, not measured -results. Compare equal thread counts first, then compare each engine's usable -parallelism separately. More CPU consumption from parallel execution can coexist -with lower latency. - -Both runtimes need cooperative cancellation: a long synchronous kernel can -prevent its worker from observing cancellation. DataFusion explicitly documents -this and supplies cooperative wrappers/optimizer support; embedding it does not -make a custom KLL operator preemptible. ASAP's new checkpoints solve the same -class of problem. See [DataFusion cooperative scheduling](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/coop.rs). - -## Arrow is a boundary contract, not a required sketch implementation - -If ASAP reuses DataFusion's standard ExecutionPlan and existing operators, their -interchange remains `SendableRecordBatchStream`. Replacing its item with an -arbitrary ASAP Batch would require adapters or a separate/forked execution -contract. Merely implementing a logical extension does not change that boundary. -See [ExecutionPlan](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/execution_plan.rs). - -There are several practical state representations: - -| Representation | Where the native sketch lives | Consequence | -| --- | --- | --- | -| Custom UDAF accumulator | Rust accumulator object, updated from Arrow arrays | Native update algorithm; encode partial state only when exporting it for merge/spill or final state output | -| Binary/LargeBinary state column | Serialized sketch payload in a standard Arrow column | Portable through generic batch transport; encode/decode cost at state-consuming boundaries | -| Struct/List state columns | Sketch components represented by supported Arrow types | May expose buffers without a monolithic encoding; requires a stable representation and reconstruction rules | -| Run-local handle column | Native state in a registry, integer/binary handle in the batch | Avoids payload serialization locally; registry lifetime, memory, retries and transport require custom handling | -| Fused custom physical operator | Native state remains internal until ordinary outputs are produced | Avoids intermediate state transport; generic optimizers cannot operate inside the fused region | - -DataFusion's `Accumulator` has `update_batch`, `state`, `merge_batch`, `evaluate` -and `size`. Its partial state can differ from its final value and contain several -fields. This is a natural fit for build/merge/finalize sketches, provided ASAP -implements parameter compatibility and the appropriate state schema. It does -not imply that an accumulator serializes itself for every input update. See -[Accumulator](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/expr-common/src/accumulator.rs). - -An Arrow extension type annotates a supported storage type with additional -semantics; it is not a universal container for a Rust trait object. Implementing -a custom Array trait object likewise does not ensure that generic take, filter, -IPC or spill code understands it. A binary extension type for “KLL version X, -parameters Y” is more interoperable than disguising an in-process pointer as a -portable value. Handles may work inside a controlled execution island, but must -not silently escape through spill, distributed exchange or persisted results. -See [Arrow extension types](https://arrow.apache.org/docs/format/Intro.html#extension-types). - -Family, parameters, encoding version, grouping layout, source/window coverage -and ownership must be validated whichever representation is selected. Ordinary -transport of bytes does not prove that two sketches can legally merge. - -## One producer and multiple consumers - -There are three separate mechanisms: - -- Detect equivalent expressions or subplans. -- Represent shared identity in the plan. -- Execute a producer once and deliver its results to independent consumers. - -DataFusion's expression CSE addresses repeated expressions; it is not a general -shared-subplan execution guarantee. Holding the same Arc from two parents is -also insufficient: ordinary parent execution can start its child independently. -See [expression CSE](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/optimizer/src/common_subexpr_eliminate.rs). - -However, “DataFusion cannot represent one producer, multiple consumers” is too -strong. Its custom ExecutionPlan implementations can coordinate shared state. -Upstream already has specialized sharing, for example one-shot scalar-subquery -execution and shared results. Repartition also coordinates producer/output -partition state, although partition distribution is not general broadcast. -These are evidence that custom coordination is possible, not an off-the-shelf -replacement for ASAP's DAG runtime. See -[ScalarSubqueryExec](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/scalar_subquery.rs). - -The motivating KLL case has a simpler alternative when the consumers are -compatible readouts of the same population/window: - -```text -raw rows → build/merge one KLL → estimate_many([0.5, 0.9, 0.99]) - → ordinary columns q50, q90, q99 -``` - -This can be a custom aggregate returning a Struct, or a build-state aggregate -followed by a multi-readout operator. It can avoid both general DAG broadcast and -repeated decoding. Three independent quantile aggregate calls do not by -themselves guarantee one KLL: an ASAP rule must explicitly select the common -state. Different filters, windows or downstream pipelines may prevent this -fusion and still require true fan-out. - -A general DataFusion integration would need an explicit run-scoped shared -producer/subscription operator or materialization service. Its contract must -cover producer identity per partition/run, consumer registration, bounded queues -or replay, slow/dropped consumers, terminal errors, cancellation, memory and plan -reuse. A new run must not accidentally reuse stale mutable state. Cross-query -reuse additionally requires cache coverage/revision invalidation; it is a -separate feature in both engines. - -A particularly important acceptance test is a diamond with asymmetric polling. -If one branch waits for the other to finish before polling, a shared producer -can fill that dormant branch's bounded queue and deadlock. The design must poll -branches appropriately, materialize/spill, or otherwise resolve the dependency. -This obligation applies to an ASAP adapter too; bounded broadcast alone does -not solve it. BufferExec's background queue is not automatically multicast. - -## Logical versus physical extension effort - -The difference is semantic scope, not simply adding an enum case. - -| Layer | Extension work | What is not automatic | -| --- | --- | --- | -| Logical node | UserDefinedLogicalNode, schema, children, expressions, reconstruction, equality/hash and explain | Correct approximation semantics, coverage, sharing identity and rewrite legality | -| Logical-to-physical lowering | Register an ExtensionPlanner mapping the node to existing or custom physical operators | Choosing a legal summary implementation and preserving ASAP guarantees | -| Physical operator | ExecutionPlan properties, child replacement, partition execution, stream, metrics and memory behavior | Efficient merging, spillable state, consumer sharing and durable maintenance | -| UDAF route | Native accumulator, state fields, update/merge/finalize, memory size; optionally specialized grouped accumulation | Arbitrary summary subtraction/join or a shared multi-consumer DAG | - -Logical extensions default to conservative predicate pushdown and expose hooks -for required columns and reconstruction. Approximation-sensitive rules must be -explicit: filtering before a summary can change its population, and removing a -grouping/identity column can invalidate it. See -[UserDefinedLogicalNode](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/expr/src/logical_plan/extension.rs). - -The physical planner already invokes registered extension planners and checks -their output schema. A conventional SummaryBuild or SummaryReadout can therefore -be implemented outside upstream DataFusion. Basic physical extension is feasible; -claiming transparent partitioning, spill, shared execution and maintenance -semantics is the larger engineering task. See -[extension planning](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/core/src/physical_planner.rs). - -For aggregate-shaped work, an AggregateUDF can reuse the existing aggregate -operator rather than introducing a custom physical node. High group cardinality -may justify implementing GroupsAccumulator to avoid a generic per-group adapter. -Reusing a partial/final aggregate pipeline still requires testing the sketch's -merge guarantees, memory reporting and exported state. See -[aggregate expression support](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-expr/src/aggregate.rs). - -ASAP would retain its accuracy/cost model, state-family rules, coverage and -revision metadata, and deployment publication/window lifecycle. DataFusion does -not infer these from an opaque extension. Optimizations must also preserve -correlations introduced when several estimates use the same randomized state; -shared estimates must not silently acquire independence assumptions. - -## How much optimization is actually reused? - -Reusing standard relational nodes gives the widest access to existing expression, -projection/filter, join, aggregation, ordering and partition optimizations. A -large opaque extension hides internal opportunities unless it supplies properties -or is lowered into supported nodes. False ordering/distribution declarations -risk correctness, while conservative declarations can reduce optimization. - -Keeping ASAP's IR and lowering directly to DataFusion physical operators is a -valid alternative to replacing the logical IR. It can reuse kernels and selected -physical optimization, but does not automatically run DataFusion's logical -optimizations over ASAP-specific nodes. Likewise, reusing logical IR alone does -not supply DataFusion physical execution to a separate native runtime. - -| Option | Reused capability | Main obligation | +| Aspect | Reuse DataFusion execution | Implement in ASAP | | --- | --- | --- | -| Native ASAP execution | Existing summary model and explicit DAG fan-out | Own relational kernels, partitioning, spill, metrics and optimizer/runtime contracts | -| DataFusion backend with extensions | Standard relational operators, physical infrastructure; logical optimizations where mapped | Arrow boundaries, custom summary semantics and shared execution adapter | -| ASAP scheduler around coarse DataFusion subplans | DataFusion relational execution inside islands; native summary edges outside | Control conversion boundaries, resource budgets, cancellation and parallelism across both layers | -| DataFusion outer plan with fused native summary regions | DataFusion surrounding relational computation; native state inside regions | Keep opaque regions coarse enough to avoid repeated conversion but expose useful properties | - -A hybrid is an option to benchmark, not automatically the best of both worlds. -Wrapping every tiny native operation in a separate DataFusion execution adds -repeated setup and conversions. Coarse subplans amortize those boundaries, but -two independent budgets or schedulers must not oversubscribe memory or threads. - -## Borrowing organization versus maintaining copied algorithms - -Adopting modules such as plan, runtime, expressions, joins, aggregate, sources -and spill is a useful ownership decision independent of execution framework. -It does not require adopting DataFusion IR or Arrow, and #462 does this now. - -Porting a hash join or external sort is much more than copying its main loop. -Those implementations rely on array kernels, expressions, row encodings, -distribution/ordering facts, memory reservations, async streams, spill formats -and tests. Retaining Arrow can reduce the port surface; replacing the data model -increases it. A maintained fork must also track upstream correctness fixes and -behavioral changes. See the concrete dependency surface in -[grouped aggregation](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs). - -A native engine can intentionally support less. That can be a sound decision -when workloads remain bounded and summary-heavy, but missing large-query -capabilities are a scope tradeoff, not evidence that their overhead has been -eliminated at equivalent functionality. - -## Experiments needed before making a performance claim - -Use the same sketch implementation, parameters, input values, grouping, windows -and accuracy contract. First measure already-bound execution; measure planning -and cold setup separately. Report both single-worker and independently tuned -parallel results. Do not compare a parallel hash join with a single-worker nested -loop and label the difference “runtime overhead.” - -| Workload | Question isolated | -| --- | --- | -| Existing KLL → 1/3/10 quantiles | Readout setup, state cloning, decoding, fusion and consumer overhead | -| Raw values → KLL → several quantiles | Build cost versus state export/transport; verify one build | -| One producer → two asymmetric branches | Buffer growth, progress, dropped consumer and cancellation behavior | -| Many tiny panes and high group cardinality | Allocation, per-group adapters, state bytes and setup amortization | -| Raw Scan → Filter/Project → grouped aggregate | Native rows versus Arrow conversion and columnar computation | -| Join and Sort/Limit under memory pressure | Algorithm choice, workspace, graceful failure versus spill | - -Record wall latency (including p50/p95 for repeated small queries), CPU time, -allocations, peak RSS, charged memory, serialized bytes, encode/decode counts, -producer starts, task/partition counts and cancellation latency. Distinguish -first-run initialization from warm execution. Verify results and summary -compatibility before interpreting timing. Acceptance also includes repeated -executions, multiple roots, reordered consumers and shared-error propagation. - -The KLL prototype should compare native fan-out, a DataFusion UDAF producing -multiple estimates, DataFusion state output plus readout, and only then a custom -shared-producer adapter if independent branches are necessary. This distinguishes -an unavoidable domain cost from a cost introduced by a particular adapter. - -## Position for PR #462 - -Proceed with the native module/runtime cleanup as scoped, while keeping the -execution-backend decision evidence-based. The current justification is direct -ownership of native summary-state edges and shared execution across both engines, -with explicitly limited generic execution capabilities. It is not established -that DataFusion is slower, cannot carry sketches, or cannot support fan-out. - -If broad relational workloads, partition scaling and spill become near-term -requirements, a DataFusion execution backend or coarse hybrid deserves a serious -prototype before porting those subsystems. If bounded summary DAGs dominate and -measured conversion/state-transport costs are material, the native path has a -stronger workload-specific case. The proposed experiments are the decision gate; -this survey alone does not establish a performance winner. +| Runtime overhead | Partition streams, dynamic dispatch and optional exchange/buffering; ordinary operators do not each spawn a task | Worker-local streams, producer queues, reader tracking, row/value dispatch and copies | +| Data representation | Standard physical edges carry Arrow RecordBatches; accumulators can hold native Rust sketches internally | Typed native batches can carry summary-state objects directly | +| Shared producers | Sharing a plan pointer does not ensure one execution; requires fusion, materialization or custom coordination | One producer per node per run, with independent consumer cursors | +| Summary extensions | Logical nodes, extension planners and physical/UDAF APIs exist; a core fork is not inherently required | ASAP owns interfaces, lowering and execution directly | +| Optimizations and algorithms | Existing relational operators and partition/spill infrastructure, subject to valid properties and state contracts | Local implementations; equivalent capabilities must be developed and maintained | +| Maintenance | Adaptation, semantic integration and upstream version changes | Algorithms, runtime contracts, regressions and feature development | + +Neither runtime is inherently cheaper. Arrow batch clones share buffers, while +conversion from rows and sketch encoding may allocate. Native state edges avoid +encoding but retain queue, allocation and row-cloning costs. Specialized kernels +and algorithms can dominate framework overhead. No comparative benchmark has run. + +## What DataFusion integration would require + +**State transport.** Native sketches can live inside a UDAF accumulator. +`update_batch` need not serialize them; `state`/`merge_batch` export and consume +partial state. Across standard physical edges, use Arrow-compatible binary or +structured state, or keep sketches inside a fused operator until scalar readout. +Run-local handles require explicit lifetime, memory and transport restrictions. +Arrow extension metadata alone does not make arbitrary Rust objects portable. + +**Sharing.** Expression CSE, shared plan identity and shared execution are distinct. +Multiple compatible quantiles can be fused into one KLL build and multi-readout; +independent downstream branches may need true fan-out. Custom shared execution +must handle per-run identity, slow/dropped consumers, errors, cancellation and +buffering. Asymmetric consumer polling can deadlock bounded broadcast queues; +ASAP's own runtime has the same scheduling obligation. + +**Optimization and partitions.** Summary nodes must preserve population, grouping, +window coverage, parameters and approximation guarantees. A legal partial/final +merge requires compatible states and nonduplicated input coverage. Logical and +physical extension APIs provide hooks; they do not establish these semantics. +Reusing physical operators alone also does not automatically reuse logical +optimization over ASAP IR. + +**Lifecycle.** Custom state must participate in memory accounting and cooperative +cancellation. Hybrid execution would additionally coordinate budgets, workers +and state ownership across runtimes. These are adapter responsibilities, not +proof that DataFusion core must change. + +## What to borrow now + +Borrow module boundaries and explicit contracts for schemas, boundedness, +emission, memory and cancellation. Keep ASAP's `SummaryExpr`, typed summary +states and shared DAG. #462 establishes those boundaries; partition parallelism, +spill and richer physical properties remain future native work. + +Copying an algorithm is a larger commitment than following organization: +DataFusion joins, sorts and aggregates depend on Arrow kernels, expressions, +partition properties, memory reservations and spill infrastructure. Any port +requires adaptation, tests and ongoing maintenance; upstream fixes do not arrive +automatically. + +Future performance comparisons should use identical sketch kernels, parameters, +inputs and worker counts. Measure setup separately from execution, including +latency, CPU, allocations, peak memory and encoded bytes. Useful workloads are +KLL multi-readout, asymmetric fan-out, grouped scans and memory-constrained +joins/sorts. Such measurements are not a prerequisite for the current decision. + +## Source scope + +Reviewed upstream at +[`e2ca7f3`](https://github.com/apache/datafusion/tree/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38) +on 2026-09-24. The SQL frontend uses DataFusion 43; upstream-main APIs below are +not a claim about that release. This PR adds no DataFusion execution dependency. + +- [ExecutionPlan and physical properties](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/execution_plan.rs) +- [Accumulator state/update/merge interface](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/expr-common/src/accumulator.rs) +- [Logical extension contracts](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/expr/src/logical_plan/extension.rs) and [physical extension planning](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/core/src/physical_planner.rs) +- [Expression CSE](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/optimizer/src/common_subexpr_eliminate.rs) and [specialized scalar-subquery sharing](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/scalar_subquery.rs) +- [Arrow RecordBatch ownership](https://arrow.apache.org/rust/arrow/array/struct.RecordBatch.html) and [extension types](https://arrow.apache.org/docs/format/Intro.html#extension-types) diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index ca678fc2..040664d4 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -1,303 +1,136 @@ # Shared physical operators and DAG execution -## Decision +## Decision and ownership -ASAP owns an independent physical operator library and DAG runtime. Precompute -and query engines bind inputs and consume outputs from the same library. The -operator defines computation; the engine supplies ingestion time or query time, -window boundaries, storage access and publication. There is no second execution -algorithm selected by phase. +ASAP implements and maintains its own physical operators, summary kernels and +DAG runtime. DataFusion is a reference for module organization and execution +contracts; #462 does not adopt its execution backend or a hybrid runtime. +`SummaryExpr` and the shared-producer DAG model remain unchanged. -The library lives in the ASAPPlanner workspace as `asap-physical-operators`, -alongside `asap-types` and logical-to-physical lowering. It depends on those local -IR types and never on ASAPQuery-backend. New IR nodes and their implementations -can be reviewed and tested in one Planner PR. Private implementation tests live -inside their owning crate; integration tests exercise exported physical DAGs. -Its native operators can execute independently of either backend engine. -Engine integration must use these operators for computation, rather than merely -using the shared scheduler around a second implementation. +`asap-physical-operators` lives beside Planner IR and lowering, depends on local +IR types, and has no ASAPQuery-backend dependency. Precompute and query engines +use the same computation: for example, Scan → KLL build/merge → quantile readout. +Deployments supply sources, evaluation windows, storage and publication. -## Engine and storage architecture - -```mermaid -flowchart TB - subgraph Engine[ASAP Query Engine or Precompute Engine] - Planner[ASAPPlanner] --> Plan[Physical DAG] - Plan --> Runtime[ASAP Runtime] - Runtime --> Operators[Physical Operator Library] - end - Operators --> API[Data Source / Storage API] - API --> Connectors[Connectors / Adapters] - Connectors --> Systems[Storage / Data Systems] -``` - -Planner describes source identities, schemas and computation. The runtime -schedules operators and owns shared execution, backpressure, cancellation and -resource accounting. Scan reads raw rows through the data-source interface; -Filter, Project, joins, aggregation and summary construction execute inside ASAP. -The same interface serves ingestion time and query time. - -Connectors own system-specific access and decoding. Prometheus, S3, Parquet, -Kafka and Iceberg are possible integrations, not built-in dependencies or claims -of implemented support. A file format adapter and a storage transport can be -composed; they need not each be a separate query engine. Asking an external -system to execute the complete query remains external execution, not raw Scan. - -The shared API exchanges typed native batches. Ordinary columns preserve -Planner types; other operator edges can carry ASAP summary states without an -Arrow representation. Raw Scan accepts ordinary rows only. Reading stored -summary state remains a distinct storage operation, with its own compatibility -and coverage requirements. - -### Raw Scan contract - -The library provides Scan, a registry keyed by Planner table/time-series source -identity, and an immutable in-memory connector. A deployment registers its -connectors before binding the physical DAG. Binding resolves metadata and -validates schemas and predicates without opening a reader. Execution lazily -opens one cursor per reachable Scan per run, even when multiple consumers share -that node. Each run gets a fresh cursor; dropping it releases connector resources. - -Scan evaluates Planner leaf predicates itself with three-valued boolean logic: -only TRUE retains a row. Predicate pushdown is not assumed. Projection, time-range -selection and aggregation remain explicit downstream operations; Scan does not -silently interpret query evaluation time as a lookback or replay Kafka offsets. -Connectors receive the execution context for cancellation and resource control; -a deployment must bind any required snapshot/offset and bound its I/O buffers. -Reader errors and schema drift fail execution, rather than becoming empty results. - -The current post-ASAP IR retains raw Scan as a leaf expression in its `Fallback` -payload. The source-aware binder recognizes only that Scan expression and runs -it locally; other retained expressions are still rejected. This does not invoke -an external fallback or introduce another operator vocabulary. Explicit stored -frontiers can still cut a DAG at a precomputed result. - -Acceptance includes a raw-only Scan → Sort → Limit DAG at both phases, shared -consumers, independent runs, cancellation before opening, schema drift, reader -errors, null predicates, empty inputs and memory limits. This establishes the -library path. A backend must register a real reader before it can serve raw-only -plans; a Prometheus reader and other external connectors are not implemented here. - -## Workspace organization - -[DataFusion's physical-plan crate](https://github.com/apache/datafusion/tree/main/datafusion/physical-plan) -is the ownership reference: it owns the execution-plan interface, concrete -operators, streams, metrics and operator tests within the same repository as -planning. ASAP follows that repository boundary, while retaining its own DAG -execution model. - -| Responsibility | ASAP owner | +| Responsibility | Owner | | --- | --- | -| post-ASAP nodes, schemas, parameters and execution phase | `asap-types` | -| Logical-to-physical lowering and candidate correctness | `asap-aware-mapping` | -| Physical operator implementations, input/output validation and streams | `asap-physical-operators` | -| Shared-producer scheduling, cancellation and resource accounting | `asap-physical-operators` | -| Summary state encoding | `asap_sketch_codec` | -| Scan and data-source interface; reference memory connector | `asap-physical-operators` | -| External connectors, durable stores, publication and serving protocols | Deployment repositories | - -The IR crate does not depend on execution. The physical operator crate depends -on the local IR crate. Operator unit tests can exercise private implementation -details; Planner integration tests check that emitted DAGs bind and execute. -Runtime values preserve the IR schema instead of redefining its type semantics. +| IR, schemas and parameters | `asap-types` | +| Lowering and candidate correctness | `asap-aware-mapping` | +| Operators, shared execution and source interface | `asap-physical-operators` | +| Summary encoding | `asap_sketch_codec` | +| External connectors, durable storage and serving | Deployment repositories | -### Module ownership +## Module organization ```text src/ - plan/ PhysicalDag, PhysicalOperator, properties and validation - runtime/ streams, shared producers, context, memory and cancellation - expressions/ scalar evaluation and Planner expression adaptation + plan/ PhysicalDag, PhysicalOperator, properties, validation + runtime/ streams, shared producers, context, memory, cancellation + expressions/ scalar evaluation and Planner expression adaptation operators/ projection.rs filter.rs joins/ - aggregate/ ordinary and temporal reductions + aggregate/ ordinary and temporal reductions sort.rs limit.rs - summary/ build, merge and readout - source.rs literal/batch sources, union and scalar conversion - sources/ raw-source API, Scan and memory connector - binding/ Planner executable DAG to physical operators - summary_operators/ mathematical summary kernels, factory and traits - stored_state/ decoding, delta application and persisted-state readout - capability.rs kernel and native operator support checks - values.rs typed rows and state payload validation + summary/ build, merge and readout + source.rs literals, batch sources, union, scalar conversion + sources/ raw-source API, Scan and memory connector + binding/ Planner executable DAG → physical operators + summary_operators/ mathematical kernels, factory and traits + stored_state/ decoding, delta application, persisted-state readout + capability.rs support checks + values.rs typed rows and state validation ``` -The graph owns topology and static checks; the runtime owns each execution's -producer state. Operator modules own both construction checks and computation. -The `Operator` enum dispatch remains a small internal routing point. This does -not change `SummaryExpr`, Planner semantics or shared-producer identity. - -`Expression` is the typed native builder; `CompiledExpression` validates and -adapts Planner scalar expressions. Both are owned by `expressions`, with shared -numeric execution. Planner-specific coercions and checked PromQL division remain -explicit at their respective binding boundaries. No expression evaluator lives -inside the projection or filter implementation. - -Deployment engines normally use `binding`, `plan`, `runtime` and `sources`. -`summary_operators` exposes update kernels for pane maintenance; `stored_state` -serves deployments reconstructing persisted panes. Kernel traits include state -serialization because persistence consumes those states, but neither the graph -nor its scheduler depends on serialization. Existing `dag`, `accumulators`, `factory`, `traits` and -`arithmetic` import paths are thin compatibility re-exports. +`plan` owns static contracts; `runtime` owns per-run state; operators own +construction checks and computation. `Expression` and `CompiledExpression` +share scalar execution under `expressions`. Summary kernels do not own scheduling +or storage selection. Kernel modules use operation names without `_accumulator`; +those old module names have no aliases. Top-level `dag`, `accumulators`, +`factory`, `traits` and `arithmetic` remain compatibility re-exports. ## Execution contract -An immutable plan describes typed nodes and dependency edges. Each execution -creates its own operator state. One producer may have multiple consumers; the -producer executes once in that run and sends the same outputs to all consumers. -Separate runs, query evaluation times and ingestion windows do not share mutable -state. Request-local caching of intermediate results is scoped to execution. - -The runtime validates dependencies, schemas, arity and cycles before sources -start. Each consumer advances independently. Bounded queues apply backpressure; -dropping one consumer does not cancel other consumers. Whole-run cancellation -wakes readers and releases queued work as streams are polled or dropped. - -Execution runs on the caller's worker without an internal thread pool. Active -streams are worker-local. Deployments poll all consumers concurrently. The byte -budget accounts for retained outputs and native operator state, including outputs -held after queue eviction. It is not an RSS limit: source-owned data, temporary -allocation peaks and allocator overhead remain outside that estimate. Blocking -operators currently have no spill implementation. Join results and membership -sets, grouping workspace, sort scratch space and merged summary-state estimates -are charged while retained. Long row loops and sort merge steps yield to the -caller, so cancellation and other consumers can progress within a single batch. -Individual kernel calls and scalar evaluations remain synchronous; memory -estimates are not allocator-exact peak bounds. - -### Finite input and emission - -`PhysicalOperator::properties` reports output boundedness and emission mode. -Unknown source boundedness is conservative: it cannot satisfy a finite-input -requirement. `PhysicalDag::properties` derives these facts together with topology -and schema validation before `start` is called on any source. - -Sort, ordinary aggregate, temporal reductions, both joins, scalar/keyed summary build and summary merge -and vector-to-scalar require bounded inputs and emit after input ends. Summary -build updates incrementally but still finalizes at end-of-input. Projection, -filter, limit, union and readout emit incrementally. A global Limit bounds its -output cardinality; a grouped Limit inherits input boundedness because new -groups may continue arriving. Neither declaration promises a time deadline. - -`RawSource::boundedness` defaults to Unknown. Connectors must explicitly promise -that a snapshot or window ends; merely receiving a query/ingestion `Scope` is -insufficient. The memory connector declares Bounded. Installed physical source -frontiers preserve their supplied properties through the checked binding wrapper. - -## Operator coverage - -Native operations include raw Scan and scalar sources, typed Project and Filter, arithmetic -and boolean expressions, exact grouped aggregation, relational joins (including semi-join), grouped Sort and -Limit, Union, vector-to-scalar conversion, and summary construction, merge and -readout. Window operators consume Planner aggregate intents for Rate, Increase, -Sum, Avg, Min, Max, Count and histogram quantiles. Deployments supply window -boundaries and bound columns; the computation is identical in either phase. -Count outputs Int64. Binary expressions use Planner arithmetic/comparison kinds -and enforce its checked-division domains. - -Grouped TopK composes Sort and Limit within each group. A weighted summary can -consume per-series rates directly: each job has its own CMS or CountSketch with a candidate heap, -with service as the item and rate as the weight. Sum accumulation inside the -summary replaces the exact grouped-sum materialization. Typed readout returns -candidate identities and estimated scores; a semi-join is not required for this -realization. Both score error and membership require accuracy guarantees. - -The `summary_operators` module owns typed summary kernels. Its modules use operation -names, such as `count_min_sketch`, `exact` and `sum`, without an -`_accumulator` filename suffix or old-path aliases. Native weighted CMS and CountSketch -use Float64 counters and preserves typed item identities, including numeric and -NULL keys. Neither uses the integer-count codec or fixed-point counter-delta -updates. The DAG binder supports column weights and explicit column/tuple item -identities for both families; unsupported families or identities are rejected. -CMS accepts nonnegative weights and estimates each score using the minimum row -counter. CountSketch accepts signed weights, uses separate bucket/sign hash seeds, -and takes the median of sign-corrected estimates across a positive odd number of -rows. Candidate heaps rank estimated scores, not absolute magnitudes. A bounded -heap alone does not establish candidate completeness, including after signed -updates or merges. Missing accuracy evidence remains a candidate requirement; -physical binding does not impose deployment's accuracy acceptance policy. -Versioned Float64 states carry their algorithm and dimensions; cross-family or -incompatible-shape merges are rejected, without integer-state compatibility decoding. -Summary construction and readout run in either ingestion or query scope, as -chosen by deployment. Each run constructs independent partition state; deployment -must supply one complete evaluation window, or an equivalent maintained snapshot. -The candidate capacity is independent of the grouped Limit's output count. - -Values retain Planner types and nullability. Native summary batches currently -support exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS and CountSketch with candidate heaps. Stored-summary decoding, delta reconstruction, exact finalization and -family-specific SketchQuery readout also live in this library. Deployment code -selects compatible panes and supplies source batches. Stored-state kernels do -not imply native batch bindings for every family. Unsupported expressions, -state families and parameters must be rejected during binding, without an -implicit external fallback. The backend retains source, storage, publication and protocol adapters. -Computation must bind to Planner operations without a second backend operator -vocabulary. Binding rejects unsupported Planner nodes before starting sources. - -Deployments provide explicit storage or ingestion source frontiers and may bind -raw Scan through the shared data-source interface. The memory connector proves -the library contract; backend raw-data access still requires a deployment connector. - -### Capability levels - -| Level | Acceptance contract | Scope | -| --- | --- | --- | -| Update kernel | `capability::validate_summary_kernel` | Family, parameters, grouping and item/update layout; used by the accumulator factory | -| Native state edge | `capability::validate_native_family` | Exact accumulators, KLL, DDSketch, HLL and Float64 weighted CMS and CountSketch with compatible parameters | -| Scalar native readout | `capability::validate_native_readout` | Supported native state plus statistic/readout arguments | -| Keyed native readout | `Operator::keyed_readout` | Weighted CMS family, heap capacity, typed identity/score schema and preserved partition columns | -| Complete physical plan | `binding::bind` / `bind_with_data_sources` | Node support, expressions, schemas, source frontiers and bounded input requirements | -| Persisted state | `stored_state` decoders and readout functions | Stored format and family-specific reconstruction/readout support | - -For example, a valid CMS update kernel does not imply a native CMS batch edge. -Stored-state support also does not register a native operator. Consumers must -use the contract for the path they intend to execute rather than treating kernel -availability as whole-plan acceptance. - -Partitioned parallel execution, disk spill, cost-based algorithm selection, -physical ordering/distribution properties and per-operator Explain/Analyze -metrics remain future extensions. This change establishes finite-input and -emission contracts without claiming those additional DataFusion capabilities. - -## DataFusion reuse vs independent implementation - -The [execution comparison survey](datafusion-execution-comparison.md) examines -runtime costs, Arrow/sketch representation, producer sharing, logical/physical -extension points, optimization reuse, native maintenance cost and a benchmark -plan against pinned upstream sources. - -| Decision dimension | DataFusion backend | Native ASAP execution | -| --- | --- | --- | -| Runtime overhead | Partition streams; ordinary operators do not each require a separate task. Conversion, state transport and exchanges depend on the chosen integration | Local streams and native state edges; producer queues, per-row values, validation and copies still have costs | -| Summary representation | Internal accumulators may remain native Rust objects; standard physical edges use Arrow batches, with binary/structured state or custom adapters | Native summary objects can cross edges directly | -| Shared execution | Shared plan references do not automatically share results; fusion, materialization or custom run-scoped producer coordination can implement reuse | One producer per node per run, with independent consumer cursors | -| Extension effort | Both logical and physical extension APIs exist; conventional custom nodes need not require an upstream fork | Direct control of both interfaces and implementations | -| Generic computation | Mature relational kernels, partitioning and spill infrastructure, subject to correct custom properties and state contracts | Supported vocabulary implemented locally; partitioned execution and spill remain deferred | -| Maintenance cost | Adapter, semantic and upstream-version integration | Ownership of operators, scheduler, resource contracts and future generic execution features | - -The native decision in this PR is scoped to direct ownership of summary-state -edges and shared DAG execution. It does not establish that DataFusion is slower, -that Arrow requires serializing every sketch update, or that DataFusion cannot -implement fan-out. Multiple quantiles of one KLL can often be fused into one -readout; independent downstream branches may still require general sharing. - -Borrowing DataFusion's module boundaries is useful regardless of backend choice. -Porting its algorithms also means adapting their array, expression, memory, -partition and spill dependencies and maintaining those adaptations. The survey -specifies the measurements needed to compare full DataFusion, native execution -and coarse hybrid subplans without confusing algorithm improvements with runtime -overhead. No comparative benchmark is claimed here. - -## Acceptance - -Independent tests must execute shared-producer diamonds without duplicated work -or deadlock, exercise slow and dropped consumers, propagate cancellation and -errors, retain memory accounting, and isolate separate executions. Operator tests -must cover types, nulls, grouped limits, state compatibility and unsupported -bindings. The same summary pipeline must run at ingestion time and query time. - -Backend integration adds deployment acceptance for source binding, window and revision -scope, durable publication and query output adaptation. External exact forwarding -does not count as evidence that a local operator was implemented. +- Validate topology, arity, schemas, supported bindings and input boundedness + before starting sources. Unsupported operations fail without external fallback. +- Execute each reachable producer once per run. Consumers have independent + cursors over shared outputs; separate runs never share mutable execution state. +- Run worker-local streams on the caller's worker. Deployments must poll consumers + concurrently: bounded queues provide backpressure, and dropping one consumer + leaves the others active. +- Propagate errors and cancellation. Long row loops and sort merge steps yield + cooperatively; individual scalar and kernel calls remain synchronous. +- Charge retained outputs and estimated operator workspace against the byte + budget. This is not an RSS or allocator-exact peak limit; source-owned data and + temporary allocation peaks are not fully covered. Blocking operators do not spill. + +`PhysicalOperator::properties` reports boundedness and emission; +`PhysicalDag::properties` derives them before execution. Sort, aggregate, temporal +reductions, joins, summary build/merge and vector-to-scalar require bounded input +and finalize after input ends. Projection, filter, limit, union and readout emit +incrementally. Global Limit bounds output cardinality; grouped Limit inherits +input boundedness. Neither promises a time deadline. + +## Sources and binding + +Raw Scan uses a registry keyed by Planner source identity. Binding checks metadata +without opening readers; execution lazily opens one cursor per reachable Scan +per run. Scan evaluates leaf predicates with three-valued logic, retaining only +TRUE. Projection, time selection and aggregation remain explicit operators. +Reader failures and schema drift fail execution. + +`RawSource::boundedness` defaults to Unknown. Connectors must explicitly declare +finite snapshots/windows and handle cancellation and I/O buffering. Only an +immutable memory connector is included; external readers are deployment work. + +The binder recognizes raw Scan inside the current `Fallback` leaf payload; +other retained expressions remain unsupported. Explicit source frontiers may +supply precomputed results. Stored-summary loading is separate from raw Scan: +deployments select compatible panes and provide coverage/revision scope. + +## Supported computation + +The library implements typed projection/filter, scalar arithmetic and booleans, +exact grouped aggregation, joins including semi-join, grouped Sort/Limit, Union, +scalar conversion, and summary build/merge/readout. Temporal reductions include +Rate, Increase, Sum, Avg, Min, Max, Count and histogram quantiles. Values preserve +Planner types/nullability and checked-division semantics; Count returns Int64. + +Native summary edges support exact Sum/Count/Min/Max/Rate/Increase, KLL, +DDSketch, HLL, and Float64 weighted CMS/CountSketch with candidate heaps. +Weighted TopK consumes finalized per-series rates into a summary per group, +reads typed candidate identities/scores, then applies grouped Sort → Limit. +CMS uses nonnegative weights and minimum-row estimates; CountSketch accepts +signed weights and uses median sign-corrected estimates at positive odd depth. +Neither uses integer-count encoding or fixed-point updates. + +Heap capacity differs from output k; ranking and merging do not prove candidate +completeness. Binding checks representation compatibility, not deployment accuracy +admission. Deployments provide a complete evaluation window or equivalent snapshot. + +| Capability | Validation entry point | +| --- | --- | +| Update kernel and parameters | `capability::validate_summary_kernel` | +| Native state representation | `capability::validate_native_family` | +| Scalar readout | `capability::validate_native_readout` | +| Keyed readout and identity/score schema | `Operator::keyed_readout` | +| Complete executable plan | `binding::bind` / `bind_with_data_sources` | +| Persisted formats and reconstruction | `stored_state` | + +Kernel or stored-state support alone does not imply executable-plan support. + +## Scope and acceptance + +Partitioned parallelism, disk spill, cost-based algorithm selection, physical +ordering/distribution properties and per-operator Explain/Analyze remain future +work. ASAP owns implementing and maintaining these capabilities when needed. +See the [DataFusion comparison](datafusion-execution-comparison.md) for tradeoffs. + +Tests cover shared-producer diamonds, slow/dropped consumers, cancellation, +errors, resource accounting and run isolation; operator coverage includes types, +nulls, grouping, state compatibility and rejected bindings. The same summary +pipeline runs in ingestion and query scopes. Deployment acceptance additionally +requires real source binding, window/revision handling, publication and output +adaptation; external query forwarding does not demonstrate local execution. From 8872466cb416b79522ab6af069ecbba9ba03e2d4 Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 16:56:04 +0000 Subject: [PATCH 17/90] fix: align physical semantics and add DataFusion-inspired contract tests --- .../src/expressions/mod.rs | 5 +- .../src/expressions/planner.rs | 32 + .../src/operators/joins/mod.rs | 13 +- .../src/operators/summary/mod.rs | 19 +- crates/asap-physical-operators/src/values.rs | 1 + .../tests/physical_semantics.rs | 685 ++++++++++++++++++ crates/types/src/pre_asap/query_expr.rs | 5 + 7 files changed, 756 insertions(+), 4 deletions(-) create mode 100644 crates/asap-physical-operators/tests/physical_semantics.rs diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs index 2ed99a36..f6dd1336 100644 --- a/crates/asap-physical-operators/src/expressions/mod.rs +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -72,7 +72,10 @@ impl Expression { }; Ok((dtype, n || m)) } - Planner(expression) => Ok(expression.dtype()), + Planner(expression) => { + expression.validate_input(input)?; + Ok(expression.dtype()) + } Column(i) => { let (t, n) = plain(input, *i)?; Ok((t.clone(), n)) diff --git a/crates/asap-physical-operators/src/expressions/planner.rs b/crates/asap-physical-operators/src/expressions/planner.rs index 39b9df5d..b2d75421 100644 --- a/crates/asap-physical-operators/src/expressions/planner.rs +++ b/crates/asap-physical-operators/src/expressions/planner.rs @@ -240,6 +240,20 @@ fn compare(op: &CompareOpKind, left: Value, right: Value) -> Result Ok(Value::Bool(true)), + CompareOpKind::Eq + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge => Ok(Value::Bool(false)), + _ => Err(Error::Invalid(format!("comparison {op:?}"))), + }; + } let ordering = cell_cmp(&left, &right) .ok_or_else(|| Error::Invalid("comparison of incompatible values".into()))?; let value = match op { @@ -353,6 +367,24 @@ impl CompiledExpression { pub(crate) fn dtype(&self) -> (DataType, bool) { self.output.clone() } + pub(crate) fn validate_input(&self, input: &Schema) -> Result<(), Error> { + if input.fields.len() != self.schema.columns.len() + || input + .fields + .iter() + .zip(&self.schema.columns) + .any(|(field, column)| { + field.dtype + != planner_types::post_asap::SummaryFamilyType::Plain(column.dtype.clone()) + || field.nullable != column.nullable + }) + { + return Err(Error::Invalid( + "expression input differs from its bound schema".into(), + )); + } + Ok(()) + } /// Evaluate a row under the same typed schema used when binding the expression. pub fn evaluate(&self, row: &[Value]) -> Result { if row.len() != self.schema.columns.len() diff --git a/crates/asap-physical-operators/src/operators/joins/mod.rs b/crates/asap-physical-operators/src/operators/joins/mod.rs index 6d10bb4f..1fece7c1 100644 --- a/crates/asap-physical-operators/src/operators/joins/mod.rs +++ b/crates/asap-physical-operators/src/operators/joins/mod.rs @@ -152,7 +152,7 @@ pub(super) fn execute<'a>( let mut work = Cooperative::new(&context); for row in &right { work.checkpoint().await?; - if right_cols.iter().all(|&i| !matches!(row[i], Value::Null)) { + if right_cols.iter().all(|&i| matchable_key(&row[i])) { let key = group_key(row, &right_cols)?; if !members.contains(&key) { workspace.grow(key_bytes(&key))?; @@ -163,7 +163,7 @@ pub(super) fn execute<'a>( let mut rows = Vec::new(); for row in left { work.checkpoint().await?; - if left_cols.iter().all(|&i| !matches!(row[i], Value::Null)) + if left_cols.iter().all(|&i| matchable_key(&row[i])) && members.contains(&group_key(&row, &left_cols)?) { workspace.grow(std::mem::size_of::>())?; @@ -176,3 +176,12 @@ pub(super) fn execute<'a>( } unreachable!() } + +// Group keys canonicalize NaNs, but equality joins must not match them. +fn matchable_key(value: &Value) -> bool { + match value { + Value::Null => false, + Value::Float64(v) => !v.is_nan(), + _ => true, + } +} diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs index 7af3ab0e..13e61ec0 100644 --- a/crates/asap-physical-operators/src/operators/summary/mod.rs +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -177,7 +177,18 @@ impl Operator { } else { DataType::Float64 }; - fields[state] = result_field("value", result_type, false); + // A state-only row represents the global population. Its extrema may + // be empty, just like an ordinary ungrouped MIN/MAX aggregate. + let nullable = fields.len() == 1 + && matches!( + fields[state].dtype, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Min + | planner_types::post_asap::ExactKind::Max, + _ + ) + ); + fields[state] = result_field("value", result_type, nullable); Ok(Self { kind: Kind::Readout { state, @@ -264,6 +275,12 @@ pub(super) fn execute<'a>( i64::try_from(count) .map_err(|_| Error::Operator("exact count exceeds Int64".into()))?, ) + } else if output.fields[*state].nullable + && matches!(statistic, crate::Statistic::Min | crate::Statistic::Max) + { + let stats = summary.aux_stats(); + let value = if *statistic == crate::Statistic::Min { stats.min } else { stats.max }; + value.map(Value::Float64).unwrap_or(Value::Null) } else { Value::Float64( summary diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs index cccfb7ad..93019f62 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -237,6 +237,7 @@ impl Batch { } pub fn bytes(&self) -> usize { std::mem::size_of::() + + self.rows.capacity() * std::mem::size_of::>() + self .rows .iter() diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs new file mode 100644 index 00000000..6d3bd022 --- /dev/null +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -0,0 +1,685 @@ +//! Contract tests inspired by DataFusion's limit, sort and join test matrices. +//! Expectations follow ASAP's IR (notably row-count and IEEE NaN equality). +//! Reference: apache/datafusion e2ca7f3, physical-plan/src/{limit.rs,sorts/sort.rs}. +use asap_physical_operators::{ + expressions::CompiledExpression, + operators::{Expression, Operator, Reduction, SortKey}, + plan::PhysicalDag, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Schema, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{CompareOpKind, DataType, JoinKind, Predicate, QueryExpr}, +}; +use std::{rc::Rc, sync::Arc}; + +fn schema(fields: &[(&str, DataType, bool)]) -> Schema { + Arc::new(SummarySchema { + fields: fields + .iter() + .map(|(name, dtype, nullable)| SummaryField { + name: (*name).into(), + dtype: SummaryFamilyType::Plain(dtype.clone()), + nullable: *nullable, + }) + .collect(), + time_index: None, + }) +} +fn context() -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits { + max_buffered_batches: 1, + ..Limits::default() + }, + ) + .unwrap() +} +fn collect(dag: &PhysicalDag<'_, Batch, Schema>, root: u64) -> Vec> { + let run = context(); + let rows = block_on(async { + let mut stream = dag.execute(&[root], run.clone()).unwrap().remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend_from_slice(batch.unwrap().rows()); + } + rows + }); + assert_eq!(run.retained_bytes(), 0); + rows +} +fn unary(input: Schema, batches: Vec>>, op: Operator) -> Vec> { + let mut dag = PhysicalDag::default(); + let batches = batches + .into_iter() + .map(|rows| Batch::try_new(input.clone(), rows).unwrap()) + .collect(); + dag.add(0, vec![], Operator::source(input, batches).unwrap()) + .unwrap(); + dag.add(1, vec![0], op).unwrap(); + collect(&dag, 1) +} +fn keys(rows: &[Vec]) -> Vec>> { + rows.iter() + .map(|r| r.iter().map(|v| v.key().unwrap()).collect()) + .collect() +} +fn eq_predicate() -> Predicate { + Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(1)), + })) +} +fn join(left: Vec, right: Vec, kind: JoinKind, keyed: bool) -> Vec> { + let input = schema(&[("key", DataType::Float64, true)]); + let output = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + input.clone() + } else { + schema(&[ + ("left", DataType::Float64, true), + ("right", DataType::Float64, true), + ]) + }; + let op = if keyed { + Operator::semi_join(input.clone(), input.clone(), vec![(0, 0)]).unwrap() + } else { + Operator::relational_join(input.clone(), input.clone(), kind, &eq_predicate(), output) + .unwrap() + }; + let mut dag = PhysicalDag::default(); + for (id, values) in [(0, left), (1, right)] { + let batches = values + .into_iter() + .map(|v| Batch::try_new(input.clone(), vec![vec![v]]).unwrap()) + .collect(); + dag.add( + id, + vec![], + Operator::source(input.clone(), batches).unwrap(), + ) + .unwrap(); + } + dag.add(2, vec![0, 1], op).unwrap(); + collect(&dag, 2) +} + +// OFFSET/FETCH must be invariant to empty batches and input batch boundaries. +#[test] +fn limit_offset_fetch_matrix() { + let input = schema(&[("v", DataType::Int64, false)]); + for chunk in [1, 2, 5, 12] { + let values = (0..9).map(|n| vec![Value::Int64(n)]).collect::>(); + let mut batches = vec![vec![]]; + for rows in values.chunks(chunk) { + batches.push(rows.to_vec()); + batches.push(vec![]); + } + for offset in [0, 1, 8, 9, 10, u64::MAX] { + for n in [0, 1, 3, 12, u64::MAX] { + let rows = unary( + input.clone(), + batches.clone(), + Operator::limit(input.clone(), n, offset, vec![]).unwrap(), + ); + let expected = values + .iter() + .skip(offset.min(9) as usize) + .take(n.min(9) as usize) + .cloned() + .collect::>(); + assert_eq!( + keys(&rows), + keys(&expected), + "chunk={chunk}, offset={offset}, n={n}" + ); + } + } + } +} + +// Zero-column batches still have rows: LIMIT must not infer cardinality from columns. +#[test] +fn limit_preserves_zero_column_row_count() { + let input = schema(&[]); + let rows = unary( + input.clone(), + vec![vec![vec![]; 5], vec![vec![]; 5]], + Operator::limit(input, 4, 3, vec![]).unwrap(), + ); + assert_eq!(rows.len(), 4); +} + +// NULL placement is independent of sort direction; ties retain original row order. +#[test] +fn sort_direction_null_placement_and_ties() { + let input = schema(&[("v", DataType::Int64, true), ("id", DataType::Int64, false)]); + let values = [Some(2), None, Some(1), Some(2), None]; + let rows = values + .iter() + .enumerate() + .map(|(i, v)| { + vec![ + v.map(Value::Int64).unwrap_or(Value::Null), + Value::Int64(i as i64), + ] + }) + .collect::>(); + for (descending, nulls_first, expected) in [ + (false, false, vec![2, 0, 3, 1, 4]), + (false, true, vec![1, 4, 2, 0, 3]), + (true, false, vec![0, 3, 2, 1, 4]), + (true, true, vec![1, 4, 0, 3, 2]), + ] { + let op = Operator::sort( + input.clone(), + vec![SortKey { + column: 0, + descending, + nulls_first, + }], + vec![], + ) + .unwrap(); + let result = unary( + input.clone(), + vec![rows[..2].to_vec(), vec![], rows[2..].to_vec()], + op, + ); + let ids = result + .iter() + .map(|r| match r[1] { + Value::Int64(n) => n, + _ => unreachable!(), + }) + .collect::>(); + assert_eq!(ids, expected); + } +} + +// Outer joins preserve unmatched NULLs, while semi/anti joins preserve left multiplicity. +#[test] +fn joins_nulls_duplicates_and_empty_sides() { + for (kind, expected_len) in [ + (JoinKind::Inner, 4), + (JoinKind::Left, 6), + (JoinKind::Right, 6), + (JoinKind::Full, 8), + (JoinKind::Semi, 2), + (JoinKind::Anti, 2), + ] { + let left = vec![ + Value::Float64(1.), + Value::Float64(1.), + Value::Float64(2.), + Value::Null, + ]; + let right = vec![ + Value::Float64(1.), + Value::Float64(1.), + Value::Float64(3.), + Value::Null, + ]; + let result = join(left, right, kind.clone(), false); + assert_eq!(result.len(), expected_len, "{kind:?}"); + } + for (kind, expected_len) in [ + (JoinKind::Inner, 0), + (JoinKind::Left, 1), + (JoinKind::Right, 0), + (JoinKind::Full, 1), + (JoinKind::Semi, 0), + (JoinKind::Anti, 1), + ] { + assert_eq!( + join(vec![Value::Float64(7.)], vec![], kind.clone(), false).len(), + expected_len, + "{kind:?}" + ); + } + let result = join(vec![Value::Float64(7.)], vec![], JoinKind::Left, false); + assert!(matches!( + result[0].as_slice(), + [Value::Float64(7.), Value::Null] + )); +} + +// Changing the semi-join algorithm must not turn IEEE NaN != NaN into a match. +#[test] +fn keyed_semijoin_obeys_ieee_equality_for_nan_and_zero() { + let left = vec![ + Value::Float64(f64::NAN), + Value::Float64(-0.), + Value::Float64(0.), + Value::Null, + ]; + let right = vec![Value::Float64(f64::NAN), Value::Float64(0.), Value::Null]; + let keyed = join(left, right, JoinKind::Semi, true); + let expected = vec![vec![Value::Float64(-0.)], vec![Value::Float64(0.)]]; + assert_eq!(keys(&keyed), keys(&expected)); +} + +// Group equality intentionally differs from predicate equality: NULL and NaNs group together. +#[test] +fn grouping_canonicalizes_null_nan_and_signed_zero() { + let input = schema(&[("v", DataType::Float64, true)]); + let op = Operator::aggregate( + input.clone(), + vec![0], + vec![("count".into(), Reduction::Count)], + ) + .unwrap(); + let values = vec![ + Value::Null, + Value::Null, + Value::Float64(0.), + Value::Float64(-0.), + Value::Float64(f64::NAN), + Value::Float64(f64::from_bits(0x7ff8000000000001)), + ]; + let result = unary( + input, + values.into_iter().map(|v| vec![vec![v]]).collect(), + op, + ); + assert_eq!(result.len(), 3); + assert!(result.iter().all(|r| matches!(r[1], Value::Int64(2)))); +} + +// Global empty input yields one aggregate row; grouped empty input yields none. +#[test] +fn aggregate_empty_and_all_null_follow_asap_contract() { + let input = schema(&[("v", DataType::Int64, true)]); + for batches in [ + vec![], + vec![vec![]], + vec![vec![vec![Value::Null], vec![Value::Null]]], + ] { + let n = batches.iter().map(Vec::len).sum::(); + let op = Operator::aggregate( + input.clone(), + vec![], + vec![ + ("count".into(), Reduction::Count), + ("min".into(), Reduction::Min(0)), + ("max".into(), Reduction::Max(0)), + ], + ) + .unwrap(); + let result = unary(input.clone(), batches, op); + assert_eq!(result.len(), 1); + assert!(matches!(result[0][0], Value::Int64(v) if v == n as i64)); + assert!(matches!(result[0][1], Value::Null)); + assert!(matches!(result[0][2], Value::Null)); + } + let op = Operator::aggregate( + input.clone(), + vec![0], + vec![("count".into(), Reduction::Count)], + ) + .unwrap(); + assert!(unary(input, vec![], op).is_empty()); +} + +// A precompiled expression with a different input contract must fail during binding. +#[test] +fn projection_rejects_expression_bound_to_another_schema() { + let original = schema(&[("a", DataType::Int64, false), ("b", DataType::Int64, false)]); + let current = schema(&[("a", DataType::Int64, false)]); + let expr = CompiledExpression::compile(&QueryExpr::Column(1), &original).unwrap(); + assert!(Operator::project(current, vec![("b".into(), Expression::planner(expr))]).is_err()); +} + +// A valid Planner MIN/MAX schema must bind even for a non-null input column. +#[test] +fn global_extrema_bind_with_planner_derived_schema() { + use asap_physical_operators::binding::bind_node; + use planner_types::{ + post_asap::*, + pre_asap::{AggIntent, Column, GroupKeys, Reduction as PlanReduction}, + }; + let input = schema(&[("v", DataType::Int64, false)]); + for measure in [ + AggIntent::Min { col: Some(0) }, + AggIntent::Max { col: Some(0) }, + ] { + let planner_input = + planner_types::pre_asap::Schema::new(vec![Column::new("v", DataType::Int64, false)]); + let derived = planner_types::pre_asap::query_expr::aggregate_output_schema( + &planner_input, + &PlanReduction::Reduce(GroupKeys::by(vec![])), + std::slice::from_ref(&measure), + &[], + ) + .unwrap(); + let result = derived.columns[0].clone(); + let output = schema(&[(&result.name, result.dtype, result.nullable)]); + let node = ExecutableDagNode { + id: PostAsapNodeId(1), + payload: ExecutableOperatorPayload::Value { + operation: ValueOperation::Exact(ExactOperation::Aggregate { + reduction: PlanReduction::Reduce(GroupKeys::by(vec![])), + measures: vec![measure], + output_names: vec![result.name], + having: None, + }), + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*output).clone(), + guarantee: None, + }; + let operator = bind_node(&node, std::slice::from_ref(&input)) + .expect("global extremum should bind to its Planner schema"); + assert!(operator.schema().fields[0].nullable); + let empty = unary(input.clone(), vec![], operator.clone()); + assert!(matches!(empty[0][0], Value::Null)); + let nonempty = unary(input.clone(), vec![vec![vec![Value::Int64(7)]]], operator); + assert!(matches!(nonempty[0][0], Value::Int64(7))); + } +} + +// NaN is a valid numeric input, not a schema error; all six comparisons obey IEEE rules. +#[test] +fn planner_comparisons_handle_nan_without_execution_errors() { + let input = schema(&[ + ("a", DataType::Float64, false), + ("b", DataType::Float64, false), + ]); + for op in [ + CompareOpKind::Eq, + CompareOpKind::Ne, + CompareOpKind::Lt, + CompareOpKind::Le, + CompareOpKind::Gt, + CompareOpKind::Ge, + ] { + let expression = QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: op.clone(), + right: Rc::new(QueryExpr::Column(1)), + }; + let compiled = CompiledExpression::compile(&expression, &input).unwrap(); + for row in [ + [Value::Float64(f64::NAN), Value::Float64(1.)], + [Value::Float64(1.), Value::Float64(f64::NAN)], + [Value::Float64(f64::NAN), Value::Float64(f64::NAN)], + ] { + let actual = compiled.evaluate(&row).unwrap(); + assert!(matches!(actual,Value::Bool(value) if value == (op == CompareOpKind::Ne))); + } + } +} + +// A bounded LIMIT branch must unsubscribe so another branch can drain the producer. +#[test] +fn limit_branch_finishes_without_blocking_shared_sibling() { + let input = schema(&[("v", DataType::Int64, false)]); + let mut dag = PhysicalDag::default(); + let batches = (0..100) + .map(|v| Batch::try_new(input.clone(), vec![vec![Value::Int64(v)]]).unwrap()) + .collect(); + dag.add(0, vec![], Operator::source(input.clone(), batches).unwrap()) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::limit(input.clone(), 1, 0, vec![]).unwrap(), + ) + .unwrap(); + dag.add(2, vec![0, 1], Operator::union(input, 2).unwrap()) + .unwrap(); + // Bound polls as well as rows so a backpressure regression cannot hang the suite. + use futures::{task::noop_waker_ref, Stream}; + use std::{ + pin::Pin, + task::{Context, Poll}, + }; + let run = context(); + let mut stream = dag.execute(&[2], run.clone()).unwrap().remove(0); + let mut cx = Context::from_waker(noop_waker_ref()); + let mut count = 0; + for _ in 0..2000 { + match Pin::new(&mut stream).poll_next(&mut cx) { + Poll::Ready(Some(batch)) => count += batch.unwrap().rows().len(), + Poll::Ready(None) => { + assert_eq!(count, 101); + drop(stream); + assert_eq!(run.retained_bytes(), 0); + return; + } + Poll::Pending => {} + } + } + panic!("shared LIMIT/Union failed to make progress"); +} + +// Mixed numeric comparisons must not round Int64 values through f64 before comparing. +#[test] +fn mixed_numeric_comparisons_preserve_large_integer_precision() { + let input = schema(&[ + ("a", DataType::Int64, false), + ("b", DataType::Float64, false), + ]); + let expr = QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Gt, + right: Rc::new(QueryExpr::Column(1)), + }; + let compiled = CompiledExpression::compile(&expr, &input).unwrap(); + for (a, b, expected) in [ + (9_007_199_254_740_993, 9_007_199_254_740_992.0, true), + (i64::MAX, 9_223_372_036_854_775_808.0, false), + (i64::MIN, f64::NEG_INFINITY, true), + ] { + assert!( + matches!(compiled.evaluate(&[Value::Int64(a),Value::Float64(b)]).unwrap(), Value::Bool(v) if v == expected) + ); + } +} + +// Both expression paths must implement all nine combinations of three-valued booleans. +#[test] +fn boolean_truth_tables_agree_between_expression_paths() { + let input = schema(&[("a", DataType::Bool, true), ("b", DataType::Bool, true)]); + for and in [true, false] { + for a in [None, Some(false), Some(true)] { + for b in [None, Some(false), Some(true)] { + let parts = vec![QueryExpr::Column(0), QueryExpr::Column(1)]; + let planner = if and { + QueryExpr::BoolAnd(parts) + } else { + QueryExpr::BoolOr(parts) + }; + let native = if and { + Expression::And( + Box::new(Expression::Column(0)), + Box::new(Expression::Column(1)), + ) + } else { + Expression::Or( + Box::new(Expression::Column(0)), + Box::new(Expression::Column(1)), + ) + }; + let expected = match (a, b, and) { + (Some(false), _, true) | (_, Some(false), true) => Some(false), + (Some(true), _, false) | (_, Some(true), false) => Some(true), + (None, _, _) | (_, None, _) => None, + (Some(a), Some(b), true) => Some(a && b), + (Some(a), Some(b), false) => Some(a || b), + } + .map(Value::Bool) + .unwrap_or(Value::Null); + let row = vec![ + a.map(Value::Bool).unwrap_or(Value::Null), + b.map(Value::Bool).unwrap_or(Value::Null), + ]; + let compiled = CompiledExpression::compile(&planner, &input).unwrap(); + assert_eq!( + compiled.evaluate(&row).unwrap().key().unwrap(), + expected.key().unwrap() + ); + let op = Operator::project(input.clone(), vec![("result".into(), native)]).unwrap(); + let result = unary(input.clone(), vec![vec![row]], op); + assert_eq!(result[0][0].key().unwrap(), expected.key().unwrap()); + } + } + } +} + +// Partial/final execution must agree with one build for an uncompacted KLL population. +#[test] +fn kll_partial_merge_and_multiple_readouts_preserve_population() { + use asap_physical_operators::Statistic; + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("v", DataType::Float64, false)]); + let family = SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), + Default::default(), + ); + let mut dag = PhysicalDag::default(); + for (id, range) in [(0, 0..64), (1, 64..128), (2, 0..128)] { + let rows = range.map(|n| vec![Value::Float64(n as f64)]).collect(); + dag.add( + id, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new(input.clone(), rows).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add( + id + 3, + vec![id], + Operator::summary_build(input.clone(), family.clone(), 0, None, vec![]).unwrap(), + ) + .unwrap(); + } + let state = Operator::summary_build(input, family, 0, None, vec![]) + .unwrap() + .schema(); + dag.add(6, vec![3, 4], Operator::union(state.clone(), 2).unwrap()) + .unwrap(); + dag.add( + 7, + vec![6], + Operator::summary_merge(state.clone(), 0, vec![]).unwrap(), + ) + .unwrap(); + let mut roots = vec![]; + for (i, q) in [0.0, 0.5, 1.0].into_iter().enumerate() { + for (j, build) in [5, 7].into_iter().enumerate() { + let id = 8 + (i * 2 + j) as u64; + dag.add( + id, + vec![build], + Operator::readout( + state.clone(), + 0, + Statistic::Quantile, + std::collections::HashMap::from([("quantile".into(), q.to_string())]), + ) + .unwrap(), + ) + .unwrap(); + roots.push(id); + } + } + for _ in 0..2 { + let run = context(); + let outputs = block_on(futures::future::join_all( + dag.execute(&roots, run.clone()) + .unwrap() + .into_iter() + .map(|s| s.collect::>()), + )); + for (pair, expected) in outputs.chunks(2).zip([0., 64., 127.]) { + let value = |batches: &[Result< + asap_physical_operators::runtime::SharedValue, + asap_physical_operators::Error, + >]| { + assert_eq!(batches.len(), 1); + match batches[0].as_ref().unwrap().rows()[0][0] { + Value::Float64(v) => v, + _ => panic!("quantile must be Float64"), + } + }; + assert_eq!(value(&pair[0]), value(&pair[1])); + assert!((value(&pair[0]) - expected).abs() <= 1.); + } + drop(outputs); + assert_eq!(run.retained_bytes(), 0); + } +} + +// Retained zero-column rows still own Vec headers and must consume the output budget. +#[test] +fn zero_column_output_obeys_memory_limit() { + use asap_physical_operators::Error; + let input = schema(&[]); + let batch = Batch::try_new(input.clone(), vec![vec![]; 200]).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(input, vec![batch]).unwrap()) + .unwrap(); + let run = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_bytes: 1024, + ..Limits::default() + }, + ) + .unwrap(); + let mut stream = dag.execute(&[0], run.clone()).unwrap().remove(0); + assert!(matches!( + block_on(stream.next()), + Some(Err(Error::MemoryLimit)) + )); + drop(stream); + assert_eq!(run.retained_bytes(), 0); +} + +// Empty exact-state finalization must preserve ordinary global MIN/MAX null semantics. +#[test] +fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { + use asap_physical_operators::Statistic; + use planner_types::post_asap::{ExactKind, ExactParams}; + let input = schema(&[("v", DataType::Float64, false)]); + for (kind, params, statistic) in [ + (ExactKind::Min, ExactParams::Min, Statistic::Min), + (ExactKind::Max, ExactParams::Max, Statistic::Max), + ] { + let build = Operator::summary_build( + input.clone(), + SummaryFamilyType::ExactAggregate(kind, params), + 0, + None, + vec![], + ) + .unwrap(); + let state = build.schema(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(input.clone(), vec![]).unwrap()) + .unwrap(); + dag.add(1, vec![0], build).unwrap(); + dag.add( + 2, + vec![1], + Operator::readout(state, 0, statistic, Default::default()).unwrap(), + ) + .unwrap(); + let rows = collect(&dag, 2); + assert_eq!(rows.len(), 1); + assert!(matches!(rows[0][0], Value::Null)); + } +} diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs index dcb73bee..87053ee1 100644 --- a/crates/types/src/pre_asap/query_expr.rs +++ b/crates/types/src/pre_asap/query_expr.rs @@ -1562,6 +1562,11 @@ pub fn aggregate_output_schema( .and_then(|id| in_schema.columns.get(*id)) .unwrap_or(&probe); let mut out = intent.output_column(in_col); + // A global extremum emits NULL for an empty input, even if its input + // column is non-nullable. Grouped extrema only emit existing groups. + if by.is_empty() && matches!(intent, AggIntent::Min { .. } | AggIntent::Max { .. }) { + out.nullable = true; + } if let Some((arg, _)) = intent .arg_selector_columns(in_schema) .map_err(QueryExprError::InvalidScalarSignature)? From 85fade34967afd47c1aea2354628447151b01b27 Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 18:44:29 +0000 Subject: [PATCH 18/90] refactor: name summary computation kernels explicitly --- crates/asap-physical-operators/README.md | 2 +- .../asap-physical-operators/src/capability.rs | 4 +--- crates/asap-physical-operators/src/lib.rs | 8 ++++---- .../src/operators/summary/mod.rs | 8 ++++---- .../src/stored_state/decoders.rs | 2 +- .../src/stored_state/delta_apply.rs | 20 +++++++++---------- .../count_min_sketch.rs | 4 ++-- .../count_min_sketch_with_heap.rs | 0 .../count_sketch.rs | 4 ++-- .../count_sketch_with_heap.rs | 2 +- .../datasketches_kll.rs | 2 +- .../dd_sketch.rs | 2 +- .../exact.rs | 0 .../factory.rs | 12 +++++------ .../hll_sketch.rs | 4 ++-- .../hydra_kll.rs | 0 .../increase.rs | 0 .../keyed_counter_state.rs | 2 +- .../keyed_max_state.rs | 0 .../keyed_min_state.rs | 0 .../keyed_sum_count.rs | 0 .../max.rs | 0 .../min.rs | 0 .../mod.rs | 1 + .../sketch_envelope.rs | 0 .../sum.rs | 0 .../traits.rs | 0 .../univmon.rs | 0 .../weighted_frequency.rs | 0 crates/asap-physical-operators/src/values.rs | 8 ++++---- .../tests/physical_dag.rs | 2 +- docs/design_docs/physical-operators.md | 2 +- 32 files changed, 44 insertions(+), 45 deletions(-) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/count_min_sketch.rs (99%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/count_min_sketch_with_heap.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/count_sketch.rs (99%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/count_sketch_with_heap.rs (99%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/datasketches_kll.rs (99%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/dd_sketch.rs (99%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/exact.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/factory.rs (99%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/hll_sketch.rs (99%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/hydra_kll.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/increase.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/keyed_counter_state.rs (99%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/keyed_max_state.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/keyed_min_state.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/keyed_sum_count.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/max.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/min.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/mod.rs (91%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/sketch_envelope.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/sum.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/traits.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/univmon.rs (100%) rename crates/asap-physical-operators/src/{summary_operators => summary_kernels}/weighted_frequency.rs (100%) diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index a526ee7c..2efa0672 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -78,7 +78,7 @@ See [the design](../../docs/design_docs/physical-operators.md). - `operators`: projection, filter, joins, aggregate/window, sort, limit and summary implementations. - `sources`: raw-source interface, Scan and the memory connector. - `binding`: Planner executable DAG binding and installed source frontiers. -- `summary_operators`: mathematical summary kernels, update adapters and accumulator traits. +- `summary_kernels`: mathematical summary kernels, update adapters and accumulator traits. - `stored_state`: persisted-state decoding, delta reconstruction and readout. - `capability`: explicit kernel and native-batch/readout validation. diff --git a/crates/asap-physical-operators/src/capability.rs b/crates/asap-physical-operators/src/capability.rs index eb6459e6..12ecabcd 100644 --- a/crates/asap-physical-operators/src/capability.rs +++ b/crates/asap-physical-operators/src/capability.rs @@ -144,9 +144,7 @@ pub fn validate_native_family(family: &SummaryFamilyType) -> Result<(), Error> { if let SummaryFamilyType::Sketch(kind, grouping) = family { if matches!(kind.algorithm(), A::CmsWithHeap | A::CountSketchWithHeap) { let (_, width, depth, _) = - crate::summary_operators::weighted_frequency::WeightedFrequency::configuration( - kind, - )?; + crate::summary_kernels::weighted_frequency::WeightedFrequency::configuration(kind)?; return if valid_matrix(width as u32, depth as u32) && grouping == &Default::default() { Ok(()) } else { diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs index 4c644bd7..dc726214 100644 --- a/crates/asap-physical-operators/src/lib.rs +++ b/crates/asap-physical-operators/src/lib.rs @@ -1,11 +1,11 @@ #![doc = include_str!("../README.md")] -pub mod summary_operators; +pub mod summary_kernels; /// Compatibility alias for existing deployments. -pub use summary_operators as accumulators; +pub use summary_kernels as accumulators; pub mod key_by_label_values; pub mod measurement; -pub use summary_operators::traits; +pub use summary_kernels::traits; mod aggregation_type; mod statistic; @@ -17,7 +17,7 @@ pub use traits::*; pub use expressions::arithmetic; pub mod capability; -pub use summary_operators::factory; +pub use summary_kernels::factory; /// The exact Planner contract used by these kernels. pub use planner_types as planner; diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs index 13e61ec0..f469f0a1 100644 --- a/crates/asap-physical-operators/src/operators/summary/mod.rs +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -7,7 +7,7 @@ impl Operator { items: Vec, groups: Vec, ) -> Result { - use crate::summary_operators::weighted_frequency::WeightedFrequency; + use crate::summary_kernels::weighted_frequency::WeightedFrequency; crate::values::validate_family(&family)?; let SummaryFamilyType::Sketch(kind, _) = &family else { return Err(invalid("keyed sketch required")); @@ -57,7 +57,7 @@ impl Operator { k: usize, output: Schema, ) -> Result { - use crate::summary_operators::weighted_frequency::WeightedFrequency; + use crate::summary_kernels::weighted_frequency::WeightedFrequency; crate::values::validate_family(&field(&input, state)?.dtype)?; let SummaryFamilyType::Sketch(kind, _) = &field(&input, state)?.dtype else { return Err(invalid("keyed readout requires summary state")); @@ -242,7 +242,7 @@ pub(super) fn execute<'a>( }; let summary = summary .as_any() - .downcast_ref::() + .downcast_ref::() .ok_or_else(|| invalid("weighted frequency typed state required"))?; for items in summary.rows(*k) { let mut values = row[..*state].to_vec(); @@ -478,7 +478,7 @@ async fn build_keyed_summary( groups: &[usize], context: &RunContext, ) -> Result>, Error> { - use crate::{summary_operators::weighted_frequency::WeightedFrequency, AggregateCore}; + use crate::{summary_kernels::weighted_frequency::WeightedFrequency, AggregateCore}; let SummaryFamilyType::Sketch(kind, _) = family else { unreachable!() }; diff --git a/crates/asap-physical-operators/src/stored_state/decoders.rs b/crates/asap-physical-operators/src/stored_state/decoders.rs index 8741d537..37928f8f 100644 --- a/crates/asap-physical-operators/src/stored_state/decoders.rs +++ b/crates/asap-physical-operators/src/stored_state/decoders.rs @@ -8,7 +8,7 @@ use asap_sketchlib::CountSketchWithHeap; use asap_sketchlib::CsHeapItem; use asap_sketchlib::MessagePackCodec; -use crate::summary_operators::count_min_sketch_with_heap::CountMinSketchWithHeapAccumulator; +use crate::summary_kernels::count_min_sketch_with_heap::CountMinSketchWithHeapAccumulator; /// Decode a `CountMinSketch` from the modified-OTLP wire bytes. /// MSGPACK path round-trips `CountMinSketch::deserialize_msgpack`; diff --git a/crates/asap-physical-operators/src/stored_state/delta_apply.rs b/crates/asap-physical-operators/src/stored_state/delta_apply.rs index a5c4a836..b8cf660e 100644 --- a/crates/asap-physical-operators/src/stored_state/delta_apply.rs +++ b/crates/asap-physical-operators/src/stored_state/delta_apply.rs @@ -84,7 +84,7 @@ impl DeltaSketchKind { sketch_cols, layers, } => SummaryState::UnivMon( - crate::summary_operators::univmon::UnivMonAccumulator::new( + crate::summary_kernels::univmon::UnivMonAccumulator::new( *heap_size as usize, *sketch_rows as usize, *sketch_cols as usize, @@ -136,7 +136,7 @@ fn decode_full( }, SketchEncoding::MsgpackFull, ) => { - let state = crate::summary_operators::univmon::UnivMonAccumulator::from_bytes(bytes) + let state = crate::summary_kernels::univmon::UnivMonAccumulator::from_bytes(bytes) .map_err(|e| e.to_string())?; if state.dimensions() != ( @@ -213,7 +213,7 @@ fn decode_full( /// folded across a window (or several) via delta application, or merged /// in from another sid's own reconstruction. pub enum SummaryState { - UnivMon(crate::summary_operators::univmon::UnivMonAccumulator), + UnivMon(crate::summary_kernels::univmon::UnivMonAccumulator), Dd(DdSketch), Hll(HllSketch), Kll(KllSketch), @@ -290,7 +290,7 @@ impl SummaryState { } // Shape (2): bucket-delta proto → additive apply via the // SAME decoder the ingest delta path uses. - use crate::summary_operators::dd_sketch::DDSketchAccumulator; + use crate::summary_kernels::dd_sketch::DDSketchAccumulator; let mut acc = DDSketchAccumulator { inner: std::mem::replace(sk, DdSketch::new(sk.alpha)), sample_p: 1.0, @@ -709,21 +709,21 @@ pub fn per_window_summary_states( // --------------------------------------------------------------------------- fn dd_from_proto(buffer: &[u8]) -> Result { - use crate::summary_operators::dd_sketch::DDSketchAccumulator; + use crate::summary_kernels::dd_sketch::DDSketchAccumulator; DDSketchAccumulator::from_sketchlib_proto_bytes(buffer) .map(|acc| acc.inner) .map_err(|e| e.to_string()) } fn kll_from_proto(buffer: &[u8]) -> Result { - use crate::summary_operators::datasketches_kll::DatasketchesKLLAccumulator; + use crate::summary_kernels::datasketches_kll::DatasketchesKLLAccumulator; DatasketchesKLLAccumulator::from_sketchlib_proto_bytes(buffer) .map(|acc| acc.inner) .map_err(|e| e.to_string()) } fn hll_from_proto(buffer: &[u8]) -> Result { - use crate::summary_operators::hll_sketch::HllSketchAccumulator; + use crate::summary_kernels::hll_sketch::HllSketchAccumulator; HllSketchAccumulator::from_sketchlib_proto_bytes(buffer) .map(|acc| acc.inner) .map_err(|e| e.to_string()) @@ -879,7 +879,7 @@ mod tests { fn hll_from_proto_matches_accumulator_decoder() { // P2-4: the warm read path and the ingest accumulator must decode // the SAME bytes to the SAME sketch (one source of truth). - use crate::summary_operators::hll_sketch::HllSketchAccumulator; + use crate::summary_kernels::hll_sketch::HllSketchAccumulator; let mut sk = HllSketch::new(HllVariant::Regular, 12); for i in 0..500u64 { sk.update(format!("item-{i}").as_bytes()); @@ -898,7 +898,7 @@ mod tests { #[test] fn dd_from_proto_matches_accumulator_decoder() { - use crate::summary_operators::dd_sketch::DDSketchAccumulator; + use crate::summary_kernels::dd_sketch::DDSketchAccumulator; let mut sk = DdSketch::new(0.01); for v in [1.0, 2.0, 5.0, 5.0, 9.0, 42.0] { sk.update(v); @@ -915,7 +915,7 @@ mod tests { #[test] fn kll_from_proto_matches_accumulator_decoder() { - use crate::summary_operators::datasketches_kll::DatasketchesKLLAccumulator; + use crate::summary_kernels::datasketches_kll::DatasketchesKLLAccumulator; let items: Vec = (0..200).map(|i| i as f64).collect(); let bytes = encode_kll(256, &items); let via_delta = kll_from_proto(&bytes).expect("delta_apply kll decode"); diff --git a/crates/asap-physical-operators/src/summary_operators/count_min_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/count_min_sketch.rs rename to crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs index 26a6ae96..cffe90d2 100644 --- a/crates/asap-physical-operators/src/summary_operators/count_min_sketch.rs +++ b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs @@ -1,4 +1,4 @@ -use crate::summary_operators::dd_sketch::normalize_sample_p; +use crate::summary_kernels::dd_sketch::normalize_sample_p; use crate::{ AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, MultipleSubpopulationAggregate, SerializableToSink, @@ -810,7 +810,7 @@ mod tests { let boxed_accs: Vec> = vec![Box::new(cms1), Box::new(cms2)]; assert!(CountMinSketchAccumulator::merge_multiple(&boxed_accs).is_err()); - use crate::summary_operators::sum::SumAccumulator; + use crate::summary_kernels::sum::SumAccumulator; let cms = CountMinSketchAccumulator::new(2, 3); let sum = SumAccumulator::new(); let mixed_accs: Vec> = vec![Box::new(cms), Box::new(sum)]; diff --git a/crates/asap-physical-operators/src/summary_operators/count_min_sketch_with_heap.rs b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/count_min_sketch_with_heap.rs rename to crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs diff --git a/crates/asap-physical-operators/src/summary_operators/count_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/count_sketch.rs rename to crates/asap-physical-operators/src/summary_kernels/count_sketch.rs index 9cfb4b24..78c1f7d7 100644 --- a/crates/asap-physical-operators/src/summary_operators/count_sketch.rs +++ b/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs @@ -96,7 +96,7 @@ impl CountSketchAccumulator { // ingest caller skips the data point) instead of building a // degenerate or huge matrix. Shares the CMS validator since the // CountSketch matrix uses the same packed-hash column layout. - crate::summary_operators::count_min_sketch::validate_sketch_dims( + crate::summary_kernels::count_min_sketch::validate_sketch_dims( "CountSketchState", rows, cols, @@ -554,7 +554,7 @@ mod tests { #[test] fn test_aggregate_core_merge_wrong_type_rejects() { - use crate::summary_operators::count_min_sketch::CountMinSketchAccumulator; + use crate::summary_kernels::count_min_sketch::CountMinSketchAccumulator; let cs = CountSketchAccumulator::new(2, 3); let cms = CountMinSketchAccumulator::new(2, 3); let result = cs.merge_with(&cms); diff --git a/crates/asap-physical-operators/src/summary_operators/count_sketch_with_heap.rs b/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/count_sketch_with_heap.rs rename to crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs index afb5e58b..1f614bcb 100644 --- a/crates/asap-physical-operators/src/summary_operators/count_sketch_with_heap.rs +++ b/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs @@ -561,7 +561,7 @@ mod tests { /// min-over-rows divergence at the sketch-math level). #[test] fn test_rejects_merge_with_cms_family_accumulator() { - use crate::summary_operators::count_min_sketch_with_heap::CountMinSketchWithHeapAccumulator; + use crate::summary_kernels::count_min_sketch_with_heap::CountMinSketchWithHeapAccumulator; let cs = CountSketchWithHeapAccumulator::new(4, 64, 10); let cms = CountMinSketchWithHeapAccumulator::new(4, 64, 10); diff --git a/crates/asap-physical-operators/src/summary_operators/datasketches_kll.rs b/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/datasketches_kll.rs rename to crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs index c031ddb1..1ab4df79 100644 --- a/crates/asap-physical-operators/src/summary_operators/datasketches_kll.rs +++ b/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs @@ -537,7 +537,7 @@ mod tests { let boxed_accs: Vec> = vec![Box::new(kll1), Box::new(kll2)]; assert!(DatasketchesKLLAccumulator::merge_multiple(&boxed_accs).is_err()); - use crate::summary_operators::sum::SumAccumulator; + use crate::summary_kernels::sum::SumAccumulator; let kll = DatasketchesKLLAccumulator::new(200); let sum = SumAccumulator::new(); let mixed_accs: Vec> = vec![Box::new(kll), Box::new(sum)]; diff --git a/crates/asap-physical-operators/src/summary_operators/dd_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/dd_sketch.rs rename to crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs index 3d55787b..a6f3b1eb 100644 --- a/crates/asap-physical-operators/src/summary_operators/dd_sketch.rs +++ b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs @@ -398,7 +398,7 @@ mod tests { #[test] fn test_aggregate_core_merge_wrong_type_rejects() { - use crate::summary_operators::count_sketch::CountSketchAccumulator; + use crate::summary_kernels::count_sketch::CountSketchAccumulator; let dd = DDSketchAccumulator::new(0.01); let cs = CountSketchAccumulator::new(2, 3); assert!(dd.merge_with(&cs).is_err()); diff --git a/crates/asap-physical-operators/src/summary_operators/exact.rs b/crates/asap-physical-operators/src/summary_kernels/exact.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/exact.rs rename to crates/asap-physical-operators/src/summary_kernels/exact.rs diff --git a/crates/asap-physical-operators/src/summary_operators/factory.rs b/crates/asap-physical-operators/src/summary_kernels/factory.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/factory.rs rename to crates/asap-physical-operators/src/summary_kernels/factory.rs index 8ea40b85..4cbeb8a5 100644 --- a/crates/asap-physical-operators/src/summary_operators/factory.rs +++ b/crates/asap-physical-operators/src/summary_kernels/factory.rs @@ -1,4 +1,4 @@ -use crate::summary_operators::{ +use crate::summary_kernels::{ CountMinSketchAccumulator, CountMinSketchWithHeapAccumulator, CountSketchAccumulator, CountSketchWithHeapAccumulator, DDSketchAccumulator, DatasketchesKLLAccumulator, HydraKllSketchAccumulator, IncreaseAccumulator, KeyedCounterState, KeyedMaxState, @@ -7,8 +7,8 @@ use crate::summary_operators::{ use crate::{AggregateCore, KeyByLabelValues, Measurement}; // Production dispatch consumes Planner SummaryAgg payloads directly. The // config adapter below is compiled only for isolated historical kernel tests. -use crate::summary_operators::hll_sketch::HllSketchAccumulator; -use crate::summary_operators::univmon::UnivMonAccumulator; +use crate::summary_kernels::hll_sketch::HllSketchAccumulator; +use crate::summary_kernels::univmon::UnivMonAccumulator; use planner_types::post_asap::{ExactKind, SketchAlgorithm, SketchParams, SummaryFamilyType}; /// Generate the two boilerplate clone-based `AccumulatorUpdater` methods @@ -913,7 +913,7 @@ pub fn create_planner_accumulator( } if matches!(family, SummaryFamilyType::ExactAggregate(..)) { return Ok(Box::new(PlannerExactUpdater { - acc: crate::summary_operators::exact::ExactAccumulator::new( + acc: crate::summary_kernels::exact::ExactAccumulator::new( family.clone(), input.item.is_some(), )?, @@ -994,7 +994,7 @@ pub fn create_planner_accumulator( } struct PlannerExactUpdater { - acc: crate::summary_operators::exact::ExactAccumulator, + acc: crate::summary_kernels::exact::ExactAccumulator, } impl AccumulatorUpdater for PlannerExactUpdater { fn update_single(&mut self, value: f64, timestamp: i64) { @@ -1005,7 +1005,7 @@ impl AccumulatorUpdater for PlannerExactUpdater { } impl_clone_accumulator_methods!(acc); fn reset(&mut self) { - self.acc = crate::summary_operators::exact::ExactAccumulator::new( + self.acc = crate::summary_kernels::exact::ExactAccumulator::new( self.acc.family().clone(), self.acc.is_keyed(), ) diff --git a/crates/asap-physical-operators/src/summary_operators/hll_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/hll_sketch.rs rename to crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs index efe15baf..4c16156d 100644 --- a/crates/asap-physical-operators/src/summary_operators/hll_sketch.rs +++ b/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs @@ -11,7 +11,7 @@ //! registers + variant + HIP accumulators losslessly, so the merge + //! store round-trip works end-to-end without that richer query surface. -use crate::summary_operators::dd_sketch::normalize_sample_p; +use crate::summary_kernels::dd_sketch::normalize_sample_p; use crate::{AggregateCore, AggregationType, KeyByLabelValues, SerializableToSink}; use asap_sketchlib::{HllSketch, HllVariant, MessagePackCodec}; use serde_json::Value; @@ -567,7 +567,7 @@ mod tests { #[test] fn test_aggregate_core_merge_wrong_type_rejects() { - use crate::summary_operators::count_sketch::CountSketchAccumulator; + use crate::summary_kernels::count_sketch::CountSketchAccumulator; let hll = HllSketchAccumulator::new(HllVariant::Regular, 2); let cs = CountSketchAccumulator::new(2, 3); assert!(hll.merge_with(&cs).is_err()); diff --git a/crates/asap-physical-operators/src/summary_operators/hydra_kll.rs b/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/hydra_kll.rs rename to crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs diff --git a/crates/asap-physical-operators/src/summary_operators/increase.rs b/crates/asap-physical-operators/src/summary_kernels/increase.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/increase.rs rename to crates/asap-physical-operators/src/summary_kernels/increase.rs diff --git a/crates/asap-physical-operators/src/summary_operators/keyed_counter_state.rs b/crates/asap-physical-operators/src/summary_kernels/keyed_counter_state.rs similarity index 99% rename from crates/asap-physical-operators/src/summary_operators/keyed_counter_state.rs rename to crates/asap-physical-operators/src/summary_kernels/keyed_counter_state.rs index 9411088f..8db0ee7e 100644 --- a/crates/asap-physical-operators/src/summary_operators/keyed_counter_state.rs +++ b/crates/asap-physical-operators/src/summary_kernels/keyed_counter_state.rs @@ -1,4 +1,4 @@ -use crate::summary_operators::IncreaseAccumulator; +use crate::summary_kernels::IncreaseAccumulator; use crate::{ AggregateCore, AggregationType, KeyByLabelValues, MergeableAccumulator, MultipleSubpopulationAggregate, SerializableToSink, SingleSubpopulationAggregate, diff --git a/crates/asap-physical-operators/src/summary_operators/keyed_max_state.rs b/crates/asap-physical-operators/src/summary_kernels/keyed_max_state.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/keyed_max_state.rs rename to crates/asap-physical-operators/src/summary_kernels/keyed_max_state.rs diff --git a/crates/asap-physical-operators/src/summary_operators/keyed_min_state.rs b/crates/asap-physical-operators/src/summary_kernels/keyed_min_state.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/keyed_min_state.rs rename to crates/asap-physical-operators/src/summary_kernels/keyed_min_state.rs diff --git a/crates/asap-physical-operators/src/summary_operators/keyed_sum_count.rs b/crates/asap-physical-operators/src/summary_kernels/keyed_sum_count.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/keyed_sum_count.rs rename to crates/asap-physical-operators/src/summary_kernels/keyed_sum_count.rs diff --git a/crates/asap-physical-operators/src/summary_operators/max.rs b/crates/asap-physical-operators/src/summary_kernels/max.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/max.rs rename to crates/asap-physical-operators/src/summary_kernels/max.rs diff --git a/crates/asap-physical-operators/src/summary_operators/min.rs b/crates/asap-physical-operators/src/summary_kernels/min.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/min.rs rename to crates/asap-physical-operators/src/summary_kernels/min.rs diff --git a/crates/asap-physical-operators/src/summary_operators/mod.rs b/crates/asap-physical-operators/src/summary_kernels/mod.rs similarity index 91% rename from crates/asap-physical-operators/src/summary_operators/mod.rs rename to crates/asap-physical-operators/src/summary_kernels/mod.rs index 2e35ed6f..72b7f1dc 100644 --- a/crates/asap-physical-operators/src/summary_operators/mod.rs +++ b/crates/asap-physical-operators/src/summary_kernels/mod.rs @@ -1,3 +1,4 @@ +//! Summary algorithms and state; DAG execution adapters live in `operators::summary`. pub mod count_min_sketch; pub mod count_min_sketch_with_heap; pub mod count_sketch; diff --git a/crates/asap-physical-operators/src/summary_operators/sketch_envelope.rs b/crates/asap-physical-operators/src/summary_kernels/sketch_envelope.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/sketch_envelope.rs rename to crates/asap-physical-operators/src/summary_kernels/sketch_envelope.rs diff --git a/crates/asap-physical-operators/src/summary_operators/sum.rs b/crates/asap-physical-operators/src/summary_kernels/sum.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/sum.rs rename to crates/asap-physical-operators/src/summary_kernels/sum.rs diff --git a/crates/asap-physical-operators/src/summary_operators/traits.rs b/crates/asap-physical-operators/src/summary_kernels/traits.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/traits.rs rename to crates/asap-physical-operators/src/summary_kernels/traits.rs diff --git a/crates/asap-physical-operators/src/summary_operators/univmon.rs b/crates/asap-physical-operators/src/summary_kernels/univmon.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/univmon.rs rename to crates/asap-physical-operators/src/summary_kernels/univmon.rs diff --git a/crates/asap-physical-operators/src/summary_operators/weighted_frequency.rs b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_operators/weighted_frequency.rs rename to crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs index 93019f62..d550a826 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -260,7 +260,7 @@ pub(crate) fn group_key(row: &[Value], columns: &[usize]) -> Result> pub(crate) use crate::capability::validate_native_family as validate_family; fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Result<(), Error> { - use crate::summary_operators::{ + use crate::summary_kernels::{ datasketches_kll::DatasketchesKLLAccumulator, dd_sketch::DDSketchAccumulator, exact::ExactAccumulator, hll_sketch::HllSketchAccumulator, }; @@ -273,7 +273,7 @@ fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Resu SketchParams::CmsWithHeap { .. } | SketchParams::CountSketchWithHeap { .. } ) => { - use crate::summary_operators::weighted_frequency::WeightedFrequency; + use crate::summary_kernels::weighted_frequency::WeightedFrequency; let (algorithm, width, depth, capacity) = WeightedFrequency::configuration(kind)?; state .as_any() @@ -296,7 +296,7 @@ fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Resu ) ) && state .as_any() - .is::()) + .is::()) } SummaryFamilyType::Sketch(kind, _) => match kind.params() { SketchParams::Kll { k } => state @@ -365,7 +365,7 @@ pub(crate) fn plain(schema: &Schema, column: usize) -> Result<(&DataType, bool), #[cfg(test)] mod weighted_state_tests { use super::*; - use crate::summary_operators::weighted_frequency::{FrequencyAlgorithm, WeightedFrequency}; + use crate::summary_kernels::weighted_frequency::{FrequencyAlgorithm, WeightedFrequency}; use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; // A state cannot acquire a different family or shape merely by relabeling its batch. diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 79aae966..7eb9c799 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -398,7 +398,7 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { // Restored state must retain its family; a mislabeled state is rejected. #[test] fn restored_exact_state_and_family_validation() { - use asap_physical_operators::{summary_operators::exact::ExactAccumulator, SerializableToSink}; + use asap_physical_operators::{summary_kernels::exact::ExactAccumulator, SerializableToSink}; let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); acc.update(None, 7., 0); diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index 040664d4..67678fc1 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -38,7 +38,7 @@ src/ source.rs literals, batch sources, union, scalar conversion sources/ raw-source API, Scan and memory connector binding/ Planner executable DAG → physical operators - summary_operators/ mathematical kernels, factory and traits + summary_kernels/ mathematical kernels, factory and traits stored_state/ decoding, delta application, persisted-state readout capability.rs support checks values.rs typed rows and state validation From f06556ceaccea6e618a268a8826632d39a67606e Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 19:01:58 +0000 Subject: [PATCH 19/90] refactor: delegate weighted frequency algorithms to sketchlib --- Cargo.lock | 7 +- crates/asap-physical-operators/Cargo.toml | 2 +- crates/asap-physical-operators/README.md | 2 +- .../src/stored_state/delta_apply.rs | 1 + .../src/summary_kernels/dd_sketch.rs | 75 +++- .../src/summary_kernels/mod.rs | 2 +- .../src/summary_kernels/weighted_frequency.rs | 324 +++--------------- crates/asap_sketch_codec/Cargo.toml | 2 +- crates/asap_sketch_codec/src/lib.rs | 6 +- docs/design_docs/physical-operators.md | 4 +- 10 files changed, 132 insertions(+), 293 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 5a54ee8a..f5ac50fd 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -382,7 +382,7 @@ dependencies = [ "asap-frontend-promql", "asap-types", "asap_sketch_codec", - "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=026cd18c7b8c23ae6c46d4d683151ba562b8cd3a)", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369)", "base64 0.21.7", "bincode", "futures", @@ -413,15 +413,16 @@ dependencies = [ name = "asap_sketch_codec" version = "0.1.0" dependencies = [ - "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=026cd18c7b8c23ae6c46d4d683151ba562b8cd3a)", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369)", "prost", ] [[package]] name = "asap_sketchlib" version = "0.3.0" -source = "git+https://github.com/ProjectASAP/asap_sketchlib?rev=026cd18c7b8c23ae6c46d4d683151ba562b8cd3a#026cd18c7b8c23ae6c46d4d683151ba562b8cd3a" +source = "git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369#5f03ccbd798ed5fec62bdd839bcb331123cab369" dependencies = [ + "bincode", "bytes", "prost", "rand 0.9.5", diff --git a/crates/asap-physical-operators/Cargo.toml b/crates/asap-physical-operators/Cargo.toml index e2314c77..e8ab710a 100644 --- a/crates/asap-physical-operators/Cargo.toml +++ b/crates/asap-physical-operators/Cargo.toml @@ -7,7 +7,7 @@ edition = "2021" futures = "0.3" planner-types = { package = "asap-types", path = "../types" } asap_sketch_codec = { path = "../asap_sketch_codec" } -asap_sketchlib = { git = "https://github.com/ProjectASAP/asap_sketchlib", rev = "026cd18c7b8c23ae6c46d4d683151ba562b8cd3a" } +asap_sketchlib = { git = "https://github.com/ProjectASAP/asap_sketchlib", rev = "5f03ccbd798ed5fec62bdd839bcb331123cab369" } serde = { version = "1", features = ["derive", "rc"] } serde_json = "1" tracing = "0.1" diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index 2efa0672..b6f9a009 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -78,7 +78,7 @@ See [the design](../../docs/design_docs/physical-operators.md). - `operators`: projection, filter, joins, aggregate/window, sort, limit and summary implementations. - `sources`: raw-source interface, Scan and the memory connector. - `binding`: Planner executable DAG binding and installed source frontiers. -- `summary_kernels`: mathematical summary kernels, update adapters and accumulator traits. +- `summary_kernels`: sketchlib state adapters, exact accumulators, update adapters and traits. - `stored_state`: persisted-state decoding, delta reconstruction and readout. - `capability`: explicit kernel and native-batch/readout validation. diff --git a/crates/asap-physical-operators/src/stored_state/delta_apply.rs b/crates/asap-physical-operators/src/stored_state/delta_apply.rs index b8cf660e..c0daf398 100644 --- a/crates/asap-physical-operators/src/stored_state/delta_apply.rs +++ b/crates/asap-physical-operators/src/stored_state/delta_apply.rs @@ -757,6 +757,7 @@ mod tests { alpha: sk.alpha, store_counts: sk.store_counts.clone(), store_offset: sk.store_offset, + ..Default::default() }; SketchEnvelope { sketch_state: Some(sketch_envelope::SketchState::Ddsketch(state)), diff --git a/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs index a6f3b1eb..1c25d940 100644 --- a/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs +++ b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs @@ -97,13 +97,8 @@ impl DDSketchAccumulator { ) .into()); } - // The DataPoint-level METRIC scalars (count/sum/min/max) were - // dropped from `DDSketchState` (ProjectASAP/sketchlib-go#243 / - // asap_sketchlib#57). Reconstruct from the bucket store only: - // `DdSketch::from_raw` now takes just (alpha, store_counts, - // store_offset) and recovers `count` by summing the bucket - // counts via `total_count()`. - let inner = DdSketch::from_raw(state.alpha, state.store_counts.clone(), state.store_offset); + // Preserve positive, negative and zero stores from the sketchlib wire state. + let inner = DdSketch::from_proto(state); Ok(Self { inner, sample_p: normalize_sample_p(sample_p), @@ -138,6 +133,12 @@ impl DDSketchAccumulator { .collect(); let delta = DdSketchDelta { buckets, + negative_buckets: pb + .negative_buckets + .into_iter() + .map(|b| (b.index, b.d_count)) + .collect(), + zero_count: pb.zero_count, ..Default::default() }; self.inner @@ -309,6 +310,7 @@ mod tests { alpha, store_counts, store_offset, + ..Default::default() }; SketchEnvelope { sketch_state: Some(sketch_envelope::SketchState::Ddsketch(state)), @@ -348,6 +350,7 @@ mod tests { alpha: 0.01, store_counts: vec![1, 2, 3, 4], store_offset: -2, + ..Default::default() }; let env = SketchEnvelope { sketch_state: Some(sketch_envelope::SketchState::Ddsketch(state)), @@ -443,6 +446,7 @@ mod tests { d_count: 20, }, ], + ..Default::default() } .encode_to_vec(); @@ -464,6 +468,7 @@ mod tests { index: i32::MAX, d_count: 1, }], + ..Default::default() } .encode_to_vec(); assert!(acc.apply_proto_delta_bytes(&bytes).is_err()); @@ -591,6 +596,7 @@ mod tests { alpha: 0.01, store_counts: vec![2, 4, 6, 8], store_offset: -2, + ..Default::default() })), ..Default::default() }; @@ -624,6 +630,7 @@ mod tests { alpha: 0.01, store_counts: vec![1, 2, 3], store_offset: 0, + ..Default::default() } .encode_to_vec(); assert_eq!( @@ -663,3 +670,57 @@ mod tests { assert_eq!(merged.sample_p, 0.1); } } + +#[cfg(test)] +mod dependency_upgrade_tests { + use super::*; + // The upgraded sketchlib state must retain negative and zero stores through both adapters. + #[test] + fn signed_state_survives_codec_and_accumulator_roundtrip() { + let mut inner = DdSketch::new(0.01); + for value in [-4.0, 0.0, 8.0] { + inner.update(value); + } + let bytes = asap_sketch_codec::encode_ddsketch(&inner); + let (wire, _) = asap_sketch_codec::ddsketch_state(&bytes).unwrap(); + assert_eq!(wire.zero_count, 1); + assert_eq!(wire.negative_store_counts.iter().sum::(), 1); + let restored = DDSketchAccumulator::from_sketchlib_proto_bytes(&bytes).unwrap(); + assert_eq!(restored.inner.total_count(), 3); + assert_eq!(restored.inner.alpha, inner.wire_alpha()); + assert_eq!(restored.inner.store_counts, inner.store_counts); + assert_eq!(restored.inner.store_offset, inner.store_offset); + assert_eq!( + restored.inner.negative_store_counts, + inner.negative_store_counts + ); + assert_eq!( + restored.inner.negative_store_offset, + inner.negative_store_offset + ); + assert_eq!(restored.inner.zero_count, inner.zero_count); + } + // Negative and zero delta fields added by sketchlib must not be discarded by the adapter. + #[test] + fn signed_delta_survives_adapter() { + use asap_sketchlib::proto::sketchlib::{DdSketchBucketDelta, DdSketchDelta as PbDelta}; + use prost::Message; + let mut accumulator = DDSketchAccumulator::new(0.01); + let bytes = PbDelta { + negative_buckets: vec![DdSketchBucketDelta { + index: 0, + d_count: 2, + }], + zero_count: 3, + ..Default::default() + } + .encode_to_vec(); + accumulator.apply_proto_delta_bytes(&bytes).unwrap(); + assert_eq!(accumulator.inner.total_count(), 5); + assert_eq!(accumulator.inner.zero_count, 3); + assert_eq!( + accumulator.inner.negative_store_counts.iter().sum::(), + 2 + ); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/mod.rs b/crates/asap-physical-operators/src/summary_kernels/mod.rs index 72b7f1dc..8ebc5d55 100644 --- a/crates/asap-physical-operators/src/summary_kernels/mod.rs +++ b/crates/asap-physical-operators/src/summary_kernels/mod.rs @@ -1,4 +1,4 @@ -//! Summary algorithms and state; DAG execution adapters live in `operators::summary`. +//! ASAP state adapters and exact accumulators; sketch algorithms live in `asap_sketchlib`. pub mod count_min_sketch; pub mod count_min_sketch_with_heap; pub mod count_sketch; diff --git a/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs index d78dc116..042dce65 100644 --- a/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs +++ b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs @@ -1,87 +1,44 @@ -//! Float64 weighted CMS and CountSketch state with typed candidate identities. Each instance -//! represents one partition at one evaluation scope; updates never round rates -//! to integer counts. Candidate membership still requires Planner evidence. +//! ASAP type and trait adapter for sketchlib's Float64 weighted frequency kernel. use crate::{values::Value, Error}; use crate::{AggregateCore, AggregationType, KeyByLabelValues, SerializableToSink, Statistic}; +pub use asap_sketchlib::FrequencyAlgorithm; +use asap_sketchlib::{FrequencyIdentity, WeightedFrequency as Kernel, WeightedFrequencyError}; use serde::{Deserialize, Serialize}; -use std::{ - cmp::Ordering, - collections::{BinaryHeap, HashMap}, - sync::Arc, -}; +use std::collections::HashMap; -#[derive(Clone, Debug, Serialize, Deserialize)] -enum Identity { - Null, - Bool(bool), - Int64(i64), - Float64(f64), - Utf8(String), -} -impl Identity { - fn from_value(value: &Value) -> Result { - Ok(match value { - Value::Null => Self::Null, - Value::Bool(v) => Self::Bool(*v), - Value::Int64(v) => Self::Int64(*v), - Value::Float64(v) if v.is_finite() => Self::Float64(if *v == 0.0 { 0.0 } else { *v }), - Value::Utf8(v) => Self::Utf8(v.to_string()), - _ => { - return Err(Error::Invalid( - "unsupported weighted frequency identity".into(), - )) - } - }) - } - fn value(&self) -> Value { - match self { - Self::Null => Value::Null, - Self::Bool(v) => Value::Bool(*v), - Self::Int64(v) => Value::Int64(*v), - Self::Float64(v) => Value::Float64(*v), - Self::Utf8(v) => Value::Utf8(Arc::from(v.as_str())), - } - } -} -#[derive(Clone, Debug, Serialize, Deserialize)] -struct Candidate { - identity: Vec, - key: Vec, - score: f64, -} -impl PartialEq for Candidate { - fn eq(&self, other: &Self) -> bool { - self.cmp(other) == Ordering::Equal +fn adapt_error(error: WeightedFrequencyError) -> Error { + match error { + WeightedFrequencyError::Invalid(message) => Error::Invalid(message), + WeightedFrequencyError::Update(message) => Error::Operator(message), } } -impl Eq for Candidate {} -impl PartialOrd for Candidate { - fn partial_cmp(&self, other: &Self) -> Option { - Some(self.cmp(other)) - } +fn identity(value: &Value) -> Result { + Ok(match value { + Value::Null => FrequencyIdentity::Null, + Value::Bool(v) => FrequencyIdentity::Bool(*v), + Value::Int64(v) => FrequencyIdentity::Int64(*v), + Value::Float64(v) => FrequencyIdentity::Float64(*v), + Value::Utf8(v) => FrequencyIdentity::Utf8(v.to_string()), + _ => { + return Err(Error::Invalid( + "unsupported weighted frequency identity".into(), + )) + } + }) } -impl Ord for Candidate { - fn cmp(&self, other: &Self) -> Ordering { - // The weakest candidate is the root of this bounded min-heap. - other - .score - .total_cmp(&self.score) - .then_with(|| other.key.cmp(&self.key)) +fn value(identity: FrequencyIdentity) -> Value { + match identity { + FrequencyIdentity::Null => Value::Null, + FrequencyIdentity::Bool(v) => Value::Bool(v), + FrequencyIdentity::Int64(v) => Value::Int64(v), + FrequencyIdentity::Float64(v) => Value::Float64(v), + FrequencyIdentity::Utf8(v) => Value::Utf8(v.into()), } } -#[derive(Copy, Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -pub enum FrequencyAlgorithm { - Cms, - CountSketch, -} #[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(transparent)] pub struct WeightedFrequency { - algorithm: FrequencyAlgorithm, - width: usize, - depth: usize, - capacity: usize, - cells: Vec, - candidates: BinaryHeap, + inner: Kernel, } impl WeightedFrequency { pub(crate) fn configuration( @@ -120,10 +77,10 @@ impl WeightedFrequency { } pub(crate) fn algorithm(&self) -> FrequencyAlgorithm { - self.algorithm + self.inner.algorithm() } pub(crate) fn shape(&self) -> (usize, usize, usize) { - (self.width, self.depth, self.capacity) + self.inner.shape() } pub fn new( algorithm: FrequencyAlgorithm, @@ -131,148 +88,26 @@ impl WeightedFrequency { depth: usize, capacity: usize, ) -> Result { - let len = width - .checked_mul(depth) - .filter(|_| { - width > 0 - && depth > 0 - && capacity > 0 - && (algorithm != FrequencyAlgorithm::CountSketch || depth % 2 == 1) - }) - .ok_or_else(|| Error::Invalid("invalid weighted frequency dimensions".into()))?; - let mut cells = Vec::new(); - cells - .try_reserve_exact(len) - .map_err(|_| Error::Invalid("weighted frequency allocation failed".into()))?; - cells.resize(len, 0.0); - Ok(Self { - algorithm, - width, - depth, - capacity, - cells, - candidates: BinaryHeap::new(), - }) + Kernel::new(algorithm, width, depth, capacity) + .map(|inner| Self { inner }) + .map_err(adapt_error) } - /// Decode only this kernel's versioned Float64 representation. Integer CMS - /// wire frames are different representations and are not accepted here. pub fn from_bytes(bytes: &[u8]) -> Result { - use bincode::Options; - let bytes = bytes - .strip_prefix(b"ASAP-WFREQ-1\0") - .ok_or_else(|| Error::Invalid("weighted frequency format/version mismatch".into()))?; - let mut state: Self = bincode::DefaultOptions::new() - .with_fixint_encoding() - .with_limit(bytes.len() as u64) - .reject_trailing_bytes() - .deserialize(bytes) - .map_err(|e| Error::Invalid(e.to_string()))?; - if state.width == 0 - || state.depth == 0 - || state.capacity == 0 - || (state.algorithm == FrequencyAlgorithm::CountSketch && state.depth.is_multiple_of(2)) - || state.width.checked_mul(state.depth) != Some(state.cells.len()) - || state - .cells - .iter() - .any(|v| !v.is_finite() || (state.algorithm == FrequencyAlgorithm::Cms && *v < 0.0)) - || state.candidates.len() > state.capacity - { - return Err(Error::Invalid("invalid weighted frequency state".into())); - } - for candidate in &state.candidates { - if candidate - .identity - .iter() - .any(|v| matches!(v, Identity::Float64(n) if !n.is_finite())) - || bincode::serialize(&candidate.identity) - .map_err(|e| Error::Invalid(e.to_string()))? - != candidate.key - { - return Err(Error::Invalid("invalid weighted frequency identity".into())); - } - } - state.retain(state.candidates.iter().cloned().collect()); - Ok(state) - } - fn indexes<'a>(&'a self, key: &'a [u8]) -> impl Iterator + 'a { - (0..self.depth).map(move |row| { - let bucket = - (xxhash_rust::xxh64::xxh64(key, 2 * row as u64) % self.width as u64) as usize; - let sign = if self.algorithm == FrequencyAlgorithm::CountSketch - && xxhash_rust::xxh64::xxh64(key, 2 * row as u64 + 1) & 1 != 0 - { - -1.0 - } else { - 1.0 - }; - (row * self.width + bucket, sign) - }) - } - fn estimate(&self, key: &[u8]) -> f64 { - let mut estimates = self - .indexes(key) - .map(|(i, sign)| self.cells[i] * sign) - .collect::>(); - match self.algorithm { - FrequencyAlgorithm::Cms => estimates.into_iter().fold(f64::INFINITY, f64::min), - FrequencyAlgorithm::CountSketch => { - let middle = estimates.len() / 2; - *estimates.select_nth_unstable_by(middle, f64::total_cmp).1 - } - } - } - fn retain(&mut self, mut candidates: Vec) { - candidates.sort_by(|a, b| a.key.cmp(&b.key)); - candidates.dedup_by(|a, b| a.key == b.key); - self.candidates.clear(); - for mut candidate in candidates { - candidate.score = self.estimate(&candidate.key); - self.candidates.push(candidate); - if self.candidates.len() > self.capacity { - self.candidates.pop(); - } - } + Kernel::from_bytes(bytes) + .map(|inner| Self { inner }) + .map_err(adapt_error) } pub fn update(&mut self, values: &[Value], weight: f64) -> Result<(), Error> { - if !weight.is_finite() || (self.algorithm == FrequencyAlgorithm::Cms && weight < 0.0) { - return Err(Error::Operator( - "weighted frequency requires finite weights; CMS additionally requires nonnegative weights".into(), - )); - } - let identity = values - .iter() - .map(Identity::from_value) - .collect::, _>>()?; - let key = bincode::serialize(&identity).map_err(|e| Error::Operator(e.to_string()))?; - let indexes = self.indexes(&key).collect::>(); - if indexes - .iter() - .any(|&(i, sign)| !(self.cells[i] + sign * weight).is_finite()) - { - return Err(Error::Operator("weighted frequency sum overflow".into())); - } - for (i, sign) in indexes { - self.cells[i] += sign * weight; - } - let mut candidates = self.candidates.iter().cloned().collect::>(); - candidates.push(Candidate { - identity, - key, - score: 0.0, - }); - self.retain(candidates); - Ok(()) + let values = values.iter().map(identity).collect::, _>>()?; + self.inner.update(&values, weight).map_err(adapt_error) } pub fn rows(&self, n: usize) -> Vec> { - let mut candidates = self.candidates.iter().collect::>(); - candidates.sort_by(|a, b| b.score.total_cmp(&a.score).then_with(|| a.key.cmp(&b.key))); - candidates + self.inner + .topk(n) .into_iter() - .take(n) - .map(|c| { - let mut row = c.identity.iter().map(Identity::value).collect::>(); - row.push(Value::Float64(c.score)); + .map(|(items, score)| { + let mut row = items.into_iter().map(value).collect::>(); + row.push(Value::Float64(score)); row }) .collect() @@ -283,9 +118,7 @@ impl SerializableToSink for WeightedFrequency { serde_json::to_value(self).expect("finite validated frequency state") } fn serialize_to_bytes(&self) -> Vec { - let mut bytes = b"ASAP-WFREQ-1\0".to_vec(); - bytes.extend(bincode::serialize(self).expect("serializable frequency state")); - bytes + self.inner.to_bytes() } } impl AggregateCore for WeightedFrequency { @@ -309,27 +142,12 @@ impl AggregateCore for WeightedFrequency { .as_any() .downcast_ref::() .ok_or("weighted frequency state type mismatch")?; - if self.algorithm != other.algorithm || self.shape() != other.shape() { - return Err("weighted frequency shape mismatch".into()); - } - let mut result = self.clone(); - for (value, rhs) in result.cells.iter_mut().zip(&other.cells) { - *value += rhs; - if !value.is_finite() { - return Err("weighted frequency merge overflow".into()); - } - } - result.retain( - self.candidates - .iter() - .chain(&other.candidates) - .cloned() - .collect(), - ); - Ok(Box::new(result)) + Ok(Box::new(Self { + inner: self.inner.merge(&other.inner)?, + })) } fn get_accumulator_type(&self) -> AggregationType { - match self.algorithm { + match self.inner.algorithm() { FrequencyAlgorithm::Cms => AggregationType::CountMinSketchWithHeap, FrequencyAlgorithm::CountSketch => AggregationType::CountSketchWithHeap, } @@ -346,30 +164,9 @@ impl AggregateCore for WeightedFrequency { Err("weighted frequency uses typed row readout".into()) } fn approx_memory_bytes(&self) -> usize { - std::mem::size_of::() - + self.cells.capacity() * 8 - + self - .candidates - .iter() - .map(|c| { - std::mem::size_of::() - + c.key.capacity() - + c.identity - .iter() - .map(|v| { - std::mem::size_of::() - + if let Identity::Utf8(s) = v { - s.capacity() - } else { - 0 - } - }) - .sum::() - }) - .sum::() + self.inner.approx_memory_bytes() } } - #[cfg(test)] mod tests { use super::*; @@ -400,25 +197,6 @@ mod tests { } } - // CountSketch uses sign-corrected median: one corrupted row cannot dominate five rows. - #[test] - fn count_sketch_median_and_odd_depth_contract() { - assert!(WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 8, 2, 2).is_err()); - let mut state = WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 16, 5, 8).unwrap(); - state.update(&[Value::Int64(4)], -0.375).unwrap(); - let key = state.candidates.peek().unwrap().key.clone(); - let indexes = state.indexes(&key).collect::>(); - for &(index, sign) in &indexes { - assert_eq!(state.cells[index] * sign, -0.375); - } - state.cells[indexes[0].0] += 1000.0; - assert_eq!(state.estimate(&key), -0.375); - let mut invalid = state.clone(); - invalid.depth = 4; - invalid.cells.truncate(64); - assert!(WeightedFrequency::from_bytes(&invalid.serialize_to_bytes()).is_err()); - } - // Invalid rates must not mutate state; typed keys cannot collide by formatting. #[test] fn fractional_updates_typed_identities_and_invalid_weights() { diff --git a/crates/asap_sketch_codec/Cargo.toml b/crates/asap_sketch_codec/Cargo.toml index 9ce14d38..2a1634b8 100644 --- a/crates/asap_sketch_codec/Cargo.toml +++ b/crates/asap_sketch_codec/Cargo.toml @@ -4,5 +4,5 @@ version = "0.1.0" edition = "2021" [dependencies] -asap_sketchlib = { git = "https://github.com/ProjectASAP/asap_sketchlib", rev = "026cd18c7b8c23ae6c46d4d683151ba562b8cd3a" } +asap_sketchlib = { git = "https://github.com/ProjectASAP/asap_sketchlib", rev = "5f03ccbd798ed5fec62bdd839bcb331123cab369" } prost = "0.13" diff --git a/crates/asap_sketch_codec/src/lib.rs b/crates/asap_sketch_codec/src/lib.rs index 279efe16..00efb284 100644 --- a/crates/asap_sketch_codec/src/lib.rs +++ b/crates/asap_sketch_codec/src/lib.rs @@ -47,11 +47,7 @@ pub fn encode_ddsketch(sketch: &DdSketch) -> Vec { producer: None, hash_spec: None, sample_p: 0.0, - sketch_state: Some(SketchState::Ddsketch(DdSketchState { - alpha: sketch.wire_alpha(), - store_counts: sketch.store_counts.clone(), - store_offset: sketch.store_offset, - })), + sketch_state: Some(SketchState::Ddsketch(sketch.to_proto())), }; envelope.encode_to_vec() } diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index 67678fc1..5a462526 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -38,7 +38,7 @@ src/ source.rs literals, batch sources, union, scalar conversion sources/ raw-source API, Scan and memory connector binding/ Planner executable DAG → physical operators - summary_kernels/ mathematical kernels, factory and traits + summary_kernels/ sketchlib adapters, exact accumulators, factory and traits stored_state/ decoding, delta application, persisted-state readout capability.rs support checks values.rs typed rows and state validation @@ -120,6 +120,8 @@ admission. Deployments provide a complete evaluation window or equivalent snapsh | Persisted formats and reconstruction | `stored_state` | Kernel or stored-state support alone does not imply executable-plan support. +Weighted CMS/CountSketch algorithms and candidate heaps live in `asap_sketchlib`; +the local adapter translates Planner parameters and typed values. ## Scope and acceptance From 433b7c4a08b4f98e3e75aaf1fbfd2c82cdd69770 Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 19:25:51 +0000 Subject: [PATCH 20/90] docs: explain shared physical execution design and acceptance boundaries --- docs/design_docs/physical-operators.md | 295 ++++++++++++++----------- 1 file changed, 169 insertions(+), 126 deletions(-) diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index 5a462526..986092ec 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -1,138 +1,181 @@ # Shared physical operators and DAG execution -## Decision and ownership +## Problem and goals -ASAP implements and maintains its own physical operators, summary kernels and -DAG runtime. DataFusion is a reference for module organization and execution -contracts; #462 does not adopt its execution backend or a hybrid runtime. -`SummaryExpr` and the shared-producer DAG model remain unchanged. +ASAPPlanner selects computations and summaries, but deployments also need concrete +implementations that execute those plans. Before this PR, there was no shared +physical-operator library for precompute and query deployments to reuse. Leaving +execution to each deployment duplicates implementation work and makes consistency +between Planner IR and executed behavior harder to maintain. -`asap-physical-operators` lives beside Planner IR and lowering, depends on local -IR types, and has no ASAPQuery-backend dependency. Precompute and query engines -use the same computation: for example, Scan → KLL build/merge → quantile readout. -Deployments supply sources, evaluation windows, storage and publication. +This design adds **both a physical-operator library and a runtime for executing +its DAGs** in `asap-physical-operators`. The intended consumers are developers of +asap-fusion and ASAPQuery. They can bind Planner-generated plans to shared +operators and execute a DAG or sub-DAG using deployment-provided inputs. -| Responsibility | Owner | -| --- | --- | -| IR, schemas and parameters | `asap-types` | -| Lowering and candidate correctness | `asap-aware-mapping` | -| Operators, shared execution and source interface | `asap-physical-operators` | -| Summary encoding | `asap_sketch_codec` | -| External connectors, durable storage and serving | Deployment repositories | +The goals are to: + +- Implement ordinary relational computation and summary build, merge and readout + behind common execution contracts. +- Preserve shared dependencies: one producer executes once per run even when + several consumers read its output. +- Share binding, cancellation and resource control across precompute and query + deployments while leaving engine orchestration with those deployments. +- Test Planner output and physical execution together so that changes to IR, + schemas and implementations stay consistent. -## Module organization +Partitioned parallelism, sharding, disk spill and cost-based physical algorithm +selection are future work. This PR provides a common place to implement them; +it does not deliver those optimizations or a complete deployment engine. + +## From a selected plan to results ```text -src/ - plan/ PhysicalDag, PhysicalOperator, properties, validation - runtime/ streams, shared producers, context, memory, cancellation - expressions/ scalar evaluation and Planner expression adaptation - operators/ - projection.rs - filter.rs - joins/ - aggregate/ ordinary and temporal reductions - sort.rs - limit.rs - summary/ build, merge and readout - source.rs literals, batch sources, union, scalar conversion - sources/ raw-source API, Scan and memory connector - binding/ Planner executable DAG → physical operators - summary_kernels/ sketchlib adapters, exact accumulators, factory and traits - stored_state/ decoding, delta application, persisted-state readout - capability.rs support checks - values.rs typed rows and state validation +SQL / PromQL + ↓ frontend +QueryExpr + ↓ ASAP planning and summary selection +SummaryNode containing SummaryExpr + ↓ compile_executable_dag(...) +ExecutableDag + ↓ binding::bind(...) + deployment-provided inputs and roots +PhysicalDag + ↓ execute(..., RunContext) +Results ``` -`plan` owns static contracts; `runtime` owns per-run state; operators own -construction checks and computation. `Expression` and `CompiledExpression` -share scalar execution under `expressions`. Summary kernels do not own scheduling -or storage selection. Kernel modules use operation names without `_accumulator`; -those old module names have no aliases. Top-level `dag`, `accumulators`, -`factory`, `traits` and `arithmetic` remain compatibility re-exports. - -## Execution contract +These representations serve different purposes: -- Validate topology, arity, schemas, supported bindings and input boundedness - before starting sources. Unsupported operations fail without external fallback. -- Execute each reachable producer once per run. Consumers have independent - cursors over shared outputs; separate runs never share mutable execution state. -- Run worker-local streams on the caller's worker. Deployments must poll consumers - concurrently: bounded queues provide backpressure, and dropping one consumer - leaves the others active. -- Propagate errors and cancellation. Long row loops and sort merge steps yield - cooperatively; individual scalar and kernel calls remain synchronous. -- Charge retained outputs and estimated operator workspace against the byte - budget. This is not an RSS or allocator-exact peak limit; source-owned data and - temporary allocation peaks are not fully covered. Blocking operators do not spill. - -`PhysicalOperator::properties` reports boundedness and emission; -`PhysicalDag::properties` derives them before execution. Sort, aggregate, temporal -reductions, joins, summary build/merge and vector-to-scalar require bounded input -and finalize after input ends. Projection, filter, limit, union and readout emit -incrementally. Global Limit bounds output cardinality; grouped Limit inherits -input boundedness. Neither promises a time deadline. - -## Sources and binding - -Raw Scan uses a registry keyed by Planner source identity. Binding checks metadata -without opening readers; execution lazily opens one cursor per reachable Scan -per run. Scan evaluates leaf predicates with three-valued logic, retaining only -TRUE. Projection, time selection and aggregation remain explicit operators. -Reader failures and schema drift fail execution. - -`RawSource::boundedness` defaults to Unknown. Connectors must explicitly declare -finite snapshots/windows and handle cancellation and I/O buffering. Only an -immutable memory connector is included; external readers are deployment work. - -The binder recognizes raw Scan inside the current `Fallback` leaf payload; -other retained expressions remain unsupported. Explicit source frontiers may -supply precomputed results. Stored-summary loading is separate from raw Scan: -deployments select compatible panes and provide coverage/revision scope. - -## Supported computation - -The library implements typed projection/filter, scalar arithmetic and booleans, -exact grouped aggregation, joins including semi-join, grouped Sort/Limit, Union, -scalar conversion, and summary build/merge/readout. Temporal reductions include -Rate, Increase, Sum, Avg, Min, Max, Count and histogram quantiles. Values preserve -Planner types/nullability and checked-division semantics; Count returns Int64. - -Native summary edges support exact Sum/Count/Min/Max/Rate/Increase, KLL, -DDSketch, HLL, and Float64 weighted CMS/CountSketch with candidate heaps. -Weighted TopK consumes finalized per-series rates into a summary per group, -reads typed candidate identities/scores, then applies grouped Sort → Limit. -CMS uses nonnegative weights and minimum-row estimates; CountSketch accepts -signed weights and uses median sign-corrected estimates at positive odd depth. -Neither uses integer-count encoding or fixed-point updates. - -Heap capacity differs from output k; ranking and merging do not prove candidate -completeness. Binding checks representation compatibility, not deployment accuracy -admission. Deployments provide a complete evaluation window or equivalent snapshot. - -| Capability | Validation entry point | +| Representation | Responsibility | +| --- | --- | +| `SummaryNode` / `SummaryExpr` | Describe the selected computation, including summary families, parameters and execution timing. | +| `ExecutableDag` | Record node identities, schemas and shared dependencies for binding. Despite its name, it contains no running physical operators. | +| `PhysicalDag` | Connect concrete physical-operator implementations that the shared runtime can execute. | + +A summary node already contains planning decisions; it is not just an unresolved +logical operator. It still does not implement batch consumption, mutable sketch +updates, output production or cancellation. Those belong to physical operators +and their runtime. Likewise, sharing a `SummaryNode` reference describes shared +computation; runtime coordination makes that sharing effective during execution. + +[`compile_executable_dag`](../../crates/types/src/post_asap/executable_dag.rs) +compiles the selected root and preserves shared node identity. The +[binder](../../crates/asap-physical-operators/src/binding/mod.rs) walks dependencies +from the requested roots, constructs supported operators and checks their +schemas. The resulting +[`PhysicalDag`](../../crates/asap-physical-operators/src/plan/mod.rs) is the input +to execution. Deployments normally use this path; they need not manually assemble +physical DAGs as low-level tests do. + +## Deployment boundary + +The shared runtime executes an individual DAG run. A deployment engine decides +which work to run, when to run it and how to use its results. + +| Shared library | asap-fusion / ASAPQuery deployment | | --- | --- | -| Update kernel and parameters | `capability::validate_summary_kernel` | -| Native state representation | `capability::validate_native_family` | -| Scalar readout | `capability::validate_native_readout` | -| Keyed readout and identity/score schema | `Operator::keyed_readout` | -| Complete executable plan | `binding::bind` / `bind_with_data_sources` | -| Persisted formats and reconstruction | `stored_state` | - -Kernel or stored-state support alone does not imply executable-plan support. -Weighted CMS/CountSketch algorithms and candidate heaps live in `asap_sketchlib`; -the local adapter translates Planner parameters and typed values. - -## Scope and acceptance - -Partitioned parallelism, disk spill, cost-based algorithm selection, physical -ordering/distribution properties and per-operator Explain/Analyze remain future -work. ASAP owns implementing and maintaining these capabilities when needed. -See the [DataFusion comparison](datafusion-execution-comparison.md) for tradeoffs. - -Tests cover shared-producer diamonds, slow/dropped consumers, cancellation, -errors, resource accounting and run isolation; operator coverage includes types, -nulls, grouping, state compatibility and rejected bindings. The same summary -pipeline runs in ingestion and query scopes. Deployment acceptance additionally -requires real source binding, window/revision handling, publication and output -adaptation; external query forwarding does not demonstrate local execution. +| Bind supported plan nodes and validate schemas | Select execution roots and approved input frontiers | +| Execute operators and coordinate shared producers | Schedule precompute/query runs and drive output streams | +| Track per-run state, cancellation and estimated memory | Set limits, evaluation windows and revision scope | +| Decode and combine supported stored-summary formats | Choose compatible stored panes and ensure coverage | +| Produce typed results | Persist/publish summaries or adapt and serve query results | + +“Deployment sources” means either raw-data connectors registered by Planner +source identity, or operators supplying results at an explicit node frontier. +For example, a query engine can supply a stored KLL state at the point where an +ingestion run would have built it. Binding stops traversing upstream dependencies +at that frontier and requires the supplied operator to match the node's schema. +This lets deployments execute the relevant sub-DAG without rebuilding upstream +computation. + +`bind_with_data_sources` resolves supported raw Scan leaves through the connector +registry. Readers open lazily during execution. Only a memory connector is +included here; external storage access belongs to deployments. Unsupported +retained expressions fail binding rather than silently forwarding execution to +another engine. A schema-compatible frontier alone does not establish window +coverage, revision correctness or summary accuracy; those remain deployment and +planning responsibilities. + +### Example: one summary, multiple answers + +An ingestion run can read a finite window, build a KLL summary and return it for +the precompute engine to persist. A query run can load compatible partial states, +merge them and feed the merged state to multiple quantile readouts. The merge +producer runs once, and each readout consumes its output independently. + +The library supplies the same build, merge and readout implementations for both +uses. The deployment supplies storage selection, scheduling and publication. +This is the before/after effect: selected plans gain a shared execution path +instead of requiring every deployment to implement these computations itself. + +## Execution contracts + +The design separates reusable plan structure from mutable per-run state: + +- Validate topology, arity, schemas, supported operations and required input + boundedness before starting sources. Errors are not empty results. +- Execute each reachable producer once per run. Consumers have independent + cursors; separate runs have independent mutable execution state. +- Use bounded queues for backpressure. Streams run on the caller's worker, and + deployments must poll multiple requested outputs concurrently. Dropping one + consumer leaves its siblings active. +- Propagate errors and cancellation, and yield cooperatively in long loops. + Individual scalar and sketch-kernel calls remain synchronous. +- Account for retained outputs and estimated operator workspace against a byte + budget. This is not an allocator-exact or process-RSS limit; source-owned data + and temporary allocation peaks are not fully covered. + +Blocking operators, including sort, aggregation, joins and summary build/merge, +require bounded input and finalize after input ends. Connectors must explicitly +declare finite snapshots or windows; unknown boundedness is insufficient. +Projection, filter and readout can emit incrementally. There is no spill path, +so workloads exceeding the tracked memory budget fail. + +## Implementation ownership and alternatives + +ASAP implements its own operators and DAG runtime, borrowing DataFusion's +separation of execution contracts and module responsibilities. This gives ASAP +direct ownership of summary-state edges and shared-producer execution, with the +cost of maintaining correctness, resource control and future optimizations. +It is not a claim of lower runtime overhead. DataFusion extension is a viable +alternative and does not inherently require modifying its core; the +[comparison](datafusion-execution-comparison.md) explains that tradeoff. + +Within the crate, `plan` owns contracts and validation, `binding` constructs +operators, `runtime` owns per-run execution, `expressions` evaluates scalars, +`operators` implements batch computation, and `sources` defines input access. +`operators/summary` adapts summary computation to the batch/DAG interface. +`summary_kernels` contains Planner-facing sketch adapters and exact accumulators; +sketch algorithms, including weighted CMS/CountSketch, belong to `asap_sketchlib`. +`stored_state` handles supported persisted-state decoding and reconstruction. +Kernel availability does not by itself imply support for native binding or every +stored-state format. + +Keeping the library in ASAPPlanner allows an IR change and its physical +implementation to be reviewed in one PR. It also permits tests across planning +and execution, including internal APIs, without coordinating changes across +separate repositories. Deployment connectors and engine policies remain outside +this library. + +## Validation and remaining gaps + +Acceptance has three levels. Each checks a different boundary: + +| Level | Required behavior | Existing automated coverage | +| --- | --- | --- | +| Individual operators | Correct values and schemas for supported types, edge cases and errors; enforce resource contracts. | Operator unit tests, [semantic tests](../../crates/asap-physical-operators/tests/physical_semantics.rs), [resource tests](../../crates/asap-physical-operators/tests/blocking_resources.rs). | +| Physical DAG execution | Compose operators correctly; respect dependencies, shared producers, cancellation and independent runs. | Runtime unit tests and [DAG integration tests](../../crates/asap-physical-operators/tests/physical_dag.rs), including summary build/merge/readout and shared consumers. | +| Planner-to-execution integration | Bind actual Planner-generated DAGs and execute with the intended types and dependencies. | [Weighted TopK integration](../../crates/asap-physical-operators/tests/weighted_topk_binding.rs) starts from PromQL and executes the selected plan with supplied rate results. | + +The third level has partial coverage: the weighted TopK test injects finalized +rates at an explicit frontier rather than reading raw time-series samples. +[Raw Scan integration](../../crates/asap-physical-operators/tests/raw_scan.rs) +executes Scan → Sort → Limit from a manually constructed `ExecutableDag`. +Neither establishes complete SQL/PromQL text → raw data → planning → native +results coverage. That full-path test remains a testing requirement, as does +broader coverage of Planner-generated operator combinations. + +The automated tests were run locally. No separate manual deployment-level +verification was performed. Deployment acceptance must additionally verify real +connectors, window/revision selection, persistence, publication and result +serving; the library tests do not establish those behaviors. From 3113435f6f652a3b2ac9d58cbf20d8c3a14e0a31 Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 19:55:53 +0000 Subject: [PATCH 21/90] docs: align physical execution design with architecture terminology --- docs/design_docs/physical-operators.md | 428 ++++++++++++++++--------- 1 file changed, 269 insertions(+), 159 deletions(-) diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index 986092ec..cafe3931 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -1,181 +1,291 @@ -# Shared physical operators and DAG execution +# Shared Physical Operators and DAG Execution -## Problem and goals +## 1. Problem -ASAPPlanner selects computations and summaries, but deployments also need concrete -implementations that execute those plans. Before this PR, there was no shared -physical-operator library for precompute and query deployments to reuse. Leaving -execution to each deployment duplicates implementation work and makes consistency -between Planner IR and executed behavior harder to maintain. +ASAPPlanner produces logical Post-ASAP candidates, but downstream systems also +need concrete implementations to execute a selected computation. Before this PR, +there was no shared physical-operator library for precompute and query deployments +to reuse. Reimplementing execution in each deployment duplicates work and makes +consistency between Planner semantics and runtime behavior harder to maintain. -This design adds **both a physical-operator library and a runtime for executing -its DAGs** in `asap-physical-operators`. The intended consumers are developers of -asap-fusion and ASAPQuery. They can bind Planner-generated plans to shared -operators and execute a DAG or sub-DAG using deployment-provided inputs. +This design introduces `asap-physical-operators`: a shared library containing +**physical operators and a DAG runtime**. asap-fusion and ASAPQuery can use it to +bind selected computations and execute them with deployment-provided inputs. +The planning library remains deployment-independent; downstream systems retain +physical feasibility checks, final commitment and engine orchestration. -The goals are to: +## 2. Goals and Non-goals -- Implement ordinary relational computation and summary build, merge and readout - behind common execution contracts. -- Preserve shared dependencies: one producer executes once per run even when - several consumers read its output. -- Share binding, cancellation and resource control across precompute and query - deployments while leaving engine orchestration with those deployments. -- Test Planner output and physical execution together so that changes to IR, - schemas and implementations stay consistent. +Goals: -Partitioned parallelism, sharding, disk spill and cost-based physical algorithm -selection are future work. This PR provides a common place to implement them; -it does not deliver those optimizations or a complete deployment engine. +- Share relational operators and summary build, merge and readout implementations. +- Execute shared dependencies once per run, with independent consumers. +- Share binding, cancellation and resource control across deployments. +- Test consistency between selected logical Post-ASAP DAGs and physical execution. -## From a selected plan to results +Non-goals for this PR are a complete precompute/query engine, external storage +connectors, partitioned parallelism, sharding, disk spill and cost-based physical +algorithm selection. The library provides a common place for future execution +optimizations; it does not implement those optimizations here. + +## 3. Architecture + +### 3.1 End-to-end flow + +Terminology follows the [ASAPPlanner design overview](architecture/README.md). +The planning library determines logical candidates. Downstream selects a +computation and commits to a feasible physical realization. The shared binder +constructs supported implementations, and the shared runtime executes them. +The deployment determines when, where and with which inputs execution happens. ```text -SQL / PromQL - ↓ frontend -QueryExpr - ↓ ASAP planning and summary selection -SummaryNode containing SummaryExpr +SQL / PromQL + planning workload and applicable evidence + ↓ frontend lowering +Canonical Pre-ASAP QueryExpr roots + ↓ logical candidate search +PlanSpace: logical Post-ASAP candidate DAG space + ↓ selection and assembly +Selected logical Post-ASAP DAG: SummaryNode / SummaryExpr ↓ compile_executable_dag(...) -ExecutableDag - ↓ binding::bind(...) + deployment-provided inputs and roots -PhysicalDag +ExecutableDag: structural representation compiled from the selected DAG + ↓ physical binding + deployment inputs and execution roots +PhysicalDag: concrete operators and their dependencies ↓ execute(..., RunContext) Results ``` -These representations serve different purposes: +This shows the selected-DAG path used by this library. `PlanSpace` is Planner's +primary output; selection and assembly are subsequent steps, not an implicit +physical deployment decision. Downstream may consume candidates directly or use +Planner's optional selection helpers. Lifecycle-aware helpers additionally +produce a `SummaryMaintenanceLifecyclePlan`; this runtime does not choose or +schedule that lifecycle. See the [planning workflows](architecture/input-output-workflow.md). + +### 3.2 Representation boundaries -| Representation | Responsibility | +| Architecture term | Rust representation | Meaning | +| --- | --- | --- | +| Pre-ASAP IR | `QueryExpr` | Original exact query semantics and output shape. | +| Logical candidate DAG space | `PlanSpace` | Alternatives and their logical guarantees/rejection reasons, before physical commitment. | +| Selected logical Post-ASAP DAG | `SummaryNode` / `SummaryExpr` | Selected logical summary semantics, including families, parameters, composition and readout. | +| Compiled representation of the selected Post-ASAP DAG | `ExecutableDag` | Node identities, schemas, execution assignments and shared dependencies exported for binding. Contains no instantiated physical operators. | +| Physical DAG used by this runtime | `asap_physical_operators::plan::PhysicalDag` | Concrete operator implementations connected for execution. | + +`ExecutableDag` is derived from the selected logical Post-ASAP DAG; it is not a +replacement name for Post-ASAP IR. A `SummaryNode` describes summary computation +but does not consume batches, update mutable sketches or coordinate cancellation. +Sharing its identity expresses a dependency; runtime coordination realizes shared +execution. + +The architecture's [physical-plan integration](architecture/physical-plan-integration.md) +also describes physical alternatives and analytical resource estimation. The +runtime `PhysicalDag` here should not be confused with a cost-model representation: +this PR does not automatically connect binding to analytical costing or search +across physical alternatives. + +### 3.3 Component responsibilities + +**The shared runtime executes one DAG run. Deployment engines decide which work +to run, when to run it and how to use its results.** + +| Component | Responsibility | | --- | --- | -| `SummaryNode` / `SummaryExpr` | Describe the selected computation, including summary families, parameters and execution timing. | -| `ExecutableDag` | Record node identities, schemas and shared dependencies for binding. Despite its name, it contains no running physical operators. | -| `PhysicalDag` | Connect concrete physical-operator implementations that the shared runtime can execute. | - -A summary node already contains planning decisions; it is not just an unresolved -logical operator. It still does not implement batch consumption, mutable sketch -updates, output production or cancellation. Those belong to physical operators -and their runtime. Likewise, sharing a `SummaryNode` reference describes shared -computation; runtime coordination makes that sharing effective during execution. - -[`compile_executable_dag`](../../crates/types/src/post_asap/executable_dag.rs) -compiles the selected root and preserves shared node identity. The -[binder](../../crates/asap-physical-operators/src/binding/mod.rs) walks dependencies -from the requested roots, constructs supported operators and checks their -schemas. The resulting -[`PhysicalDag`](../../crates/asap-physical-operators/src/plan/mod.rs) is the input -to execution. Deployments normally use this path; they need not manually assemble -physical DAGs as low-level tests do. - -## Deployment boundary - -The shared runtime executes an individual DAG run. A deployment engine decides -which work to run, when to run it and how to use its results. - -| Shared library | asap-fusion / ASAPQuery deployment | +| ASAPPlanner planning library | Generate logical Post-ASAP candidates; optionally select and assemble logical DAGs. | +| `compile_executable_dag` in `asap-types` | Export a selected logical Post-ASAP DAG as an `ExecutableDag`, preserving shared identity. | +| Binder | Construct supported physical implementations and validate their connections. | +| Physical operators | Perform relational and summary computation. | +| Shared runtime | Coordinate dependencies, shared producers and per-run resources. | +| asap-fusion / ASAPQuery | Decide physical feasibility and deployment; schedule runs, select storage/windows/revisions, persist/publish and serve results. | + +Keeping implementations in the ASAPPlanner repository does not move deployment +commitment into the planning library. The deployments reuse this crate as part +of their physical implementation rather than duplicating its DAG execution. + +## 4. Execution Model + +### 4.1 Binding + +The deployment supplies an `ExecutableDag`, execution roots and inputs. The +[binder](../../crates/asap-physical-operators/src/binding/mod.rs) traverses reachable +dependencies and constructs supported operators. A supplied intermediate result +cuts traversal at that node, enabling sub-DAG execution. Section 6 defines this +input boundary. Unsupported computation fails binding. + +### 4.2 DAG execution + +[`PhysicalDag::execute`](../../crates/asap-physical-operators/src/plan/mod.rs) +creates per-run execution state using a `RunContext` and returns output streams. +The deployment drives these streams and handles their results. Execution is +worker-local; the library does not supply a deployment scheduler or thread pool. +Separate runs reuse the plan structure while keeping mutable execution state +independent. + +### 4.3 Shared producers + +A reachable producer executes once per run, even when several consumers depend +on it. Each consumer has its own cursor over the shared output. Bounded queues +apply backpressure, so deployments must poll multiple requested output streams +concurrently. Dropping one consumer leaves other consumers active. + +### 4.4 Example: summary build → merge → readout + +```text +Precompute deployment Query deployment + +raw data compatible stored KLL states + ↓ ↓ +KLL build KLL merge + ↓ ┌───┴───┐ +stored KLL state ↓ ↓ +(deployment persists) p50 readout p99 readout +``` + +The precompute engine provides a finite input window and persists the returned +state. The query engine selects compatible stored states and supplies them at +an input frontier. Binding includes only the required downstream computation. +The merge executes once, and both readouts independently consume its output. + +Both engines reuse the same build, merge and readout implementations. This is +the E2E change introduced by the PR: deployments gain a shared binding and +execution path for selected computations, while retaining storage and scheduling. + +## 5. Execution Contracts + +| Contract | Invariant | | --- | --- | -| Bind supported plan nodes and validate schemas | Select execution roots and approved input frontiers | -| Execute operators and coordinate shared producers | Schedule precompute/query runs and drive output streams | -| Track per-run state, cancellation and estimated memory | Set limits, evaluation windows and revision scope | -| Decode and combine supported stored-summary formats | Choose compatible stored panes and ensure coverage | -| Produce typed results | Persist/publish summaries or adapt and serve query results | - -“Deployment sources” means either raw-data connectors registered by Planner -source identity, or operators supplying results at an explicit node frontier. -For example, a query engine can supply a stored KLL state at the point where an -ingestion run would have built it. Binding stops traversing upstream dependencies -at that frontier and requires the supplied operator to match the node's schema. -This lets deployments execute the relevant sub-DAG without rebuilding upstream -computation. - -`bind_with_data_sources` resolves supported raw Scan leaves through the connector -registry. Readers open lazily during execution. Only a memory connector is -included here; external storage access belongs to deployments. Unsupported -retained expressions fail binding rather than silently forwarding execution to -another engine. A schema-compatible frontier alone does not establish window -coverage, revision correctness or summary accuracy; those remain deployment and -planning responsibilities. - -### Example: one summary, multiple answers - -An ingestion run can read a finite window, build a KLL summary and return it for -the precompute engine to persist. A query run can load compatible partial states, -merge them and feed the merged state to multiple quantile readouts. The merge -producer runs once, and each readout consumes its output independently. - -The library supplies the same build, merge and readout implementations for both -uses. The deployment supplies storage selection, scheduling and publication. -This is the before/after effect: selected plans gain a shared execution path -instead of requiring every deployment to implement these computations itself. - -## Execution contracts - -The design separates reusable plan structure from mutable per-run state: - -- Validate topology, arity, schemas, supported operations and required input - boundedness before starting sources. Errors are not empty results. -- Execute each reachable producer once per run. Consumers have independent - cursors; separate runs have independent mutable execution state. -- Use bounded queues for backpressure. Streams run on the caller's worker, and - deployments must poll multiple requested outputs concurrently. Dropping one - consumer leaves its siblings active. -- Propagate errors and cancellation, and yield cooperatively in long loops. - Individual scalar and sketch-kernel calls remain synchronous. -- Account for retained outputs and estimated operator workspace against a byte - budget. This is not an allocator-exact or process-RSS limit; source-owned data - and temporary allocation peaks are not fully covered. +| C1 — Validate before execution | Topology, schemas, arity, supported operations and required boundedness are checked before sources start. Runtime data and reader failures remain execution errors. | +| C2 — Execute each producer once per run | Multiple consumers share one producer execution. | +| C3 — Isolate runs | Independent executions do not share mutable operator execution state. | +| C4 — Apply backpressure | Producer/consumer communication uses bounded queues and independent cursors. | +| C5 — Propagate failure and cancellation | Errors are not empty results; long computation loops cooperate with cancellation. | +| C6 — Enforce the tracked resource budget | Retained outputs and estimated operator workspace count against the run's byte budget. | Blocking operators, including sort, aggregation, joins and summary build/merge, -require bounded input and finalize after input ends. Connectors must explicitly -declare finite snapshots or windows; unknown boundedness is insufficient. -Projection, filter and readout can emit incrementally. There is no spill path, -so workloads exceeding the tracked memory budget fail. - -## Implementation ownership and alternatives - -ASAP implements its own operators and DAG runtime, borrowing DataFusion's -separation of execution contracts and module responsibilities. This gives ASAP -direct ownership of summary-state edges and shared-producer execution, with the -cost of maintaining correctness, resource control and future optimizations. -It is not a claim of lower runtime overhead. DataFusion extension is a viable -alternative and does not inherently require modifying its core; the -[comparison](datafusion-execution-comparison.md) explains that tradeoff. - -Within the crate, `plan` owns contracts and validation, `binding` constructs -operators, `runtime` owns per-run execution, `expressions` evaluates scalars, -`operators` implements batch computation, and `sources` defines input access. -`operators/summary` adapts summary computation to the batch/DAG interface. -`summary_kernels` contains Planner-facing sketch adapters and exact accumulators; -sketch algorithms, including weighted CMS/CountSketch, belong to `asap_sketchlib`. -`stored_state` handles supported persisted-state decoding and reconstruction. -Kernel availability does not by itself imply support for native binding or every -stored-state format. - -Keeping the library in ASAPPlanner allows an IR change and its physical -implementation to be reviewed in one PR. It also permits tests across planning -and execution, including internal APIs, without coordinating changes across -separate repositories. Deployment connectors and engine policies remain outside -this library. - -## Validation and remaining gaps - -Acceptance has three levels. Each checks a different boundary: - -| Level | Required behavior | Existing automated coverage | +require bounded input and finalize after input ends. Unknown source boundedness +is insufficient. Projection, filter and readout can emit incrementally. + +Cancellation is cooperative: individual scalar and sketch-kernel calls remain +synchronous. Resource accounting is not an allocator-exact or process-RSS limit; +source-owned data and temporary allocation peaks are not fully covered. There +is no spill path, so exceeding the tracked byte budget fails execution. + +## 6. Inputs and Execution Frontiers + +A deployment can provide raw data at a Scan or already-computed results at an +intermediate node. These are two ways to supply the inputs of one execution. + +### 6.1 Raw data sources + +`bind_with_data_sources` resolves supported raw Scan leaves through a registry +keyed by Planner source identity. Binding checks metadata; execution lazily opens +readers. Connectors must declare finite snapshots/windows when required and +handle cancellation and I/O buffering. Reader failures and schema drift fail +execution. Only a memory connector is included in this PR. + +The current binder recognizes raw Scan inside a retained Pre-ASAP leaf payload. +Other unsupported retained expressions fail binding; arbitrary Pre-ASAP execution +is not implied by support for Scan. + +### 6.2 Supplied intermediate results + +```text +Scan → KLL build → KLL merge → quantile readout + ↑ + deployment may supply this node's output from stored state +``` + +When a deployment supplies a compatible result, binding treats that node as an +execution frontier and does not bind its upstream dependencies. The supplied +operator must match the node's declared schema. Deployments therefore need not +manually construct physical DAGs to reuse precomputed work. + +Schema compatibility does not establish window coverage, revision correctness, +maintenance readiness or accuracy guarantees. Planning and deployment must +establish these before committing to execution. Stored-state decoding provides +format support, not a policy for selecting valid stored panes. + +## 7. Implementation Organization + +| Module | Owns | +| --- | --- | +| `plan` | Physical DAG/operator contracts, properties and validation | +| `binding` | Compiled Post-ASAP representation → concrete operators | +| `runtime` | Per-run execution, streams, shared producers and resource control | +| `operators` | Batch implementations of relational and temporal computation | +| `expressions` | Scalar evaluation and Planner expression adaptation | +| `operators/summary` | Physical summary build, merge and readout | +| `summary_kernels` | Planner-facing sketch adapters and exact accumulators | +| `sources` | Input interfaces, Scan and memory connector | +| `stored_state` | Persisted summary decoding, delta application and reconstruction | + +Sketch algorithms themselves, including weighted CMS/CountSketch, belong to +`asap_sketchlib`. Kernel availability does not imply support for native binding, +every readout or every stored-state format; these capabilities are checked +separately. + +Locating the library alongside Planner lets one PR change an IR node and its +physical implementation. It also enables tests across internal planning and +execution APIs without coordinating repositories. Deployment policies and +external connectors remain outside the library. + +## 8. Alternatives Considered + +### 8.1 ASAP-owned runtime — selected + +ASAP owns physical operators and DAG execution, following DataFusion's separation +of contracts, runtime and concrete implementations. This provides direct control +over native summary-state edges and shared producers, common computation across +deployments, and tests spanning logical planning and execution. + +The cost is ownership of operator correctness, resource management and future +parallelism, spill and physical optimization. There is no measured claim that +this runtime has lower overhead than DataFusion. + +### 8.2 DataFusion extension + +DataFusion offers established physical implementations and execution machinery. +Reusing it would require integrating ASAP's logical summary semantics, state +transport, sharing, partitioning and lifecycle contracts. Extensions do not +inherently require changes to DataFusion core, but existing optimizations are +usable only when those contracts preserve ASAP semantics. + +This PR chooses native execution; a DataFusion backend or hybrid runtime is +outside its scope. The [DataFusion comparison](datafusion-execution-comparison.md) +provides the detailed tradeoffs. + +## 9. Validation + +Acceptance has three levels: + +| Level | Required behavior | Coverage today | | --- | --- | --- | -| Individual operators | Correct values and schemas for supported types, edge cases and errors; enforce resource contracts. | Operator unit tests, [semantic tests](../../crates/asap-physical-operators/tests/physical_semantics.rs), [resource tests](../../crates/asap-physical-operators/tests/blocking_resources.rs). | -| Physical DAG execution | Compose operators correctly; respect dependencies, shared producers, cancellation and independent runs. | Runtime unit tests and [DAG integration tests](../../crates/asap-physical-operators/tests/physical_dag.rs), including summary build/merge/readout and shared consumers. | -| Planner-to-execution integration | Bind actual Planner-generated DAGs and execute with the intended types and dependencies. | [Weighted TopK integration](../../crates/asap-physical-operators/tests/weighted_topk_binding.rs) starts from PromQL and executes the selected plan with supplied rate results. | +| Operator correctness | Correct results and schemas for supported types, edge cases and errors; resource contracts hold. | Unit tests, [semantic tests](../../crates/asap-physical-operators/tests/physical_semantics.rs) and [resource tests](../../crates/asap-physical-operators/tests/blocking_resources.rs). | +| Physical DAG correctness | Correct composition, dependencies, shared producers, cancellation and independent runs. | Runtime unit tests and [DAG integration tests](../../crates/asap-physical-operators/tests/physical_dag.rs). | +| Planner → execution correctness | Selected logical Post-ASAP DAGs compile, bind and execute with the intended semantics. | [Weighted TopK integration](../../crates/asap-physical-operators/tests/weighted_topk_binding.rs) starts from PromQL, selects a candidate and executes with supplied rate results. | -The third level has partial coverage: the weighted TopK test injects finalized -rates at an explicit frontier rather than reading raw time-series samples. +The third level is partial: weighted TopK injects finalized rates at a frontier, +so it does not execute raw time-series input through rate calculation. [Raw Scan integration](../../crates/asap-physical-operators/tests/raw_scan.rs) -executes Scan → Sort → Limit from a manually constructed `ExecutableDag`. -Neither establishes complete SQL/PromQL text → raw data → planning → native -results coverage. That full-path test remains a testing requirement, as does -broader coverage of Planner-generated operator combinations. - -The automated tests were run locally. No separate manual deployment-level -verification was performed. Deployment acceptance must additionally verify real -connectors, window/revision selection, persistence, publication and result -serving; the library tests do not establish those behaviors. +covers Scan → Sort → Limit using a manually constructed `ExecutableDag`. + +Still required are a complete SQL/PromQL → raw input → planning → binding → native +results test and broader combinations of Planner-generated operators. Deployment +acceptance additionally needs real connectors, window/revision selection, +persistence, publication and serving. + +Existing automated tests were run locally. No separate manual deployment-level +verification was performed. The documentation update does not add test coverage. + +## 10. Limitations and Future Work + +Execution currently targets supported operations over bounded inputs where +blocking computation is required. It is not complete SQL/PromQL execution, and +successful logical candidate construction does not prove physical feasibility. +Downstream must reject candidates without a complete supported realization for +the chosen execution boundary. + +Partitioned parallelism, sharding, spill, physical algorithm selection and richer +ordering/distribution properties require further design and implementation. +Connecting runtime implementations to analytical physical costing is also a +separate integration task. These extensions must preserve C1–C6 and the planning +and deployment ownership boundaries above. From 359eaa1917ea4e09c4f2a3d4f64279ec4b99b905 Mon Sep 17 00:00:00 2001 From: zz_y Date: Thu, 24 Sep 2026 21:50:46 +0000 Subject: [PATCH 22/90] docs: simplify physical execution design to three ownership layers --- docs/design_docs/physical-operators.md | 435 +++++++++++++------------ 1 file changed, 222 insertions(+), 213 deletions(-) diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md index cafe3931..bebad582 100644 --- a/docs/design_docs/physical-operators.md +++ b/docs/design_docs/physical-operators.md @@ -2,290 +2,299 @@ ## 1. Problem -ASAPPlanner produces logical Post-ASAP candidates, but downstream systems also -need concrete implementations to execute a selected computation. Before this PR, -there was no shared physical-operator library for precompute and query deployments -to reuse. Reimplementing execution in each deployment duplicates work and makes -consistency between Planner semantics and runtime behavior harder to maintain. - -This design introduces `asap-physical-operators`: a shared library containing -**physical operators and a DAG runtime**. asap-fusion and ASAPQuery can use it to -bind selected computations and execute them with deployment-provided inputs. -The planning library remains deployment-independent; downstream systems retain -physical feasibility checks, final commitment and engine orchestration. +ASAPPlanner describes valid summary computations. Precompute and query engines +need concrete operators to execute them. Implementing these operators and their +execution separately duplicates work and makes it harder to preserve Planner +semantics across deployments. + +This design provides a shared physical execution library for asap-fusion and +ASAPQuery. It contains both physical operators and the runtime for executing a +DAG. Deployment compilers use the shared physical operator vocabulary to describe +computation; deployment engines supply inputs and invoke the shared executor. + +The intended outcome is that the same summary build, merge and readout can run +in either engine with the same semantics and execution contracts. ## 2. Goals and Non-goals Goals: -- Share relational operators and summary build, merge and readout implementations. -- Execute shared dependencies once per run, with independent consumers. -- Share binding, cancellation and resource control across deployments. -- Test consistency between selected logical Post-ASAP DAGs and physical execution. +- Share relational and summary operators across precompute and query execution. +- Preserve logical semantics when compiling a computation to physical operators. +- Execute shared producers once per run, with independent consumers. +- Provide common failure, cancellation, backpressure and resource contracts. +- Keep deployment decisions separate from computation execution. -Non-goals for this PR are a complete precompute/query engine, external storage -connectors, partitioned parallelism, sharding, disk spill and cost-based physical -algorithm selection. The library provides a common place for future execution -optimizations; it does not implement those optimizations here. +This library does not choose deployment placement, manage durable storage, +schedule maintenance or serve requests. Partitioned parallelism, sharding, spill +and cost-based algorithm selection are outside the initial scope. ## 3. Architecture -### 3.1 End-to-end flow +### 3.1 Three layers Terminology follows the [ASAPPlanner design overview](architecture/README.md). -The planning library determines logical candidates. Downstream selects a -computation and commits to a feasible physical realization. The shared binder -constructs supported implementations, and the shared runtime executes them. -The deployment determines when, where and with which inputs execution happens. +There are three layers: ```text -SQL / PromQL + planning workload and applicable evidence - ↓ frontend lowering -Canonical Pre-ASAP QueryExpr roots - ↓ logical candidate search -PlanSpace: logical Post-ASAP candidate DAG space - ↓ selection and assembly -Selected logical Post-ASAP DAG: SummaryNode / SummaryExpr - ↓ compile_executable_dag(...) -ExecutableDag: structural representation compiled from the selected DAG - ↓ physical binding + deployment inputs and execution roots -PhysicalDag: concrete operators and their dependencies - ↓ execute(..., RunContext) -Results +Logical planning — ASAPPlanner + PlanSpace → selected logical Post-ASAP DAG + ↓ +Deployment compilation — deployment PhysicalPlanCompiler + Deployment plan: physical DAGs + input bindings + operational configuration + ↓ +Shared execution — physical operators and DAG runtime + Bound inputs + physical DAG + run context → results ``` -This shows the selected-DAG path used by this library. `PlanSpace` is Planner's -primary output; selection and assembly are subsequent steps, not an implicit -physical deployment decision. Downstream may consume candidates directly or use -Planner's optional selection helpers. Lifecycle-aware helpers additionally -produce a `SummaryMaintenanceLifecyclePlan`; this runtime does not choose or -schedule that lifecycle. See the [planning workflows](architecture/input-output-workflow.md). +| Layer | Responsibility | Output | +| --- | --- | --- | +| Logical planning | Define legal computations, summary families, parameters and guarantees. | Logical candidates and a selected Post-ASAP DAG. | +| Deployment compilation | Choose a feasible physical realization and establish its inputs, materializations, placement and operational configuration. | A deployment plan containing physical DAGs and their deployment contracts. | +| Shared execution | Run the specified physical computation and coordinate dependencies and resources. | Result streams or an explicit failure. | + +`PlanSpace` is the primary output of logical candidate search. Selection and +assembly produce the logical DAG used for deployment compilation. Downstream +physical evidence may inform selection through Planner's interfaces; the flow +does not require choosing a candidate without considering physical feasibility. +Planner-owned semantics and guarantees remain authoritative. See the +[planning workflows](architecture/input-output-workflow.md). + +**The deployment decides what to run, where, when and with which inputs. The +shared executor runs that computation.** Scheduling, plan installation, +persistence and serving are deployment activities, not additional planning +layers. -### 3.2 Representation boundaries +### 3.2 Two computation graphs -| Architecture term | Rust representation | Meaning | +| Graph | Describes | Owner | | --- | --- | --- | -| Pre-ASAP IR | `QueryExpr` | Original exact query semantics and output shape. | -| Logical candidate DAG space | `PlanSpace` | Alternatives and their logical guarantees/rejection reasons, before physical commitment. | -| Selected logical Post-ASAP DAG | `SummaryNode` / `SummaryExpr` | Selected logical summary semantics, including families, parameters, composition and readout. | -| Compiled representation of the selected Post-ASAP DAG | `ExecutableDag` | Node identities, schemas, execution assignments and shared dependencies exported for binding. Contains no instantiated physical operators. | -| Physical DAG used by this runtime | `asap_physical_operators::plan::PhysicalDag` | Concrete operator implementations connected for execution. | - -`ExecutableDag` is derived from the selected logical Post-ASAP DAG; it is not a -replacement name for Post-ASAP IR. A `SummaryNode` describes summary computation -but does not consume batches, update mutable sketches or coordinate cancellation. -Sharing its identity expresses a dependency; runtime coordination realizes shared -execution. - -The architecture's [physical-plan integration](architecture/physical-plan-integration.md) -also describes physical alternatives and analytical resource estimation. The -runtime `PhysicalDag` here should not be confused with a cost-model representation: -this PR does not automatically connect binding to analytical costing or search -across physical alternatives. - -### 3.3 Component responsibilities - -**The shared runtime executes one DAG run. Deployment engines decide which work -to run, when to run it and how to use its results.** - -| Component | Responsibility | -| --- | --- | -| ASAPPlanner planning library | Generate logical Post-ASAP candidates; optionally select and assemble logical DAGs. | -| `compile_executable_dag` in `asap-types` | Export a selected logical Post-ASAP DAG as an `ExecutableDag`, preserving shared identity. | -| Binder | Construct supported physical implementations and validate their connections. | -| Physical operators | Perform relational and summary computation. | -| Shared runtime | Coordinate dependencies, shared producers and per-run resources. | -| asap-fusion / ASAPQuery | Decide physical feasibility and deployment; schedule runs, select storage/windows/revisions, persist/publish and serve results. | +| Logical Post-ASAP DAG | Selected computation semantics, summary operations and shared dependencies. | Planner | +| Physical DAG | Concrete operator choices, configuration, ordered dependencies and input slots. | Shared physical-plan contract, constructed by deployment compilation | -Keeping implementations in the ASAPPlanner repository does not move deployment -commitment into the planning library. The deployments reuse this crate as part -of their physical implementation rather than duplicating its DAG execution. +A logical node may expand into several physical operators. Several consumers +may share one physical producer. A materialized intermediate result may become +an input, excluding its upstream computation from that execution. -## 4. Execution Model +There is no additional execution IR between these graphs. Exporting a logical +DAG as node IDs and edges is a serialization detail. Loading a physical DAG and +instantiating its operator objects is an execution detail. Neither operation +introduces a new semantic plan. + +A deployment plan is a container for physical DAGs and their operational +configuration, not another computation graph. It can hold distinct precompute +and query DAGs that use the same physical operator vocabulary. + +### 3.3 Deployment compiler and shared executor -### 4.1 Binding +The backend `PhysicalPlanCompiler` is the deployment compilation entry point. +It turns a selected logical Post-ASAP DAG into a feasible deployment contract: -The deployment supplies an `ExecutableDag`, execution roots and inputs. The -[binder](../../crates/asap-physical-operators/src/binding/mod.rs) traverses reachable -dependencies and constructs supported operators. A supplied intermediate result -cuts traversal at that node, enabling sub-DAG execution. Section 6 defines this -input boundary. Unsupported computation fails binding. +- Lower computation to the shared physical operator vocabulary. +- Select input boundaries and bind them to raw sources or materializations. +- Establish concrete window/state layout, placement and transmission contracts. +- Produce consistent precompute and query plans referencing the same state + identities and schemas. -### 4.2 DAG execution +The shared library owns operator definitions, validation and implementations. +The compiler uses these contracts to construct physical DAGs; it does not +maintain a separate query or precompute operator hierarchy. -[`PhysicalDag::execute`](../../crates/asap-physical-operators/src/plan/mod.rs) -creates per-run execution state using a `RunContext` and returns output streams. -The deployment drives these streams and handles their results. Execution is -worker-local; the library does not supply a deployment scheduler or thread pool. -Separate runs reuse the plan structure while keeping mutable execution state -independent. +`QueryPlan` contains query identity, input/materialization bindings, a physical +DAG and fallback policy. `PrecomputePlan` contains physical DAGs together with +their trigger, input, state and publication contracts. Catalog and transmission +configuration connect producers and consumers without redefining computation. -### 4.3 Shared producers +At execution time, the deployment resolves input slots to readers or supplied +state and instantiates the specified operators. This step does not reselect +algorithms, change summary semantics or make new deployment decisions. The +executor runs the resulting graph; the deployment handles its results. -A reachable producer executes once per run, even when several consumers depend -on it. Each consumer has its own cursor over the shared output. Bounded queues -apply backpressure, so deployments must poll multiple requested output streams -concurrently. Dropping one consumer leaves other consumers active. +## 4. Execution Model -### 4.4 Example: summary build → merge → readout +### 4.1 Example: one KLL summary, two quantiles ```text -Precompute deployment Query deployment - -raw data compatible stored KLL states - ↓ ↓ -KLL build KLL merge - ↓ ┌───┴───┐ -stored KLL state ↓ ↓ -(deployment persists) p50 readout p99 readout +Precompute DAG Query DAG + +raw input slot stored-state input slot + ↓ ↓ + KLL build KLL merge + ↓ ┌────┴────┐ + state output ↓ ↓ + p50 readout p99 readout ``` -The precompute engine provides a finite input window and persists the returned -state. The query engine selects compatible stored states and supplies them at -an input frontier. Binding includes only the required downstream computation. -The merge executes once, and both readouts independently consume its output. +Logical planning selects KLL computation and the required readouts. Deployment +compilation decides where summaries are built and stored, assigns materialization +identities and defines the input contract for each DAG. + +The precompute engine supplies a finite input window, executes the build DAG +and persists its output. The query engine supplies compatible stored states and +executes the query DAG. Merge runs once; both quantile operators independently +consume its output. + +Storage retrieval and publication are deployment responsibilities. Build, merge, +readout and dependency execution use the shared library in both engines. -Both engines reuse the same build, merge and readout implementations. This is -the E2E change introduced by the PR: deployments gain a shared binding and -execution path for selected computations, while retaining storage and scheduling. +### 4.2 One execution + +The executor takes a physical DAG, resolved inputs, selected output roots and +a run context. Before opening sources, it validates the reachable computation +and input contracts. It then creates fresh mutable execution state and returns +output streams. + +The deployment drives those streams and handles completion or failure. Multiple +requested streams must be polled concurrently so a slow consumer does not prevent +progress through a shared producer's bounded queues. Dropping one consumer leaves +its siblings active. + +The same physical DAG can execute repeatedly. Operator state, consumer cursors +and resource reservations belong to an individual run. Persistent state is +supplied explicitly through deployment inputs rather than implicitly retained +between runs. ## 5. Execution Contracts | Contract | Invariant | | --- | --- | -| C1 — Validate before execution | Topology, schemas, arity, supported operations and required boundedness are checked before sources start. Runtime data and reader failures remain execution errors. | -| C2 — Execute each producer once per run | Multiple consumers share one producer execution. | -| C3 — Isolate runs | Independent executions do not share mutable operator execution state. | -| C4 — Apply backpressure | Producer/consumer communication uses bounded queues and independent cursors. | -| C5 — Propagate failure and cancellation | Errors are not empty results; long computation loops cooperate with cancellation. | +| C1 — Validate before execution | Reject invalid topology, arity, schemas, operator configurations and input boundedness before sources start. | +| C2 — Execute shared producers once | A reachable physical producer executes once per run, regardless of consumer count. | +| C3 — Isolate runs | Independent runs have independent mutable execution state. | +| C4 — Apply backpressure | Bounded communication queues limit retained output; consumers have independent cursors. | +| C5 — Propagate failure and cancellation | Errors are explicit, and cancellation terminates affected work and releases its resources. | | C6 — Enforce the tracked resource budget | Retained outputs and estimated operator workspace count against the run's byte budget. | -Blocking operators, including sort, aggregation, joins and summary build/merge, -require bounded input and finalize after input ends. Unknown source boundedness -is insufficient. Projection, filter and readout can emit incrementally. +Blocking operators require finite input and finalize after it ends. Unknown +boundedness is insufficient for these operators. Incremental operators may emit +before the input completes. Operator properties must make this distinction +available to validation. -Cancellation is cooperative: individual scalar and sketch-kernel calls remain -synchronous. Resource accounting is not an allocator-exact or process-RSS limit; -source-owned data and temporary allocation peaks are not fully covered. There -is no spill path, so exceeding the tracked byte budget fails execution. +Cancellation is cooperative, including within long computation loops. Individual +synchronous kernel calls limit cancellation responsiveness. Memory accounting +covers tracked allocations, not process RSS or every temporary allocation peak. +Without spill support, exceeding the tracked budget fails the run. -## 6. Inputs and Execution Frontiers +The executor reports errors; deployment policy decides whether to retry or invoke +an explicit fallback. It must not silently replace failed computation or turn an +error into an empty result. -A deployment can provide raw data at a Scan or already-computed results at an -intermediate node. These are two ways to supply the inputs of one execution. +## 6. Inputs and Execution Boundaries -### 6.1 Raw data sources +A physical DAG exposes typed input slots. The deployment plan binds each slot +to raw data or an already-computed result. -`bind_with_data_sources` resolves supported raw Scan leaves through a registry -keyed by Planner source identity. Binding checks metadata; execution lazily opens -readers. Connectors must declare finite snapshots/windows when required and -handle cancellation and I/O buffering. Reader failures and schema drift fail -execution. Only a memory connector is included in this PR. +### 6.1 Raw inputs -The current binder recognizes raw Scan inside a retained Pre-ASAP leaf payload. -Other unsupported retained expressions fail binding; arbitrary Pre-ASAP execution -is not implied by support for Scan. +A raw input contract identifies the source, schema and required data scope. +The deployment supplies a reader satisfying that contract. Readers declare +boundedness, support cancellation and report schema drift or I/O failures. +Opening readers is deferred until execution validation succeeds. -### 6.2 Supplied intermediate results +### 6.2 Materialized inputs + +Deployment compilation can replace an upstream computation with a compatible +materialized result: ```text -Scan → KLL build → KLL merge → quantile readout - ↑ - deployment may supply this node's output from stored state +Logical computation: Scan → KLL build → merge → readout + +Physical query DAG: stored-state input → merge → readout ``` -When a deployment supplies a compatible result, binding treats that node as an -execution frontier and does not bind its upstream dependencies. The supplied -operator must match the node's declared schema. Deployments therefore need not -manually construct physical DAGs to reuse precomputed work. +This boundary is chosen during deployment compilation. The executor receives an +explicit input slot; it does not discover a materialization or decide to omit +upstream work at serving time. -Schema compatibility does not establish window coverage, revision correctness, -maintenance readiness or accuracy guarantees. Planning and deployment must -establish these before committing to execution. Stored-state decoding provides -format support, not a policy for selecting valid stored panes. +Compatibility includes summary family and parameters, schema, grouping, window +coverage and revision scope. The deployment must establish materialization +readiness and the applicable accuracy guarantee before using the result. +Successful byte decoding or schema matching alone is insufficient. ## 7. Implementation Organization +These are modules within the shared execution library, not architectural layers: + | Module | Owns | | --- | --- | -| `plan` | Physical DAG/operator contracts, properties and validation | -| `binding` | Compiled Post-ASAP representation → concrete operators | -| `runtime` | Per-run execution, streams, shared producers and resource control | -| `operators` | Batch implementations of relational and temporal computation | -| `expressions` | Scalar evaluation and Planner expression adaptation | -| `operators/summary` | Physical summary build, merge and readout | +| `plan` | Shared physical operator vocabulary, DAG structure, properties and validation | +| `binding` | Instantiate specified operators and resolve supplied input slots | +| `runtime` | Per-run state, streams, shared producers and resource accounting | +| `operators` | Relational and temporal physical implementations | +| `operators/summary` | Summary build, merge and readout implementations | +| `expressions` | Scalar expression evaluation | | `summary_kernels` | Planner-facing sketch adapters and exact accumulators | -| `sources` | Input interfaces, Scan and memory connector | -| `stored_state` | Persisted summary decoding, delta application and reconstruction | +| `sources` | Reader interfaces and input adapters | +| `stored_state` | Supported state decoding, delta application and reconstruction | -Sketch algorithms themselves, including weighted CMS/CountSketch, belong to -`asap_sketchlib`. Kernel availability does not imply support for native binding, -every readout or every stored-state format; these capabilities are checked -separately. +Sketch algorithms belong to `asap_sketchlib`. Storage selection and engine +policies belong to deployment repositories. The deployment compiler consumes +the shared operator contracts to lower logical computation. -Locating the library alongside Planner lets one PR change an IR node and its -physical implementation. It also enables tests across internal planning and -execution APIs without coordinating repositories. Deployment policies and -external connectors remain outside the library. +Keeping the execution library alongside Planner allows one PR to change an IR +operation and its implementation, and enables tests across internal planning +and execution interfaces. Repository location does not transfer deployment +ownership to the planning library. ## 8. Alternatives Considered -### 8.1 ASAP-owned runtime — selected +### 8.1 Shared ASAP execution — selected -ASAP owns physical operators and DAG execution, following DataFusion's separation -of contracts, runtime and concrete implementations. This provides direct control -over native summary-state edges and shared producers, common computation across -deployments, and tests spanning logical planning and execution. +One physical operator vocabulary and executor support both precompute and query +engines. ASAP controls summary-state edges and shared-producer semantics directly. +The cost is maintaining operator correctness, resource management and future +execution optimizations. -The cost is ownership of operator correctness, resource management and future -parallelism, spill and physical optimization. There is no measured claim that -this runtime has lower overhead than DataFusion. +Separate operator systems in each deployment would duplicate these responsibilities +and require repeated consistency work. Deployment differences are expressed +through input and operational contracts instead. -### 8.2 DataFusion extension +### 8.2 DataFusion execution -DataFusion offers established physical implementations and execution machinery. -Reusing it would require integrating ASAP's logical summary semantics, state -transport, sharing, partitioning and lifecycle contracts. Extensions do not -inherently require changes to DataFusion core, but existing optimizations are -usable only when those contracts preserve ASAP semantics. +DataFusion provides established operators and execution infrastructure. Reuse +requires integrating ASAP's summary semantics, state transport, sharing and +lifecycle contracts. Extensions do not inherently require changing DataFusion +core, but optimizations must preserve the supplied semantics. -This PR chooses native execution; a DataFusion backend or hybrid runtime is -outside its scope. The [DataFusion comparison](datafusion-execution-comparison.md) -provides the detailed tradeoffs. +This design selects native execution, following DataFusion's separation of +operator contracts and implementation responsibilities. It makes no claim of +lower runtime overhead. See the [DataFusion comparison](datafusion-execution-comparison.md). ## 9. Validation -Acceptance has three levels: +The design requires three levels of automated validation: -| Level | Required behavior | Coverage today | -| --- | --- | --- | -| Operator correctness | Correct results and schemas for supported types, edge cases and errors; resource contracts hold. | Unit tests, [semantic tests](../../crates/asap-physical-operators/tests/physical_semantics.rs) and [resource tests](../../crates/asap-physical-operators/tests/blocking_resources.rs). | -| Physical DAG correctness | Correct composition, dependencies, shared producers, cancellation and independent runs. | Runtime unit tests and [DAG integration tests](../../crates/asap-physical-operators/tests/physical_dag.rs). | -| Planner → execution correctness | Selected logical Post-ASAP DAGs compile, bind and execute with the intended semantics. | [Weighted TopK integration](../../crates/asap-physical-operators/tests/weighted_topk_binding.rs) starts from PromQL, selects a candidate and executes with supplied rate results. | - -The third level is partial: weighted TopK injects finalized rates at a frontier, -so it does not execute raw time-series input through rate calculation. -[Raw Scan integration](../../crates/asap-physical-operators/tests/raw_scan.rs) -covers Scan → Sort → Limit using a manually constructed `ExecutableDag`. - -Still required are a complete SQL/PromQL → raw input → planning → binding → native -results test and broader combinations of Planner-generated operators. Deployment -acceptance additionally needs real connectors, window/revision selection, -persistence, publication and serving. - -Existing automated tests were run locally. No separate manual deployment-level -verification was performed. The documentation update does not add test coverage. +| Level | Required behavior | +| --- | --- | +| Operators | Correct results and schemas across supported types, empty/null inputs, errors and resource limits. Summary build/merge/readout preserves the intended state semantics. | +| Physical DAG execution | Correct composition, shared producers, independent consumers, repeated runs, cancellation and resource release. | +| Planning and deployment integration | A selected Post-ASAP DAG compiles to consistent deployment contracts and executes through the shared library with correct input identities and results. | + +The KLL example must verify both precompute output and query readouts, including +one merge execution for two consumers. Materialized-input tests must reject +incompatible state and establish that excluded upstream work does not run. +Full-path tests must start with SQL/PromQL and raw inputs, then exercise planning, +deployment compilation and native execution rather than injecting all computed +intermediates. + +Existing operator, DAG and resource tests provide a foundation. The weighted +TopK integration exercises a Planner-selected computation with supplied rate +results; raw Scan integration uses a manually constructed DAG. These do not +establish the complete three-layer integration. Automated tests have been run +locally; no separate manual deployment-level verification has been performed. + +Deployment acceptance additionally covers real connectors, installation, +window/revision handling, persistence, publication and serving. This document +defines the target design; it does not claim those integrations are complete. ## 10. Limitations and Future Work -Execution currently targets supported operations over bounded inputs where -blocking computation is required. It is not complete SQL/PromQL execution, and -successful logical candidate construction does not prove physical feasibility. -Downstream must reject candidates without a complete supported realization for -the chosen execution boundary. - -Partitioned parallelism, sharding, spill, physical algorithm selection and richer -ordering/distribution properties require further design and implementation. -Connecting runtime implementations to analytical physical costing is also a -separate integration task. These extensions must preserve C1–C6 and the planning -and deployment ownership boundaries above. +The initial execution scope is supported computation with bounded input wherever +blocking operators require it. Logical validity alone does not establish physical +feasibility: deployment compilation must reject an unsupported realization before +committing it, or select an explicit deployment fallback. + +Parallelism, sharding, spill, richer ordering/distribution properties and physical +algorithm costing can extend the same physical-plan contract. They must preserve +C1–C6 and the three ownership boundaries. They do not require additional semantic +IR layers. From 6f423388b1c2747b060702431e2d839bc29540c6 Mon Sep 17 00:00:00 2001 From: zz_y Date: Fri, 25 Sep 2026 17:46:04 +0000 Subject: [PATCH 23/90] docs: clarify maintenance physical planning and deployment boundaries --- crates/asap-physical-operators/README.md | 2 +- docs/design_docs/physical-operators.md | 300 ------------------ .../physical-planning-and-deployment.md | 195 ++++++++++++ 3 files changed, 196 insertions(+), 301 deletions(-) delete mode 100644 docs/design_docs/physical-operators.md create mode 100644 docs/design_docs/physical-planning-and-deployment.md diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index b6f9a009..c4a4d321 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -68,7 +68,7 @@ Deployments supply storage/ingestion sources and adapt output protocols. The library has no ASAPQuery-backend dependency. Backend raw Scan remains a separate deployment capability. -See [the design](../../docs/design_docs/physical-operators.md). +See [the design](../../docs/design_docs/physical-planning-and-deployment.md). ## Module boundaries diff --git a/docs/design_docs/physical-operators.md b/docs/design_docs/physical-operators.md deleted file mode 100644 index bebad582..00000000 --- a/docs/design_docs/physical-operators.md +++ /dev/null @@ -1,300 +0,0 @@ -# Shared Physical Operators and DAG Execution - -## 1. Problem - -ASAPPlanner describes valid summary computations. Precompute and query engines -need concrete operators to execute them. Implementing these operators and their -execution separately duplicates work and makes it harder to preserve Planner -semantics across deployments. - -This design provides a shared physical execution library for asap-fusion and -ASAPQuery. It contains both physical operators and the runtime for executing a -DAG. Deployment compilers use the shared physical operator vocabulary to describe -computation; deployment engines supply inputs and invoke the shared executor. - -The intended outcome is that the same summary build, merge and readout can run -in either engine with the same semantics and execution contracts. - -## 2. Goals and Non-goals - -Goals: - -- Share relational and summary operators across precompute and query execution. -- Preserve logical semantics when compiling a computation to physical operators. -- Execute shared producers once per run, with independent consumers. -- Provide common failure, cancellation, backpressure and resource contracts. -- Keep deployment decisions separate from computation execution. - -This library does not choose deployment placement, manage durable storage, -schedule maintenance or serve requests. Partitioned parallelism, sharding, spill -and cost-based algorithm selection are outside the initial scope. - -## 3. Architecture - -### 3.1 Three layers - -Terminology follows the [ASAPPlanner design overview](architecture/README.md). -There are three layers: - -```text -Logical planning — ASAPPlanner - PlanSpace → selected logical Post-ASAP DAG - ↓ -Deployment compilation — deployment PhysicalPlanCompiler - Deployment plan: physical DAGs + input bindings + operational configuration - ↓ -Shared execution — physical operators and DAG runtime - Bound inputs + physical DAG + run context → results -``` - -| Layer | Responsibility | Output | -| --- | --- | --- | -| Logical planning | Define legal computations, summary families, parameters and guarantees. | Logical candidates and a selected Post-ASAP DAG. | -| Deployment compilation | Choose a feasible physical realization and establish its inputs, materializations, placement and operational configuration. | A deployment plan containing physical DAGs and their deployment contracts. | -| Shared execution | Run the specified physical computation and coordinate dependencies and resources. | Result streams or an explicit failure. | - -`PlanSpace` is the primary output of logical candidate search. Selection and -assembly produce the logical DAG used for deployment compilation. Downstream -physical evidence may inform selection through Planner's interfaces; the flow -does not require choosing a candidate without considering physical feasibility. -Planner-owned semantics and guarantees remain authoritative. See the -[planning workflows](architecture/input-output-workflow.md). - -**The deployment decides what to run, where, when and with which inputs. The -shared executor runs that computation.** Scheduling, plan installation, -persistence and serving are deployment activities, not additional planning -layers. - -### 3.2 Two computation graphs - -| Graph | Describes | Owner | -| --- | --- | --- | -| Logical Post-ASAP DAG | Selected computation semantics, summary operations and shared dependencies. | Planner | -| Physical DAG | Concrete operator choices, configuration, ordered dependencies and input slots. | Shared physical-plan contract, constructed by deployment compilation | - -A logical node may expand into several physical operators. Several consumers -may share one physical producer. A materialized intermediate result may become -an input, excluding its upstream computation from that execution. - -There is no additional execution IR between these graphs. Exporting a logical -DAG as node IDs and edges is a serialization detail. Loading a physical DAG and -instantiating its operator objects is an execution detail. Neither operation -introduces a new semantic plan. - -A deployment plan is a container for physical DAGs and their operational -configuration, not another computation graph. It can hold distinct precompute -and query DAGs that use the same physical operator vocabulary. - -### 3.3 Deployment compiler and shared executor - -The backend `PhysicalPlanCompiler` is the deployment compilation entry point. -It turns a selected logical Post-ASAP DAG into a feasible deployment contract: - -- Lower computation to the shared physical operator vocabulary. -- Select input boundaries and bind them to raw sources or materializations. -- Establish concrete window/state layout, placement and transmission contracts. -- Produce consistent precompute and query plans referencing the same state - identities and schemas. - -The shared library owns operator definitions, validation and implementations. -The compiler uses these contracts to construct physical DAGs; it does not -maintain a separate query or precompute operator hierarchy. - -`QueryPlan` contains query identity, input/materialization bindings, a physical -DAG and fallback policy. `PrecomputePlan` contains physical DAGs together with -their trigger, input, state and publication contracts. Catalog and transmission -configuration connect producers and consumers without redefining computation. - -At execution time, the deployment resolves input slots to readers or supplied -state and instantiates the specified operators. This step does not reselect -algorithms, change summary semantics or make new deployment decisions. The -executor runs the resulting graph; the deployment handles its results. - -## 4. Execution Model - -### 4.1 Example: one KLL summary, two quantiles - -```text -Precompute DAG Query DAG - -raw input slot stored-state input slot - ↓ ↓ - KLL build KLL merge - ↓ ┌────┴────┐ - state output ↓ ↓ - p50 readout p99 readout -``` - -Logical planning selects KLL computation and the required readouts. Deployment -compilation decides where summaries are built and stored, assigns materialization -identities and defines the input contract for each DAG. - -The precompute engine supplies a finite input window, executes the build DAG -and persists its output. The query engine supplies compatible stored states and -executes the query DAG. Merge runs once; both quantile operators independently -consume its output. - -Storage retrieval and publication are deployment responsibilities. Build, merge, -readout and dependency execution use the shared library in both engines. - -### 4.2 One execution - -The executor takes a physical DAG, resolved inputs, selected output roots and -a run context. Before opening sources, it validates the reachable computation -and input contracts. It then creates fresh mutable execution state and returns -output streams. - -The deployment drives those streams and handles completion or failure. Multiple -requested streams must be polled concurrently so a slow consumer does not prevent -progress through a shared producer's bounded queues. Dropping one consumer leaves -its siblings active. - -The same physical DAG can execute repeatedly. Operator state, consumer cursors -and resource reservations belong to an individual run. Persistent state is -supplied explicitly through deployment inputs rather than implicitly retained -between runs. - -## 5. Execution Contracts - -| Contract | Invariant | -| --- | --- | -| C1 — Validate before execution | Reject invalid topology, arity, schemas, operator configurations and input boundedness before sources start. | -| C2 — Execute shared producers once | A reachable physical producer executes once per run, regardless of consumer count. | -| C3 — Isolate runs | Independent runs have independent mutable execution state. | -| C4 — Apply backpressure | Bounded communication queues limit retained output; consumers have independent cursors. | -| C5 — Propagate failure and cancellation | Errors are explicit, and cancellation terminates affected work and releases its resources. | -| C6 — Enforce the tracked resource budget | Retained outputs and estimated operator workspace count against the run's byte budget. | - -Blocking operators require finite input and finalize after it ends. Unknown -boundedness is insufficient for these operators. Incremental operators may emit -before the input completes. Operator properties must make this distinction -available to validation. - -Cancellation is cooperative, including within long computation loops. Individual -synchronous kernel calls limit cancellation responsiveness. Memory accounting -covers tracked allocations, not process RSS or every temporary allocation peak. -Without spill support, exceeding the tracked budget fails the run. - -The executor reports errors; deployment policy decides whether to retry or invoke -an explicit fallback. It must not silently replace failed computation or turn an -error into an empty result. - -## 6. Inputs and Execution Boundaries - -A physical DAG exposes typed input slots. The deployment plan binds each slot -to raw data or an already-computed result. - -### 6.1 Raw inputs - -A raw input contract identifies the source, schema and required data scope. -The deployment supplies a reader satisfying that contract. Readers declare -boundedness, support cancellation and report schema drift or I/O failures. -Opening readers is deferred until execution validation succeeds. - -### 6.2 Materialized inputs - -Deployment compilation can replace an upstream computation with a compatible -materialized result: - -```text -Logical computation: Scan → KLL build → merge → readout - -Physical query DAG: stored-state input → merge → readout -``` - -This boundary is chosen during deployment compilation. The executor receives an -explicit input slot; it does not discover a materialization or decide to omit -upstream work at serving time. - -Compatibility includes summary family and parameters, schema, grouping, window -coverage and revision scope. The deployment must establish materialization -readiness and the applicable accuracy guarantee before using the result. -Successful byte decoding or schema matching alone is insufficient. - -## 7. Implementation Organization - -These are modules within the shared execution library, not architectural layers: - -| Module | Owns | -| --- | --- | -| `plan` | Shared physical operator vocabulary, DAG structure, properties and validation | -| `binding` | Instantiate specified operators and resolve supplied input slots | -| `runtime` | Per-run state, streams, shared producers and resource accounting | -| `operators` | Relational and temporal physical implementations | -| `operators/summary` | Summary build, merge and readout implementations | -| `expressions` | Scalar expression evaluation | -| `summary_kernels` | Planner-facing sketch adapters and exact accumulators | -| `sources` | Reader interfaces and input adapters | -| `stored_state` | Supported state decoding, delta application and reconstruction | - -Sketch algorithms belong to `asap_sketchlib`. Storage selection and engine -policies belong to deployment repositories. The deployment compiler consumes -the shared operator contracts to lower logical computation. - -Keeping the execution library alongside Planner allows one PR to change an IR -operation and its implementation, and enables tests across internal planning -and execution interfaces. Repository location does not transfer deployment -ownership to the planning library. - -## 8. Alternatives Considered - -### 8.1 Shared ASAP execution — selected - -One physical operator vocabulary and executor support both precompute and query -engines. ASAP controls summary-state edges and shared-producer semantics directly. -The cost is maintaining operator correctness, resource management and future -execution optimizations. - -Separate operator systems in each deployment would duplicate these responsibilities -and require repeated consistency work. Deployment differences are expressed -through input and operational contracts instead. - -### 8.2 DataFusion execution - -DataFusion provides established operators and execution infrastructure. Reuse -requires integrating ASAP's summary semantics, state transport, sharing and -lifecycle contracts. Extensions do not inherently require changing DataFusion -core, but optimizations must preserve the supplied semantics. - -This design selects native execution, following DataFusion's separation of -operator contracts and implementation responsibilities. It makes no claim of -lower runtime overhead. See the [DataFusion comparison](datafusion-execution-comparison.md). - -## 9. Validation - -The design requires three levels of automated validation: - -| Level | Required behavior | -| --- | --- | -| Operators | Correct results and schemas across supported types, empty/null inputs, errors and resource limits. Summary build/merge/readout preserves the intended state semantics. | -| Physical DAG execution | Correct composition, shared producers, independent consumers, repeated runs, cancellation and resource release. | -| Planning and deployment integration | A selected Post-ASAP DAG compiles to consistent deployment contracts and executes through the shared library with correct input identities and results. | - -The KLL example must verify both precompute output and query readouts, including -one merge execution for two consumers. Materialized-input tests must reject -incompatible state and establish that excluded upstream work does not run. -Full-path tests must start with SQL/PromQL and raw inputs, then exercise planning, -deployment compilation and native execution rather than injecting all computed -intermediates. - -Existing operator, DAG and resource tests provide a foundation. The weighted -TopK integration exercises a Planner-selected computation with supplied rate -results; raw Scan integration uses a manually constructed DAG. These do not -establish the complete three-layer integration. Automated tests have been run -locally; no separate manual deployment-level verification has been performed. - -Deployment acceptance additionally covers real connectors, installation, -window/revision handling, persistence, publication and serving. This document -defines the target design; it does not claim those integrations are complete. - -## 10. Limitations and Future Work - -The initial execution scope is supported computation with bounded input wherever -blocking operators require it. Logical validity alone does not establish physical -feasibility: deployment compilation must reject an unsupported realization before -committing it, or select an explicit deployment fallback. - -Parallelism, sharding, spill, richer ordering/distribution properties and physical -algorithm costing can extend the same physical-plan contract. They must preserve -C1–C6 and the three ownership boundaries. They do not require additional semantic -IR layers. diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md new file mode 100644 index 00000000..2d7f0726 --- /dev/null +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -0,0 +1,195 @@ +# Physical Planning, Summary Maintenance, and Deployment + +## 1. Architecture + +A Post-ASAP computation is progressively realized through four layers: + +```mermaid +flowchart LR + L["Logical Post-ASAP DAG
Computation semantics"] + M["Summary Maintenance Lifecycle
State lifecycle"] + P["Physical DAG
Executable computation"] + D["Deployment Plan / DAG
System instantiation"] + + L -->|"Summary Maintenance
Selection"| M + M -->|"Physical Plan
Compiler"| P + P -->|"Deployment Plan
Compiler"| D +``` + +| Layer | Defines | +| --- | --- | +| **Logical Post-ASAP DAG** | What computation should happen | +| **Summary Maintenance Lifecycle** | How summary state is maintained | +| **Physical DAG** | How the computation is executed | +| **Deployment Plan / DAG** | How it is instantiated in a concrete system | + +ASAPPlanner owns computation, maintenance selection, physical planning, and the +shared physical operator implementation library. Deployment systems such as +ASAPQuery and asap-fusion own deployment compilation and operation. + +## 2. Logical Post-ASAP DAG → Summary Maintenance Lifecycle + +The Logical Post-ASAP DAG defines computation semantics: + +```text +Scan → KLLBuild(k=200) → KLLMerge ─┬→ Quantile(0.50) + └→ Quantile(0.99) +``` + +It specifies operators, summary parameters, dependencies, sharing, and guarantees. +It does not specify how summary state is maintained. + +**Summary Maintenance Selection** chooses the lifecycle of each summary producer +using workload demand, window/freshness requirements, physical feasibility, and cost. + +Typical strategies are: + +- build per request; +- build, retain, and reuse; or +- continuously maintain. + +The result is a **Summary Maintenance Lifecycle** describing the selected strategy +and applicable window, freshness, reuse, and retention requirements. + +```text +Logical Post-ASAP DAG ++ workload requirements ++ physical feasibility/cost + ↓ +Summary Maintenance Selection + ↓ +Summary Maintenance Lifecycle +``` + +The lifecycle is a planning contract associated with the logical computation, not +a separate computation IR. Selection may request physical candidates and use their +feasibility and cost to reconsider maintenance candidates; this is not an +irreversible pass. + +## 3. Summary Maintenance Lifecycle → Physical DAG + +The **Physical Plan Compiler** lowers the logical computation and selected lifecycle +into executable physical operators: + +```text +Logical Post-ASAP DAG ++ Summary Maintenance Lifecycle ++ physical capabilities + ↓ +Physical Plan Compiler + ↓ +Physical DAG(s) +``` + +It selects physical implementations, compiles expressions, resolves types and +schemas, preserves dependencies and sharing, creates typed input boundaries, and +validates physical requirements. + +For example: + +```text +InputSlot(k=200, grouping=G) + ↓ +NativeKllMerge(k=200) + ├──→ NativeQuantile(0.50) + └──→ NativeQuantile(0.99) +``` + +A **Physical DAG** contains concrete operators, compiled expressions, typed input +slots, dependencies, output roots, and execution properties. + +It remains deployment-independent: materialization IDs, storage locations, +placement, and scheduling are not part of the Physical DAG. + +If the selected lifecycle cannot be physically realized, compilation fails and +planning may reconsider the candidate. + +## 4. Physical DAG → Deployment Plan / DAG + +The **Deployment Plan Compiler** binds a Physical DAG to a concrete deployment: + +```text +Physical DAG(s) ++ Summary Maintenance Lifecycle ++ deployment catalog/state ++ sources/materializations ++ operational policy + ↓ +Deployment Plan Compiler + ↓ +Deployment Plan / DAG +``` + +It determines: + +- concrete source and materialization bindings; +- storage and placement; +- scheduling and lifecycle execution; +- readiness and revision checks; and +- persistence, publication, or serving behavior. + +For example: + +```text +InputSlot + ↓ +materialization "latency-kll-5m" + ↓ +object-store reader +``` + +The deployment compiler does not lower logical operators or choose a different +physical algorithm. If the selected physical computation cannot be bound correctly, +it fails or requests replanning. A Deployment Plan / DAG is an operational +instantiation, not another computation IR. + +## 5. Materialized Boundaries + +A selected maintenance strategy may place a materialized boundary inside the +logical computation: + +```text +Scan → KLLBuild → KLLMerge → Quantile + ↑ + materialized +``` + +The corresponding query Physical DAG becomes: + +```text +InputSlot → KLLMerge → Quantile +``` + +Responsibilities remain separated: + +- **Summary Maintenance Selection** decides that the KLL state should be maintained + and reused. +- **Physical Plan Compiler** constructs the Physical DAG with an explicit typed + boundary. +- **Deployment Plan Compiler** binds that boundary to a concrete compatible + materialization. + +Reuse requires compatible summary family and parameters, schema, grouping, +population, window coverage, revision scope, and guarantees. + +## 6. Execution + +The deployment engine resolves the Deployment Plan and calls ASAPPlanner's shared +physical operator implementation library, `asap-physical-operators`: + +```text +Deployment Plan / DAG + ↓ deployment engine resolves inputs and invokes +Physical DAG + resolved inputs + RunContext + ↓ +ASAPPlanner shared physical operator implementation library + concrete operators + DAG runtime + ↓ +Results +``` + +The runtime executes the supplied Physical DAG. It does not select maintenance +strategies, discover materializations, or make deployment decisions. + +Shared producers execute once per run, while failure, cancellation, backpressure, +and resource management follow the common runtime contract. From 9b0d74fe2547f21acf1949da88c8979e23f8ca9c Mon Sep 17 00:00:00 2001 From: zz_y Date: Fri, 25 Sep 2026 17:55:21 +0000 Subject: [PATCH 24/90] docs: explain planning boundaries with running KLL pane example --- .../physical-planning-and-deployment.md | 342 ++++++++++++------ 1 file changed, 235 insertions(+), 107 deletions(-) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 2d7f0726..675ea6f0 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -6,10 +6,10 @@ A Post-ASAP computation is progressively realized through four layers: ```mermaid flowchart LR - L["Logical Post-ASAP DAG
Computation semantics"] - M["Summary Maintenance Lifecycle
State lifecycle"] - P["Physical DAG
Executable computation"] - D["Deployment Plan / DAG
System instantiation"] + L["Logical Post-ASAP DAG
What computation?"] + M["Summary Maintenance Lifecycle
How is state maintained?"] + P["Physical DAG(s)
How is it executed?"] + D["Deployment Plan / DAG
How is it instantiated?"] L -->|"Summary Maintenance
Selection"| M M -->|"Physical Plan
Compiler"| P @@ -18,58 +18,149 @@ flowchart LR | Layer | Defines | | --- | --- | -| **Logical Post-ASAP DAG** | What computation should happen | -| **Summary Maintenance Lifecycle** | How summary state is maintained | -| **Physical DAG** | How the computation is executed | -| **Deployment Plan / DAG** | How it is instantiated in a concrete system | +| **Logical Post-ASAP DAG** | Computation semantics | +| **Summary Maintenance Lifecycle** | Build, retention, reuse, and window strategy | +| **Physical DAG(s)** | Concrete executable operators and input boundaries | +| **Deployment Plan / DAG** | Concrete data/state bindings and operational lifecycle | -ASAPPlanner owns computation, maintenance selection, physical planning, and the -shared physical operator implementation library. Deployment systems such as -ASAPQuery and asap-fusion own deployment compilation and operation. +ASAPPlanner owns the first three layers and the shared physical operator +implementation library. Deployment systems such as ASAPQuery and asap-fusion +own deployment compilation and operation. The lifecycle is a planning contract +associated with the logical DAG, not a separate computation IR. -## 2. Logical Post-ASAP DAG → Summary Maintenance Lifecycle +### Running example + +Suppose p50 and p99 are requested over the same five-minute latency population, +and ASAP selects KLL with `k=200`. Assume query windows align with one-minute +pane boundaries and that the selected parameters satisfy the required guarantees. +Operator names below are illustrative; the example defines the design, not a +claim that the entire deployment integration is implemented. -The Logical Post-ASAP DAG defines computation semantics: +The example evolves through the architecture as follows: ```text -Scan → KLLBuild(k=200) → KLLMerge ─┬→ Quantile(0.50) - └→ Quantile(0.99) -``` +1. Logical Post-ASAP DAG -It specifies operators, summary parameters, dependencies, sharing, and guarantees. -It does not specify how summary state is maintained. +raw latency + ↓ +KLLBuild(k=200) + ↓ +KLLMerge + ┌─┴─────┐ + ↓ ↓ + p50 p99 -**Summary Maintenance Selection** chooses the lifecycle of each summary producer -using workload demand, window/freshness requirements, physical feasibility, and cost. + │ + │ Summary Maintenance Selection + ▼ -Typical strategies are: +2. Summary Maintenance Lifecycle -- build per request; -- build, retain, and reuse; or -- continuously maintain. +KLLBuild(k=200) + strategy = continuously maintain + window = 1-minute panes + reuse = p50 + p99 + query = merge panes covering requested aligned 5 minutes -The result is a **Summary Maintenance Lifecycle** describing the selected strategy -and applicable window, freshness, reuse, and retention requirements. + │ + │ Physical Plan Compiler + ▼ -```text -Logical Post-ASAP DAG -+ workload requirements -+ physical feasibility/cost +3. Physical DAGs + +Maintenance DAG: +RawInput + ↓ +NativeKllBuild(k=200) + ↓ +KllStateOutput + +Query DAG: +InputSlot[5 panes] ↓ -Summary Maintenance Selection +NativeKllMerge(k=200) + ┌─┴────────┐ + ↓ ↓ + NativeP50 NativeP99 + + │ + │ Deployment Plan Compiler + ▼ + +4. Deployment Plan / DAG + +Maintenance: +OTLP latency source + ↓ +run KLL build over each complete 1-minute input pane ↓ -Summary Maintenance Lifecycle +store as latency-kll-1m/ + +Query: +resolve five latency-kll-1m states + ↓ +execute query Physical DAG + ↓ +return p50 / p99 +``` + +Each stage adds a different class of decision while preserving the preceding +contracts. Here, continuous maintenance means recurring production of pane state; +the bounded build DAG does not itself implement an unbounded streaming window. + +## 2. Logical Post-ASAP DAG → Summary Maintenance Lifecycle + +The **Logical Post-ASAP DAG** defines computation semantics: + +```text +Scan(latency) + ↓ +KLLBuild(k=200) + ↓ +KLLMerge + ┌─┴────────────┐ + ↓ ↓ +Quantile(.5) Quantile(.99) +``` + +It establishes that KLL with `k=200` is used and that the merge is shared by the +two readouts. It does not determine when KLL states are built or retained. + +**Summary Maintenance Selection** makes that decision using workload demand, +window/freshness requirements, and physical feasibility/cost. + +For the running example, assume it selects: + +```text +producer: KLLBuild(k=200) + +strategy: + continuously maintain + +window realization: + 1-minute panes + +query requirement: + combine panes covering the requested aligned 5-minute range + +reuse: + one merged state serves p50 and p99 ``` -The lifecycle is a planning contract associated with the logical computation, not -a separate computation IR. Selection may request physical candidates and use their -feasibility and cost to reconsider maintenance candidates; this is not an -irreversible pass. +This produces the **Summary Maintenance Lifecycle**. + +The lifecycle specifies how the selected logical summary should be maintained, +but not its concrete operator implementation or storage location. + +Physical feasibility may feed back into selection. For example, if the required +pane-based maintenance cannot be implemented, this lifecycle candidate cannot be +selected. One-minute panes alone also cannot cover an arbitrarily phased query +window; that requires supported boundary handling or a different candidate. ## 3. Summary Maintenance Lifecycle → Physical DAG -The **Physical Plan Compiler** lowers the logical computation and selected lifecycle -into executable physical operators: +The **Physical Plan Compiler** consumes both computation semantics and maintenance +requirements: ```text Logical Post-ASAP DAG @@ -81,35 +172,68 @@ Physical Plan Compiler Physical DAG(s) ``` -It selects physical implementations, compiles expressions, resolves types and -schemas, preserves dependencies and sharing, creates typed input boundaries, and -validates physical requirements. +For the running example, the lifecycle creates two execution boundaries. -For example: +### Maintenance Physical DAG ```text -InputSlot(k=200, grouping=G) +RawInputSlot( + window = 1m, + bounded = true +) + ↓ +NativeKllBuild(k=200) + ↓ +KllStateOutput(k=200) +``` + +This DAG implements construction of each maintained one-minute pane. Its input +contract requires the complete pane population; the deployment supplies that +bounded input from its source integration. + +### Query Physical DAG + +```text +InputSlot( + k = 200, + coverage = requested aligned 5m +) ↓ NativeKllMerge(k=200) - ├──→ NativeQuantile(0.50) - └──→ NativeQuantile(0.99) + ┌─┴──────────────────┐ + ↓ ↓ +NativeQuantile(.50) NativeQuantile(.99) +``` + +The Physical Plan Compiler chooses `NativeKllBuild`, `NativeKllMerge`, and the +physical quantile implementations, validates state compatibility, and preserves +the shared merge. It also resolves expressions, schemas, ordered dependencies +and execution properties. + +The resulting Physical DAGs know that compatible KLL states are required, but +do not know where those states are stored. + +For example: + +```text +InputSlot ``` -A **Physical DAG** contains concrete operators, compiled expressions, typed input -slots, dependencies, output roots, and execution properties. +is physical, while: -It remains deployment-independent: materialization IDs, storage locations, -placement, and scheduling are not part of the Physical DAG. +```text +s3://.../latency-kll/12:01 +``` -If the selected lifecycle cannot be physically realized, compilation fails and -planning may reconsider the candidate. +is deployment-specific. Placement and scheduling also remain outside the Physical +DAG. If the required behavior cannot be realized, physical compilation fails. ## 4. Physical DAG → Deployment Plan / DAG -The **Deployment Plan Compiler** binds a Physical DAG to a concrete deployment: +The **Deployment Plan Compiler** binds the Physical DAGs to the concrete deployment: ```text -Physical DAG(s) +Physical DAGs + Summary Maintenance Lifecycle + deployment catalog/state + sources/materializations @@ -120,76 +244,80 @@ Deployment Plan Compiler Deployment Plan / DAG ``` -It determines: - -- concrete source and materialization bindings; -- storage and placement; -- scheduling and lifecycle execution; -- readiness and revision checks; and -- persistence, publication, or serving behavior. - -For example: +For the maintenance DAG, it may produce: ```text -InputSlot - ↓ -materialization "latency-kll-5m" - ↓ -object-store reader -``` - -The deployment compiler does not lower logical operators or choose a different -physical algorithm. If the selected physical computation cannot be bound correctly, -it fails or requests replanning. A Deployment Plan / DAG is an operational -instantiation, not another computation IR. +Source: + RawInputSlot + → complete bounded panes from the OTLP latency source -## 5. Materialized Boundaries +Schedule: + each 1-minute pane, once its completion requirements are met -A selected maintenance strategy may place a materialized boundary inside the -logical computation: +Execution: + RawInput → NativeKllBuild(k=200) -```text -Scan → KLLBuild → KLLMerge → Quantile - ↑ - materialized +Output: + KllStateOutput + → latency-kll-1m/ ``` -The corresponding query Physical DAG becomes: +For a query over `(12:00, 12:05]`, its input-binding rule resolves: ```text -InputSlot → KLLMerge → Quantile +InputSlot[5 panes] + ├── latency-kll-1m/(12:00,12:01] + ├── latency-kll-1m/(12:01,12:02] + ├── latency-kll-1m/(12:02,12:03] + ├── latency-kll-1m/(12:03,12:04] + └── latency-kll-1m/(12:04,12:05] + ↓ + Query Physical DAG + ↓ + p50, p99 ``` -Responsibilities remain separated: +The Deployment Plan Compiler establishes bindings and checks that their contracts +satisfy the physical inputs and selected lifecycle, including KLL parameters, +grouping, population, window coverage and revision scope. The deployment engine +resolves request-specific states and checks their actual coverage, revisions and +readiness at execution time. A compiled plan cannot establish future readiness. -- **Summary Maintenance Selection** decides that the KLL state should be maintained - and reused. -- **Physical Plan Compiler** constructs the Physical DAG with an explicit typed - boundary. -- **Deployment Plan Compiler** binds that boundary to a concrete compatible - materialization. +The compiler does not replace `NativeKllMerge`, choose another sketch, or decide +to maintain different windows. Such changes require replanning. A Deployment +Plan / DAG is an operational instantiation, not another computation IR. -Reuse requires compatible summary family and parameters, schema, grouping, -population, window coverage, revision scope, and guarantees. +## 5. Responsibility Boundary -## 6. Execution +The complete example makes the ownership boundary explicit: -The deployment engine resolves the Deployment Plan and calls ASAPPlanner's shared -physical operator implementation library, `asap-physical-operators`: +| Stage | KLL example decision | +| --- | --- | +| **Logical Post-ASAP DAG** | Use `KLL(k=200)` with shared merge for p50/p99 | +| **Summary Maintenance Selection** | Maintain 1-minute panes and reuse them for aligned five-minute queries | +| **Summary Maintenance Lifecycle** | Record pane/window/freshness/reuse requirements | +| **Physical Plan Compiler** | Lower to native KLL build, merge, and readout operators | +| **Physical DAG** | Define maintenance and query DAGs with typed input/output boundaries | +| **Deployment Plan Compiler** | Bind raw input and KLL state slots to concrete sources/materializations | +| **Deployment Plan / DAG** | Specify maintenance schedules, stored-pane resolution and query execution | ```text -Deployment Plan / DAG - ↓ deployment engine resolves inputs and invokes -Physical DAG + resolved inputs + RunContext - ↓ -ASAPPlanner shared physical operator implementation library - concrete operators + DAG runtime - ↓ -Results -``` +Logical: + "Use KLL for p50/p99." + +Lifecycle: + "Maintain reusable 1-minute KLL panes." -The runtime executes the supplied Physical DAG. It does not select maintenance -strategies, discover materializations, or make deployment decisions. +Physical: + "Execute NativeKllBuild and + NativeKllMerge → {p50, p99}." + +Deployment: + "Read OTLP here, store panes here, + and bind these five panes for this aligned query." +``` -Shared producers execute once per run, while failure, cancellation, backpressure, -and resource management follow the common runtime contract. +The deployment engine executes the bound Physical DAGs through ASAPPlanner's +shared physical operator implementation library, `asap-physical-operators`, and +its DAG runtime. The merge executes once per run for both consumers. Execution +does not introduce additional planning decisions. From 71f9884154956a8373b6274b0cadb43505a54ec2 Mon Sep 17 00:00:00 2001 From: zz_y Date: Fri, 25 Sep 2026 20:48:36 +0000 Subject: [PATCH 25/90] feat: compile physical DAG candidates independently of deployment readers --- crates/asap-physical-operators/README.md | 22 ++- crates/asap-physical-operators/src/dag/mod.rs | 4 +- crates/asap-physical-operators/src/lib.rs | 2 +- .../src/physical_planner/compiled.rs | 151 ++++++++++++++++++ .../src/{binding => physical_planner}/mod.rs | 103 +++++++----- .../tests/physical_semantics.rs | 4 +- .../asap-physical-operators/tests/raw_scan.rs | 55 +++++++ .../tests/weighted_topk_binding.rs | 9 +- 8 files changed, 300 insertions(+), 50 deletions(-) create mode 100644 crates/asap-physical-operators/src/physical_planner/compiled.rs rename crates/asap-physical-operators/src/{binding => physical_planner}/mod.rs (89%) diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index c4a4d321..e1d98069 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -47,12 +47,12 @@ assert!(matches!(batch.rows()[0][0], Value::Int64(-7))); # Ok::<(), asap_physical_operators::dag::Error>(()) ``` -`binding::bind` accepts a post-ASAP DAG and explicit source bindings for -installed ingestion/storage frontiers. It rejects unsupported operations and +`physical_planner::compile` accepts a post-ASAP DAG and typed input contracts. +The resulting candidate is instantiated with deployment readers after selection. It rejects unsupported operations and schema mismatches before starting a source. Implement `PhysicalOperator` for a deployment source, including asynchronous I/O; computation operators remain in the library. The public `planner` export identifies the exact Planner types used -by the crate. The native binder currently supports a subset of those types and +by the crate. The physical compiler currently supports a subset of those types and operations; it does not interpret an unknown node as external fallback. Plain values preserve Planner scalar/collection types and nullability. Numeric @@ -77,7 +77,7 @@ See [the design](../../docs/design_docs/physical-planning-and-deployment.md). - `expressions`: scalar evaluation; typed builders and the Planner expression adapter. - `operators`: projection, filter, joins, aggregate/window, sort, limit and summary implementations. - `sources`: raw-source interface, Scan and the memory connector. -- `binding`: Planner executable DAG binding and installed source frontiers. +- `physical_planner`: native operator lowering, typed input contracts and checked instantiation. - `summary_kernels`: sketchlib state adapters, exact accumulators, update adapters and traits. - `stored_state`: persisted-state decoding, delta reconstruction and readout. - `capability`: explicit kernel and native-batch/readout validation. @@ -97,3 +97,17 @@ row processing and sort merges. Cancellation releases reservations when the stream is polled or dropped. Individual scalar evaluations, bounded sort chunks and sketch kernel calls are synchronous; this is not preemptive execution. There is no spill or partitioned parallel execution in this implementation. + +## Physical compilation and deployment inputs + +`physical_planner::compile` accepts a Planner `ExecutableDag`, typed +`InputContract`s and output roots. It returns a reusable `CompiledPhysicalDag` +containing selected native operators and no live readers. Compilation validates +schemas, input ordering, sharing and boundedness before deployment source access. + +A deployment calls `CompiledPhysicalDag::instantiate` with exactly the declared +inputs. This checks source schemas and execution properties and constructs the +runnable graph without repeating logical lowering. The graph executes through +the shared runtime with independent per-run state. Window coverage, revision and +maintenance-policy admission remain deployment/planning contracts; this compiler +does not discover storage or silently change a selected maintenance strategy. diff --git a/crates/asap-physical-operators/src/dag/mod.rs b/crates/asap-physical-operators/src/dag/mod.rs index a3ea3a7e..c7837672 100644 --- a/crates/asap-physical-operators/src/dag/mod.rs +++ b/crates/asap-physical-operators/src/dag/mod.rs @@ -1,8 +1,8 @@ -//! Compatibility imports. New code should use plan, runtime, operators, binding and sources directly. +//! Compatibility imports. New code should use plan, runtime, operators, physical_planner and sources directly. pub use crate::plan::{NodeId, PhysicalDag, PhysicalOperator}; pub use crate::runtime::batch_execution; pub use crate::runtime::{ Input, Limits, OutputStream, Reservation, RunContext, Scope, SharedValue, }; pub use crate::Error; -pub use crate::{binding as planner, expressions, operators, sources as scan, values}; +pub use crate::{expressions, operators, physical_planner as planner, sources as scan, values}; diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs index dc726214..5709a586 100644 --- a/crates/asap-physical-operators/src/lib.rs +++ b/crates/asap-physical-operators/src/lib.rs @@ -28,9 +28,9 @@ pub mod stored_state; mod error; pub use error::Error; -pub mod binding; pub mod expressions; pub mod operators; +pub mod physical_planner; pub mod plan; pub mod runtime; pub mod sources; diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs new file mode 100644 index 00000000..4b4f0884 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -0,0 +1,151 @@ +//! Reader-independent physical computation and checked deployment instantiation. +use super::*; + +/// A typed execution boundary, without storage identity or a live reader. +#[derive(Clone, Debug)] +pub struct InputContract { + pub schema: Schema, + pub properties: PlanProperties, +} +impl InputContract { + pub fn bounded(schema: Schema) -> Self { + Self { + schema, + properties: PlanProperties { + boundedness: Boundedness::Bounded, + emission: Emission::Unknown, + }, + } + } + pub fn from_source(source: &dyn PhysicalOperator) -> Self { + Self { + schema: source.output_schema(), + properties: source.properties(&[]), + } + } +} +#[derive(Clone)] +enum Node { + Input(InputContract), + Operator { + inputs: Vec, + operator: Operator, + }, +} + +/// Selected native operators and input slots. Rebinding never repeats lowering. +#[derive(Clone)] +pub struct CompiledPhysicalDag { + nodes: BTreeMap, + roots: Vec, +} +impl CompiledPhysicalDag { + pub(super) fn new(roots: Vec) -> Self { + Self { + nodes: BTreeMap::new(), + roots, + } + } + pub(super) fn add_input(&mut self, id: NodeId, contract: InputContract) -> Result<(), Error> { + self.insert(id, Node::Input(contract)) + } + pub(super) fn add( + &mut self, + id: NodeId, + inputs: Vec, + operator: Operator, + ) -> Result<(), Error> { + self.insert(id, Node::Operator { inputs, operator }) + } + fn insert(&mut self, id: NodeId, node: Node) -> Result<(), Error> { + if self.nodes.insert(id, node).is_some() { + return Err(invalid(format!("duplicate physical node {id}"))); + } + Ok(()) + } + pub fn roots(&self) -> &[NodeId] { + &self.roots + } + pub fn input_contracts(&self) -> impl Iterator { + self.nodes.iter().filter_map(|(&id, node)| match node { + Node::Input(contract) => Some((id, contract)), + Node::Operator { .. } => None, + }) + } + /// Validate using contract-only sources. No deployment reader is available. + pub fn validate(&self) -> Result<(), Error> { + let sources = self + .input_contracts() + .map(|(id, c)| (id, Box::new(c.clone()) as Source<'_>)) + .collect(); + self.instantiate(sources).map(|_| ()) + } + /// Resolve exactly the declared inputs and validate before any source starts. + pub fn instantiate<'a>( + &self, + mut sources: BTreeMap>, + ) -> Result, Error> { + let mut graph = PhysicalDag::default(); + for (&id, node) in &self.nodes { + match node { + Node::Input(contract) => { + let source = sources + .remove(&id) + .ok_or_else(|| invalid(format!("missing physical input {id}")))?; + let actual = source.properties(&[]); + if !source.input_schemas().is_empty() + || source.output_schema() != contract.schema + || (contract.properties.boundedness != Boundedness::Unknown + && actual.boundedness != contract.properties.boundedness) + || (contract.properties.emission != Emission::Unknown + && actual.emission != contract.properties.emission) + { + return Err(invalid(format!( + "physical input {id} violates its compiled contract" + ))); + } + graph.add_boxed( + id, + vec![], + Box::new(CheckedSource { + source, + output: contract.schema.clone(), + }), + )?; + } + Node::Operator { inputs, operator } => { + graph.add(id, inputs.clone(), operator.clone())?; + } + } + } + if !sources.is_empty() { + return Err(invalid("unexpected physical input binding")); + } + graph.validate(&self.roots)?; + Ok(graph) + } +} +impl PhysicalOperator for InputContract { + fn name(&self) -> &str { + "UnresolvedInput" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.schema.clone() + } + fn properties(&self, _: &[PlanProperties]) -> PlanProperties { + self.properties + } + fn output_bytes(&self, batch: &Batch) -> usize { + batch.bytes() + } + fn start<'a>( + &'a self, + _: Vec>, + _: crate::runtime::RunContext, + ) -> Result, Error> { + Err(invalid("physical input must be resolved before execution")) + } +} diff --git a/crates/asap-physical-operators/src/binding/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs similarity index 89% rename from crates/asap-physical-operators/src/binding/mod.rs rename to crates/asap-physical-operators/src/physical_planner/mod.rs index 03dad551..0596a8a3 100644 --- a/crates/asap-physical-operators/src/binding/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -1,8 +1,8 @@ -//! Bind a post-ASAP DAG to native operators. Sources are explicit execution -//! frontiers supplied by the deployment; unsupported computation is an error. +//! Compile logical computation to native operators with typed external inputs. +//! Compilation needs no readers; deployment resolves inputs after selection. use crate::{ operators::{Expression, Operator, Reduction, SortKey}, - plan::{NodeId, PhysicalDag, PhysicalOperator}, + plan::{Boundedness, Emission, NodeId, PhysicalDag, PhysicalOperator, PlanProperties}, values::{Batch, Schema}, Error, }; @@ -28,31 +28,74 @@ fn invalid(message: impl Into) -> Error { /// A deployment must authorize these frontiers before calling this function. pub type Source<'a> = Box + 'a>; +mod compiled; +pub use compiled::{CompiledPhysicalDag, InputContract}; + +/// Compile computation without opening or retaining deployment readers. +/// Input contracts identify explicit boundaries selected by maintenance planning. +pub fn compile( + dag: &ExecutableDag, + inputs: BTreeMap, + roots: &[NodeId], +) -> Result { + compile_internal(dag, inputs, roots) +} + +/// Convenience for callers that already resolved inputs. Lowering still uses +/// only their contracts, and instantiation checks those contracts again. pub fn bind<'a>( dag: &ExecutableDag, sources: BTreeMap>, roots: &[NodeId], ) -> Result, Error> { - bind_internal(dag, sources, roots, None) + let inputs = sources + .iter() + .map(|(&id, source)| (id, InputContract::from_source(source.as_ref()))) + .collect(); + compile(dag, inputs, roots)?.instantiate(sources) } -/// Bind raw Planner Scan leaves through registered connectors. Other retained -/// pre-ASAP expressions remain unsupported; they are not executed externally. +/// Resolve raw scan connectors before invoking the reader-independent compiler. pub fn bind_with_data_sources<'a>( dag: &ExecutableDag, - sources: BTreeMap>, + mut sources: BTreeMap>, roots: &[NodeId], data_sources: &crate::sources::DataSources, ) -> Result, Error> { - bind_internal(dag, sources, roots, Some(data_sources)) + // Only resolve scans reachable below the selected input boundaries. + let mut pending = roots.to_vec(); + let mut seen = BTreeSet::new(); + while let Some(id) = pending.pop() { + if !seen.insert(id) || sources.contains_key(&id) { + continue; + } + let node = dag + .nodes + .iter() + .find(|n| u64::from(n.id.0) == id) + .ok_or_else(|| invalid(format!("missing node {id}")))?; + if let Payload::Fallback { + expression: expression @ QueryExpr::Scan { .. }, + } = &node.payload + { + sources.insert(id, Box::new(data_sources.bind(expression)?)); + } else { + pending.extend( + dag.edges + .iter() + .filter(|e| u64::from(e.consumer.0) == id) + .map(|e| u64::from(e.producer.0)), + ); + } + } + bind(dag, sources, roots) } -fn bind_internal<'a>( +fn compile_internal( dag: &ExecutableDag, - mut sources: BTreeMap>, + mut sources: BTreeMap, roots: &[NodeId], - data_sources: Option<&crate::sources::DataSources>, -) -> Result, Error> { +) -> Result { preflight_depth(dag)?; dag.validate().map_err(|e| invalid(e.to_string()))?; let nodes = dag @@ -103,32 +146,17 @@ fn bind_internal<'a>( } } } - let mut graph = PhysicalDag::default(); + let mut graph = CompiledPhysicalDag::new(roots.to_vec()); let mut auxiliary = u64::MAX; for id in ordered { let node = nodes[&id]; let output = Arc::new(node.output_schema.clone()); crate::values::validate_schema(&output)?; - let (operator, inputs) = if let Some(source) = sources.remove(&id) { - if !source.input_schemas().is_empty() || source.output_schema() != output { - return Err(invalid("frontier is not a source with the declared schema")); + if let Some(source) = sources.remove(&id) { + if source.schema != output { + return Err(invalid("frontier does not have the declared schema")); } - ( - Box::new(CheckedSource { source, output }) as Source<'a>, - vec![], - ) - } else if let ( - Some(registry), - Payload::Fallback { - expression: expression @ QueryExpr::Scan { .. }, - }, - ) = (data_sources, &node.payload) - { - let scan = registry.bind(expression)?; - if scan.output_schema() != output { - return Err(invalid("Scan output differs from post-ASAP schema")); - } - (Box::new(scan) as Source<'a>, vec![]) + graph.add_input(id, source)?; } else { let mut inputs = dependencies.get(&id).cloned().unwrap_or_default(); let mut schemas = inputs @@ -148,19 +176,18 @@ fn bind_internal<'a>( auxiliary -= 1; schemas.truncate(1); } - let operator = bind_node(node, &schemas) + let operator = compile_node(node, &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; - (Box::new(operator) as Source<'a>, inputs) - }; - graph.add_boxed(id, inputs, operator)?; + graph.add(id, inputs, operator)?; + } } - graph.validate(roots)?; + graph.validate()?; Ok(graph) } /// Bind a Planner node against the schemas supplied by its deployment edges. /// This is the same checked path used by complete DAG binding. -pub fn bind_node(node: &ExecutableDagNode, inputs: &[Schema]) -> Result { +pub fn compile_node(node: &ExecutableDagNode, inputs: &[Schema]) -> Result { for schema in inputs { crate::values::validate_schema(schema)?; } diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs index 6d3bd022..7270bd1e 100644 --- a/crates/asap-physical-operators/tests/physical_semantics.rs +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -339,7 +339,7 @@ fn projection_rejects_expression_bound_to_another_schema() { // A valid Planner MIN/MAX schema must bind even for a non-null input column. #[test] fn global_extrema_bind_with_planner_derived_schema() { - use asap_physical_operators::binding::bind_node; + use asap_physical_operators::physical_planner::compile_node; use planner_types::{ post_asap::*, pre_asap::{AggIntent, Column, GroupKeys, Reduction as PlanReduction}, @@ -374,7 +374,7 @@ fn global_extrema_bind_with_planner_derived_schema() { output_schema: (*output).clone(), guarantee: None, }; - let operator = bind_node(&node, std::slice::from_ref(&input)) + let operator = compile_node(&node, std::slice::from_ref(&input)) .expect("global extremum should bind to its Planner schema"); assert!(operator.schema().fields[0].nullable); let empty = unary(input.clone(), vec![], operator.clone()); diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs index 5aa9e10b..25851c4f 100644 --- a/crates/asap-physical-operators/tests/raw_scan.rs +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -329,3 +329,58 @@ fn empty_sources_and_three_valued_predicates() { }); } } + +// A physical candidate can be compiled once without readers and rebound per run. +#[test] +fn compile_without_readers_and_rebind_inputs() { + use asap_physical_operators::{ + operators::Operator, + physical_planner::{compile, InputContract, Source}, + }; + let (scan, schema, batches) = fixture(); + let dag = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let compiled = compile( + &dag, + BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), + &[2], + ) + .unwrap(); + assert_eq!(compiled.input_contracts().count(), 1); + for _ in 0..2 { + let sources = BTreeMap::from([( + 0, + Box::new(Operator::source(schema.clone(), batches.clone()).unwrap()) as Source<'_>, + )]); + let graph = compiled.instantiate(sources).unwrap(); + let mut outputs = graph.execute(compiled.roots(), context()).unwrap(); + let result = block_on(outputs.remove(0).collect::>()); + assert!(result.iter().all(Result::is_ok)); + assert_eq!( + result + .iter() + .map(|b| b.as_ref().unwrap().rows().len()) + .sum::(), + 2 + ); + } + assert!(compiled.instantiate(BTreeMap::new()).is_err()); +} + +// Input boundedness must be proved during compilation, before readers exist. +#[test] +fn compilation_rejects_unknown_boundedness_for_sort() { + use asap_physical_operators::{ + physical_planner::{compile, InputContract}, + plan::{Boundedness, Emission, PlanProperties}, + }; + let (scan, schema, _) = fixture(); + let dag = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let input = InputContract { + schema, + properties: PlanProperties { + boundedness: Boundedness::Unknown, + emission: Emission::Unknown, + }, + }; + assert!(compile(&dag, BTreeMap::from([(0, input)]), &[2]).is_err()); +} diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index 7761161d..e37c2120 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -8,7 +8,7 @@ use asap_aware_mapping::{ }; use asap_physical_operators::dag::{ operators::Operator, - planner::{bind, Source}, + planner::{compile, InputContract, Source}, values::{Batch, Value}, Limits, RunContext, Scope, }; @@ -153,12 +153,15 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S .unwrap(); let source = Box::new(Operator::source(rates.clone(), vec![batch.clone()]).unwrap()) as Source<'static>; - let graph = bind( + let compiled = compile( &placed, - BTreeMap::from([(rate_id.0 as u64, source)]), + BTreeMap::from([(rate_id.0 as u64, InputContract::bounded(rates.clone()))]), &[dag.root.0 as u64], ) .unwrap(); + let graph = compiled + .instantiate(BTreeMap::from([(rate_id.0 as u64, source)])) + .unwrap(); let context = RunContext::new(scope, Limits::default()).unwrap(); let output = block_on(async { let mut output = Vec::new(); From ad4e611702d655fb979952a13b756d1184c2ad1f Mon Sep 17 00:00:00 2001 From: zz_y Date: Fri, 25 Sep 2026 20:50:05 +0000 Subject: [PATCH 26/90] feat: compose compiled physical fragments with typed inputs --- .../src/physical_planner/compiled.rs | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs index 4b4f0884..bf9f7fda 100644 --- a/crates/asap-physical-operators/src/physical_planner/compiled.rs +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -40,6 +40,23 @@ pub struct CompiledPhysicalDag { roots: Vec, } impl CompiledPhysicalDag { + /// Assemble already-lowered operators and typed external inputs. This is + /// useful for engines that compose multiple compiled computation fragments. + pub fn from_operators( + inputs: BTreeMap, + operators: BTreeMap, Operator)>, + roots: Vec, + ) -> Result { + let mut result = Self::new(roots); + for (id, contract) in inputs { + result.add_input(id, contract)?; + } + for (id, (inputs, operator)) in operators { + result.add(id, inputs, operator)?; + } + result.validate()?; + Ok(result) + } pub(super) fn new(roots: Vec) -> Self { Self { nodes: BTreeMap::new(), From d0c67b20703e0a7a27c0e4b59209e7c167ab958f Mon Sep 17 00:00:00 2001 From: zz_y Date: Wed, 23 Sep 2026 03:28:56 +0000 Subject: [PATCH 27/90] feat: model bounded classic HLL confidence without an RSE shortcut --- .../asap-aware-mapping/src/hll_confidence.rs | 208 ++++++++++++++++++ crates/asap-aware-mapping/src/lib.rs | 1 + 2 files changed, 209 insertions(+) create mode 100644 crates/asap-aware-mapping/src/hll_confidence.rs diff --git a/crates/asap-aware-mapping/src/hll_confidence.rs b/crates/asap-aware-mapping/src/hll_confidence.rs new file mode 100644 index 00000000..bc8f36d9 --- /dev/null +++ b/crates/asap-aware-mapping/src/hll_confidence.rs @@ -0,0 +1,208 @@ +//! Estimator-specific confidence for classic HLL's linear-counting branch. +//! +//! This is conditional on independent uniform bucket hashes and an enforced +//! upper bound on distinct items in the complete readout population (including +//! all merged panes). It is not an RSE-to-normal conversion or an ERP fit. + +use asap_types::post_asap::{ + BoundExpr, ErrorMetric, GuaranteeSource, ProbabilityExpr, ResultGuarantee, +}; + +/// A finite-population contract for `m * ln(m / zero_registers)` with the +/// classic HLL small-range switch. Hashing is assumed independent and uniform. +/// The deployment must establish the population bound; observations alone do +/// not establish it. Unsupported precisions/populations return no certificate. +#[derive(Debug, Clone, Copy)] +pub struct ClassicHllConfidence { + max_distinct: u32, + relative_error: f64, +} + +impl ClassicHllConfidence { + pub fn new(max_distinct: u32, relative_error: f64) -> Option { + (max_distinct > 0 + && max_distinct <= 4096 + && relative_error.is_finite() + && (1e-6..1.0).contains(&relative_error)) + .then_some(Self { + max_distinct, + relative_error, + }) + } + + pub fn guarantee(&self, precision: u8) -> Option { + let delta = self.failure_probability(precision)?; + Some(ResultGuarantee { + metric: ErrorMetric::Cardinality, + bound: BoundExpr::Constant { + value: self.relative_error, + }, + failure_probability: ProbabilityExpr::Constant { value: delta }, + provenance: vec![GuaranteeSource::SketchReadout { + algorithm: "Hll".into(), + contract: "classic_hll_linear_counting_collision_bound_v1".into(), + params: serde_json::json!({"precision": precision, + "max_distinct": self.max_distinct, "relative_error": self.relative_error, + "hash_assumption": "independent_uniform_buckets", + "population_scope": "complete_readout_including_merged_panes"}), + query: "Cardinality".into(), + }], + }) + } + + pub fn precision(&self, delta: f64) -> Option { + if !delta.is_finite() || !(0.0..1.0).contains(&delta) || delta == 0.0 { + return None; + } + (4..=18).find(|&p| self.failure_probability(p).is_some_and(|d| d <= delta)) + } + + /// Finite bound, not an asymptotic RSE fit. With N distinct hashes and K + /// occupied buckets, C=N-K collision arrivals satisfy + /// P(C>=t) <= lambda^t/t!, lambda=N(N-1)/(2m): each arrival's conditional + /// collision probability is at most (i-1)/m, and a union bound over t + /// arrivals is bounded by the t-th power of their sum divided by t!. + /// + /// N<=m/2 makes the classic raw estimate <=2*alpha_m*m<2.5m, + /// so the small-range switch always uses L=-m*ln(1-K/m). Then + /// K<=L<=N + N^2/(2(m-N)). The latter bounds overestimation + /// deterministically; underestimation implies C>epsilon*N. + /// We maximize the collision bound over EVERY integer N in the contract, + /// not just its upper endpoint (small-cardinality tails matter). + fn failure_probability(&self, precision: u8) -> Option { + if !(4..=18).contains(&precision) { + return None; + } + let m = f64::from(1u32 << precision); + let max_n = f64::from(self.max_distinct); + // Reserve numerical slack; do not certify sub-floating-point error. + let eps = self.relative_error * (1.0 - 1e-8); + if max_n > m / 2.0 || max_n / (2.0 * (m - max_n)) > eps { + return None; + } + let mut log_factorial = vec![0.0; self.max_distinct as usize + 1]; + for i in 1..log_factorial.len() { + log_factorial[i] = log_factorial[i - 1] + (i as f64).ln(); + } + let mut worst = 0.0_f64; + for n in 2..=self.max_distinct { + let nf = f64::from(n); + // Including a boundary collision event is conservative. + let t = ((eps * nf).floor() as usize + 1).min(n as usize); + let lambda = nf * (nf - 1.0) / (2.0 * m); + let log_tail = (t as f64) * lambda.ln() - log_factorial[t]; + worst = worst.max(log_tail.min(0.0).exp()); + } + // Never return a spurious zero from underflow or numeric cancellation. + Some((worst * (1.0 + 1e-10) + 1e-12).min(1.0)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// A supported estimator contract supplies a probability, unlike generic HLL RSE. + #[test] + fn bounded_classic_hll_has_a_feasible_confidence_target() { + let model = ClassicHllConfidence::new(128, 0.05).unwrap(); + let precision = model.precision(0.01).expect("finite confidence-sized HLL"); + let guarantee = model.guarantee(precision).unwrap(); + assert!(!guarantee.has_unknown()); + assert!(guarantee.failure_probability.evaluate().unwrap() <= 0.01); + assert_eq!(guarantee.bound.evaluate(), Some(0.05)); + } + /// Tighter confidence must increase precision or explicitly become unavailable. + #[test] + fn sizing_and_domain_limits_are_consistent() { + let model = ClassicHllConfidence::new(128, 0.05).unwrap(); + assert!(model.precision(0.001).unwrap() > model.precision(0.01).unwrap()); + assert!(model.precision(1e-12).is_none()); + assert!(model.precision(0.0).is_none()); + assert!(model.precision(f64::NAN).is_none()); + assert!(model.guarantee(3).is_none()); + assert!(model.guarantee(19).is_none()); + assert!(model.guarantee(7).is_none()); + for (n, e) in [(0, 0.05), (4097, 0.05), (128, 0.0), (128, f64::NAN)] { + assert!(ClassicHllConfidence::new(n, e).is_none()); + } + } + + /// Exact occupancy probabilities independently check both tails for every N. + #[test] + fn probability_bound_dominates_exact_occupancy_distribution() { + for precision in 4..=10 { + let m = 1usize << precision; + let max_n = 64.min(m / 2); + for eps in [0.05, 0.2, 0.6] { + let model = ClassicHllConfidence::new(max_n as u32, eps).unwrap(); + let Some(bound) = model.failure_probability(precision) else { + continue; + }; + let mut occupancy = vec![0.0; max_n + 1]; + occupancy[0] = 1.0; + for n in 1..=max_n { + let mut next = vec![0.0; max_n + 1]; + for k in 0..n { + next[k] += occupancy[k] * k as f64 / m as f64; + next[k + 1] += occupancy[k] * (m - k) as f64 / m as f64; + } + occupancy = next; + let actual: f64 = occupancy + .iter() + .enumerate() + .filter_map(|(k, &prob)| { + let estimate = -(m as f64) * (-(k as f64) / (m as f64)).ln_1p(); + ((estimate - n as f64).abs() > eps * n as f64).then_some(prob) + }) + .sum(); + assert!( + actual <= bound + 1e-12, + "p={precision} n={n} eps={eps}: {actual}>{bound}" + ); + } + } + } + } + /// The model's readout formula matches the actual classic estimator after merge. + #[test] + fn native_classic_estimator_and_merged_registers_use_the_same_contract() { + use asap_sketchlib::sketches::hll::{Classic, HyperLogLogP16}; + let model = ClassicHllConfidence::new(128, 0.05).unwrap(); + assert!( + model + .guarantee(16) + .unwrap() + .failure_probability + .evaluate() + .unwrap() + < 0.01 + ); + let mut single = HyperLogLogP16::::new(); + let mut left = HyperLogLogP16::::new(); + let mut right = HyperLogLogP16::::new(); + for n in 0..128u64 { + // SplitMix64 supplies deterministic test hashes, not a proof of randomness. + let mut h = n.wrapping_add(0x9e3779b97f4a7c15); + h = (h ^ (h >> 30)).wrapping_mul(0xbf58476d1ce4e5b9); + h = (h ^ (h >> 27)).wrapping_mul(0x94d049bb133111eb); + h ^= h >> 31; + single.insert_with_hash(h); + if n % 2 == 0 { + left.insert_with_hash(h); + } else { + right.insert_with_hash(h); + } + } + left.merge(&right); + assert_eq!(single.registers_as_slice(), left.registers_as_slice()); + let zeroes = left + .registers_as_slice() + .iter() + .filter(|&&r| r == 0) + .count(); + let expected = (65536.0 * (65536.0 / zeroes as f64).ln()) as usize; + assert_eq!(left.estimate(), expected); + assert!((expected as f64 - 128.0).abs() / 128.0 <= 0.05); + } +} diff --git a/crates/asap-aware-mapping/src/lib.rs b/crates/asap-aware-mapping/src/lib.rs index 420d5a2f..b9356220 100644 --- a/crates/asap-aware-mapping/src/lib.rs +++ b/crates/asap-aware-mapping/src/lib.rs @@ -161,6 +161,7 @@ pub mod exact_composition; pub mod explanation; mod function_rules; pub mod grouping; +pub mod hll_confidence; pub mod pane_sharing; pub mod physical_handoff_cost; pub mod physical_operator_statistics; From 20aff01b203e081298e866451e333a86c9990428 Mon Sep 17 00:00:00 2001 From: zz_y Date: Wed, 23 Sep 2026 03:31:15 +0000 Subject: [PATCH 28/90] fix: reserve absolute relative-error slack for HLL arithmetic --- crates/asap-aware-mapping/src/hll_confidence.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/crates/asap-aware-mapping/src/hll_confidence.rs b/crates/asap-aware-mapping/src/hll_confidence.rs index bc8f36d9..d7abcd26 100644 --- a/crates/asap-aware-mapping/src/hll_confidence.rs +++ b/crates/asap-aware-mapping/src/hll_confidence.rs @@ -76,7 +76,7 @@ impl ClassicHllConfidence { let m = f64::from(1u32 << precision); let max_n = f64::from(self.max_distinct); // Reserve numerical slack; do not certify sub-floating-point error. - let eps = self.relative_error * (1.0 - 1e-8); + let eps = self.relative_error - 1e-8; if max_n > m / 2.0 || max_n / (2.0 * (m - max_n)) > eps { return None; } From 2d8d98d096174c6ae269da349fcb86b8459205e8 Mon Sep 17 00:00:00 2001 From: zz_y Date: Fri, 25 Sep 2026 20:55:56 +0000 Subject: [PATCH 29/90] fix: rank mixed logical candidates using explicit candidate costs --- crates/asap-aware-mapping/src/replacement.rs | 54 ++++++++++++++++++-- 1 file changed, 51 insertions(+), 3 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index ffad44bb..8d105e32 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -4316,9 +4316,17 @@ fn rank_group<'a>( // purpose. `total_cmp` gives deterministic placement to a model's NaN // placeholders without dropping any candidate. ranked.sort_by(|a, b| { - cost_model - .estimate_cost(a, &target) - .total_cmp(&cost_model.estimate_cost(b, &target)) + match ( + cost_model.candidate_cost(a, &target), + cost_model.candidate_cost(b, &target), + ) { + (Some(a), Some(b)) => a.0.total_cmp(&b.0), + (Some(_), None) => std::cmp::Ordering::Less, + (None, Some(_)) => std::cmp::Ordering::Greater, + (None, None) => cost_model + .estimate_cost(a, &target) + .total_cmp(&cost_model.estimate_cost(b, &target)), + } }); ranked } @@ -8458,6 +8466,46 @@ mod tests { ); } + // Mixed candidate ranking must honor explicit costs, not legacy estimates. + #[test] + fn mixed_candidate_ranking_uses_explicit_candidate_costs() { + struct ExplicitCosts; + impl CostModel for ExplicitCosts { + fn rank_candidates( + &self, + _: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + fn candidate_cost( + &self, + candidate: &ReplacementSubDAG, + _: &TargetSubDAG<'_>, + ) -> Option { + Some(Cost( + if candidate.provenance == ReplacementProvenance::LogicalRewrite { + 1.0 + } else { + 100.0 + }, + )) + } + } + let root = Rc::new(lower_promql( + "sum by(job)(sum_over_time(a[1m]))", + AccuracyTarget::Exact, + )); + let space = search_workload(vec![("q", root)]); + let selection = space.global_selection(&ExplicitCosts); + let selected = selection + .for_target(&space.roots[0].1) + .unwrap() + .chosen + .unwrap(); + assert_eq!(selected.provenance, ReplacementProvenance::LogicalRewrite); + } + #[test] fn global_selection_compares_a_logical_rewrite_with_the_cse_choice() { struct PreferLogicalRewrite; From 4e833e7f590deb87153084b8f790a3e2b16e4dd0 Mon Sep 17 00:00:00 2001 From: zz_y Date: Fri, 25 Sep 2026 20:58:31 +0000 Subject: [PATCH 30/90] fix: expose executable grouped temporal accumulator candidates --- crates/asap-aware-mapping/src/replacement.rs | 33 ++++++++++++++++++++ crates/asap-aware-mapping/src/rewrite.rs | 2 +- 2 files changed, 34 insertions(+), 1 deletion(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 8d105e32..07bd8e9e 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -1286,6 +1286,22 @@ impl<'a> SketchAlgorithmStrategy<'a> { /// differs — see [`realize_child_with`]). fn propose_with(&self, root: &Rc, intent_override: Option<&AggIntent>) -> Proposals { let mut proposals = Proposals::default(); + // A selected logical rewrite otherwise remains KeepPreAsap during DAG + // assembly. Also expose its concrete summary realization for selection. + if intent_override.is_none() { + if let Some(rewritten) = crate::rewrite::composed_aggregate_rewrite(root) { + if let Ok(node) = realize_child_with(&rewritten, self.planning_inputs, None) { + if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) { + proposals.candidates.push(ReplacementSubDAG { + replacement: Replacement::Summary(node), + strategy: "SketchAlgorithmStrategy", + provenance: ReplacementProvenance::SummaryRealization, + rationale: "realize a schema-preserving composition of temporal and grouped accumulators".into(), + }); + } + } + } + } if let Ok(Some(node)) = exact_topk_over_temporal_values(root, self.planning_inputs) { proposals.candidates.push(ReplacementSubDAG { replacement: Replacement::Summary(node), @@ -8466,6 +8482,23 @@ mod tests { ); } + // Composable temporal/grouped Sum must be executable as one producer. + #[test] + fn grouped_temporal_sum_has_one_summary_producer_candidate() { + let root = Rc::new(lower_promql( + "sum by(job)(sum_over_time(a[1m]))", + AccuracyTarget::Exact, + )); + let candidates = + SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + assert!(candidates + .iter() + .any(|candidate| matches!(&candidate.replacement, + Replacement::Summary(node) if matches!(&node.expr, + SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } + if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))))); + } + // Mixed candidate ranking must honor explicit costs, not legacy estimates. #[test] fn mixed_candidate_ranking_uses_explicit_candidate_costs() { diff --git a/crates/asap-aware-mapping/src/rewrite.rs b/crates/asap-aware-mapping/src/rewrite.rs index 6081105e..44b685f8 100644 --- a/crates/asap-aware-mapping/src/rewrite.rs +++ b/crates/asap-aware-mapping/src/rewrite.rs @@ -247,7 +247,7 @@ fn build_rewrite(root: &Rc) -> Option> { /// Compose adjacent per-entity and cross-entity accumulators when their /// algebra, rather than a query-language spelling, proves equivalence. -fn composed_aggregate_rewrite(root: &Rc) -> Option> { +pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option> { let original_schema = root.output_schema().ok()?; let QueryExpr::Aggregate { reduction: outer_reduction @ Reduction::Reduce(_), From 70809561b20648b6090d6747201a1106fd919f7f Mon Sep 17 00:00:00 2001 From: zz_y Date: Fri, 25 Sep 2026 21:04:30 +0000 Subject: [PATCH 31/90] fix: preserve selected composed summaries during DAG assembly --- crates/asap-aware-mapping/src/replacement.rs | 40 +++++++++++++++++++- 1 file changed, 39 insertions(+), 1 deletion(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 07bd8e9e..23feb80c 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -4582,7 +4582,12 @@ impl<'a> GlobalSelection<'a> { if let Some(node) = self.assembled_nodes.borrow().get(&ptr) { return Ok(Rc::clone(node)); } - let node = if query_time_nested_sum(target) { + let selected_summary = self + .groups + .get(&ptr) + .and_then(|sel| sel.chosen) + .is_some_and(|candidate| matches!(candidate.replacement, Replacement::Summary(_))); + let node = if query_time_nested_sum(target) && !selected_summary { self.assemble_residual(target)? } else { match self @@ -8497,6 +8502,39 @@ mod tests { Replacement::Summary(node) if matches!(&node.expr, SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))))); + struct PreferComposed; + impl CostModel for PreferComposed { + fn rank_candidates( + &self, + _: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + fn candidate_cost( + &self, + candidate: &ReplacementSubDAG, + _: &TargetSubDAG<'_>, + ) -> Option { + Some(Cost( + if matches!(&candidate.replacement, + Replacement::Summary(node) if matches!(&node.expr, + SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } + if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))) + { + 1.0 + } else { + 100.0 + }, + )) + } + } + let space = search_workload(vec![("q", root.clone())]); + let selected = space.global_selection(&PreferComposed); + let node = selected.assemble_target(&space.roots[0].1).unwrap(); + assert!(matches!(&node.expr, + SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } + if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))); } // Mixed candidate ranking must honor explicit costs, not legacy estimates. From 206f93a8bdc8001601e814cd07d0c7de58501e8e Mon Sep 17 00:00:00 2001 From: zz_y Date: Fri, 25 Sep 2026 21:09:34 +0000 Subject: [PATCH 32/90] fix: keep nested aggregate dependencies explicit unless composition removes them --- crates/asap-aware-mapping/src/replacement.rs | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 23feb80c..58adc692 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -4582,12 +4582,15 @@ impl<'a> GlobalSelection<'a> { if let Some(node) = self.assembled_nodes.borrow().get(&ptr) { return Ok(Rc::clone(node)); } - let selected_summary = self + let selected_composed_summary = self .groups .get(&ptr) .and_then(|sel| sel.chosen) - .is_some_and(|candidate| matches!(candidate.replacement, Replacement::Summary(_))); - let node = if query_time_nested_sum(target) && !selected_summary { + .is_some_and(|candidate| matches!(&candidate.replacement, + Replacement::Summary(node) if matches!(&node.expr, + SummaryExpr::SummaryAgg { child, .. } + if matches!(&child.expr, SummaryExpr::KeepPreAsap(raw) if !contains_aggregate(raw))))); + let node = if query_time_nested_sum(target) && !selected_composed_summary { self.assemble_residual(target)? } else { match self From 1aab370729a60d15b3907f611025d0714f4409a1 Mon Sep 17 00:00:00 2001 From: zz_y Date: Fri, 25 Sep 2026 21:21:05 +0000 Subject: [PATCH 33/90] feat: honor summary candidate physical feasibility during selection --- crates/asap-aware-mapping/src/cost_model.rs | 7 +++ crates/asap-aware-mapping/src/replacement.rs | 52 +++++++++++++++----- 2 files changed, 48 insertions(+), 11 deletions(-) diff --git a/crates/asap-aware-mapping/src/cost_model.rs b/crates/asap-aware-mapping/src/cost_model.rs index 366f0289..b2c42953 100644 --- a/crates/asap-aware-mapping/src/cost_model.rs +++ b/crates/asap-aware-mapping/src/cost_model.rs @@ -876,6 +876,13 @@ pub trait CostModel { self.raw_query_recompute_cost(target) .map(|per_read| Cost(per_read.0 * expected_reads)) } + /// Physical feasibility evidence for a complete summary candidate. + /// `None` defers admission to physical/deployment compilation; `Some(false)` + /// excludes the candidate without changing its computation or parameters. + fn summary_support_evidence(&self, _summary: &SummaryNode) -> Option { + None + } + /// Which mixed exact/summary execution shapes the downstream runtime /// advertises (issue #171). Gates candidate *generation* in /// [`crate::exact_composition::ExactCompositionStrategy`]: a shape the diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 58adc692..5d710abd 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -520,15 +520,15 @@ impl ReplacementSubDAG { ) } - /// Runtime support for this candidate. Summary implementations remain - /// unknown until backend binding; a pure logical rewrite needs no new - /// physical operator. `Some(false)` disproves mixed-operation support. + /// Physical feasibility evidence for this candidate. A pure logical + /// rewrite needs no new operator. Unknown support is checked during + /// physical/deployment compilation; explicit rejection prevents selection. pub fn runtime_support_evidence(&self, cost_model: &dyn CostModel) -> Option { match &self.replacement { Replacement::ExactComposition(composition) => { cost_model.value_operation_support_evidence(&composition.op, composition.placement) } - Replacement::Summary(_) => None, + Replacement::Summary(node) => cost_model.summary_support_evidence(node), Replacement::Rewrite(_) => Some(true), } } @@ -4936,7 +4936,7 @@ fn composition_options<'a>( None => child_group.candidates.iter().collect(), }; for child_candidate in child_candidates { - if child_candidate.has_missing_accuracy_evidence() { + if !is_automatically_selectable(child_candidate, cost_model) { continue; } let Replacement::Summary(summary) = &child_candidate.replacement else { @@ -5113,7 +5113,7 @@ impl PlanSpace { .candidates .iter() .filter(|candidate| !is_composition_candidate(candidate)) - .filter(|candidate| is_automatically_selectable(candidate)) + .filter(|candidate| is_automatically_selectable(candidate, cost_model)) .filter_map(|candidate| { costs .get(&group.target, candidate) @@ -5139,7 +5139,7 @@ impl PlanSpace { .filter(|candidate| { !is_cse_candidate(candidate) && !is_composition_candidate(candidate) - && is_automatically_selectable(candidate) + && is_automatically_selectable(candidate, cost_model) }) .filter_map(|candidate| { cost_model @@ -5196,7 +5196,7 @@ impl PlanSpace { .filter(|candidate| { !is_cse_candidate(candidate) && !is_composition_candidate(candidate) - && is_automatically_selectable(candidate) + && is_automatically_selectable(candidate, cost_model) }) .filter_map(|candidate| { cost_model @@ -5241,7 +5241,7 @@ impl PlanSpace { // children (see `multiplier`'s `_ => effective` arm). None => rank_group(group, cost_model).into_iter().find(|candidate| { !is_composition_candidate(candidate) - && is_automatically_selectable(candidate) + && is_automatically_selectable(candidate, cost_model) && (cost_model .candidate_cost( candidate, @@ -5258,7 +5258,7 @@ impl PlanSpace { .find(|candidate| { !is_cse_candidate(candidate) && !is_composition_candidate(candidate) - && is_automatically_selectable(candidate) + && is_automatically_selectable(candidate, cost_model) && (cost_model .candidate_cost(candidate, &effective_target) .is_some() @@ -5344,8 +5344,9 @@ fn is_cse_candidate(candidate: &ReplacementSubDAG) -> bool { ) } -fn is_automatically_selectable(candidate: &ReplacementSubDAG) -> bool { +fn is_automatically_selectable(candidate: &ReplacementSubDAG, cost_model: &dyn CostModel) -> bool { !candidate.has_missing_accuracy_evidence() + && candidate.runtime_support_evidence(cost_model) != Some(false) } /// How much one direct reference to `parent_ptr` actually costs, once @@ -8490,6 +8491,35 @@ mod tests { ); } + // A cheap but physically infeasible candidate must not be selected. + #[test] + fn explicit_summary_infeasibility_prevents_selection() { + struct Unsupported; + impl CostModel for Unsupported { + fn rank_candidates( + &self, + _: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + fn candidate_cost(&self, _: &ReplacementSubDAG, _: &TargetSubDAG<'_>) -> Option { + Some(Cost(1.0)) + } + fn summary_support_evidence(&self, _: &SummaryNode) -> Option { + Some(false) + } + } + let root = Rc::new(lower_promql("sum_over_time(a[1m])", AccuracyTarget::Exact)); + let space = search_workload(vec![("q", root)]); + let selected = space.global_selection(&Unsupported); + assert!(selected + .for_target(&space.roots[0].1) + .unwrap() + .chosen + .is_none()); + } + // Composable temporal/grouped Sum must be executable as one producer. #[test] fn grouped_temporal_sum_has_one_summary_producer_candidate() { From 32facc975fc660cd7295c832e1d58ca87a31dec5 Mon Sep 17 00:00:00 2001 From: zz_y Date: Fri, 25 Sep 2026 21:32:21 +0000 Subject: [PATCH 34/90] fix: consider exact count candidates for approximate accuracy targets --- crates/asap-aware-mapping/src/replacement.rs | 24 +++++++++++++++++++- 1 file changed, 23 insertions(+), 1 deletion(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 5d710abd..745cc7f3 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -834,6 +834,11 @@ pub(crate) fn realizations_for_intent( )), ], AccuracyTarget::Exact => vec![exact_realization(intent)], + _ if matches!(intent, AggIntent::Count { .. }) => { + let mut candidates = sketch_realizations(intent, accuracy, cost_model); + candidates.push(exact_realization(intent)); + candidates + } _ => sketch_realizations(intent, accuracy, cost_model), }, @@ -6647,6 +6652,23 @@ mod tests { ); } + // Exact counting remains a legal candidate under an approximate target. + #[test] + fn approximate_count_includes_exact_accumulator_candidate() { + let intent = AggIntent::Count { + accuracy: eps(0.01), + }; + assert!(realizations_for_intent(&intent, &DefaultCostModel) + .iter() + .any(|candidate| matches!( + candidate, + Realization::ExactAggregate { + kind: ExactKind::Count, + .. + } + ))); + } + #[test] fn epsilon_delta_sizes_cms_depth() { let intent = AggIntent::Count { @@ -7432,7 +7454,7 @@ mod tests { assert_eq!(agg_group.consumer_count, 1); assert_eq!( agg_group.candidates.len(), - 5, + 6, "Hydra candidates with unknown evidence remain available: {:?}", agg_group.candidates ); From d333c1111aafd91d56a18b0e4203a22936275aba Mon Sep 17 00:00:00 2001 From: zz_y Date: Fri, 25 Sep 2026 21:35:45 +0000 Subject: [PATCH 35/90] feat: certify bounded mean and quantile ratio candidates --- .../src/accuracy/composition.rs | 4 +- crates/asap-aware-mapping/src/replacement.rs | 84 +++++++++++++++++-- 2 files changed, 77 insertions(+), 11 deletions(-) diff --git a/crates/asap-aware-mapping/src/accuracy/composition.rs b/crates/asap-aware-mapping/src/accuracy/composition.rs index a284ab32..2fa1a11b 100644 --- a/crates/asap-aware-mapping/src/accuracy/composition.rs +++ b/crates/asap-aware-mapping/src/accuracy/composition.rs @@ -276,10 +276,10 @@ impl DefaultAccuracyModel { if inputs.len() != 2 || inputs .iter() - .any(|input| input.metric != ErrorMetric::RelativeValue) + .any(|input| input.metric != ErrorMetric::RelativeValue && !input.is_exact()) { return Err(unsupported( - "division needs exactly two RelativeValue guarantees".into(), + "division needs two relative-value or exact guarantees".into(), )); } let Some(numerator) = inputs[0].bound.evaluate() else { diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 745cc7f3..bd693060 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -1842,11 +1842,30 @@ fn realize_binary( return Ok(None); } let domains = lhs_domain.zip(rhs_domain).map(|(lhs, rhs)| [lhs, rhs]); + let has_mean = [lhs, rhs] + .iter() + .any(|expr| matches!(bindable_intent(expr), Some(AggIntent::Avg { .. }))); + if has_mean + && domains.as_ref().is_none_or(|domains| { + domains.iter().any(|domain| { + !(domain.lower.abs().max(domain.upper.abs()) * domain.max_samples as f64) + .is_finite() + }) + }) + { + return Ok(None); + } lhs_node = realize_ddsketch_quantile_operand(lhs, planning_inputs, &target)?; rhs_node = realize_ddsketch_quantile_operand(rhs, planning_inputs, &target)?; if let Some(domains) = domains.as_ref() { for (domain, node) in domains.iter().zip([&lhs_node, &rhs_node]) { if !ddsketch_quantile_alpha(node) + .or_else(|| { + node.guarantee + .as_ref() + .is_some_and(ResultGuarantee::is_exact) + .then_some(alpha) + }) .is_some_and(|alpha| domain.supports_ddsketch(alpha)) { return Ok(None); @@ -1928,9 +1947,13 @@ fn realize_binary( let guarantee = if matches!(op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)) && direct_ddsketch_ratio && has_ratio_domains - && ddsketch_quantile_alpha(&lhs_node).is_some() - && ddsketch_quantile_alpha(&rhs_node).is_some() - { + && [&lhs_node, &rhs_node].iter().all(|node| { + ddsketch_quantile_alpha(node).is_some() + || node + .guarantee + .as_ref() + .is_some_and(ResultGuarantee::is_exact) + }) { [lhs_node.guarantee.clone(), rhs_node.guarantee.clone()] .into_iter() .collect::>>() @@ -2039,9 +2062,8 @@ fn is_promql_scalar(expr: &QueryExpr) -> bool { ) } -/// Both direct quantile operands inherit the workload target during PromQL -/// lowering. Reuse that one target for the ratio rather than interpreting it -/// as two independent error budgets. +/// Quantile operands inherit one workload target. A temporal mean is exact +/// on its checked finite domain and needs no approximation budget. fn shared_quantile_target(lhs: &QueryExpr, rhs: &QueryExpr) -> Option { let quantile_target = |expr: &QueryExpr| match bindable_intent(expr) { Some(AggIntent::Quantile { accuracy, q, .. }) @@ -2051,9 +2073,16 @@ fn shared_quantile_target(lhs: &QueryExpr, rhs: &QueryExpr) -> Option None, }; - let lhs = quantile_target(lhs)?; - let rhs = quantile_target(rhs)?; - (lhs == rhs).then_some(lhs) + match (quantile_target(lhs), quantile_target(rhs)) { + (Some(lhs), Some(rhs)) => (lhs == rhs).then_some(lhs), + (Some(target), None) if matches!(bindable_intent(rhs), Some(AggIntent::Avg { .. })) => { + Some(target) + } + (None, Some(target)) if matches!(bindable_intent(lhs), Some(AggIntent::Avg { .. })) => { + Some(target) + } + _ => None, + } } /// For `a / b`, two DDSketches with the same relative bound `alpha` produce @@ -6356,6 +6385,43 @@ mod tests { } } + // A bounded exact mean can share the relative division proof with a quantile. + #[test] + fn bounded_mean_quantile_ratio_is_certified() { + struct Domain; + impl AccuracyEvidenceProvider for Domain { + fn quantile_input_domain( + &self, + _: &QueryExpr, + ) -> Option { + Some(crate::accuracy::QuantileInputDomain { + lower: 1.0, + upper: 1000.0, + max_samples: 10000, + contract: "finite test population".into(), + }) + } + } + let target = AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }; + let inputs = CandidatePlanningInputs { + evidence: &Domain, + ..CandidatePlanningInputs::with_default_accuracy(&DefaultCostModel) + }; + for query in [ + "avg_over_time(a[5m]) / quantile_over_time(0.5,a[5m])", + "quantile_over_time(0.5,a[5m]) / avg_over_time(a[5m])", + ] { + let root = Rc::new(lower_promql(query, target.clone())); + let node = realize_binary(&root, inputs, Some(&target)) + .unwrap() + .expect("bounded ratio candidate"); + assert!(DefaultAccuracyModel.satisfies(node.guarantee.as_ref().unwrap(), &target)); + } + } + // Missing domain proof permits an uncertified direct quantile ratio only. #[test] fn quantile_ratio_without_input_proof_has_no_root_guarantee() { From 78e5a81cf247f2b54d8af6dd55f19a5c300709c6 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 03:36:45 +0000 Subject: [PATCH 36/90] feat: compile and select physical precompute frontier candidates --- .../src/physical_planner/candidates.rs | 142 +++++++++ .../src/physical_planner/compiled.rs | 21 ++ .../src/physical_planner/mod.rs | 6 + .../tests/precompute_candidates.rs | 300 ++++++++++++++++++ .../physical-planning-and-deployment.md | 44 +++ 5 files changed, 513 insertions(+) create mode 100644 crates/asap-physical-operators/src/physical_planner/candidates.rs create mode 100644 crates/asap-physical-operators/tests/precompute_candidates.rs diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs new file mode 100644 index 00000000..b6101479 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -0,0 +1,142 @@ +//! Compile maintenance-selected frontiers without deployment-specific graph rewrites. +use super::*; + +/// One computation realization; lifecycle/window/revision requirements accompany +/// it during optimization and deployment. Stored outputs have no storage identity. +#[derive(Clone)] +pub struct PhysicalCandidate { + pub precompute: Option, + pub query: CompiledPhysicalDag, + pub materialized_outputs: BTreeMap, +} + +/// Compile an explicit materialization frontier selected by Planner maintenance +/// search. Operators upstream of that frontier run in precompute, including +/// readouts/reductions; query execution receives their typed output values. +/// Empty frontiers retain the full computation in the query DAG. +/// +/// Repeated windows must be instantiated with the same evaluation/population +/// contract used to build each output. This API never treats a result from a +/// different window or revision as interchangeable merely because types match. +pub fn compile_candidate( + dag: &ExecutableDag, + inputs: BTreeMap, + roots: &[NodeId], + frontier: &[NodeId], +) -> Result { + if frontier.is_empty() { + return Ok(PhysicalCandidate { + precompute: None, + query: compile(dag, inputs, roots)?, + materialized_outputs: BTreeMap::new(), + }); + } + let frontier_set: BTreeSet<_> = frontier.iter().copied().collect(); + if frontier_set.len() != frontier.len() || frontier.iter().any(|id| inputs.contains_key(id)) { + return Err(invalid("frontier must contain distinct computed outputs")); + } + let full = compile(dag, inputs.clone(), roots)?; + let precompute = compile(dag, inputs.clone(), frontier)?; + let mut materialized_outputs = BTreeMap::new(); + for &id in frontier { + // Also proves that the frontier is reachable from the requested roots. + full.output_contract(id)?; + let mut output = precompute.output_contract(id)?; + if output.properties.boundedness != Boundedness::Bounded { + return Err(invalid("materialized output requires bounded execution")); + } + // A stored reader may stream batches even when the producer blocked. + // Its timing is independent; the retained result still must be finite. + output.properties.emission = Emission::Unknown; + materialized_outputs.insert(id, output); + } + let mut query_inputs = inputs; + query_inputs.extend(materialized_outputs.clone()); + let query = compile(dag, query_inputs, roots)?; + let used: BTreeSet<_> = query.input_contracts().map(|(id, _)| id).collect(); + if !frontier.iter().all(|id| used.contains(id)) { + return Err(invalid( + "frontier contains an output shadowed by another boundary", + )); + } + Ok(PhysicalCandidate { + precompute: Some(precompute), + query, + materialized_outputs, + }) +} + +/// Lower every maintenance candidate before feasibility/cost evaluation. Keep +/// individual failures visible; do not substitute another computation on error. +pub fn compile_candidates( + dag: &ExecutableDag, + inputs: BTreeMap, + roots: &[NodeId], + frontiers: &[Vec], +) -> Vec> { + frontiers + .iter() + .map(|frontier| compile_candidate(dag, inputs.clone(), roots, frontier)) + .collect() +} + +/// Complete workload cost supplied by scoped optimizer/deployment evidence. +/// The evaluator includes build/update work, retained state, shared producers +/// and recurrent reads over the same horizon; these are not per-query timings. +#[derive(Clone, Debug)] +pub struct CandidateCost { + pub workload_scope: String, + pub horizon_seconds: f64, + pub total_cost: f64, +} + +pub struct CandidateSelection { + pub candidate: PhysicalCandidate, + pub candidate_index: usize, + pub cost: CandidateCost, +} + +/// Select only compiled and deployment-feasible physical candidates. `None` +/// rejects an unbindable candidate before pricing. Comparable scoped costs are +/// required; deployment never rewrites the selected frontier after this step. +pub fn select_candidate( + candidates: Vec>, + mut evaluate: impl FnMut(&PhysicalCandidate) -> Result, Error>, +) -> Result { + let mut scope: Option<(String, f64)> = None; + let mut selected: Option = None; + for (candidate_index, candidate) in candidates.into_iter().enumerate() { + let Ok(candidate) = candidate else { continue }; + let Some(cost) = evaluate(&candidate)? else { + continue; + }; + if cost.workload_scope.is_empty() + || !cost.horizon_seconds.is_finite() + || cost.horizon_seconds <= 0. + || !cost.total_cost.is_finite() + || cost.total_cost < 0. + { + return Err(invalid( + "candidate cost lacks a valid workload scope/horizon", + )); + } + let current_scope = (cost.workload_scope.clone(), cost.horizon_seconds); + if scope.as_ref().is_some_and(|scope| scope != ¤t_scope) { + return Err(invalid( + "candidate costs describe different workloads or horizons", + )); + } + scope = Some(current_scope); + if selected + .as_ref() + .is_none_or(|selected| cost.total_cost < selected.cost.total_cost) + { + selected = Some(CandidateSelection { + candidate, + candidate_index, + cost, + }); + } + } + selected.ok_or_else(|| invalid("no feasible priced physical candidate")) +} diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs index bf9f7fda..0e2e1b61 100644 --- a/crates/asap-physical-operators/src/physical_planner/compiled.rs +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -89,6 +89,27 @@ impl CompiledPhysicalDag { Node::Operator { .. } => None, }) } + /// Derive a reachable output contract without opening deployment readers. + pub fn output_contract(&self, id: NodeId) -> Result { + let sources = self + .input_contracts() + .map(|(id, contract)| (id, Box::new(contract.clone()) as Source<'_>)) + .collect(); + let graph = self.instantiate(sources)?; + let properties = graph.properties(&self.roots)?; + let properties = *properties + .get(&id) + .ok_or_else(|| invalid("output is not reachable"))?; + let schema = match self + .nodes + .get(&id) + .ok_or_else(|| invalid("missing output"))? + { + Node::Input(contract) => contract.schema.clone(), + Node::Operator { operator, .. } => operator.output_schema(), + }; + Ok(InputContract { schema, properties }) + } /// Validate using contract-only sources. No deployment reader is available. pub fn validate(&self) -> Result<(), Error> { let sources = self diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 0596a8a3..70483c56 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -28,6 +28,12 @@ fn invalid(message: impl Into) -> Error { /// A deployment must authorize these frontiers before calling this function. pub type Source<'a> = Box + 'a>; +mod candidates; +pub use candidates::{ + compile_candidate, compile_candidates, select_candidate, CandidateCost, CandidateSelection, + PhysicalCandidate, +}; + mod compiled; pub use compiled::{CompiledPhysicalDag, InputContract}; diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs new file mode 100644 index 00000000..0193f23c --- /dev/null +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -0,0 +1,300 @@ +//! Materialized frontiers are compiled by Planner, never rewritten by deployment. +use asap_aware_mapping::{cost_model::DefaultCostModel, search_workload}; +use asap_physical_operators::{ + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{ + compile_candidates, select_candidate, CandidateCost, CompiledPhysicalDag, InputContract, + Source, + }, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{post_asap::*, pre_asap::DataType, types::AccuracyTarget, workload::*}; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; + +fn grouped_rate() -> ExecutableDag { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("sum by(job)(rate(m[1m]))".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let root = Rc::new( + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0), + ); + let space = search_workload(vec![("grouped-rate", root)]); + let selected = space + .global_selection(&DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + compile_executable_dag(&selected).unwrap() +} +fn run(plan: &CompiledPhysicalDag, inputs: BTreeMap, scope: Scope) -> Vec { + let sources = inputs + .into_iter() + .map(|(id, batch)| { + let source = Operator::source(batch.schema().clone(), vec![batch]).unwrap(); + (id, Box::new(source) as Source<'_>) + }) + .collect(); + let dag = plan.instantiate(sources).unwrap(); + block_on(async { + let context = RunContext::new(scope, Limits::default()).unwrap(); + let mut output = dag.execute(plan.roots(), context).unwrap().remove(0); + let mut batches = vec![]; + while let Some(batch) = output.next().await { + batches.push((*batch.unwrap()).clone()); + } + batches + }) +} + +/// Rate readouts and grouped Sum can run together during bounded precompute; +/// storing per-series rates instead leaves the same Sum in the query DAG. +#[test] +fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { + let dag = grouped_rate(); + let state = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + ExecutableOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) + .unwrap(); + let readout = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + ExecutableOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator + } + ) + }) + .unwrap(); + let input_schema = Arc::new(state.output_schema.clone()); + let (family, update, grouping) = match &state.payload { + ExecutableOperatorPayload::SummaryAgg { + family, + input, + grouping, + .. + } => (family, input, grouping), + _ => unreachable!(), + }; + let mut expected_rate_sum = 0.; + let rows = [[100., 0., 100.], [100., 200., 0.]] + .into_iter() + .enumerate() + .map(|(index, values)| { + let mut accumulator = create_planner_accumulator(family, update, grouping).unwrap(); + for (i, value) in values.into_iter().enumerate() { + accumulator.update_single(value, i as i64 * 1000); + } + let state = accumulator.into_accumulator(); + expected_rate_sum += state + .query_statistic( + asap_physical_operators::Statistic::Rate, + &None, + &Default::default(), + ) + .unwrap(); + let summary = Value::Summary { + family: family.clone(), + state: Arc::from(state), + }; + input_schema + .fields + .iter() + .map(|field| match &field.dtype { + SummaryFamilyType::ExactAggregate(..) => summary.clone(), + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(2000), + SummaryFamilyType::Plain(DataType::Utf8) => { + Value::Utf8(if field.name == "job" { + "api".into() + } else { + format!("series-{index}").into() + }) + } + _ => panic!("unexpected input field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(input_schema.clone(), rows).unwrap(); + let root = u64::from(dag.root.0); + let state_id = u64::from(state.id.0); + let rate_id = u64::from(readout.id.0); + let candidates = compile_candidates( + &dag, + BTreeMap::from([(state_id, InputContract::bounded(input_schema))]), + &[root], + &[vec![], vec![rate_id], vec![root]], + ); + // Scoped cost fixtures select either precompute boundary. No readers are + // opened during candidate construction or selection. + for prefer_grouped in [false, true] { + let inventory = compile_candidates( + &dag, + BTreeMap::from([( + state_id, + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]), + &[root], + &[vec![rate_id], vec![root]], + ); + let selected = select_candidate(inventory, |candidate| { + let grouped = candidate.materialized_outputs.contains_key(&root); + Ok(Some(CandidateCost { + workload_scope: "reset-counter-workload".into(), + horizon_seconds: 300., + total_cost: if grouped == prefer_grouped { 1. } else { 100. }, + })) + }) + .unwrap(); + assert_eq!( + selected.candidate.materialized_outputs.contains_key(&root), + prefer_grouped + ); + assert_eq!(selected.cost.total_cost, 1.); + } + let contracts = BTreeMap::from([( + state_id, + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + for frontier in [vec![rate_id, rate_id], vec![root, rate_id], vec![999]] { + assert!( + asap_physical_operators::physical_planner::compile_candidate( + &dag, + contracts.clone(), + &[root], + &frontier + ) + .is_err() + ); + } + let inventory = compile_candidates( + &dag, + contracts.clone(), + &[root], + &[vec![rate_id], vec![root]], + ); + let selected = select_candidate(inventory, |candidate| { + if candidate.materialized_outputs.contains_key(&root) { + return Ok(None); + } + Ok(Some(CandidateCost { + workload_scope: "same-workload".into(), + horizon_seconds: 300., + total_cost: 100., + })) + }) + .unwrap(); + assert!(selected + .candidate + .materialized_outputs + .contains_key(&rate_id)); + let inventory = compile_candidates(&dag, contracts, &[root], &[vec![rate_id], vec![root]]); + assert!( + select_candidate(inventory, |candidate| Ok(Some(CandidateCost { + workload_scope: "same-workload".into(), + horizon_seconds: if candidate.materialized_outputs.contains_key(&root) { + 60. + } else { + 300. + }, + total_cost: 1., + }))) + .is_err() + ); + let query_scope = Scope::Query { + evaluation_time_ms: 2000, + revision: 1, + }; + let maintenance_scope = Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }; + let mut results = vec![]; + for candidate in candidates { + let candidate = candidate.unwrap(); + let inputs = if let Some(precompute) = &candidate.precompute { + let stored = run( + precompute, + BTreeMap::from([(state_id, batch.clone())]), + maintenance_scope.clone(), + ); + assert_eq!(stored.len(), 1); + let boundary = precompute.roots()[0]; + assert_eq!( + candidate.materialized_outputs[&boundary].schema, + *stored[0].schema() + ); + BTreeMap::from([(boundary, stored[0].clone())]) + } else { + BTreeMap::from([(state_id, batch.clone())]) + }; + let output = run(&candidate.query, inputs, query_scope.clone()); + assert_eq!(output.len(), 1); + assert_eq!(output[0].rows().len(), 1); + assert!(matches!(&output[0].rows()[0][0], Value::Utf8(job) if job.as_ref() == "api")); + assert!( + matches!(output[0].rows()[0][1], Value::Float64(value) if value == expected_rate_sum) + ); + results.push( + output[0].rows()[0] + .iter() + .map(|value| value.key().unwrap()) + .collect::>(), + ); + } + assert_eq!(results[0], results[1]); + assert_eq!(results[1], results[2]); + let mut wrong_order = create_planner_accumulator(family, update, grouping).unwrap(); + for (i, value) in [200., 200., 100.].into_iter().enumerate() { + wrong_order.update_single(value, i as i64 * 1000); + } + let rate_of_sum = wrong_order + .into_accumulator() + .query_statistic( + asap_physical_operators::Statistic::Rate, + &None, + &Default::default(), + ) + .unwrap(); + assert_ne!( + expected_rate_sum, rate_of_sum, + "counter resets prohibit moving Sum before Rate" + ); +} diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 675ea6f0..062e681b 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -228,6 +228,50 @@ s3://.../latency-kll/12:01 is deployment-specific. Placement and scheduling also remain outside the Physical DAG. If the required behavior cannot be realized, physical compilation fails. +### Physical candidates include precompute computation + +Materialization frontiers are Planner decisions. A candidate records both the +precompute Physical DAG and the query Physical DAG, with typed outputs connecting +them. The deployment compiler binds those outputs; it does not move operators. + +For `sum by(job)(rate(m[1m]))`, legal physical candidates can include: + +```text +Candidate A: + precompute: compatible per-series counter states → per-series Rate + materialized output: per-series rate values for window/evaluation/revision + query: stored per-series rate values → grouped Sum + +Candidate B: + precompute: compatible per-series counter states → per-series Rate → grouped Sum + materialized output: grouped values for window/evaluation/revision + query: stored grouped values → result +``` + +Both preserve reset-aware Rate before Sum. Summing raw counters before Rate is +not equivalent. The counter-state build may be another maintenance DAG; typed +state inputs do not imply that a deployment can construct or bind those states. + +The shared library exposes `physical_planner::compile_candidates(...)` to lower +explicit frontier candidates to `PhysicalCandidate { precompute, query, +materialized_outputs }`. `select_candidate(...)` accepts deployment feasibility +and scoped complete-workload costs and chooses the lowest-cost feasible +candidate. Costs must describe the same workload and planning horizon; missing +feasibility is rejected before pricing. The optimizer supplies candidate +frontiers and cost evidence, including updates, retention, recurrence and sharing. +This interface does not yet enumerate every possible frontier automatically. + +Physical compilation opens no readers. Bounded precompute outputs become typed +query inputs. Their build window, evaluation time, population, readiness and +revision contracts must accompany the selected lifecycle and be checked during +deployment binding. Type compatibility alone does not establish reuse legality. + +The Planner integration test executes both candidates through the shared runtime +and reverses the selected frontier with two controlled cost fixtures. It also +rejects shadowed/duplicate boundaries and incomparable planning horizons. This +establishes Planner capability; it does not establish that ASAPQuery currently +supports persisting every scalar/result-output frontier. + ## 4. Physical DAG → Deployment Plan / DAG The **Deployment Plan Compiler** binds the Physical DAGs to the concrete deployment: From 9bed5c6b1d09ee4a94b13303b38e7d6eb7c28a5e Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 03:57:42 +0000 Subject: [PATCH 37/90] feat: enumerate bounded physical materialization frontiers --- .../src/physical_planner/candidates.rs | 72 +++++++++++++++++++ .../src/physical_planner/mod.rs | 4 +- .../tests/precompute_candidates.rs | 20 ++++++ .../physical-planning-and-deployment.md | 2 +- 4 files changed, 95 insertions(+), 3 deletions(-) diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs index b6101479..65617ed8 100644 --- a/crates/asap-physical-operators/src/physical_planner/candidates.rs +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -66,6 +66,78 @@ pub fn compile_candidate( }) } +/// Enumerate bounded, reachable materialization frontiers above explicit inputs. +/// Each frontier is an antichain: storing an output and its ancestor together +/// would leave the ancestor unused by query execution. Lifecycle eligibility +/// and deployment feasibility are evaluated separately before cost selection. +/// Exceeding the search budget returns an error, never a partial inventory. +pub fn enumerate_frontiers( + dag: &ExecutableDag, + inputs: &BTreeMap, + roots: &[NodeId], + max_candidates: usize, +) -> Result>, Error> { + if max_candidates == 0 { + return Err(invalid( + "frontier search requires a positive candidate budget", + )); + } + let compiled = compile(dag, inputs.clone(), roots)?; + let mut ancestors = BTreeMap::>::new(); + let mut eligible = Vec::new(); + for node in &dag.nodes { + let id = u64::from(node.id.0); + if inputs.contains_key(&id) { + continue; + } + let Ok(contract) = compiled.output_contract(id) else { + continue; + }; + if contract.properties.boundedness != Boundedness::Bounded { + continue; + } + let mut seen = BTreeSet::new(); + let mut pending = vec![id]; + while let Some(current) = pending.pop() { + if !seen.insert(current) || inputs.contains_key(¤t) { + continue; + } + pending.extend( + dag.edges + .iter() + .filter(|edge| u64::from(edge.consumer.0) == current) + .map(|edge| u64::from(edge.producer.0)), + ); + } + ancestors.insert(id, seen); + eligible.push(id); + } + eligible.sort_unstable(); + let mut frontiers = vec![vec![]]; + for id in eligible { + let additions = frontiers + .iter() + .filter(|frontier| { + frontier.iter().all(|previous| { + !ancestors[&id].contains(previous) && !ancestors[previous].contains(&id) + }) + }) + .map(|frontier| { + let mut next = frontier.clone(); + next.push(id); + next + }) + .collect::>(); + if additions.len() > max_candidates.saturating_sub(frontiers.len()) { + return Err(invalid( + "materialization frontier search exceeds candidate budget", + )); + } + frontiers.extend(additions); + } + Ok(frontiers) +} + /// Lower every maintenance candidate before feasibility/cost evaluation. Keep /// individual failures visible; do not substitute another computation on error. pub fn compile_candidates( diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 70483c56..2b54eaf8 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -30,8 +30,8 @@ pub type Source<'a> = Box + 'a>; mod candidates; pub use candidates::{ - compile_candidate, compile_candidates, select_candidate, CandidateCost, CandidateSelection, - PhysicalCandidate, + compile_candidate, compile_candidates, enumerate_frontiers, select_candidate, CandidateCost, + CandidateSelection, PhysicalCandidate, }; mod compiled; diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index 0193f23c..45603015 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -155,6 +155,26 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { let root = u64::from(dag.root.0); let state_id = u64::from(state.id.0); let rate_id = u64::from(readout.id.0); + let frontiers = asap_physical_operators::physical_planner::enumerate_frontiers( + &dag, + &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), + &[root], + 128, + ) + .unwrap(); + assert!(frontiers.contains(&vec![])); + assert!(frontiers.contains(&vec![rate_id])); + assert!(frontiers.contains(&vec![root])); + assert!(!frontiers.contains(&vec![root, rate_id])); + assert!( + asap_physical_operators::physical_planner::enumerate_frontiers( + &dag, + &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), + &[root], + 1, + ) + .is_err() + ); let candidates = compile_candidates( &dag, BTreeMap::from([(state_id, InputContract::bounded(input_schema))]), diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 062e681b..8a5212b2 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -259,7 +259,7 @@ and scoped complete-workload costs and chooses the lowest-cost feasible candidate. Costs must describe the same workload and planning horizon; missing feasibility is rejected before pricing. The optimizer supplies candidate frontiers and cost evidence, including updates, retention, recurrence and sharing. -This interface does not yet enumerate every possible frontier automatically. +`enumerate_frontiers` constructs bounded, reachable antichain frontiers above explicit input boundaries, including query-only and fully precomputed results. It fails explicitly when the candidate budget is exceeded. Maintenance selection must still reject frontiers that violate window, freshness, or reuse requirements; deployment feasibility is checked before pricing. Physical compilation opens no readers. Bounded precompute outputs become typed query inputs. Their build window, evaluation time, population, readiness and From 874ae1c6081dde89a46dca4dbc7ebcb4f2f627f1 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 04:05:07 +0000 Subject: [PATCH 38/90] fix: admit exact temporal ranking for approximate requests --- crates/asap-aware-mapping/src/replacement.rs | 25 ++++++++++++++++---- 1 file changed, 20 insertions(+), 5 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index bd693060..8ebe08aa 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -1659,11 +1659,7 @@ fn exact_topk_over_temporal_values( else { return Ok(None); }; - let [AggIntent::TopK { - k, - accuracy: AccuracyTarget::Exact, - }] = measures.as_slice() - else { + let [AggIntent::TopK { k, .. }] = measures.as_slice() else { return Ok(None); }; let QueryExpr::Aggregate { @@ -6332,6 +6328,25 @@ mod tests { ); } + // Approximate requests also admit exact temporal ranking candidates. + #[test] + fn approximate_temporal_topk_admits_exact_maintained_values() { + let root = Rc::new(lower_promql( + "topk by(job)(1,count_over_time(a[5m]))", + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + )); + let planning_inputs = + CandidatePlanningInputs::with_default_accuracy(&crate::cost_model::DefaultCostModel); + let node = exact_topk_over_temporal_values(&root, planning_inputs) + .unwrap() + .expect("exact ranking is legal for an approximate request"); + assert!(node.guarantee.as_ref().unwrap().is_exact()); + asap_types::post_asap::compile_executable_dag(&node).unwrap(); + } + // Exact Top-K consumes the Planner's maintained temporal values. #[test] fn exact_temporal_topk_has_a_maintained_value_candidate() { From b5beb532e0c742746bcc8cb8ad61a5b787d06dd7 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 04:10:12 +0000 Subject: [PATCH 39/90] refactor: retain deployment metadata in Planner candidate selection --- .../src/physical_planner/candidates.rs | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs index 65617ed8..a84c3394 100644 --- a/crates/asap-physical-operators/src/physical_planner/candidates.rs +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -162,8 +162,8 @@ pub struct CandidateCost { pub total_cost: f64, } -pub struct CandidateSelection { - pub candidate: PhysicalCandidate, +pub struct CandidateSelection { + pub candidate: T, pub candidate_index: usize, pub cost: CandidateCost, } @@ -171,12 +171,14 @@ pub struct CandidateSelection { /// Select only compiled and deployment-feasible physical candidates. `None` /// rejects an unbindable candidate before pricing. Comparable scoped costs are /// required; deployment never rewrites the selected frontier after this step. -pub fn select_candidate( - candidates: Vec>, - mut evaluate: impl FnMut(&PhysicalCandidate) -> Result, Error>, -) -> Result { +/// The payload is generic so deployments can retain binding/diagnostic metadata +/// alongside each compiled computation without duplicating winner selection. +pub fn select_candidate( + candidates: Vec>, + mut evaluate: impl FnMut(&T) -> Result, Error>, +) -> Result, Error> { let mut scope: Option<(String, f64)> = None; - let mut selected: Option = None; + let mut selected: Option> = None; for (candidate_index, candidate) in candidates.into_iter().enumerate() { let Ok(candidate) = candidate else { continue }; let Some(cost) = evaluate(&candidate)? else { From 13a4e2809fe00d11e8f4d94ee17ecf34e5b831cc Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 04:21:51 +0000 Subject: [PATCH 40/90] fix: omit sparse counter series in shared physical readouts --- .../src/operators/summary/mod.rs | 2 + .../src/stored_state/readout.rs | 88 ++++++++++++++++++- 2 files changed, 89 insertions(+), 1 deletion(-) diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs index f469f0a1..86415a5c 100644 --- a/crates/asap-physical-operators/src/operators/summary/mod.rs +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -261,6 +261,8 @@ pub(super) fn execute<'a>( .map(move |batch| { let batch = batch?; let mut rows = batch.rows().to_vec(); + rows.retain(|row| !matches!(&row[*state], Value::Summary { state: summary, .. } + if crate::stored_state::readout::insufficient_counter_samples(summary.as_ref(), *statistic))); for row in &mut rows { let Value::Summary { state: summary, .. } = &row[*state] else { return Err(invalid("summary value required")); diff --git a/crates/asap-physical-operators/src/stored_state/readout.rs b/crates/asap-physical-operators/src/stored_state/readout.rs index 6d99fce4..02a04572 100644 --- a/crates/asap-physical-operators/src/stored_state/readout.rs +++ b/crates/asap-physical-operators/src/stored_state/readout.rs @@ -98,6 +98,15 @@ pub fn exact_readout( key: &Option, parameters: &std::collections::HashMap, ) -> Result { + let merged = merge_exact_states(states)?; + merged + .query_statistic(statistic, key, parameters) + .map_err(|e| e.to_string()) +} + +fn merge_exact_states( + states: impl IntoIterator>, +) -> Result, String> { let mut states = states.into_iter(); let mut merged = states .next() @@ -108,7 +117,84 @@ pub fn exact_readout( .merge_with(state.as_ref()) .map_err(|e| e.to_string())?; } + Ok(merged) +} + +/// PromQL counter readouts omit a series with fewer than two samples. Other +/// state/type/range failures remain errors rather than empty results. +pub fn insufficient_counter_samples( + state: &dyn crate::AggregateCore, + statistic: crate::Statistic, +) -> bool { + matches!( + statistic, + crate::Statistic::Rate | crate::Statistic::Increase + ) && state + .as_any() + .downcast_ref::() + .is_some_and(|state| state.sample_count < 2) +} + +pub fn exact_readout_optional( + states: impl IntoIterator>, + statistic: crate::Statistic, + key: &Option, + parameters: &std::collections::HashMap, +) -> Result, String> { + let merged = merge_exact_states(states)?; + if insufficient_counter_samples(merged.as_ref(), statistic) { + return Ok(None); + } merged .query_statistic(statistic, key, parameters) - .map_err(|e| e.to_string()) + .map(Some) + .map_err(|error| error.to_string()) +} + +#[cfg(test)] +mod counter_tests { + use super::*; + use crate::{summary_kernels::IncreaseAccumulator, AggregateCore, Measurement, Statistic}; + use std::sync::Arc; + + #[test] + fn sparse_counter_is_absent_but_invalid_ranges_still_fail() { + let mut state = + IncreaseAccumulator::new(Measurement::new(10.), 10_000, Measurement::new(10.), 10_000); + let parameters = std::collections::HashMap::from([ + ("range_start_ms".into(), "0".into()), + ("range_end_ms".into(), "60000".into()), + ]); + assert_eq!( + exact_readout_optional( + [Arc::new(state.clone()) as Arc], + Statistic::Rate, + &None, + ¶meters + ) + .unwrap(), + None + ); + state.update(Measurement::new(20.), 20_000); + assert!(exact_readout_optional( + [Arc::new(state.clone()) as Arc], + Statistic::Rate, + &None, + ¶meters + ) + .unwrap() + .is_some()); + let invalid = std::collections::HashMap::from([ + ("range_start_ms".into(), "60000".into()), + ("range_end_ms".into(), "0".into()), + ]); + assert!(exact_readout_optional( + [Arc::new(state) as Arc], + Statistic::Rate, + &None, + &invalid + ) + .is_err()); + assert!(exact_readout_optional([], Statistic::Rate, &None, ¶meters).is_err()); + } } From ae8817a151d14c00bff023057eaff0ac23adfa60 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 04:41:04 +0000 Subject: [PATCH 41/90] Preserve logical counter windows across physical candidates --- .../src/operators/mod.rs | 58 +++++++++++++++++++ .../src/operators/summary/mod.rs | 5 +- .../src/physical_planner/mod.rs | 29 +++++++++- .../asap-physical-operators/src/plan/mod.rs | 4 ++ .../src/runtime/mod.rs | 9 +++ .../src/stored_state/readout.rs | 16 ++++- .../tests/precompute_candidates.rs | 24 +++++++- 7 files changed, 138 insertions(+), 7 deletions(-) diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index 4fee793e..ecc30279 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -95,6 +95,61 @@ pub struct Operator { output: Schema, } impl Operator { + pub(crate) fn is_counter_readout(&self) -> bool { + matches!( + self.kind, + Kind::Readout { + statistic: crate::Statistic::Rate | crate::Statistic::Increase, + .. + } + ) + } + pub(crate) fn with_counter_lookback(mut self, lookback: i64) -> Result { + if lookback <= 0 { + return Err(invalid("counter lookback must be positive")); + } + if let Kind::Readout { parameters, .. } = &mut self.kind { + parameters.insert("logical_lookback_ms".into(), lookback.to_string()); + } + Ok(self) + } + pub(super) fn readout_parameters( + &self, + context: &RunContext, + ) -> Result, Error> { + let Kind::Readout { parameters, .. } = &self.kind else { + return Ok(Default::default()); + }; + let mut parameters = parameters.clone(); + if let Some(lookback) = parameters.remove("logical_lookback_ms") { + let lookback: i64 = lookback + .parse() + .map_err(|_| invalid("invalid counter lookback"))?; + let end = match context.scope { + crate::runtime::Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + crate::runtime::Scope::Ingestion { window_end_ms, .. } => window_end_ms, + }; + let start = end + .checked_sub(lookback) + .ok_or_else(|| invalid("counter window overflows Int64"))?; + if let crate::runtime::Scope::Ingestion { + window_start_ms, .. + } = context.scope + { + if window_start_ms != start { + return Err(invalid( + "maintenance window differs from logical counter window", + )); + } + } + parameters.insert("range_start_ms".into(), start.to_string()); + parameters.insert("range_end_ms".into(), end.to_string()); + } + Ok(parameters) + } + pub(crate) fn with_output_schema(mut self, output: Schema) -> Result { if self.output.fields.len() != output.fields.len() || self @@ -171,6 +226,9 @@ impl PhysicalOperator for Operator { Kind::Readout { .. } => "SummaryReadout", } } + fn validate_context(&self, context: &RunContext) -> Result<(), Error> { + self.readout_parameters(context).map(|_| ()) + } fn input_schemas(&self) -> Vec { self.inputs.clone() } diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs index 86415a5c..70fbf79a 100644 --- a/crates/asap-physical-operators/src/operators/summary/mod.rs +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -205,6 +205,7 @@ pub(super) fn execute<'a>( mut inputs: Vec>, context: RunContext, ) -> Result, Error> { + let parameters = operator.readout_parameters(&context)?; let output = operator.output.clone(); let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; match &operator.kind { @@ -256,7 +257,7 @@ pub(super) fn execute<'a>( Kind::Readout { state, statistic, - parameters, + .. } => Ok(input .map(move |batch| { let batch = batch?; @@ -286,7 +287,7 @@ pub(super) fn execute<'a>( } else { Value::Float64( summary - .query_statistic(*statistic, &None, parameters) + .query_statistic(*statistic, &None, ¶meters) .map_err(|e| Error::Operator(e.to_string()))?, ) }; diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 2b54eaf8..d7474271 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -182,8 +182,35 @@ fn compile_internal( auxiliary -= 1; schemas.truncate(1); } - let operator = compile_node(node, &schemas) + let mut operator = compile_node(node, &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; + if operator.is_counter_readout() { + let mut pending = vec![id]; + let mut visited = BTreeSet::new(); + let mut ranges = BTreeSet::new(); + while let Some(ancestor) = pending.pop() { + if !visited.insert(ancestor) { + continue; + } + if let Payload::Fallback { + expression: QueryExpr::TimeRange { range, .. }, + } = &nodes[&ancestor].payload + { + ranges.insert( + i64::try_from(range.as_millis()) + .map_err(|_| invalid("counter lookback exceeds Int64"))?, + ); + continue; + } + pending.extend(dependencies.get(&ancestor).into_iter().flatten().copied()); + } + if ranges.len() > 1 { + return Err(invalid("counter readout has ambiguous logical windows")); + } + if let Some(lookback) = ranges.into_iter().next() { + operator = operator.with_counter_lookback(lookback)?; + } + } graph.add(id, inputs, operator)?; } } diff --git a/crates/asap-physical-operators/src/plan/mod.rs b/crates/asap-physical-operators/src/plan/mod.rs index ffc6ebd9..0dce4dd8 100644 --- a/crates/asap-physical-operators/src/plan/mod.rs +++ b/crates/asap-physical-operators/src/plan/mod.rs @@ -25,6 +25,10 @@ pub trait PhysicalOperator { false } + /// Validate run-specific contracts before any source is opened. + fn validate_context(&self, _context: &RunContext) -> Result<(), Error> { + Ok(()) + } fn input_schemas(&self) -> Vec; fn output_schema(&self) -> S; fn start<'a>( diff --git a/crates/asap-physical-operators/src/runtime/mod.rs b/crates/asap-physical-operators/src/runtime/mod.rs index cd0721e8..f72a57ef 100644 --- a/crates/asap-physical-operators/src/runtime/mod.rs +++ b/crates/asap-physical-operators/src/runtime/mod.rs @@ -50,6 +50,15 @@ pub(crate) fn execute<'r, V: 'r, S: Clone + PartialEq + Debug + 'r>( return Err(Error::Cancelled); } dag.validate(roots)?; + let mut pending = roots.to_vec(); + let mut visited = std::collections::BTreeSet::new(); + while let Some(id) = pending.pop() { + if visited.insert(id) { + let node = &dag.nodes[&id]; + node.operator.validate_context(&context)?; + pending.extend(node.inputs.iter().copied()); + } + } fn build<'r, V: 'r, S: 'r>( dag: &'r PhysicalDag<'_, V, S>, id: NodeId, diff --git a/crates/asap-physical-operators/src/stored_state/readout.rs b/crates/asap-physical-operators/src/stored_state/readout.rs index 02a04572..32200289 100644 --- a/crates/asap-physical-operators/src/stored_state/readout.rs +++ b/crates/asap-physical-operators/src/stored_state/readout.rs @@ -132,7 +132,9 @@ pub fn insufficient_counter_samples( ) && state .as_any() .downcast_ref::() - .is_some_and(|state| state.sample_count < 2) + .is_some_and(|state| { + state.sample_count < 2 || state.last_seen_timestamp == state.starting_timestamp + }) } pub fn exact_readout_optional( @@ -175,6 +177,18 @@ mod counter_tests { .unwrap(), None ); + let mut repeated = state.clone(); + repeated.update(Measurement::new(10.), 10_000); + assert_eq!( + exact_readout_optional( + [Arc::new(repeated) as Arc], + Statistic::Rate, + &None, + ¶meters + ) + .unwrap(), + None + ); state.update(Measurement::new(20.), 20_000); assert!(exact_readout_optional( [Arc::new(state.clone()) as Arc], diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index 45603015..12e43178 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -112,6 +112,10 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { } => (family, input, grouping), _ => unreachable!(), }; + let range_parameters = std::collections::HashMap::from([ + ("range_start_ms".into(), "-58000".into()), + ("range_end_ms".into(), "2000".into()), + ]); let mut expected_rate_sum = 0.; let rows = [[100., 0., 100.], [100., 200., 0.]] .into_iter() @@ -126,7 +130,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .query_statistic( asap_physical_operators::Statistic::Rate, &None, - &Default::default(), + &range_parameters, ) .unwrap(); let summary = Value::Summary { @@ -262,7 +266,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { revision: 1, }; let maintenance_scope = Scope::Ingestion { - window_start_ms: 0, + window_start_ms: -58_000, window_end_ms: 2000, revision: 1, }; @@ -270,6 +274,20 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { for candidate in candidates { let candidate = candidate.unwrap(); let inputs = if let Some(precompute) = &candidate.precompute { + let source = Operator::source(batch.schema().clone(), vec![batch.clone()]).unwrap(); + let invalid = precompute + .instantiate(BTreeMap::from([(state_id, Box::new(source) as Source<'_>)])) + .unwrap(); + let context = RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + assert!(invalid.execute(precompute.roots(), context).is_err()); let stored = run( precompute, BTreeMap::from([(state_id, batch.clone())]), @@ -310,7 +328,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .query_statistic( asap_physical_operators::Statistic::Rate, &None, - &Default::default(), + &range_parameters, ) .unwrap(); assert_ne!( From 99e980506dbe712507f577e74f15fdbf735b841a Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 04:47:51 +0000 Subject: [PATCH 42/90] Omit sparse keyed counter populations through shared readout --- .../src/stored_state/readout.rs | 32 ++++++++++++++++++- 1 file changed, 31 insertions(+), 1 deletion(-) diff --git a/crates/asap-physical-operators/src/stored_state/readout.rs b/crates/asap-physical-operators/src/stored_state/readout.rs index 32200289..fc810be3 100644 --- a/crates/asap-physical-operators/src/stored_state/readout.rs +++ b/crates/asap-physical-operators/src/stored_state/readout.rs @@ -144,7 +144,15 @@ pub fn exact_readout_optional( parameters: &std::collections::HashMap, ) -> Result, String> { let merged = merge_exact_states(states)?; - if insufficient_counter_samples(merged.as_ref(), statistic) { + let counter = key.as_ref().and_then(|key| { + merged + .as_any() + .downcast_ref::() + .and_then(|state| state.increases.get(key)) + }); + if insufficient_counter_samples(merged.as_ref(), statistic) + || counter.is_some_and(|counter| insufficient_counter_samples(counter, statistic)) + { return Ok(None); } merged @@ -189,6 +197,28 @@ mod counter_tests { .unwrap(), None ); + let mut keyed = crate::summary_kernels::KeyedCounterState::new(); + let label = crate::KeyByLabelValues::new_with_labels(vec!["checkout".into()]); + keyed.update(label.clone(), state.clone()); + assert_eq!( + exact_readout_optional( + [Arc::new(keyed.clone()) as Arc], + Statistic::Rate, + &Some(label), + ¶meters + ) + .unwrap(), + None + ); + assert!(exact_readout_optional( + [Arc::new(keyed) as Arc], + Statistic::Rate, + &Some(crate::KeyByLabelValues::new_with_labels(vec![ + "missing".into() + ])), + ¶meters + ) + .is_err()); state.update(Measurement::new(20.), 20_000); assert!(exact_readout_optional( [Arc::new(state.clone()) as Arc], From fbb620b3687c9ad48b29a162563d579d30726fe6 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 04:59:09 +0000 Subject: [PATCH 43/90] Handle sparse Planner exact counter state in shared readouts --- .../src/stored_state/readout.rs | 47 ++++++++++++++++++- .../src/summary_kernels/exact.rs | 22 +++++++++ 2 files changed, 67 insertions(+), 2 deletions(-) diff --git a/crates/asap-physical-operators/src/stored_state/readout.rs b/crates/asap-physical-operators/src/stored_state/readout.rs index fc810be3..d62a33e5 100644 --- a/crates/asap-physical-operators/src/stored_state/readout.rs +++ b/crates/asap-physical-operators/src/stored_state/readout.rs @@ -129,12 +129,16 @@ pub fn insufficient_counter_samples( matches!( statistic, crate::Statistic::Rate | crate::Statistic::Increase - ) && state + ) && (state .as_any() .downcast_ref::() .is_some_and(|state| { state.sample_count < 2 || state.last_seen_timestamp == state.starting_timestamp }) + || state + .as_any() + .downcast_ref::() + .is_some_and(|state| state.insufficient_counter_samples(statistic, &None))) } pub fn exact_readout_optional( @@ -150,7 +154,12 @@ pub fn exact_readout_optional( .downcast_ref::() .and_then(|state| state.increases.get(key)) }); - if insufficient_counter_samples(merged.as_ref(), statistic) + let exact_insufficient = merged + .as_any() + .downcast_ref::() + .is_some_and(|state| state.insufficient_counter_samples(statistic, key)); + if exact_insufficient + || insufficient_counter_samples(merged.as_ref(), statistic) || counter.is_some_and(|counter| insufficient_counter_samples(counter, statistic)) { return Ok(None); @@ -167,6 +176,40 @@ mod counter_tests { use crate::{summary_kernels::IncreaseAccumulator, AggregateCore, Measurement, Statistic}; use std::sync::Arc; + #[test] + fn planner_counter_population_omits_insufficient_samples() { + use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; + for (kind, params, statistic) in [ + (ExactKind::Rate, ExactParams::Rate, Statistic::Rate), + ( + ExactKind::Increase, + ExactParams::Increase, + Statistic::Increase, + ), + ] { + for keyed in [false, true] { + let mut state = crate::summary_kernels::exact::ExactAccumulator::new( + SummaryFamilyType::ExactAggregate(kind.clone(), params.clone()), + keyed, + ) + .unwrap(); + let key = keyed + .then(|| crate::KeyByLabelValues::new_with_labels(vec!["checkout".into()])); + state.update(key.as_ref(), 10., 10_000); + assert_eq!( + exact_readout_optional( + [Arc::new(state) as Arc], + statistic, + &key, + &Default::default() + ) + .unwrap(), + None + ); + } + } + } + #[test] fn sparse_counter_is_absent_but_invalid_ranges_still_fail() { let mut state = diff --git a/crates/asap-physical-operators/src/summary_kernels/exact.rs b/crates/asap-physical-operators/src/summary_kernels/exact.rs index 058cb5b7..8c43f8c1 100644 --- a/crates/asap-physical-operators/src/summary_kernels/exact.rs +++ b/crates/asap-physical-operators/src/summary_kernels/exact.rs @@ -53,6 +53,28 @@ impl ExactAccumulator { pub fn family(&self) -> &SummaryFamilyType { &self.family } + pub(crate) fn insufficient_counter_samples( + &self, + statistic: Statistic, + key: &Option, + ) -> bool { + if statistic != self.statistic() { + return false; + } + let state = match (&self.keyed, key) { + (Some(states), Some(key)) => states.get(key), + (None, None) => Some(&self.scalar), + _ => None, + }; + match state { + Some(ScalarState::Counter(None)) => true, + Some(ScalarState::Counter(Some(counter))) => { + counter.sample_count < 2 + || counter.last_seen_timestamp == counter.starting_timestamp + } + _ => false, + } + } pub fn is_keyed(&self) -> bool { self.keyed.is_some() } From 8ecf5f230aa14fe8de7d977a64b8e227dcb0cf20 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 06:20:22 +0000 Subject: [PATCH 44/90] perf: reuse run-local scratch for canonical exact-state merges --- .../src/stored_state/readout.rs | 22 ++++++- .../src/summary_kernels/exact.rs | 42 ++++++------ .../src/summary_kernels/increase.rs | 64 +++++++++---------- 3 files changed, 70 insertions(+), 58 deletions(-) diff --git a/crates/asap-physical-operators/src/stored_state/readout.rs b/crates/asap-physical-operators/src/stored_state/readout.rs index d62a33e5..df13f33c 100644 --- a/crates/asap-physical-operators/src/stored_state/readout.rs +++ b/crates/asap-physical-operators/src/stored_state/readout.rs @@ -108,10 +108,26 @@ fn merge_exact_states( states: impl IntoIterator>, ) -> Result, String> { let mut states = states.into_iter(); - let mut merged = states + let first = states .next() - .ok_or_else(|| "empty exact state input".to_string())? - .clone_boxed_core(); + .ok_or_else(|| "empty exact state input".to_string())?; + if let Some(first) = first + .as_any() + .downcast_ref::() + { + let mut merged = first.clone(); + for state in states { + let other = state + .as_any() + .downcast_ref::() + .ok_or_else(|| "merge requires Planner exact state".to_string())?; + merged + .merge_from(other) + .map_err(|error| error.to_string())?; + } + return Ok(Box::new(merged)); + } + let mut merged = first.clone_boxed_core(); for state in states { merged = merged .merge_with(state.as_ref()) diff --git a/crates/asap-physical-operators/src/summary_kernels/exact.rs b/crates/asap-physical-operators/src/summary_kernels/exact.rs index 8c43f8c1..8652bec4 100644 --- a/crates/asap-physical-operators/src/summary_kernels/exact.rs +++ b/crates/asap-physical-operators/src/summary_kernels/exact.rs @@ -29,6 +29,26 @@ pub struct ExactAccumulator { } impl ExactAccumulator { + /// Accumulate into run-local scratch state. Persistent input states remain + /// immutable; a failed merge discards this scratch state. + pub(crate) fn merge_from(&mut self, other: &Self) -> Result<(), Error> { + if self.family != other.family || self.is_keyed() != other.is_keyed() { + return Err("cannot merge different Planner families or layouts".into()); + } + if let (Some(target), Some(source)) = (&mut self.keyed, &other.keyed) { + for (key, state) in source { + let combined = match target.get(key) { + Some(old) => merge_scalar(old, state)?, + None => state.clone(), + }; + target.insert(key.clone(), combined); + } + } else { + self.scalar = merge_scalar(&self.scalar, &other.scalar)?; + } + Ok(()) + } + pub fn new(family: SummaryFamilyType, keyed: bool) -> Result { use ExactKind as K; use ExactParams as P; @@ -155,12 +175,7 @@ fn merge_scalar(left: &ScalarState, right: &ScalarState) -> Result ScalarState::Counter(match (a, b) { - (Some(a), Some(b)) => Some( - >::merge_accumulators(vec![ - a.clone(), - b.clone(), - ])?, - ), + (Some(a), Some(b)) => Some(IncreaseAccumulator::merge_pair(a, b)), (a, b) => a.clone().or_else(|| b.clone()), }), _ => return Err("exact scalar state families differ".into()), @@ -194,21 +209,8 @@ impl AggregateCore for ExactAccumulator { .as_any() .downcast_ref::() .ok_or("merge requires Planner exact state")?; - if self.family != other.family || self.is_keyed() != other.is_keyed() { - return Err("cannot merge different Planner families or layouts".into()); - } let mut merged = self.clone(); - if let (Some(target), Some(source)) = (&mut merged.keyed, &other.keyed) { - for (key, state) in source { - let combined = match target.get(key) { - Some(old) => merge_scalar(old, state)?, - None => state.clone(), - }; - target.insert(key.clone(), combined); - } - } else { - merged.scalar = merge_scalar(&self.scalar, &other.scalar)?; - } + merged.merge_from(other)?; Ok(Box::new(merged)) } fn get_accumulator_type(&self) -> AggregationType { diff --git a/crates/asap-physical-operators/src/summary_kernels/increase.rs b/crates/asap-physical-operators/src/summary_kernels/increase.rs index ec609532..a6cdb19c 100644 --- a/crates/asap-physical-operators/src/summary_kernels/increase.rs +++ b/crates/asap-physical-operators/src/summary_kernels/increase.rs @@ -28,6 +28,33 @@ pub struct IncreaseAccumulator { } impl IncreaseAccumulator { + /// Merge two counter intervals without a temporary collection. Ties retain + /// the left input, matching the stable ordering of multi-pane merges. + pub(crate) fn merge_pair(left: &Self, right: &Self) -> Self { + let (first, second) = if left.starting_timestamp <= right.starting_timestamp { + (left, right) + } else { + (right, left) + }; + let mut merged = first.clone(); + if second.starting_timestamp > merged.last_seen_timestamp { + merged.total_increase += + if second.starting_measurement.value >= merged.last_seen_measurement.value { + second.starting_measurement.value - merged.last_seen_measurement.value + } else { + second.starting_measurement.value + }; + } + merged.total_increase += second.total_increase; + merged.sample_count = merged.sample_count.saturating_add(second.sample_count); + if second.last_seen_timestamp > merged.last_seen_timestamp { + merged.last_seen_measurement = second.last_seen_measurement.clone(); + merged.last_seen_timestamp = second.last_seen_timestamp; + } + + merged + } + /// Return the number of bytes occupied by one accumulator at the start of /// `buffer`. Old persisted values end after `last_seen_timestamp`; reset- /// aware values carry a magic-prefixed extension. The magic makes this @@ -292,20 +319,7 @@ impl MergeableAccumulator for IncreaseAccumulator { let mut result = accumulators[0].clone(); for acc in &accumulators[1..] { - if acc.starting_timestamp > result.last_seen_timestamp { - result.total_increase += - if acc.starting_measurement.value >= result.last_seen_measurement.value { - acc.starting_measurement.value - result.last_seen_measurement.value - } else { - acc.starting_measurement.value - }; - } - result.total_increase += acc.total_increase; - result.sample_count = result.sample_count.saturating_add(acc.sample_count); - if acc.last_seen_timestamp > result.last_seen_timestamp { - result.last_seen_measurement = acc.last_seen_measurement.clone(); - result.last_seen_timestamp = acc.last_seen_timestamp; - } + result = Self::merge_pair(&result, acc); } Ok(result) @@ -348,27 +362,7 @@ impl AggregateCore for IncreaseAccumulator { .downcast_ref::() .ok_or("Failed to downcast to IncreaseAccumulator")?; - let (first, second) = if self.starting_timestamp <= other_increase.starting_timestamp { - (self, other_increase) - } else { - (other_increase, self) - }; - let mut merged = first.clone(); - if second.starting_timestamp > merged.last_seen_timestamp { - merged.total_increase += - if second.starting_measurement.value >= merged.last_seen_measurement.value { - second.starting_measurement.value - merged.last_seen_measurement.value - } else { - second.starting_measurement.value - }; - } - merged.total_increase += second.total_increase; - merged.sample_count = merged.sample_count.saturating_add(second.sample_count); - if second.last_seen_timestamp > merged.last_seen_timestamp { - merged.last_seen_measurement = second.last_seen_measurement.clone(); - merged.last_seen_timestamp = second.last_seen_timestamp; - } - + let merged = Self::merge_pair(self, other_increase); Ok(Box::new(merged)) } From be05d9a5b479fc366bd8eed243f5bd944f3a2aa2 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 07:05:50 +0000 Subject: [PATCH 45/90] fix: merge finalized pane populations once in cumulative readout --- .../src/stored_state/delta_apply.rs | 65 ++++++++++++------- .../physical-planning-and-deployment.md | 4 ++ 2 files changed, 45 insertions(+), 24 deletions(-) diff --git a/crates/asap-physical-operators/src/stored_state/delta_apply.rs b/crates/asap-physical-operators/src/stored_state/delta_apply.rs index c0daf398..e1381b40 100644 --- a/crates/asap-physical-operators/src/stored_state/delta_apply.rs +++ b/crates/asap-physical-operators/src/stored_state/delta_apply.rs @@ -550,28 +550,14 @@ pub fn cumulative_summary_state( kind: DeltaSketchKind, ) -> Result, String> { let mut rolling: Option = None; - for (_window_end, state) in samples { - match state.encoding { - SketchEncoding::ProtoFull | SketchEncoding::MsgpackFull => { - let new_state = decode_full(&kind, &state.bytes, state.encoding)?; - rolling = Some(match rolling.take() { - None => new_state, - Some(mut prev) => { - prev.merge_same_family(&new_state)?; - prev - } - }); - } - SketchEncoding::ProtoDelta | SketchEncoding::MsgpackDelta => { - if rolling.is_none() { - rolling = Some(kind.bootstrap_empty()); - } - if let Some(rs) = rolling.as_mut() { - rs.apply_delta_bytes(&state.bytes, state.encoding)?; - } - } + visit_window_summary_states(samples, kind, |_, state| { + if let Some(acc) = rolling.as_mut() { + acc.merge_same_family(&state)?; + } else { + rolling = Some(state); } - } + Ok(()) + })?; Ok(rolling) } @@ -648,6 +634,20 @@ pub fn per_window_summary_states( kind: DeltaSketchKind, ) -> Result<(Vec<(i64, SummaryState)>, usize), String> { let mut out: Vec<(i64, SummaryState)> = Vec::new(); + let skipped = visit_window_summary_states(samples, kind, |end, state| { + out.push((end, state)); + Ok(()) + })?; + Ok((out, skipped)) +} + +// Both readout modes must reconstruct the same final pane population. The +// visitor lets cumulative merging stream panes without retaining every state. +fn visit_window_summary_states( + samples: &[(i64, &SketchSampleState)], + kind: DeltaSketchKind, + mut emit: impl FnMut(i64, SummaryState) -> Result<(), String>, +) -> Result { let mut skipped = 0usize; // Rolling state for the CURRENT window only. Reset to None whenever @@ -661,7 +661,7 @@ pub fn per_window_summary_states( // state, then reset the base so this window starts from empty. if cur_end != Some(*window_end) { if let (Some(prev_end), Some(rs)) = (cur_end, rolling.take()) { - out.push((prev_end, rs)); + emit(prev_end, rs)?; } cur_end = Some(*window_end); } @@ -688,10 +688,10 @@ pub fn per_window_summary_states( // Flush the final window. if let (Some(prev_end), Some(rs)) = (cur_end, rolling.take()) { - out.push((prev_end, rs)); + emit(prev_end, rs)?; } - Ok((out, skipped)) + Ok(skipped) } // --------------------------------------------------------------------------- @@ -959,6 +959,23 @@ mod tests { sk } + /// A full re-snapshot replaces its pane's earlier frames; cumulative + /// readout must merge the finalized panes without counting updates twice. + #[test] + fn cumulative_readout_counts_resnapshot_population_once() { + let first = full(encode_dd(&dd_over(0.01, &[1., 2.]))); + let updated = full(encode_dd(&dd_over(0.01, &[1., 2., 3.]))); + let next = delta(encode_dd(&dd_over(0.01, &[9.]))); + let samples = [(1000, &first), (1000, &updated), (2000, &next)]; + let state = cumulative_summary_state(&samples, DeltaSketchKind::DDSketch { alpha: 0.01 }) + .unwrap() + .unwrap(); + let SummaryState::Dd(state) = state else { + panic!("expected DDSketch state"); + }; + assert_eq!(state.store_counts.iter().sum::(), 4); + } + /// PWR across 3 windows: window 1 is `[Full]`, windows 2 & 3 are /// `[Delta-from-empty]` (NO Full carry-in). Each window must /// reconstruct its OWN distribution's median — not empty (the old diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 8a5212b2..0c494478 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -365,3 +365,7 @@ The deployment engine executes the bound Physical DAGs through ASAPPlanner's shared physical operator implementation library, `asap-physical-operators`, and its DAG runtime. The merge executes once per run for both consumers. Execution does not introduce additional planning decisions. + +Each maintained pane contributes its finalized population once. A replacement +snapshot updates that pane's state; it does not introduce another population +when the shared runtime merges panes for a query. From d02eabd0f843f87025073d78dc1d78a34733a7e9 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 13:42:49 +0000 Subject: [PATCH 46/90] docs: remove DataFusion execution comparison --- .../datafusion-execution-comparison.md | 89 ------------------- 1 file changed, 89 deletions(-) delete mode 100644 docs/design_docs/datafusion-execution-comparison.md diff --git a/docs/design_docs/datafusion-execution-comparison.md b/docs/design_docs/datafusion-execution-comparison.md deleted file mode 100644 index b388fa16..00000000 --- a/docs/design_docs/datafusion-execution-comparison.md +++ /dev/null @@ -1,89 +0,0 @@ -# DataFusion versus native ASAP execution - -## Decision for #462 - -Implement and maintain all ASAP physical operators, summary operators and shared -DAG execution locally, following DataFusion's separation of plan contracts, -runtime, expressions and concrete operators. A DataFusion backend or hybrid -runtime is outside this PR's chosen direction. - -This gives ASAP direct ownership of native summary-state edges and shared -execution across precompute and query engines. It also makes ASAP responsible -for operator correctness, resource control and future parallelism/spill. -It is an architectural choice, not a measured performance advantage. - -## Key differences - -| Aspect | Reuse DataFusion execution | Implement in ASAP | -| --- | --- | --- | -| Runtime overhead | Partition streams, dynamic dispatch and optional exchange/buffering; ordinary operators do not each spawn a task | Worker-local streams, producer queues, reader tracking, row/value dispatch and copies | -| Data representation | Standard physical edges carry Arrow RecordBatches; accumulators can hold native Rust sketches internally | Typed native batches can carry summary-state objects directly | -| Shared producers | Sharing a plan pointer does not ensure one execution; requires fusion, materialization or custom coordination | One producer per node per run, with independent consumer cursors | -| Summary extensions | Logical nodes, extension planners and physical/UDAF APIs exist; a core fork is not inherently required | ASAP owns interfaces, lowering and execution directly | -| Optimizations and algorithms | Existing relational operators and partition/spill infrastructure, subject to valid properties and state contracts | Local implementations; equivalent capabilities must be developed and maintained | -| Maintenance | Adaptation, semantic integration and upstream version changes | Algorithms, runtime contracts, regressions and feature development | - -Neither runtime is inherently cheaper. Arrow batch clones share buffers, while -conversion from rows and sketch encoding may allocate. Native state edges avoid -encoding but retain queue, allocation and row-cloning costs. Specialized kernels -and algorithms can dominate framework overhead. No comparative benchmark has run. - -## What DataFusion integration would require - -**State transport.** Native sketches can live inside a UDAF accumulator. -`update_batch` need not serialize them; `state`/`merge_batch` export and consume -partial state. Across standard physical edges, use Arrow-compatible binary or -structured state, or keep sketches inside a fused operator until scalar readout. -Run-local handles require explicit lifetime, memory and transport restrictions. -Arrow extension metadata alone does not make arbitrary Rust objects portable. - -**Sharing.** Expression CSE, shared plan identity and shared execution are distinct. -Multiple compatible quantiles can be fused into one KLL build and multi-readout; -independent downstream branches may need true fan-out. Custom shared execution -must handle per-run identity, slow/dropped consumers, errors, cancellation and -buffering. Asymmetric consumer polling can deadlock bounded broadcast queues; -ASAP's own runtime has the same scheduling obligation. - -**Optimization and partitions.** Summary nodes must preserve population, grouping, -window coverage, parameters and approximation guarantees. A legal partial/final -merge requires compatible states and nonduplicated input coverage. Logical and -physical extension APIs provide hooks; they do not establish these semantics. -Reusing physical operators alone also does not automatically reuse logical -optimization over ASAP IR. - -**Lifecycle.** Custom state must participate in memory accounting and cooperative -cancellation. Hybrid execution would additionally coordinate budgets, workers -and state ownership across runtimes. These are adapter responsibilities, not -proof that DataFusion core must change. - -## What to borrow now - -Borrow module boundaries and explicit contracts for schemas, boundedness, -emission, memory and cancellation. Keep ASAP's `SummaryExpr`, typed summary -states and shared DAG. #462 establishes those boundaries; partition parallelism, -spill and richer physical properties remain future native work. - -Copying an algorithm is a larger commitment than following organization: -DataFusion joins, sorts and aggregates depend on Arrow kernels, expressions, -partition properties, memory reservations and spill infrastructure. Any port -requires adaptation, tests and ongoing maintenance; upstream fixes do not arrive -automatically. - -Future performance comparisons should use identical sketch kernels, parameters, -inputs and worker counts. Measure setup separately from execution, including -latency, CPU, allocations, peak memory and encoded bytes. Useful workloads are -KLL multi-readout, asymmetric fan-out, grouped scans and memory-constrained -joins/sorts. Such measurements are not a prerequisite for the current decision. - -## Source scope - -Reviewed upstream at -[`e2ca7f3`](https://github.com/apache/datafusion/tree/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38) -on 2026-09-24. The SQL frontend uses DataFusion 43; upstream-main APIs below are -not a claim about that release. This PR adds no DataFusion execution dependency. - -- [ExecutionPlan and physical properties](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/execution_plan.rs) -- [Accumulator state/update/merge interface](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/expr-common/src/accumulator.rs) -- [Logical extension contracts](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/expr/src/logical_plan/extension.rs) and [physical extension planning](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/core/src/physical_planner.rs) -- [Expression CSE](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/optimizer/src/common_subexpr_eliminate.rs) and [specialized scalar-subquery sharing](https://github.com/apache/datafusion/blob/e2ca7f38051744b2010cf09b80db8e9dfa4b5d38/datafusion/physical-plan/src/scalar_subquery.rs) -- [Arrow RecordBatch ownership](https://arrow.apache.org/rust/arrow/array/struct.RecordBatch.html) and [extension types](https://arrow.apache.org/docs/format/Intro.html#extension-types) From 34fad57386392f0075a96b3a7009a9e1bb0350c6 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 13:59:52 +0000 Subject: [PATCH 47/90] test: cover lifecycle planning and shared physical execution end to end --- Cargo.lock | 2 + .../src/operators/summary/mod.rs | 12 +- .../src/physical_planner/mod.rs | 5 +- .../tests/precompute_candidates.rs | 28 +- crates/integration-tests/Cargo.toml | 3 + .../tests/kll_pane_execution.rs | 294 ++++++++++++++++++ .../tests/physical_common/mod.rs | 42 +++ .../tests/promql_to_post_asap.rs | 9 +- .../tests/sql_to_physical.rs | 165 ++++++++++ .../summary_maintenance_lifecycle_e2e.rs | 272 +++++++++++++--- .../physical-planning-and-deployment.md | 19 ++ 11 files changed, 796 insertions(+), 55 deletions(-) create mode 100644 crates/integration-tests/tests/kll_pane_execution.rs create mode 100644 crates/integration-tests/tests/physical_common/mod.rs create mode 100644 crates/integration-tests/tests/sql_to_physical.rs diff --git a/Cargo.lock b/Cargo.lock index f5ac50fd..e7b6b6c6 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -368,8 +368,10 @@ dependencies = [ "asap-aware-mapping", "asap-frontend-promql", "asap-frontend-sql", + "asap-physical-operators", "asap-types", "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", + "futures", "serde_json", "tokio", ] diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs index 70fbf79a..75102d8b 100644 --- a/crates/asap-physical-operators/src/operators/summary/mod.rs +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -90,8 +90,8 @@ impl Operator { ) -> Result { crate::values::validate_family(&family)?; validate_groups(&input, &groups)?; - if plain(&input, value)? != (&DataType::Float64, false) { - return Err(invalid("summary numeric update requires non-null Float64")); + if plain(&input, value)?.0 != &DataType::Float64 { + return Err(invalid("summary numeric update requires Float64")); } if let Some(time) = time { if plain(&input, time)? != (&DataType::Timestamp, false) { @@ -374,8 +374,12 @@ async fn build_summary( } let (_, updater, memory, overhead, previous) = states.get_mut(&key).expect("inserted group"); - let Value::Float64(value) = row[value] else { - return Err(invalid("summary update type")); + // SQL aggregates ignore NULL samples while retaining the group. + // A missing counter sample also contributes no observation. + let value = match row[value] { + Value::Float64(value) => value, + Value::Null => continue, + _ => return Err(invalid("summary update type")), }; let timestamp = if let Some(time) = time { let Value::Timestamp(time) = row[time] else { diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index d7474271..7b8b91fa 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -494,7 +494,10 @@ fn summary_column(input: &Schema) -> Result { } fn named_column(input: &Schema, column: &ColumnRef) -> Result { let name = match column { - ColumnRef::Named(name) => name.as_str(), + // Executable SummarySchema retains column names, not table qualifiers. + // Frontend binding has resolved the qualifier; still reject ambiguous + // names here rather than guessing a join side. + ColumnRef::Named(name) | ColumnRef::Qualified { name, .. } => name.as_str(), ColumnRef::SampleValue => "value", _ => { return Err(invalid( diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index 12e43178..470272e3 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -195,9 +195,12 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { InputContract::bounded(Arc::new(state.output_schema.clone())), )]), &[root], - &[vec![rate_id], vec![root]], + &[vec![999], vec![rate_id], vec![root]], ); + assert!(inventory[0].is_err()); + let mut evaluated = 0; let selected = select_candidate(inventory, |candidate| { + evaluated += 1; let grouped = candidate.materialized_outputs.contains_key(&root); Ok(Some(CandidateCost { workload_scope: "reset-counter-workload".into(), @@ -211,6 +214,29 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { prefer_grouped ); assert_eq!(selected.cost.total_cost, 1.); + assert_eq!(evaluated, 2, "uncompilable candidates must never be priced"); + let candidate = selected.candidate; + let precompute = candidate.precompute.as_ref().unwrap(); + let stored = run( + precompute, + BTreeMap::from([(state_id, batch.clone())]), + Scope::Ingestion { + window_start_ms: -58_000, + window_end_ms: 2000, + revision: 1, + }, + ); + let output = run( + &candidate.query, + BTreeMap::from([(precompute.roots()[0], stored[0].clone())]), + Scope::Query { + evaluation_time_ms: 2000, + revision: 1, + }, + ); + assert!( + matches!(output[0].rows()[0][1], Value::Float64(value) if value == expected_rate_sum) + ); } let contracts = BTreeMap::from([( state_id, diff --git a/crates/integration-tests/Cargo.toml b/crates/integration-tests/Cargo.toml index 7b0c0d3e..5a5de990 100644 --- a/crates/integration-tests/Cargo.toml +++ b/crates/integration-tests/Cargo.toml @@ -13,3 +13,6 @@ asap-aware-mapping = { path = "../asap-aware-mapping" } asap_sketchlib = { workspace = true } serde_json = "1" tokio = { version = "1", features = ["rt", "macros", "rt-multi-thread"] } + +asap-physical-operators = { path = "../asap-physical-operators" } +futures = "0.3" diff --git a/crates/integration-tests/tests/kll_pane_execution.rs b/crates/integration-tests/tests/kll_pane_execution.rs new file mode 100644 index 00000000..2273f965 --- /dev/null +++ b/crates/integration-tests/tests/kll_pane_execution.rs @@ -0,0 +1,294 @@ +//! Maintenance -> wire state -> independently bound query execution. +mod physical_common; +use asap_physical_operators::{ + operators::Operator, + physical_planner::{CompiledPhysicalDag, InputContract, Source}, + plan::{PhysicalDag, PhysicalOperator, PlanProperties}, + runtime::{Input, Limits, OutputStream, RunContext, Scope}, + summary_kernels::datasketches_kll::DatasketchesKLLAccumulator, + values::{Batch, Schema, Value}, + Error, Statistic, +}; +use asap_types::{ + post_asap::{ + SketchAlgorithm, SketchKind, SketchParams, SummaryFamilyType, SummaryField, SummarySchema, + }, + pre_asap::DataType, +}; +use futures::{executor::block_on, StreamExt}; +use std::{ + collections::{BTreeMap, HashMap}, + sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, + }, +}; + +fn family(k: u32) -> SummaryFamilyType { + SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), + Default::default(), + ) +} +fn raw_schema() -> Schema { + Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }) +} +fn query_scope() -> Scope { + Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + } +} +fn fixture() -> (CompiledPhysicalDag, Operator, Schema) { + let raw = raw_schema(); + let build = Operator::summary_build(raw.clone(), family(200), 0, None, vec![]).unwrap(); + let state = build.schema(); + let maintenance = CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(raw))]), + BTreeMap::from([(1, (vec![0], build))]), + vec![1], + ) + .unwrap(); + let merge = Operator::summary_merge(state.clone(), 0, vec![]).unwrap(); + (maintenance, merge, state) +} +fn pane_bytes(maintenance: &CompiledPhysicalDag, pane: i64) -> Vec { + // Twenty samples in each (start,end] one-minute pane; k=200 avoids + // compaction so quantiles and sample counts have deterministic oracles. + let raw = raw_schema(); + let rows = (0..20) + .map(|i| vec![Value::Float64((pane * 20 + i) as f64)]) + .collect(); + let state = physical_common::execute( + maintenance, + BTreeMap::from([(0, Batch::try_new(raw, rows).unwrap())]), + Scope::Ingestion { + window_start_ms: pane * 60_000, + window_end_ms: (pane + 1) * 60_000, + revision: 1, + }, + ); + let Value::Summary { state, .. } = &state[0][0].rows()[0][0] else { + panic!("missing KLL") + }; + state.serialize_to_bytes() +} +fn restore(schema: Schema, bytes: &[Vec]) -> Batch { + Batch::try_new( + schema, + bytes + .iter() + .map(|bytes| { + vec![Value::Summary { + family: family(200), + state: Arc::new(DatasketchesKLLAccumulator::from_msgpack_bytes(bytes).unwrap()), + }] + }) + .collect(), + ) + .unwrap() +} +fn readout(schema: Schema, q: f64) -> Operator { + Operator::readout( + schema, + 0, + Statistic::Quantile, + HashMap::from([("quantile".into(), q.to_string())]), + ) + .unwrap() +} +struct CountStarts { + operator: Operator, + starts: Arc, +} +impl PhysicalOperator for CountStarts { + fn name(&self) -> &str { + self.operator.name() + } + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + self.operator.properties(inputs) + } + fn requires_bounded_input(&self) -> bool { + self.operator.requires_bounded_input() + } + fn input_schemas(&self) -> Vec { + self.operator.input_schemas() + } + fn output_schema(&self) -> Schema { + self.operator.output_schema() + } + fn output_bytes(&self, batch: &Batch) -> usize { + self.operator.output_bytes(batch) + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + self.starts.fetch_add(1, Ordering::SeqCst); + self.operator.start(inputs, context) + } +} + +/// Actual codec bytes survive destruction of maintenance state; one shared +/// native merge supplies p50, p99 and the population-count oracle per run. +#[test] +fn five_panes_roundtrip_and_shared_merge_runs_once() { + let (maintenance, merge, schema) = fixture(); + let bytes: Vec<_> = (0..6).map(|pane| pane_bytes(&maintenance, pane)).collect(); + drop(maintenance); + let compiled = CompiledPhysicalDag::from_operators( + (0..5) + .map(|id| (id, InputContract::bounded(schema.clone()))) + .collect(), + BTreeMap::from([ + ( + 5, + ( + vec![0, 1, 2, 3, 4], + Operator::union(schema.clone(), 5).unwrap(), + ), + ), + (6, (vec![5], merge.clone())), + (7, (vec![6], readout(schema.clone(), 0.5))), + (8, (vec![6], readout(schema.clone(), 0.99))), + ]), + vec![6, 7, 8], + ) + .unwrap(); + let layout = asap_types::post_asap::PaneLayout { + pane_width_ms: 60_000, + pane_origin_ms: Some(0), + }; + assert!( + asap_types::post_asap::validate_pane_coverage( + &layout, + Some(330_000), + &asap_types::post_asap::WindowEdgeCoverage::PaneAligned + ) + .is_err(), + "moving window edges require residual computation" + ); + for offset in [0, 1] { + let restored = restore(schema.clone(), &bytes[offset..offset + 5]); + let evaluation_time_ms = (5 + offset as i64) * 60_000; + asap_types::post_asap::validate_pane_coverage( + &layout, + Some(evaluation_time_ms), + &asap_types::post_asap::WindowEdgeCoverage::PaneAligned, + ) + .unwrap(); + let inputs: BTreeMap<_, _> = (0..5) + .map(|id| { + ( + id as u64, + restore(schema.clone(), &bytes[offset + id..offset + id + 1]), + ) + }) + .collect(); + // A five-pane deployment cannot bind only four state slots. + let incomplete: BTreeMap<_, _> = inputs + .iter() + .take(4) + .map(|(&id, batch)| { + ( + id, + Box::new(Operator::source(schema.clone(), vec![batch.clone()]).unwrap()) + as Source<'_>, + ) + }) + .collect(); + assert!(compiled.instantiate(incomplete).is_err()); + let result = physical_common::execute( + &compiled, + inputs, + Scope::Query { + evaluation_time_ms, + revision: 1, + }, + ); + let Value::Summary { state, .. } = &result[0][0].rows()[0][0] else { + panic!("missing merged state") + }; + let kll = state + .as_any() + .downcast_ref::() + .unwrap(); + assert_eq!(kll.inner.count(), 100); + let value = |index: usize| match result[index][0].rows()[0][0] { + Value::Float64(value) => value, + _ => panic!("missing quantile"), + }; + assert!((value(1) - (50 + offset * 20) as f64).abs() <= 1.); + assert!((value(2) - (99 + offset * 20) as f64).abs() <= 1.); + let starts = Arc::new(AtomicUsize::new(0)); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(schema.clone(), vec![restored]).unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + CountStarts { + operator: merge.clone(), + starts: starts.clone(), + }, + ) + .unwrap(); + dag.add(2, vec![1], readout(schema.clone(), 0.5)).unwrap(); + dag.add(3, vec![1], readout(schema.clone(), 0.99)).unwrap(); + let outputs = block_on(futures::future::join_all( + dag.execute( + &[2, 3], + RunContext::new(query_scope(), Limits::default()).unwrap(), + ) + .unwrap() + .into_iter() + .map(|stream| stream.collect::>()), + )); + assert!(outputs + .iter() + .all(|output| output.len() == 1 && output[0].is_ok())); + assert_eq!(starts.load(Ordering::SeqCst), 1); + } +} + +/// Corrupt wire data, relabelled KLL parameters and missing bindings fail +/// explicitly; decoding success is not a license to change the state contract. +#[test] +fn restored_panes_reject_corruption_parameters_schema_and_missing_binding() { + let (maintenance, merge, schema) = fixture(); + let bytes = pane_bytes(&maintenance, 0); + assert!(DatasketchesKLLAccumulator::from_msgpack_bytes(&bytes[..bytes.len() / 2]).is_err()); + let wrong = Value::Summary { + family: family(200), + state: Arc::new(DatasketchesKLLAccumulator::new(128)), + }; + assert!(Batch::try_new(schema.clone(), vec![vec![wrong]]).is_err()); + let compiled = CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema))]), + BTreeMap::from([(1, (vec![0], merge))]), + vec![1], + ) + .unwrap(); + assert!(compiled.instantiate(BTreeMap::new()).is_err()); + let raw = raw_schema(); + let source = Operator::source( + raw.clone(), + vec![Batch::try_new(raw, vec![vec![Value::Float64(1.)]]).unwrap()], + ) + .unwrap(); + assert!(compiled + .instantiate(BTreeMap::from([(0, Box::new(source) as Source<'_>)])) + .is_err()); +} diff --git a/crates/integration-tests/tests/physical_common/mod.rs b/crates/integration-tests/tests/physical_common/mod.rs new file mode 100644 index 00000000..aca4d4ef --- /dev/null +++ b/crates/integration-tests/tests/physical_common/mod.rs @@ -0,0 +1,42 @@ +use asap_physical_operators::{ + operators::Operator, + physical_planner::{CompiledPhysicalDag, Source}, + runtime::{Limits, RunContext, Scope}, + values::Batch, +}; +use futures::{executor::block_on, StreamExt}; +use std::collections::BTreeMap; + +pub fn execute( + plan: &CompiledPhysicalDag, + inputs: BTreeMap, + scope: Scope, +) -> Vec> { + let sources = inputs + .into_iter() + .map(|(id, batch)| { + ( + id, + Box::new(Operator::source(batch.schema().clone(), vec![batch]).unwrap()) + as Source<'_>, + ) + }) + .collect(); + let dag = plan.instantiate(sources).unwrap(); + block_on(async { + let streams = dag + .execute( + plan.roots(), + RunContext::new(scope, Limits::default()).unwrap(), + ) + .unwrap(); + futures::future::join_all(streams.into_iter().map(|mut stream| async move { + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push((*batch.unwrap()).clone()); + } + batches + })) + .await + }) +} diff --git a/crates/integration-tests/tests/promql_to_post_asap.rs b/crates/integration-tests/tests/promql_to_post_asap.rs index f94a02af..539db7bf 100644 --- a/crates/integration-tests/tests/promql_to_post_asap.rs +++ b/crates/integration-tests/tests/promql_to_post_asap.rs @@ -831,7 +831,7 @@ fn execute_topk_reference(plan: &SummaryNode) -> Vec<(String, f64)> { } #[test] -fn planner_topk_reference_execution_matches_ground_truth() { +fn planner_heap_topk_reference_execution_matches_ground_truth() { // Pin numeric results independently of the emitted IR: swapping weights, // losing identity, changing the window, or dropping k changes the answer. for (query, expected) in [ @@ -865,8 +865,11 @@ fn planner_topk_reference_execution_matches_ground_truth() { &EqualSplitAllocator, &SeparatedTopK, ); - let candidates = strategy.replacements(&TargetSubDAG::new(&pre)); - assert!(!candidates.is_empty(), "no plan for {query}"); + // This reference executor consumes keyed heap updates. The inventory + // also contains maintained exact values followed by sort/limit; those + // have a different execution contract and must not enter this fixture. + let candidates: Vec<_> = strategy.replacements(&TargetSubDAG::new(&pre)).into_iter().filter(|candidate| matches!(&candidate.replacement, Replacement::Summary(plan) if matches!(plan.expr, SummaryExpr::SummaryEstimate { query: SketchQuery::TopK { .. }, .. }))).collect(); + assert!(!candidates.is_empty(), "no heap candidate for {query}"); for candidate in candidates { let Replacement::Summary(plan) = candidate.replacement else { panic!("expected summary plan for {query}") diff --git a/crates/integration-tests/tests/sql_to_physical.rs b/crates/integration-tests/tests/sql_to_physical.rs new file mode 100644 index 00000000..22cb92a4 --- /dev/null +++ b/crates/integration-tests/tests/sql_to_physical.rs @@ -0,0 +1,165 @@ +//! SQL frontend, candidate selection, physical compilation and fresh-run execution. +use asap_aware_mapping::{search_workload, DefaultCostModel}; +use asap_frontend_sql::{lower_sql, SqlCatalog}; +use asap_physical_operators::{ + physical_planner::{compile, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + sources::{DataSources, MemorySource}, + values::{Batch, Value}, +}; +use asap_types::{ + post_asap::{compile_executable_dag, ExecutableOperatorPayload, SummaryFamilyType}, + pre_asap::{Column, DataType, QueryExpr, Schema}, + types::AccuracyTarget, +}; +use futures::StreamExt; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; + +/// SQL filtering and grouped aggregation survive logical/physical lowering; +/// rebinding the compiled DAG runs against new data rather than cached results. +#[tokio::test] +async fn sql_filter_grouped_sum_executes_and_rebinds() { + let catalog = SqlCatalog::new().with_table( + "metrics", + Schema::new(vec![ + Column::new("service", DataType::Utf8, false), + Column::new("value", DataType::Float64, true), + ]), + ); + for query in [ + "SELECT service, SUM(value) AS total FROM metrics WHERE value > 1 GROUP BY service", + "SELECT service, SUM(value) AS total FROM metrics GROUP BY service", + ] { + let logical = Rc::new( + lower_sql(query, &catalog, AccuracyTarget::Exact) + .await + .unwrap(), + ); + let space = search_workload(vec![("sql", logical)]); + let selected = space + .global_selection(&DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + let dag = compile_executable_dag(&selected).unwrap(); + let scan = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + ExecutableOperatorPayload::Fallback { + expression: QueryExpr::Scan { .. } + } + ) + }) + .expect("raw SQL scan"); + let schema = Arc::new(scan.output_schema.clone()); + assert!(schema + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + let plan = compile( + &dag, + BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + for multiplier in [1., 2.] { + let rows = [ + ("api", Some(2.)), + ("api", Some(3.)), + ("api", None), + ("batch", Some(4.)), + ("batch", Some(1.)), + ] + .into_iter() + .map(|(service, value)| { + schema + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "value" => { + value.map_or(Value::Null, |value| Value::Float64(value * multiplier)) + } + _ => panic!("unexpected field {field:?}"), + }) + .collect() + }) + .collect(); + let ExecutableOperatorPayload::Fallback { expression } = &scan.payload else { + unreachable!() + }; + let QueryExpr::Scan { source, .. } = expression else { + unreachable!() + }; + let mut sources = DataSources::default(); + sources + .register( + source.clone(), + Arc::new( + MemorySource::new( + schema.clone(), + vec![Batch::try_new(schema.clone(), rows).unwrap()], + ) + .unwrap(), + ), + ) + .unwrap(); + let bound = plan + .instantiate(BTreeMap::from([( + u64::from(scan.id.0), + Box::new(sources.bind(expression).unwrap()) as Source<'_>, + )])) + .unwrap(); + let mut stream = bound + .execute( + plan.roots(), + RunContext::new( + Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(), + ) + .unwrap() + .remove(0); + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push(batch.unwrap()); + } + let mut actual: Vec<_> = batches + .iter() + .flat_map(|batch| batch.rows()) + .map(|row| { + let Value::Utf8(service) = &row[0] else { + panic!("missing service") + }; + let Value::Float64(value) = row[1] else { + panic!("missing sum") + }; + (service.to_string(), value) + }) + .collect(); + actual.sort_by(|a, b| a.0.cmp(&b.0)); + assert_eq!( + actual, + vec![ + ("api".into(), 5. * multiplier), + ( + "batch".into(), + 4. * multiplier + + if query.contains("WHERE") && multiplier == 1. { + 0. + } else { + multiplier + } + ) + ] + ); + } + } +} diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index 186dd04b..b2a5ede1 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -119,52 +119,7 @@ fn dashboard_workload() -> PlanningWorkload { #[test] fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() { let workload = dashboard_workload(); - workload.validate().unwrap(); - - let lowered = lower_promql_workload(&workload, 0) - .expect("valid PromQL workload") - .into_iter() - .next() - .expect("one normalized workload entry"); - let root = Rc::new(lowered); - let strategies = asap_aware_mapping::default_strategies_with(&FullyCostedRuntime); - let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); - let target = Rc::clone(&space.roots[0].1); - let capabilities = SummaryMaintenanceLifecycleCapabilities { - supports_ephemeral: true, - supports_prepared: false, - supports_shared: false, - supports_continuously_maintained: true, - }; - - let selection = global_selection_with_summary_maintenance_lifecycles( - &space, - WorkloadDemand { - workload: &workload.query_workload, - data_workload: workload.data_workload.as_ref(), - entry_indices: &[1], - }, - NOW_MS, - Some(Horizon(100.0)), - capabilities, - &FullyCostedRuntime, - ) - .unwrap(); - let plan = assemble_selected_dag_with_summary_maintenance_lifecycles( - &selection, - &target, - WorkloadDemand::new_with_data( - &workload.query_workload, - workload.data_workload.as_ref().unwrap(), - &[1], - ), - NOW_MS, - Some(Horizon(100.0)), - capabilities, - &FullyCostedRuntime, - ) - .unwrap() - .expect("selected summary plan"); + let plan = selected_plan(&workload); assert!(!plan.selected_raw_recompute); assert_eq!(plan.expected_reads, Some(100.0)); @@ -231,3 +186,228 @@ fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() "continuously_maintained" ); } + +fn selected_plan( + workload: &PlanningWorkload, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + workload.validate().unwrap(); + + let lowered = lower_promql_workload(workload, 0) + .expect("valid PromQL workload") + .into_iter() + .next() + .expect("one normalized workload entry"); + let root = Rc::new(lowered); + let strategies = asap_aware_mapping::default_strategies_with(&FullyCostedRuntime); + let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); + let target = Rc::clone(&space.roots[0].1); + let capabilities = SummaryMaintenanceLifecycleCapabilities { + supports_ephemeral: true, + supports_prepared: false, + supports_shared: false, + supports_continuously_maintained: true, + }; + + let selection = global_selection_with_summary_maintenance_lifecycles( + &space, + WorkloadDemand { + workload: &workload.query_workload, + data_workload: workload.data_workload.as_ref(), + entry_indices: &[1], + }, + NOW_MS, + Some(Horizon(100.0)), + capabilities, + &FullyCostedRuntime, + ) + .unwrap(); + assemble_selected_dag_with_summary_maintenance_lifecycles( + &selection, + &target, + WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + NOW_MS, + Some(Horizon(100.0)), + capabilities, + &FullyCostedRuntime, + ) + .unwrap() + .expect("selected summary plan") +} + +mod physical_common; + +/// A selected continuous lifecycle supplies a materialization boundary; its +/// maintenance and query DAGs execute the selected KLL computation in fresh runs. +#[test] +fn continuous_lifecycle_compiles_and_executes_spatial_kll() { + use asap_physical_operators::{ + physical_planner::{compile_candidate, InputContract}, + runtime::Scope, + values::{Batch, Value}, + }; + use asap_types::{ + post_asap::{compile_executable_dag, ExecutableOperatorPayload, SummaryFamilyType}, + pre_asap::DataType, + }; + use std::{collections::BTreeMap, sync::Arc}; + let mut workload = dashboard_workload(); + workload.query_workload.query_batch.as_mut().unwrap()[0].query = + Query("quantile(0.99, latency)".into()); + workload.query_workload.repeating_queries.as_mut().unwrap()[0].query = + Query("quantile(0.99, latency)".into()); + let selected = selected_plan(&workload); + assert_eq!( + selected.deployments[0] + .summary_maintenance_lifecycle_guarantee + .as_ref() + .unwrap() + .summary_maintenance_lifecycle, + SummaryMaintenanceLifecycle::ContinuouslyMaintained + ); + let dag = compile_executable_dag(&selected.root).unwrap(); + let build = dag + .nodes + .iter() + .find(|node| matches!(node.payload, ExecutableOperatorPayload::SummaryAgg { .. })) + .unwrap(); + let input = dag + .edges + .iter() + .find(|edge| edge.consumer == build.id) + .unwrap() + .producer; + let raw = dag.nodes.iter().find(|node| node.id == input).unwrap(); + let schema = Arc::new(raw.output_schema.clone()); + let candidate = compile_candidate( + &dag, + BTreeMap::from([(u64::from(input.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &[u64::from(build.id.0)], + ) + .unwrap(); + + // A continuous input without a finite pane boundary cannot implement this + // blocking builder. Retain lifecycle ownership in the candidate payload; + // only the legal bounded request candidate reaches workload pricing. + let mut unbounded = InputContract::bounded(schema.clone()); + unbounded.properties.boundedness = asap_physical_operators::plan::Boundedness::Unbounded; + let rejected = compile_candidate( + &dag, + BTreeMap::from([(u64::from(input.0), unbounded)]), + &[u64::from(dag.root.0)], + &[u64::from(build.id.0)], + ); + assert!(rejected.is_err()); + let request = compile_candidate( + &dag, + BTreeMap::from([(u64::from(input.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &[], + ) + .unwrap(); + let mut priced = 0; + let feedback = asap_physical_operators::physical_planner::select_candidate( + vec![ + rejected.map(|candidate| { + ( + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + candidate, + ) + }), + Ok((SummaryMaintenanceLifecycle::Ephemeral, request)), + ], + |_| { + priced += 1; + Ok(Some( + asap_physical_operators::physical_planner::CandidateCost { + workload_scope: "dashboard".into(), + horizon_seconds: 100., + total_cost: 1000., + }, + )) + }, + ) + .unwrap(); + assert_eq!(priced, 1); + assert_eq!(feedback.candidate.0, SummaryMaintenanceLifecycle::Ephemeral); + for revision in [1, 2] { + let rows = (1..=100) + .map(|value| { + schema + .fields + .iter() + .map(|field| match field.dtype { + SummaryFamilyType::Plain(DataType::Float64) => { + Value::Float64(f64::from(value)) + } + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), + _ => panic!("unexpected field {field:?}"), + }) + .collect() + }) + .collect(); + let raw_batch = Batch::try_new(schema.clone(), rows).unwrap(); + let direct = physical_common::execute( + &feedback.candidate.1.query, + BTreeMap::from([(u64::from(input.0), raw_batch.clone())]), + Scope::Query { + evaluation_time_ms: 300_000, + revision, + }, + ); + let state = physical_common::execute( + candidate.precompute.as_ref().unwrap(), + BTreeMap::from([(u64::from(input.0), raw_batch)]), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 300_000, + revision, + }, + ); + let result = physical_common::execute( + &candidate.query, + BTreeMap::from([(u64::from(build.id.0), state[0][0].clone())]), + Scope::Query { + evaluation_time_ms: 300_000, + revision, + }, + ); + let values: Vec<_> = result[0] + .iter() + .flat_map(|batch| batch.rows()) + .flat_map(|row| row.iter()) + .filter_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .collect(); + let direct_values: Vec<_> = direct[0] + .iter() + .flat_map(|batch| batch.rows()) + .flat_map(|row| row.iter()) + .filter_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .collect(); + assert_eq!( + values, direct_values, + "maintenance and request candidates preserve the same population" + ); + assert_eq!(values.len(), 1); + assert!( + (98. ..=100.).contains(&values[0]), + "p99 rank must reflect the supplied population" + ); + } +} diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 0c494478..c98fe47c 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -369,3 +369,22 @@ does not introduce additional planning decisions. Each maintained pane contributes its finalized population once. A replacement snapshot updates that pane's state; it does not introduce another population when the shared runtime merges panes for a query. + +## 6. Executable acceptance coverage + +The tests distinguish optimizer-selected lifecycle execution from explicit +physical pane construction: + +| Test | Contract exercised | +| --- | --- | +| `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled maintenance/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate returns the same population | +| `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute maintenance DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | +| `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | +| `precompute_candidates::grouped_rate_can_be_materialized_before_or_after_grouped_sum` | Cost changes select different legal precompute frontiers; both selected candidates execute with the same reset-sensitive result; uncompilable candidates are not priced | +| `sql_to_physical::sql_filter_grouped_sum_executes_and_rebinds` | SQL text → candidate search → physical compilation → shared Scan predicates and grouped summary execution; NULL samples are ignored and fresh bindings produce new results | + +The pane test uses an explicit physical realization. It does not establish that +maintenance selection automatically emits the complete temporal pane DAG. +Pane phase validation uses the Planner coverage contract; concrete stored-pane +identity, revision, readiness and population coverage remain deployment checks. +Real storage and HTTP execution belong to deployment-repository E2E tests. From 07d4937e96e7b0ea152c097294e110e82903978b Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 14:36:27 +0000 Subject: [PATCH 48/90] feat: compile selected temporal KLL maintenance into pane DAGs --- crates/asap-physical-operators/README.md | 13 + .../src/operators/mod.rs | 24 +- .../src/operators/panes.rs | 228 ++++++++ .../src/physical_planner/mod.rs | 6 + .../src/physical_planner/temporal_panes.rs | 332 ++++++++++++ .../summary_maintenance_lifecycle_e2e.rs | 488 +++++++++++++++++- .../physical-planning-and-deployment.md | 28 +- 7 files changed, 1106 insertions(+), 13 deletions(-) create mode 100644 crates/asap-physical-operators/src/operators/panes.rs create mode 100644 crates/asap-physical-operators/src/physical_planner/temporal_panes.rs diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index e1d98069..31bbc49a 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -111,3 +111,16 @@ runnable graph without repeating logical lowering. The graph executes through the shared runtime with independent per-run state. Window coverage, revision and maintenance-policy admission remain deployment/planning contracts; this compiler does not discover storage or silently change a selected maintenance strategy. + +`physical_planner::compile_temporal_pane_candidate` lowers a selected continuous +KLL lifecycle and Sliding/Tumbling framework into maintenance and query DAGs. +`TemporalPaneMaintenance` supplies pane geometry and a resolved complete entity +identity contract. The compiler inserts population guards, scan predicates, +pane construction, ordered state slots, a shared merge and quantile readouts. +Pane outputs have distinct physical identities from the logical whole-window +summary, and the returned candidate retains the maintenance contract for binding. +Each run checks phase, pane timestamps and duplicate entity states. The initial +realization uses complete bounded snapshots; partial edges, exponential +histograms and cross-run delta accumulation are unsupported. Storage identities, +revision selection, completeness/readiness evidence and scheduling stay with +deployment. diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index ecc30279..0b5b8d47 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -19,6 +19,7 @@ mod aggregate; mod filter; mod joins; mod limit; +mod panes; mod projection; mod sort; mod source; @@ -28,6 +29,14 @@ pub use sort::SortKey; #[derive(Clone)] enum Kind { Source(Vec), + PaneInput { + coordinate: usize, + layout: planner_types::post_asap::PaneLayout, + offset_ms: Option, + }, + ScopeTimestamp { + columns: Vec>, + }, Union, VectorToScalar { column: usize, @@ -199,7 +208,14 @@ impl PhysicalOperator for Operator { }; PlanProperties { boundedness, - emission: if self.requires_bounded_input() { + emission: if matches!( + self.kind, + Kind::PaneInput { .. } | Kind::ScopeTimestamp { .. } + ) { + inputs + .first() + .map_or(Emission::Unknown, |input| input.emission) + } else if self.requires_bounded_input() { Emission::AfterInput } else { Emission::Incremental @@ -210,6 +226,8 @@ impl PhysicalOperator for Operator { fn name(&self) -> &str { match self.kind { Kind::Source(_) => "Source", + Kind::PaneInput { .. } => "PaneInput", + Kind::ScopeTimestamp { .. } => "ScopeTimestamp", Kind::Union => "Union", Kind::VectorToScalar { .. } => "VectorToScalar", Kind::Project(_) => "Project", @@ -227,6 +245,7 @@ impl PhysicalOperator for Operator { } } fn validate_context(&self, context: &RunContext) -> Result<(), Error> { + panes::validate_context(self, context)?; self.readout_parameters(context).map(|_| ()) } fn input_schemas(&self) -> Vec { @@ -248,6 +267,9 @@ impl PhysicalOperator for Operator { source::execute(self, inputs, context) } Kind::Project(_) => projection::execute(self, inputs, context), + Kind::PaneInput { .. } | Kind::ScopeTimestamp { .. } => { + panes::execute(self, inputs, context) + } Kind::Filter(_) => filter::execute(self, inputs, context), Kind::Limit { .. } => limit::execute(self, inputs, context), Kind::Sort { .. } => sort::execute(self, inputs, context), diff --git a/crates/asap-physical-operators/src/operators/panes.rs b/crates/asap-physical-operators/src/operators/panes.rs new file mode 100644 index 00000000..4748db09 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/panes.rs @@ -0,0 +1,228 @@ +//! Run-scoped pane population checks and timestamp restoration after reduction. +use super::*; +use crate::runtime::Scope; +use planner_types::post_asap::{validate_pane_coverage, PaneLayout, WindowEdgeCoverage}; + +impl Operator { + pub(crate) fn pane_input( + input: Schema, + coordinate: usize, + layout: PaneLayout, + offset_ms: Option, + ) -> Result { + if plain(&input, coordinate)? != (&DataType::Timestamp, false) { + return Err(invalid("pane input requires a non-null timestamp")); + } + if layout.pane_width_ms > i64::MAX as u64 { + return Err(invalid("pane width exceeds timestamp range")); + } + validate_pane_coverage( + &layout, + layout.pane_origin_ms, + &WindowEdgeCoverage::PaneAligned, + ) + .map_err(|error| Error::Invalid(format!("invalid pane layout: {error:?}")))?; + if offset_ms.is_some_and(|offset| offset < 0) { + return Err(invalid("negative pane offset")); + } + Ok(Self { + kind: Kind::PaneInput { + coordinate, + layout, + offset_ms, + }, + inputs: vec![input.clone()], + output: input, + }) + } + + pub(crate) fn scope_timestamp(input: Schema, output: Schema) -> Result { + crate::values::validate_schema(&output)?; + let coordinate = output + .time_index + .ok_or_else(|| invalid("temporal output requires a time index"))?; + if plain(&output, coordinate)? != (&DataType::Timestamp, false) { + return Err(invalid("temporal output requires a non-null timestamp")); + } + let mut columns = Vec::new(); + let mut used = std::collections::BTreeSet::new(); + for (index, field) in output.fields.iter().enumerate() { + if index == coordinate { + columns.push(None); + continue; + } + let matches: Vec<_> = input + .fields + .iter() + .enumerate() + .filter(|(_, candidate)| { + candidate.dtype == field.dtype + && candidate.nullable == field.nullable + && (candidate.name == field.name + || !matches!(field.dtype, SummaryFamilyType::Plain(_))) + }) + .map(|(index, _)| index) + .collect(); + let [column] = matches.as_slice() else { + return Err(invalid("temporal output column missing or ambiguous")); + }; + if !used.insert(*column) { + return Err(invalid("temporal output repeats an input column")); + } + columns.push(Some(*column)); + } + if used.len() != input.fields.len() { + return Err(invalid("temporal output drops an input column")); + } + Ok(Self { + kind: Kind::ScopeTimestamp { columns }, + inputs: vec![input], + output, + }) + } +} + +pub(super) fn validate_context(operator: &Operator, context: &RunContext) -> Result<(), Error> { + let Kind::PaneInput { + layout, offset_ms, .. + } = &operator.kind + else { + return Ok(()); + }; + let end = match (&context.scope, offset_ms) { + ( + Scope::Ingestion { + window_start_ms, + window_end_ms, + .. + }, + None, + ) => { + if window_end_ms.checked_sub(*window_start_ms) != Some(layout.pane_width_ms as i64) { + return Err(invalid("maintenance run must cover exactly one pane")); + } + *window_end_ms + } + ( + Scope::Query { + evaluation_time_ms, .. + }, + Some(offset), + ) => { + let end = evaluation_time_ms + .checked_sub(*offset) + .ok_or_else(|| invalid("query pane timestamp overflows"))?; + end.checked_sub(layout.pane_width_ms as i64) + .ok_or_else(|| invalid("query pane start overflows"))?; + end + } + _ => return Err(invalid("pane operator received the wrong execution scope")), + }; + validate_pane_coverage(layout, Some(end), &WindowEdgeCoverage::PaneAligned).map_err(|error| { + Error::Invalid(format!( + "query requires aligned panes or boundary residuals: {error:?}" + )) + }) +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + validate_context(operator, &context)?; + let input = inputs.pop().ok_or_else(|| invalid("pane input missing"))?; + let output = operator.output.clone(); + let mut seen = std::collections::BTreeSet::new(); + let mut memory = context.reserve(0)?; + let mut key_bytes = 0; + Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + match &operator.kind { + Kind::PaneInput { + coordinate, + offset_ms, + .. + } => { + let groups: Vec<_> = output + .fields + .iter() + .enumerate() + .filter(|(index, field)| { + *index != *coordinate + && matches!(field.dtype, SummaryFamilyType::Plain(_)) + }) + .map(|(index, _)| index) + .collect(); + for row in batch.rows() { + let Value::Timestamp(timestamp) = row[*coordinate] else { + return Err(invalid("pane timestamp type mismatch")); + }; + match (&context.scope, offset_ms) { + ( + Scope::Ingestion { + window_start_ms, + window_end_ms, + .. + }, + None, + ) if timestamp > *window_start_ms && timestamp <= *window_end_ms => {} + ( + Scope::Query { + evaluation_time_ms, .. + }, + Some(offset), + ) if timestamp + == evaluation_time_ms + .checked_sub(*offset) + .ok_or_else(|| invalid("pane timestamp overflows"))? => + { + let key = group_key(row, &groups)?; + if seen.contains(&key) { + return Err(invalid("duplicate entity state within a pane")); + } + key_bytes += key.iter().map(Vec::len).sum::() + + key.len() * std::mem::size_of::>() + + 64; + memory.resize(key_bytes)?; + seen.insert(key); + } + _ => { + return Err(invalid("input population differs from required pane")) + } + } + } + Ok(batch.value().clone()) + } + Kind::ScopeTimestamp { columns } => { + let timestamp = match context.scope { + Scope::Ingestion { window_end_ms, .. } => window_end_ms, + Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + }; + let rows = batch + .rows() + .iter() + .map(|row| { + columns + .iter() + .map(|column| { + column.map_or(Value::Timestamp(timestamp), |column| { + row[column].clone() + }) + }) + .collect() + }) + .collect(); + Batch::try_new(output.clone(), rows) + } + _ => unreachable!(), + } + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 7b8b91fa..eff79f6a 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -34,6 +34,12 @@ pub use candidates::{ CandidateSelection, PhysicalCandidate, }; +mod temporal_panes; +pub use temporal_panes::{ + compile_temporal_pane_candidate, TemporalEntityIdentity, TemporalPaneCandidate, + TemporalPaneMaintenance, +}; + mod compiled; pub use compiled::{CompiledPhysicalDag, InputContract}; diff --git a/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs b/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs new file mode 100644 index 00000000..46218fd0 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs @@ -0,0 +1,332 @@ +//! Lower a selected temporal maintenance contract; deployment supplies readers. +use super::*; +use planner_types::post_asap::{ + EvaluationSchedule, OutputRepresentation, PaneLayout, SketchAlgorithm, + SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, + SummaryWindowFramework, +}; + +/// Resolved source identity, supplied with physical capability evidence. +/// A schemaless PromQL projection cannot establish the complete label set. +#[derive(Clone, Debug)] +pub enum TemporalEntityIdentity { + /// The input resolver guarantees that the slot contains one entity. + SingleEntity, + /// All entity keys are represented by these columns; there are no hidden + /// labels distinguishing two rows with the same key. + Columns(Vec), +} + +/// Planner-selected lifecycle/window requirements for one temporal producer. +/// Pane geometry is semantic input, not a storage identity or scheduling policy. +#[derive(Clone, Debug)] +pub struct TemporalPaneMaintenance { + pub summary_node: NodeId, + pub lifecycle: SummaryMaintenanceLifecycleGuarantee, + pub framework: SummaryWindowFramework, + pub layout: PaneLayout, + pub entity_identity: TemporalEntityIdentity, +} + +/// Generated maintenance and query computation. `pane_inputs` is ordered from +/// the oldest complete pane to the newest; each run checks actual timestamps. +#[derive(Clone)] +pub struct TemporalPaneCandidate { + pub physical: PhysicalCandidate, + pub maintenance: TemporalPaneMaintenance, + pub pane_inputs: Vec, + pub merged_state: NodeId, + pub window_width_ms: u64, +} + +/// Compile bounded pane construction and a shared pane merge for temporal KLL +/// quantile roots. The selected contract remains authoritative; unsupported +/// lifecycle/framework/operator shapes fail rather than being substituted. +/// This initial realization consumes complete pane populations and emits full +/// state snapshots. Cross-run delta accumulation belongs to other candidates. +pub fn compile_temporal_pane_candidate( + dag: &ExecutableDag, + inputs: BTreeMap, + roots: &[NodeId], + maintenance: &TemporalPaneMaintenance, +) -> Result { + dag.validate().map_err(|error| invalid(error.to_string()))?; + if maintenance.lifecycle.summary_maintenance_lifecycle + != SummaryMaintenanceLifecycle::ContinuouslyMaintained + || maintenance.lifecycle.summary_maintenance_mode != SummaryMaintenanceMode::Incremental + || maintenance.lifecycle.evaluation_schedule != EvaluationSchedule::PerUpdate + || maintenance.lifecycle.output_representation != OutputRepresentation::SummaryState + { + return Err(invalid( + "pane candidate requires continuous incremental summary maintenance", + )); + } + let build = dag + .nodes + .iter() + .find(|node| u64::from(node.id.0) == maintenance.summary_node) + .ok_or_else(|| invalid("unknown maintained producer"))?; + let Payload::SummaryAgg { + family, + input: update, + reduction: PlannerReduction::PerEntity, + grouping, + } = &build.payload + else { + return Err(invalid( + "pane candidate requires a temporal per-entity summary", + )); + }; + if !matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll) + || update.item.is_some() + { + return Err(invalid("pane candidate supports unkeyed temporal KLL only")); + } + crate::capability::validate_summary_kernel(family, update, grouping).map_err(Error::Invalid)?; + let dependencies: Vec<_> = dag + .edges + .iter() + .filter(|edge| edge.consumer == build.id) + .map(|edge| edge.producer) + .collect(); + let [raw_id] = dependencies.as_slice() else { + return Err(invalid("temporal producer requires one raw input")); + }; + let raw = dag + .nodes + .iter() + .find(|node| node.id == *raw_id) + .ok_or_else(|| invalid("missing raw input"))?; + let Payload::Fallback { + expression: QueryExpr::TimeRange { range, child }, + } = &raw.payload + else { + return Err(invalid( + "temporal producer requires an explicit logical time range", + )); + }; + let QueryExpr::Scan { predicates, .. } = child.as_ref() else { + return Err(invalid("temporal pane source requires a raw scan")); + }; + let window_width_ms: u64 = range + .as_millis() + .try_into() + .map_err(|_| invalid("temporal window overflows"))?; + if window_width_ms == 0 + || window_width_ms > i64::MAX as u64 + || range.subsec_nanos() % 1_000_000 != 0 + { + return Err(invalid( + "temporal window requires positive integral milliseconds", + )); + } + let width = maintenance.layout.pane_width_ms; + if width == 0 || width > window_width_ms || !window_width_ms.is_multiple_of(width) { + return Err(invalid("temporal window must contain whole panes")); + } + match maintenance.framework { + SummaryWindowFramework::Sliding => {} + SummaryWindowFramework::Tumbling if width == window_width_ms => {} + _ => return Err(invalid("unsupported temporal window realization")), + } + let count = window_width_ms / width; + if count > 4096 { + return Err(invalid("temporal pane candidate exceeds input budget")); + } + let raw_id = u64::from(raw_id.0); + if inputs.len() != 1 { + return Err(invalid( + "pane candidate requires exactly its raw input contract", + )); + } + let contract = inputs + .get(&raw_id) + .ok_or_else(|| invalid("missing raw input contract"))?; + let raw_schema = Arc::new(raw.output_schema.clone()); + if contract.schema != raw_schema || contract.properties.boundedness != Boundedness::Bounded { + return Err(invalid("pane source requires its declared bounded schema")); + } + let coordinate = raw_schema + .time_index + .ok_or_else(|| invalid("temporal source requires a time index"))?; + let SummaryInputExpr::Column(value) = &update.weight else { + return Err(invalid("pane builder requires a value column")); + }; + let value = named_column(&raw_schema, value)?; + let groups: Vec<_> = (0..raw_schema.fields.len()) + .filter(|&index| index != coordinate && index != value) + .collect(); + match &maintenance.entity_identity { + TemporalEntityIdentity::SingleEntity if groups.is_empty() => {} + TemporalEntityIdentity::Columns(columns) + if !columns.is_empty() + && columns.len() == columns.iter().collect::>().len() + && columns.iter().copied().collect::>() + == groups.iter().copied().collect() => {} + _ => { + return Err(invalid( + "pane input requires its complete resolved entity identity", + )) + } + } + let mut next = dag + .nodes + .iter() + .map(|node| u64::from(node.id.0)) + .max() + .unwrap_or(0) + + 1; + let mut allocate = || { + let id = next; + next += 1; + id + }; + let mut operators = BTreeMap::new(); + let guard = allocate(); + operators.insert( + guard, + ( + vec![raw_id], + Operator::pane_input( + raw_schema.clone(), + coordinate, + maintenance.layout.clone(), + None, + )?, + ), + ); + let mut previous = guard; + for predicate in predicates { + let id = allocate(); + operators.insert( + id, + ( + vec![previous], + Operator::filter(raw_schema.clone(), expression(&predicate.0, &raw_schema)?)?, + ), + ); + previous = id; + } + let native = + Operator::summary_build(raw_schema, family.clone(), value, Some(coordinate), groups)?; + let compact_state = native.schema(); + let native_id = allocate(); + operators.insert(native_id, (vec![previous], native)); + let state_schema = Arc::new(build.output_schema.clone()); + let pane_output = allocate(); + operators.insert( + pane_output, + ( + vec![native_id], + Operator::scope_timestamp(compact_state, state_schema.clone())?, + ), + ); + let precompute = CompiledPhysicalDag::from_operators(inputs, operators, vec![pane_output])?; + let state_coordinate = state_schema + .time_index + .ok_or_else(|| invalid("pane state requires a time index"))?; + let state_column = summary_column(&state_schema)?; + let mut query_inputs = BTreeMap::new(); + let mut operators = BTreeMap::new(); + let mut pane_inputs = Vec::new(); + let mut guarded_inputs = Vec::new(); + for pane in 0..count { + let input = allocate(); + let guard = allocate(); + query_inputs.insert(input, InputContract::bounded(state_schema.clone())); + let offset = ((count - 1 - pane) * width) as i64; + operators.insert( + guard, + ( + vec![input], + Operator::pane_input( + state_schema.clone(), + state_coordinate, + maintenance.layout.clone(), + Some(offset), + )?, + ), + ); + pane_inputs.push(input); + guarded_inputs.push(guard); + } + let union = allocate(); + operators.insert( + union, + ( + guarded_inputs, + Operator::union(state_schema.clone(), count as usize)?, + ), + ); + let merge = Operator::summary_merge( + state_schema.clone(), + state_column, + (0..state_schema.fields.len()) + .filter(|&index| index != state_coordinate && index != state_column) + .collect(), + )?; + let merged_schema = merge.schema(); + let merged_state = allocate(); + operators.insert(merged_state, (vec![union], merge)); + if roots.is_empty() || roots.iter().copied().collect::>().len() != roots.len() { + return Err(invalid("temporal query requires distinct output roots")); + } + for &root in roots { + let node = dag + .nodes + .iter() + .find(|node| u64::from(node.id.0) == root) + .ok_or_else(|| invalid("unknown temporal output root"))?; + let Payload::SummaryEstimate { + query: SketchQuery::Quantile { q }, + } = node.payload + else { + return Err(invalid("temporal pane root must be a KLL quantile")); + }; + if !q.is_finite() || !(0. ..=1.).contains(&q) { + return Err(invalid("invalid temporal quantile")); + } + let dependencies: Vec<_> = dag + .edges + .iter() + .filter(|edge| edge.consumer == node.id) + .map(|edge| u64::from(edge.producer.0)) + .collect(); + if dependencies != [maintenance.summary_node] { + return Err(invalid( + "temporal readout must consume the maintained producer", + )); + } + let readout = Operator::readout( + merged_schema.clone(), + summary_column(&merged_schema)?, + crate::Statistic::Quantile, + std::collections::HashMap::from([("quantile".into(), q.to_string())]), + )?; + let readout_schema = readout.schema(); + let readout_id = allocate(); + operators.insert(readout_id, (vec![merged_state], readout)); + operators.insert( + root, + ( + vec![readout_id], + Operator::scope_timestamp(readout_schema, Arc::new(node.output_schema.clone()))?, + ), + ); + } + let query = CompiledPhysicalDag::from_operators(query_inputs, operators, roots.to_vec())?; + let mut output = precompute.output_contract(pane_output)?; + // Persisted readers have independent timing from the blocking builder. + output.properties.emission = Emission::Unknown; + Ok(TemporalPaneCandidate { + physical: PhysicalCandidate { + precompute: Some(precompute), + query, + materialized_outputs: BTreeMap::from([(pane_output, output)]), + }, + maintenance: maintenance.clone(), + pane_inputs, + merged_state, + window_width_ms, + }) +} diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index b2a5ede1..61003a50 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -189,6 +189,21 @@ fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() fn selected_plan( workload: &PlanningWorkload, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + selected_plan_with_model(workload, &FullyCostedRuntime) +} + +fn selected_plan_with_model( + workload: &PlanningWorkload, + model: &dyn CostModel, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + selected_plan_with_horizon(workload, model, Horizon(100.)) +} + +fn selected_plan_with_horizon( + workload: &PlanningWorkload, + model: &dyn CostModel, + horizon: Horizon, ) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { workload.validate().unwrap(); @@ -198,7 +213,7 @@ fn selected_plan( .next() .expect("one normalized workload entry"); let root = Rc::new(lowered); - let strategies = asap_aware_mapping::default_strategies_with(&FullyCostedRuntime); + let strategies = asap_aware_mapping::default_strategies_with(model); let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); let target = Rc::clone(&space.roots[0].1); let capabilities = SummaryMaintenanceLifecycleCapabilities { @@ -216,9 +231,9 @@ fn selected_plan( entry_indices: &[1], }, NOW_MS, - Some(Horizon(100.0)), + Some(horizon), capabilities, - &FullyCostedRuntime, + model, ) .unwrap(); assemble_selected_dag_with_summary_maintenance_lifecycles( @@ -230,9 +245,9 @@ fn selected_plan( &[1], ), NOW_MS, - Some(Horizon(100.0)), + Some(horizon), capabilities, - &FullyCostedRuntime, + model, ) .unwrap() .expect("selected summary plan") @@ -411,3 +426,466 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { ); } } + +struct SlidingPaneModel; +impl CostModel for SlidingPaneModel { + fn raw_query_recompute_total_cost( + &self, + target: &asap_types::pre_asap::QueryExpr, + reads: f64, + ) -> Option { + let _ = (target, reads); + Some(Cost(100_000.0)) + } + fn rank_candidates( + &self, + intent: &AggIntent, + candidates: &[asap_types::post_asap::SketchAlgorithm], + ) -> Vec { + FullyCostedRuntime.rank_candidates(intent, candidates) + } + fn summary_maintenance_lifecycle_cost_inputs( + &self, + summary: &SummaryNode, + ) -> SummaryMaintenanceLifecycleCostInputs { + let mut costs = FullyCostedRuntime.summary_maintenance_lifecycle_cost_inputs(summary); + // Controlled workload evidence makes repeated raw construction more + // expensive than retaining and updating the same temporal population. + costs.build_cost = Some(Cost(1000.)); + costs + } + fn summary_maintenance_capabilities( + &self, + summary: &SummaryNode, + ) -> SummaryMaintenanceCapabilities { + FullyCostedRuntime.summary_maintenance_capabilities(summary) + } + fn complete_summary_candidate_estimate( + &self, + _root: &SummaryNode, + _target: Option<&asap_types::pre_asap::QueryExpr>, + deployments: &[asap_aware_mapping::cost_model::CostedSummaryDeployment<'_>], + _horizon: Option, + _reads: Option, + _accuracy: &[AccuracyTarget], + ) -> Option { + Some(asap_aware_mapping::CompleteSummaryCandidateEstimate { + cost: Cost( + deployments + .iter() + .map(|deployment| deployment.selected_cost.0) + .sum(), + ), + physical_plan_id: Some("bounded-sliding-pane-evidence".into()), + window_frameworks: deployments + .iter() + .map(|deployment| { + (deployment.guarantee.summary_maintenance_lifecycle + == SummaryMaintenanceLifecycle::ContinuouslyMaintained) + .then_some(asap_types::post_asap::SummaryWindowFramework::Sliding) + }) + .collect(), + window_accuracy_guarantee: None, + }) + } +} + +/// Workload and optimizer-selected lifecycle generate both physical DAGs. +/// No computational operators or graph edges are constructed by this fixture. +#[test] +fn selected_temporal_lifecycle_compiles_panes_and_executes() { + use asap_physical_operators::{ + operators::Operator, + physical_planner::{ + compile_temporal_pane_candidate, InputContract, Source, TemporalEntityIdentity, + TemporalPaneMaintenance, + }, + runtime::{Limits, RunContext, Scope}, + summary_kernels::datasketches_kll::DatasketchesKLLAccumulator, + values::{Batch, Value}, + }; + use asap_types::{ + post_asap::{ + compile_executable_dag, plan_pane_phase, ExecutableOperatorPayload, SummaryFamilyType, + SummaryWindowFramework, + }, + pre_asap::DataType, + workload::TimestampMs, + }; + use futures::{executor::block_on, StreamExt}; + use std::{collections::BTreeMap, sync::Arc}; + for quantile in [0.5, 0.99] { + let mut workload = dashboard_workload(); + let query = Query(format!( + "quantile_over_time({quantile}, latency{{job=\"api\"}}[5m])" + )); + workload.query_workload.query_batch.as_mut().unwrap()[0].query = query.clone(); + workload.query_workload.repeating_queries.as_mut().unwrap()[0].query = query; + + workload + .data_workload + .as_mut() + .unwrap() + .data_ingestion_interval + .value = Some(DurationMs(60_000)); + workload.query_workload.repeating_queries.as_mut().unwrap()[0].demand = + RepeatedDemand::FixedIntervalAt { + interval: RepetitionInterval(60_000), + evaluation_phase: TimestampMs(300_000), + }; + let plan = selected_plan_with_horizon(&workload, &SlidingPaneModel, Horizon(1000.)); + assert!(!plan.selected_raw_recompute); + assert_eq!(plan.deployments.len(), 1); + let deployment = &plan.deployments[0]; + assert_eq!( + deployment.selected_window_framework, + Some(SummaryWindowFramework::Sliding) + ); + let dag = compile_executable_dag(&plan.root).unwrap(); + let build = dag + .nodes + .iter() + .find(|node| matches!(node.payload, ExecutableOperatorPayload::SummaryAgg { .. })) + .unwrap(); + assert_eq!(build.id, deployment.post_asap_node_id); + let raw = dag + .nodes + .iter() + .find(|node| matches!(node.payload, ExecutableOperatorPayload::Fallback { .. })) + .unwrap(); + let schema = Arc::new(raw.output_schema.clone()); + let width = workload + .data_workload + .as_ref() + .unwrap() + .data_ingestion_interval + .value + .unwrap() + .0; + let layout = plan_pane_phase( + &workload.query_workload.repeating_queries.as_ref().unwrap()[0].demand, + width, + ) + .unwrap(); + let maintenance = TemporalPaneMaintenance { + summary_node: u64::from(build.id.0), + lifecycle: deployment + .summary_maintenance_lifecycle_guarantee + .clone() + .unwrap(), + framework: deployment.selected_window_framework.clone().unwrap(), + layout, + // The memory source has exactly the declared label columns; a + // schemaless deployment must resolve all entity keys first. + entity_identity: TemporalEntityIdentity::Columns( + schema + .fields + .iter() + .enumerate() + .filter(|(_, field)| field.name == "job") + .map(|(index, _)| index) + .collect(), + ), + }; + let candidate = compile_temporal_pane_candidate( + &dag, + BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &maintenance, + ) + .unwrap(); + assert_eq!(candidate.window_width_ms, 300_000); + assert_eq!(candidate.pane_inputs.len(), 5); + assert_ne!( + candidate.physical.precompute.as_ref().unwrap().roots()[0], + maintenance.summary_node, + "a one-minute pane is not the logical five-minute summary output" + ); + let source_opens = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let mut stored = Vec::new(); + for pane in 0..6 { + let rows = (0..20) + .flat_map(|sample| { + ["api", "batch"] + .into_iter() + .map(move |entity| (sample, entity)) + }) + .map(|(sample, entity)| { + schema + .fields + .iter() + .map(|field| match field.dtype { + SummaryFamilyType::Plain(DataType::Timestamp) => { + Value::Timestamp(pane * 60_000 + (sample + 1) * 3000) + } + SummaryFamilyType::Plain(DataType::Float64) => Value::Float64( + (pane * 20 + sample) as f64 + + if entity == "batch" { 100_000. } else { 0. }, + ), + SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8(entity.into()), + _ => panic!("unexpected raw field {field:?}"), + }) + .collect() + }) + .collect(); + let result = physical_common::execute( + candidate.physical.precompute.as_ref().unwrap(), + BTreeMap::from([( + u64::from(raw.id.0), + Batch::try_new(schema.clone(), rows).unwrap(), + )]), + Scope::Ingestion { + window_start_ms: pane * 60_000, + window_end_ms: (pane + 1) * 60_000, + revision: 1, + }, + ); + let batch = &result[0][0]; + let rows = batch + .rows() + .iter() + .map(|row| { + row.iter() + .map(|value| match value { + Value::Summary { family, state } => Value::Summary { + family: family.clone(), + state: Arc::new( + DatasketchesKLLAccumulator::from_msgpack_bytes( + &state.serialize_to_bytes(), + ) + .unwrap(), + ), + }, + value => value.clone(), + }) + .collect() + }) + .collect(); + stored.push(Batch::try_new(batch.schema().clone(), rows).unwrap()); + } + for offset in [0, 1] { + let inputs: BTreeMap<_, _> = candidate + .pane_inputs + .iter() + .enumerate() + .map(|(index, &id)| (id, stored[offset + index].clone())) + .collect(); + let scope = Scope::Query { + evaluation_time_ms: (5 + offset as i64) * 60_000, + revision: 1, + }; + let result = + physical_common::execute(&candidate.physical.query, inputs.clone(), scope.clone()); + let rows = result[0][0].rows(); + assert_eq!(rows.len(), 1); + assert!(rows[0] + .iter() + .any(|value| matches!(value, Value::Utf8(label) if label.as_ref() == "api"))); + let Value::Float64(value) = rows[0][1] else { + panic!("missing p99") + }; + assert!( + (value - ((if quantile == 0.5 { 50 } else { 99 }) + offset * 20) as f64).abs() + <= 1. + ); + assert!( + matches!(rows[0][0], Value::Timestamp(timestamp) if timestamp == (5 + offset as i64) * 60_000) + ); + let sources = |inputs: BTreeMap| -> BTreeMap> { + inputs + .into_iter() + .map(|(id, batch)| { + ( + id, + Box::new(TemporalCountingSource { + operator: Operator::source(batch.schema().clone(), vec![batch]) + .unwrap(), + opens: source_opens.clone(), + }) as Source<'static>, + ) + }) + .collect() + }; + let bound = candidate + .physical + .query + .instantiate(sources(inputs.clone())) + .unwrap(); + let population = block_on( + bound + .execute( + &[candidate.merged_state], + RunContext::new(scope.clone(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0) + .collect::>(), + ); + let Value::Summary { state, .. } = population[0].as_ref().unwrap().rows()[0] + .iter() + .find(|value| matches!(value, Value::Summary { .. })) + .unwrap() + else { + panic!("missing merged state") + }; + assert_eq!( + state + .as_any() + .downcast_ref::() + .unwrap() + .inner + .count(), + 100 + ); + let mut missing = inputs.clone(); + missing.remove(&candidate.pane_inputs[0]); + assert!(candidate + .physical + .query + .instantiate(sources(missing)) + .is_err()); + let mut duplicate = inputs.clone(); + duplicate.insert(candidate.pane_inputs[1], stored[offset].clone()); + let bad = candidate + .physical + .query + .instantiate(sources(duplicate)) + .unwrap(); + let errors = block_on( + bad.execute( + candidate.physical.query.roots(), + RunContext::new(scope.clone(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0) + .collect::>(), + ); + assert!( + errors.iter().any(Result::is_err), + "duplicate pane must not be merged twice" + ); + let mut duplicate_entity = inputs.clone(); + let pane = &stored[offset]; + duplicate_entity.insert( + candidate.pane_inputs[0], + Batch::try_new( + pane.schema().clone(), + vec![pane.rows()[0].clone(), pane.rows()[0].clone()], + ) + .unwrap(), + ); + let bad = candidate + .physical + .query + .instantiate(sources(duplicate_entity)) + .unwrap(); + let errors = block_on( + bad.execute( + candidate.physical.query.roots(), + RunContext::new(scope.clone(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0) + .collect::>(), + ); + assert!( + errors.iter().any(Result::is_err), + "duplicate snapshots within a pane must fail" + ); + let bound = candidate + .physical + .query + .instantiate(sources(inputs)) + .unwrap(); + source_opens.store(0, std::sync::atomic::Ordering::SeqCst); + assert!(bound + .execute( + candidate.physical.query.roots(), + RunContext::new( + Scope::Query { + evaluation_time_ms: 330_000, + revision: 1 + }, + Limits::default() + ) + .unwrap() + ) + .is_err()); + assert_eq!( + source_opens.load(std::sync::atomic::Ordering::SeqCst), + 0, + "invalid phase must fail before opening readers" + ); + } + // A selected framework cannot be silently replaced by physical planning. + let mut wrong_framework = maintenance.clone(); + wrong_framework.framework = SummaryWindowFramework::ExponentialHistogram; + let mut wrong_identity = maintenance.clone(); + wrong_identity.entity_identity = TemporalEntityIdentity::SingleEntity; + let mut unknown_phase = maintenance.clone(); + unknown_phase.layout.pane_origin_ms = None; + let mut partial_panes = maintenance.clone(); + partial_panes.layout.pane_width_ms = 90_000; + let mut wrong_lifecycle = maintenance.clone(); + wrong_lifecycle.lifecycle.summary_maintenance_lifecycle = + SummaryMaintenanceLifecycle::Ephemeral; + for unsupported in [ + wrong_framework, + wrong_identity, + unknown_phase, + partial_panes, + wrong_lifecycle, + ] { + assert!(compile_temporal_pane_candidate( + &dag, + BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &unsupported + ) + .is_err()); + } + } +} + +struct TemporalCountingSource { + operator: asap_physical_operators::operators::Operator, + opens: std::sync::Arc, +} +impl + asap_physical_operators::plan::PhysicalOperator< + asap_physical_operators::values::Batch, + asap_physical_operators::values::Schema, + > for TemporalCountingSource +{ + fn name(&self) -> &str { + "TemporalCountingSource" + } + fn properties( + &self, + inputs: &[asap_physical_operators::plan::PlanProperties], + ) -> asap_physical_operators::plan::PlanProperties { + self.operator.properties(inputs) + } + fn input_schemas(&self) -> Vec { + self.operator.input_schemas() + } + fn output_schema(&self) -> asap_physical_operators::values::Schema { + self.operator.output_schema() + } + fn output_bytes(&self, batch: &asap_physical_operators::values::Batch) -> usize { + batch.bytes() + } + fn start<'a>( + &'a self, + inputs: Vec< + asap_physical_operators::runtime::Input<'a, asap_physical_operators::values::Batch>, + >, + context: asap_physical_operators::runtime::RunContext, + ) -> Result< + asap_physical_operators::runtime::OutputStream<'a, asap_physical_operators::values::Batch>, + asap_physical_operators::Error, + > { + self.opens.fetch_add(1, std::sync::atomic::Ordering::SeqCst); + self.operator.start(inputs, context) + } +} diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index c98fe47c..f7f2ffe2 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -372,19 +372,33 @@ when the shared runtime merges panes for a query. ## 6. Executable acceptance coverage -The tests distinguish optimizer-selected lifecycle execution from explicit -physical pane construction: +The tests cover optimizer-selected lifecycle execution and automatic temporal +pane compilation, alongside independent operator/runtime fixtures: | Test | Contract exercised | | --- | --- | +| `summary_maintenance_lifecycle_e2e::selected_temporal_lifecycle_compiles_panes_and_executes` | PromQL p50/p99 workloads → selected continuous lifecycle and Sliding framework → automatically generated maintenance/query DAGs → real codec round-trip → adjacent aligned windows; checks filters, entity identity, sample counts, missing/duplicate panes and phase rejection before opening readers | | `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled maintenance/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate returns the same population | | `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute maintenance DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | | `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | | `precompute_candidates::grouped_rate_can_be_materialized_before_or_after_grouped_sum` | Cost changes select different legal precompute frontiers; both selected candidates execute with the same reset-sensitive result; uncompilable candidates are not priced | | `sql_to_physical::sql_filter_grouped_sum_executes_and_rebinds` | SQL text → candidate search → physical compilation → shared Scan predicates and grouped summary execution; NULL samples are ignored and fresh bindings produce new results | -The pane test uses an explicit physical realization. It does not establish that -maintenance selection automatically emits the complete temporal pane DAG. -Pane phase validation uses the Planner coverage contract; concrete stored-pane -identity, revision, readiness and population coverage remain deployment checks. -Real storage and HTTP execution belong to deployment-repository E2E tests. +`physical_planner::compile_temporal_pane_candidate` consumes the logical DAG, +selected lifecycle/framework and a generic pane/entity input contract. It +generates pane construction, scan predicates, ordered state slots, a shared +merge, quantile readouts and run-scoped timestamps. A physical pane output has +its own identity: one minute of state cannot masquerade as the logical +five-minute summary. The returned candidate retains the maintenance contract. + +This initial realization supports bounded, complete KLL panes with known phase +and resolved entity identity, for Sliding windows or a single Tumbling window. +Source capability evidence must declare all entity keys or isolate one entity; +usage-derived PromQL columns alone cannot establish that identity. Partial edge +panes, exponential histograms and cross-run delta accumulation require further +physical candidates and are rejected by this entry point. + +Physical execution checks pane timestamps and duplicate entity states. Concrete +stored identity, revisions, readiness and completeness evidence remain +deployment responsibilities. Real storage and HTTP execution belong to +deployment-repository E2E tests. From 62bdadf9c767076c3c7474829a1522d08377035c Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 14:03:17 +0000 Subject: [PATCH 49/90] docs: clarify summary source grouping and pane coverage --- .../physical-planning-and-deployment.md | 49 ++++++++----------- 1 file changed, 20 insertions(+), 29 deletions(-) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index f7f2ffe2..e59669ff 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -30,12 +30,17 @@ associated with the logical DAG, not a separate computation IR. ### Running example -Suppose p50 and p99 are requested over the same five-minute latency population, +Suppose p50 and p99 are requested over the same latency samples in a five-minute window, and ASAP selects KLL with `k=200`. Assume query windows align with one-minute pane boundaries and that the selected parameters satisfy the required guarantees. Operator names below are illustrative; the example defines the design, not a claim that the entire deployment integration is implemented. +The data source identifies where samples come from. Filters, grouping and the +window determine which samples enter each summary. Here `pane_duration: 1m` +means each stored pane covers one minute; the query range is five minutes. +Neither duration specifies how often maintenance runs or how long state is kept. + The example evolves through the architecture as follows: ```text @@ -188,7 +193,7 @@ KllStateOutput(k=200) ``` This DAG implements construction of each maintained one-minute pane. Its input -contract requires the complete pane population; the deployment supplies that +contract requires all input samples matching the source, filters and group within that pane; the deployment supplies that bounded input from its source integration. ### Query Physical DAG @@ -262,7 +267,7 @@ frontiers and cost evidence, including updates, retention, recurrence and sharin `enumerate_frontiers` constructs bounded, reachable antichain frontiers above explicit input boundaries, including query-only and fully precomputed results. It fails explicitly when the candidate budget is exceeded. Maintenance selection must still reject frontiers that violate window, freshness, or reuse requirements; deployment feasibility is checked before pricing. Physical compilation opens no readers. Bounded precompute outputs become typed -query inputs. Their build window, evaluation time, population, readiness and +query inputs. Their source, filters, grouping, build window, evaluation time, readiness and revision contracts must accompany the selected lifecycle and be checked during deployment binding. Type compatibility alone does not establish reuse legality. @@ -323,7 +328,7 @@ InputSlot[5 panes] The Deployment Plan Compiler establishes bindings and checks that their contracts satisfy the physical inputs and selected lifecycle, including KLL parameters, -grouping, population, window coverage and revision scope. The deployment engine +source, filters, grouping, window coverage and revision scope. The deployment engine resolves request-specific states and checks their actual coverage, revisions and readiness at execution time. A compiled plan cannot establish future readiness. @@ -366,39 +371,25 @@ shared physical operator implementation library, `asap-physical-operators`, and its DAG runtime. The merge executes once per run for both consumers. Execution does not introduce additional planning decisions. -Each maintained pane contributes its finalized population once. A replacement -snapshot updates that pane's state; it does not introduce another population -when the shared runtime merges panes for a query. +Each maintained pane contributes its input samples once. A replacement snapshot +replaces that pane's state; query merging must not count both the old and new +snapshots as separate inputs. ## 6. Executable acceptance coverage -The tests cover optimizer-selected lifecycle execution and automatic temporal -pane compilation, alongside independent operator/runtime fixtures: +The tests distinguish optimizer-selected lifecycle execution from explicit +physical pane construction: | Test | Contract exercised | | --- | --- | -| `summary_maintenance_lifecycle_e2e::selected_temporal_lifecycle_compiles_panes_and_executes` | PromQL p50/p99 workloads → selected continuous lifecycle and Sliding framework → automatically generated maintenance/query DAGs → real codec round-trip → adjacent aligned windows; checks filters, entity identity, sample counts, missing/duplicate panes and phase rejection before opening readers | -| `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled maintenance/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate returns the same population | +| `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled maintenance/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | | `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute maintenance DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | | `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | | `precompute_candidates::grouped_rate_can_be_materialized_before_or_after_grouped_sum` | Cost changes select different legal precompute frontiers; both selected candidates execute with the same reset-sensitive result; uncompilable candidates are not priced | | `sql_to_physical::sql_filter_grouped_sum_executes_and_rebinds` | SQL text → candidate search → physical compilation → shared Scan predicates and grouped summary execution; NULL samples are ignored and fresh bindings produce new results | -`physical_planner::compile_temporal_pane_candidate` consumes the logical DAG, -selected lifecycle/framework and a generic pane/entity input contract. It -generates pane construction, scan predicates, ordered state slots, a shared -merge, quantile readouts and run-scoped timestamps. A physical pane output has -its own identity: one minute of state cannot masquerade as the logical -five-minute summary. The returned candidate retains the maintenance contract. - -This initial realization supports bounded, complete KLL panes with known phase -and resolved entity identity, for Sliding windows or a single Tumbling window. -Source capability evidence must declare all entity keys or isolate one entity; -usage-derived PromQL columns alone cannot establish that identity. Partial edge -panes, exponential histograms and cross-run delta accumulation require further -physical candidates and are rejected by this entry point. - -Physical execution checks pane timestamps and duplicate entity states. Concrete -stored identity, revisions, readiness and completeness evidence remain -deployment responsibilities. Real storage and HTTP execution belong to -deployment-repository E2E tests. +The pane test uses an explicit physical realization. It does not establish that +maintenance selection automatically emits the complete temporal pane DAG. +Pane phase validation uses the Planner coverage contract; concrete stored-pane +identity, revision, readiness and complete coverage of the required input samples remain deployment checks. +Real storage and HTTP execution belong to deployment-repository E2E tests. From fa2fd23e8cacc1fe7fabdd09cfa70d8a11ed15e3 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 14:20:13 +0000 Subject: [PATCH 50/90] docs: distinguish input scope from complete summary semantics --- .../physical-planning-and-deployment.md | 35 +++++++++++++++++++ 1 file changed, 35 insertions(+) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index e59669ff..82969e3e 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -28,6 +28,41 @@ implementation library. Deployment systems such as ASAPQuery and asap-fusion own deployment compilation and operation. The lifecycle is a planning contract associated with the logical DAG, not a separate computation IR. +### Input semantics and summary semantics + +`source`, `filter`, `grouping` and `window` describe input-data semantics: +where records originate, which records qualify, how they are grouped and which +time interval applies. They are not a complete description of arbitrary summary +computation. In particular, the same four fields can summarize different value +expressions or produce different states. + +| Concern | Required semantic information | +| --- | --- | +| Input computation | Source identities and schemas, filters, joins/transforms and their order, or a reference to the canonical input sub-DAG | +| Values and grouping | Value expressions, item identities and weights where applicable, group keys and types, and operation-defined null/duplicate handling | +| Time | Time column and interpretation, interval bounds, evaluation alignment, and distinction between query range and maintained panes | +| Summary computation | Exact operation or sketch family, algorithm and parameters, and supported build/merge behavior | +| Output | State versus finalized value, output schema/type, and readout parameters when part of the output computation | + +For example, KLL over `latency_seconds` and KLL over `log(latency_seconds)` differ +even with identical source, filter, grouping and window. Likewise, weighted +frequency state needs both item and weight expressions. More complex inputs +must retain their computation DAG; four descriptive fields cannot replace it. + +The canonical selected computation is authoritative. These categories describe +what must be preserved, not a new flat IR or a second expression language. +Operator-defined behavior should be referenced through its canonical contract, +not independently configured in deployment metadata. Unsupported or unresolved +semantics cannot be treated as compatible. + +Logical planning defines the semantics; physical compilation realizes them as +operators and typed boundaries. Deployment binds concrete readers and state +records that satisfy those requirements. A stored summary definition records or +references the relevant semantics for compatibility checks. Matching a definition +alone does not establish actual window coverage, revision compatibility or +readiness; those require runtime checks. Physical location, encoding, scheduling +and retention are separate execution/deployment contracts. + ### Running example Suppose p50 and p99 are requested over the same latency samples in a five-minute window, From 880abe49b3f6746b5e04f2dc41bb9faa0e62a855 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sat, 26 Sep 2026 17:10:54 +0000 Subject: [PATCH 51/90] feat(types): export versioned summary semantic dependency closures --- Cargo.lock | 1 + crates/types/Cargo.toml | 1 + crates/types/src/post_asap/mod.rs | 2 + .../src/post_asap/semantic_definition.rs | 579 ++++++++++++++++++ .../physical-planning-and-deployment.md | 8 + 5 files changed, 591 insertions(+) create mode 100644 crates/types/src/post_asap/semantic_definition.rs diff --git a/Cargo.lock b/Cargo.lock index e7b6b6c6..f54e6055 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -408,6 +408,7 @@ version = "0.1.0" dependencies = [ "serde", "serde_json", + "sha2", "thiserror 2.0.18", ] diff --git a/crates/types/Cargo.toml b/crates/types/Cargo.toml index 6b353657..6cf3c2b3 100644 --- a/crates/types/Cargo.toml +++ b/crates/types/Cargo.toml @@ -17,3 +17,4 @@ edition = "2021" serde = { version = "1", features = ["derive", "rc"] } serde_json = "1" thiserror = "2" +sha2 = "0.10" diff --git a/crates/types/src/post_asap/mod.rs b/crates/types/src/post_asap/mod.rs index 6bf31742..038276ea 100644 --- a/crates/types/src/post_asap/mod.rs +++ b/crates/types/src/post_asap/mod.rs @@ -35,6 +35,8 @@ pub mod maintained_population; pub mod post_asap_dag; pub mod query_time; pub mod schema; +pub mod semantic_definition; +pub use semantic_definition::SummarySemanticFragment; pub mod sketch; pub mod summary_maintenance; pub mod summary_maintenance_lifecycle; diff --git a/crates/types/src/post_asap/semantic_definition.rs b/crates/types/src/post_asap/semantic_definition.rs new file mode 100644 index 00000000..ebe61560 --- /dev/null +++ b/crates/types/src/post_asap/semantic_definition.rs @@ -0,0 +1,579 @@ +//! Persistable dependency closure using Planner's typed operation vocabulary. +//! Node hashes are local semantic references, not executable or deployed IDs. +use crate::post_asap::{ + EdgeRole, ExecutableDag, ExecutableOperatorPayload, PostAsapNodeId, SummarySchema, +}; +use serde::{Deserialize, Serialize}; +use sha2::{Digest, Sha256}; +use std::collections::{BTreeMap, BTreeSet}; + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SummarySemanticFragment { + pub format_version: u32, + pub output: String, + pub nodes: BTreeMap, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SemanticOperation { + // Wire ownership must be Send + Sync. These values are checked against the + // Planner types on export and on recovery; arbitrary JSON is not accepted. + pub operation: serde_json::Value, + pub output_schema: serde_json::Value, + pub inputs: Vec, + /// The direct input range is supplied by the stored record, not by a query lookback. + #[serde(default, skip_serializing_if = "std::ops::Not::not")] + pub record_range: bool, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SemanticInput { + pub role: EdgeRole, + pub node: String, +} + +pub(crate) fn canonical_bytes(value: &impl Serialize) -> Result, String> { + fn canonical(value: serde_json::Value) -> serde_json::Value { + match value { + serde_json::Value::Object(values) => serde_json::Value::Object( + values + .into_iter() + .map(|(k, v)| (k, canonical(v))) + .collect::>() + .into_iter() + .collect(), + ), + serde_json::Value::Array(values) => { + serde_json::Value::Array(values.into_iter().map(canonical).collect()) + } + value => value, + } + } + serde_json::to_vec(&canonical( + serde_json::to_value(value).map_err(|e| e.to_string())?, + )) + .map_err(|e| e.to_string()) +} +fn hash(value: &impl Serialize) -> Result { + Ok(format!("{:x}", Sha256::digest(canonical_bytes(value)?))) +} +fn role(role: EdgeRole) -> u8 { + match role { + EdgeRole::Input => 0, + EdgeRole::Left => 1, + EdgeRole::Right => 2, + } +} + +impl SummarySemanticFragment { + pub fn from_stored_output(dag: &ExecutableDag, output: PostAsapNodeId) -> Result { + Self::export(dag, output, true) + } + + pub fn from_dag(dag: &ExecutableDag, output: PostAsapNodeId) -> Result { + Self::export(dag, output, false) + } + + fn export( + dag: &ExecutableDag, + output: PostAsapNodeId, + parameterize_range: bool, + ) -> Result { + let mut included = BTreeSet::new(); + let mut pending = vec![output]; + while let Some(id) = pending.pop() { + if included.insert(id) { + pending.extend( + dag.edges + .iter() + .filter(|e| e.consumer == id) + .map(|e| e.producer), + ); + } + } + let dag = ExecutableDag { + nodes: dag + .nodes + .iter() + .filter(|n| included.contains(&n.id)) + .cloned() + .collect(), + edges: dag + .edges + .iter() + .filter(|e| included.contains(&e.consumer)) + .cloned() + .collect(), + root: output, + }; + dag.validate().map_err(|e| e.to_string())?; + if dag.nodes.len() > 4096 { + return Err("semantic fragment exceeds node budget".into()); + } + // Open PromQL entities carry all labels. Nullable label columns demanded + // only by a downstream consumer do not change a per-entity scalar state. + let mut dag = dag; + let sample_only = matches!( + &dag.nodes + .iter() + .find(|n| n.id == output) + .ok_or("missing output")? + .payload, + ExecutableOperatorPayload::SummaryAgg { + reduction: crate::pre_asap::Reduction::PerEntity, + grouping: crate::post_asap::GroupingStrategy::PerSubpopulationInstance, + input: crate::post_asap::SummaryUpdate { + item: None, + weight: crate::post_asap::SummaryInputExpr::Column( + crate::pre_asap::ColumnRef::SampleValue + ), + .. + }, + .. + } + ); + if parameterize_range && sample_only { + let direct: BTreeSet<_> = dag + .edges + .iter() + .filter(|e| e.consumer == output) + .map(|e| e.producer) + .collect(); + let mut normalized = false; + for node in &mut dag.nodes { + if !direct.contains(&node.id) { + continue; + } + if let ExecutableOperatorPayload::Fallback { expression } = &mut node.payload { + let source = match expression { + crate::pre_asap::QueryExpr::TimeRange { child, .. } => { + std::rc::Rc::make_mut(child) + } + other => other, + }; + if let crate::pre_asap::QueryExpr::Scan { + source: crate::pre_asap::Source::TimeSeries { .. }, + predicates, + schema, + } = source + { + if !schema.closed + && predicates.is_empty() + && schema.unique_keys.is_empty() + && schema + .columns + .iter() + .take_while(|c| { + !(c.nullable && c.dtype == crate::pre_asap::DataType::Utf8) + }) + .count() + + schema + .columns + .iter() + .rev() + .take_while(|c| { + c.nullable && c.dtype == crate::pre_asap::DataType::Utf8 + }) + .count() + == schema.columns.len() + { + schema.columns.retain(|c| { + !(c.nullable && c.dtype == crate::pre_asap::DataType::Utf8) + }); + node.output_schema.fields.retain(|c| { + !(c.nullable + && c.dtype + == crate::post_asap::SummaryFamilyType::Plain( + crate::pre_asap::DataType::Utf8, + )) + }); + normalized = true; + } + } + } + } + if normalized { + dag.nodes + .iter_mut() + .find(|n| n.id == output) + .unwrap() + .output_schema + .fields + .retain(|c| { + !(c.nullable + && c.dtype + == crate::post_asap::SummaryFamilyType::Plain( + crate::pre_asap::DataType::Utf8, + )) + }); + } + } + let nodes: BTreeMap<_, _> = dag.nodes.iter().map(|n| (n.id, n)).collect(); + let mut hashes: BTreeMap = BTreeMap::new(); + let mut result = Self { + format_version: 1, + output: String::new(), + nodes: BTreeMap::new(), + }; + let mut stack = vec![(output, false)]; + while let Some((id, finish)) = stack.pop() { + if hashes.contains_key(&id) { + continue; + } + let node = nodes.get(&id).ok_or("missing semantic output")?; + let edges: Vec<_> = dag.edges.iter().filter(|e| e.consumer == id).collect(); + if !finish { + stack.push((id, true)); + for edge in &edges { + stack.push((edge.producer, false)); + } + continue; + } + let mut inputs = edges + .iter() + .map(|e| SemanticInput { + role: e.role, + node: hashes[&e.producer].clone(), + }) + .collect::>(); + inputs.sort_by(|a, b| (role(a.role), &a.node).cmp(&(role(b.role), &b.node))); + let mut payload = node.payload.clone(); + let mut record_range = false; + if parameterize_range + && matches!( + nodes[&output].payload, + ExecutableOperatorPayload::SummaryAgg { .. } + ) + && dag + .edges + .iter() + .any(|e| e.consumer == output && e.producer == id) + { + if let ExecutableOperatorPayload::Fallback { + expression: crate::pre_asap::QueryExpr::TimeRange { child, .. }, + } = &payload + { + payload = ExecutableOperatorPayload::Fallback { + expression: child.as_ref().clone(), + }; + record_range = true; + } + } + if let ExecutableOperatorPayload::RelationalJoin { pruning, .. } = &mut payload { + *pruning = None; + } + let operation = SemanticOperation { + record_range, + operation: serde_json::to_value(&payload).map_err(|e| e.to_string())?, + output_schema: serde_json::to_value(&node.output_schema) + .map_err(|e| e.to_string())?, + inputs, + }; + let key = hash(&operation)?; + result.nodes.insert(key.clone(), operation); + hashes.insert(id, key); + } + result.output = hashes + .remove(&output) + .ok_or("missing semantic output hash")?; + result.validate()?; + Ok(result) + } + + pub fn validate(&self) -> Result<(), String> { + if self.format_version != 1 + || self.nodes.is_empty() + || self.nodes.len() > 4096 + || canonical_bytes(self)?.len() > 4 * 1024 * 1024 + { + return Err("unsupported semantic fragment version or size".into()); + } + for (key, node) in &self.nodes { + let payload: ExecutableOperatorPayload = + serde_json::from_value(node.operation.clone()).map_err(|e| e.to_string())?; + if node.record_range { + let root = self + .nodes + .get(&self.output) + .ok_or("missing semantic root")?; + let root_payload: ExecutableOperatorPayload = + serde_json::from_value(root.operation.clone()).map_err(|e| e.to_string())?; + if !matches!(payload, ExecutableOperatorPayload::Fallback { .. }) + || !matches!(root_payload, ExecutableOperatorPayload::SummaryAgg { .. }) + || !root.inputs.iter().any(|input| &input.node == key) + { + return Err("record range must belong to a direct summary input".into()); + } + } + let _: SummarySchema = + serde_json::from_value(node.output_schema.clone()).map_err(|e| e.to_string())?; + if hash(node)? != *key + || node + .inputs + .iter() + .any(|i| !self.nodes.contains_key(&i.node)) + { + return Err("semantic fragment hash or dependency mismatch".into()); + } + if node + .inputs + .windows(2) + .any(|p| (role(p[0].role), &p[0].node) > (role(p[1].role), &p[1].node)) + { + return Err("noncanonical semantic input order".into()); + } + } + let mut seen = BTreeSet::new(); + let mut stack = vec![self.output.as_str()]; + while let Some(id) = stack.pop() { + let node = self.nodes.get(id).ok_or("missing semantic fragment root")?; + if seen.insert(id) { + stack.extend(node.inputs.iter().map(|i| i.node.as_str())); + } + } + if seen.len() != self.nodes.len() { + return Err("unrelated semantic fragment nodes".into()); + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::post_asap::{compile_executable_dag, SummaryExpr, SummaryNode}; + use crate::pre_asap::{Column, DataType, QueryExpr, Schema, Source}; + use std::rc::Rc; + + fn fixture(metric: &str) -> ExecutableDag { + let scan = QueryExpr::Scan { + source: Source::TimeSeries { + metric: metric.into(), + }, + predicates: vec![], + schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), + }; + let schema = SummarySchema { + fields: vec![crate::post_asap::SummaryField { + name: "value".into(), + dtype: crate::post_asap::SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }; + compile_executable_dag(&Rc::new(SummaryNode { + expr: SummaryExpr::KeepPreAsap(Rc::new(scan)), + schema, + guarantee: None, + })) + .unwrap() + } + + // Storage identity must ignore temporary identifiers and execution placement. + #[test] + fn identity_ignores_node_ids_and_phase() { + let dag = fixture("latency"); + let expected = SummarySemanticFragment::from_dag(&dag, dag.root).unwrap(); + let mut other = dag.clone(); + other.root = PostAsapNodeId(71); + other.nodes[0].id = other.root; + other.nodes[0].output_state.timing = crate::post_asap::ExecutionTiming::IngestionTime; + assert_eq!( + canonical_bytes(&expected).unwrap(), + canonical_bytes(&SummarySemanticFragment::from_dag(&other, other.root).unwrap()) + .unwrap() + ); + } + + // Source identity and supported semantic format survive restart independently. + #[test] + fn semantics_roundtrip_and_reject_unknown_version() { + let a = fixture("latency"); + let b = fixture("bytes"); + let a = SummarySemanticFragment::from_dag(&a, a.root).unwrap(); + let b = SummarySemanticFragment::from_dag(&b, b.root).unwrap(); + assert_ne!(canonical_bytes(&a).unwrap(), canonical_bytes(&b).unwrap()); + let mut restored: SummarySemanticFragment = + serde_json::from_slice(&canonical_bytes(&a).unwrap()).unwrap(); + restored.validate().unwrap(); + restored.format_version += 1; + assert!(restored.validate().is_err()); + } + // A transformed value cannot share state identity with its source column. + #[test] + fn value_expression_is_semantic_and_nonfinite_constants_are_rejected() { + use crate::pre_asap::{ProjectItem, ScalarValue}; + let original = fixture("latency"); + let expected = SummarySemanticFragment::from_dag(&original, original.root).unwrap(); + let mut transformed = original.clone(); + let ExecutableOperatorPayload::Fallback { expression } = &mut transformed.nodes[0].payload + else { + unreachable!() + }; + *expression = QueryExpr::Project { + cols: vec![ProjectItem { + alias: Some("value".into()), + expr: QueryExpr::FunctionCall { + name: "ln".into(), + args: vec![QueryExpr::Column(0)], + }, + }], + qualifier: None, + child: Rc::new(expression.clone()), + }; + let logged = SummarySemanticFragment::from_dag(&transformed, transformed.root).unwrap(); + assert_ne!( + canonical_bytes(&expected).unwrap(), + canonical_bytes(&logged).unwrap() + ); + let ExecutableOperatorPayload::Fallback { + expression: QueryExpr::Project { cols, .. }, + } = &mut transformed.nodes[0].payload + else { + unreachable!() + }; + cols[0].expr = QueryExpr::Literal(ScalarValue::Float64(f64::NAN)); + assert!(SummarySemanticFragment::from_dag(&transformed, transformed.root).is_err()); + } + + // Changing a downstream consumer cannot change the persisted input definition. + #[test] + fn only_output_dependency_closure_is_exported() { + let mut dag = fixture("latency"); + let stored = dag.root; + let mut consumer = dag.nodes[0].clone(); + consumer.id = PostAsapNodeId(9); + consumer.payload = ExecutableOperatorPayload::Value { + operation: crate::post_asap::ValueOperation::Project { + cols: vec![], + qualifier: None, + }, + }; + dag.edges.push(crate::post_asap::ExecutableDagEdge { + producer: stored, + consumer: consumer.id, + role: EdgeRole::Input, + intermediate_schema: dag.nodes[0].output_schema.clone(), + data_state: dag.nodes[0].output_state, + grouping: crate::post_asap::GroupingEdgeCompatibility::NotApplicable, + window: crate::post_asap::WindowEdgeCompatibility::NotApplicable, + }); + dag.root = consumer.id; + dag.nodes.push(consumer); + let before = SummarySemanticFragment::from_dag(&dag, stored).unwrap(); + dag.nodes.reverse(); + let after = SummarySemanticFragment::from_dag(&dag, stored).unwrap(); + assert_eq!(before, after); + assert_eq!(before.nodes.len(), 1); + } + // Query lookback does not become the identity of each stored input pane. + #[test] + fn stored_input_range_is_parameterized_but_logical_range_is_preserved() { + use crate::post_asap::*; + use crate::pre_asap::{ColumnRef, Reduction}; + let make = |seconds| { + let mut dag = fixture("latency"); + let ExecutableOperatorPayload::Fallback { expression } = &mut dag.nodes[0].payload + else { + unreachable!() + }; + *expression = QueryExpr::TimeRange { + range: std::time::Duration::from_secs(seconds), + child: Rc::new(expression.clone()), + }; + let mut output = dag.nodes[0].clone(); + output.id = PostAsapNodeId(1); + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + output.payload = ExecutableOperatorPayload::SummaryAgg { + family: family.clone(), + input: SummaryUpdate { + item: None, + weight: SummaryInputExpr::Column(ColumnRef::SampleValue), + weight_domain: Default::default(), + }, + reduction: Reduction::PerEntity, + grouping: Default::default(), + }; + output.output_schema.fields[0].dtype = family; + output.output_state.primitive = DataPrimitive::SummaryState; + dag.edges.push(ExecutableDagEdge { + producer: dag.root, + consumer: output.id, + role: EdgeRole::Input, + intermediate_schema: dag.nodes[0].output_schema.clone(), + data_state: dag.nodes[0].output_state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }); + dag.root = output.id; + dag.nodes.push(output); + dag + }; + let one = make(60); + let five = make(300); + assert_ne!( + SummarySemanticFragment::from_dag(&one, one.root).unwrap(), + SummarySemanticFragment::from_dag(&five, five.root).unwrap() + ); + assert_eq!( + SummarySemanticFragment::from_stored_output(&one, one.root).unwrap(), + SummarySemanticFragment::from_stored_output(&five, five.root).unwrap() + ); + // Open entities retain their full label identity; consumer-demanded + // optional labels do not change the per-entity stored computation. + let mut open = one.clone(); + let ExecutableOperatorPayload::Fallback { + expression: QueryExpr::TimeRange { child, .. }, + } = &mut open.nodes[0].payload + else { + unreachable!() + }; + let QueryExpr::Scan { schema, .. } = Rc::make_mut(child) else { + unreachable!() + }; + schema.closed = false; + let expected = SummarySemanticFragment::from_stored_output(&open, open.root).unwrap(); + let ExecutableOperatorPayload::Fallback { + expression: QueryExpr::TimeRange { child, .. }, + } = &mut open.nodes[0].payload + else { + unreachable!() + }; + let QueryExpr::Scan { schema, .. } = Rc::make_mut(child) else { + unreachable!() + }; + schema + .columns + .push(Column::new("job", DataType::Utf8, true)); + let label = SummaryField { + name: "job".into(), + dtype: SummaryFamilyType::Plain(DataType::Utf8), + nullable: true, + }; + for node in &mut open.nodes { + node.output_schema.fields.push(label.clone()); + } + open.edges[0].intermediate_schema.fields.push(label); + assert_eq!( + expected, + SummarySemanticFragment::from_stored_output(&open, open.root).unwrap() + ); + let mut forged = SummarySemanticFragment::from_stored_output(&one, one.root).unwrap(); + forged.nodes.values_mut().next().unwrap().operation = + serde_json::json!({"kind": "unknown"}); + assert!(forged.validate().is_err()); + } + // Persisted semantic format changes require an explicit migration/version review. + #[test] + fn semantic_format_v1_has_stable_wire_identity() { + let dag = fixture("latency"); + let exported = SummarySemanticFragment::from_dag(&dag, dag.root).unwrap(); + assert_eq!( + exported.output, + "488a0550f37763997397403ae5ed3588dee5d2a09fa7840f4110b775095fe594" + ); + } +} diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 82969e3e..58a1b0a0 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -63,6 +63,14 @@ alone does not establish actual window coverage, revision compatibility or readiness; those require runtime checks. Physical location, encoding, scheduling and retention are separate execution/deployment contracts. +The persisted semantic format contains only the dependency closure of the +selected output. It excludes execution timing, temporary node IDs and deployment +bindings. For a stored raw-input aggregate, its direct input interval is the +record's `(start, end]` interval; a consuming query's lookback is not the identity +of each pane. Nested computations retain their own time semantics. Planner exports +this contract through `SummarySemanticFragment`; changing its semantic wire +vocabulary requires an explicit format-version review. + ### Running example Suppose p50 and p99 are requested over the same latency samples in a five-minute window, From 6a1329d44fbbc65952a9b927007697367ccc10b7 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 12:46:42 +0000 Subject: [PATCH 52/90] feat: retain unpriced computation candidates before physical compilation --- crates/asap-aware-mapping/src/replacement.rs | 139 +++++++++++++++++++ 1 file changed, 139 insertions(+) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 8ebe08aa..d591d066 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -3711,6 +3711,120 @@ impl PlanSpace { } } +/// DAG candidates assembled from an unpriced search space. +/// This is an internal planning stage: callers must still validate lifecycle +/// requirements and compile supported physical operators before deployment. +/// The caller supplies a finite expansion budget; exceeding it is an error, +/// never a silently truncated inventory presented as exhaustive. +#[derive(Debug)] +pub struct CandidateDagInventory { + pub candidates: Vec)>>, + pub rejected_assemblies: Vec, +} + +type CandidateDagChoice<'a> = (Option<&'a ReplacementSubDAG>, Option>); + +impl PlanSpace { + pub fn enumerate_candidate_dags( + &self, + expansion_limit: usize, + ) -> Result, RealizationError> { + // Composition plans carry the proofs established during discovery. + // No cost ranking is consulted while expanding these choices. + let options: Vec>> = self + .order + .iter() + .map(|ptr| { + let group = &self.groups[ptr]; + let mut choices = vec![(None, None)]; + for candidate in &group.candidates { + match &candidate.replacement { + Replacement::ExactComposition(operation) => { + for prepared in &self.composition_plans { + if prepared.target == *ptr + && prepared.operation.placement == operation.placement + && prepared.operation.op == operation.op + && Rc::ptr_eq( + &prepared.operation.child_target, + &operation.child_target, + ) + { + choices + .push((Some(candidate), Some(Rc::clone(&prepared.plan)))); + } + } + } + _ => choices.push((Some(candidate), None)), + } + } + choices + }) + .collect(); + let combinations = options + .iter() + .try_fold(1usize, |n, choices| n.checked_mul(choices.len())) + .filter(|n| *n <= expansion_limit) + .ok_or(RealizationError::PhysicalRealization( + "candidate expansion budget exceeded; no partial inventory returned", + ))?; + let mut inventory = CandidateDagInventory { + candidates: Vec::new(), + rejected_assemblies: Vec::new(), + }; + for mut ordinal in 0..combinations { + let mut groups = HashMap::new(); + let mut assembled_nodes = HashMap::new(); + for (ptr, choices) in self.order.iter().zip(&options) { + let (chosen, prepared) = &choices[ordinal % choices.len()]; + ordinal /= choices.len(); + let group = &self.groups[ptr]; + if let Some(node) = prepared { + assembled_nodes.insert(*ptr, Rc::clone(node)); + } + groups.insert( + *ptr, + TargetSubDAGSelection { + target: &group.target, + consumer_count: group.consumer_count, + effective_consumer_count: group.consumer_count, + chosen: *chosen, + composition: None, + }, + ); + } + let assembly = GlobalSelection { + order: self.order.clone(), + groups, + assembled_nodes: RefCell::new(assembled_nodes), + }; + let roots = self + .roots + .iter() + .map(|(id, root)| { + assembly + .assemble_target(root) + .map(|node| (id.clone(), node)) + }) + .collect::, _>>(); + match roots { + Ok(roots) => { + let roots = asap_types::post_asap::share_common_summary_subtrees(roots); + if !inventory.candidates.contains(&roots) { + inventory.candidates.push(roots); + } + } + Err(error) => { + let reason = error.to_string(); + if !inventory.rejected_assemblies.contains(&reason) { + inventory.rejected_assemblies.push(reason); + } + } + } + } + Ok(inventory) + } +} + /// Lifecycle-aware whole-subplan costs keyed by target and candidate identity. #[derive(Default)] pub(crate) struct CandidateCostOverrides { @@ -6295,6 +6409,31 @@ mod tests { use asap_types::types::AccuracyTarget; use std::collections::HashMap; + #[test] + fn unpriced_inventory_retains_quantile_families_and_raw_execution() { + let query = Rc::new(agg(vec![2], default_quantile(0.9), metric_scan(&["job"]))); + let space = search_workload(vec![(0usize, query)]); + let inventory = space.enumerate_candidate_dags(4096).unwrap(); + let roots = inventory + .candidates + .iter() + .map(|forest| format!("{:?}", forest[0].1)) + .collect::>(); + assert!(roots.iter().any(|root| root.contains("Kll"))); + assert!(roots.iter().any(|root| root.contains("DDSketch"))); + assert!(inventory + .candidates + .iter() + .any(|forest| matches!(forest[0].1.expr, SummaryExpr::KeepPreAsap(_)))); + } + + #[test] + fn inventory_budget_never_returns_a_silent_partial_search() { + let query = Rc::new(agg(vec![2], default_quantile(0.9), metric_scan(&["job"]))); + let space = search_workload(vec![(0usize, query)]); + assert!(space.enumerate_candidate_dags(0).is_err()); + } + fn equi_pred(left: ColumnId, right: ColumnId) -> Predicate { Predicate(Rc::new(QueryExpr::Compare { left: Rc::new(QueryExpr::Column(left)), From e8b4e54dfc02ca4411bc59666757dda2d64e774b Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 12:50:05 +0000 Subject: [PATCH 53/90] docs: define physical candidate handoff and deployment selection ownership --- .../physical-planning-and-deployment.md | 63 ++++++++++++++++--- 1 file changed, 55 insertions(+), 8 deletions(-) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 58a1b0a0..5f8571a9 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -11,7 +11,7 @@ flowchart LR P["Physical DAG(s)
How is it executed?"] D["Deployment Plan / DAG
How is it instantiated?"] - L -->|"Summary Maintenance
Selection"| M + L -->|"Summary Maintenance
Candidate Generation"| M M -->|"Physical Plan
Compiler"| P P -->|"Deployment Plan
Compiler"| D ``` @@ -20,14 +20,59 @@ flowchart LR | --- | --- | | **Logical Post-ASAP DAG** | Computation semantics | | **Summary Maintenance Lifecycle** | Build, retention, reuse, and window strategy | -| **Physical DAG(s)** | Concrete executable operators and input boundaries | -| **Deployment Plan / DAG** | Concrete data/state bindings and operational lifecycle | +| **Physical DAG(s)** | Supported physical candidates, executable operators and typed input boundaries | +| **Deployment Plan / DAG** | Selected candidate, concrete data/state bindings and operational lifecycle | ASAPPlanner owns the first three layers and the shared physical operator implementation library. Deployment systems such as ASAPQuery and asap-fusion own deployment compilation and operation. The lifecycle is a planning contract associated with the logical DAG, not a separate computation IR. +### Candidate generation and deployment selection + +Planner exposes the supported, semantically legal **physical plan candidates**. +It does not discard a computation family or materialization placement merely +because a deployment-independent cost estimate prefers another candidate. +Logical candidates are an internal search stage, not the deployment handoff. + +```text +Query semantics + accuracy and lifecycle requirements + ↓ Planner +Supported Physical DAG candidates + typed inputs/outputs + requirements + ↓ backend +Binding feasibility + runtime statistics + resource limits + ERP + ↓ backend deployment compiler +Selected PrecomputePlan + QueryPlan + StoredOutputReferences +``` + +Planner owns operators, dependencies, sharing, and each candidate's +materialization frontier. The backend rejects candidates it cannot realize and +prices feasible candidates over a comparable workload and time horizon. It binds +the selected candidate; it does not lower the logical computation again, exchange +operators, or move an operator across the selected frontier. A missing quote is +not a zero-cost implementation. ERP evidence cannot authorize an illegal rewrite. + +The candidate inventory must identify its supported search scope and budget. +If a configured exhaustive enumeration exceeds its budget, planning fails +explicitly instead of selecting from an undisclosed partial inventory. Reports +separate unsupported compilation, deployment infeasibility, missing evidence, +and a feasible candidate that loses on cost. Absence is not a cost comparison. + +For `sum by(job)(rate(m[1m]))`, Rate remains per series before grouped Sum. +When lifecycle requirements permit it, a candidate may finalize Rate and Sum +within a bounded precompute run and persist the grouped value. Another may leave +those operators in the query DAG. Storing a value requires its exact evaluation +window, revision, readiness and serving cadence to match the query contract. + +For instant-vector TopK, CMS/CountSketch with a candidate heap requires explicit +series identity and a supported latest-value input protocol. Appending historical +sample values does not preserve instant-vector semantics. Replacement, rank +decrease, expiry, grouping and the required approximation guarantee must be +validated before admitting that physical candidate. + +This is the target ownership contract. A backend path that still reconstructs +operators from logical candidates has not completed this integration. + ### Input semantics and summary semantics `source`, `filter`, `grouping` and `window` describe input-data semantics: @@ -74,7 +119,7 @@ vocabulary requires an explicit format-version review. ### Running example Suppose p50 and p99 are requested over the same latency samples in a five-minute window, -and ASAP selects KLL with `k=200`. Assume query windows align with one-minute +and one Planner candidate uses KLL with `k=200`. Assume query windows align with one-minute pane boundaries and that the selected parameters satisfy the required guarantees. Operator names below are illustrative; the example defines the design, not a claim that the entire deployment integration is implemented. @@ -99,7 +144,7 @@ KLLMerge p50 p99 │ - │ Summary Maintenance Selection + │ Summary Maintenance Candidate Generation ▼ 2. Summary Maintenance Lifecycle @@ -174,8 +219,10 @@ Quantile(.5) Quantile(.99) It establishes that KLL with `k=200` is used and that the merge is shared by the two readouts. It does not determine when KLL states are built or retained. -**Summary Maintenance Selection** makes that decision using workload demand, -window/freshness requirements, and physical feasibility/cost. +**Summary Maintenance Candidate Generation** enumerates legal lifecycle choices +using workload demand, window/freshness requirements and supported physical +implementations. Backend selection uses runtime feasibility and cost after +physical compilation. The following example follows one candidate. For the running example, assume it selects: @@ -386,7 +433,7 @@ The complete example makes the ownership boundary explicit: | Stage | KLL example decision | | --- | --- | | **Logical Post-ASAP DAG** | Use `KLL(k=200)` with shared merge for p50/p99 | -| **Summary Maintenance Selection** | Maintain 1-minute panes and reuse them for aligned five-minute queries | +| **Summary Maintenance Candidate Generation** | Maintain 1-minute panes and reuse them for aligned five-minute queries | | **Summary Maintenance Lifecycle** | Record pane/window/freshness/reuse requirements | | **Physical Plan Compiler** | Lower to native KLL build, merge, and readout operators | | **Physical DAG** | Define maintenance and query DAGs with typed input/output boundaries | From 75c68b6d76198846d3eae4e226e21e4b508ce5de Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 13:04:10 +0000 Subject: [PATCH 54/90] feat: expose heap candidates over finalized per-series counter rates --- crates/asap-aware-mapping/src/replacement.rs | 66 +++++++- .../src/operators/summary/mod.rs | 18 +- .../tests/precompute_candidates.rs | 40 +++++ .../tests/weighted_topk_binding.rs | 155 ++++++++++++++++++ crates/frontend-promql/src/promql.rs | 24 ++- 5 files changed, 297 insertions(+), 6 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index d591d066..bb3e2a81 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -2401,9 +2401,10 @@ fn is_counter_weighted_topk(intent: &AggIntent, child: &QueryExpr) -> bool { matches!(intent, AggIntent::TopK { .. }) && matches!(child, QueryExpr::Aggregate { measures, child, .. } - if matches!(measures.as_slice(), [AggIntent::Sum { .. }]) - && matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } - if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase]))) + if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase]) + || (matches!(measures.as_slice(), [AggIntent::Sum { .. }]) + && matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])))) } /// Translate an [`Realization`] into the `(family, needs a @@ -2459,6 +2460,7 @@ type PhysicalSummaryInputRule = fn( /// `construct_summary_agg`. const PHYSICAL_SUMMARY_INPUT_RULES: &[PhysicalSummaryInputRule] = &[ realize_value_frequency_summary_input, + realize_counter_value_summary_input, realize_keyed_additive_summary_input, ]; @@ -3061,6 +3063,64 @@ fn ranking_score_index( Ok(index) } +/// Rebuild a heap from this evaluation's finalized per-series counter values. +/// The rate window is preserved; raw counter samples never become CMS weights. +fn realize_counter_value_summary_input( + intent: &AggIntent, + family: &SummaryFamilyType, + output_reduction: &Reduction, + child: &Rc, +) -> PhysicalSummaryInputRuleResult { + if !matches!(intent, AggIntent::TopK { .. }) + || !matches!(family, SummaryFamilyType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap)) + || !matches!(child.as_ref(), QueryExpr::Aggregate { reduction: Reduction::PerEntity, measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])) + { + return PhysicalSummaryInputRuleResult::NotApplicable; + } + let Ok(schema) = child.output_schema() else { + return PhysicalSummaryInputRuleResult::Unsupported( + "counter ranking needs a valid value schema", + ); + }; + if !schema.closed { + return PhysicalSummaryInputRuleResult::Unsupported( + "counter ranking needs the complete resolved series identity", + ); + } + let Reduction::Reduce(groups) = output_reduction else { + return PhysicalSummaryInputRuleResult::Unsupported( + "counter ranking requires explicit partitions", + ); + }; + if groups.is_without() { + return PhysicalSummaryInputRuleResult::Unsupported( + "counter ranking requires resolved partitions", + ); + } + // Retain the evaluation timestamp in each returned row. This sketch is a + // snapshot, not an additive history of successive rate evaluations. + let items = schema + .columns + .iter() + .enumerate() + .filter(|(index, column)| column.name != "value" && !groups.contains(index)) + .map(|(index, _)| schema_column_ref(child, index).map(SummaryInputExpr::Column)) + .collect::>>(); + let Some(items) = items.filter(|items| !items.is_empty()) else { + return PhysicalSummaryInputRuleResult::Unsupported("counter ranking has no item columns"); + }; + PhysicalSummaryInputRuleResult::Realized(PhysicalSummaryInput { + child: Rc::clone(child), + input: SummaryUpdate { + item: Some(SummaryInputExpr::Tuple(items)), + weight: SummaryInputExpr::Column(ColumnRef::SampleValue), + weight_domain: WeightDomain::NonNegative { + proof: NonNegativeWeightProof::ResetAwareCounterDerivative, + }, + }, + }) +} + /// Realize the composite heavy-hitter realization for /// `TopK(Count GROUP BY key)`. The heap sketch consumes the raw keyed stream; /// it does not consume an independently materialized Count result. diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs index 75102d8b..745b524f 100644 --- a/crates/asap-physical-operators/src/operators/summary/mod.rs +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -23,6 +23,7 @@ impl Operator { if !matches!( plain(&input, item)?.0, DataType::Utf8 + | DataType::Timestamp | DataType::Int64 | DataType::Float64 | DataType::Bool @@ -248,6 +249,15 @@ pub(super) fn execute<'a>( for items in summary.rows(*k) { let mut values = row[..*state].to_vec(); values.extend(items); + // The typed output schema restores epoch-millisecond + // timestamp keys from the kernel's Int64 representation. + for (value, field) in values.iter_mut().zip(&output.fields) { + if field.dtype == SummaryFamilyType::Plain(DataType::Timestamp) { + if let Value::Int64(time) = value { + *value = Value::Timestamp(*time); + } + } + } rows.push(values); } } @@ -524,7 +534,13 @@ async fn build_keyed_summary( return Err(invalid("weighted frequency weight type")); }; summary.update( - &items.iter().map(|&i| row[i].clone()).collect::>(), + &items + .iter() + .map(|&i| match &row[i] { + Value::Timestamp(time) => Value::Int64(*time), + value => value.clone(), + }) + .collect::>(), weight, )?; reservation.resize(summary.approx_memory_bytes() + *overhead)?; diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index 470272e3..f965a215 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -362,3 +362,43 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { "counter resets prohibit moving Sum before Rate" ); } + +/// Enumerated frontiers include both grouped-result and per-series readout +/// persistence; an explicit Rate-state input retains its original semantics. +#[test] +fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { + use asap_physical_operators::physical_planner::enumerate_frontiers; + let dag = grouped_rate(); + let state = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + ExecutableOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) + .unwrap(); + let inputs = BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + let roots = [u64::from(dag.root.0)]; + let frontiers = enumerate_frontiers(&dag, &inputs, &roots, 4096).unwrap(); + let candidates = compile_candidates(&dag, inputs.clone(), &roots, &frontiers) + .into_iter() + .collect::, _>>() + .unwrap(); + assert!(candidates.iter().any(|c| c.precompute.is_none())); + assert!(candidates + .iter() + .any(|c| c.materialized_outputs.contains_key(&roots[0]))); + assert!(candidates + .iter() + .any(|c| !c.materialized_outputs.is_empty() + && !c.materialized_outputs.contains_key(&roots[0]))); + assert!(enumerate_frontiers(&dag, &inputs, &roots, 1).is_err()); +} diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index e37c2120..8037eb14 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -264,3 +264,158 @@ fn rate_updates_cannot_enter_integer_heap_factory() { .is_err() ); } + +/// A catalog-resolved per-series rate can feed a heap sketch directly, without +/// requiring an otherwise unnecessary grouped Sum between Rate and TopK. +#[test] +fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { + let mut logical = + lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(); + fn resolve_catalog(node: &mut QueryExpr) { + match node { + QueryExpr::Aggregate { child, .. } | QueryExpr::TimeRange { child, .. } => { + resolve_catalog(Rc::make_mut(child)) + } + QueryExpr::Scan { schema, .. } => { + schema.closed = true; + schema + .columns + .push(planner_types::pre_asap::schema::Column::new( + "service", + DataType::Utf8, + false, + )); + } + _ => panic!("unexpected input shape: {node:?}"), + } + } + resolve_catalog(&mut logical); + let root = Rc::new(logical); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let candidates = strategy.replacements(&TargetSubDAG::new(&root)); + for algorithm in [ + SketchAlgorithm::CmsWithHeap, + SketchAlgorithm::CountSketchWithHeap, + ] { + let candidate = candidates + .iter() + .find_map(|candidate| match &candidate.replacement { + Replacement::Summary(node) + if candidate.rationale.contains(&format!("{algorithm:?}")) => + { + Some(node) + } + _ => None, + }) + .unwrap_or_else(|| panic!("missing {algorithm:?} over direct Rate")); + let dag = compile_executable_dag(candidate).unwrap(); + assert!(dag.nodes.iter().any(|node| matches!(&node.payload, + ExecutableOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); + let build = dag.nodes.iter().find(|node| matches!(&node.payload, + ExecutableOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); + let input_id = dag + .edges + .iter() + .find(|edge| edge.consumer == build.id) + .unwrap() + .producer; + let schema = Arc::new( + dag.nodes + .iter() + .find(|node| node.id == input_id) + .unwrap() + .output_schema + .clone(), + ); + let compiled = compile( + &dag, + BTreeMap::from([( + u64::from(input_id.0), + InputContract::bounded(schema.clone()), + )]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + for (time, values, expected) in [ + ( + 60_000, + vec![("auth", 3.), ("checkout", 2.), ("search", 1.)], + vec![2., 3.], + ), + ( + 61_000, + vec![("auth", 0.), ("checkout", 2.), ("search", 4.)], + vec![2., 4.], + ), + (62_000, vec![("auth", 0.), ("checkout", 2.)], vec![0., 2.]), + ] { + let rows = values + .into_iter() + .map(|(service, value)| { + schema + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "job" => Value::Utf8("api".into()), + "value" => Value::Float64(value), + "ts" => Value::Timestamp(time), + _ => panic!("unexpected rate field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(schema.clone(), rows).unwrap(); + let source = + Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'static>; + let graph = compiled + .instantiate(BTreeMap::from([(u64::from(input_id.0), source)])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: time, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + let mut scores = block_on(async { + let mut scores = vec![]; + let mut stream = graph + .execute(&[u64::from(dag.root.0)], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + let batch = batch.unwrap(); + for row in batch.rows() { + assert!(row.iter().any( + |value| matches!(value, Value::Timestamp(actual) if *actual == time) + )); + scores.push( + row.iter() + .find_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .unwrap(), + ); + } + } + scores + }); + scores.sort_by(f64::total_cmp); + assert_eq!( + scores, expected, + "heap snapshots must not accumulate across evaluations" + ); + } + } +} diff --git a/crates/frontend-promql/src/promql.rs b/crates/frontend-promql/src/promql.rs index f0cf907f..c3df6c67 100644 --- a/crates/frontend-promql/src/promql.rs +++ b/crates/frontend-promql/src/promql.rs @@ -625,7 +625,10 @@ fn build_over_subtree(outer: Outer, keys: Vec, child: Unresolved) -> && matches!(sum_child.as_ref(), Unresolved::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])) ); - if descending && weighted_counter_ranking { + let direct_counter_ranking = matches!(&child, Unresolved::Aggregate { + measures, reduction: Reduction::PerEntity, .. + } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])); + if descending && (weighted_counter_ranking || direct_counter_ranking) { return Ok(outer_aggregate( keys, AggIntent::TopK { @@ -1487,10 +1490,27 @@ fn build(inner: Inner, keys: Vec, outer: Outer) -> Result }) } Outer::TopK { k, descending } => { + // Preserve the counter-value ranking intent. Physical candidates + // may rebuild a heap over finalized rates or use exact Sort/Limit; + // neither is allowed to sum raw counter samples as ranking weights. + if descending && matches!(inner.func, Some(InnerFunc::Rate | InnerFunc::Increase)) { + let intent = inner_intent(inner.func.as_ref().expect("counter function")); + let ranked = windowed_aggregate(inner, vec![], intent); + return Ok(Unresolved::Aggregate { + reduction: Reduction::Reduce(keys.into()), + measures: vec![AggIntent::TopK { + k: k as usize, + accuracy: current_accuracy(), + }], + output_names: vec![], + having: None, + child: Rc::new(ranked), + }); + } // Heavy-hitter only when ranking by an additive measure (`count` // or `sum`): that is a // first-class aggregate intent → `TopK`. Any other ranking (topk - // over avg/quantile/rate, a bare selector's raw value, all bottomk) + // over avg/quantile, a bare selector's raw value, all bottomk) // is a generic order-by-value + limit and stays as the `Sort + Limit` // operator pair. The descending-plus-measure rule is shared with the // canonicalize-pass promotion so the two cannot drift (issue #38). From 4cfbfae313041af5607c3cabfb763d551ab107c1 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 13:07:20 +0000 Subject: [PATCH 55/90] perf: bucket candidate equality checks without changing admission --- crates/asap-aware-mapping/src/replacement.rs | 39 +++++++++++++++++++- 1 file changed, 38 insertions(+), 1 deletion(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index bb3e2a81..80d259b9 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -3831,6 +3831,9 @@ impl PlanSpace { candidates: Vec::new(), rejected_assemblies: Vec::new(), }; + // Hash buckets avoid quadratic comparisons across a large workload + // inventory. Equality still decides deduplication, including collisions. + let mut seen = HashMap::>::new(); for mut ordinal in 0..combinations { let mut groups = HashMap::new(); let mut assembled_nodes = HashMap::new(); @@ -3869,7 +3872,41 @@ impl PlanSpace { match roots { Ok(roots) => { let roots = asap_types::post_asap::share_common_summary_subtrees(roots); - if !inventory.candidates.contains(&roots) { + use std::hash::{Hash, Hasher}; + let mut hash = std::collections::hash_map::DefaultHasher::new(); + for (_, node) in &roots { + let mut value = serde_json::to_value((&node.schema, &node.guarantee)) + .map_err(|_| { + RealizationError::PhysicalRealization( + "candidate identity serialization failed", + ) + })?; + fn normalize(value: &mut serde_json::Value) { + match value { + serde_json::Value::Number(number) + if number.as_f64() == Some(0.0) => + { + *value = serde_json::json!(0); + } + serde_json::Value::Array(values) => { + values.iter_mut().for_each(normalize) + } + serde_json::Value::Object(values) => { + values.values_mut().for_each(normalize) + } + _ => {} + } + } + normalize(&mut value); + value.sort_all_objects(); + value.to_string().hash(&mut hash); + } + let bucket = seen.entry(hash.finish()).or_default(); + if !bucket + .iter() + .any(|&index| inventory.candidates[index] == roots) + { + bucket.push(inventory.candidates.len()); inventory.candidates.push(roots); } } From 2359a30e560ab1d7b6f734dac6fe3b98a28d30e0 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 13:15:31 +0000 Subject: [PATCH 56/90] test: verify counter heap snapshots in query and ingestion scopes --- crates/asap-aware-mapping/src/replacement.rs | 48 ++++++++++- .../tests/weighted_topk_binding.rs | 86 ++++++++++--------- 2 files changed, 91 insertions(+), 43 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 80d259b9..04f199cf 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -3874,9 +3874,29 @@ impl PlanSpace { let roots = asap_types::post_asap::share_common_summary_subtrees(roots); use std::hash::{Hash, Hasher}; let mut hash = std::collections::hash_map::DefaultHasher::new(); - for (_, node) in &roots { - let mut value = serde_json::to_value((&node.schema, &node.guarantee)) - .map_err(|_| { + let mut pending = roots + .iter() + .map(|(_, node)| node.as_ref()) + .collect::>(); + while let Some(node) = pending.pop() { + std::mem::discriminant(&node.expr).hash(&mut hash); + let raw = match &node.expr { + SummaryExpr::KeepPreAsap(raw) => Some(raw.as_ref()), + _ => None, + }; + let operation = match &node.expr { + SummaryExpr::ValueOperation { + timing, operation, .. + } => serde_json::json!((timing, operation)), + SummaryExpr::BinaryOp { + timing, operator, .. + } => serde_json::json!((timing, operator)), + SummaryExpr::SummaryMerge { timing, .. } => serde_json::json!(timing), + _ => serde_json::Value::Null, + }; + let mut value = + serde_json::to_value((&node.schema, &node.guarantee, raw, operation)) + .map_err(|_| { RealizationError::PhysicalRealization( "candidate identity serialization failed", ) @@ -3900,6 +3920,28 @@ impl PlanSpace { normalize(&mut value); value.sort_all_objects(); value.to_string().hash(&mut hash); + match &node.expr { + SummaryExpr::KeepPreAsap(_) => {} + SummaryExpr::BinaryOp { lhs, rhs, .. } => { + pending.extend([lhs.as_ref(), rhs.as_ref()]) + } + SummaryExpr::RelationalJoin { left, right, .. } + | SummaryExpr::SummarySubtract { left, right } => { + pending.extend([left.as_ref(), right.as_ref()]) + } + SummaryExpr::ValueOperation { child, .. } + | SummaryExpr::SummaryAgg { child, .. } => pending.push(child.as_ref()), + SummaryExpr::SummaryJoin { outer, inner, .. } => { + pending.extend([outer.as_ref(), inner.as_ref()]) + } + SummaryExpr::SummaryDelete { summary_input, .. } + | SummaryExpr::SummaryEstimate { summary_input, .. } => { + pending.push(summary_input.as_ref()) + } + SummaryExpr::SummaryMerge { children, .. } => { + pending.extend(children.iter().map(|child| child.as_ref())) + } + } } let bucket = seen.entry(hash.finish()).or_default(); if !bucket diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index 8037eb14..3ece6a24 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -371,51 +371,57 @@ fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { }) .collect(); let batch = Batch::try_new(schema.clone(), rows).unwrap(); - let source = - Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'static>; - let graph = compiled - .instantiate(BTreeMap::from([(u64::from(input_id.0), source)])) - .unwrap(); - let context = RunContext::new( + for scope in [ Scope::Query { evaluation_time_ms: time, revision: 1, }, - Limits::default(), - ) - .unwrap(); - let mut scores = block_on(async { - let mut scores = vec![]; - let mut stream = graph - .execute(&[u64::from(dag.root.0)], context) - .unwrap() - .remove(0); - while let Some(batch) = stream.next().await { - let batch = batch.unwrap(); - for row in batch.rows() { - assert!(row.iter().any( - |value| matches!(value, Value::Timestamp(actual) if *actual == time) - )); - scores.push( - row.iter() - .find_map(|value| { - if let Value::Float64(value) = value { - Some(*value) - } else { - None - } - }) - .unwrap(), - ); + Scope::Ingestion { + window_start_ms: time - 60_000, + window_end_ms: time, + revision: 1, + }, + ] { + let source = + Box::new(Operator::source(schema.clone(), vec![batch.clone()]).unwrap()) + as Source<'static>; + let graph = compiled + .instantiate(BTreeMap::from([(u64::from(input_id.0), source)])) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let mut scores = block_on(async { + let mut scores = vec![]; + let mut stream = graph + .execute(&[u64::from(dag.root.0)], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + let batch = batch.unwrap(); + for row in batch.rows() { + assert!(row.iter().any( + |value| matches!(value, Value::Timestamp(actual) if *actual == time) + )); + scores.push( + row.iter() + .find_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .unwrap(), + ); + } } - } - scores - }); - scores.sort_by(f64::total_cmp); - assert_eq!( - scores, expected, - "heap snapshots must not accumulate across evaluations" - ); + scores + }); + scores.sort_by(f64::total_cmp); + assert_eq!( + scores, expected, + "heap snapshots must not accumulate across evaluations" + ); + } } } } From 3404ec232edb96eea3e037d33680f59687fa0716 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 14:22:15 +0000 Subject: [PATCH 57/90] feat: enumerate candidate roots without workload Cartesian expansion --- crates/asap-aware-mapping/src/replacement.rs | 136 ++++++++++++++++++- 1 file changed, 130 insertions(+), 6 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 04f199cf..1c9efdaa 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -3789,10 +3789,65 @@ impl PlanSpace { &self, expansion_limit: usize, ) -> Result, RealizationError> { + self.enumerate_candidate_roots(&self.roots, expansion_limit) + } + + /// Enumerate one workload root without expanding independent roots' choices. + /// Discovery and composition proofs still come from the shared workload + /// space. Deployment may price combinations lazily; this API does not rank + /// candidates or claim that independently cheapest roots minimize shared cost. + pub fn enumerate_candidate_dags_for_root( + &self, + id: &Id, + expansion_limit: usize, + ) -> Result, RealizationError> { + let roots = self + .roots + .iter() + .filter(|(candidate, _)| candidate == id) + .cloned() + .collect::>(); + if roots.len() != 1 { + return Err(RealizationError::PhysicalRealization( + "candidate enumeration requires one uniquely identified workload root", + )); + } + self.enumerate_candidate_roots(&roots, expansion_limit) + } + + fn enumerate_candidate_roots( + &self, + roots: &[(Id, Rc)], + expansion_limit: usize, + ) -> Result, RealizationError> { + let mut reachable = Vec::new(); + let mut nodes = HashMap::new(); + let mut counts = HashMap::new(); + for (_, root) in roots { + walk(root, &mut reachable, &mut nodes, &mut counts); + } + // Rewrites may introduce descendants absent from the original root. + let mut cursor = 0; + while cursor < reachable.len() { + let ptr = reachable[cursor]; + cursor += 1; + if let Some(group) = self.groups.get(&ptr) { + for candidate in &group.candidates { + if let Replacement::Rewrite(rewritten) = &candidate.replacement { + walk(rewritten, &mut reachable, &mut nodes, &mut counts); + } + } + } + } + let order = self + .order + .iter() + .copied() + .filter(|ptr| counts.contains_key(ptr)) + .collect::>(); // Composition plans carry the proofs established during discovery. // No cost ranking is consulted while expanding these choices. - let options: Vec>> = self - .order + let options: Vec>> = order .iter() .map(|ptr| { let group = &self.groups[ptr]; @@ -3837,7 +3892,7 @@ impl PlanSpace { for mut ordinal in 0..combinations { let mut groups = HashMap::new(); let mut assembled_nodes = HashMap::new(); - for (ptr, choices) in self.order.iter().zip(&options) { + for (ptr, choices) in order.iter().zip(&options) { let (chosen, prepared) = &choices[ordinal % choices.len()]; ordinal /= choices.len(); let group = &self.groups[ptr]; @@ -3856,12 +3911,11 @@ impl PlanSpace { ); } let assembly = GlobalSelection { - order: self.order.clone(), + order: order.clone(), groups, assembled_nodes: RefCell::new(assembled_nodes), }; - let roots = self - .roots + let roots = roots .iter() .map(|(id, root)| { assembly @@ -6566,6 +6620,76 @@ mod tests { .any(|forest| matches!(forest[0].1.expr, SummaryExpr::KeepPreAsap(_)))); } + // Independent roots must not require materializing their Cartesian product. + #[test] + fn root_inventory_preserves_choices_without_workload_cartesian_expansion() { + let roots = (0..24usize) + .map(|id| { + ( + id, + Rc::new(agg( + vec![2], + default_quantile((id + 1) as f64 / 25.0), + metric_scan(&["job"]), + )), + ) + }) + .collect(); + let space = search_workload(roots); + assert!(space.enumerate_candidate_dags(4096).is_err()); + for id in 0..24 { + let inventory = space.enumerate_candidate_dags_for_root(&id, 4096).unwrap(); + assert!(inventory + .candidates + .iter() + .all(|forest| forest.len() == 1 && forest[0].0 == id)); + let descriptions = inventory + .candidates + .iter() + .map(|forest| format!("{:?}", forest[0].1)) + .collect::>(); + assert!(descriptions.iter().any(|node| node.contains("Kll"))); + assert!(descriptions.iter().any(|node| node.contains("DDSketch"))); + assert!(inventory + .candidates + .iter() + .any(|forest| matches!(forest[0].1.expr, SummaryExpr::KeepPreAsap(_)))); + } + assert!(space.enumerate_candidate_dags_for_root(&24, 4096).is_err()); + assert!(space.enumerate_candidate_dags_for_root(&0, 0).is_err()); + } + + // Factoring changes enumeration, not the set of root computations. + #[test] + fn root_inventory_matches_projection_of_exhaustive_workload_inventory() { + let roots = (0..2usize) + .map(|id| { + ( + id, + Rc::new(agg( + vec![2], + default_quantile(0.5 + id as f64 * 0.4), + metric_scan(&["job"]), + )), + ) + }) + .collect(); + let space = search_workload(roots); + let full = space.enumerate_candidate_dags(4096).unwrap(); + for id in 0..2 { + let inventory = space.enumerate_candidate_dags_for_root(&id, 4096).unwrap(); + for forest in &full.candidates { + let node = &forest.iter().find(|(root, _)| *root == id).unwrap().1; + assert!(inventory.candidates.iter().any(|one| &one[0].1 == node)); + } + for one in &inventory.candidates { + assert!(full.candidates.iter().any(|forest| forest + .iter() + .any(|(root, node)| *root == id && node == &one[0].1))); + } + } + } + #[test] fn inventory_budget_never_returns_a_silent_partial_search() { let query = Rc::new(agg(vec![2], default_quantile(0.9), metric_scan(&["job"]))); From 824b26d3bf6801ad382a62a4d9ce76fb905fe34c Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 19:58:48 +0000 Subject: [PATCH 58/90] fix: compile resolved per-series counter windows before heap ranking --- .../src/physical_planner/mod.rs | 60 ++++++++++ .../tests/weighted_topk_binding.rs | 109 ++++++++++++++++++ 2 files changed, 169 insertions(+) diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index eff79f6a..09b9bb7e 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -188,6 +188,66 @@ fn compile_internal( auxiliary -= 1; schemas.truncate(1); } + // Per-entity identity is safe only when the source catalog closes + // the label set. An open PromQL projection can hide distinct series. + if let Payload::SummaryAgg { + family, + input: update, + reduction: PlannerReduction::PerEntity, + grouping, + } = &node.payload + { + let [input_id] = inputs.as_slice() else { + return Err(invalid("per-entity summary requires one input")); + }; + let Payload::Fallback { + expression: QueryExpr::TimeRange { child, .. }, + } = &nodes[input_id].payload + else { + return Err(invalid( + "per-entity summary requires a resolved raw time range", + )); + }; + let QueryExpr::Scan { schema, .. } = child.as_ref() else { + return Err(invalid("per-entity summary requires a resolved source")); + }; + if !schema.closed || update.item.is_some() { + return Err(invalid( + "per-entity summary requires complete source identity", + )); + } + crate::capability::validate_summary_kernel(family, update, grouping) + .map_err(Error::Invalid)?; + let SummaryInputExpr::Column(value) = &update.weight else { + return Err(invalid( + "per-entity update requires a projected value column", + )); + }; + let input = schemas[0].clone(); + let value = named_column(&input, value)?; + let coordinate = input + .time_index + .ok_or_else(|| invalid("temporal input lacks time"))?; + let groups = (0..input.fields.len()) + .filter(|&column| column != value && column != coordinate) + .collect(); + let build = Operator::summary_build( + input, + family.clone(), + value, + Some(coordinate), + groups, + )?; + let compact = build.schema(); + graph.add(auxiliary, inputs, build)?; + graph.add( + id, + vec![auxiliary], + Operator::scope_timestamp(compact, output)?, + )?; + auxiliary -= 1; + continue; + } let mut operator = compile_node(node, &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; if operator.is_counter_readout() { diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index 3ece6a24..fc51e87f 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -332,6 +332,115 @@ fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { .output_schema .clone(), ); + let raw = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + ExecutableOperatorPayload::Fallback { + expression: QueryExpr::TimeRange { .. } + } + ) + }) + .unwrap_or_else(|| panic!("no raw counter source: {dag:?}")); + let raw_schema = Arc::new(raw.output_schema.clone()); + let raw_compiled = compile( + &dag, + BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(raw_schema.clone()), + )]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + // Each evaluation receives a complete raw window. A reset, a stopped + // series and an expired leader must not retain last run's heap weights. + for (end, series, expected) in [ + ( + 60_000, + vec![ + ("auth", vec![10., 30., 50.]), + ("checkout", vec![10., 50., 90.]), + ("search", vec![10., 70., 130.]), + ], + vec![11. / 6., 8. / 3.], + ), + ( + 120_000, + vec![ + ("auth", vec![100., 10., 50.]), + ("checkout", vec![100., 100., 100.]), + ], + vec![0., 1.25], + ), + ] { + let mut raw_rows = Vec::new(); + for (service, samples) in series { + for (offset, value) in [10_000, 30_000, 50_000].into_iter().zip(samples) { + raw_rows.push( + raw_schema + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "job" => Value::Utf8("api".into()), + "value" => Value::Float64(value), + "ts" => Value::Timestamp(end - 60_000 + offset), + _ => panic!("unexpected raw field"), + }) + .collect(), + ); + } + } + let raw_batch = Batch::try_new(raw_schema.clone(), raw_rows).unwrap(); + for scope in [ + Scope::Ingestion { + window_start_ms: end - 60_000, + window_end_ms: end, + revision: 1, + }, + Scope::Query { + evaluation_time_ms: end, + revision: 1, + }, + ] { + let source = Box::new( + Operator::source(raw_schema.clone(), vec![raw_batch.clone()]).unwrap(), + ) as Source<'static>; + let graph = raw_compiled + .instantiate(BTreeMap::from([(u64::from(raw.id.0), source)])) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let mut raw_scores = block_on(async { + let mut scores = Vec::new(); + let mut stream = graph + .execute(&[u64::from(dag.root.0)], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + for row in batch.unwrap().rows() { + assert!(row.iter().any( + |value| matches!(value, Value::Timestamp(time) if *time == end) + )); + scores.extend(row.iter().filter_map(|value| match value { + Value::Float64(value) => Some(*value), + _ => None, + })); + } + } + scores + }); + raw_scores.sort_by(f64::total_cmp); + assert_eq!(raw_scores.len(), expected.len()); + for (actual, expected) in raw_scores.iter().zip(&expected) { + assert!( + (actual - expected).abs() < 1e-12, + "raw counter semantics must precede heap ranking: {raw_scores:?}" + ); + } + } + } let compiled = compile( &dag, BTreeMap::from([( From 595ca6ef6107c6cbbe8fcf99eabe7512966ea2b9 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 20:09:19 +0000 Subject: [PATCH 59/90] feat: persist and validate selected physical candidates without logical lowering --- .../src/expressions/mod.rs | 2 +- .../src/expressions/planner.rs | 12 +- .../src/operators/aggregate/mod.rs | 2 +- .../src/operators/mod.rs | 7 +- .../src/operators/persisted.rs | 118 ++++++++++++++++++ .../src/operators/sort.rs | 2 +- .../src/physical_planner/candidates.rs | 79 ++++++++++++ .../src/physical_planner/compiled.rs | 41 +++++- .../src/plan/properties.rs | 6 +- crates/asap-physical-operators/src/values.rs | 3 +- .../tests/physical_plan_recovery.rs | 102 +++++++++++++++ .../tests/weighted_topk_binding.rs | 3 + 12 files changed, 365 insertions(+), 12 deletions(-) create mode 100644 crates/asap-physical-operators/src/operators/persisted.rs create mode 100644 crates/asap-physical-operators/tests/physical_plan_recovery.rs diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs index f6dd1336..b9429220 100644 --- a/crates/asap-physical-operators/src/expressions/mod.rs +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -7,7 +7,7 @@ use planner_types::pre_asap::{ArithmeticOpKind, DataType}; pub mod arithmetic; mod planner; pub use planner::CompiledExpression; -#[derive(Clone, Debug)] +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] pub enum Expression { Binary { operator: planner_types::post_asap::BinaryOperator, diff --git a/crates/asap-physical-operators/src/expressions/planner.rs b/crates/asap-physical-operators/src/expressions/planner.rs index b2d75421..2130a2f7 100644 --- a/crates/asap-physical-operators/src/expressions/planner.rs +++ b/crates/asap-physical-operators/src/expressions/planner.rs @@ -329,13 +329,17 @@ fn cell_cmp(left: &Value, right: &Value) -> Option { } } -#[derive(Clone, Debug)] +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] pub struct CompiledExpression { expression: QueryExpr, schema: planner_types::pre_asap::Schema, output: (DataType, bool), } impl CompiledExpression { + pub(crate) fn expression(&self) -> &QueryExpr { + &self.expression + } + pub fn compile(expression: &QueryExpr, input: &Schema) -> Result { let schema = input .fields @@ -368,6 +372,12 @@ impl CompiledExpression { self.output.clone() } pub(crate) fn validate_input(&self, input: &Schema) -> Result<(), Error> { + let checked = Self::compile(&self.expression, input)?; + if checked.output != self.output { + return Err(Error::Invalid( + "persisted expression type differs from its semantics".into(), + )); + } if input.fields.len() != self.schema.columns.len() || input .fields diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs index 0270942c..c57b4fce 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/mod.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -113,7 +113,7 @@ impl Operator { }) } } -#[derive(Clone, Debug)] +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] pub enum Reduction { Count, Sum(usize), diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index 0b5b8d47..7c38835e 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -20,14 +20,16 @@ mod filter; mod joins; mod limit; mod panes; +mod persisted; mod projection; mod sort; mod source; mod summary; pub use aggregate::Reduction; pub use sort::SortKey; -#[derive(Clone)] +#[derive(Clone, serde::Serialize, serde::Deserialize)] enum Kind { + #[serde(skip)] Source(Vec), PaneInput { coordinate: usize, @@ -97,7 +99,8 @@ enum Kind { }, } /// A bound operation has a fully checked input/output contract before execution. -#[derive(Clone)] +#[derive(Clone, serde::Serialize, serde::Deserialize)] +#[serde(try_from = "persisted::StoredOperator")] pub struct Operator { kind: Kind, inputs: Vec, diff --git a/crates/asap-physical-operators/src/operators/persisted.rs b/crates/asap-physical-operators/src/operators/persisted.rs new file mode 100644 index 00000000..17d05061 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/persisted.rs @@ -0,0 +1,118 @@ +//! Recovery validates operator contracts, without selecting or lowering a plan. +use super::*; + +#[derive(serde::Deserialize)] +#[serde(deny_unknown_fields)] +pub(super) struct StoredOperator { + kind: Kind, + inputs: Vec, + output: Schema, +} +impl TryFrom for Operator { + type Error = Error; + fn try_from(stored: StoredOperator) -> Result { + let expected_kind = + serde_json::to_value(&stored.kind).map_err(|error| invalid(&error.to_string()))?; + let StoredOperator { + kind, + inputs, + output, + } = stored; + for schema in inputs.iter().chain(std::iter::once(&output)) { + crate::values::validate_schema(schema)?; + } + let input = |index| { + inputs + .get(index) + .cloned() + .ok_or_else(|| invalid("missing persisted input")) + }; + let op = match kind { + Kind::Source(_) => return Err(invalid("physical plans cannot persist live sources")), + Kind::PaneInput { + coordinate, + layout, + offset_ms, + } => Operator::pane_input(input(0)?, coordinate, layout, offset_ms)?, + Kind::ScopeTimestamp { .. } => Operator::scope_timestamp(input(0)?, output.clone())?, + Kind::Union => Operator::union(input(0)?, inputs.len())?, + Kind::VectorToScalar { column } => Operator::vector_to_scalar(input(0)?, column)?, + Kind::Project(expressions) => { + if expressions.len() != output.fields.len() { + return Err(invalid("persisted projection width mismatch")); + } + Operator::project( + input(0)?, + output + .fields + .iter() + .zip(expressions) + .map(|(f, e)| (f.name.clone(), e)) + .collect(), + )? + } + Kind::Filter(expression) => Operator::filter(input(0)?, expression)?, + Kind::Limit { n, offset, groups } => Operator::limit(input(0)?, n, offset, groups)?, + Kind::Sort { keys, groups } => Operator::sort(input(0)?, keys, groups)?, + Kind::Window { + intent, + coordinate, + value, + groups, + window, + } => Operator::window(input(0)?, *intent, coordinate, value, groups, window)?, + Kind::Aggregate { groups, measures } => { + if groups.len() + measures.len() != output.fields.len() { + return Err(invalid("persisted aggregate width mismatch")); + } + let names = output.fields[groups.len()..].iter().map(|f| f.name.clone()); + Operator::aggregate(input(0)?, groups, names.zip(measures).collect())? + } + Kind::SemiJoin { keys } => Operator::semi_join(input(0)?, input(1)?, keys)?, + Kind::Join { kind, predicate } => Operator::relational_join( + input(0)?, + input(1)?, + kind, + &planner_types::pre_asap::Predicate(std::rc::Rc::new( + predicate.expression().clone(), + )), + output.clone(), + )?, + Kind::SummaryBuild { + family, + value, + time, + groups, + } => Operator::summary_build(input(0)?, family, value, time, groups)?, + Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + } => Operator::keyed_summary_build(input(0)?, family, value, items, groups)?, + Kind::KeyedReadout { state, k } => { + Operator::keyed_readout(input(0)?, state, k, output.clone())? + } + Kind::SummaryMerge { state, groups } => { + Operator::summary_merge(input(0)?, state, groups)? + } + Kind::Readout { + state, + statistic, + parameters, + } => Operator::readout(input(0)?, state, statistic, parameters)?, + } + .with_output_schema(output)?; + if serde_json::to_value(&op.kind).map_err(|error| invalid(&error.to_string()))? + != expected_kind + { + return Err(invalid( + "persisted operator contains inconsistent compiled fields", + )); + } + if op.inputs != inputs { + return Err(invalid("persisted operator input contracts differ")); + } + Ok(op) + } +} diff --git a/crates/asap-physical-operators/src/operators/sort.rs b/crates/asap-physical-operators/src/operators/sort.rs index be9c1d7d..71f81de0 100644 --- a/crates/asap-physical-operators/src/operators/sort.rs +++ b/crates/asap-physical-operators/src/operators/sort.rs @@ -14,7 +14,7 @@ impl Operator { }) } } -#[derive(Clone, Debug)] +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] pub struct SortKey { pub column: usize, pub descending: bool, diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs index a84c3394..148035d7 100644 --- a/crates/asap-physical-operators/src/physical_planner/candidates.rs +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -214,3 +214,82 @@ pub fn select_candidate( } selected.ok_or_else(|| invalid("no feasible priced physical candidate")) } + +#[derive(serde::Serialize, serde::Deserialize)] +#[serde(deny_unknown_fields)] +struct StoredCandidate { + version: u32, + precompute: Option, + query: serde_json::Value, + materialized_outputs: BTreeMap, +} + +impl PhysicalCandidate { + /// Validate the physical handoff, including the producer/reader boundary. + pub fn validate(&self) -> Result<(), Error> { + self.query.validate()?; + let Some(precompute) = &self.precompute else { + return if self.materialized_outputs.is_empty() { + Ok(()) + } else { + Err(invalid("materialized outputs have no producer DAG")) + }; + }; + precompute.validate()?; + let outputs: BTreeSet<_> = self.materialized_outputs.keys().copied().collect(); + if outputs.is_empty() || outputs != precompute.roots().iter().copied().collect() { + return Err(invalid("physical frontier differs from precompute outputs")); + } + let readers: BTreeMap<_, _> = self.query.input_contracts().collect(); + for (&id, contract) in &self.materialized_outputs { + let produced = precompute.output_contract(id)?; + // Direct frontiers retain their node IDs. Temporal candidates can + // read several window instances through distinct input slots; + // their deployment bindings must validate those slots separately. + let reader = readers.get(&id); + if contract.schema != produced.schema + || reader.is_some_and(|reader| contract.schema != reader.schema) + || produced.properties.boundedness != Boundedness::Bounded + || contract.properties.boundedness != Boundedness::Bounded + || reader + .is_some_and(|reader| reader.properties.boundedness != Boundedness::Bounded) + { + return Err(invalid("physical frontier schema or boundedness mismatch")); + } + } + Ok(()) + } + pub fn encode(&self) -> Result, Error> { + self.validate()?; + let graph = |dag: &CompiledPhysicalDag| -> Result { + serde_json::from_slice(&dag.encode()?).map_err(|error| invalid(error.to_string())) + }; + serde_json::to_vec(&StoredCandidate { + version: 1, + precompute: self.precompute.as_ref().map(graph).transpose()?, + query: graph(&self.query)?, + materialized_outputs: self.materialized_outputs.clone(), + }) + .map_err(|error| invalid(error.to_string())) + } + /// Recover the selected physical candidate; no logical IR is accepted here. + pub fn decode(bytes: &[u8]) -> Result { + let stored: StoredCandidate = + serde_json::from_slice(bytes).map_err(|error| invalid(error.to_string()))?; + if stored.version != 1 { + return Err(invalid("unsupported physical candidate format")); + } + let graph = |value| -> Result { + CompiledPhysicalDag::decode( + &serde_json::to_vec(&value).map_err(|error| invalid(error.to_string()))?, + ) + }; + let candidate = Self { + precompute: stored.precompute.map(graph).transpose()?, + query: graph(stored.query)?, + materialized_outputs: stored.materialized_outputs, + }; + candidate.validate()?; + Ok(candidate) + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs index 0e2e1b61..3edbf536 100644 --- a/crates/asap-physical-operators/src/physical_planner/compiled.rs +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -2,7 +2,7 @@ use super::*; /// A typed execution boundary, without storage identity or a live reader. -#[derive(Clone, Debug)] +#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)] pub struct InputContract { pub schema: Schema, pub properties: PlanProperties, @@ -24,7 +24,7 @@ impl InputContract { } } } -#[derive(Clone)] +#[derive(Clone, serde::Serialize, serde::Deserialize)] enum Node { Input(InputContract), Operator { @@ -39,7 +39,44 @@ pub struct CompiledPhysicalDag { nodes: BTreeMap, roots: Vec, } +#[derive(serde::Serialize, serde::Deserialize)] +#[serde(deny_unknown_fields)] +struct StoredDag { + version: u32, + nodes: BTreeMap, + roots: Vec, +} + impl CompiledPhysicalDag { + /// Persist selected physical operators and input slots, never live readers + /// or mutable summary state. Recovery does not run logical plan lowering. + pub fn encode(&self) -> Result, Error> { + self.validate()?; + let bytes = serde_json::to_vec(&StoredDag { + version: 1, + nodes: self.nodes.clone(), + roots: self.roots.clone(), + }) + .map_err(|error| invalid(error.to_string()))?; + // JSON cannot preserve non-finite literal values. Fail at publication, + // rather than persisting a document that cannot be recovered. + Self::decode(&bytes)?; + Ok(bytes) + } + pub fn decode(bytes: &[u8]) -> Result { + let stored: StoredDag = + serde_json::from_slice(bytes).map_err(|error| invalid(error.to_string()))?; + if stored.version != 1 { + return Err(invalid("unsupported physical plan format")); + } + let result = Self { + nodes: stored.nodes, + roots: stored.roots, + }; + result.validate()?; + Ok(result) + } + /// Assemble already-lowered operators and typed external inputs. This is /// useful for engines that compose multiple compiled computation fragments. pub fn from_operators( diff --git a/crates/asap-physical-operators/src/plan/properties.rs b/crates/asap-physical-operators/src/plan/properties.rs index de1c7474..7ec770ea 100644 --- a/crates/asap-physical-operators/src/plan/properties.rs +++ b/crates/asap-physical-operators/src/plan/properties.rs @@ -1,5 +1,5 @@ //! Execution facts used to reject operators that cannot finish on their inputs. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug, PartialEq, Eq)] pub enum Boundedness { /// The source or operator promises a finite result for this run. Bounded, @@ -18,14 +18,14 @@ impl Boundedness { } } } -#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug, PartialEq, Eq)] pub enum Emission { Incremental, /// Produces its result only after all inputs end, even if accumulation is incremental. AfterInput, Unknown, } -#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug, PartialEq, Eq)] pub struct PlanProperties { pub boundedness: Boundedness, pub emission: Emission, diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs index d550a826..60dc1c70 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -7,7 +7,7 @@ use planner_types::{ }; use std::{cmp::Ordering, sync::Arc}; pub type Schema = Arc; -#[derive(Clone)] +#[derive(Clone, serde::Serialize, serde::Deserialize)] pub enum Value { Null, Bool(bool), @@ -24,6 +24,7 @@ pub enum Value { List(Arc<[Value]>), Struct(Arc<[Value]>), Map(Arc<[(Value, Value)]>), + #[serde(skip)] Summary { family: SummaryFamilyType, state: Arc, diff --git a/crates/asap-physical-operators/tests/physical_plan_recovery.rs b/crates/asap-physical-operators/tests/physical_plan_recovery.rs new file mode 100644 index 00000000..15a00ba5 --- /dev/null +++ b/crates/asap-physical-operators/tests/physical_plan_recovery.rs @@ -0,0 +1,102 @@ +//! Persisted physical plans recover selected operators without logical lowering. +use asap_physical_operators::{ + operators::{Operator, SortKey}, + physical_planner::{CompiledPhysicalDag, InputContract}, +}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, +}; +use std::{collections::BTreeMap, sync::Arc}; + +fn sorted() -> CompiledPhysicalDag { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }); + CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), + BTreeMap::from([( + 1, + ( + vec![0], + Operator::sort( + schema, + vec![SortKey { + column: 0, + descending: true, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + ), + )]), + vec![1], + ) + .unwrap() +} + +#[test] +fn recovery_retains_selected_operator_and_rejects_invalid_contracts() { + let bytes = sorted().encode().unwrap(); + let recovered = CompiledPhysicalDag::decode(&bytes).unwrap(); + assert_eq!(recovered.encode().unwrap(), bytes); + for mutation in ["version", "column", "edge", "output"] { + let mut wire: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + match mutation { + "version" => wire["version"] = 999.into(), + "column" => { + wire["nodes"]["1"]["Operator"]["operator"]["kind"]["Sort"]["keys"][0]["column"] = + 7.into() + } + "edge" => wire["nodes"]["1"]["Operator"]["inputs"][0] = 999.into(), + "output" => { + wire["nodes"]["1"]["Operator"]["operator"]["output"]["fields"][0]["dtype"] = + serde_json::json!({"Plain":"utf8"}) + } + _ => unreachable!(), + } + assert!( + CompiledPhysicalDag::decode(&serde_json::to_vec(&wire).unwrap()).is_err(), + "accepted {mutation}" + ); + } +} + +#[test] +fn candidate_recovery_preserves_materialization_boundary() { + use asap_physical_operators::physical_planner::PhysicalCandidate; + let precompute = sorted(); + let output = InputContract::bounded(precompute.output_contract(1).unwrap().schema); + let query = CompiledPhysicalDag::from_operators( + BTreeMap::from([(1, output.clone())]), + BTreeMap::from([( + 2, + ( + vec![1], + Operator::limit(output.schema.clone(), 3, 0, vec![]).unwrap(), + ), + )]), + vec![2], + ) + .unwrap(); + let candidate = PhysicalCandidate { + precompute: Some(precompute), + query, + materialized_outputs: BTreeMap::from([(1, output)]), + }; + let bytes = candidate.encode().unwrap(); + let restored = PhysicalCandidate::decode(&bytes).unwrap(); + assert_eq!(restored.precompute.as_ref().unwrap().roots(), &[1]); + assert_eq!(restored.query.roots(), &[2]); + assert_eq!(restored.encode().unwrap(), bytes); + let mut wire: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + wire["materialized_outputs"]["1"]["schema"]["fields"][0]["dtype"] = + serde_json::json!({"Plain":"utf8"}); + assert!(PhysicalCandidate::decode(&serde_json::to_vec(&wire).unwrap()).is_err()); +} diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index fc51e87f..39c12758 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -354,6 +354,9 @@ fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { &[u64::from(dag.root.0)], ) .unwrap(); + let bytes = raw_compiled.encode().unwrap(); + let raw_compiled = + asap_physical_operators::physical_planner::CompiledPhysicalDag::decode(&bytes).unwrap(); // Each evaluation receives a complete raw window. A reset, a stopped // series and an expired leader must not retain last run's heap weights. for (end, series, expected) in [ From c470366b596c947558266ee4c332e62f529724e3 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 20:52:25 +0000 Subject: [PATCH 60/90] feat: persist typed physical output batches with explicit summary codecs --- .../src/stored_state/mod.rs | 1 + .../src/stored_state/native.rs | 350 ++++++++++++++++++ 2 files changed, 351 insertions(+) create mode 100644 crates/asap-physical-operators/src/stored_state/native.rs diff --git a/crates/asap-physical-operators/src/stored_state/mod.rs b/crates/asap-physical-operators/src/stored_state/mod.rs index 0a3d0e7b..a2dbc24b 100644 --- a/crates/asap-physical-operators/src/stored_state/mod.rs +++ b/crates/asap-physical-operators/src/stored_state/mod.rs @@ -1,6 +1,7 @@ //! Portable stored-summary payloads and reconstruction, independent of storage engines. pub mod decoders; pub mod delta_apply; +pub mod native; pub mod readout; #[derive(Debug, Clone)] diff --git a/crates/asap-physical-operators/src/stored_state/native.rs b/crates/asap-physical-operators/src/stored_state/native.rs new file mode 100644 index 00000000..f5624821 --- /dev/null +++ b/crates/asap-physical-operators/src/stored_state/native.rs @@ -0,0 +1,350 @@ +//! Versioned physical output batches. Deployment identities and coverage remain +//! outside this payload and must be checked before decoding with the bound schema. +use crate::{ + summary_kernels::{ + datasketches_kll::DatasketchesKLLAccumulator, dd_sketch::DDSketchAccumulator, + exact::ExactAccumulator, hll_sketch::HllSketchAccumulator, + weighted_frequency::WeightedFrequency, SumAccumulator, + }, + values::{Batch, Schema, Value}, + AggregateCore, Error, +}; +use planner_types::post_asap::{SummaryFamilyType, SummarySchema}; +use serde::{Deserialize, Serialize}; +use std::sync::Arc; + +#[derive(Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct StoredBatch { + version: u32, + schema: SummarySchema, + rows: Vec>, +} + +#[derive(Serialize, Deserialize)] +enum Cell { + Plain(Value), + Summary { + family: SummaryFamilyType, + codec: StateCodec, + bytes: Vec, + }, +} + +// Codec identity is distinct from algorithm identity: old integer CMS/CS bytes +// must never be interpreted as Float64 weighted state with typed item tuples. +#[derive(Serialize, Deserialize)] +enum StateCodec { + WeightedFrequencyV1, + ExactAccumulatorV1, + SumAccumulatorV1, + KllMsgpackV1, + DdMsgpackV1, + HllMsgpackV1, +} +fn invalid(message: impl ToString) -> Error { + Error::Invalid(message.to_string()) +} +impl StateCodec { + fn for_state(state: &dyn AggregateCore) -> Result { + let state = state.as_any(); + if state.is::() { + Ok(Self::WeightedFrequencyV1) + } else if state.is::() { + Ok(Self::ExactAccumulatorV1) + } else if state.is::() { + Ok(Self::SumAccumulatorV1) + } else if state.is::() { + Ok(Self::KllMsgpackV1) + } else if state.is::() { + Ok(Self::DdMsgpackV1) + } else if state.is::() { + Ok(Self::HllMsgpackV1) + } else { + Err(invalid("physical summary has no persisted native codec")) + } + } + fn decode(&self, bytes: &[u8]) -> Result, Error> { + Ok(match self { + Self::WeightedFrequencyV1 => Arc::new(WeightedFrequency::from_bytes(bytes)?), + Self::ExactAccumulatorV1 => { + Arc::new(ExactAccumulator::deserialize_from_bytes(bytes).map_err(invalid)?) + } + Self::SumAccumulatorV1 => { + Arc::new(SumAccumulator::deserialize_from_bytes(bytes).map_err(invalid)?) + } + Self::KllMsgpackV1 => { + Arc::new(DatasketchesKLLAccumulator::from_msgpack_bytes(bytes).map_err(invalid)?) + } + Self::DdMsgpackV1 => { + Arc::new(DDSketchAccumulator::from_msgpack_bytes(bytes).map_err(invalid)?) + } + Self::HllMsgpackV1 => { + Arc::new(HllSketchAccumulator::from_msgpack_bytes(bytes).map_err(invalid)?) + } + }) + } +} + +/// Encode a validated physical output, preserving Float64 and typed identities. +/// This format is independent of the logical and physical plan wire formats. +pub fn encode_batch(batch: &Batch) -> Result, Error> { + let rows = batch + .rows() + .iter() + .map(|row| { + row.iter() + .map(|value| { + Ok(match value { + Value::Summary { family, state } => Cell::Summary { + family: family.clone(), + codec: StateCodec::for_state(state.as_ref())?, + bytes: state.serialize_to_bytes(), + }, + value => Cell::Plain(value.clone()), + }) + }) + .collect::, Error>>() + }) + .collect::, Error>>()?; + rmp_serde::to_vec_named(&StoredBatch { + version: 1, + schema: batch.schema().as_ref().clone(), + rows, + }) + .map_err(invalid) +} + +/// Decode only against the installed output contract. The caller supplies its +/// per-read payload limit; checking state parameters is part of Batch validation. +pub fn decode_batch(bytes: &[u8], expected: Schema, max_bytes: usize) -> Result { + if bytes.len() > max_bytes { + return Err(invalid("native output payload exceeds read budget")); + } + let stored: StoredBatch = rmp_serde::from_slice(bytes).map_err(invalid)?; + if stored.version != 1 { + return Err(invalid("unsupported native output format")); + } + if stored.schema != *expected { + return Err(invalid( + "native output schema differs from installed contract", + )); + } + let rows = stored + .rows + .into_iter() + .map(|row| { + row.into_iter() + .map(|cell| { + Ok(match cell { + Cell::Plain(value) => value, + Cell::Summary { + family, + codec, + bytes, + } => Value::Summary { + family, + state: codec.decode(&bytes)?, + }, + }) + }) + .collect::, Error>>() + }) + .collect::, Error>>()?; + Batch::try_new(expected, rows) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::summary_kernels::weighted_frequency::FrequencyAlgorithm; + use planner_types::{ + post_asap::{SketchAlgorithm, SketchKind, SketchParams, SummaryField}, + pre_asap::DataType, + }; + + fn weighted(algorithm: SketchAlgorithm) -> Batch { + let (native, params) = match algorithm { + SketchAlgorithm::CmsWithHeap => ( + FrequencyAlgorithm::Cms, + SketchParams::CmsWithHeap { + width: 64, + depth: 5, + heap_size: 8, + }, + ), + _ => ( + FrequencyAlgorithm::CountSketch, + SketchParams::CountSketchWithHeap { + width: 64, + depth: 5, + heap_size: 8, + }, + ), + }; + let family = + SummaryFamilyType::Sketch(SketchKind::new(algorithm, params), Default::default()); + let schema = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "group".into(), + dtype: SummaryFamilyType::Plain(DataType::Utf8), + nullable: false, + }, + SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }, + ], + time_index: None, + }); + let mut state = WeightedFrequency::new(native, 64, 5, 8).unwrap(); + state + .update(&[Value::Int64(7), Value::Utf8("service-a".into())], 0.125) + .unwrap(); + state + .update( + &[Value::Utf8("7".into()), Value::Utf8("service-b".into())], + 0.25, + ) + .unwrap(); + Batch::try_new( + schema, + vec![vec![ + Value::Utf8("job-a".into()), + Value::Summary { + family, + state: Arc::new(state), + }, + ]], + ) + .unwrap() + } + + // Fractional rates and distinct typed item tuples survive both heap codecs. + #[test] + fn weighted_outputs_roundtrip_without_integer_conversion() { + for algorithm in [ + SketchAlgorithm::CmsWithHeap, + SketchAlgorithm::CountSketchWithHeap, + ] { + let batch = weighted(algorithm); + let bytes = encode_batch(&batch).unwrap(); + let restored = decode_batch(&bytes, batch.schema().clone(), bytes.len()).unwrap(); + let scores = |b: &Batch| { + let Value::Summary { state, .. } = &b.rows()[0][1] else { + panic!() + }; + state + .as_any() + .downcast_ref::() + .unwrap() + .rows(8) + .iter() + .map(|row| row.iter().map(|v| v.key().unwrap()).collect::>()) + .collect::>() + }; + assert_eq!(scores(&batch), scores(&restored)); + assert!(decode_batch(&bytes, batch.schema().clone(), bytes.len() - 1).is_err()); + } + } + + // Every admitted native summary codec survives the same typed boundary. + #[test] + fn native_summary_families_and_nonfinite_plain_values_roundtrip() { + use planner_types::post_asap::{ExactKind, ExactParams}; + let exact = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let sketch = |algorithm, params| { + SummaryFamilyType::Sketch(SketchKind::new(algorithm, params), Default::default()) + }; + let cases: Vec<(SummaryFamilyType, Arc)> = vec![ + ( + exact.clone(), + Arc::new(ExactAccumulator::new(exact.clone(), false).unwrap()), + ), + (exact, Arc::new(SumAccumulator::new())), + ( + sketch(SketchAlgorithm::Kll, SketchParams::Kll { k: 200 }), + Arc::new(DatasketchesKLLAccumulator::new(200)), + ), + ( + sketch( + SketchAlgorithm::DDSketch, + SketchParams::DDSketch { alpha: 0.01 }, + ), + Arc::new(DDSketchAccumulator::new(0.01)), + ), + ( + sketch(SketchAlgorithm::Hll, SketchParams::Hll { precision: 12 }), + Arc::new(HllSketchAccumulator::new( + asap_sketchlib::HllVariant::Regular, + 12, + )), + ), + ]; + for (family, state) in cases { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }], + time_index: None, + }); + let batch = + Batch::try_new(schema.clone(), vec![vec![Value::Summary { family, state }]]) + .unwrap(); + let bytes = encode_batch(&batch).unwrap(); + let restored = decode_batch(&bytes, schema, bytes.len()).unwrap(); + assert_eq!(encode_batch(&restored).unwrap(), bytes); + } + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }); + let batch = Batch::try_new( + schema.clone(), + vec![ + vec![Value::Float64(f64::NAN)], + vec![Value::Float64(f64::INFINITY)], + ], + ) + .unwrap(); + let restored = decode_batch(&encode_batch(&batch).unwrap(), schema, usize::MAX).unwrap(); + assert!(matches!(restored.rows()[0][0], Value::Float64(v) if v.is_nan())); + assert!(matches!(restored.rows()[1][0], Value::Float64(v) if v == f64::INFINITY)); + } + + // Recovery validates the format, bound schema and actual sketch parameters. + #[test] + fn corrupt_or_relabelled_output_is_rejected() { + let batch = weighted(SketchAlgorithm::CmsWithHeap); + let bytes = encode_batch(&batch).unwrap(); + let wrong = weighted(SketchAlgorithm::CountSketchWithHeap); + assert!(decode_batch(&bytes, wrong.schema().clone(), usize::MAX).is_err()); + let mut stored: StoredBatch = rmp_serde::from_slice(&bytes).unwrap(); + stored.version = 2; + assert!(decode_batch( + &rmp_serde::to_vec_named(&stored).unwrap(), + batch.schema().clone(), + usize::MAX + ) + .is_err()); + stored.version = 1; + let Cell::Summary { bytes: payload, .. } = &mut stored.rows[0][1] else { + panic!() + }; + *payload = b"legacy integer heap".to_vec(); + assert!(decode_batch( + &rmp_serde::to_vec_named(&stored).unwrap(), + batch.schema().clone(), + usize::MAX + ) + .is_err()); + } +} From 5aee349f0a46a51c231a05cdee9945cae56ce08d Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 21:01:56 +0000 Subject: [PATCH 61/90] fix: reject native physical output frames in legacy sketch readers --- .../src/stored_state/delta_apply.rs | 3 +++ .../src/stored_state/mod.rs | 2 ++ .../src/stored_state/native.rs | 18 ++++++++++++++++++ 3 files changed, 23 insertions(+) diff --git a/crates/asap-physical-operators/src/stored_state/delta_apply.rs b/crates/asap-physical-operators/src/stored_state/delta_apply.rs index e1381b40..3f638503 100644 --- a/crates/asap-physical-operators/src/stored_state/delta_apply.rs +++ b/crates/asap-physical-operators/src/stored_state/delta_apply.rs @@ -667,6 +667,9 @@ fn visit_window_summary_states( } match state.encoding { + SketchEncoding::NativeBatchV1 => { + return Err("native physical outputs require the bound native batch decoder".into()) + } SketchEncoding::ProtoFull | SketchEncoding::MsgpackFull => { // A Full (re)sets this window's base. rolling = Some(decode_full(&kind, &state.bytes, state.encoding)?); diff --git a/crates/asap-physical-operators/src/stored_state/mod.rs b/crates/asap-physical-operators/src/stored_state/mod.rs index a2dbc24b..89ab9578 100644 --- a/crates/asap-physical-operators/src/stored_state/mod.rs +++ b/crates/asap-physical-operators/src/stored_state/mod.rs @@ -17,4 +17,6 @@ pub enum SketchEncoding { ProtoDelta, MsgpackFull, MsgpackDelta, + /// Versioned typed physical output; never a legacy sketch frame. + NativeBatchV1, } diff --git a/crates/asap-physical-operators/src/stored_state/native.rs b/crates/asap-physical-operators/src/stored_state/native.rs index f5624821..4cf0e41b 100644 --- a/crates/asap-physical-operators/src/stored_state/native.rs +++ b/crates/asap-physical-operators/src/stored_state/native.rs @@ -250,6 +250,24 @@ mod tests { } } + // A storage tag cannot send weighted physical output through an integer heap decoder. + #[test] + fn legacy_sketch_reader_rejects_native_batch_frames() { + let sample = super::super::SketchSampleState { + bytes: encode_batch(&weighted(SketchAlgorithm::CmsWithHeap)).unwrap(), + encoding: super::super::SketchEncoding::NativeBatchV1, + }; + let result = super::super::delta_apply::per_window_summary_states( + &[(60_000, &sample)], + super::super::delta_apply::DeltaSketchKind::CmsWithHeap { + rows: 5, + cols: 64, + heap_size: 8, + }, + ); + assert!(result.is_err()); + } + // Every admitted native summary codec survives the same typed boundary. #[test] fn native_summary_families_and_nonfinite_plain_values_roundtrip() { From 0b0ba603e560d05d555a193a04ab78551c5c5b6f Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 22:16:03 +0000 Subject: [PATCH 62/90] feat: preserve dynamic series identity through rate and snapshot heaps --- .../src/operators/current_series.rs | 135 +++++++++ .../src/operators/mod.rs | 11 + .../src/operators/persisted.rs | 6 + .../src/physical_planner/mod.rs | 2 + .../src/physical_planner/promql_rows.rs | 144 +++++++++ .../tests/current_series_heap.rs | 282 ++++++++++++++++++ .../tests/weighted_topk_binding.rs | 67 ++++- docs/develop_docs/native-promql-inputs.md | 42 +++ 8 files changed, 687 insertions(+), 2 deletions(-) create mode 100644 crates/asap-physical-operators/src/operators/current_series.rs create mode 100644 crates/asap-physical-operators/src/physical_planner/promql_rows.rs create mode 100644 crates/asap-physical-operators/tests/current_series_heap.rs create mode 100644 docs/develop_docs/native-promql-inputs.md diff --git a/crates/asap-physical-operators/src/operators/current_series.rs b/crates/asap-physical-operators/src/operators/current_series.rs new file mode 100644 index 00000000..937036c5 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/current_series.rs @@ -0,0 +1,135 @@ +//! A bounded instant-vector snapshot: select latest before removing stale markers. +use super::*; + +impl Operator { + pub fn current_series( + input: Schema, + identity: usize, + coordinate: usize, + value: usize, + lookback_ms: i64, + ) -> Result { + if lookback_ms <= 0 + || plain(&input, identity)? != (&DataType::Utf8, false) + || plain(&input, coordinate)? != (&DataType::Timestamp, false) + || plain(&input, value)? != (&DataType::Float64, false) + { + return Err(invalid("invalid current-series input contract")); + } + Ok(Self { + kind: Kind::CurrentSeries { + identity, + coordinate, + value, + lookback_ms, + }, + inputs: vec![input.clone()], + output: input, + }) + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::CurrentSeries { + identity, + coordinate, + value, + lookback_ms, + } = operator.kind + else { + unreachable!() + }; + let input = inputs + .pop() + .ok_or_else(|| invalid("current-series input missing"))?; + let output = operator.output.clone(); + let (start, end) = window(lookback_ms, &context)?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let mut latest = BTreeMap::, usize>::new(); + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + for (index, row) in rows.iter().enumerate() { + work.checkpoint().await?; + let Value::Timestamp(timestamp) = row[coordinate] else { + unreachable!() + }; + if timestamp <= start || timestamp > end { + continue; + } + let key = row[identity].key()?; + if let Some(&previous) = latest.get(&key) { + let Value::Timestamp(previous_time) = rows[previous][coordinate] else { + unreachable!() + }; + if timestamp < previous_time { + continue; + } + if timestamp == previous_time { + let (Value::Float64(a), Value::Float64(b)) = + (&row[value], &rows[previous][value]) + else { + unreachable!() + }; + if a.to_bits() != b.to_bits() { + return Err(invalid("conflicting samples for one series timestamp")); + } + continue; + } + } else { + workspace.grow(64 + key.len())?; + } + latest.insert(key, index); + } + let mut result = Vec::new(); + for index in latest.into_values() { + work.checkpoint().await?; + let Value::Float64(sample) = rows[index][value] else { + unreachable!() + }; + if sample.to_bits() == 0x7ff0_0000_0000_0002 { + continue; + } + workspace.grow(row_bytes(&rows[index]))?; + let mut row = rows[index].clone(); + row[coordinate] = Value::Timestamp(end); + result.push(row); + } + Batch::try_new(output, result) + }) + .boxed_local()) +} + +fn window(lookback_ms: i64, context: &RunContext) -> Result<(i64, i64), Error> { + let end = match context.scope { + crate::runtime::Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + crate::runtime::Scope::Ingestion { window_end_ms, .. } => window_end_ms, + }; + let start = end + .checked_sub(lookback_ms) + .ok_or_else(|| invalid("current-series window overflows"))?; + if let crate::runtime::Scope::Ingestion { + window_start_ms, .. + } = context.scope + { + if window_start_ms != start { + return Err(invalid( + "current-series maintenance window differs from lookback", + )); + } + } + Ok((start, end)) +} + +pub(super) fn validate_context(operator: &Operator, context: &RunContext) -> Result<(), Error> { + if let Kind::CurrentSeries { lookback_ms, .. } = operator.kind { + window(lookback_ms, context)?; + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index 7c38835e..0aba8b86 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -16,6 +16,7 @@ use crate::expressions::ordered; pub use crate::expressions::Expression; use common::*; mod aggregate; +mod current_series; mod filter; mod joins; mod limit; @@ -39,6 +40,12 @@ enum Kind { ScopeTimestamp { columns: Vec>, }, + CurrentSeries { + identity: usize, + coordinate: usize, + value: usize, + lookback_ms: i64, + }, Union, VectorToScalar { column: usize, @@ -193,6 +200,7 @@ impl PhysicalOperator for Operator { matches!( self.kind, Kind::Sort { .. } + | Kind::CurrentSeries { .. } | Kind::Aggregate { .. } | Kind::Window { .. } | Kind::Join { .. } @@ -232,6 +240,7 @@ impl PhysicalOperator for Operator { Kind::PaneInput { .. } => "PaneInput", Kind::ScopeTimestamp { .. } => "ScopeTimestamp", Kind::Union => "Union", + Kind::CurrentSeries { .. } => "CurrentSeries", Kind::VectorToScalar { .. } => "VectorToScalar", Kind::Project(_) => "Project", Kind::Filter(_) => "Filter", @@ -249,6 +258,7 @@ impl PhysicalOperator for Operator { } fn validate_context(&self, context: &RunContext) -> Result<(), Error> { panes::validate_context(self, context)?; + current_series::validate_context(self, context)?; self.readout_parameters(context).map(|_| ()) } fn input_schemas(&self) -> Vec { @@ -270,6 +280,7 @@ impl PhysicalOperator for Operator { source::execute(self, inputs, context) } Kind::Project(_) => projection::execute(self, inputs, context), + Kind::CurrentSeries { .. } => current_series::execute(self, inputs, context), Kind::PaneInput { .. } | Kind::ScopeTimestamp { .. } => { panes::execute(self, inputs, context) } diff --git a/crates/asap-physical-operators/src/operators/persisted.rs b/crates/asap-physical-operators/src/operators/persisted.rs index 17d05061..85add57a 100644 --- a/crates/asap-physical-operators/src/operators/persisted.rs +++ b/crates/asap-physical-operators/src/operators/persisted.rs @@ -36,6 +36,12 @@ impl TryFrom for Operator { } => Operator::pane_input(input(0)?, coordinate, layout, offset_ms)?, Kind::ScopeTimestamp { .. } => Operator::scope_timestamp(input(0)?, output.clone())?, Kind::Union => Operator::union(input(0)?, inputs.len())?, + Kind::CurrentSeries { + identity, + coordinate, + value, + lookback_ms, + } => Operator::current_series(input(0)?, identity, coordinate, value, lookback_ms)?, Kind::VectorToScalar { column } => Operator::vector_to_scalar(input(0)?, column)?, Kind::Project(expressions) => { if expressions.len() != output.fields.len() { diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 09b9bb7e..012e5941 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -28,6 +28,8 @@ fn invalid(message: impl Into) -> Error { /// A deployment must authorize these frontiers before calling this function. pub type Source<'a> = Box + 'a>; +pub mod promql_rows; + mod candidates; pub use candidates::{ compile_candidate, compile_candidates, enumerate_frontiers, select_candidate, CandidateCost, diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs new file mode 100644 index 00000000..f8ab7d40 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -0,0 +1,144 @@ +//! A bounded PromQL source row carries the entire label set, not just labels +//! mentioned by the query. The source adapter owns this lossless encoding. +use super::*; +use planner_types::pre_asap::{Column, DataType, Source as LogicalSource}; +use std::rc::Rc; + +/// Not a legal PromQL label name, so it cannot shadow a user label. +pub const SERIES_IDENTITY_COLUMN: &str = "$promql_series_identity"; + +/// Canonical, reversible identity. JSON object encoding preserves label names, +/// empty values and escaping; sorting makes ingestion order irrelevant. +pub fn encode_series_identity(labels: &BTreeMap) -> Result { + serde_json::to_string(labels).map_err(|error| invalid(error.to_string())) +} + +pub fn decode_series_identity(encoded: &str) -> Result, Error> { + let labels: BTreeMap = + serde_json::from_str(encoded).map_err(|error| invalid(error.to_string()))?; + if encode_series_identity(&labels)? != encoded { + return Err(invalid("series identity is not canonically encoded")); + } + Ok(labels) +} + +/// Resolve the row representation before candidate search. `closed` describes +/// physical columns here: the final column contains every dynamic source label. +/// It does not assert that the query's projected labels are the full label set. +/// +/// This realization supports explicit `by` grouping and per-series computation. +/// Operators that rewrite or implicitly match dynamic label sets require their +/// own realization; they must not accidentally treat the opaque identity as a +/// user label or silently discard it. +pub fn with_series_identity(root: &QueryExpr) -> Result { + let mut root = root.clone(); + fn visit(node: &mut QueryExpr) -> Result<(), Error> { + use planner_types::pre_asap::Reduction; + match node { + QueryExpr::Scan { + source: LogicalSource::TimeSeries { .. }, + schema, + .. + } => { + if schema + .columns + .iter() + .any(|column| column.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "source already contains a physical series identity", + )); + } + if schema.closed { + return Err(invalid( + "dynamic series identity requires an open PromQL source", + )); + } + schema + .columns + .push(Column::new(SERIES_IDENTITY_COLUMN, DataType::Utf8, false)); + schema.closed = true; + Ok(()) + } + QueryExpr::TimeRange { child, .. } | QueryExpr::Limit { child, .. } => { + visit(Rc::make_mut(child)) + } + QueryExpr::Aggregate { + child, reduction, .. + } => { + if matches!(reduction, Reduction::Reduce(keys) if keys.is_without()) { + return Err(invalid( + "dynamic without grouping requires label-set projection", + )); + } + visit(Rc::make_mut(child)) + } + QueryExpr::Sort { + child, + partition_by, + .. + } => { + if partition_by.is_without() { + return Err(invalid( + "dynamic without ranking requires label-set projection", + )); + } + visit(Rc::make_mut(child)) + } + _ => Err(invalid( + "operator has no dynamic series-identity realization", + )), + } + } + visit(&mut root)?; + root.output_schema() + .map_err(|error| invalid(error.to_string()))?; + Ok(root) +} + +/// Construct source rows only from full identities. The named label columns +/// are projections of that same identity and cannot independently redefine it. +pub fn series_row( + schema: &Schema, + labels: &BTreeMap, + timestamp: i64, + value: f64, +) -> Result, Error> { + use crate::values::Value; + let identity = encode_series_identity(labels)?; + let mut found = false; + let row = schema + .fields + .iter() + .enumerate() + .map(|(index, field)| { + if field.name == SERIES_IDENTITY_COLUMN { + if field.dtype != SummaryFamilyType::Plain(DataType::Utf8) + || field.nullable + || found + { + return Err(invalid("invalid series identity column")); + } + found = true; + Ok(Value::Utf8(identity.clone().into())) + } else if Some(index) == schema.time_index { + Ok(Value::Timestamp(timestamp)) + } else if field.name == "value" + && field.dtype == SummaryFamilyType::Plain(DataType::Float64) + { + Ok(Value::Float64(value)) + } else if field.dtype == SummaryFamilyType::Plain(DataType::Utf8) { + Ok(labels.get(&field.name).map_or_else( + || Value::Utf8("".into()), + |value| Value::Utf8(value.clone().into()), + )) + } else { + Err(invalid("unsupported PromQL source column")) + } + }) + .collect::, _>>()?; + if !found { + return Err(invalid("source lacks its full series identity")); + } + Ok(row) +} diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs new file mode 100644 index 00000000..0a7059e7 --- /dev/null +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -0,0 +1,282 @@ +//! Spatial heap weights come from a fresh instant vector, never sample history. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{ + promql_rows::{decode_series_identity, series_row, SERIES_IDENTITY_COLUMN}, + CompiledPhysicalDag, InputContract, Source, + }, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{post_asap::*, pre_asap::DataType}; +use std::{collections::BTreeMap, sync::Arc}; + +fn schema() -> Arc { + Arc::new(SummarySchema { + fields: [ + ("ts", DataType::Timestamp), + ("value", DataType::Float64), + ("job", DataType::Utf8), + (SERIES_IDENTITY_COLUMN, DataType::Utf8), + ] + .into_iter() + .map(|(name, dtype)| SummaryField { + name: name.into(), + dtype: SummaryFamilyType::Plain(dtype), + nullable: false, + }) + .collect(), + time_index: Some(0), + }) +} +fn run(program: &CompiledPhysicalDag, data: Batch, end: i64) -> Result, String> { + let recovered = CompiledPhysicalDag::decode(&program.encode().map_err(|e| e.to_string())?) + .map_err(|e| e.to_string())?; + let graph = recovered + .instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(data.schema().clone(), vec![data]).unwrap()) as Source<'_>, + )])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: end, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = graph.execute(recovered.roots(), context).unwrap().remove(0); + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push((*batch.map_err(|e| e.to_string())?).clone()); + } + Ok(batches) + }) +} +fn input(samples: &[(&str, i64, f64)]) -> Batch { + let schema = schema(); + let rows = samples + .iter() + .map(|(instance, time, value)| { + series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("hidden_instance".into(), (*instance).into()), + ]), + *time, + *value, + ) + .unwrap() + }) + .collect(); + Batch::try_new(schema, rows).unwrap() +} +fn snapshot_plan() -> CompiledPhysicalDag { + CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema()))]), + BTreeMap::from([( + 1, + ( + vec![0], + Operator::current_series(schema(), 3, 0, 1, 60_000).unwrap(), + ), + )]), + vec![1], + ) + .unwrap() +} + +// Replacement, expiry and stale markers act before sketch updates. Hidden labels +// survive even when every series has the same projected `job` value. +#[test] +fn latest_snapshot_replaces_decreases_expires_and_retains_full_identity() { + let plan = snapshot_plan(); + let batches = run( + &plan, + input(&[ + ("decrease", 10_000, 100.), + ("decrease", 50_000, 1.), + ("steady", 40_000, 20.), + ("expired", 0, 1_000.), + ("stale", 20_000, 500.), + ("stale", 55_000, f64::from_bits(0x7ff0_0000_0000_0002)), + ("future", 60_001, 2_000.), + ]), + 60_000, + ) + .unwrap(); + let values = batches + .iter() + .flat_map(|batch| batch.rows()) + .map(|row| { + let Value::Utf8(identity) = &row[3] else { + panic!() + }; + let Value::Float64(value) = row[1] else { + panic!() + }; + assert!(matches!(row[0], Value::Timestamp(60_000))); + ( + decode_series_identity(identity).unwrap()["hidden_instance"].clone(), + value, + ) + }) + .collect::>(); + assert_eq!( + values, + BTreeMap::from([("decrease".into(), 1.), ("steady".into(), 20.)]) + ); + assert!(run(&plan, input(&[("steady", 40_000, 20.)]), 100_000) + .unwrap() + .iter() + .all(|batch| batch.rows().is_empty())); + assert!(run( + &plan, + input(&[("conflict", 50_000, 1.), ("conflict", 50_000, 2.)]), + 60_000 + ) + .is_err()); +} + +#[test] +fn spatial_heap_ranks_latest_values_in_independent_runs() { + for algorithm in [ + SketchAlgorithm::CmsWithHeap, + SketchAlgorithm::CountSketchWithHeap, + ] { + let params = match algorithm { + SketchAlgorithm::CmsWithHeap => SketchParams::CmsWithHeap { + width: 2048, + depth: 5, + heap_size: 100, + }, + _ => SketchParams::CountSketchWithHeap { + width: 2048, + depth: 5, + heap_size: 100, + }, + }; + let family = + SummaryFamilyType::Sketch(SketchKind::new(algorithm, params), Default::default()); + let build = Operator::keyed_summary_build(schema(), family, 1, vec![3], vec![2]).unwrap(); + let output = Arc::new(SummarySchema { + fields: vec![ + schema().fields[2].clone(), + schema().fields[3].clone(), + schema().fields[1].clone(), + ], + time_index: None, + }); + let read = Operator::keyed_readout(build.schema(), 1, 1, output).unwrap(); + let plan = CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema()))]), + BTreeMap::from([ + ( + 1, + ( + vec![0], + Operator::current_series(schema(), 3, 0, 1, 60_000).unwrap(), + ), + ), + (2, (vec![1], build)), + (3, (vec![2], read)), + ]), + vec![3], + ) + .unwrap(); + for (samples, end, winner, score) in [ + ( + vec![("a", 10_000, 100.), ("a", 50_000, 1.), ("b", 50_000, 20.)], + 60_000, + "b", + 20., + ), + ( + vec![("a", 110_000, 3.), ("b", 50_000, 20.)], + 120_000, + "a", + 3., + ), + ] { + let batches = run(&plan, input(&samples), end).unwrap(); + let rows = batches + .iter() + .flat_map(|batch| batch.rows()) + .collect::>(); + assert_eq!(rows.len(), 1); + let Value::Utf8(encoded) = &rows[0][1] else { + panic!() + }; + assert_eq!( + decode_series_identity(encoded).unwrap()["hidden_instance"], + winner + ); + assert!(matches!(rows[0][2], Value::Float64(actual) if actual == score)); + } + } +} + +// Blocking membership selection shares the run's cancellation and byte budget. +#[test] +fn current_series_observes_resource_limits() { + use asap_physical_operators::Error; + let plan = snapshot_plan(); + for cancelled in [false, true] { + let data = input(&[("one", 50_000, 1.)]); + let graph = plan + .instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(data.schema().clone(), vec![data]).unwrap()) + as Source<'_>, + )])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 60_000, + revision: 0, + }, + Limits { + max_bytes: if cancelled { 1 << 20 } else { 1 }, + ..Limits::default() + }, + ) + .unwrap(); + if cancelled { + context.cancel(); + } + let result = match graph.execute(&[1], context.clone()) { + Err(error) => Err(error), + Ok(mut streams) => block_on(streams.remove(0).next()).unwrap().map(|_| ()), + }; + assert!(matches!( + (cancelled, result), + (true, Err(Error::Cancelled)) | (false, Err(Error::MemoryLimit)) + )); + assert_eq!(context.retained_bytes(), 0); + } +} + +#[test] +fn identity_encoding_is_lossless_and_rejects_noncanonical_inputs() { + use asap_physical_operators::physical_planner::promql_rows::encode_series_identity; + let labels = BTreeMap::from([ + ("a".into(), "quote\"slash\\".into()), + ("other".into(), "".into()), + ]); + assert_eq!( + decode_series_identity(&encode_series_identity(&labels).unwrap()).unwrap(), + labels + ); + for invalid in [ + "[]", + "{\"a\":1}", + "{\"a\":\"x\",\"a\":\"x\"}", + "{ \"a\":\"x\"}", + ] { + assert!(decode_series_identity(invalid).is_err(), "{invalid}"); + } +} diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index 39c12758..269139c6 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -269,6 +269,19 @@ fn rate_updates_cannot_enter_integer_heap_factory() { /// requiring an otherwise unnecessary grouped Sum between Rate and TopK. #[test] fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { + check_direct_rate_topk(false); +} + +// Unreferenced labels still distinguish series throughout Rate and heap readout. +#[test] +fn direct_rate_topk_preserves_dynamic_unreferenced_labels() { + check_direct_rate_topk(true); +} + +fn check_direct_rate_topk(dynamic: bool) { + use asap_physical_operators::physical_planner::promql_rows::{ + decode_series_identity, series_row, with_series_identity, SERIES_IDENTITY_COLUMN, + }; let mut logical = lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(); fn resolve_catalog(node: &mut QueryExpr) { @@ -289,7 +302,11 @@ fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { _ => panic!("unexpected input shape: {node:?}"), } } - resolve_catalog(&mut logical); + if dynamic { + logical = with_series_identity(&logical).unwrap(); + } else { + resolve_catalog(&mut logical); + } let root = Rc::new(logical); let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( &DefaultCostModel, @@ -381,6 +398,22 @@ fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { let mut raw_rows = Vec::new(); for (service, samples) in series { for (offset, value) in [10_000, 30_000, 50_000].into_iter().zip(samples) { + if dynamic { + raw_rows.push( + series_row( + &raw_schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("service".into(), service.into()), + ("unreferenced".into(), format!("{service}-extra")), + ]), + end - 60_000 + offset, + value, + ) + .unwrap(), + ); + continue; + } raw_rows.push( raw_schema .fields @@ -422,7 +455,25 @@ fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { .unwrap() .remove(0); while let Some(batch) = stream.next().await { - for row in batch.unwrap().rows() { + let batch = batch.unwrap(); + for row in batch.rows() { + if dynamic { + let column = batch + .schema() + .fields + .iter() + .position(|field| field.name == SERIES_IDENTITY_COLUMN) + .unwrap(); + let Value::Utf8(encoded) = &row[column] else { + panic!("identity lost"); + }; + let labels = decode_series_identity(encoded).unwrap(); + assert_eq!(labels["job"], "api"); + assert_eq!( + labels["unreferenced"], + format!("{}-extra", labels["service"]) + ); + } assert!(row.iter().any( |value| matches!(value, Value::Timestamp(time) if *time == end) )); @@ -469,6 +520,18 @@ fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { let rows = values .into_iter() .map(|(service, value)| { + if dynamic { + return series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("service".into(), service.into()), + ]), + time, + value, + ) + .unwrap(); + } schema .fields .iter() diff --git a/docs/develop_docs/native-promql-inputs.md b/docs/develop_docs/native-promql-inputs.md new file mode 100644 index 00000000..66b83ae6 --- /dev/null +++ b/docs/develop_docs/native-promql-inputs.md @@ -0,0 +1,42 @@ +# Native PromQL source rows + +Audience: source-adapter and physical-executor developers. + +A PromQL query only names some labels. Those columns cannot establish series +identity for Rate or TopK: two series with the same `job` may have different +unreferenced instance labels. + +`physical_planner::promql_rows::with_series_identity` resolves supported unary +PromQL computations to a bounded row representation before candidate search. +It appends `$promql_series_identity`, a non-null UTF-8 column containing the +canonical JSON encoding of the full label map. The name cannot collide with a +legal PromQL label. The resulting schema is closed over physical columns; the +label map remains dynamic and is not restricted to labels named in the query. +This realization rejects unsupported label rewriting, implicit vector matching, +and `without` operations rather than dropping hidden labels. + +Source adapters construct batches with `series_row`. Named label columns are +projections of the same complete identity; absent named labels project to empty +strings. `decode_series_identity` restores all labels on result conversion and +rejects noncanonical encodings. A query adapter must still apply the selected +operator's metric-name/result-label rules. Source selection, complete window +coverage and revision admission remain deployment responsibilities. + +The native `CurrentSeries` operator selects the latest sample per complete +identity in `(evaluation_time - lookback, evaluation_time]`. It removes stale +markers after selecting the latest sample, so an older value cannot reappear. +It rejects conflicting values at one series timestamp and emits the evaluation +timestamp. Each run builds a new snapshot; decreased values and expired series +cannot retain earlier heap weights. It reserves workspace and observes the +run's cancellation and byte budget. Precompute scopes must match the declared +lookback before any input is polled. + +CMS/CountSketch heap operators can consume this snapshot. CMS still requires +nonnegative weights; legal approximate TopK admission still requires the +Planner's accuracy/membership evidence. Executing a heap does not establish +that its result satisfies a query's accuracy requirements. + +Tests cover open-label Rate → CMS/CountSketch heaps, hidden-label round trips, +reset and zero-rate cases, snapshot replacement/decrease/expiry/staleness, +serialized physical recovery, and resource rejection. These are shared-library +tests, not proof of Backend candidate selection or durable deployment execution. From 74234d42faaa8e0d5c335845b87e0ad64b71773d Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 22:28:25 +0000 Subject: [PATCH 63/90] feat: compile current-series population boundaries into physical readouts --- .../src/maintained_population.rs | 7 +- .../src/physical_planner/mod.rs | 99 ++++++++++++++- .../src/physical_planner/promql_rows.rs | 89 ++++++++++++- .../tests/current_series_heap.rs | 119 +++++++++++++++++- .../src/post_asap/maintained_population.rs | 2 +- crates/types/src/pre_asap/schema.rs | 14 +++ docs/develop_docs/native-promql-inputs.md | 5 + 7 files changed, 328 insertions(+), 7 deletions(-) diff --git a/crates/asap-aware-mapping/src/maintained_population.rs b/crates/asap-aware-mapping/src/maintained_population.rs index 180fc224..3dbe2489 100644 --- a/crates/asap-aware-mapping/src/maintained_population.rs +++ b/crates/asap-aware-mapping/src/maintained_population.rs @@ -142,8 +142,11 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou if value_column.is_some_and(|c| schema.columns.get(c).is_none_or(|c| c.name != "value")) { return None; } - // Open time-series schemas distinguish instant PromQL populations from table rows. - if metric.is_empty() || schema.closed || schema.time_index.is_none() { + // PromQL can retain open labels or resolve them into a complete identity column. + if metric.is_empty() + || (schema.closed && !schema.has_promql_series_identity()) + || schema.time_index.is_none() + { return None; } let label = |col: usize| -> Option { diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 012e5941..7064ce22 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -190,8 +190,103 @@ fn compile_internal( auxiliary -= 1; schemas.truncate(1); } - // Per-entity identity is safe only when the source catalog closes - // the label set. An open PromQL projection can hide distinct series. + if let Payload::Value { + operation: ValueOperation::MaintainPopulation { population }, + } = &node.payload + { + use planner_types::post_asap::maintained_population::PopulationInput; + let PopulationInput::CurrentSeries(spec) = &population.input else { + return Err(invalid( + "native maintained population requires a current-series input", + )); + }; + let [input] = schemas.as_slice() else { + return Err(invalid("current-series population requires one input")); + }; + if spec.without { + return Err(invalid( + "dynamic without grouping requires label-set projection", + )); + } + let identity = named_column( + input, + &ColumnRef::Named(promql_rows::SERIES_IDENTITY_COLUMN.into()), + )?; + let coordinate = input + .time_index + .ok_or_else(|| invalid("current-series input lacks timestamp"))?; + let value = named_column(input, &ColumnRef::SampleValue)?; + let lookback = i64::try_from(spec.lookback_ms) + .map_err(|_| invalid("current-series lookback overflows"))?; + graph.add( + id, + inputs, + Operator::current_series(input.clone(), identity, coordinate, value, lookback)? + .with_output_schema(output)?, + )?; + continue; + } + if let Payload::Value { + operation: ValueOperation::ReadPopulation { readout }, + } = &node.payload + { + use planner_types::post_asap::maintained_population::{ + PopulationInput, PopulationReadout, + }; + let PopulationReadout::TopK { k } = readout else { + return Err(invalid( + "native population readout does not support this operation", + )); + }; + let [producer] = inputs.as_slice() else { + return Err(invalid("population readout requires one input")); + }; + let Payload::Value { + operation: ValueOperation::MaintainPopulation { population }, + } = &nodes[producer].payload + else { + return Err(invalid( + "population readout requires its declared population", + )); + }; + let PopulationInput::CurrentSeries(spec) = &population.input else { + return Err(invalid("current-series population required")); + }; + if spec.without { + return Err(invalid( + "dynamic without ranking requires label-set projection", + )); + } + let input = schemas[0].clone(); + let groups = spec + .grouping + .iter() + .map(|name| named_column(&input, &ColumnRef::Named(name.clone()))) + .collect::, _>>()?; + let value = named_column(&input, &ColumnRef::SampleValue)?; + graph.add( + auxiliary, + inputs, + Operator::sort( + input.clone(), + vec![SortKey { + column: value, + descending: true, + nulls_first: false, + }], + groups.clone(), + )?, + )?; + graph.add( + id, + vec![auxiliary], + Operator::limit(input, *k as u64, 0, groups)?.with_output_schema(output)?, + )?; + auxiliary -= 1; + continue; + } + // A closed row must include either all source labels or the explicit + // complete-label identity. Projected labels alone are insufficient. if let Payload::SummaryAgg { family, input: update, diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index f8ab7d40..7431c966 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -5,7 +5,7 @@ use planner_types::pre_asap::{Column, DataType, Source as LogicalSource}; use std::rc::Rc; /// Not a legal PromQL label name, so it cannot shadow a user label. -pub const SERIES_IDENTITY_COLUMN: &str = "$promql_series_identity"; +pub use planner_types::pre_asap::schema::PROMQL_SERIES_IDENTITY as SERIES_IDENTITY_COLUMN; /// Canonical, reversible identity. JSON object encoding preserves label names, /// empty values and escaping; sorting makes ingestion order irrelevant. @@ -142,3 +142,90 @@ pub fn series_row( } Ok(row) } + +/// Compile the selected TopK computation above an existing maintained-population +/// source. The boundary supplies the complete eligible vector, not a truncated +/// TopK result; ranking remains a native physical operator. +pub fn compile_current_series_readout( + selected: &Rc, +) -> Result { + use planner_types::post_asap::{ + compile_executable_dag, maintained_population::PopulationReadout, SummaryField, + }; + let mut dag = compile_executable_dag(selected).map_err(|error| invalid(error.to_string()))?; + if dag.nodes.len() != 3 + || !dag.nodes.iter().any(|node| { + node.id == dag.root + && matches!( + node.payload, + Payload::Value { + operation: ValueOperation::ReadPopulation { + readout: PopulationReadout::TopK { .. } + } + } + ) + }) + { + return Err(invalid( + "expected one selected current-series TopK computation", + )); + } + let mut frontier = None; + for node in &mut dag.nodes { + match &mut node.payload { + Payload::Fallback { expression } => { + *expression = with_series_identity(expression)?; + } + Payload::Value { + operation: ValueOperation::MaintainPopulation { .. }, + } => { + frontier = Some(u64::from(node.id.0)); + } + Payload::Value { + operation: + ValueOperation::ReadPopulation { + readout: PopulationReadout::TopK { .. }, + }, + } => {} + _ => return Err(invalid("unsupported current-series readout dependency")), + } + if node + .output_schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "current-series input already has a physical identity column", + )); + } + node.output_schema.fields.push(SummaryField { + name: SERIES_IDENTITY_COLUMN.into(), + dtype: SummaryFamilyType::Plain(DataType::Utf8), + nullable: false, + }); + } + for edge in &mut dag.edges { + edge.intermediate_schema = dag + .nodes + .iter() + .find(|node| node.id == edge.producer) + .unwrap() + .output_schema + .clone(); + } + let frontier = frontier.ok_or_else(|| invalid("missing current-series population"))?; + let schema = Arc::new( + dag.nodes + .iter() + .find(|node| u64::from(node.id.0) == frontier) + .unwrap() + .output_schema + .clone(), + ); + compile( + &dag, + BTreeMap::from([(frontier, InputContract::bounded(schema))]), + &[u64::from(dag.root.0)], + ) +} diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs index 0a7059e7..7abbeb04 100644 --- a/crates/asap-physical-operators/tests/current_series_heap.rs +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -33,9 +33,10 @@ fn schema() -> Arc { fn run(program: &CompiledPhysicalDag, data: Batch, end: i64) -> Result, String> { let recovered = CompiledPhysicalDag::decode(&program.encode().map_err(|e| e.to_string())?) .map_err(|e| e.to_string())?; + let input_id = recovered.input_contracts().next().unwrap().0; let graph = recovered .instantiate(BTreeMap::from([( - 0, + input_id, Box::new(Operator::source(data.schema().clone(), vec![data]).unwrap()) as Source<'_>, )])) .unwrap(); @@ -280,3 +281,119 @@ fn identity_encoding_is_lossless_and_rejects_noncanonical_inputs() { assert!(decode_series_identity(invalid).is_err(), "{invalid}"); } } + +// The actual Planner population candidate lowers to native operators; this +// test does not manually assemble the computation or its dependency edges. +#[test] +fn planner_current_series_candidate_compiles_with_dynamic_identity() { + use asap_physical_operators::physical_planner::{compile, promql_rows::with_series_identity}; + use planner_types::{types::AccuracyTarget, workload::*}; + use std::rc::Rc; + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("topk by(job)(1, m)".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(60_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let original = asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0); + let open_root = Rc::new(original.clone()); + let open_selected = + asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&open_root), + ) + .candidate(&open_root) + .unwrap(); + let snapshot_program = + asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + &open_selected, + ) + .unwrap(); + let encoded = String::from_utf8(snapshot_program.encode().unwrap()).unwrap(); + assert!( + !encoded.contains("CurrentSeries"), + "maintained input must not be rebuilt" + ); + assert!(encoded.contains("Sort") && encoded.contains("Limit")); + assert_eq!(snapshot_program.input_contracts().count(), 1); + let root = Rc::new(with_series_identity(&original).unwrap()); + let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&root), + ) + .candidate(&root) + .unwrap(); + let logical = compile_executable_dag(&selected).unwrap(); + let raw = logical + .nodes + .iter() + .find(|node| matches!(node.payload, ExecutableOperatorPayload::Fallback { .. })) + .unwrap(); + let raw_schema = Arc::new(raw.output_schema.clone()); + let physical = compile( + &logical, + BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(raw_schema.clone()), + )]), + &[u64::from(logical.root.0)], + ) + .unwrap(); + let bytes = String::from_utf8(physical.encode().unwrap()).unwrap(); + assert!(bytes.contains("CurrentSeries")); + assert!(bytes.contains("Sort")); + assert!(bytes.contains("Limit")); + let rows = [("a", 10_000, 100.), ("a", 50_000, 1.), ("b", 50_000, 20.)] + .into_iter() + .map(|(member, at, value)| { + series_row( + &raw_schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("unreferenced".into(), member.into()), + ]), + at, + value, + ) + .unwrap() + }) + .collect(); + let batches = run(&physical, Batch::try_new(raw_schema, rows).unwrap(), 60_000).unwrap(); + let rows = batches + .iter() + .flat_map(|batch| batch.rows()) + .collect::>(); + assert_eq!(rows.len(), 1); + assert!(matches!(rows[0][1], Value::Float64(20.))); + let id = batches[0] + .schema() + .fields + .iter() + .position(|field| field.name == SERIES_IDENTITY_COLUMN) + .unwrap(); + let Value::Utf8(encoded) = &rows[0][id] else { + panic!() + }; + assert_eq!( + decode_series_identity(encoded).unwrap()["unreferenced"], + "b" + ); +} diff --git a/crates/types/src/post_asap/maintained_population.rs b/crates/types/src/post_asap/maintained_population.rs index 30dacf82..3939c7c1 100644 --- a/crates/types/src/post_asap/maintained_population.rs +++ b/crates/types/src/post_asap/maintained_population.rs @@ -63,7 +63,7 @@ impl CurrentSeriesInput { }; if self.metric.is_empty() || *metric != self.metric - || schema.closed + || (schema.closed && !schema.has_promql_series_identity()) || schema.time_index.is_none() { return false; diff --git a/crates/types/src/pre_asap/schema.rs b/crates/types/src/pre_asap/schema.rs index 81b4ae1f..fbcb9d67 100644 --- a/crates/types/src/pre_asap/schema.rs +++ b/crates/types/src/pre_asap/schema.rs @@ -154,7 +154,21 @@ pub struct Schema { pub closed: bool, } +/// Reserved physical row column carrying canonical JSON of a complete PromQL +/// label map. `$` cannot occur in a user PromQL label name. +pub const PROMQL_SERIES_IDENTITY: &str = "$promql_series_identity"; + impl Schema { + pub fn has_promql_series_identity(&self) -> bool { + self.closed + && self.columns.iter().any(|column| { + column.name == PROMQL_SERIES_IDENTITY + && column.dtype == DataType::Utf8 + && !column.nullable + && column.table.is_none() + }) + } + /// Construct a `Schema` from columns alone — no time index, no /// unique-key constraint. Used by `Scan` over a tabular source /// when the catalog supplies no primary-key metadata. diff --git a/docs/develop_docs/native-promql-inputs.md b/docs/develop_docs/native-promql-inputs.md index 66b83ae6..55117e79 100644 --- a/docs/develop_docs/native-promql-inputs.md +++ b/docs/develop_docs/native-promql-inputs.md @@ -22,6 +22,11 @@ rejects noncanonical encodings. A query adapter must still apply the selected operator's metric-name/result-label rules. Source selection, complete window coverage and revision admission remain deployment responsibilities. +Planner's maintained-population candidate recognizes this explicit identity +representation. Its TopK readout compiles automatically to `CurrentSeries`, +`Sort`, and `Limit`; deployment supplies the raw boundary or an already maintained +population boundary. Compilation does not open either source. + The native `CurrentSeries` operator selects the latest sample per complete identity in `(evaluation_time - lookback, evaluation_time]`. It removes stale markers after selecting the latest sample, so an older value cannot reappear. From 89bf83c69cc2a27a4fd72b0d56be49d8ca1e1988 Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 22:52:31 +0000 Subject: [PATCH 64/90] feat: expose signed spatial TopK heap physical candidates --- .../src/maintained_population.rs | 1 + crates/asap-aware-mapping/src/replacement.rs | 191 ++++++++++++++++-- .../src/physical_planner/promql_rows.rs | 23 +++ .../tests/weighted_topk_binding.rs | 131 ++++++++++++ docs/develop_docs/native-promql-inputs.md | 8 + 5 files changed, 339 insertions(+), 15 deletions(-) diff --git a/crates/asap-aware-mapping/src/maintained_population.rs b/crates/asap-aware-mapping/src/maintained_population.rs index 3dbe2489..504c8100 100644 --- a/crates/asap-aware-mapping/src/maintained_population.rs +++ b/crates/asap-aware-mapping/src/maintained_population.rs @@ -50,6 +50,7 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou AggIntent::Quantile { q, col, .. } if q.is_finite() => { (*col, PopulationReadout::Quantile { q: *q }) } + AggIntent::TopK { k, .. } => (None, PopulationReadout::TopK { k: *k }), AggIntent::Sum { col } => (*col, PopulationReadout::Sum), AggIntent::Count { .. } => (None, PopulationReadout::Count), AggIntent::Avg { col } => (*col, PopulationReadout::Average), diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 1c9efdaa..fa2d8b48 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -1282,6 +1282,65 @@ impl<'a> SketchAlgorithmStrategy<'a> { } } + /// Preserve the canonical Sort/Limit representation while exploring heap + /// realizations of an instant-vector ranking under the caller's target. + /// The input must carry the complete dynamic series identity. This never + /// treats a range of historical samples as the instant vector. + pub fn current_series_topk_candidates( + &self, + root: &Rc, + accuracy: &AccuracyTarget, + ) -> Proposals { + let QueryExpr::Limit { + n, + offset: 0, + child, + } = root.as_ref() + else { + return Proposals::default(); + }; + let QueryExpr::Sort { + keys, + partition_by, + child, + } = child.as_ref() + else { + return Proposals::default(); + }; + let [key] = keys.as_slice() else { + return Proposals::default(); + }; + let QueryExpr::Column(value) = key.expr else { + return Proposals::default(); + }; + let Ok(schema) = child.output_schema() else { + return Proposals::default(); + }; + if key.ascending + || key.nulls_first + || partition_by.is_without() + || !schema.has_promql_series_identity() + || !schema + .columns + .get(value) + .is_some_and(|column| column.name == "value") + || !is_current_series_source(child) + { + return Proposals::default(); + } + let ranked = Rc::new(QueryExpr::Aggregate { + reduction: Reduction::Reduce(partition_by.clone()), + measures: vec![AggIntent::TopK { + k: *n, + accuracy: accuracy.clone(), + }], + output_names: vec![], + having: None, + child: Rc::clone(child), + }); + self.propose_with(&ranked, None) + } + pub(crate) fn from_planning_inputs(planning_inputs: CandidatePlanningInputs<'a>) -> Self { Self { planning_inputs } } @@ -2290,7 +2349,7 @@ pub(crate) fn construct_summary_with( child_target, allocation, )?; - if is_counter_weighted_topk(intent, child) { + if is_snapshot_weighted_topk(intent, child) { return finish_weighted_topk(candidate, expr, intent); } return Ok(candidate); @@ -2397,14 +2456,25 @@ fn finish_weighted_topk( Ok(result) } -fn is_counter_weighted_topk(intent: &AggIntent, child: &QueryExpr) -> bool { +fn is_current_series_source(child: &QueryExpr) -> bool { + let source = match child { + QueryExpr::TimeRange { child, .. } => child.as_ref(), + source => source, + }; + matches!(source, QueryExpr::Scan { + source: asap_types::pre_asap::Source::TimeSeries { .. }, schema, .. + } if schema.has_promql_series_identity()) +} + +fn is_snapshot_weighted_topk(intent: &AggIntent, child: &QueryExpr) -> bool { matches!(intent, AggIntent::TopK { .. }) - && matches!(child, + && (is_current_series_source(child) + || matches!(child, QueryExpr::Aggregate { measures, child, .. } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase]) || (matches!(measures.as_slice(), [AggIntent::Sum { .. }]) && matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } - if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])))) + if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase]))))) } /// Translate an [`Realization`] into the `(family, needs a @@ -2461,6 +2531,7 @@ type PhysicalSummaryInputRule = fn( const PHYSICAL_SUMMARY_INPUT_RULES: &[PhysicalSummaryInputRule] = &[ realize_value_frequency_summary_input, realize_counter_value_summary_input, + realize_current_series_summary_input, realize_keyed_additive_summary_input, ]; @@ -2616,10 +2687,10 @@ fn construct_summary_agg( SummaryFamilyType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap) ); - let rate_weighted = matches!(node, QueryExpr::Aggregate { child, .. } - if is_counter_weighted_topk(intent, child)); + let snapshot_weighted = matches!(node, QueryExpr::Aggregate { child, .. } + if is_snapshot_weighted_topk(intent, child)); let mut family = family; - let score_population = if rate_weighted { + let score_population = if snapshot_weighted { let bound = planning_inputs.evidence.topk_max_distinct_items(node); if bound.is_some_and(|n| n == 0 || n > (1u64 << 53)) { return Err(RealizationError::PhysicalRealization( @@ -2643,7 +2714,7 @@ fn construct_summary_agg( } else { None }; - let physical_reduction = if rate_weighted { + let physical_reduction = if snapshot_weighted { let QueryExpr::Aggregate { child, .. } = node else { unreachable!() }; @@ -2694,7 +2765,7 @@ fn construct_summary_agg( let state_idx = summary_col_index(&out_schema, &by, per_series); let readout_schema = if keyed_heap - && matches!(node, QueryExpr::Aggregate { child, .. } if is_counter_weighted_topk(intent, child)) + && matches!(node, QueryExpr::Aggregate { child, .. } if is_snapshot_weighted_topk(intent, child)) { keyed_heap_readout_schema(&input, node)? } else { @@ -2703,7 +2774,7 @@ fn construct_summary_agg( let summary_input = input.input; let query = estimate.then(|| { - if rate_weighted { + if snapshot_weighted { if let SummaryFamilyType::Sketch(kind, _) = &family { let capacity = match kind.params() { SketchParams::CmsWithHeap { heap_size, .. } @@ -2722,7 +2793,7 @@ fn construct_summary_agg( if keyed_heap { let mut state = state_schema.fields[state_idx].clone(); state.dtype = family.clone(); - let mut fields = if rate_weighted { + let mut fields = if snapshot_weighted { readout_schema.fields[..reduction.group_keys().map_or(0, |keys| keys.len())].to_vec() } else { Vec::new() @@ -2744,13 +2815,30 @@ fn construct_summary_agg( let bound_child = realize_child_with( &input.child, planning_inputs, - if rate_weighted { + if snapshot_weighted { Some(&AccuracyTarget::Exact) } else { child_target }, )?; - let bound_child = if rate_weighted { + let bound_child = if snapshot_weighted && is_current_series_source(&input.child) { + // Explicit snapshot selection prevents historical observations from + // becoming repeated weights in an instant-vector heap. + let root = Rc::new(node.clone()); + let population = crate::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&root), + ) + .candidate(&root) + .ok_or(RealizationError::PhysicalRealization( + "snapshot ranking requires a supported current-series population", + ))?; + let SummaryExpr::ValueOperation { child, .. } = &population.expr else { + return Err(RealizationError::PhysicalRealization( + "missing population readout", + )); + }; + Rc::clone(child) + } else if snapshot_weighted { // A fresh query-time summary consumes this evaluation's finalized rates. // Moving rate snapshots must never accumulate across evaluations. finalize_exact_accumulator(bound_child, &input.child)? @@ -2777,7 +2865,7 @@ fn construct_summary_agg( planning_inputs.evidence.estimator_contract(node), local_target, ); - let membership_query = if rate_weighted { + let membership_query = if snapshot_weighted { Some(readout(intent, &summary_input, planning_inputs.cost)) } else { query.clone() @@ -2792,7 +2880,7 @@ fn construct_summary_agg( allocation, )?; - if rate_weighted { + if snapshot_weighted { use asap_types::post_asap::{BoundExpr, ProbabilityExpr}; let target = accuracy_target(intent).expect("TopK target"); guarantee = if let Some(mut score) = @@ -3016,6 +3104,19 @@ fn ranking_score_index( logical: &QueryExpr, values: &SummarySchema, ) -> Result { + if is_current_series_source(logical) { + return values + .fields + .iter() + .position(|field| { + field.name == "value" + && field.dtype + == SummaryFamilyType::Plain(asap_types::pre_asap::DataType::Float64) + }) + .ok_or(RealizationError::PhysicalRealization( + "snapshot ranking requires the sample value column", + )); + } let QueryExpr::Aggregate { reduction, measures, @@ -3121,6 +3222,66 @@ fn realize_counter_value_summary_input( }) } +/// An instant-vector source has one current value per full series identity. +/// Rebuild the state for each evaluation; historical samples are not updates. +fn realize_current_series_summary_input( + intent: &AggIntent, + family: &SummaryFamilyType, + output_reduction: &Reduction, + child: &Rc, +) -> PhysicalSummaryInputRuleResult { + if !matches!(intent, AggIntent::TopK { .. }) || !is_current_series_source(child) { + return PhysicalSummaryInputRuleResult::NotApplicable; + } + let SummaryFamilyType::Sketch(kind, _) = family else { + return PhysicalSummaryInputRuleResult::NotApplicable; + }; + match kind.algorithm() { + SketchAlgorithm::CountSketchWithHeap => {} + SketchAlgorithm::CmsWithHeap => { + return PhysicalSummaryInputRuleResult::Unsupported( + "current sample values do not prove non-negative CMS weights", + ) + } + _ => return PhysicalSummaryInputRuleResult::NotApplicable, + } + let Reduction::Reduce(groups) = output_reduction else { + return PhysicalSummaryInputRuleResult::Unsupported( + "snapshot ranking requires explicit partitions", + ); + }; + if groups.is_without() { + return PhysicalSummaryInputRuleResult::Unsupported( + "snapshot ranking requires resolved partitions", + ); + } + let Ok(schema) = child.output_schema() else { + return PhysicalSummaryInputRuleResult::Unsupported( + "snapshot ranking requires a valid source schema", + ); + }; + let items = schema + .columns + .iter() + .enumerate() + .filter(|(index, column)| column.name != "value" && !groups.contains(index)) + .map(|(index, _)| schema_column_ref(child, index).map(SummaryInputExpr::Column)) + .collect::>>(); + let Some(items) = items.filter(|items| !items.is_empty()) else { + return PhysicalSummaryInputRuleResult::Unsupported( + "snapshot ranking has no item identity", + ); + }; + PhysicalSummaryInputRuleResult::Realized(PhysicalSummaryInput { + child: Rc::clone(child), + input: SummaryUpdate { + item: Some(SummaryInputExpr::Tuple(items)), + weight: SummaryInputExpr::Column(ColumnRef::SampleValue), + weight_domain: WeightDomain::UnknownOrSigned, + }, + }) +} + /// Realize the composite heavy-hitter realization for /// `TopK(Count GROUP BY key)`. The heap sketch consumes the raw keyed stream; /// it does not consume an independently materialized Count result. diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index 7431c966..d55982c7 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -153,6 +153,29 @@ pub fn compile_current_series_readout( compile_executable_dag, maintained_population::PopulationReadout, SummaryField, }; let mut dag = compile_executable_dag(selected).map_err(|error| invalid(error.to_string()))?; + // Typed snapshot candidates already carry full identity throughout the DAG. + // Cut at the population output, preserving all selected heap/readout nodes. + let populations = dag.nodes.iter().filter(|node| matches!(&node.payload, + Payload::Value { operation: ValueOperation::MaintainPopulation { population } } + if matches!(population.input, planner_types::post_asap::maintained_population::PopulationInput::CurrentSeries(_)) + )).collect::>(); + if let [population] = populations.as_slice() { + if population + .output_schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + { + return compile( + &dag, + BTreeMap::from([( + u64::from(population.id.0), + InputContract::bounded(Arc::new(population.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + ); + } + } if dag.nodes.len() != 3 || !dag.nodes.iter().any(|node| { node.id == dag.root diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index 269139c6..84f54364 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -600,3 +600,134 @@ fn check_direct_rate_topk(dynamic: bool) { } } } + +// Spatial ranking consumes one eligible instant vector. Signed values require +// CountSketch; a raw metric does not establish the non-negative CMS contract. +#[test] +fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { + use asap_physical_operators::physical_planner::promql_rows::{ + decode_series_identity, series_row, with_series_identity, SERIES_IDENTITY_COLUMN, + }; + let logical = lower_promql("topk by(job)(1, m)", AccuracyTarget::Epsilon(0.1)).unwrap(); + let root = Rc::new(with_series_identity(&logical).unwrap()); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let candidates = strategy + .current_series_topk_candidates(&root, &AccuracyTarget::Epsilon(0.1)) + .candidates; + assert!(!candidates + .iter() + .any(|c| c.rationale.contains("CmsWithHeap"))); + let selected = candidates + .iter() + .find_map(|candidate| match &candidate.replacement { + Replacement::Summary(node) if candidate.rationale.contains("CountSketchWithHeap") => { + Some(node) + } + _ => None, + }) + .expect("signed spatial TopK must expose CountSketch with heap"); + let dag = compile_executable_dag(selected).unwrap(); + let raw = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + ExecutableOperatorPayload::Fallback { + expression: QueryExpr::TimeRange { .. } + } + ) + }) + .unwrap(); + let schema = Arc::new(raw.output_schema.clone()); + let program = compile( + &dag, + BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + let snapshot_program = + asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + selected, + ) + .unwrap(); + let encoded: serde_json::Value = + serde_json::from_slice(&snapshot_program.encode().unwrap()).unwrap(); + assert!(!encoded.to_string().contains("CurrentSeries")); + assert!(encoded.to_string().contains("KeyedSummaryBuild")); + assert!(encoded.to_string().contains("KeyedReadout")); + for (values, expected, score) in [ + ([100., 20.], "a", 100.), + ([1., 20.], "b", 20.), + ([-10., -2.], "b", -2.), + ] { + let rows = ["a", "b"] + .into_iter() + .zip(values) + .map(|(instance, value)| { + series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("unreferenced".into(), instance.into()), + ]), + 60_000, + value, + ) + .unwrap() + }) + .collect(); + let batch = Batch::try_new(schema.clone(), rows).unwrap(); + let graph = program + .instantiate(BTreeMap::from([( + u64::from(raw.id.0), + Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'_>, + )])) + .unwrap(); + block_on(async { + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 60_000, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut result = Vec::new(); + while let Some(batch) = stream.next().await { + let batch = batch.unwrap(); + let identity = batch + .schema() + .fields + .iter() + .position(|f| f.name == SERIES_IDENTITY_COLUMN) + .unwrap(); + let value = batch + .schema() + .fields + .iter() + .position(|f| f.name == "value") + .unwrap(); + for row in batch.rows() { + let Value::Utf8(labels) = &row[identity] else { + panic!() + }; + let Value::Float64(v) = row[value] else { + panic!() + }; + result.push(( + decode_series_identity(labels).unwrap()["unreferenced"].clone(), + v, + )); + } + } + assert_eq!(result, vec![(expected.into(), score)]); + }); + } +} diff --git a/docs/develop_docs/native-promql-inputs.md b/docs/develop_docs/native-promql-inputs.md index 55117e79..aa1a52b6 100644 --- a/docs/develop_docs/native-promql-inputs.md +++ b/docs/develop_docs/native-promql-inputs.md @@ -45,3 +45,11 @@ Tests cover open-label Rate → CMS/CountSketch heaps, hidden-label round trips, reset and zero-rate cases, snapshot replacement/decrease/expiry/staleness, serialized physical recovery, and resource rejection. These are shared-library tests, not proof of Backend candidate selection or durable deployment execution. + +Spatial heap candidates use the same complete series identity. Planner's +`current_series_topk_candidates` explores a CountSketch-with-heap realization +of canonical Sort/Limit under an explicit accuracy target. The physical graph +selects the latest eligible samples before building a fresh heap. A maintained +population boundary can supply that snapshot directly. Arbitrary signed metric +values do not authorize CMS; counter Rate's non-negative proof is separate. +These candidates still require membership/score evidence for deployment admission. From c9df5a66369a9699f213e986daa89894122f486b Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 23:24:57 +0000 Subject: [PATCH 65/90] feat: compile native ranking above exact stored Rate readouts --- .../src/physical_planner/promql_rows.rs | 64 +++++++++++++++++++ .../tests/weighted_topk_binding.rs | 22 +++++++ 2 files changed, 86 insertions(+) diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index d55982c7..854fd0ce 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -252,3 +252,67 @@ pub fn compile_current_series_readout( &[u64::from(dag.root.0)], ) } + +/// Compile a selected per-series Rate -> ranking computation above its exact +/// counter readout. Deployments bind complete window readouts at this boundary; +/// the heap is rebuilt independently for each evaluation. This does not move +/// that frontier to ingestion time or authorize combining finalized rates. +pub fn compile_rate_ranking( + selected: &Rc, +) -> Result< + ( + Rc, + CompiledPhysicalDag, + ), + Error, +> { + use planner_types::post_asap::{ + compile_executable_dag_with_node_ids, ExactKind, SummaryExpr, SummaryNode, + }; + fn frontier(node: &Rc) -> Option> { + match &node.expr { + SummaryExpr::ValueOperation { + child, + operation: ValueOperation::FinalizeExactAccumulator, + .. + } if matches!(&child.expr, SummaryExpr::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + reduction: planner_types::pre_asap::Reduction::PerEntity, + child: raw, .. + } if matches!(&raw.expr, SummaryExpr::KeepPreAsap(expr) if matches!(expr.as_ref(), QueryExpr::TimeRange { .. }))) => + { + Some(Rc::clone(node)) + } + SummaryExpr::ValueOperation { child, .. } | SummaryExpr::SummaryAgg { child, .. } => { + frontier(child) + } + SummaryExpr::SummaryEstimate { summary_input, .. } => frontier(summary_input), + _ => None, + } + } + let source = frontier(selected) + .ok_or_else(|| invalid("ranking requires one exact per-series Rate frontier"))?; + if !source + .schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid("Rate ranking requires complete series identity")); + } + let compiled = compile_executable_dag_with_node_ids(selected) + .map_err(|error| invalid(error.to_string()))?; + let id = u64::from( + compiled + .node_ids + .node_id(&source) + .ok_or_else(|| invalid("missing Rate frontier"))? + .0, + ); + let program = compile( + &compiled.dag, + BTreeMap::from([(id, InputContract::bounded(Arc::new(source.schema.clone())))]), + &[u64::from(compiled.dag.root.0)], + )?; + Ok((source, program)) +} diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index 84f54364..edcc711f 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -330,6 +330,28 @@ fn check_direct_rate_topk(dynamic: bool) { _ => None, }) .unwrap_or_else(|| panic!("missing {algorithm:?} over direct Rate")); + if dynamic { + let (source, ranked) = + asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking( + candidate, + ) + .unwrap(); + assert!(matches!( + source.expr, + SummaryExpr::ValueOperation { + operation: ValueOperation::FinalizeExactAccumulator, + .. + } + )); + assert_eq!(ranked.input_contracts().count(), 1); + let encoded = String::from_utf8(ranked.encode().unwrap()).unwrap(); + assert!(encoded.contains("KeyedSummaryBuild")); + assert!(encoded.contains("KeyedReadout")); + assert!( + !encoded.contains("\"Rate\""), + "Rate must be supplied by its exact stored-state readout" + ); + } let dag = compile_executable_dag(candidate).unwrap(); assert!(dag.nodes.iter().any(|node| matches!(&node.payload, ExecutableOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); From 15de21e36bed4bc49783237b3fb6a2af4829456b Mon Sep 17 00:00:00 2001 From: zz_y Date: Sun, 27 Sep 2026 23:55:22 +0000 Subject: [PATCH 66/90] Expose fixed-window Rate heap physical candidates and verify stored-state execution --- crates/asap-aware-mapping/src/replacement.rs | 49 ++++- .../src/physical_planner/promql_rows.rs | 53 ++++- .../tests/weighted_topk_binding.rs | 205 ++++++++++++++++++ 3 files changed, 305 insertions(+), 2 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index fa2d8b48..eb963687 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -1341,6 +1341,53 @@ impl<'a> SketchAlgorithmStrategy<'a> { self.propose_with(&ranked, None) } + /// Fixed-window maintenance can finalize each series' counter state and + /// build a fresh heap for that evaluation window. Deployment must provide + /// a complete, synchronized population and bind the matching window; this + /// candidate never incrementally adds one window's rates to another. + pub fn fixed_window_rate_topk_candidates(&self, root: &Rc) -> Proposals { + fn place(node: &Rc) -> Option> { + let mut next = node.as_ref().clone(); + match &mut next.expr { + SummaryExpr::ValueOperation { + child, + operation: ValueOperation::FinalizeExactAccumulator, + timing, + } if matches!(&child.expr, SummaryExpr::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + reduction: Reduction::PerEntity, child: source, .. + } if matches!(&source.expr, SummaryExpr::KeepPreAsap(source) if matches!(source.as_ref(), QueryExpr::TimeRange { .. }))) => + { + *timing = ExecutionTiming::IngestionTime; + } + SummaryExpr::ValueOperation { child, .. } + | SummaryExpr::SummaryAgg { child, .. } => *child = place(child)?, + SummaryExpr::SummaryEstimate { summary_input, .. } => { + *summary_input = place(summary_input)? + } + _ => return None, + } + Some(Rc::new(next)) + } + let mut proposals = self.propose_with(root, None); + proposals.candidates.retain_mut(|candidate| { + let Replacement::Summary(node) = &candidate.replacement else { return false }; + let Ok(dag) = asap_types::post_asap::compile_executable_dag(node) else { return false }; + if !dag.nodes.iter().any(|node| matches!(&node.payload, + asap_types::post_asap::ExecutableOperatorPayload::SummaryAgg { + family: SummaryFamilyType::Sketch(kind, _), .. + } if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap))) { + return false; + } + let Some(placed) = place(node) else { return false }; + if asap_types::post_asap::compile_executable_dag(&placed).is_err() { return false; } + candidate.replacement = Replacement::Summary(placed); + candidate.rationale.push_str("; fixed-window precompute over complete per-series counter states"); + true + }); + proposals + } + pub(crate) fn from_planning_inputs(planning_inputs: CandidatePlanningInputs<'a>) -> Self { Self { planning_inputs } } @@ -1605,7 +1652,7 @@ fn describe_realization(intent: &AggIntent, realization: &Realization) -> String match realization { Realization::Sketch(kind) => format!( "{} realizes as a {:?} sketch — one of summary_candidates' \ - alternatives for this intent (asap_aware_mapping::replacement::realizations_for_intent)", + candidates for this intent (asap_aware_mapping::replacement::realizations_for_intent)", describe_intent(intent), kind.algorithm() ), diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index 854fd0ce..c69a6b59 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -274,7 +274,7 @@ pub fn compile_rate_ranking( SummaryExpr::ValueOperation { child, operation: ValueOperation::FinalizeExactAccumulator, - .. + timing: planner_types::post_asap::ExecutionTiming::QueryTime, } if matches!(&child.expr, SummaryExpr::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), reduction: planner_types::pre_asap::Reduction::PerEntity, @@ -316,3 +316,54 @@ pub fn compile_rate_ranking( )?; Ok((source, program)) } + +/// The selected logical placement requires a fresh heap for each closed window. +/// Compile both physical graphs before deployment chooses storage or scheduling. +/// The input is the complete collection of per-series exact counter states. +pub fn compile_fixed_window_rate_ranking( + selected: &Rc, +) -> Result { + use planner_types::post_asap::{ + compile_executable_dag, ExactKind, ExecutionTiming, SketchAlgorithm, + }; + let dag = compile_executable_dag(selected).map_err(|e| invalid(e.to_string()))?; + let sources = dag + .nodes + .iter() + .filter(|n| { + matches!( + &n.payload, + Payload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + reduction: planner_types::pre_asap::Reduction::PerEntity, + .. + } + ) + }) + .collect::>(); + let heaps = dag.nodes.iter().filter(|n| n.output_state.timing == ExecutionTiming::IngestionTime && matches!(&n.payload, Payload::SummaryAgg { + family: SummaryFamilyType::Sketch(kind, _), .. + } if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap))).collect::>(); + let ([source], [heap]) = (sources.as_slice(), heaps.as_slice()) else { + return Err(invalid("expected one selected fixed-window Rate heap")); + }; + if !source + .output_schema + .fields + .iter() + .any(|f| f.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "fixed-window Rate heap requires complete series identity", + )); + } + compile_candidate( + &dag, + BTreeMap::from([( + u64::from(source.id.0), + InputContract::bounded(Arc::new(source.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &[u64::from(heap.id.0)], + ) +} diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index edcc711f..a81a1841 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -753,3 +753,208 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { }); } } + +// Placement changes execution ownership only. Every fixed-window candidate +// contains Rate finalization before a fresh heap, with query readout downstream. +#[test] +fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { + use asap_physical_operators::physical_planner::{ + compile_candidate, promql_rows::with_series_identity, + }; + let root = Rc::new( + with_series_identity( + &lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(), + ) + .unwrap(), + ); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let candidates = strategy.fixed_window_rate_topk_candidates(&root).candidates; + assert_eq!(candidates.len(), 2); + for candidate in candidates { + let Replacement::Summary(root) = candidate.replacement else { + panic!() + }; + let dag = compile_executable_dag(&root).unwrap(); + let state = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + ExecutableOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) + .unwrap(); + let heap = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + ExecutableOperatorPayload::SummaryAgg { + family: SummaryFamilyType::Sketch(..), + .. + } + ) + }) + .unwrap(); + assert_eq!(heap.output_state.timing, ExecutionTiming::IngestionTime); + let physical = compile_candidate( + &dag, + BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &[u64::from(heap.id.0)], + ) + .unwrap(); + let exported = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_ranking(&root).unwrap(); + assert_eq!(exported.encode().unwrap(), physical.encode().unwrap()); + assert!( + asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking(&root) + .is_err(), + "query binding must not move the selected precompute frontier" + ); + // Execute the selected split across a state serialization boundary. + // Each run builds fresh weights from that window's counters. + let execute = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDag, + input: Batch, + scope: Scope| { + let id = plan.input_contracts().next().unwrap().0; + let source = Box::new(Operator::source(input.schema().clone(), vec![input]).unwrap()) + as Source<'static>; + let graph = plan.instantiate(BTreeMap::from([(id, source)])).unwrap(); + block_on(async { + let mut stream = graph + .execute( + plan.roots(), + RunContext::new(scope, Limits::default()).unwrap(), + ) + .unwrap() + .remove(0); + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push((*batch.unwrap()).clone()); + } + assert_eq!(batches.len(), 1); + batches.remove(0) + }) + }; + let (family, input, grouping) = match &state.payload { + ExecutableOperatorPayload::SummaryAgg { + family, + input, + grouping, + .. + } => (family, input, grouping), + _ => unreachable!(), + }; + for (end, samples, leader) in [ + ( + 60_000, + [[0., 100., 200.], [0., 10., 20.], [0., 1., 2.]], + "a", + ), + ( + 120_000, + [[200., 200., 200.], [100., 0., 300.], [2., 3., 4.]], + "b", + ), + ] { + let schema = Arc::new(state.output_schema.clone()); + let rows = samples + .into_iter() + .zip(["a", "b", "c"]) + .map(|(samples, label)| { + let mut accumulator = + asap_physical_operators::factory::create_planner_accumulator( + family, input, grouping, + ) + .unwrap(); + for (offset, value) in [10_000, 30_000, 50_000].into_iter().zip(samples) { + accumulator.update_single(value, end - 60_000 + offset); + } + let summary = Value::Summary { + family: family.clone(), + state: Arc::from(accumulator.into_accumulator()), + }; + schema + .fields + .iter() + .map(|field| match &field.dtype { + SummaryFamilyType::ExactAggregate(..) => summary.clone(), + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(end), + SummaryFamilyType::Plain(DataType::Utf8) + if field.name == "$promql_series_identity" => + { + Value::Utf8( + serde_json::to_string(&BTreeMap::from([ + ("job", "api"), + ("instance", label), + ])) + .unwrap() + .into(), + ) + } + SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8("api".into()), + _ => panic!("unexpected state field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(schema, rows).unwrap(); + let precompute = physical.precompute.as_ref().unwrap(); + let heap = execute( + precompute, + batch, + Scope::Ingestion { + window_start_ms: end - 60_000, + window_end_ms: end, + revision: 1, + }, + ); + let bytes = asap_physical_operators::stored_state::native::encode_batch(&heap).unwrap(); + let restored = asap_physical_operators::stored_state::native::decode_batch( + &bytes, + heap.schema().clone(), + 1 << 24, + ) + .unwrap(); + let result = execute( + &physical.query, + restored, + Scope::Query { + evaluation_time_ms: end, + revision: 1, + }, + ); + let identity = result + .schema() + .fields + .iter() + .position(|f| f.name == "$promql_series_identity") + .unwrap(); + let Value::Utf8(encoded) = &result.rows()[0][identity] else { + panic!() + }; + let labels: BTreeMap = serde_json::from_str(encoded).unwrap(); + assert_eq!(labels["instance"], leader); + assert_eq!(result.rows().len(), 2); + } + let precompute = String::from_utf8(physical.precompute.unwrap().encode().unwrap()).unwrap(); + assert!(precompute.contains("KeyedSummaryBuild")); + assert!(precompute.contains("Rate")); + assert!(!String::from_utf8(physical.query.encode().unwrap()) + .unwrap() + .contains("KeyedSummaryBuild")); + } +} From 87bb344a817b14c9735b837d2bd94a755c67acd6 Mon Sep 17 00:00:00 2001 From: zz_y Date: Mon, 28 Sep 2026 00:26:10 +0000 Subject: [PATCH 67/90] Expose grouped Rate Sum at both maintenance and query placements --- crates/asap-aware-mapping/src/replacement.rs | 81 ++++++++++++++++--- .../src/physical_planner/promql_rows.rs | 38 ++++++--- .../tests/weighted_topk_binding.rs | 60 +++++++++++++- 3 files changed, 158 insertions(+), 21 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index eb963687..6e371194 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -1342,10 +1342,10 @@ impl<'a> SketchAlgorithmStrategy<'a> { } /// Fixed-window maintenance can finalize each series' counter state and - /// build a fresh heap for that evaluation window. Deployment must provide + /// build a fresh heap or grouped Sum for that evaluation window. Deployment must provide /// a complete, synchronized population and bind the matching window; this /// candidate never incrementally adds one window's rates to another. - pub fn fixed_window_rate_topk_candidates(&self, root: &Rc) -> Proposals { + pub fn fixed_window_rate_candidates(&self, root: &Rc) -> Proposals { fn place(node: &Rc) -> Option> { let mut next = node.as_ref().clone(); match &mut next.expr { @@ -1371,18 +1371,79 @@ impl<'a> SketchAlgorithmStrategy<'a> { } let mut proposals = self.propose_with(root, None); proposals.candidates.retain_mut(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { return false }; - let Ok(dag) = asap_types::post_asap::compile_executable_dag(node) else { return false }; - if !dag.nodes.iter().any(|node| matches!(&node.payload, + let Replacement::Summary(node) = &candidate.replacement else { + return false; + }; + let Ok(dag) = asap_types::post_asap::compile_executable_dag(node) else { + return false; + }; + if !dag.nodes.iter().any(|node| match &node.payload { asap_types::post_asap::ExecutableOperatorPayload::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), .. - } if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap))) { + family: SummaryFamilyType::Sketch(kind, _), + .. + } => matches!( + kind.algorithm(), + SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap + ), + asap_types::post_asap::ExecutableOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + .. + } => true, + _ => false, + }) { + return false; + } + let Some(placed) = place(node) else { + return false; + }; + if asap_types::post_asap::compile_executable_dag(&placed).is_err() { return false; } - let Some(placed) = place(node) else { return false }; - if asap_types::post_asap::compile_executable_dag(&placed).is_err() { return false; } + let Ok(placed) = finalize_exact_accumulator(placed, root) else { + return false; + }; candidate.replacement = Replacement::Summary(placed); - candidate.rationale.push_str("; fixed-window precompute over complete per-series counter states"); + candidate + .rationale + .push_str("; fixed-window precompute over complete per-series counter states"); + true + }); + proposals + } + + /// Retain grouped Sum after a per-series Rate readout as a query-time + /// candidate alongside its complete-window maintenance placement. + pub fn query_time_rate_aggregation_candidates(&self, root: &Rc) -> Proposals { + fn query_time(node: &Rc) -> Rc { + let mut next = node.as_ref().clone(); + match &mut next.expr { + SummaryExpr::ValueOperation { + child, + operation: ValueOperation::FinalizeExactAccumulator, + timing, + } if matches!( + &child.expr, + SummaryExpr::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) => + { + *timing = ExecutionTiming::QueryTime; + } + SummaryExpr::ValueOperation { child, .. } + | SummaryExpr::SummaryAgg { child, .. } => *child = query_time(child), + _ => {} + } + Rc::new(next) + } + let mut proposals = self.fixed_window_rate_candidates(root); + proposals.candidates.retain_mut(|candidate| { + let Replacement::Summary(node) = &candidate.replacement else { return false }; + if !matches!(&node.expr, SummaryExpr::ValueOperation { child, operation: ValueOperation::FinalizeExactAccumulator, .. } + if matches!(&child.expr, SummaryExpr::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), .. })) { return false; } + candidate.replacement = Replacement::Summary(query_time(node)); + candidate.rationale = "query-time grouped Sum over complete per-series Rate readouts".into(); true }); proposals diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index c69a6b59..83208f50 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -253,8 +253,8 @@ pub fn compile_current_series_readout( ) } -/// Compile a selected per-series Rate -> ranking computation above its exact -/// counter readout. Deployments bind complete window readouts at this boundary; +/// Compile selected ranking or aggregation above an exact per-series Rate +/// readout. Deployments bind complete window readouts at this boundary; /// the heap is rebuilt independently for each evaluation. This does not move /// that frontier to ingestion time or authorize combining finalized rates. pub fn compile_rate_ranking( @@ -317,10 +317,10 @@ pub fn compile_rate_ranking( Ok((source, program)) } -/// The selected logical placement requires a fresh heap for each closed window. +/// The selected logical placement requires fresh aggregate state per closed window. /// Compile both physical graphs before deployment chooses storage or scheduling. /// The input is the complete collection of per-series exact counter states. -pub fn compile_fixed_window_rate_ranking( +pub fn compile_fixed_window_rate_aggregation( selected: &Rc, ) -> Result { use planner_types::post_asap::{ @@ -341,11 +341,31 @@ pub fn compile_fixed_window_rate_ranking( ) }) .collect::>(); - let heaps = dag.nodes.iter().filter(|n| n.output_state.timing == ExecutionTiming::IngestionTime && matches!(&n.payload, Payload::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), .. - } if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap))).collect::>(); + let heaps = dag + .nodes + .iter() + .filter(|n| { + n.output_state.timing == ExecutionTiming::IngestionTime + && match &n.payload { + Payload::SummaryAgg { + family: SummaryFamilyType::Sketch(kind, _), + .. + } => matches!( + kind.algorithm(), + SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap + ), + Payload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + .. + } => true, + _ => false, + } + }) + .collect::>(); let ([source], [heap]) = (sources.as_slice(), heaps.as_slice()) else { - return Err(invalid("expected one selected fixed-window Rate heap")); + return Err(invalid( + "expected one selected fixed-window Rate aggregation", + )); }; if !source .output_schema @@ -354,7 +374,7 @@ pub fn compile_fixed_window_rate_ranking( .any(|f| f.name == SERIES_IDENTITY_COLUMN) { return Err(invalid( - "fixed-window Rate heap requires complete series identity", + "fixed-window Rate aggregation requires complete series identity", )); } compile_candidate( diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index a81a1841..e4be66c1 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -773,7 +773,7 @@ fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { &EqualSplitAllocator, &Evidence, ); - let candidates = strategy.fixed_window_rate_topk_candidates(&root).candidates; + let candidates = strategy.fixed_window_rate_candidates(&root).candidates; assert_eq!(candidates.len(), 2); for candidate in candidates { let Replacement::Summary(root) = candidate.replacement else { @@ -817,7 +817,7 @@ fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { &[u64::from(heap.id.0)], ) .unwrap(); - let exported = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_ranking(&root).unwrap(); + let exported = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_aggregation(&root).unwrap(); assert_eq!(exported.encode().unwrap(), physical.encode().unwrap()); assert!( asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking(&root) @@ -958,3 +958,59 @@ fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { .contains("KeyedSummaryBuild")); } } + +// Grouped Rate has a legal stored Sum candidate as well as query-time reduction. +#[test] +fn grouped_rate_exposes_precomputed_sum_with_query_readout() { + let root = Rc::new( + asap_physical_operators::physical_planner::promql_rows::with_series_identity( + &lower_promql("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact).unwrap(), + ) + .unwrap(), + ); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let direct = strategy.query_time_rate_aggregation_candidates(&root); + assert!( + direct.candidates.iter().any(|candidate| { + let Replacement::Summary(root) = &candidate.replacement else { + return false; + }; + let Ok((_, program)) = + asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking(root) + else { + return false; + }; + let output = program.output_contract(program.roots()[0]).unwrap(); + output + .schema + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_))) + }), + "query-time grouped Rate must finalize Sum inside the physical graph" + ); + let candidates = strategy.fixed_window_rate_candidates(&root).candidates; + assert!( + !candidates.is_empty(), + "Planner must expose Rate -> grouped Sum at ingestion" + ); + for candidate in candidates { + let Replacement::Summary(root) = candidate.replacement else { + panic!() + }; + let physical = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_aggregation(&root).unwrap(); + let precompute = String::from_utf8(physical.precompute.unwrap().encode().unwrap()).unwrap(); + assert!( + precompute.contains("SummaryBuild") + && precompute.contains("Rate") + && precompute.contains("Sum") + ); + let query = String::from_utf8(physical.query.encode().unwrap()).unwrap(); + assert!(query.contains("Readout") && !query.contains("SummaryBuild")); + } +} From b860407ff1c67091a76396519b8c5401027ae37d Mon Sep 17 00:00:00 2001 From: zz_y Date: Mon, 28 Sep 2026 14:00:30 +0000 Subject: [PATCH 68/90] feat: bind persisted semantic definitions to logical dataset identity --- crates/types/src/post_asap/mod.rs | 2 +- .../src/post_asap/semantic_definition.rs | 76 ++++++++++++++++++- 2 files changed, 76 insertions(+), 2 deletions(-) diff --git a/crates/types/src/post_asap/mod.rs b/crates/types/src/post_asap/mod.rs index 038276ea..67e38645 100644 --- a/crates/types/src/post_asap/mod.rs +++ b/crates/types/src/post_asap/mod.rs @@ -36,7 +36,7 @@ pub mod post_asap_dag; pub mod query_time; pub mod schema; pub mod semantic_definition; -pub use semantic_definition::SummarySemanticFragment; +pub use semantic_definition::{LogicalDatasetIdentity, SummarySemanticFragment}; pub mod sketch; pub mod summary_maintenance; pub mod summary_maintenance_lifecycle; diff --git a/crates/types/src/post_asap/semantic_definition.rs b/crates/types/src/post_asap/semantic_definition.rs index ebe61560..1337222f 100644 --- a/crates/types/src/post_asap/semantic_definition.rs +++ b/crates/types/src/post_asap/semantic_definition.rs @@ -7,10 +7,31 @@ use serde::{Deserialize, Serialize}; use sha2::{Digest, Sha256}; use std::collections::{BTreeMap, BTreeSet}; +/// Stable identity of a logical input dataset, independent of its endpoint. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct LogicalDatasetIdentity { + pub namespace: String, + pub dataset: String, +} + +impl LogicalDatasetIdentity { + pub fn validate(&self) -> Result<(), String> { + if self.namespace.trim().is_empty() || self.dataset.trim().is_empty() { + return Err("dataset namespace and identity must be nonempty".into()); + } + Ok(()) + } +} + #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct SummarySemanticFragment { pub format_version: u32, + /// Version 1 fragments are unbound structural descriptions. Persisted, + /// dataset-bound descriptions use version 2. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub dataset_identity: Option, pub output: String, pub nodes: BTreeMap, } @@ -73,6 +94,20 @@ impl SummarySemanticFragment { Self::export(dag, output, true) } + /// All source names in this DAG resolve within this logical dataset. + pub fn from_stored_output_in_dataset( + dag: &ExecutableDag, + output: PostAsapNodeId, + dataset: LogicalDatasetIdentity, + ) -> Result { + dataset.validate()?; + let mut fragment = Self::export(dag, output, true)?; + fragment.format_version = 2; + fragment.dataset_identity = Some(dataset); + fragment.validate()?; + Ok(fragment) + } + pub fn from_dag(dag: &ExecutableDag, output: PostAsapNodeId) -> Result { Self::export(dag, output, false) } @@ -215,6 +250,7 @@ impl SummarySemanticFragment { let mut hashes: BTreeMap = BTreeMap::new(); let mut result = Self { format_version: 1, + dataset_identity: None, output: String::new(), nodes: BTreeMap::new(), }; @@ -284,7 +320,12 @@ impl SummarySemanticFragment { } pub fn validate(&self) -> Result<(), String> { - if self.format_version != 1 + match (&self.dataset_identity, self.format_version) { + (None, 1) => (), + (Some(dataset), 2) => dataset.validate()?, + _ => return Err("semantic version and dataset binding disagree".into()), + } + if !matches!(self.format_version, 1 | 2) || self.nodes.is_empty() || self.nodes.len() > 4096 || canonical_bytes(self)?.len() > 4 * 1024 * 1024 @@ -372,6 +413,39 @@ mod tests { .unwrap() } + // Equal source names in different datasets must not alias persisted meaning. + #[test] + fn dataset_identity_is_semantic_and_roundtrips() { + let dag = fixture("latency"); + let export = |namespace: &str| { + SummarySemanticFragment::from_stored_output_in_dataset( + &dag, + dag.root, + LogicalDatasetIdentity { + namespace: namespace.into(), + dataset: "requests".into(), + }, + ) + .unwrap() + }; + let a = export("tenant-a"); + assert_ne!( + canonical_bytes(&a).unwrap(), + canonical_bytes(&export("tenant-b")).unwrap() + ); + assert_eq!(a, export("tenant-a")); // No endpoint enters the semantic API. + let restored: SummarySemanticFragment = + serde_json::from_slice(&canonical_bytes(&a).unwrap()).unwrap(); + restored.validate().unwrap(); + assert_eq!(a, restored); + let mut bad = a.clone(); + bad.dataset_identity.as_mut().unwrap().namespace.clear(); + assert!(bad.validate().is_err()); + bad = a; + bad.dataset_identity = None; + assert!(bad.validate().is_err()); + } + // Storage identity must ignore temporary identifiers and execution placement. #[test] fn identity_ignores_node_ids_and_phase() { From a07456885bddd5c4b9fdea949dd93daee304cef3 Mon Sep 17 00:00:00 2001 From: zz_y Date: Mon, 28 Sep 2026 14:16:57 +0000 Subject: [PATCH 69/90] docs: specify dataset-bound semantic export contract --- .../physical-planning-and-deployment.md | 42 +++++++++++++++---- 1 file changed, 35 insertions(+), 7 deletions(-) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 5f8571a9..6961fe9f 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -467,19 +467,47 @@ snapshots as separate inputs. ## 6. Executable acceptance coverage -The tests distinguish optimizer-selected lifecycle execution from explicit -physical pane construction: +The tests cover optimizer-selected lifecycle execution and automatic temporal +pane compilation, alongside independent operator/runtime fixtures: | Test | Contract exercised | | --- | --- | +| `summary_maintenance_lifecycle_e2e::selected_temporal_lifecycle_compiles_panes_and_executes` | PromQL p50/p99 workloads → selected continuous lifecycle and Sliding framework → automatically generated maintenance/query DAGs → real codec round-trip → adjacent aligned windows; checks filters, entity identity, sample counts, missing/duplicate panes and phase rejection before opening readers | | `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled maintenance/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | | `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute maintenance DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | | `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | | `precompute_candidates::grouped_rate_can_be_materialized_before_or_after_grouped_sum` | Cost changes select different legal precompute frontiers; both selected candidates execute with the same reset-sensitive result; uncompilable candidates are not priced | | `sql_to_physical::sql_filter_grouped_sum_executes_and_rebinds` | SQL text → candidate search → physical compilation → shared Scan predicates and grouped summary execution; NULL samples are ignored and fresh bindings produce new results | -The pane test uses an explicit physical realization. It does not establish that -maintenance selection automatically emits the complete temporal pane DAG. -Pane phase validation uses the Planner coverage contract; concrete stored-pane -identity, revision, readiness and complete coverage of the required input samples remain deployment checks. -Real storage and HTTP execution belong to deployment-repository E2E tests. +`physical_planner::compile_temporal_pane_candidate` consumes the logical DAG, +selected lifecycle/framework and a generic pane/entity input contract. It +generates pane construction, scan predicates, ordered state slots, a shared +merge, quantile readouts and run-scoped timestamps. A physical pane output has +its own identity: one minute of state cannot masquerade as the logical +five-minute summary. The returned candidate retains the maintenance contract. + +This initial realization supports bounded, complete KLL panes with known phase +and resolved entity identity, for Sliding windows or a single Tumbling window. +Source capability evidence must declare all entity keys or isolate one entity; +usage-derived PromQL columns alone cannot establish that identity. Partial edge +panes, exponential histograms and cross-run delta accumulation require further +physical candidates and are rejected by this entry point. + +Physical execution checks pane timestamps and duplicate entity states. Concrete +stored identity, revisions, readiness and complete coverage of required input samples remain +deployment responsibilities. Real storage and HTTP execution belong to +deployment-repository E2E tests. + +### Dataset-bound semantic export + +Backend supplies a stable `LogicalDatasetIdentity { namespace, dataset }` before +Planner exports a persisted definition. Source names within the exported DAG are +resolved in that dataset. Tenant A's `KLL(latency)` and tenant B's `KLL(latency)` +therefore differ; relocating the same dataset to another endpoint does not. + +`SummarySemanticFragment::from_stored_output_in_dataset` exports version 2 with +this identity. Version 1 remains an unbound structural description; Backend's new +planning path uses version 2 for persisted outputs and validates the identity +against the installed input binding. Endpoint and replica information do not enter +semantic identity. Definitions do not authorize cross-deployment reads or adoption +of another plan version's state. From 39ad0e4526d3ffad4cd93fc5d130a3f81f97948e Mon Sep 17 00:00:00 2001 From: zz_y Date: Mon, 28 Sep 2026 14:17:10 +0000 Subject: [PATCH 70/90] docs: group dataset identity with semantic input contracts --- .../physical-planning-and-deployment.md | 28 +++++++++---------- 1 file changed, 14 insertions(+), 14 deletions(-) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 6961fe9f..30a28857 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -116,6 +116,20 @@ of each pane. Nested computations retain their own time semantics. Planner expor this contract through `SummarySemanticFragment`; changing its semantic wire vocabulary requires an explicit format-version review. +### Dataset-bound semantic export + +Backend supplies a stable `LogicalDatasetIdentity { namespace, dataset }` before +Planner exports a persisted definition. Source names within the exported DAG are +resolved in that dataset. Tenant A's `KLL(latency)` and tenant B's `KLL(latency)` +therefore differ; relocating the same dataset to another endpoint does not. + +`SummarySemanticFragment::from_stored_output_in_dataset` exports version 2 with +this identity. Version 1 remains an unbound structural description; Backend's new +planning path uses version 2 for persisted outputs and validates the identity +against the installed input binding. Endpoint and replica information do not enter +semantic identity. Definitions do not authorize cross-deployment reads or adoption +of another plan version's state. + ### Running example Suppose p50 and p99 are requested over the same latency samples in a five-minute window, @@ -497,17 +511,3 @@ Physical execution checks pane timestamps and duplicate entity states. Concrete stored identity, revisions, readiness and complete coverage of required input samples remain deployment responsibilities. Real storage and HTTP execution belong to deployment-repository E2E tests. - -### Dataset-bound semantic export - -Backend supplies a stable `LogicalDatasetIdentity { namespace, dataset }` before -Planner exports a persisted definition. Source names within the exported DAG are -resolved in that dataset. Tenant A's `KLL(latency)` and tenant B's `KLL(latency)` -therefore differ; relocating the same dataset to another endpoint does not. - -`SummarySemanticFragment::from_stored_output_in_dataset` exports version 2 with -this identity. Version 1 remains an unbound structural description; Backend's new -planning path uses version 2 for persisted outputs and validates the identity -against the installed input binding. Endpoint and replica information do not enter -semantic identity. Definitions do not authorize cross-deployment reads or adoption -of another plan version's state. From ec66cf552314f31f9ae9ca738c516372e8db8e03 Mon Sep 17 00:00:00 2001 From: zz_y Date: Mon, 28 Sep 2026 20:07:16 +0000 Subject: [PATCH 71/90] fix: preserve grouped samples through temporal physical plans --- .../src/operators/mod.rs | 9 ++ .../src/physical_planner/compiled.rs | 20 ++++ .../tests/physical_dag.rs | 102 ++++++++++++++++++ crates/types/src/pre_asap/query_expr.rs | 49 +++++++-- 4 files changed, 172 insertions(+), 8 deletions(-) diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index 0aba8b86..bef6b913 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -114,6 +114,15 @@ pub struct Operator { output: Schema, } impl Operator { + pub(crate) fn row_preserving_input(&self) -> Option { + match self.kind { + Kind::Filter(_) | Kind::Sort { .. } | Kind::Limit { .. } | Kind::SemiJoin { .. } => { + Some(0) + } + _ => None, + } + } + pub(crate) fn is_counter_readout(&self) -> bool { matches!( self.kind, diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs index 3edbf536..da97def5 100644 --- a/crates/asap-physical-operators/src/physical_planner/compiled.rs +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -117,6 +117,26 @@ impl CompiledPhysicalDag { } Ok(()) } + /// Identify the external input whose rows survive unchanged at this output. + /// Protocol adapters can retain labels that are outside a closed physical schema. + pub fn row_source(&self, id: NodeId) -> Option { + match self.nodes.get(&id)? { + Node::Input(_) => Some(id), + Node::Operator { inputs, operator } => { + let index = operator.row_preserving_input()?; + self.row_source(*inputs.get(index)?) + } + } + } + + /// Selected operator name, for plan inspection without decoding its wire format. + pub fn operator_name(&self, id: NodeId) -> Option<&str> { + match self.nodes.get(&id)? { + Node::Input(_) => Some("Input"), + Node::Operator { operator, .. } => Some(operator.name()), + } + } + pub fn roots(&self) -> &[NodeId] { &self.roots } diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 7eb9c799..974b47ab 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -1087,3 +1087,105 @@ fn assert_weighted_rate_topk(count_sketch: bool) { assert_eq!(services, vec!["auth", "checkout", "ingest", "export"]); } } + +// The grouped temporal reducer's sample schema must survive physical Sort/Limit binding. +#[test] +fn grouped_temporal_schema_compiles_and_executes_topk() { + use asap_physical_operators::physical_planner::{ + compile_node, CompiledPhysicalDag, InputContract, Source, + }; + use planner_types::post_asap::{ + ExecutableDagNode, ExecutableOperatorPayload, ExecutionDataState, PostAsapNodeId, + ValueOperation, + }; + use planner_types::pre_asap::{ + aggregate_output_schema, AggIntent, Column, GroupKeys, QueryExpr, Reduction as IrReduction, + Schema as IrSchema, + }; + let grouped = IrSchema::new(vec![ + Column::new("job", DataType::Utf8, false), + Column::new("sum", DataType::Float64, false), + ]); + let output = aggregate_output_schema( + &grouped, + &IrReduction::PerEntity, + &[AggIntent::Avg { col: None }], + &[], + ) + .unwrap(); + let input = schema( + &output + .columns + .iter() + .map(|c| (c.name.as_str(), c.dtype.clone(), c.nullable)) + .collect::>(), + ); + let node = |id, operation| ExecutableDagNode { + id: PostAsapNodeId(id), + payload: ExecutableOperatorPayload::Value { operation }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*input).clone(), + guarantee: None, + }; + let sort = compile_node( + &node( + 1, + ValueOperation::Sort { + keys: vec![planner_types::pre_asap::SortKey { + expr: QueryExpr::Column(1), + ascending: false, + nulls_first: false, + }], + partition_by: GroupKeys::none(), + }, + ), + &[input.clone()], + ) + .unwrap(); + let limit = compile_node( + &node( + 2, + ValueOperation::Limit { + n: 1, + offset: 0, + partition_by: GroupKeys::none(), + }, + ), + &[input.clone()], + ) + .unwrap(); + let compiled = CompiledPhysicalDag::from_operators( + [(0, InputContract::bounded(input.clone()))].into(), + [(1, (vec![0], sort)), (2, (vec![1], limit))].into(), + vec![2], + ) + .unwrap(); + let recovered = CompiledPhysicalDag::decode(&compiled.encode().unwrap()).unwrap(); + assert_eq!(recovered.row_source(2), Some(0)); + assert_eq!(recovered.operator_name(2), Some("Limit")); + let expected = vec![Value::Utf8("api".into()), Value::Float64(9.)]; + let batch = Batch::try_new( + input.clone(), + vec![ + vec![Value::Utf8("worker".into()), Value::Float64(2.)], + expected.clone(), + ], + ) + .unwrap(); + let source = Box::new(Operator::source(input, vec![batch]).unwrap()) as Source<'_>; + let physical = recovered.instantiate([(0, source)].into()).unwrap(); + let mut stream = physical + .execute(&[2], RunContext::new(query(), Limits::default()).unwrap()) + .unwrap() + .remove(0); + let rows = block_on(async { + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend_from_slice(batch.unwrap().rows()); + } + rows + }); + assert_eq!(rows.len(), 1); + assert!(matches!(&rows[0][0], Value::Utf8(label) if label.as_ref() == "api")); + assert!(matches!(rows[0][1], Value::Float64(9.))); +} diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs index 87053ee1..dec0107a 100644 --- a/crates/types/src/pre_asap/query_expr.rs +++ b/crates/types/src/pre_asap/query_expr.rs @@ -65,6 +65,8 @@ pub enum QueryExprError { /// used by `Project`'s own `output_schema` arm instead). #[error("a scalar expression has no row schema of its own")] ScalarHasNoRowSchema, + #[error("invalid per-series sample column: {0}")] + InvalidSampleColumn(String), } // ── Leaf / supporting types ─────────────────────────────────────────────────── @@ -1451,12 +1453,23 @@ impl QueryExpr { /// one value per series, so every label column of `input` is preserved and only /// the sample value is replaced — kept named `value` so the PromQL sample-value /// convention (and any outer `SampleValue` reference) still resolves it by name. -fn per_series_reduction_schema(input: &Schema, agg: &AggIntent) -> Schema { - let value_idx = input - .column_id("value") - .or_else(|| (0..input.columns.len()).find(|&i| Some(i) != input.time_index)); +fn per_series_reduction_schema(input: &Schema, agg: &AggIntent) -> Result { + let vi = if let Some(index) = agg.input_cols().first() { + *index + } else { + super::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, input) + .map_err(|error| QueryExprError::InvalidSampleColumn(error.to_string()))? + }; + if !matches!( + input.columns.get(vi).map(|column| &column.dtype), + Some(DataType::Float64 | DataType::Int64) + ) { + return Err(QueryExprError::InvalidSampleColumn(format!( + "column {vi} is not numeric" + ))); + } let mut columns = input.columns.clone(); - if let Some(vi) = value_idx { + { let mut out = agg.output_column(&columns[vi]); out.name = "value".into(); // A per-series range reduction produces a PromQL sample value, which is @@ -1466,14 +1479,14 @@ fn per_series_reduction_schema(input: &Schema, agg: &AggIntent) -> Schema { out.dtype = DataType::Float64; columns[vi] = out; } - Schema { + Ok(Schema { columns, time_index: input.time_index, unique_keys: input.unique_keys.clone(), // Per-series reduction is label-preserving: it inherits its input's // completeness (an open scan stays open; a closed one stays closed). closed: input.closed, - } + }) } /// The output schema of an `Aggregate { reduction, measures }` over `in_schema` — @@ -1500,7 +1513,7 @@ pub fn aggregate_output_schema( 1, "a per-entity reduction is single-aggregate" ); - return Ok(per_series_reduction_schema(in_schema, &measures[0])); + return per_series_reduction_schema(in_schema, &measures[0]); } Reduction::Reduce(by) => by, }; @@ -2286,6 +2299,26 @@ mod tests { assert_eq!(back, s); } + // Nested temporal aggregation must replace the sample, never the grouping label. + #[test] + fn temporal_reduction_of_grouped_sum_preserves_job() { + let input = Schema::new(vec![ + col("job", DataType::Utf8, true), + col("sum", DataType::Float64, false), + ]); + for aggregate in [ + AggIntent::Avg { col: None }, + AggIntent::Avg { col: Some(1) }, + AggIntent::Rate, + ] { + let output = + aggregate_output_schema(&input, &Reduction::PerEntity, &[aggregate], &[]).unwrap(); + assert_eq!(output.columns[0], input.columns[0]); + assert_eq!(output.columns[1].name, "value"); + assert_eq!(output.columns[1].dtype, DataType::Float64); + } + } + #[test] fn per_series_rate_preserves_labels() { // A per-series range reduction (`rate`) is label-preserving: it produces From 70893e6e28048f9fc9c0d3204654c7f7995de229 Mon Sep 17 00:00:00 2001 From: zz_y Date: Mon, 28 Sep 2026 20:41:46 +0000 Subject: [PATCH 72/90] fix: finalize exact state at exposed query candidate roots --- crates/asap-aware-mapping/src/replacement.rs | 28 +++ .../tests/precompute_candidates.rs | 160 +++++++++++++++++- 2 files changed, 186 insertions(+), 2 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 6e371194..2ec5f8fd 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -4189,6 +4189,9 @@ impl PlanSpace { .map(|(id, root)| { assembly .assemble_target(root) + // Exposed query candidates return values. Internal assembly + // still retains accumulator states for sharing and storage. + .and_then(|node| finalize_exact_accumulator(node, root)) .map(|node| (id.clone(), node)) }) .collect::, _>>(); @@ -6871,6 +6874,31 @@ mod tests { use asap_types::types::AccuracyTarget; use std::collections::HashMap; + // Every exposed query result has a readout; internal accumulator frontiers stay states. + #[test] + fn query_candidate_roots_do_not_leak_exact_accumulator_state() { + for query in [ + "sum by(job)(rate(m[1m]))", + "sum by(job)(m)", + "sum_over_time(m[1m])", + ] { + let root = Rc::new(lower_promql(query, AccuracyTarget::Exact)); + let space = search_workload(vec![(0usize, root.clone())]); + let inventory = space.enumerate_candidate_dags(4096).unwrap(); + assert!(!inventory.candidates.is_empty()); + for node in inventory.candidates.iter().map(|forest| &forest[0].1) { + assert!( + node.schema + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_))), + "{query}: query root leaks state: {:?}", + node.schema + ); + } + } + } + #[test] fn unpriced_inventory_retains_quantile_families_and_raw_execution() { let query = Rc::new(agg(vec![2], default_quantile(0.9), metric_scan(&["job"]))); diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index f965a215..2acb39a0 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -14,7 +14,7 @@ use futures::{executor::block_on, StreamExt}; use planner_types::{post_asap::*, pre_asap::DataType, types::AccuracyTarget, workload::*}; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; -fn grouped_rate() -> ExecutableDag { +fn grouped_rate_space() -> asap_aware_mapping::PlanSpace<&'static str> { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -44,7 +44,15 @@ fn grouped_rate() -> ExecutableDag { .unwrap() .remove(0), ); - let space = search_workload(vec![("grouped-rate", root)]); + let root = Rc::new( + asap_physical_operators::physical_planner::promql_rows::with_series_identity(&root) + .unwrap(), + ); + search_workload(vec![("grouped-rate", root)]) +} + +fn grouped_rate() -> ExecutableDag { + let space = grouped_rate_space(); let selected = space .global_selection(&DefaultCostModel) .assemble_selected_dag(&space.roots[0].1) @@ -402,3 +410,151 @@ fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { && !c.materialized_outputs.contains_key(&roots[0]))); assert!(enumerate_frontiers(&dag, &inputs, &roots, 1).is_err()); } + +#[test] +fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { + let inventory = grouped_rate_space().enumerate_candidate_dags(4096).unwrap(); + let mut executed = 0; + for forest in inventory.candidates { + let root = &forest[0].1; + let dag = compile_executable_dag(root).unwrap(); + let Some(state) = dag.nodes.iter().find(|node| { + matches!( + node.payload, + ExecutableOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) else { + continue; + }; + let boundary = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + ExecutableOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + .. + } + ) + }) + .map(|node| u64::from(node.id.0)) + .unwrap_or(u64::from(dag.root.0)); + let physical_candidates = compile_candidates( + &dag, + BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &[vec![], vec![boundary]], + ); + let (family, input, grouping) = match &state.payload { + ExecutableOperatorPayload::SummaryAgg { + family, + input, + grouping, + .. + } => (family, input, grouping), + _ => unreachable!(), + }; + let schema = Arc::new(state.output_schema.clone()); + let rows = ["a", "b"] + .into_iter() + .map(|instance| { + let mut accumulator = create_planner_accumulator(family, input, grouping).unwrap(); + for (timestamp, value) in [(1_000, 1.), (31_000, 31.), (59_000, 59.)] { + accumulator.update_single(value, timestamp); + } + let summary = Value::Summary { + family: family.clone(), + state: Arc::from(accumulator.into_accumulator()), + }; + schema + .fields + .iter() + .map(|field| match &field.dtype { + SummaryFamilyType::ExactAggregate(..) => summary.clone(), + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), + SummaryFamilyType::Plain(DataType::Utf8) + if field.name == "$promql_series_identity" => + { + Value::Utf8( + serde_json::to_string(&BTreeMap::from([ + ("job", "api"), + ("instance", instance), + ])) + .unwrap() + .into(), + ) + } + SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8("api".into()), + _ => panic!("unexpected input field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(schema, rows).unwrap(); + for physical in physical_candidates { + let physical = physical.unwrap(); + let inputs = if let Some(precompute) = &physical.precompute { + let source_id = precompute.input_contracts().next().unwrap().0; + let stored = run( + precompute, + BTreeMap::from([(source_id, batch.clone())]), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 60_000, + revision: 1, + }, + ); + assert_eq!(stored.len(), 1); + // Persist/recover the actual materialization boundary before reading it. + let bytes = asap_physical_operators::stored_state::native::encode_batch(&stored[0]) + .unwrap(); + let recovered = asap_physical_operators::stored_state::native::decode_batch( + &bytes, + stored[0].schema().clone(), + 1 << 20, + ) + .unwrap(); + BTreeMap::from([(precompute.roots()[0], recovered)]) + } else { + BTreeMap::from([( + physical.query.input_contracts().next().unwrap().0, + batch.clone(), + )]) + }; + let output = run( + &physical.query, + inputs, + Scope::Query { + evaluation_time_ms: 60_000, + revision: 1, + }, + ); + assert_eq!(output.len(), 1); + assert_eq!(output[0].rows().len(), 1); + assert!(output[0] + .schema() + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + assert!( + output[0].rows()[0] + .iter() + .any(|value| matches!(value, Value::Float64(x) if (*x - 2.).abs() < 1e-12)), + "{:?}", + output[0].rows() + ); + executed += 1; + } + } + assert!( + executed >= 2, + "must execute both stored and query-time grouped Rate candidates: {executed}" + ); +} From f674e4572ba97ff31785673fc1ab9a00b019111d Mon Sep 17 00:00:00 2001 From: zz_y Date: Mon, 28 Sep 2026 20:48:42 +0000 Subject: [PATCH 73/90] fix: expose finalized globally selected query results --- crates/asap-aware-mapping/src/replacement.rs | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 2ec5f8fd..2bdc6592 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -5186,6 +5186,18 @@ impl<'a> GlobalSelection<'a> { self.assemble_target(target).map(Some) } + /// Assemble a complete query result, including an exact-state readout when + /// needed. `assemble_selected_dag` also serves internal state frontiers; + /// callers exposing query results must use this boundary instead. + pub fn assemble_selected_query( + &self, + target: &Rc, + ) -> Result>, RealizationError> { + self.assemble_selected_dag(target)? + .map(|node| finalize_exact_accumulator(node, target)) + .transpose() + } + fn assemble_target(&self, target: &Rc) -> Result, RealizationError> { let ptr = Rc::as_ptr(target); if let Some(node) = self.assembled_nodes.borrow().get(&ptr) { From c3a4e0ad97e520b905f675a672fcff23e5153f91 Mon Sep 17 00:00:00 2001 From: zz_y Date: Mon, 28 Sep 2026 20:49:47 +0000 Subject: [PATCH 74/90] test: cover selected query result schema boundary --- crates/asap-aware-mapping/src/replacement.rs | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 2bdc6592..a2c0819d 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -6898,7 +6898,17 @@ mod tests { let space = search_workload(vec![(0usize, root.clone())]); let inventory = space.enumerate_candidate_dags(4096).unwrap(); assert!(!inventory.candidates.is_empty()); - for node in inventory.candidates.iter().map(|forest| &forest[0].1) { + let selected = space + .global_selection(&DefaultCostModel) + .assemble_selected_query(&space.roots[0].1) + .unwrap() + .unwrap(); + for node in inventory + .candidates + .iter() + .map(|forest| &forest[0].1) + .chain(std::iter::once(&selected)) + { assert!( node.schema .fields From 1c4417e4d660f9b010a17d46efb43c7eaab91a5e Mon Sep 17 00:00:00 2001 From: zz_y Date: Mon, 28 Sep 2026 21:07:37 +0000 Subject: [PATCH 75/90] fix: expose Planner query finalization for direct candidate proposals --- crates/asap-aware-mapping/src/replacement.rs | 36 ++++++++++++++------ 1 file changed, 25 insertions(+), 11 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index a2c0819d..014c682e 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -1399,7 +1399,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { if asap_types::post_asap::compile_executable_dag(&placed).is_err() { return false; } - let Ok(placed) = finalize_exact_accumulator(placed, root) else { + let Ok(placed) = finalize_query_candidate(placed, root) else { return false; }; candidate.replacement = Replacement::Summary(placed); @@ -1849,7 +1849,7 @@ fn exact_topk_over_temporal_values( { return Ok(None); } - let values = finalize_exact_accumulator(values, child)?; + let values = finalize_query_candidate(values, child)?; let partition_by = reduction .group_keys() .ok_or(RealizationError::PhysicalRealization( @@ -2103,8 +2103,8 @@ fn realize_binary( return Ok(None); } - lhs_node = finalize_exact_accumulator(lhs_node, lhs)?; - rhs_node = finalize_exact_accumulator(rhs_node, rhs)?; + lhs_node = finalize_query_candidate(lhs_node, lhs)?; + rhs_node = finalize_query_candidate(rhs_node, rhs)?; let has_ratio_domains = ratio_domains.is_some(); let guarantee = if matches!(op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)) @@ -2168,7 +2168,7 @@ fn realize_binary( /// Put an explicit read boundary between maintained exact state and a /// query-time value consumer. Approximate summaries must already carry a /// `SummaryEstimate`, so they deliberately do not pass this predicate. -fn finalize_exact_accumulator( +pub fn finalize_query_candidate( node: Rc, logical_output: &QueryExpr, ) -> Result, RealizationError> { @@ -2949,7 +2949,7 @@ fn construct_summary_agg( } else if snapshot_weighted { // A fresh query-time summary consumes this evaluation's finalized rates. // Moving rate snapshots must never accumulate across evaluations. - finalize_exact_accumulator(bound_child, &input.child)? + finalize_query_candidate(bound_child, &input.child)? } else { let child = finalize_exact_accumulator_at( bound_child, @@ -4191,7 +4191,7 @@ impl PlanSpace { .assemble_target(root) // Exposed query candidates return values. Internal assembly // still retains accumulator states for sharing and storage. - .and_then(|node| finalize_exact_accumulator(node, root)) + .and_then(|node| finalize_query_candidate(node, root)) .map(|node| (id.clone(), node)) }) .collect::, _>>(); @@ -5194,7 +5194,7 @@ impl<'a> GlobalSelection<'a> { target: &Rc, ) -> Result>, RealizationError> { self.assemble_selected_dag(target)? - .map(|node| finalize_exact_accumulator(node, target)) + .map(|node| finalize_query_candidate(node, target)) .transpose() } @@ -5262,8 +5262,8 @@ impl<'a> GlobalSelection<'a> { let Some(pred) = normalized_pred else { return keep_pre_asap(target); }; - let left = finalize_exact_accumulator(self.assemble_target(left)?, left)?; - let right = finalize_exact_accumulator(self.assemble_target(right)?, right)?; + let left = finalize_query_candidate(self.assemble_target(left)?, left)?; + let right = finalize_query_candidate(self.assemble_target(right)?, right)?; let guarantee = relational_join_guarantee(left.guarantee.as_ref(), right.guarantee.as_ref()); let node = Rc::new(SummaryNode { @@ -5334,7 +5334,7 @@ impl<'a> GlobalSelection<'a> { ), _ => return keep_pre_asap(target), }; - let child = finalize_exact_accumulator(self.assemble_target(child_target)?, child_target)?; + let child = finalize_query_candidate(self.assemble_target(child_target)?, child_target)?; let guarantee = child.guarantee.clone(); let node = Rc::new(SummaryNode { expr: SummaryExpr::ValueOperation { @@ -6898,6 +6898,20 @@ mod tests { let space = search_workload(vec![(0usize, root.clone())]); let inventory = space.enumerate_candidate_dags(4096).unwrap(); assert!(!inventory.candidates.is_empty()); + let strategy = SketchAlgorithmStrategy::new(&DefaultCostModel); + for candidate in strategy.propose(&TargetSubDAG::new(&root)).candidates { + if let Replacement::Summary(node) = candidate.replacement { + let output = finalize_query_candidate(node, &root).unwrap(); + assert!( + output + .schema + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_))), + "direct candidate {query} leaks state" + ); + } + } let selected = space .global_selection(&DefaultCostModel) .assemble_selected_query(&space.roots[0].1) From 94864946c2342a50b7dd8a35a2583755581fae95 Mon Sep 17 00:00:00 2001 From: zz_y Date: Mon, 28 Sep 2026 21:27:17 +0000 Subject: [PATCH 76/90] test: borrow schemas directly in physical query regression --- crates/asap-physical-operators/tests/physical_dag.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 974b47ab..e19cbed2 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -1139,7 +1139,7 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { partition_by: GroupKeys::none(), }, ), - &[input.clone()], + std::slice::from_ref(&input), ) .unwrap(); let limit = compile_node( @@ -1151,7 +1151,7 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { partition_by: GroupKeys::none(), }, ), - &[input.clone()], + std::slice::from_ref(&input), ) .unwrap(); let compiled = CompiledPhysicalDag::from_operators( From 0066214e80f2f2fe05d1373af00cd95ae40fd2f4 Mon Sep 17 00:00:00 2001 From: zzylol Date: Mon, 28 Sep 2026 21:38:40 +0000 Subject: [PATCH 77/90] docs: name the precompute DAG after the field that holds it The design doc called one half of a PhysicalCandidate the "Maintenance Physical DAG" while the struct calls it `precompute`, and used both words elsewhere in the same document. The query half had one name throughout. The crates already keep these apart by layer: asap-aware-mapping owns the Summary Maintenance Lifecycle, and asap-physical-operators owns the physical DAGs a lifecycle compiles to. Borrowing the lifecycle's word for a physical object collapses that distinction, and consumers inherit the fork: the backend currently carries both vocabularies for the same thing, with a type named InstalledPostAsapDag whose own validator is called validate_maintenance. Use `precompute` for the DAG half, matching PhysicalCandidate, and say once where the halves are introduced that `maintenance` remains the lifecycle's word. Section 2 and every lifecycle reference are unchanged. Co-Authored-By: Claude Opus 5 (1M context) --- .../physical-planning-and-deployment.md | 26 ++++++++++++------- 1 file changed, 17 insertions(+), 9 deletions(-) diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 30a28857..73180984 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -175,7 +175,7 @@ KLLBuild(k=200) 3. Physical DAGs -Maintenance DAG: +Precompute DAG: RawInput ↓ NativeKllBuild(k=200) @@ -196,7 +196,7 @@ NativeKllMerge(k=200) 4. Deployment Plan / DAG -Maintenance: +Precompute: OTLP latency source ↓ run KLL build over each complete 1-minute input pane @@ -283,7 +283,15 @@ Physical DAG(s) For the running example, the lifecycle creates two execution boundaries. -### Maintenance Physical DAG +These two halves are named as `PhysicalCandidate` names them, `precompute` +and `query`. *Maintenance* stays the lifecycle's word (section 2): it covers +how state is built, retained, reused and scheduled. A precompute DAG is the +physical object that a maintenance lifecycle compiles to, so reusing +*maintenance* for it collapses two layers that the crates keep apart: +`asap-aware-mapping::summary_maintenance_*` owns the lifecycle, and +`asap-physical-operators::physical_planner` owns the DAGs. + +### Precompute Physical DAG ```text RawInputSlot( @@ -358,7 +366,7 @@ Candidate B: ``` Both preserve reset-aware Rate before Sum. Summing raw counters before Rate is -not equivalent. The counter-state build may be another maintenance DAG; typed +not equivalent. The counter-state build may be another precompute DAG; typed state inputs do not imply that a deployment can construct or bind those states. The shared library exposes `physical_planner::compile_candidates(...)` to lower @@ -397,7 +405,7 @@ Deployment Plan Compiler Deployment Plan / DAG ``` -For the maintenance DAG, it may produce: +For the precompute DAG, it may produce: ```text Source: @@ -450,7 +458,7 @@ The complete example makes the ownership boundary explicit: | **Summary Maintenance Candidate Generation** | Maintain 1-minute panes and reuse them for aligned five-minute queries | | **Summary Maintenance Lifecycle** | Record pane/window/freshness/reuse requirements | | **Physical Plan Compiler** | Lower to native KLL build, merge, and readout operators | -| **Physical DAG** | Define maintenance and query DAGs with typed input/output boundaries | +| **Physical DAG** | Define precompute and query DAGs with typed input/output boundaries | | **Deployment Plan Compiler** | Bind raw input and KLL state slots to concrete sources/materializations | | **Deployment Plan / DAG** | Specify maintenance schedules, stored-pane resolution and query execution | @@ -486,9 +494,9 @@ pane compilation, alongside independent operator/runtime fixtures: | Test | Contract exercised | | --- | --- | -| `summary_maintenance_lifecycle_e2e::selected_temporal_lifecycle_compiles_panes_and_executes` | PromQL p50/p99 workloads → selected continuous lifecycle and Sliding framework → automatically generated maintenance/query DAGs → real codec round-trip → adjacent aligned windows; checks filters, entity identity, sample counts, missing/duplicate panes and phase rejection before opening readers | -| `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled maintenance/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | -| `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute maintenance DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | +| `summary_maintenance_lifecycle_e2e::selected_temporal_lifecycle_compiles_panes_and_executes` | PromQL p50/p99 workloads → selected continuous lifecycle and Sliding framework → automatically generated precompute/query DAGs → real codec round-trip → adjacent aligned windows; checks filters, entity identity, sample counts, missing/duplicate panes and phase rejection before opening readers | +| `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled precompute/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | +| `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute precompute DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | | `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | | `precompute_candidates::grouped_rate_can_be_materialized_before_or_after_grouped_sum` | Cost changes select different legal precompute frontiers; both selected candidates execute with the same reset-sensitive result; uncompilable candidates are not priced | | `sql_to_physical::sql_filter_grouped_sum_executes_and_rebinds` | SQL text → candidate search → physical compilation → shared Scan predicates and grouped summary execution; NULL samples are ignored and fresh bindings produce new results | From 2af6666f9fbcb640dab311e223f6c84e4a74c65a Mon Sep 17 00:00:00 2001 From: zzylol Date: Mon, 28 Sep 2026 22:06:29 +0000 Subject: [PATCH 78/90] docs: name the precompute half consistently in the physical layer Two doc comments still called one half of a PhysicalCandidate "maintenance" while the struct field is `precompute`, and one of them sat directly above the field it described. The identifiers were already right: TemporalPaneMaintenance does hold lifecycle requirements, and asap-aware-mapping's SummaryMaintenanceDagExport does export the lifecycle plan, so neither moves. Co-Authored-By: Claude Opus 5 (1M context) --- .../src/physical_planner/temporal_panes.rs | 2 +- crates/asap-physical-operators/src/summary_kernels/factory.rs | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs b/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs index 46218fd0..4e4d5722 100644 --- a/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs +++ b/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs @@ -28,7 +28,7 @@ pub struct TemporalPaneMaintenance { pub entity_identity: TemporalEntityIdentity, } -/// Generated maintenance and query computation. `pane_inputs` is ordered from +/// Generated precompute and query computation. `pane_inputs` is ordered from /// the oldest complete pane to the newest; each run checks actual timestamps. #[derive(Clone)] pub struct TemporalPaneCandidate { diff --git a/crates/asap-physical-operators/src/summary_kernels/factory.rs b/crates/asap-physical-operators/src/summary_kernels/factory.rs index 4cbeb8a5..e2ed34b7 100644 --- a/crates/asap-physical-operators/src/summary_kernels/factory.rs +++ b/crates/asap-physical-operators/src/summary_kernels/factory.rs @@ -38,7 +38,7 @@ macro_rules! impl_clone_accumulator_methods { }; } -/// Shared update interface for query-time and maintenance-time accumulation. +/// Shared update interface for query-time and precompute-time accumulation. /// /// This provides a uniform interface over all accumulator types so that the /// worker loop doesn't need to know which concrete type it's dealing with. From 0fde51846acad682d0edcd1e9e21468d51fc8f22 Mon Sep 17 00:00:00 2001 From: zzylol Date: Mon, 28 Sep 2026 22:20:05 +0000 Subject: [PATCH 79/90] docs: align the accumulator input comment with its own trait doc The trait doc above it now reads "query-time and precompute-time accumulation"; the method doc two lines down still said maintenance for the same input. Co-Authored-By: Claude Opus 5 (1M context) --- crates/asap-physical-operators/src/summary_kernels/factory.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/crates/asap-physical-operators/src/summary_kernels/factory.rs b/crates/asap-physical-operators/src/summary_kernels/factory.rs index e2ed34b7..cd57454a 100644 --- a/crates/asap-physical-operators/src/summary_kernels/factory.rs +++ b/crates/asap-physical-operators/src/summary_kernels/factory.rs @@ -43,7 +43,7 @@ macro_rules! impl_clone_accumulator_methods { /// This provides a uniform interface over all accumulator types so that the /// worker loop doesn't need to know which concrete type it's dealing with. pub trait AccumulatorUpdater: Send { - /// Validate an immutable maintenance input before an updater can silently + /// Validate an immutable precompute input before an updater can silently /// discard a value outside its representable domain. fn validate_single_input(&self, value: f64) -> Result<(), String> { if value.is_finite() { From 4338cb22c91bfd2114382d13e46db9cfa0d9e2e4 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 13:17:58 +0000 Subject: [PATCH 80/90] feat: compile scalar and dynamic-label vector computation before binding --- .../src/expressions/mod.rs | 82 ++++++ .../src/operators/mod.rs | 8 + .../src/operators/persisted.rs | 4 + .../src/operators/vector_binary.rs | 232 +++++++++++++++ .../src/physical_planner/mod.rs | 7 + .../src/physical_planner/promql_values.rs | 155 ++++++++++ .../tests/promql_binary.rs | 268 ++++++++++++++++++ .../tests/promql_values.rs | 123 ++++++++ 8 files changed, 879 insertions(+) create mode 100644 crates/asap-physical-operators/src/operators/vector_binary.rs create mode 100644 crates/asap-physical-operators/src/physical_planner/promql_values.rs create mode 100644 crates/asap-physical-operators/tests/promql_binary.rs create mode 100644 crates/asap-physical-operators/tests/promql_values.rs diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs index b9429220..bdeb821a 100644 --- a/crates/asap-physical-operators/src/expressions/mod.rs +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -16,6 +16,12 @@ pub enum Expression { }, Planner(Box), Column(usize), + ExactFloat64(usize), + LabelSet { + column: usize, + labels: Vec, + without: bool, + }, Literal { value: Value, dtype: DataType, @@ -76,6 +82,36 @@ impl Expression { expression.validate_input(input)?; Ok(expression.dtype()) } + ExactFloat64(column) => { + let (dtype, nullable) = plain(input, *column)?; + if nullable || !matches!(dtype, DataType::Int64 | DataType::Float64) { + return Err(invalid( + "exact Float64 conversion requires non-null numeric input", + )); + } + Ok((DataType::Float64, false)) + } + LabelSet { column, labels, .. } => { + let (dtype, nullable) = plain(input, *column)?; + let expected = DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }; + if dtype != &expected + || nullable + || labels + .iter() + .collect::>() + .len() + != labels.len() + { + return Err(invalid( + "label projection requires a non-null Utf8 map and unique label names", + )); + } + Ok((expected, false)) + } Column(i) => { let (t, n) = plain(input, *i)?; Ok((t.clone(), n)) @@ -142,6 +178,52 @@ impl Expression { pub(crate) fn evaluate(&self, row: &[Value]) -> Result { use Expression::*; Ok(match self { + ExactFloat64(column) => match row[*column] { + Value::Float64(value) => Value::Float64(value), + Value::Int64(value) if value.unsigned_abs() <= (1u64 << 53) => { + Value::Float64(value as f64) + } + _ => { + return Err(invalid( + "numeric result cannot be represented exactly as Float64", + )) + } + }, + LabelSet { + column, + labels, + without, + } => { + let Value::Map(entries) = &row[*column] else { + return Err(invalid("label projection requires a map")); + }; + let mut selected = std::collections::BTreeMap::new(); + let mut seen = std::collections::BTreeSet::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("label projection requires Utf8 entries")); + }; + if !seen.insert(key.clone()) { + return Err(invalid("duplicate label name")); + } + let keep = if *without { + key.as_ref() != "__name__" + && !labels.iter().any(|label| label.as_str() == key.as_ref()) + } else { + labels.iter().any(|label| label.as_str() == key.as_ref()) + }; + if keep && !value.is_empty() { + selected.insert(key.clone(), value.clone()); + } + } + Value::Map( + selected + .into_iter() + .map(|(k, v)| (Value::Utf8(k), Value::Utf8(v))) + .collect::>() + .into(), + ) + } Binary { operator, left, diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index bef6b913..662414a1 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -26,6 +26,7 @@ mod projection; mod sort; mod source; mod summary; +pub(crate) mod vector_binary; pub use aggregate::Reduction; pub use sort::SortKey; #[derive(Clone, serde::Serialize, serde::Deserialize)] @@ -50,6 +51,10 @@ enum Kind { VectorToScalar { column: usize, }, + VectorBinary { + operator: planner_types::post_asap::BinaryOperator, + return_bool: bool, + }, Project(Vec), Filter(Expression), Limit { @@ -209,6 +214,7 @@ impl PhysicalOperator for Operator { matches!( self.kind, Kind::Sort { .. } + | Kind::VectorBinary { .. } | Kind::CurrentSeries { .. } | Kind::Aggregate { .. } | Kind::Window { .. } @@ -251,6 +257,7 @@ impl PhysicalOperator for Operator { Kind::Union => "Union", Kind::CurrentSeries { .. } => "CurrentSeries", Kind::VectorToScalar { .. } => "VectorToScalar", + Kind::VectorBinary { .. } => "VectorBinary", Kind::Project(_) => "Project", Kind::Filter(_) => "Filter", Kind::Limit { .. } => "Limit", @@ -288,6 +295,7 @@ impl PhysicalOperator for Operator { Kind::Source(_) | Kind::Union | Kind::VectorToScalar { .. } => { source::execute(self, inputs, context) } + Kind::VectorBinary { .. } => vector_binary::execute(self, inputs, context), Kind::Project(_) => projection::execute(self, inputs, context), Kind::CurrentSeries { .. } => current_series::execute(self, inputs, context), Kind::PaneInput { .. } | Kind::ScopeTimestamp { .. } => { diff --git a/crates/asap-physical-operators/src/operators/persisted.rs b/crates/asap-physical-operators/src/operators/persisted.rs index 85add57a..d6c53898 100644 --- a/crates/asap-physical-operators/src/operators/persisted.rs +++ b/crates/asap-physical-operators/src/operators/persisted.rs @@ -43,6 +43,10 @@ impl TryFrom for Operator { lookback_ms, } => Operator::current_series(input(0)?, identity, coordinate, value, lookback_ms)?, Kind::VectorToScalar { column } => Operator::vector_to_scalar(input(0)?, column)?, + Kind::VectorBinary { + operator, + return_bool, + } => Operator::vector_binary(input(0)?, input(1)?, operator, return_bool)?, Kind::Project(expressions) => { if expressions.len() != output.fields.len() { return Err(invalid("persisted projection width mismatch")); diff --git a/crates/asap-physical-operators/src/operators/vector_binary.rs b/crates/asap-physical-operators/src/operators/vector_binary.rs new file mode 100644 index 00000000..0ef77d80 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/vector_binary.rs @@ -0,0 +1,232 @@ +//! Label matching and scalar broadcasting are physical computation, not source binding. +use super::*; +use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; + +pub(crate) fn value_schema(scalar: bool) -> Schema { + let mut fields = Vec::new(); + if !scalar { + fields.push(result_field( + "labels", + DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }, + false, + )); + } + fields.push(result_field( + if scalar { "$promql_scalar" } else { "value" }, + DataType::Float64, + false, + )); + schema(fields) +} + +fn is_scalar(input: &Schema) -> Result { + for scalar in [true, false] { + let expected = value_schema(scalar); + if input.fields.len() == expected.fields.len() + && input + .fields + .iter() + .zip(&expected.fields) + .all(|(a, b)| a.dtype == b.dtype && !a.nullable) + { + return Ok(scalar); + } + } + Err(invalid( + "vector binary requires Float64 scalars or complete label-map vectors", + )) +} + +impl Operator { + pub fn vector_binary( + left: Schema, + right: Schema, + operator: BinaryOperator, + return_bool: bool, + ) -> Result { + let scalar = is_scalar(&left)? && is_scalar(&right)?; + is_scalar(&right)?; + let expression = Expression::Binary { + operator: operator.clone(), + left: Box::new(Expression::Column(0)), + right: Box::new(Expression::Column(1)), + }; + expression.dtype(&schema(vec![ + result_field("left", DataType::Float64, false), + result_field("right", DataType::Float64, false), + ]))?; + let comparison = matches!(operator.kind, BinaryOpKind::Compare(_)); + if (return_bool && !comparison) || (scalar && comparison && !return_bool) { + return Err(invalid("invalid scalar/vector comparison bool mode")); + } + Ok(Self { + inputs: vec![left, right], + output: value_schema(scalar), + kind: Kind::VectorBinary { + operator, + return_bool, + }, + }) + } +} + +type Labels = BTreeMap, Arc>; +fn labels(row: &[Value]) -> Result { + let Some(Value::Map(entries)) = row.first() else { + return Err(invalid("vector requires label map")); + }; + let mut result = BTreeMap::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("labels must be Utf8")); + }; + if result.insert(key.clone(), value.clone()).is_some() { + return Err(invalid("duplicate label name")); + } + } + Ok(result) +} +fn identity(mut labels: Labels) -> Labels { + labels.remove("__name__"); + labels.retain(|_, value| !value.is_empty()); + labels +} +fn value(row: &[Value]) -> Result { + match row.last() { + Some(Value::Float64(value)) => Ok(*value), + _ => Err(invalid("binary value must be Float64")), + } +} +fn label_bytes(labels: &Labels) -> usize { + labels.iter().map(|(k, v)| 64 + k.len() + v.len()).sum() +} + +pub(super) fn execute<'a>( + op: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::VectorBinary { + operator, + return_bool, + } = &op.kind + else { + unreachable!() + }; + let left_scalar = is_scalar(&op.inputs[0])?; + let right_scalar = is_scalar(&op.inputs[1])?; + let right = inputs.pop().ok_or_else(|| invalid("missing right input"))?; + let left = inputs.pop().ok_or_else(|| invalid("missing left input"))?; + Ok(futures::stream::once(async move { + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + if (left_scalar && left.len() != 1) || (right_scalar && right.len() != 1) { + return Err(invalid("scalar input must contain exactly one value")); + } + let mut workspace = Workspace::new(&context)?; + let mut work = Cooperative::new(&context); + let mut rows = Vec::new(); + let mut emit = |labels: Labels, a: f64, b: f64| -> Result<(), Error> { + let arithmetic = matches!(operator.kind, BinaryOpKind::Arithmetic(_)); + let result = match crate::expressions::arithmetic::evaluate_binary(operator, a, b)? { + Value::Float64(value) => value, + Value::Bool(value) if *return_bool => { + if value { + 1. + } else { + 0. + } + } + Value::Bool(true) => { + if left_scalar { + b + } else { + a + } + } + Value::Bool(false) => return Ok(()), + _ => return Err(invalid("invalid binary result")), + }; + let mut row = Vec::new(); + if !left_scalar || !right_scalar { + let labels = if arithmetic || *return_bool { + identity(labels) + } else { + labels + }; + workspace.grow( + label_bytes(&labels) + + std::mem::size_of::>() + + 2 * std::mem::size_of::(), + )?; + row.push(Value::Map( + labels + .into_iter() + .map(|(k, v)| (Value::Utf8(k), Value::Utf8(v))) + .collect::>() + .into(), + )); + } else { + workspace.grow(std::mem::size_of::>() + std::mem::size_of::())?; + } + row.push(Value::Float64(result)); + rows.push(row); + Ok(()) + }; + if left_scalar || right_scalar { + let vectors = if left_scalar { &right } else { &left }; + for row in vectors { + work.checkpoint().await?; + let labels = if left_scalar && right_scalar { + Labels::new() + } else { + labels(row)? + }; + emit( + labels, + if left_scalar { + value(&left[0])? + } else { + value(row)? + }, + if right_scalar { + value(&right[0])? + } else { + value(row)? + }, + )?; + } + } else { + let mut rhs = BTreeMap::new(); + // Keep matching workspace separate from the output reservation captured by emit. + let mut matching = Workspace::new(&context)?; + for row in &right { + work.checkpoint().await?; + let key = identity(labels(row)?); + matching.grow(label_bytes(&key) + 64)?; + if rhs.insert(key, value(row)?).is_some() { + return Err(invalid("duplicate vector matching labels")); + } + } + let mut seen = std::collections::BTreeSet::new(); + for row in &left { + work.checkpoint().await?; + let labels = labels(row)?; + let key = identity(labels.clone()); + matching.grow(label_bytes(&key) + 64)?; + if !seen.insert(key.clone()) { + return Err(invalid("duplicate vector matching labels")); + } + if let Some(b) = rhs.get(&key) { + emit(labels, value(row)?, *b)?; + } + } + } + Batch::try_new(op.output.clone(), rows) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 7064ce22..942c56d9 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -29,6 +29,7 @@ fn invalid(message: impl Into) -> Error { pub type Source<'a> = Box + 'a>; pub mod promql_rows; +pub mod promql_values; mod candidates; pub use candidates::{ @@ -391,6 +392,12 @@ pub fn compile_node(node: &ExecutableDagNode, inputs: &[Schema]) -> Result Result { + if let Payload::Binary { operator } = &node.payload { + let [left, right] = inputs else { + return Err(invalid("binary requires two inputs")); + }; + return Operator::vector_binary(left.clone(), right.clone(), operator.clone(), false); + } if let Payload::RelationalJoin { join_kind, pred, diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/asap-physical-operators/src/physical_planner/promql_values.rs new file mode 100644 index 00000000..9e0c24b8 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/promql_values.rs @@ -0,0 +1,155 @@ +//! Physical scalar/vector contracts preserve complete label sets across native computation. +use super::*; + +pub fn scalar_schema() -> Schema { + crate::operators::vector_binary::value_schema(true) +} +pub fn vector_schema() -> Schema { + crate::operators::vector_binary::value_schema(false) +} + +/// Compile before deployment chooses readers. Input slots 0 and 1 retain operand order. +pub fn compile_binary( + operator: &planner_types::post_asap::BinaryOperator, + return_bool: bool, + left_scalar: bool, + right_scalar: bool, +) -> Result { + let left = crate::operators::vector_binary::value_schema(left_scalar); + let right = crate::operators::vector_binary::value_schema(right_scalar); + let op = Operator::vector_binary(left.clone(), right.clone(), operator.clone(), return_bool)?; + CompiledPhysicalDag::from_operators( + BTreeMap::from([ + (0, InputContract::bounded(left)), + (1, InputContract::bounded(right)), + ]), + BTreeMap::from([(2, (vec![0, 1], op))]), + vec![2], + ) +} + +fn unary(operators: Vec, input: Schema) -> Result { + let root = operators.len() as u64; + CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(input))]), + operators + .into_iter() + .enumerate() + .map(|(i, op)| ((i + 1) as u64, (vec![i as u64], op))) + .collect(), + vec![root], + ) +} + +fn grouped(grouping: &GroupKeys) -> Result { + let labels = grouping + .keys() + .iter() + .map(|key| match key { + ColumnRef::Named(label) => Ok(label.clone()), + _ => Err(invalid("vector grouping requires label names")), + }) + .collect::, _>>()?; + Operator::project( + vector_schema(), + vec![ + ("labels".into(), Expression::Column(0)), + ("value".into(), Expression::Column(1)), + ( + "group".into(), + Expression::LabelSet { + column: 0, + labels, + without: grouping.is_without(), + }, + ), + ], + ) +} + +fn vector_output(input: Schema, labels: usize, value: usize) -> Result { + let value = Expression::ExactFloat64(value); + Operator::project( + input, + vec![ + ("labels".into(), Expression::Column(labels)), + ("value".into(), value), + ], + ) +} + +pub fn compile_aggregate( + intent: &AggIntent, + grouping: &GroupKeys, +) -> Result { + let project = grouped(grouping)?; + let reduction = match intent { + AggIntent::Sum { .. } => Reduction::Sum(1), + AggIntent::Avg { .. } => Reduction::Avg(1), + AggIntent::Count { .. } => Reduction::Count, + AggIntent::Min { .. } => Reduction::Min(1), + AggIntent::Max { .. } => Reduction::Max(1), + _ => return Err(invalid("unsupported vector aggregate")), + }; + let aggregate = + Operator::aggregate(project.schema(), vec![2], vec![("value".into(), reduction)])?; + let output = vector_output(aggregate.schema(), 0, 1)?; + unary(vec![project, aggregate, output], vector_schema()) +} + +pub fn compile_sort( + descending: bool, + grouping: &GroupKeys, +) -> Result { + let project = grouped(grouping)?; + let sort = Operator::sort( + project.schema(), + vec![SortKey { + column: 1, + descending, + nulls_first: false, + }], + vec![2], + )?; + let output = vector_output(sort.schema(), 0, 1)?; + unary(vec![project, sort, output], vector_schema()) +} + +pub fn compile_limit( + n: u64, + offset: u64, + grouping: &GroupKeys, +) -> Result { + let project = grouped(grouping)?; + let limit = Operator::limit(project.schema(), n, offset, vec![2])?; + let output = vector_output(limit.schema(), 0, 1)?; + unary(vec![project, limit, output], vector_schema()) +} + +pub fn compile_negate(scalar: bool) -> Result { + let input = if scalar { + scalar_schema() + } else { + vector_schema() + }; + let mut columns = Vec::new(); + if !scalar { + columns.push(("labels".into(), Expression::Column(0))); + } + columns.push(( + if scalar { + "$promql_scalar".into() + } else { + "value".into() + }, + Expression::Negate(Box::new(Expression::Column(if scalar { 0 } else { 1 }))), + )); + unary(vec![Operator::project(input.clone(), columns)?], input) +} + +pub fn compile_vector_to_scalar() -> Result { + unary( + vec![Operator::vector_to_scalar(vector_schema(), 1)?.with_output_schema(scalar_schema())?], + vector_schema(), + ) +} diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs new file mode 100644 index 00000000..9bc51ac9 --- /dev/null +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -0,0 +1,268 @@ +//! Binary computation must be fully compiled before deployment binds values. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{compile_node, CompiledPhysicalDag, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Schema, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{ + BinaryOperator, ExecutableDagNode, ExecutableOperatorPayload, ExecutionDataState, + PostAsapNodeId, SummaryFamilyType, SummaryField, SummarySchema, + }, + pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +fn schema() -> Schema { + Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "labels".into(), + dtype: SummaryFamilyType::Plain(DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }), + nullable: false, + }, + SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }, + ], + time_index: None, + }) +} +fn row(name: &str, job: &str, value: f64) -> Vec { + vec![ + Value::Map( + vec![ + (Value::Utf8("__name__".into()), Value::Utf8(name.into())), + (Value::Utf8("job".into()), Value::Utf8(job.into())), + ] + .into(), + ), + Value::Float64(value), + ] +} +fn program() -> CompiledPhysicalDag { + let schema = schema(); + let node = ExecutableDagNode { + id: PostAsapNodeId(2), + payload: ExecutableOperatorPayload::Binary { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + vector_match: None, + checked_relative_division: true, + checked_finite_division: false, + }, + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*schema).clone(), + guarantee: None, + }; + let operator = compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(); + let graph = CompiledPhysicalDag::from_operators( + BTreeMap::from([ + (0, InputContract::bounded(schema.clone())), + (1, InputContract::bounded(schema)), + ]), + BTreeMap::from([(2, (vec![0, 1], operator))]), + vec![2], + ) + .unwrap(); + CompiledPhysicalDag::decode(&graph.encode().unwrap()).unwrap() +} +fn evaluate( + left: Vec>, + right: Vec>, +) -> Result>, asap_physical_operators::Error> { + let graph = program(); + let sources = [left, right] + .into_iter() + .enumerate() + .map(|(id, rows)| { + let batch = Batch::try_new(schema(), rows).unwrap(); + ( + id as u64, + Box::new(Operator::source(schema(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect(); + let bound = graph.instantiate(sources)?; + let ctx = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 0, + }, + Limits::default(), + )?; + block_on(async { + let mut stream = bound.execute(&[2], ctx)?.remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok(rows) + }) +} + +#[test] +fn compiled_binary_matches_series_and_preserves_checked_division() { + let rows = evaluate( + vec![row("left", "api", 6.), row("left", "unmatched", 8.)], + vec![row("right", "api", 2.)], + ) + .unwrap(); + assert_eq!( + serde_json::to_value(&rows).unwrap(), + serde_json::to_value(vec![vec![ + Value::Map(vec![(Value::Utf8("job".into()), Value::Utf8("api".into()))].into()), + Value::Float64(3.) + ]]) + .unwrap() + ); + assert!(evaluate(vec![row("a", "api", 1.)], vec![row("b", "api", 0.)]).is_err()); +} + +#[test] +fn duplicate_matching_identity_is_rejected() { + assert!(evaluate( + vec![row("a", "api", 1.)], + vec![row("b", "api", 2.), row("c", "api", 3.)] + ) + .is_err()); +} + +// Scalar broadcasting and comparison filtering keep the vector operand's value. +#[test] +fn scalar_broadcast_and_bool_comparison_are_distinct() { + use asap_physical_operators::physical_planner::promql_values; + use planner_types::pre_asap::CompareOpKind; + for return_bool in [false, true] { + let graph = promql_values::compile_binary( + &BinaryOperator { + kind: BinaryOpKind::Compare(CompareOpKind::Lt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + true, + false, + ) + .unwrap(); + let graph = CompiledPhysicalDag::decode(&graph.encode().unwrap()).unwrap(); + let scalar = promql_values::scalar_schema(); + let vector = promql_values::vector_schema(); + let sources = BTreeMap::from([ + ( + 0, + Box::new( + Operator::source( + scalar.clone(), + vec![Batch::try_new(scalar, vec![vec![Value::Float64(2.)]]).unwrap()], + ) + .unwrap(), + ) as Source<'_>, + ), + ( + 1, + Box::new( + Operator::source( + vector.clone(), + vec![Batch::try_new( + vector, + vec![row("requests", "api", 4.), row("requests", "worker", 1.)], + ) + .unwrap()], + ) + .unwrap(), + ) as Source<'_>, + ), + ]); + let bound = graph.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + let result = block_on(async { + bound + .execute(&[2], context) + .unwrap() + .remove(0) + .next() + .await + .unwrap() + .unwrap() + }); + assert_eq!(result.rows().len(), if return_bool { 2 } else { 1 }); + assert!( + matches!(result.rows()[0].last(), Some(Value::Float64(v)) if *v == if return_bool { 1. } else { 4. }) + ); + let Value::Map(labels) = &result.rows()[0][0] else { + panic!("missing labels") + }; + assert_eq!( + labels + .iter() + .any(|(key, _)| matches!(key, Value::Utf8(s) if s.as_ref() == "__name__")), + !return_bool + ); + } +} + +// Terminal request controls retain their native error classification. +#[test] +fn binary_obeys_memory_and_cancellation() { + for cancel in [false, true] { + let graph = program(); + let sources = (0..2) + .map(|id| { + ( + id, + Box::new( + Operator::source( + schema(), + vec![Batch::try_new(schema(), vec![row("x", "api", 1.)]).unwrap()], + ) + .unwrap(), + ) as Source<'_>, + ) + }) + .collect(); + let bound = graph.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 0, + }, + Limits { + max_bytes: if cancel { 10000 } else { 1 }, + ..Limits::default() + }, + ) + .unwrap(); + if cancel { + context.cancel(); + } + let result = block_on(async { + match bound.execute(&[2], context) { + Err(error) => Err(error), + Ok(mut streams) => streams.remove(0).next().await.unwrap().map(|_| ()), + } + }); + assert!(matches!( + (cancel, result), + (true, Err(asap_physical_operators::Error::Cancelled)) + | (false, Err(asap_physical_operators::Error::MemoryLimit)) + )); + } +} diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs new file mode 100644 index 00000000..1010f0cc --- /dev/null +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -0,0 +1,123 @@ +//! Compile, persist and rebind dynamic-label computation without deployment lowering. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{promql_values::*, CompiledPhysicalDag, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::pre_asap::{AggIntent, ColumnRef, GroupKeys}; +use std::collections::BTreeMap; + +fn row(labels: &[(&str, &str)], value: f64) -> Vec { + vec![ + Value::Map( + labels + .iter() + .map(|(k, v)| (Value::Utf8((*k).into()), Value::Utf8((*v).into()))) + .collect::>() + .into(), + ), + Value::Float64(value), + ] +} +fn run(graph: CompiledPhysicalDag, rows: Vec>) -> Vec> { + let graph = CompiledPhysicalDag::decode(&graph.encode().unwrap()).unwrap(); + let input = Batch::try_new(vector_schema(), rows).unwrap(); + let source = Box::new(Operator::source(vector_schema(), vec![input]).unwrap()) as Source<'_>; + let bound = graph.instantiate(BTreeMap::from([(0, source)])).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = bound.execute(graph.roots(), context).unwrap().remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + rows.extend(batch.unwrap().rows().iter().cloned()); + } + rows + }) +} +fn equal_rows(actual: Vec>, expected: Vec>) { + let mut actual = actual + .into_iter() + .map(|r| serde_json::to_string(&r).unwrap()) + .collect::>(); + let mut expected = expected + .into_iter() + .map(|r| serde_json::to_string(&r).unwrap()) + .collect::>(); + actual.sort(); + expected.sort(); + assert_eq!(actual, expected); +} + +#[test] +fn grouping_preserves_unenumerated_labels_and_empty_label_semantics() { + let rows = vec![ + row(&[("__name__", "m"), ("instance", "a"), ("job", "api")], 1.), + row(&[("__name__", "m"), ("instance", "b"), ("job", "api")], 2.), + row(&[("instance", "c"), ("job", "")], 4.), + row(&[("instance", "d")], 8.), + ]; + equal_rows( + run( + compile_aggregate( + &AggIntent::Sum { col: None }, + &GroupKeys::without(vec![ColumnRef::Named("instance".into())]), + ) + .unwrap(), + rows.clone(), + ), + vec![row(&[("job", "api")], 3.), row(&[], 12.)], + ); + equal_rows( + run( + compile_aggregate( + &AggIntent::Count { + accuracy: planner_types::types::AccuracyTarget::Exact, + }, + &GroupKeys::by(vec![ColumnRef::Named("job".into())]), + ) + .unwrap(), + rows, + ), + vec![row(&[("job", "api")], 2.), row(&[], 2.)], + ); +} + +#[test] +fn ranking_and_grouped_limit_preserve_full_selected_series() { + let grouping = GroupKeys::by(vec![ColumnRef::Named("job".into())]); + let rows = vec![ + row(&[("instance", "a"), ("job", "api")], 1.), + row(&[("instance", "b"), ("job", "api")], 3.), + row(&[("instance", "c"), ("job", "worker")], 2.), + ]; + let sorted = run(compile_sort(true, &grouping).unwrap(), rows); + let selected = run(compile_limit(1, 0, &grouping).unwrap(), sorted); + equal_rows( + selected, + vec![ + row(&[("instance", "b"), ("job", "api")], 3.), + row(&[("instance", "c"), ("job", "worker")], 2.), + ], + ); + assert!(run(compile_limit(0, 0, &grouping).unwrap(), vec![row(&[], 1.)]).is_empty()); +} + +#[test] +fn empty_vector_aggregation_stays_empty() { + assert!(run( + compile_aggregate(&AggIntent::Sum { col: None }, &GroupKeys::default()).unwrap(), + vec![] + ) + .is_empty()); + let scalar = run(compile_vector_to_scalar().unwrap(), vec![]); + assert!(matches!(scalar[0][0],Value::Float64(v) if v.is_nan())); +} From 3a99b6ff855cd681f992263d067ceb89db2338b7 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 13:33:18 +0000 Subject: [PATCH 81/90] feat: retain window operators and compose shared physical fragments --- .../src/operators/aggregate/mod.rs | 2 +- .../src/operators/aggregate/temporal.rs | 2 +- .../src/operators/mod.rs | 21 +- .../src/operators/persisted.rs | 3 + .../src/operators/source.rs | 20 +- .../src/operators/vector_window.rs | 147 +++++++++++++ .../src/physical_planner/compiled.rs | 73 ++++++ .../src/physical_planner/promql_values.rs | 53 +++++ .../tests/promql_values.rs | 207 +++++++++++++++++- 9 files changed, 513 insertions(+), 15 deletions(-) create mode 100644 crates/asap-physical-operators/src/operators/vector_window.rs diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs index c57b4fce..71e7134c 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/mod.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -159,7 +159,7 @@ pub(super) fn execute<'a>( .boxed_local()) } -mod temporal; +pub(super) mod temporal; async fn reduce( rows: Vec>, groups: &[usize], diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs index fb74965a..83751883 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -13,7 +13,7 @@ use crate::{ use planner_types::pre_asap::{AggIntent, ColumnRef}; use std::collections::BTreeMap; -pub(super) async fn reduce( +pub(in crate::operators) async fn reduce( rows: Vec>, intent: &AggIntent, groups: &[usize], diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index 662414a1..752febe6 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -27,12 +27,17 @@ mod sort; mod source; mod summary; pub(crate) mod vector_binary; +pub(crate) mod vector_window; pub use aggregate::Reduction; pub use sort::SortKey; #[derive(Clone, serde::Serialize, serde::Deserialize)] enum Kind { #[serde(skip)] Source(Vec), + Constant { + value: Value, + dtype: DataType, + }, PaneInput { coordinate: usize, layout: planner_types::post_asap::PaneLayout, @@ -55,6 +60,10 @@ enum Kind { operator: planner_types::post_asap::BinaryOperator, return_bool: bool, }, + RangeWindow { + intent: Box>, + }, + HistogramQuantile, Project(Vec), Filter(Expression), Limit { @@ -215,6 +224,8 @@ impl PhysicalOperator for Operator { self.kind, Kind::Sort { .. } | Kind::VectorBinary { .. } + | Kind::RangeWindow { .. } + | Kind::HistogramQuantile | Kind::CurrentSeries { .. } | Kind::Aggregate { .. } | Kind::Window { .. } @@ -228,7 +239,7 @@ impl PhysicalOperator for Operator { } fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { let boundedness = match &self.kind { - Kind::Source(_) => Boundedness::Bounded, + Kind::Source(_) | Kind::Constant { .. } => Boundedness::Bounded, Kind::Limit { groups, .. } if groups.is_empty() => Boundedness::Bounded, _ => Boundedness::from_inputs(inputs), }; @@ -252,12 +263,15 @@ impl PhysicalOperator for Operator { fn name(&self) -> &str { match self.kind { Kind::Source(_) => "Source", + Kind::Constant { .. } => "Constant", Kind::PaneInput { .. } => "PaneInput", Kind::ScopeTimestamp { .. } => "ScopeTimestamp", Kind::Union => "Union", Kind::CurrentSeries { .. } => "CurrentSeries", Kind::VectorToScalar { .. } => "VectorToScalar", Kind::VectorBinary { .. } => "VectorBinary", + Kind::RangeWindow { .. } => "RangeWindow", + Kind::HistogramQuantile => "HistogramQuantile", Kind::Project(_) => "Project", Kind::Filter(_) => "Filter", Kind::Limit { .. } => "Limit", @@ -292,10 +306,13 @@ impl PhysicalOperator for Operator { context: RunContext, ) -> Result, Error> { match self.kind { - Kind::Source(_) | Kind::Union | Kind::VectorToScalar { .. } => { + Kind::Source(_) | Kind::Constant { .. } | Kind::Union | Kind::VectorToScalar { .. } => { source::execute(self, inputs, context) } Kind::VectorBinary { .. } => vector_binary::execute(self, inputs, context), + Kind::RangeWindow { .. } | Kind::HistogramQuantile => { + vector_window::execute(self, inputs, context) + } Kind::Project(_) => projection::execute(self, inputs, context), Kind::CurrentSeries { .. } => current_series::execute(self, inputs, context), Kind::PaneInput { .. } | Kind::ScopeTimestamp { .. } => { diff --git a/crates/asap-physical-operators/src/operators/persisted.rs b/crates/asap-physical-operators/src/operators/persisted.rs index d6c53898..c98dafad 100644 --- a/crates/asap-physical-operators/src/operators/persisted.rs +++ b/crates/asap-physical-operators/src/operators/persisted.rs @@ -29,6 +29,7 @@ impl TryFrom for Operator { }; let op = match kind { Kind::Source(_) => return Err(invalid("physical plans cannot persist live sources")), + Kind::Constant { value, dtype } => Operator::scalar(value, dtype)?, Kind::PaneInput { coordinate, layout, @@ -47,6 +48,8 @@ impl TryFrom for Operator { operator, return_bool, } => Operator::vector_binary(input(0)?, input(1)?, operator, return_bool)?, + Kind::RangeWindow { intent } => Operator::range_window(*intent)?, + Kind::HistogramQuantile => Operator::histogram_quantile(), Kind::Project(expressions) => { if expressions.len() != output.fields.len() { return Err(invalid("persisted projection width mismatch")); diff --git a/crates/asap-physical-operators/src/operators/source.rs b/crates/asap-physical-operators/src/operators/source.rs index 060a5381..42701e17 100644 --- a/crates/asap-physical-operators/src/operators/source.rs +++ b/crates/asap-physical-operators/src/operators/source.rs @@ -12,15 +12,17 @@ impl Operator { }) } pub fn scalar(value: Value, dtype: DataType) -> Result { - let schema = schema(vec![result_field( + let output = schema(vec![result_field( "value", - dtype, + dtype.clone(), matches!(value, Value::Null), )]); - Self::source( - schema.clone(), - vec![Batch::try_new(schema, vec![vec![value]])?], - ) + Batch::try_new(output.clone(), vec![vec![value.clone()]])?; + Ok(Self { + kind: Kind::Constant { value, dtype }, + inputs: vec![], + output, + }) } pub fn vector_to_scalar(input: Schema, column: usize) -> Result { if plain(&input, column)? != (&DataType::Float64, false) { @@ -52,6 +54,12 @@ pub(super) fn execute<'a>( if let Kind::Source(batches) = &operator.kind { return Ok(futures::stream::iter(batches.iter().cloned().map(Ok)).boxed_local()); } + if let Kind::Constant { value, .. } = &operator.kind { + return Ok(futures::stream::once(async move { + Batch::try_new(output, vec![vec![value.clone()]]) + }) + .boxed_local()); + } if matches!(operator.kind, Kind::Union) { return Ok(futures::stream::select_all(inputs) .map(|batch| batch.map(|batch| batch.value().clone())) diff --git a/crates/asap-physical-operators/src/operators/vector_window.rs b/crates/asap-physical-operators/src/operators/vector_window.rs new file mode 100644 index 00000000..407f04ea --- /dev/null +++ b/crates/asap-physical-operators/src/operators/vector_window.rs @@ -0,0 +1,147 @@ +//! Window bounds are typed input data; aggregation and histogram semantics stay native. +use super::*; +use planner_types::pre_asap::AggIntent; + +pub(crate) fn matrix_schema() -> Schema { + let mut fields = vector_binary::value_schema(false).fields.clone(); + fields.insert(1, result_field("timestamp", DataType::Timestamp, false)); + fields.push(result_field("window_start", DataType::Timestamp, false)); + fields.push(result_field("window_end", DataType::Timestamp, false)); + Arc::new(SummarySchema { + fields, + time_index: Some(1), + }) +} + +impl Operator { + pub fn range_window(intent: AggIntent) -> Result { + // Reuse the window constructor's semantic admission, without fixing request time. + Self::window(matrix_schema(), intent.clone(), 1, 2, vec![0], Some((0, 1)))?; + Ok(Self { + kind: Kind::RangeWindow { + intent: Box::new(intent), + }, + inputs: vec![matrix_schema()], + output: vector_binary::value_schema(false), + }) + } + pub fn histogram_quantile() -> Self { + Self { + kind: Kind::HistogramQuantile, + inputs: vec![ + vector_binary::value_schema(true), + vector_binary::value_schema(false), + ], + output: vector_binary::value_schema(false), + } + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + match &operator.kind { + Kind::RangeWindow { intent } => { + let input = inputs + .pop() + .ok_or_else(|| invalid("missing matrix input"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let mut window = None; + let mut work = Cooperative::new(&context); + for row in &rows { + work.checkpoint().await?; + let (Value::Timestamp(start), Value::Timestamp(end)) = (&row[3], &row[4]) + else { + return Err(invalid("missing matrix window bounds")); + }; + if start >= end || window.is_some_and(|bounds| bounds != (*start, *end)) { + return Err(invalid("matrix rows must share one nonempty window")); + } + window = Some((*start, *end)); + } + let mut rows = + aggregate::temporal::reduce(rows, intent, &[0], 1, 2, window, &context).await?; + for row in &mut rows { + work.checkpoint().await?; + row[1] = Expression::ExactFloat64(1).evaluate(row)?; + } + Batch::try_new(operator.output.clone(), rows) + }) + .boxed_local()) + } + Kind::HistogramQuantile => { + let buckets = inputs + .pop() + .ok_or_else(|| invalid("missing histogram buckets"))?; + let quantile = inputs.pop().ok_or_else(|| invalid("missing quantile"))?; + Ok(futures::stream::once(async move { + let ((quantile, _q_memory), (buckets, _bucket_memory)) = futures::try_join!( + collect_rows(quantile, &context), + collect_rows(buckets, &context) + )?; + let [row] = quantile.as_slice() else { + return Err(invalid("histogram quantile requires one scalar")); + }; + let [Value::Float64(q)] = row.as_slice() else { + return Err(invalid("invalid quantile scalar")); + }; + let mut rows = Vec::new(); + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + for row in buckets { + work.checkpoint().await?; + let Value::Map(entries) = &row[0] else { + return Err(invalid("histogram buckets require labels")); + }; + let mut bound = None; + let mut labels = BTreeMap::new(); + let mut seen = std::collections::BTreeSet::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("histogram labels must be Utf8")); + }; + if !seen.insert(key) { + return Err(invalid("duplicate histogram label")); + } + if key.as_ref() == "le" { + bound = value.parse::().ok(); + } else if key.as_ref() != "__name__" && !value.is_empty() { + labels.insert(key.clone(), value.clone()); + } + } + if let Some(bound) = bound { + let projected = vec![ + Value::Map( + labels + .into_iter() + .map(|(k, v)| (Value::Utf8(k), Value::Utf8(v))) + .collect::>() + .into(), + ), + Value::Float64(bound), + row[1].clone(), + ]; + workspace.grow(row_bytes(&projected))?; + rows.push(projected); + } + } + let result = aggregate::temporal::reduce( + rows, + &AggIntent::HistogramQuantile { q: *q }, + &[0], + 1, + 2, + None, + &context, + ) + .await?; + Batch::try_new(operator.output.clone(), result) + }) + .boxed_local()) + } + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs index da97def5..f450742f 100644 --- a/crates/asap-physical-operators/src/physical_planner/compiled.rs +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -48,6 +48,79 @@ struct StoredDag { } impl CompiledPhysicalDag { + /// Link already-selected physical fragments without lowering operators again. + /// Fragment keys and source keys share a namespace; repeated dependency IDs + /// therefore remain one producer in the composed graph. + pub fn compose( + sources: BTreeMap, + fragments: BTreeMap, Self)>, + roots: Vec, + ) -> Result { + if sources.keys().any(|id| fragments.contains_key(id)) { + return Err(invalid("physical source and fragment IDs overlap")); + } + let mut contracts = sources.clone(); + for (&id, (_, fragment)) in &fragments { + fragment.validate()?; + let [root] = fragment.roots() else { + return Err(invalid("composed fragment requires one root")); + }; + if fragment.input_contracts().any(|(id, _)| id == *root) { + return Err(invalid("fragment root must be a computed output")); + } + contracts.insert(id, fragment.output_contract(*root)?); + } + let mut next = contracts + .keys() + .next_back() + .copied() + .unwrap_or(0) + .checked_add(1) + .ok_or_else(|| invalid("physical node ID overflow"))?; + let mut result = Self::new(roots); + for (id, contract) in sources { + result.add_input(id, contract)?; + } + for (id, (inputs, fragment)) in fragments { + if inputs.len() != fragment.input_contracts().count() { + return Err(invalid("physical fragment input arity mismatch")); + } + let mut mapping = BTreeMap::new(); + for ((local, expected), global) in fragment.input_contracts().zip(inputs) { + let actual = contracts + .get(&global) + .ok_or_else(|| invalid("missing physical fragment dependency"))?; + if expected.schema != actual.schema + || (expected.properties.boundedness == Boundedness::Bounded + && actual.properties.boundedness != Boundedness::Bounded) + { + return Err(invalid("physical fragment dependency contract mismatch")); + } + mapping.insert(local, global); + } + mapping.insert(fragment.roots[0], id); + for local in fragment.nodes.keys() { + if !mapping.contains_key(local) { + mapping.insert(*local, next); + next = next + .checked_add(1) + .ok_or_else(|| invalid("physical node ID overflow"))?; + } + } + for (local, node) in fragment.nodes { + if let Node::Operator { inputs, operator } = node { + result.add( + mapping[&local], + inputs.into_iter().map(|input| mapping[&input]).collect(), + operator, + )?; + } + } + } + result.validate()?; + Ok(result) + } + /// Persist selected physical operators and input slots, never live readers /// or mutable summary state. Recovery does not run logical plan lowering. pub fn encode(&self) -> Result, Error> { diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/asap-physical-operators/src/physical_planner/promql_values.rs index 9e0c24b8..7957cb40 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_values.rs @@ -8,6 +8,59 @@ pub fn vector_schema() -> Schema { crate::operators::vector_binary::value_schema(false) } +pub fn matrix_schema() -> Schema { + crate::operators::vector_window::matrix_schema() +} + +pub fn compile_scalar(value: f64) -> Result { + let operator = Operator::scalar( + crate::values::Value::Float64(value), + planner_types::pre_asap::DataType::Float64, + )? + .with_output_schema(scalar_schema())?; + CompiledPhysicalDag::from_operators( + BTreeMap::new(), + BTreeMap::from([(0, (vec![], operator))]), + vec![0], + ) +} + +pub fn compile_temporal( + intent: &AggIntent, + preserve_metric_name: bool, +) -> Result { + let operator = Operator::range_window(intent.clone())?; + let mut operators = vec![operator]; + if !preserve_metric_name { + operators.push(Operator::project( + vector_schema(), + vec![ + ( + "labels".into(), + Expression::LabelSet { + column: 0, + labels: vec![], + without: true, + }, + ), + ("value".into(), Expression::Column(1)), + ], + )?); + } + unary(operators, matrix_schema()) +} + +pub fn compile_histogram_quantile() -> Result { + CompiledPhysicalDag::from_operators( + BTreeMap::from([ + (0, InputContract::bounded(scalar_schema())), + (1, InputContract::bounded(vector_schema())), + ]), + BTreeMap::from([(2, (vec![0, 1], Operator::histogram_quantile()))]), + vec![2], + ) +} + /// Compile before deployment chooses readers. Input slots 0 and 1 retain operand order. pub fn compile_binary( operator: &planner_types::post_asap::BinaryOperator, diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs index 1010f0cc..507f1953 100644 --- a/crates/asap-physical-operators/tests/promql_values.rs +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -22,10 +22,25 @@ fn row(labels: &[(&str, &str)], value: f64) -> Vec { ] } fn run(graph: CompiledPhysicalDag, rows: Vec>) -> Vec> { + run_inputs(graph, vec![Batch::try_new(vector_schema(), rows).unwrap()]).unwrap() +} +fn run_inputs( + graph: CompiledPhysicalDag, + batches: Vec, +) -> Result>, asap_physical_operators::Error> { let graph = CompiledPhysicalDag::decode(&graph.encode().unwrap()).unwrap(); - let input = Batch::try_new(vector_schema(), rows).unwrap(); - let source = Box::new(Operator::source(vector_schema(), vec![input]).unwrap()) as Source<'_>; - let bound = graph.instantiate(BTreeMap::from([(0, source)])).unwrap(); + let sources = batches + .into_iter() + .enumerate() + .map(|(id, batch)| { + ( + id as u64, + Box::new(Operator::source(batch.schema().clone(), vec![batch]).unwrap()) + as Source<'_>, + ) + }) + .collect::>(); + let bound = graph.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Query { evaluation_time_ms: 0, @@ -38,9 +53,9 @@ fn run(graph: CompiledPhysicalDag, rows: Vec>) -> Vec> { let mut stream = bound.execute(graph.roots(), context).unwrap().remove(0); let mut rows = Vec::new(); while let Some(batch) = stream.next().await { - rows.extend(batch.unwrap().rows().iter().cloned()); + rows.extend(batch?.rows().iter().cloned()); } - rows + Ok(rows) }) } fn equal_rows(actual: Vec>, expected: Vec>) { @@ -121,3 +136,185 @@ fn empty_vector_aggregation_stays_empty() { let scalar = run(compile_vector_to_scalar().unwrap(), vec![]); assert!(matches!(scalar[0][0],Value::Float64(v) if v.is_nan())); } + +// One persisted temporal graph accepts different request windows and detects resets. +#[test] +fn temporal_graph_uses_bound_window_without_recompilation() { + let graph = compile_temporal(&AggIntent::Rate, false).unwrap(); + for start in [0, 60_000] { + let labels = row(&[("__name__", "counter"), ("job", "api")], 0.)[0].clone(); + let samples = [(0, 5.), (30_000, 1.), (60_000, 7.)]; + let rows = samples + .into_iter() + .map(|(time, value)| { + vec![ + labels.clone(), + Value::Timestamp(start + time), + Value::Float64(value), + Value::Timestamp(start), + Value::Timestamp(start + 60_000), + ] + }) + .collect(); + let output = run_inputs( + graph.clone(), + vec![Batch::try_new(matrix_schema(), rows).unwrap()], + ) + .unwrap(); + equal_rows(output, vec![row(&[("job", "api")], 7. / 60.)]); + } + let labels = row(&[("job", "api")], 0.)[0].clone(); + let rows = vec![ + vec![ + labels.clone(), + Value::Timestamp(0), + Value::Float64(1.), + Value::Timestamp(0), + Value::Timestamp(1000), + ], + vec![ + labels, + Value::Timestamp(1000), + Value::Float64(2.), + Value::Timestamp(0), + Value::Timestamp(2000), + ], + ]; + assert!(run_inputs(graph, vec![Batch::try_new(matrix_schema(), rows).unwrap()]).is_err()); +} + +// The quantile is an ordinary scalar input, and bucket labels are native computation. +#[test] +fn histogram_quantile_keeps_each_label_group() { + let graph = compile_histogram_quantile().unwrap(); + let buckets = vec![ + row(&[("job", "api"), ("le", "1")], 2.), + row(&[("job", "api"), ("le", "2")], 4.), + row(&[("job", "api"), ("le", "+Inf")], 4.), + ]; + let output = run_inputs( + graph, + vec![ + Batch::try_new(scalar_schema(), vec![vec![Value::Float64(0.75)]]).unwrap(), + Batch::try_new(vector_schema(), buckets).unwrap(), + ], + ) + .unwrap(); + equal_rows(output, vec![row(&[("job", "api")], 1.5)]); +} + +// Linking an ensemble preserves its shared producer and every selected operator. +#[test] +fn composed_ensemble_shares_a_producer_across_roots() { + use asap_physical_operators::{ + physical_planner::InputContract, + plan::{PhysicalOperator, PlanProperties}, + runtime::{Input, OutputStream}, + values::Schema, + }; + use planner_types::{ + post_asap::BinaryOperator, + pre_asap::{ArithmeticOpKind, BinaryOpKind}, + }; + struct Counted { + source: Operator, + starts: std::rc::Rc>, + } + impl PhysicalOperator for Counted { + fn name(&self) -> &str { + "CountedInput" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.source.schema() + } + fn output_bytes(&self, batch: &Batch) -> usize { + batch.bytes() + } + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + self.source.properties(inputs) + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, asap_physical_operators::Error> { + self.starts.set(self.starts.get() + 1); + self.source.start(inputs, context) + } + } + let aggregate = compile_aggregate( + &AggIntent::Sum { col: None }, + &GroupKeys::by(vec![ColumnRef::Named("job".into())]), + ) + .unwrap(); + let binary = compile_binary( + &BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_finite_division: false, + checked_relative_division: false, + }, + false, + false, + false, + ) + .unwrap(); + let graph = CompiledPhysicalDag::compose( + BTreeMap::from([(0, InputContract::bounded(vector_schema()))]), + BTreeMap::from([ + (10, (vec![0], aggregate)), + (20, (vec![10, 10], binary)), + (30, (vec![10], compile_negate(false).unwrap())), + ]), + vec![20, 30], + ) + .unwrap(); + let graph = CompiledPhysicalDag::decode(&graph.encode().unwrap()).unwrap(); + assert_eq!(graph.input_contracts().count(), 1); + let starts = std::rc::Rc::new(std::cell::Cell::new(0)); + for _ in 0..2 { + let input = Batch::try_new(vector_schema(), vec![row(&[("job", "api")], 3.)]).unwrap(); + let source = Counted { + source: Operator::source(vector_schema(), vec![input]).unwrap(), + starts: starts.clone(), + }; + let bound = graph + .instantiate(BTreeMap::from([(0, Box::new(source) as Source<'_>)])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_buffered_batches: 1, + ..Limits::default() + }, + ) + .unwrap(); + let results = + block_on(futures::future::join_all( + bound + .execute(graph.roots(), context) + .unwrap() + .into_iter() + .map(|mut stream| async move { + stream.next().await.unwrap().unwrap().rows().to_vec() + }), + )); + equal_rows(results[0].clone(), vec![row(&[("job", "api")], 6.)]); + equal_rows(results[1].clone(), vec![row(&[("job", "api")], -3.)]); + } + assert_eq!(starts.get(), 2); +} + +#[test] +fn compiled_constant_needs_no_deployment_source() { + let graph = compile_scalar(3.).unwrap(); + assert_eq!(graph.input_contracts().count(), 0); + let result = run_inputs(graph, vec![]).unwrap(); + assert!(matches!(result[0][0], Value::Float64(3.))); +} From ed7f82c8308200b63f66746874e38d54106ff553 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 13:52:13 +0000 Subject: [PATCH 82/90] fix: enforce certified pruning coverage in native semi-joins --- .../src/operators/joins/mod.rs | 45 ++++++- .../src/operators/mod.rs | 1 + .../src/operators/persisted.rs | 12 +- .../src/physical_planner/compiled.rs | 11 +- .../src/physical_planner/mod.rs | 12 +- .../tests/physical_dag.rs | 111 ++++++++++++++++++ 6 files changed, 186 insertions(+), 6 deletions(-) diff --git a/crates/asap-physical-operators/src/operators/joins/mod.rs b/crates/asap-physical-operators/src/operators/joins/mod.rs index 1fece7c1..a4397033 100644 --- a/crates/asap-physical-operators/src/operators/joins/mod.rs +++ b/crates/asap-physical-operators/src/operators/joins/mod.rs @@ -14,11 +14,33 @@ impl Operator { } } Ok(Self { - kind: Kind::SemiJoin { keys }, + kind: Kind::SemiJoin { + keys, + require_complete_right: false, + }, inputs: vec![left.clone(), right], output: left, }) } + pub(crate) fn require_complete_right(mut self) -> Self { + if let Kind::SemiJoin { + require_complete_right, + .. + } = &mut self.kind + { + *require_complete_right = true; + } + self + } + pub(crate) fn certified_pruning_keys(&self) -> Option<&[(usize, usize)]> { + match &self.kind { + Kind::SemiJoin { + keys, + require_complete_right: true, + } => Some(keys), + _ => None, + } + } pub fn relational_join( left: Schema, right: Schema, @@ -138,7 +160,11 @@ pub(super) fn execute<'a>( }) .boxed_local()); } - if let Kind::SemiJoin { keys } = &operator.kind { + if let Kind::SemiJoin { + keys, + require_complete_right, + } = &operator.kind + { let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; return Ok(futures::stream::once(async move { @@ -158,18 +184,33 @@ pub(super) fn execute<'a>( workspace.grow(key_bytes(&key))?; members.insert(key); } + } else if *require_complete_right { + return Err(invalid( + "certified pruning candidate has an unmatchable key", + )); } } let mut rows = Vec::new(); + let mut covered = std::collections::BTreeSet::new(); for row in left { work.checkpoint().await?; if left_cols.iter().all(|&i| matchable_key(&row[i])) && members.contains(&group_key(&row, &left_cols)?) { + if *require_complete_right { + let key = group_key(&row, &left_cols)?; + if !covered.contains(&key) { + workspace.grow(key_bytes(&key))?; + covered.insert(key); + } + } workspace.grow(std::mem::size_of::>())?; rows.push(row); } } + if *require_complete_right && members != covered { + return Err(invalid("certified pruning key has no authoritative value")); + } Batch::try_new(output, rows) }) .boxed_local()); diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index 752febe6..6c5e435b 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -88,6 +88,7 @@ enum Kind { }, SemiJoin { keys: Vec<(usize, usize)>, + require_complete_right: bool, }, Join { kind: planner_types::pre_asap::JoinKind, diff --git a/crates/asap-physical-operators/src/operators/persisted.rs b/crates/asap-physical-operators/src/operators/persisted.rs index c98dafad..ff945fa7 100644 --- a/crates/asap-physical-operators/src/operators/persisted.rs +++ b/crates/asap-physical-operators/src/operators/persisted.rs @@ -81,7 +81,17 @@ impl TryFrom for Operator { let names = output.fields[groups.len()..].iter().map(|f| f.name.clone()); Operator::aggregate(input(0)?, groups, names.zip(measures).collect())? } - Kind::SemiJoin { keys } => Operator::semi_join(input(0)?, input(1)?, keys)?, + Kind::SemiJoin { + keys, + require_complete_right, + } => { + let operator = Operator::semi_join(input(0)?, input(1)?, keys)?; + if require_complete_right { + operator.require_complete_right() + } else { + operator + } + } Kind::Join { kind, predicate } => Operator::relational_join( input(0)?, input(1)?, diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs index f450742f..ffaf29b8 100644 --- a/crates/asap-physical-operators/src/physical_planner/compiled.rs +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -126,7 +126,7 @@ impl CompiledPhysicalDag { pub fn encode(&self) -> Result, Error> { self.validate()?; let bytes = serde_json::to_vec(&StoredDag { - version: 1, + version: 2, nodes: self.nodes.clone(), roots: self.roots.clone(), }) @@ -139,7 +139,7 @@ impl CompiledPhysicalDag { pub fn decode(bytes: &[u8]) -> Result { let stored: StoredDag = serde_json::from_slice(bytes).map_err(|error| invalid(error.to_string()))?; - if stored.version != 1 { + if stored.version != 2 { return Err(invalid("unsupported physical plan format")); } let result = Self { @@ -203,6 +203,13 @@ impl CompiledPhysicalDag { } /// Selected operator name, for plan inspection without decoding its wire format. + /// Certified candidate pruning checks authoritative-key coverage inside this operator. + pub fn certified_pruning_keys(&self, id: NodeId) -> Option<&[(usize, usize)]> { + match self.nodes.get(&id)? { + Node::Operator { operator, .. } => operator.certified_pruning_keys(), + Node::Input(_) => None, + } + } pub fn operator_name(&self, id: NodeId) -> Option<&str> { match self.nodes.get(&id)? { Node::Input(_) => Some("Input"), diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 942c56d9..dec3ad10 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -417,9 +417,19 @@ fn bind_operation(node: &ExecutableDagNode, inputs: &[Schema]) -> Result, + ) + }) + .collect::>(); + let bound = graph.instantiate(sources).unwrap(); + let result = block_on(async { + let mut stream = bound + .execute( + graph.roots(), + RunContext::new(query(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok::<_, asap_physical_operators::Error>(rows) + }); + if certified && !complete { + assert!(result + .unwrap_err() + .to_string() + .contains("no authoritative value")); + } else { + let rows = result.unwrap(); + assert_eq!(rows.len(), 1); + assert!(matches!(&rows[0][0], Value::Utf8(key) if key.as_ref() == "a")); + } + } + } +} From 23cf9eb0ddbeaf8a73b2060aea41fe217e59add2 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 13:59:34 +0000 Subject: [PATCH 83/90] feat: compile aligned ingestion arithmetic in physical plans --- .../src/operators/aligned_binary.rs | 144 ++++++++++++++++++ .../src/operators/mod.rs | 9 ++ .../src/operators/persisted.rs | 5 + .../src/physical_planner/mod.rs | 40 +++++ .../tests/physical_dag.rs | 98 ++++++++++++ 5 files changed, 296 insertions(+) create mode 100644 crates/asap-physical-operators/src/operators/aligned_binary.rs diff --git a/crates/asap-physical-operators/src/operators/aligned_binary.rs b/crates/asap-physical-operators/src/operators/aligned_binary.rs new file mode 100644 index 00000000..941a9f61 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/aligned_binary.rs @@ -0,0 +1,144 @@ +//! Arithmetic on complete, aligned population/window rows used by precomputation. +use super::*; +use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; +use std::collections::BTreeSet; + +impl Operator { + /// Match every row by the declared identity columns. Unlike an inner join, + /// incomplete or duplicate keys are errors: dropping an update changes state. + pub fn aligned_binary( + left: Schema, + right: Schema, + keys: Vec<(usize, usize)>, + values: (usize, usize), + operator: BinaryOperator, + ) -> Result { + if keys.is_empty() + || !matches!(operator.kind, BinaryOpKind::Arithmetic(_)) + || operator.vector_match.is_some() + { + return Err(invalid( + "aligned arithmetic requires explicit keys and arithmetic semantics", + )); + } + for (input, value) in [(&left, values.0), (&right, values.1)] { + if input.fields.get(value).is_none_or(|f| { + f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Float64) + }) { + return Err(invalid( + "aligned arithmetic requires non-null Float64 values", + )); + } + } + let mut left_keys = BTreeSet::new(); + let mut right_keys = BTreeSet::new(); + for &(l, r) in &keys { + if l == values.0 + || r == values.1 + || !left_keys.insert(l) + || !right_keys.insert(r) + || left + .fields + .get(l) + .zip(right.fields.get(r)) + .is_none_or(|(l, r)| l.nullable || r.nullable || l.dtype != r.dtype) + { + return Err(invalid("invalid aligned arithmetic keys")); + } + } + if left_keys.len() + 1 != left.fields.len() || right_keys.len() + 1 != right.fields.len() { + return Err(invalid( + "aligned arithmetic must account for every input column", + )); + } + Ok(Self { + output: left.clone(), + inputs: vec![left, right], + kind: Kind::AlignedBinary { + keys, + values, + operator, + }, + }) + } +} + +pub(super) fn execute<'a>( + op: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::AlignedBinary { + keys, + values, + operator, + } = &op.kind + else { + unreachable!() + }; + let right = inputs + .pop() + .ok_or_else(|| invalid("missing aligned right input"))?; + let left = inputs + .pop() + .ok_or_else(|| invalid("missing aligned left input"))?; + Ok(futures::stream::once(async move { + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + if left.is_empty() || left.len() != right.len() { + return Err(invalid( + "aligned arithmetic requires matching nonempty key sets", + )); + } + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + let columns = |side: bool| { + keys.iter() + .map(|&(l, r)| if side { r } else { l }) + .collect::>() + }; + let left_columns = columns(false); + let right_columns = columns(true); + let mut indexed = BTreeMap::new(); + for row in right { + work.checkpoint().await?; + let key = group_key(&row, &right_columns)?; + let Value::Float64(value) = row[values.1] else { + return Err(invalid("invalid aligned value")); + }; + if !value.is_finite() { + return Err(invalid("aligned arithmetic input is non-finite")); + } + workspace.grow(key_bytes(&key) + 64)?; + if indexed.insert(key, value).is_some() { + return Err(invalid("aligned arithmetic input has duplicate keys")); + } + } + let mut rows = Vec::new(); + for mut row in left { + work.checkpoint().await?; + let key = group_key(&row, &left_columns)?; + let right = indexed + .remove(&key) + .ok_or_else(|| invalid("aligned arithmetic input has missing or duplicate keys"))?; + let Value::Float64(left) = row[values.0] else { + return Err(invalid("invalid aligned value")); + }; + if !left.is_finite() { + return Err(invalid("aligned arithmetic input is non-finite")); + } + let result = crate::expressions::arithmetic::evaluate_binary(operator, left, right)?; + if !matches!(result, Value::Float64(value) if value.is_finite()) { + return Err(invalid("aligned arithmetic produced a non-finite update")); + } + row[values.0] = result; + workspace.grow(std::mem::size_of::>())?; + rows.push(row); + } + if !indexed.is_empty() { + return Err(invalid("aligned arithmetic has unmatched input keys")); + } + Batch::try_new(op.output.clone(), rows) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index 6c5e435b..5cdcd530 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -1,4 +1,5 @@ //! Native physical operators. Each module owns its constructors and execution. +mod aligned_binary; use crate::plan::{Boundedness, Emission, PhysicalOperator, PlanProperties}; use crate::{ runtime::{Cooperative, Input, OutputStream, Reservation, RunContext}, @@ -60,6 +61,11 @@ enum Kind { operator: planner_types::post_asap::BinaryOperator, return_bool: bool, }, + AlignedBinary { + keys: Vec<(usize, usize)>, + values: (usize, usize), + operator: planner_types::post_asap::BinaryOperator, + }, RangeWindow { intent: Box>, }, @@ -224,6 +230,7 @@ impl PhysicalOperator for Operator { matches!( self.kind, Kind::Sort { .. } + | Kind::AlignedBinary { .. } | Kind::VectorBinary { .. } | Kind::RangeWindow { .. } | Kind::HistogramQuantile @@ -271,6 +278,7 @@ impl PhysicalOperator for Operator { Kind::CurrentSeries { .. } => "CurrentSeries", Kind::VectorToScalar { .. } => "VectorToScalar", Kind::VectorBinary { .. } => "VectorBinary", + Kind::AlignedBinary { .. } => "AlignedBinary", Kind::RangeWindow { .. } => "RangeWindow", Kind::HistogramQuantile => "HistogramQuantile", Kind::Project(_) => "Project", @@ -311,6 +319,7 @@ impl PhysicalOperator for Operator { source::execute(self, inputs, context) } Kind::VectorBinary { .. } => vector_binary::execute(self, inputs, context), + Kind::AlignedBinary { .. } => aligned_binary::execute(self, inputs, context), Kind::RangeWindow { .. } | Kind::HistogramQuantile => { vector_window::execute(self, inputs, context) } diff --git a/crates/asap-physical-operators/src/operators/persisted.rs b/crates/asap-physical-operators/src/operators/persisted.rs index ff945fa7..05ca3561 100644 --- a/crates/asap-physical-operators/src/operators/persisted.rs +++ b/crates/asap-physical-operators/src/operators/persisted.rs @@ -48,6 +48,11 @@ impl TryFrom for Operator { operator, return_bool, } => Operator::vector_binary(input(0)?, input(1)?, operator, return_bool)?, + Kind::AlignedBinary { + keys, + values, + operator, + } => Operator::aligned_binary(input(0)?, input(1)?, keys, values, operator)?, Kind::RangeWindow { intent } => Operator::range_window(*intent)?, Kind::HistogramQuantile => Operator::histogram_quantile(), Kind::Project(expressions) => { diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index dec3ad10..00a2ca75 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -396,6 +396,46 @@ fn bind_operation(node: &ExecutableDagNode, inputs: &[Schema]) -> Result Result { + let columns = schema + .fields + .iter() + .enumerate() + .filter(|(_, field)| { + field.dtype + == SummaryFamilyType::Plain(planner_types::pre_asap::DataType::Float64) + }) + .map(|(i, _)| i) + .collect::>(); + match columns.as_slice() { + [value] => Ok(*value), + _ => Err(invalid("aligned binary requires one value column")), + } + }; + let (l, r) = (value(left)?, value(right)?); + let keys = left + .fields + .iter() + .enumerate() + .filter(|(i, _)| *i != l) + .map(|(i, field)| { + right + .fields + .iter() + .position(|other| other.name == field.name && other.dtype == field.dtype) + .map(|j| (i, j)) + .ok_or_else(|| invalid("aligned input identities differ")) + }) + .collect::, _>>()?; + return Operator::aligned_binary( + left.clone(), + right.clone(), + keys, + (l, r), + operator.clone(), + ); + } return Operator::vector_binary(left.clone(), right.clone(), operator.clone(), false); } if let Payload::RelationalJoin { diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index f38e735c..d33a54b2 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -1300,3 +1300,101 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { } } } + +// Precompute arithmetic must match population/window identities, never zip arrival order. +#[test] +fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { + use asap_physical_operators::physical_planner::{ + compile_node, CompiledPhysicalDag, InputContract, Source, + }; + use planner_types::{ + post_asap::*, + pre_asap::{ArithmeticOpKind, BinaryOpKind}, + }; + use std::collections::BTreeMap; + let input = schema(&[ + ("population", DataType::Utf8, false), + ("time", DataType::Timestamp, false), + ("value", DataType::Float64, false), + ]); + let node = ExecutableDagNode { + id: PostAsapNodeId(2), + output_schema: (*input).clone(), + output_state: ExecutionDataState::INGESTION_ROWS, + guarantee: None, + payload: ExecutableOperatorPayload::Binary { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + }, + }; + let program = CompiledPhysicalDag::from_operators( + [ + (0, InputContract::bounded(input.clone())), + (1, InputContract::bounded(input.clone())), + ] + .into(), + [( + 2, + ( + vec![0, 1], + compile_node(&node, &[input.clone(), input.clone()]).unwrap(), + ), + )] + .into(), + vec![2], + ) + .unwrap(); + let program = CompiledPhysicalDag::decode(&program.encode().unwrap()).unwrap(); + for (right, expected) in [ + (vec![("b", 2, 3.), ("a", 1, 2.)], Some(vec![8., 17.])), + (vec![("b", 2, 3.)], None), + (vec![("a", 1, 2.), ("a", 1, 2.)], None), + (vec![("a", 2, 2.), ("b", 1, 3.)], None), + (vec![("a", 1, f64::NAN), ("b", 2, 3.)], None), + ] { + let sources = [vec![("a", 1, 10.), ("b", 2, 20.)], right] + .into_iter() + .enumerate() + .map(|(i, rows)| { + let rows = rows + .into_iter() + .map(|(group, time, value)| { + vec![ + Value::Utf8(group.into()), + Value::Timestamp(time), + Value::Float64(value), + ] + }) + .collect(); + let batch = Batch::try_new(input.clone(), rows).unwrap(); + ( + i as u64, + Box::new(Operator::source(input.clone(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect::>(); + let graph = program.instantiate(sources).unwrap(); + let result = block_on(async { + let mut stream = graph + .execute( + program.roots(), + RunContext::new(query(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok::<_, asap_physical_operators::Error>(rows) + }); + match expected { + Some(values) => assert_eq!(floats(&result.unwrap(), 2), values), + None => assert!(result.is_err()), + } + } +} From c4c248506a44a7dadde618dea6ba2e9e027eadcc Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 14:08:47 +0000 Subject: [PATCH 84/90] feat: compile immutable population precompute DAGs in Planner --- .../src/expressions/mod.rs | 11 + .../src/physical_planner/mod.rs | 1 + .../src/physical_planner/precompute.rs | 371 ++++++++++++++++++ .../tests/precompute_population.rs | 177 +++++++++ 4 files changed, 560 insertions(+) create mode 100644 crates/asap-physical-operators/src/physical_planner/precompute.rs create mode 100644 crates/asap-physical-operators/tests/precompute_population.rs diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs index bdeb821a..2916cca7 100644 --- a/crates/asap-physical-operators/src/expressions/mod.rs +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -17,6 +17,7 @@ pub enum Expression { Planner(Box), Column(usize), ExactFloat64(usize), + FiniteFloat64(Box), LabelSet { column: usize, labels: Vec, @@ -82,6 +83,12 @@ impl Expression { expression.validate_input(input)?; Ok(expression.dtype()) } + FiniteFloat64(expression) => { + if expression.dtype(input)? != (DataType::Float64, false) { + return Err(invalid("finite update requires non-null Float64")); + } + Ok((DataType::Float64, false)) + } ExactFloat64(column) => { let (dtype, nullable) = plain(input, *column)?; if nullable || !matches!(dtype, DataType::Int64 | DataType::Float64) { @@ -178,6 +185,10 @@ impl Expression { pub(crate) fn evaluate(&self, row: &[Value]) -> Result { use Expression::*; Ok(match self { + FiniteFloat64(expression) => match expression.evaluate(row)? { + Value::Float64(value) if value.is_finite() => Value::Float64(value), + _ => return Err(invalid("summary update must be finite")), + }, ExactFloat64(column) => match row[*column] { Value::Float64(value) => Value::Float64(value), Value::Int64(value) if value.unsigned_abs() <= (1u64 << 53) => { diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 00a2ca75..7534f422 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -28,6 +28,7 @@ fn invalid(message: impl Into) -> Error { /// A deployment must authorize these frontiers before calling this function. pub type Source<'a> = Box + 'a>; +pub mod precompute; pub mod promql_rows; pub mod promql_values; diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs new file mode 100644 index 00000000..c37e64e1 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -0,0 +1,371 @@ +//! Compile immutable summary-input computation with explicit population and pane identity. +use super::*; +use planner_types::{ + post_asap::{ExecutionTiming, GroupingStrategy, SummarySchema}, + pre_asap::DataType, +}; + +/// Physical rows carry the population and pane coordinate alongside the logical value. +/// These fields preserve identities which are implicit in a stored summary instance. +pub fn population_schema(family: SummaryFamilyType) -> Schema { + Arc::new(SummarySchema { + fields: vec![ + planner_types::post_asap::SummaryField { + name: "$population".into(), + dtype: SummaryFamilyType::Plain(DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }), + nullable: false, + }, + planner_types::post_asap::SummaryField { + name: "$window_end".into(), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), + nullable: false, + }, + planner_types::post_asap::SummaryField { + name: "value".into(), + dtype: family, + nullable: false, + }, + ], + time_index: Some(1), + }) +} + +/// Validate the adapter layout during installed-plan recovery without lowering operators. +pub fn source_schema(logical: &SummarySchema) -> Result { + let states = logical + .fields + .iter() + .filter(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + .collect::>(); + let [state] = states.as_slice() else { + return Err(invalid( + "stored population requires one typed summary state", + )); + }; + if logical.fields.iter().enumerate().any(|(i, field)| matches!(&field.dtype, SummaryFamilyType::Plain(dtype) + if field.nullable || !matches!(dtype, DataType::Utf8) && !(Some(i) == logical.time_index && *dtype == DataType::Timestamp))) { + return Err(invalid("stored population metadata cannot reconstruct extra value columns")); + } + if state.nullable { + return Err(invalid("stored population state cannot be null")); + } + Ok(population_schema(state.dtype.clone())) +} + +pub fn is_population_schema(schema: &Schema) -> bool { + schema + .fields + .get(2) + .is_some_and(|field| *schema == population_schema(field.dtype.clone())) +} + +/// Compile a complete selected precompute sub-DAG. Inputs are already-computed +/// state boundaries; the deployment supplies groups, panes and states, never operations. +pub fn compile( + dag: &ExecutableDag, + frontiers: &[NodeId], + roots: &[NodeId], +) -> Result { + preflight_depth(dag)?; + dag.validate().map_err(|e| invalid(e.to_string()))?; + let nodes = dag + .nodes + .iter() + .map(|n| (u64::from(n.id.0), n)) + .collect::>(); + let frontier = frontiers.iter().copied().collect::>(); + if frontier.len() != frontiers.len() || roots.iter().any(|r| frontier.contains(r)) { + return Err(invalid( + "precompute boundaries must be distinct from outputs", + )); + } + let mut dependencies = BTreeMap::>::new(); + let mut edges = dag.edges.iter().collect::>(); + edges.sort_by_key(|edge| { + ( + edge.consumer.0, + match edge.role { + planner_types::post_asap::EdgeRole::Left => 0, + planner_types::post_asap::EdgeRole::Input => 1, + planner_types::post_asap::EdgeRole::Right => 2, + }, + ) + }); + for edge in edges { + dependencies + .entry(u64::from(edge.consumer.0)) + .or_default() + .push(u64::from(edge.producer.0)); + } + let mut ordered = Vec::new(); + let mut seen = BTreeSet::new(); + let mut pending = roots.iter().map(|&id| (id, false)).collect::>(); + while let Some((id, expanded)) = pending.pop() { + if expanded { + ordered.push(id); + continue; + } + if !seen.insert(id) { + continue; + } + if !nodes.contains_key(&id) { + return Err(invalid("missing precompute node")); + } + pending.push((id, true)); + if !frontier.contains(&id) { + pending.extend( + dependencies + .get(&id) + .into_iter() + .flatten() + .map(|id| (*id, false)), + ); + } + } + let mut sources = BTreeMap::new(); + let mut fragments = BTreeMap::new(); + let mut outputs = BTreeMap::::new(); + for id in ordered { + let node = nodes[&id]; + if frontier.contains(&id) { + let schema = source_schema(&node.output_schema)?; + sources.insert(id, InputContract::bounded(schema.clone())); + outputs.insert(id, schema); + continue; + } + if node.output_state.timing != ExecutionTiming::IngestionTime { + return Err(invalid("precompute graph contains a query-time operation")); + } + let inputs = dependencies.get(&id).cloned().unwrap_or_default(); + let schemas = inputs + .iter() + .map(|id| { + outputs + .get(id) + .cloned() + .ok_or_else(|| invalid("missing precompute input")) + }) + .collect::, _>>()?; + let graph = fragment( + node, + &schemas, + &inputs.iter().map(|id| nodes[id]).collect::>(), + )?; + outputs.insert(id, graph.output_contract(graph.roots()[0])?.schema); + fragments.insert(id, (inputs, graph)); + } + CompiledPhysicalDag::compose(sources, fragments, roots.to_vec()) +} + +fn validate_value_output(node: &ExecutableDagNode) -> Result<(), Error> { + let schema = &node.output_schema; + let values = schema + .fields + .iter() + .enumerate() + .filter(|(i, _)| Some(*i) != schema.time_index) + .collect::>(); + if !matches!(values.as_slice(), [(_, field)] if !field.nullable && field.dtype == SummaryFamilyType::Plain(DataType::Float64)) + || schema.time_index.is_some_and(|i| { + schema.fields.get(i).is_none_or(|f| { + f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Timestamp) + }) + }) + { + return Err(invalid( + "precompute value schema requires Float64 and an optional declared timestamp", + )); + } + Ok(()) +} + +fn fragment( + node: &ExecutableDagNode, + schemas: &[Schema], + parents: &[&ExecutableDagNode], +) -> Result { + let sources = schemas + .iter() + .enumerate() + .map(|(id, schema)| (id as u64, InputContract::bounded(schema.clone()))) + .collect(); + let mut operators = BTreeMap::new(); + let mut next = schemas.len() as u64; + let mut add = |inputs: Vec, op: Operator| -> Result { + let id = next; + next += 1; + operators.insert(id, (inputs, op)); + Ok(id) + }; + let root = match &node.payload { + Payload::Binary { operator } => { + validate_value_output(node)?; + if node.output_schema.time_index.is_none() + || parents.iter().any(|p| p.output_schema.time_index.is_none()) + { + return Err(invalid( + "precompute binary requires declared window timestamps", + )); + } + let [left, right] = schemas else { + return Err(invalid("precompute binary requires two inputs")); + }; + add( + vec![0, 1], + Operator::aligned_binary( + left.clone(), + right.clone(), + vec![(0, 0), (1, 1)], + (2, 2), + operator.clone(), + )?, + )? + } + Payload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + } => { + let [input] = schemas else { + return Err(invalid("finalize requires one state input")); + }; + validate_value_output(node)?; + let statistic = match &input.fields[2].dtype { + SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Sum, _) => { + crate::Statistic::Sum + } + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Count, + _, + ) => crate::Statistic::Count, + _ => { + return Err(invalid( + "precompute finalization requires explicit Sum or Count semantics", + )) + } + }; + let read = Operator::readout(input.clone(), 2, statistic, Default::default())?; + let output = read.schema(); + let read = add(vec![0], read)?; + let project = Operator::project( + output, + vec![ + ("$population".into(), Expression::Column(0)), + ("$window_end".into(), Expression::Column(1)), + ( + "value".into(), + Expression::FiniteFloat64(Box::new(Expression::ExactFloat64(2))), + ), + ], + )? + .with_output_schema(population_schema(SummaryFamilyType::Plain( + DataType::Float64, + )))?; + add(vec![read], project)? + } + Payload::SummaryAgg { + family, + input: update, + reduction, + grouping, + } => { + let [input] = schemas else { + return Err(invalid("summary update requires one input")); + }; + if update.item.is_some() + || !matches!(grouping, GroupingStrategy::PerSubpopulationInstance) + { + return Err(invalid( + "precompute keyed/shared update needs its dedicated physical candidate", + )); + } + crate::capability::validate_summary_kernel(family, update, grouping) + .map_err(Error::Invalid)?; + let labels = match reduction { + PlannerReduction::PerEntity => Expression::Column(0), + PlannerReduction::Reduce(keys) => Expression::LabelSet { + column: 0, + labels: keys + .keys() + .iter() + .map(|key| { + parents[0] + .output_schema + .fields + .get(*key) + .map(|f| f.name.clone()) + .ok_or_else(|| invalid("summary grouping column is missing")) + }) + .collect::, _>>()?, + without: keys.is_without(), + }, + }; + let weight = match &update.weight { + SummaryInputExpr::Constant(value) => Expression::Literal { + value: crate::values::Value::Float64(*value), + dtype: DataType::Float64, + }, + SummaryInputExpr::Column(ColumnRef::SampleValue) => Expression::Column(2), + SummaryInputExpr::Column(ColumnRef::Named(name)) + if parents[0].output_schema.fields.iter().any(|f| { + f.name == *name && f.dtype == SummaryFamilyType::Plain(DataType::Float64) + }) => + { + Expression::Column(2) + } + _ => { + return Err(invalid( + "summary weight does not resolve to the input value", + )) + } + }; + let project = Operator::project( + input.clone(), + vec![ + ("$population".into(), labels), + ("$window_end".into(), Expression::Column(1)), + ("value".into(), Expression::FiniteFloat64(Box::new(weight))), + ], + )? + .with_output_schema(population_schema(SummaryFamilyType::Plain( + DataType::Float64, + )))?; + let projected = project.schema(); + let project = add(vec![0], project)?; + let build = Operator::summary_build(projected, family.clone(), 2, Some(1), vec![0])?; + let built = build.schema(); + let build = add(vec![project], build)?; + add( + vec![build], + Operator::scope_timestamp(built, population_schema(family.clone()))?, + )? + } + Payload::SummaryMerge => { + let Some(input) = schemas.first() else { + return Err(invalid("summary merge requires inputs")); + }; + if schemas.iter().any(|s| s != input) { + return Err(invalid("summary merge inputs differ")); + } + let union = add( + (0..schemas.len() as u64).collect(), + Operator::union(input.clone(), schemas.len())?, + )?; + let merge = Operator::summary_merge(input.clone(), 2, vec![0])?; + let merged = merge.schema(); + let merge = add(vec![union], merge)?; + add( + vec![merge], + Operator::scope_timestamp(merged, input.clone())?, + )? + } + _ => { + return Err(invalid( + "precompute operation has no native population implementation", + )) + } + }; + CompiledPhysicalDag::from_operators(sources, operators, vec![root]) +} diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs new file mode 100644 index 00000000..ca8aec04 --- /dev/null +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -0,0 +1,177 @@ +//! Persisted precompute graphs preserve group/window identity and execute state-to-state computation. +use asap_physical_operators::{ + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{precompute, CompiledPhysicalDag, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, + Statistic, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{ArithmeticOpKind, BinaryOpKind, ColumnRef, DataType, GroupKeys, Reduction}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +#[test] +fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let schema = |dtype| SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype, + nullable: false, + }], + time_index: None, + }; + let state_schema = schema(family.clone()); + let mut value_schema = schema(SummaryFamilyType::Plain(DataType::Float64)); + value_schema.fields.push(SummaryField { + name: "time".into(), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), + nullable: false, + }); + value_schema.time_index = Some(1); + for (weight, expected) in [ + (SummaryInputExpr::Column(ColumnRef::SampleValue), 60.), + (SummaryInputExpr::Constant(1.), 4.), + ] { + let nodes = vec![ + ExecutableDagNode { + id: PostAsapNodeId(0), + payload: ExecutableOperatorPayload::SummaryMerge, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: state_schema.clone(), + guarantee: None, + }, + ExecutableDagNode { + id: PostAsapNodeId(1), + payload: ExecutableOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + }, + output_state: ExecutionDataState::INGESTION_ROWS, + output_schema: value_schema.clone(), + guarantee: None, + }, + ExecutableDagNode { + id: PostAsapNodeId(2), + payload: ExecutableOperatorPayload::Binary { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + }, + output_state: ExecutionDataState::INGESTION_ROWS, + output_schema: value_schema.clone(), + guarantee: None, + }, + ExecutableDagNode { + id: PostAsapNodeId(3), + payload: ExecutableOperatorPayload::SummaryAgg { + family: family.clone(), + input: SummaryUpdate { + weight, + ..SummaryUpdate::column(ColumnRef::SampleValue) + }, + reduction: Reduction::Reduce(GroupKeys::by(vec![])), + grouping: GroupingStrategy::PerSubpopulationInstance, + }, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: state_schema.clone(), + guarantee: None, + }, + ]; + let edges = [ + (0, 1, EdgeRole::Input), + (1, 2, EdgeRole::Left), + (1, 2, EdgeRole::Right), + (2, 3, EdgeRole::Input), + ] + .into_iter() + .map(|(producer, consumer, role)| ExecutableDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role, + intermediate_schema: nodes[producer as usize].output_schema.clone(), + data_state: nodes[producer as usize].output_state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }) + .collect(); + let dag = ExecutableDag { + nodes, + edges, + root: PostAsapNodeId(3), + }; + let program = precompute::compile(&dag, &[0], &[3]).unwrap(); + let program = CompiledPhysicalDag::decode(&program.encode().unwrap()).unwrap(); + assert_eq!(program.input_contracts().count(), 1); + for revision in [1, 2] { + let rows = [ + ("a", 1000, 2.), + ("a", 2000, 4.), + ("b", 1000, 8.), + ("b", 2000, 16.), + ] + .into_iter() + .map(|(group, time, value)| { + let mut state = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + state.update_single(value, time); + vec![ + Value::Map( + vec![(Value::Utf8("instance".into()), Value::Utf8(group.into()))].into(), + ), + Value::Timestamp(time), + Value::Summary { + family: family.clone(), + state: Arc::from(state.into_accumulator()), + }, + ] + }) + .collect(); + let input = + Batch::try_new(precompute::population_schema(family.clone()), rows).unwrap(); + let sources = BTreeMap::from([( + 0, + Box::new(Operator::source(input.schema().clone(), vec![input]).unwrap()) + as Source<'_>, + )]); + let graph = program.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision, + }, + Limits::default(), + ) + .unwrap(); + let output = block_on(async { + let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let batch = stream.next().await.unwrap().unwrap(); + assert!(stream.next().await.is_none()); + batch + }); + assert_eq!(output.rows().len(), 1); + assert!(matches!(&output.rows()[0][0], Value::Map(labels) if labels.is_empty())); + assert!(matches!(&output.rows()[0][1], Value::Timestamp(2000))); + let Value::Summary { state, .. } = &output.rows()[0][2] else { + panic!("state expected") + }; + assert_eq!( + state + .query_statistic(Statistic::Sum, &None, &Default::default()) + .unwrap(), + expected + ); + } + } +} From da8392a1a30a0179b6473c10d0a07448f9127577 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 14:24:19 +0000 Subject: [PATCH 85/90] fix: reject non-label population grouping during physical compilation --- .../src/physical_planner/precompute.rs | 8 +++++++- .../tests/precompute_population.rs | 11 +++++++++++ 2 files changed, 18 insertions(+), 1 deletion(-) diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index c37e64e1..bf9e2438 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -295,8 +295,14 @@ fn fragment( .output_schema .fields .get(*key) + .filter(|field| { + !field.nullable + && field.dtype == SummaryFamilyType::Plain(DataType::Utf8) + }) .map(|f| f.name.clone()) - .ok_or_else(|| invalid("summary grouping column is missing")) + .ok_or_else(|| { + invalid("summary grouping must identify population labels") + }) }) .collect::, _>>()?, without: keys.is_without(), diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs index ca8aec04..3917d054 100644 --- a/crates/asap-physical-operators/tests/precompute_population.rs +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -106,6 +106,17 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { edges, root: PostAsapNodeId(3), }; + let mut invalid_grouping = dag.clone(); + let ExecutableOperatorPayload::SummaryAgg { reduction, .. } = + &mut invalid_grouping.nodes[3].payload + else { + unreachable!() + }; + *reduction = Reduction::Reduce(GroupKeys::by(vec![0])); + assert!( + precompute::compile(&invalid_grouping, &[0], &[3]).is_err(), + "numeric values cannot be reinterpreted as population labels" + ); let program = precompute::compile(&dag, &[0], &[3]).unwrap(); let program = CompiledPhysicalDag::decode(&program.encode().unwrap()).unwrap(); assert_eq!(program.input_contracts().count(), 1); From e65445a5cde32d1175ca6ffe4ba7616630b37da6 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 14:36:45 +0000 Subject: [PATCH 86/90] test: verify immutable precompute graph population contracts --- .../tests/precompute_population.rs | 220 ++++++++++++++++++ 1 file changed, 220 insertions(+) diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs index 3917d054..e860f07c 100644 --- a/crates/asap-physical-operators/tests/precompute_population.rs +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -186,3 +186,223 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { } } } + +fn logical_schema(family: SummaryFamilyType) -> SummarySchema { + SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: family, + nullable: false, + }], + time_index: None, + } +} +fn state_graph( + family: SummaryFamilyType, + target: Option, + merge: bool, +) -> CompiledPhysicalDag { + let mut nodes = vec![ExecutableDagNode { + id: PostAsapNodeId(0), + payload: ExecutableOperatorPayload::SummaryMerge, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: logical_schema(family.clone()), + guarantee: None, + }]; + if merge { + nodes.push(ExecutableDagNode { + id: PostAsapNodeId(1), + payload: ExecutableOperatorPayload::SummaryMerge, + ..nodes[0].clone() + }); + } + let read_id = nodes.len() as u32; + nodes.push(ExecutableDagNode { + id: PostAsapNodeId(read_id), + payload: ExecutableOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + }, + output_state: ExecutionDataState::INGESTION_ROWS, + output_schema: logical_schema(SummaryFamilyType::Plain(DataType::Float64)), + guarantee: None, + }); + if let Some(target) = target { + nodes.push(ExecutableDagNode { + id: PostAsapNodeId(nodes.len() as u32), + payload: ExecutableOperatorPayload::SummaryAgg { + family: target.clone(), + input: SummaryUpdate::column(ColumnRef::SampleValue), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + }, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: logical_schema(target), + guarantee: None, + }); + } + let edges = (1..nodes.len()) + .map(|i| ExecutableDagEdge { + producer: nodes[i - 1].id, + consumer: nodes[i].id, + role: EdgeRole::Input, + intermediate_schema: nodes[i - 1].output_schema.clone(), + data_state: nodes[i - 1].output_state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }) + .collect(); + let root = nodes.last().unwrap().id; + precompute::compile( + &ExecutableDag { nodes, edges, root }, + &[0], + &[u64::from(root.0)], + ) + .unwrap() +} +fn native_run( + program: &CompiledPhysicalDag, + family: SummaryFamilyType, + states: Vec>, + context: RunContext, +) -> Result>, asap_physical_operators::Error> { + let program = CompiledPhysicalDag::decode(&program.encode()?)?; + let rows = states + .into_iter() + .enumerate() + .map(|(i, state)| { + vec![ + Value::Map(vec![].into()), + Value::Timestamp((i as i64 + 1) * 1000), + Value::Summary { + family: family.clone(), + state, + }, + ] + }) + .collect(); + let input = Batch::try_new(precompute::population_schema(family), rows)?; + let graph = program.instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(input.schema().clone(), vec![input])?) as Source<'_>, + )]))?; + block_on(async { + let mut rows = Vec::new(); + let mut stream = graph.execute(program.roots(), context)?.remove(0); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok(rows) + }) +} +fn ingestion_context(limits: Limits) -> RunContext { + RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }, + limits, + ) + .unwrap() +} +fn sum_state(value: f64) -> Arc { + Arc::new(asap_physical_operators::summary_kernels::SumAccumulator::with_sum(value)) +} + +// Only an explicit merge may collapse distinct pane updates before finalization. +#[test] +fn explicit_merge_changes_pane_cardinality() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + for (merge, expected) in [(false, vec![2., 7.]), (true, vec![9.])] { + let program = state_graph(family.clone(), None, merge); + let rows = native_run( + &program, + family.clone(), + vec![sum_state(2.), sum_state(7.)], + ingestion_context(Limits::default()), + ) + .unwrap(); + let values = rows + .iter() + .map(|row| match row[2] { + Value::Float64(v) => v, + _ => panic!("numeric readout expected"), + }) + .collect::>(); + assert_eq!(values, expected); + assert!(matches!(rows.last().unwrap()[1], Value::Timestamp(2000))); + } +} + +// Typed updates reject invalid domains before publishing any target state. +#[test] +fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { + let source = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let target = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::DDSketch, + SketchParams::DDSketch { alpha: 0.01 }, + ), + GroupingStrategy::default(), + ); + let program = state_graph(source.clone(), Some(target), false); + assert!(native_run( + &program, + source.clone(), + vec![sum_state(20.)], + ingestion_context(Limits::default()) + ) + .is_ok()); + for value in [-20., 0., f64::MAX, f64::NAN, f64::INFINITY] { + assert!(native_run( + &program, + source.clone(), + vec![sum_state(value)], + ingestion_context(Limits::default()) + ) + .is_err()); + } +} + +// An exact count must not silently lose units when exposed through Float64 rows. +#[test] +fn precompute_count_conversion_checks_precision() { + use asap_physical_operators::summary_kernels::exact::ExactAccumulator; + let family = SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count); + let program = state_graph(family.clone(), None, false); + for (count, valid) in [(3u64, true), ((1u64 << 53) + 1, false)] { + let mut state = + serde_json::to_value(ExactAccumulator::new(family.clone(), false).unwrap()).unwrap(); + state["scalar"]["Count"] = count.into(); + let state: ExactAccumulator = serde_json::from_value(state).unwrap(); + let result = native_run( + &program, + family.clone(), + vec![Arc::new(state)], + ingestion_context(Limits::default()), + ); + assert_eq!(result.is_ok(), valid); + } +} + +// Graph execution retains terminal cancellation and shared workspace limits. +#[test] +fn precompute_graph_enforces_cancellation_and_budget() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let program = state_graph(family.clone(), None, true); + let context = ingestion_context(Limits::default()); + context.cancel(); + let error = native_run(&program, family.clone(), vec![sum_state(1.)], context).unwrap_err(); + assert!(format!("{error:?}").contains("Cancelled")); + let error = native_run( + &program, + family, + vec![sum_state(1.)], + ingestion_context(Limits { + max_bytes: 1, + ..Limits::default() + }), + ) + .unwrap_err(); + assert!(format!("{error:?}").contains("MemoryLimit")); +} From 97743d436955fb9a804decace839b4b70a337400 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 14:50:24 +0000 Subject: [PATCH 87/90] fix: reject colliding scalar broadcast result identities --- .../src/operators/vector_binary.rs | 5 ++ .../tests/promql_values.rs | 77 +++++++++++++++++++ 2 files changed, 82 insertions(+) diff --git a/crates/asap-physical-operators/src/operators/vector_binary.rs b/crates/asap-physical-operators/src/operators/vector_binary.rs index 0ef77d80..cc60ff4c 100644 --- a/crates/asap-physical-operators/src/operators/vector_binary.rs +++ b/crates/asap-physical-operators/src/operators/vector_binary.rs @@ -130,6 +130,7 @@ pub(super) fn execute<'a>( let mut workspace = Workspace::new(&context)?; let mut work = Cooperative::new(&context); let mut rows = Vec::new(); + let mut result_identities = std::collections::BTreeSet::new(); let mut emit = |labels: Labels, a: f64, b: f64| -> Result<(), Error> { let arithmetic = matches!(operator.kind, BinaryOpKind::Arithmetic(_)); let result = match crate::expressions::arithmetic::evaluate_binary(operator, a, b)? { @@ -158,6 +159,10 @@ pub(super) fn execute<'a>( } else { labels }; + workspace.grow(label_bytes(&labels) + 64)?; + if !result_identities.insert(labels.clone()) { + return Err(invalid("duplicate vector result labels")); + } workspace.grow( label_bytes(&labels) + std::mem::size_of::>() diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs index 507f1953..b80b0a4a 100644 --- a/crates/asap-physical-operators/tests/promql_values.rs +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -318,3 +318,80 @@ fn compiled_constant_needs_no_deployment_source() { let result = run_inputs(graph, vec![]).unwrap(); assert!(matches!(result[0][0], Value::Float64(3.))); } + +// Scalar broadcasting cannot silently create duplicate result identities when +// arithmetic or bool comparisons remove the metric name. +#[test] +fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { + use planner_types::{ + post_asap::BinaryOperator, + pre_asap::{ArithmeticOpKind, BinaryOpKind, CompareOpKind}, + }; + for left_scalar in [false, true] { + for names in [["a", "a"], ["a", "b"]] { + for (kind, return_bool) in [ + (BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), false), + (BinaryOpKind::Compare(CompareOpKind::Gt), true), + ] { + let graph = compile_binary( + &BinaryOperator { + kind, + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + left_scalar, + !left_scalar, + ) + .unwrap(); + let vector = Batch::try_new( + vector_schema(), + vec![ + row(&[("__name__", names[0]), ("job", "api")], 2.), + row(&[("__name__", names[1]), ("job", "api")], 3.), + ], + ) + .unwrap(); + let scalar = + Batch::try_new(scalar_schema(), vec![vec![Value::Float64(1.)]]).unwrap(); + let result = run_inputs( + graph, + if left_scalar { + vec![scalar, vector] + } else { + vec![vector, scalar] + }, + ); + assert!(result.is_err(), "duplicate output label sets were accepted"); + } + } + } + let graph = compile_binary( + &BinaryOperator { + kind: BinaryOpKind::Compare(CompareOpKind::Gt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + false, + false, + true, + ) + .unwrap(); + let rows = vec![ + row(&[("__name__", "a"), ("job", "api")], 2.), + row(&[("__name__", "b"), ("job", "api")], 3.), + ]; + equal_rows( + run_inputs( + graph, + vec![ + Batch::try_new(vector_schema(), rows.clone()).unwrap(), + Batch::try_new(scalar_schema(), vec![vec![Value::Float64(1.)]]).unwrap(), + ], + ) + .unwrap(), + rows, + ); +} From 06efd1db655bb0a1b1a79ab7ba6bfc1c991a0694 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 16:10:54 +0000 Subject: [PATCH 88/90] fix: preserve opaque summary columns in physical projections --- .../src/operators/projection.rs | 9 + .../src/physical_planner/mod.rs | 5 +- .../tests/summary_projection.rs | 158 ++++++++++++++++++ 3 files changed, 171 insertions(+), 1 deletion(-) create mode 100644 crates/asap-physical-operators/tests/summary_projection.rs diff --git a/crates/asap-physical-operators/src/operators/projection.rs b/crates/asap-physical-operators/src/operators/projection.rs index 9337f7ab..4a174ed9 100644 --- a/crates/asap-physical-operators/src/operators/projection.rs +++ b/crates/asap-physical-operators/src/operators/projection.rs @@ -4,6 +4,15 @@ impl Operator { let fields = columns .iter() .map(|(name, e)| { + if let Expression::Column(index) = e { + let mut field = input + .fields + .get(*index) + .ok_or_else(|| invalid("projection column out of range"))? + .clone(); + field.name = name.clone(); + return Ok(field); + } let (t, n) = e.dtype(&input)?; Ok(result_field(name, t, n)) }) diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 7534f422..809ed410 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -498,7 +498,10 @@ fn bind_operation(node: &ExecutableDagNode, inputs: &[Schema]) -> Result Expression::Column(*index), + expr => expression(expr, input)?, + }, )) }) .collect::>()?, diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs new file mode 100644 index 00000000..f8a2acbe --- /dev/null +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -0,0 +1,158 @@ +//! Opaque state travels through a retained physical projection without scalar decoding. +use asap_physical_operators::{ + expressions::Expression, + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{compile, CompiledPhysicalDag, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{ColumnRef, DataType, ProjectItem, QueryExpr}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +// A Post-ASAP projection may reorder/rename summary columns; recovery must retain +// the family and pass through the same immutable state, without decoding the payload. +#[test] +fn post_asap_summary_projection_survives_recovery() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let schema = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }, + SummaryField { + name: "service".into(), + dtype: SummaryFamilyType::Plain(DataType::Utf8), + nullable: false, + }, + ], + time_index: None, + }); + let output = SummarySchema { + fields: vec![ + schema.fields[1].clone(), + SummaryField { + name: "renamed".into(), + ..schema.fields[0].clone() + }, + ], + time_index: None, + }; + let dag = ExecutableDag { + nodes: vec![ + ExecutableDagNode { + id: PostAsapNodeId(0), + payload: ExecutableOperatorPayload::SummaryMerge, + output_schema: (*schema).clone(), + output_state: ExecutionDataState::INGESTION_SUMMARY, + guarantee: None, + }, + ExecutableDagNode { + id: PostAsapNodeId(1), + payload: ExecutableOperatorPayload::Value { + operation: ValueOperation::Project { + cols: vec![1, 0] + .into_iter() + .map(|index| ProjectItem { + alias: None, + expr: QueryExpr::Column(index), + }) + .collect(), + qualifier: None, + }, + }, + output_schema: output.clone(), + output_state: ExecutionDataState::INGESTION_SUMMARY, + guarantee: None, + }, + ], + edges: vec![ExecutableDagEdge { + producer: PostAsapNodeId(0), + consumer: PostAsapNodeId(1), + role: EdgeRole::Input, + intermediate_schema: (*schema).clone(), + data_state: ExecutionDataState::INGESTION_SUMMARY, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }], + root: PostAsapNodeId(1), + }; + let program = compile( + &dag, + BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), + &[1], + ) + .unwrap(); + let encoded = program.encode().unwrap(); + let program = CompiledPhysicalDag::decode(&encoded).unwrap(); + let mut forged: serde_json::Value = serde_json::from_slice(&encoded).unwrap(); + forged["nodes"]["1"]["Operator"]["operator"]["output"]["fields"][1]["dtype"] = + serde_json::json!({"Plain": "float64"}); + assert!(CompiledPhysicalDag::decode(&serde_json::to_vec(&forged).unwrap()).is_err()); + assert!(Operator::project( + schema.clone(), + vec![("invalid".into(), Expression::Column(2))] + ) + .is_err()); + assert!(Operator::project( + schema.clone(), + vec![( + "invalid".into(), + Expression::Negate(Box::new(Expression::Column(0))) + )] + ) + .is_err()); + assert_eq!(*program.output_contract(1).unwrap().schema, output); + let mut accumulator = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + accumulator.update_single(7., 1); + let state = Arc::from(accumulator.into_accumulator()); + let batch = Batch::try_new( + schema.clone(), + vec![vec![ + Value::Summary { + family, + state: Arc::clone(&state), + }, + Value::Utf8("api".into()), + ]], + ) + .unwrap(); + let graph = program + .instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, + )])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut output = graph.execute(&[1], context).unwrap().remove(0); + let batch = output.next().await.unwrap().unwrap(); + assert!(matches!(&batch.rows()[0][0], Value::Utf8(label) if label.as_ref() == "api")); + let Value::Summary { + state: projected, .. + } = &batch.rows()[0][1] + else { + panic!("missing summary") + }; + assert!(Arc::ptr_eq(&state, projected)); + assert!(output.next().await.is_none()); + }); +} From aea2321ed2c890d8f5fcb1adb3c98fb759c46e5b Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 16:25:02 +0000 Subject: [PATCH 89/90] feat: retain exact summary readouts as physical graphs --- .../src/physical_planner/promql_values.rs | 63 +++++++++++++ .../tests/promql_values.rs | 93 +++++++++++++++++++ 2 files changed, 156 insertions(+) diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/asap-physical-operators/src/physical_planner/promql_values.rs index 7957cb40..54a0a47a 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_values.rs @@ -206,3 +206,66 @@ pub fn compile_vector_to_scalar() -> Result { vector_schema(), ) } + +/// A stored exact-state input retains the complete population identity. The +/// deployment supplies eligible panes; merging and finalization are computation. +pub fn exact_state_schema(family: SummaryFamilyType) -> Result { + if !matches!(family, SummaryFamilyType::ExactAggregate(..)) { + return Err(invalid("exact-state input requires an exact family")); + } + crate::values::validate_family(&family)?; + let mut schema = (*vector_schema()).clone(); + schema.fields[1].dtype = family; + Ok(Arc::new(schema)) +} + +/// Retain exact readout semantics before any deployment state is opened. +pub fn compile_exact_readout( + family: SummaryFamilyType, + lookback_ms: u64, + preserve_metric_name: bool, +) -> Result { + use planner_types::post_asap::ExactKind; + let statistic = match &family { + SummaryFamilyType::ExactAggregate(kind, _) => match kind { + ExactKind::Sum => crate::Statistic::Sum, + ExactKind::Count => crate::Statistic::Count, + ExactKind::Min => crate::Statistic::Min, + ExactKind::Max => crate::Statistic::Max, + ExactKind::Rate => crate::Statistic::Rate, + ExactKind::Increase => crate::Statistic::Increase, + ExactKind::IRate => return Err(invalid("instant-rate state readout is not supported")), + }, + _ => return Err(invalid("exact readout requires an exact family")), + }; + let input = exact_state_schema(family)?; + let merge = Operator::summary_merge(input.clone(), 1, vec![0])?; + let mut readout = Operator::readout(merge.schema(), 1, statistic, Default::default())?; + if matches!( + statistic, + crate::Statistic::Rate | crate::Statistic::Increase + ) { + readout = readout.with_counter_lookback( + i64::try_from(lookback_ms).map_err(|_| invalid("counter lookback exceeds Int64"))?, + )?; + } + let project = Operator::project( + readout.schema(), + vec![ + ( + "labels".into(), + if preserve_metric_name { + Expression::Column(0) + } else { + Expression::LabelSet { + column: 0, + labels: vec![], + without: true, + } + }, + ), + ("value".into(), Expression::ExactFloat64(1)), + ], + )?; + unary(vec![merge, readout, project], input) +} diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs index b80b0a4a..cec1744e 100644 --- a/crates/asap-physical-operators/tests/promql_values.rs +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -395,3 +395,96 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { rows, ); } + +// Persisted exact readout graphs, rather than the storage adapter, merge panes, +// finalize each population, and preserve the requested metric-name semantics. +#[test] +fn exact_state_readouts_recover_and_finalize_panes() { + use asap_physical_operators::factory::create_planner_accumulator; + use planner_types::post_asap::*; + use std::sync::Arc; + for (kind, params, expected) in [ + (ExactKind::Sum, ExactParams::Sum, 12.), + (ExactKind::Count, ExactParams::Count, 4.), + (ExactKind::Min, ExactParams::Min, 1.), + (ExactKind::Max, ExactParams::Max, 5.), + ] { + let family = SummaryFamilyType::ExactAggregate(kind, params); + for preserve in [false, true] { + let rows = [[1., 2.], [4., 5.]] + .into_iter() + .map(|samples| { + let mut state = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + for sample in samples { + state.update_single(sample, 0); + } + let labels = row(&[("__name__", "m"), ("instance", "a")], 0.).remove(0); + vec![ + labels, + Value::Summary { + family: family.clone(), + state: Arc::from(state.into_accumulator()), + }, + ] + }) + .collect(); + let output = run_inputs( + compile_exact_readout(family.clone(), 60_000, preserve).unwrap(), + vec![Batch::try_new(exact_state_schema(family.clone()).unwrap(), rows).unwrap()], + ) + .unwrap(); + let labels = if preserve { + vec![("__name__", "m"), ("instance", "a")] + } else { + vec![("instance", "a")] + }; + equal_rows(output, vec![row(&labels, expected)]); + } + } +} + +#[test] +fn recovered_exact_counter_uses_window_and_omits_insufficient_samples() { + use asap_physical_operators::factory::create_planner_accumulator; + use planner_types::post_asap::*; + use std::sync::Arc; + for (kind, params, expected) in [ + (ExactKind::Rate, ExactParams::Rate, 1.), + (ExactKind::Increase, ExactParams::Increase, 60.), + ] { + let family = SummaryFamilyType::ExactAggregate(kind, params); + let rows = [1, 2] + .into_iter() + .map(|count| { + let mut state = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + state.update_single(100., -50_000); + if count == 2 { + state.update_single(140., -10_000); + } + vec![ + row(&[("instance", if count == 1 { "one" } else { "two" })], 0.).remove(0), + Value::Summary { + family: family.clone(), + state: Arc::from(state.into_accumulator()), + }, + ] + }) + .collect(); + let output = run_inputs( + compile_exact_readout(family.clone(), 60_000, false).unwrap(), + vec![Batch::try_new(exact_state_schema(family).unwrap(), rows).unwrap()], + ) + .unwrap(); + equal_rows(output, vec![row(&[("instance", "two")], expected)]); + } +} From 76fbbf16cc44b19f56a780bfdb47a95327e84711 Mon Sep 17 00:00:00 2001 From: zzylol Date: Tue, 29 Sep 2026 17:58:07 +0000 Subject: [PATCH 90/90] refactor: adopt PostAsapDag names in the physical layer Follow #470: the physical compiler consumes the logical Post-ASAP DAG. The design doc now names the Pre-ASAP and Post-ASAP DAGs as the two logical layers and records that placement ownership between per-node phases and frontier enumeration is still open. Co-Authored-By: Claude Opus 5.5 --- crates/asap-aware-mapping/src/replacement.rs | 10 ++-- crates/asap-physical-operators/README.md | 4 +- .../asap-physical-operators/src/capability.rs | 2 +- .../src/physical_planner/candidates.rs | 6 +- .../src/physical_planner/mod.rs | 16 ++--- .../src/physical_planner/precompute.rs | 8 +-- .../src/physical_planner/promql_rows.rs | 12 ++-- .../src/physical_planner/temporal_panes.rs | 2 +- .../tests/current_series_heap.rs | 4 +- .../tests/physical_dag.rs | 58 +++++++++---------- .../tests/physical_semantics.rs | 4 +- .../tests/precompute_candidates.rs | 20 +++---- .../tests/precompute_population.rs | 42 +++++++------- .../tests/promql_binary.rs | 8 +-- .../asap-physical-operators/tests/raw_scan.rs | 14 ++--- .../tests/summary_projection.rs | 12 ++-- .../tests/weighted_topk_binding.rs | 24 ++++---- .../tests/sql_to_physical.rs | 8 +-- .../summary_maintenance_lifecycle_e2e.rs | 14 ++--- .../src/post_asap/semantic_definition.rs | 57 +++++++++--------- .../physical-planning-and-deployment.md | 11 +++- 21 files changed, 172 insertions(+), 164 deletions(-) diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 014c682e..f8f55a1d 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -1374,18 +1374,18 @@ impl<'a> SketchAlgorithmStrategy<'a> { let Replacement::Summary(node) = &candidate.replacement else { return false; }; - let Ok(dag) = asap_types::post_asap::compile_executable_dag(node) else { + let Ok(dag) = asap_types::post_asap::compile_post_asap_dag(node) else { return false; }; if !dag.nodes.iter().any(|node| match &node.payload { - asap_types::post_asap::ExecutableOperatorPayload::SummaryAgg { + asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } => matches!( kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap ), - asap_types::post_asap::ExecutableOperatorPayload::SummaryAgg { + asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), .. } => true, @@ -1396,7 +1396,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { let Some(placed) = place(node) else { return false; }; - if asap_types::post_asap::compile_executable_dag(&placed).is_err() { + if asap_types::post_asap::compile_post_asap_dag(&placed).is_err() { return false; } let Ok(placed) = finalize_query_candidate(placed, root) else { @@ -7079,7 +7079,7 @@ mod tests { .unwrap() .expect("exact ranking is legal for an approximate request"); assert!(node.guarantee.as_ref().unwrap().is_exact()); - asap_types::post_asap::compile_executable_dag(&node).unwrap(); + asap_types::post_asap::compile_post_asap_dag(&node).unwrap(); } // Exact Top-K consumes the Planner's maintained temporal values. diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index 31bbc49a..c00739f0 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -47,7 +47,7 @@ assert!(matches!(batch.rows()[0][0], Value::Int64(-7))); # Ok::<(), asap_physical_operators::dag::Error>(()) ``` -`physical_planner::compile` accepts a post-ASAP DAG and typed input contracts. +`physical_planner::compile` accepts a logical Post-ASAP DAG (`PostAsapDag`) and typed input contracts. The resulting candidate is instantiated with deployment readers after selection. It rejects unsupported operations and schema mismatches before starting a source. Implement `PhysicalOperator` for a deployment source, including asynchronous I/O; computation operators remain in @@ -100,7 +100,7 @@ There is no spill or partitioned parallel execution in this implementation. ## Physical compilation and deployment inputs -`physical_planner::compile` accepts a Planner `ExecutableDag`, typed +`physical_planner::compile` accepts a Planner `PostAsapDag`, typed `InputContract`s and output roots. It returns a reusable `CompiledPhysicalDag` containing selected native operators and no live readers. Compilation validates schemas, input ordering, sharing and boundedness before deployment source access. diff --git a/crates/asap-physical-operators/src/capability.rs b/crates/asap-physical-operators/src/capability.rs index 12ecabcd..8ca526e1 100644 --- a/crates/asap-physical-operators/src/capability.rs +++ b/crates/asap-physical-operators/src/capability.rs @@ -4,7 +4,7 @@ //! native batch representation. `validate_native_family` and //! `validate_native_readout` check native state and scalar readout support. //! Keyed weighted-frequency readouts are checked by `Operator::keyed_readout`. -//! A successful kernel check alone does not mean an executable DAG will bind. +//! A successful kernel check alone does not mean a physical DAG will bind. //! //! Persisted state uses `stored_state` decoding and readout contracts; support //! there does not imply a native build/merge operator. Full plan acceptance is diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs index 148035d7..d173abe0 100644 --- a/crates/asap-physical-operators/src/physical_planner/candidates.rs +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -19,7 +19,7 @@ pub struct PhysicalCandidate { /// contract used to build each output. This API never treats a result from a /// different window or revision as interchangeable merely because types match. pub fn compile_candidate( - dag: &ExecutableDag, + dag: &PostAsapDag, inputs: BTreeMap, roots: &[NodeId], frontier: &[NodeId], @@ -72,7 +72,7 @@ pub fn compile_candidate( /// and deployment feasibility are evaluated separately before cost selection. /// Exceeding the search budget returns an error, never a partial inventory. pub fn enumerate_frontiers( - dag: &ExecutableDag, + dag: &PostAsapDag, inputs: &BTreeMap, roots: &[NodeId], max_candidates: usize, @@ -141,7 +141,7 @@ pub fn enumerate_frontiers( /// Lower every maintenance candidate before feasibility/cost evaluation. Keep /// individual failures visible; do not substitute another computation on error. pub fn compile_candidates( - dag: &ExecutableDag, + dag: &PostAsapDag, inputs: BTreeMap, roots: &[NodeId], frontiers: &[Vec], diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 809ed410..630f9da2 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -8,7 +8,7 @@ use crate::{ }; use planner_types::{ post_asap::{ - ExactOperation, ExecutableDag, ExecutableDagNode, ExecutableOperatorPayload as Payload, + ExactOperation, PostAsapDag, PostAsapDagNode, PostAsapOperatorPayload as Payload, SketchQuery, SummaryFamilyType, SummaryInputExpr, ValueOperation, }, pre_asap::{ @@ -50,7 +50,7 @@ pub use compiled::{CompiledPhysicalDag, InputContract}; /// Compile computation without opening or retaining deployment readers. /// Input contracts identify explicit boundaries selected by maintenance planning. pub fn compile( - dag: &ExecutableDag, + dag: &PostAsapDag, inputs: BTreeMap, roots: &[NodeId], ) -> Result { @@ -60,7 +60,7 @@ pub fn compile( /// Convenience for callers that already resolved inputs. Lowering still uses /// only their contracts, and instantiation checks those contracts again. pub fn bind<'a>( - dag: &ExecutableDag, + dag: &PostAsapDag, sources: BTreeMap>, roots: &[NodeId], ) -> Result, Error> { @@ -73,7 +73,7 @@ pub fn bind<'a>( /// Resolve raw scan connectors before invoking the reader-independent compiler. pub fn bind_with_data_sources<'a>( - dag: &ExecutableDag, + dag: &PostAsapDag, mut sources: BTreeMap>, roots: &[NodeId], data_sources: &crate::sources::DataSources, @@ -108,7 +108,7 @@ pub fn bind_with_data_sources<'a>( } fn compile_internal( - dag: &ExecutableDag, + dag: &PostAsapDag, mut sources: BTreeMap, roots: &[NodeId], ) -> Result { @@ -385,14 +385,14 @@ fn compile_internal( /// Bind a Planner node against the schemas supplied by its deployment edges. /// This is the same checked path used by complete DAG binding. -pub fn compile_node(node: &ExecutableDagNode, inputs: &[Schema]) -> Result { +pub fn compile_node(node: &PostAsapDagNode, inputs: &[Schema]) -> Result { for schema in inputs { crate::values::validate_schema(schema)?; } bind_operation(node, inputs)?.with_output_schema(Arc::new(node.output_schema.clone())) } -fn bind_operation(node: &ExecutableDagNode, inputs: &[Schema]) -> Result { +fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result { if let Payload::Binary { operator } = &node.payload { let [left, right] = inputs else { return Err(invalid("binary requires two inputs")); @@ -798,7 +798,7 @@ impl PhysicalOperator for CheckedSource<'_> { } // Bound recursion before invoking the upstream recursive provenance validator. -fn preflight_depth(dag: &ExecutableDag) -> Result<(), Error> { +fn preflight_depth(dag: &PostAsapDag) -> Result<(), Error> { let mut remaining = dag .nodes .iter() diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index bf9e2438..7284b70c 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -66,7 +66,7 @@ pub fn is_population_schema(schema: &Schema) -> bool { /// Compile a complete selected precompute sub-DAG. Inputs are already-computed /// state boundaries; the deployment supplies groups, panes and states, never operations. pub fn compile( - dag: &ExecutableDag, + dag: &PostAsapDag, frontiers: &[NodeId], roots: &[NodeId], ) -> Result { @@ -161,7 +161,7 @@ pub fn compile( CompiledPhysicalDag::compose(sources, fragments, roots.to_vec()) } -fn validate_value_output(node: &ExecutableDagNode) -> Result<(), Error> { +fn validate_value_output(node: &PostAsapDagNode) -> Result<(), Error> { let schema = &node.output_schema; let values = schema .fields @@ -184,9 +184,9 @@ fn validate_value_output(node: &ExecutableDagNode) -> Result<(), Error> { } fn fragment( - node: &ExecutableDagNode, + node: &PostAsapDagNode, schemas: &[Schema], - parents: &[&ExecutableDagNode], + parents: &[&PostAsapDagNode], ) -> Result { let sources = schemas .iter() diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index 83208f50..cb680096 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -150,9 +150,9 @@ pub fn compile_current_series_readout( selected: &Rc, ) -> Result { use planner_types::post_asap::{ - compile_executable_dag, maintained_population::PopulationReadout, SummaryField, + compile_post_asap_dag, maintained_population::PopulationReadout, SummaryField, }; - let mut dag = compile_executable_dag(selected).map_err(|error| invalid(error.to_string()))?; + let mut dag = compile_post_asap_dag(selected).map_err(|error| invalid(error.to_string()))?; // Typed snapshot candidates already carry full identity throughout the DAG. // Cut at the population output, preserving all selected heap/readout nodes. let populations = dag.nodes.iter().filter(|node| matches!(&node.payload, @@ -267,7 +267,7 @@ pub fn compile_rate_ranking( Error, > { use planner_types::post_asap::{ - compile_executable_dag_with_node_ids, ExactKind, SummaryExpr, SummaryNode, + compile_post_asap_dag_with_node_ids, ExactKind, SummaryExpr, SummaryNode, }; fn frontier(node: &Rc) -> Option> { match &node.expr { @@ -300,7 +300,7 @@ pub fn compile_rate_ranking( { return Err(invalid("Rate ranking requires complete series identity")); } - let compiled = compile_executable_dag_with_node_ids(selected) + let compiled = compile_post_asap_dag_with_node_ids(selected) .map_err(|error| invalid(error.to_string()))?; let id = u64::from( compiled @@ -324,9 +324,9 @@ pub fn compile_fixed_window_rate_aggregation( selected: &Rc, ) -> Result { use planner_types::post_asap::{ - compile_executable_dag, ExactKind, ExecutionTiming, SketchAlgorithm, + compile_post_asap_dag, ExactKind, ExecutionTiming, SketchAlgorithm, }; - let dag = compile_executable_dag(selected).map_err(|e| invalid(e.to_string()))?; + let dag = compile_post_asap_dag(selected).map_err(|e| invalid(e.to_string()))?; let sources = dag .nodes .iter() diff --git a/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs b/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs index 4e4d5722..fae56c6b 100644 --- a/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs +++ b/crates/asap-physical-operators/src/physical_planner/temporal_panes.rs @@ -45,7 +45,7 @@ pub struct TemporalPaneCandidate { /// This initial realization consumes complete pane populations and emits full /// state snapshots. Cross-run delta accumulation belongs to other candidates. pub fn compile_temporal_pane_candidate( - dag: &ExecutableDag, + dag: &PostAsapDag, inputs: BTreeMap, roots: &[NodeId], maintenance: &TemporalPaneMaintenance, diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs index 7abbeb04..afa99cda 100644 --- a/crates/asap-physical-operators/tests/current_series_heap.rs +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -341,11 +341,11 @@ fn planner_current_series_candidate_compiles_with_dynamic_identity() { ) .candidate(&root) .unwrap(); - let logical = compile_executable_dag(&selected).unwrap(); + let logical = compile_post_asap_dag(&selected).unwrap(); let raw = logical .nodes .iter() - .find(|node| matches!(node.payload, ExecutableOperatorPayload::Fallback { .. })) + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) .unwrap(); let raw_schema = Arc::new(raw.output_schema.clone()); let physical = compile( diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index d33a54b2..95abe685 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -454,32 +454,32 @@ fn bind_post_asap_before_execution() { use asap_physical_operators::dag::planner::bind; use planner_types::{ post_asap::{ - EdgeRole, ExecutableDag, ExecutableDagEdge, ExecutableDagNode, - ExecutableOperatorPayload, ExecutionDataState, GroupingEdgeCompatibility, - PostAsapNodeId, ValueOperation, WindowEdgeCompatibility, + EdgeRole, ExecutionDataState, GroupingEdgeCompatibility, PostAsapDag, PostAsapDagEdge, + PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, ValueOperation, + WindowEdgeCompatibility, }, pre_asap::{ArithmeticOpKind, ProjectItem, QueryExpr, ScalarValue}, }; use std::{collections::BTreeMap, rc::Rc}; let schema = schema(&[("value", DataType::Float64, false)]); - let node = |id, payload| ExecutableDagNode { + let node = |id, payload| PostAsapDagNode { id: PostAsapNodeId(id), payload, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*schema).clone(), guarantee: None, }; - let mut dag = ExecutableDag { + let mut dag = PostAsapDag { nodes: vec![ node( 0, - ExecutableOperatorPayload::Fallback { + PostAsapOperatorPayload::Fallback { expression: QueryExpr::promql_scalar(1.), }, ), node( 1, - ExecutableOperatorPayload::Value { + PostAsapOperatorPayload::Value { operation: ValueOperation::Project { cols: vec![ProjectItem { alias: None, @@ -494,7 +494,7 @@ fn bind_post_asap_before_execution() { }, ), ], - edges: vec![ExecutableDagEdge { + edges: vec![PostAsapDagEdge { producer: PostAsapNodeId(0), consumer: PostAsapNodeId(1), role: EdgeRole::Input, @@ -520,7 +520,7 @@ fn bind_post_asap_before_execution() { let native = bind(&dag, sources(), &[1]).unwrap(); assert_eq!(floats(&run(&native, 1, query()), 0), vec![3.]); assert!(bind(&dag, BTreeMap::new(), &[1]).is_err()); - dag.nodes[1].payload = ExecutableOperatorPayload::Value { + dag.nodes[1].payload = PostAsapOperatorPayload::Value { operation: ValueOperation::Extension { name: "unknown".into(), }, @@ -555,8 +555,8 @@ fn source_batches_must_match_the_bound_schema() { use asap_physical_operators::dag::{self, PhysicalOperator}; use planner_types::{ post_asap::{ - ExecutableDag, ExecutableDagNode, ExecutableOperatorPayload, ExecutionDataState, - PostAsapNodeId, + ExecutionDataState, PostAsapDag, PostAsapDagNode, PostAsapNodeId, + PostAsapOperatorPayload, }, pre_asap::QueryExpr, }; @@ -592,10 +592,10 @@ fn source_batches_must_match_the_bound_schema() { } let expected = schema(&[("value", DataType::Float64, false)]); let starts = Rc::new(Cell::new(0)); - let plan = ExecutableDag { - nodes: vec![ExecutableDagNode { + let plan = PostAsapDag { + nodes: vec![PostAsapDagNode { id: PostAsapNodeId(0), - payload: ExecutableOperatorPayload::Fallback { + payload: PostAsapOperatorPayload::Fallback { expression: QueryExpr::promql_scalar(1.), }, output_state: ExecutionDataState::QUERY_ROWS, @@ -673,14 +673,14 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { ("score", DataType::Float64, false), ]); let keys_schema = schema(&[("key", DataType::Utf8, false)]); - let node = |id, payload, schema: &Schema| ExecutableDagNode { + let node = |id, payload, schema: &Schema| PostAsapDagNode { id: PostAsapNodeId(id), payload, output_schema: (**schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, guarantee: None, }; - let edge = |producer, consumer, role, schema: &Schema| ExecutableDagEdge { + let edge = |producer, consumer, role, schema: &Schema| PostAsapDagEdge { producer: PostAsapNodeId(producer), consumer: PostAsapNodeId(consumer), role, @@ -690,25 +690,25 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { window: WindowEdgeCompatibility::NotApplicable, }; let groups = GroupKeys::by(vec![0]); - let dag = ExecutableDag { + let dag = PostAsapDag { nodes: vec![ node( 0, - ExecutableOperatorPayload::Fallback { + PostAsapOperatorPayload::Fallback { expression: QueryExpr::promql_scalar(0.), }, &rows_schema, ), node( 1, - ExecutableOperatorPayload::Fallback { + PostAsapOperatorPayload::Fallback { expression: QueryExpr::promql_scalar(0.), }, &keys_schema, ), node( 2, - ExecutableOperatorPayload::RelationalJoin { + PostAsapOperatorPayload::RelationalJoin { join_kind: JoinKind::Semi, pruning: None, pred: Predicate(Rc::new(QueryExpr::Compare { @@ -721,7 +721,7 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { ), node( 3, - ExecutableOperatorPayload::Value { + PostAsapOperatorPayload::Value { operation: ValueOperation::Sort { keys: vec![SortKey { expr: QueryExpr::Column(2), @@ -735,7 +735,7 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { ), node( 4, - ExecutableOperatorPayload::Value { + PostAsapOperatorPayload::Value { operation: ValueOperation::Limit { n: 1, offset: 0, @@ -1095,7 +1095,7 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { compile_node, CompiledPhysicalDag, InputContract, Source, }; use planner_types::post_asap::{ - ExecutableDagNode, ExecutableOperatorPayload, ExecutionDataState, PostAsapNodeId, + ExecutionDataState, PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, ValueOperation, }; use planner_types::pre_asap::{ @@ -1120,9 +1120,9 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { .map(|c| (c.name.as_str(), c.dtype.clone(), c.nullable)) .collect::>(), ); - let node = |id, operation| ExecutableDagNode { + let node = |id, operation| PostAsapDagNode { id: PostAsapNodeId(id), - payload: ExecutableOperatorPayload::Value { operation }, + payload: PostAsapOperatorPayload::Value { operation }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*input).clone(), guarantee: None, @@ -1203,12 +1203,12 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { use std::{collections::BTreeMap, rc::Rc}; let schema = schema(&[("key", DataType::Utf8, false)]); for certified in [false, true] { - let node = ExecutableDagNode { + let node = PostAsapDagNode { id: PostAsapNodeId(2), output_schema: (*schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, guarantee: None, - payload: ExecutableOperatorPayload::RelationalJoin { + payload: PostAsapOperatorPayload::RelationalJoin { join_kind: JoinKind::Semi, pred: Predicate(Rc::new(QueryExpr::Compare { left: Rc::new(QueryExpr::Column(0)), @@ -1317,12 +1317,12 @@ fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { ("time", DataType::Timestamp, false), ("value", DataType::Float64, false), ]); - let node = ExecutableDagNode { + let node = PostAsapDagNode { id: PostAsapNodeId(2), output_schema: (*input).clone(), output_state: ExecutionDataState::INGESTION_ROWS, guarantee: None, - payload: ExecutableOperatorPayload::Binary { + payload: PostAsapOperatorPayload::Binary { operator: BinaryOperator { kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), vector_match: None, diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs index 7270bd1e..51854d2d 100644 --- a/crates/asap-physical-operators/tests/physical_semantics.rs +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -360,9 +360,9 @@ fn global_extrema_bind_with_planner_derived_schema() { .unwrap(); let result = derived.columns[0].clone(); let output = schema(&[(&result.name, result.dtype, result.nullable)]); - let node = ExecutableDagNode { + let node = PostAsapDagNode { id: PostAsapNodeId(1), - payload: ExecutableOperatorPayload::Value { + payload: PostAsapOperatorPayload::Value { operation: ValueOperation::Exact(ExactOperation::Aggregate { reduction: PlanReduction::Reduce(GroupKeys::by(vec![])), measures: vec![measure], diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index 2acb39a0..3b64596e 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -51,14 +51,14 @@ fn grouped_rate_space() -> asap_aware_mapping::PlanSpace<&'static str> { search_workload(vec![("grouped-rate", root)]) } -fn grouped_rate() -> ExecutableDag { +fn grouped_rate() -> PostAsapDag { let space = grouped_rate_space(); let selected = space .global_selection(&DefaultCostModel) .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - compile_executable_dag(&selected).unwrap() + compile_post_asap_dag(&selected).unwrap() } fn run(plan: &CompiledPhysicalDag, inputs: BTreeMap, scope: Scope) -> Vec { let sources = inputs @@ -91,7 +91,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .find(|node| { matches!( node.payload, - ExecutableOperatorPayload::SummaryAgg { + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), .. } @@ -104,7 +104,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .find(|node| { matches!( node.payload, - ExecutableOperatorPayload::Value { + PostAsapOperatorPayload::Value { operation: ValueOperation::FinalizeExactAccumulator } ) @@ -112,7 +112,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .unwrap(); let input_schema = Arc::new(state.output_schema.clone()); let (family, update, grouping) = match &state.payload { - ExecutableOperatorPayload::SummaryAgg { + PostAsapOperatorPayload::SummaryAgg { family, input, grouping, @@ -383,7 +383,7 @@ fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { .find(|node| { matches!( &node.payload, - ExecutableOperatorPayload::SummaryAgg { + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), .. } @@ -417,11 +417,11 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { let mut executed = 0; for forest in inventory.candidates { let root = &forest[0].1; - let dag = compile_executable_dag(root).unwrap(); + let dag = compile_post_asap_dag(root).unwrap(); let Some(state) = dag.nodes.iter().find(|node| { matches!( node.payload, - ExecutableOperatorPayload::SummaryAgg { + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), .. } @@ -435,7 +435,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { .find(|node| { matches!( node.payload, - ExecutableOperatorPayload::SummaryAgg { + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), .. } @@ -453,7 +453,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { &[vec![], vec![boundary]], ); let (family, input, grouping) = match &state.payload { - ExecutableOperatorPayload::SummaryAgg { + PostAsapOperatorPayload::SummaryAgg { family, input, grouping, diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs index e860f07c..30a9bb64 100644 --- a/crates/asap-physical-operators/tests/precompute_population.rs +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -38,25 +38,25 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { (SummaryInputExpr::Constant(1.), 4.), ] { let nodes = vec![ - ExecutableDagNode { + PostAsapDagNode { id: PostAsapNodeId(0), - payload: ExecutableOperatorPayload::SummaryMerge, + payload: PostAsapOperatorPayload::SummaryMerge, output_state: ExecutionDataState::INGESTION_SUMMARY, output_schema: state_schema.clone(), guarantee: None, }, - ExecutableDagNode { + PostAsapDagNode { id: PostAsapNodeId(1), - payload: ExecutableOperatorPayload::Value { + payload: PostAsapOperatorPayload::Value { operation: ValueOperation::FinalizeExactAccumulator, }, output_state: ExecutionDataState::INGESTION_ROWS, output_schema: value_schema.clone(), guarantee: None, }, - ExecutableDagNode { + PostAsapDagNode { id: PostAsapNodeId(2), - payload: ExecutableOperatorPayload::Binary { + payload: PostAsapOperatorPayload::Binary { operator: BinaryOperator { kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), vector_match: None, @@ -68,9 +68,9 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { output_schema: value_schema.clone(), guarantee: None, }, - ExecutableDagNode { + PostAsapDagNode { id: PostAsapNodeId(3), - payload: ExecutableOperatorPayload::SummaryAgg { + payload: PostAsapOperatorPayload::SummaryAgg { family: family.clone(), input: SummaryUpdate { weight, @@ -91,7 +91,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { (2, 3, EdgeRole::Input), ] .into_iter() - .map(|(producer, consumer, role)| ExecutableDagEdge { + .map(|(producer, consumer, role)| PostAsapDagEdge { producer: PostAsapNodeId(producer), consumer: PostAsapNodeId(consumer), role, @@ -101,13 +101,13 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { window: WindowEdgeCompatibility::NotApplicable, }) .collect(); - let dag = ExecutableDag { + let dag = PostAsapDag { nodes, edges, root: PostAsapNodeId(3), }; let mut invalid_grouping = dag.clone(); - let ExecutableOperatorPayload::SummaryAgg { reduction, .. } = + let PostAsapOperatorPayload::SummaryAgg { reduction, .. } = &mut invalid_grouping.nodes[3].payload else { unreachable!() @@ -202,24 +202,24 @@ fn state_graph( target: Option, merge: bool, ) -> CompiledPhysicalDag { - let mut nodes = vec![ExecutableDagNode { + let mut nodes = vec![PostAsapDagNode { id: PostAsapNodeId(0), - payload: ExecutableOperatorPayload::SummaryMerge, + payload: PostAsapOperatorPayload::SummaryMerge, output_state: ExecutionDataState::INGESTION_SUMMARY, output_schema: logical_schema(family.clone()), guarantee: None, }]; if merge { - nodes.push(ExecutableDagNode { + nodes.push(PostAsapDagNode { id: PostAsapNodeId(1), - payload: ExecutableOperatorPayload::SummaryMerge, + payload: PostAsapOperatorPayload::SummaryMerge, ..nodes[0].clone() }); } let read_id = nodes.len() as u32; - nodes.push(ExecutableDagNode { + nodes.push(PostAsapDagNode { id: PostAsapNodeId(read_id), - payload: ExecutableOperatorPayload::Value { + payload: PostAsapOperatorPayload::Value { operation: ValueOperation::FinalizeExactAccumulator, }, output_state: ExecutionDataState::INGESTION_ROWS, @@ -227,9 +227,9 @@ fn state_graph( guarantee: None, }); if let Some(target) = target { - nodes.push(ExecutableDagNode { + nodes.push(PostAsapDagNode { id: PostAsapNodeId(nodes.len() as u32), - payload: ExecutableOperatorPayload::SummaryAgg { + payload: PostAsapOperatorPayload::SummaryAgg { family: target.clone(), input: SummaryUpdate::column(ColumnRef::SampleValue), reduction: Reduction::by(vec![]), @@ -241,7 +241,7 @@ fn state_graph( }); } let edges = (1..nodes.len()) - .map(|i| ExecutableDagEdge { + .map(|i| PostAsapDagEdge { producer: nodes[i - 1].id, consumer: nodes[i].id, role: EdgeRole::Input, @@ -253,7 +253,7 @@ fn state_graph( .collect(); let root = nodes.last().unwrap().id; precompute::compile( - &ExecutableDag { nodes, edges, root }, + &PostAsapDag { nodes, edges, root }, &[0], &[u64::from(root.0)], ) diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs index 9bc51ac9..ca0d6d7e 100644 --- a/crates/asap-physical-operators/tests/promql_binary.rs +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -8,8 +8,8 @@ use asap_physical_operators::{ use futures::{executor::block_on, StreamExt}; use planner_types::{ post_asap::{ - BinaryOperator, ExecutableDagNode, ExecutableOperatorPayload, ExecutionDataState, - PostAsapNodeId, SummaryFamilyType, SummaryField, SummarySchema, + BinaryOperator, ExecutionDataState, PostAsapDagNode, PostAsapNodeId, + PostAsapOperatorPayload, SummaryFamilyType, SummaryField, SummarySchema, }, pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, }; @@ -50,9 +50,9 @@ fn row(name: &str, job: &str, value: f64) -> Vec { } fn program() -> CompiledPhysicalDag { let schema = schema(); - let node = ExecutableDagNode { + let node = PostAsapDagNode { id: PostAsapNodeId(2), - payload: ExecutableOperatorPayload::Binary { + payload: PostAsapOperatorPayload::Binary { operator: BinaryOperator { kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), vector_match: None, diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs index 25851c4f..78e4c8c0 100644 --- a/crates/asap-physical-operators/tests/raw_scan.rs +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -53,15 +53,15 @@ fn fixture() -> (QueryExpr, Schema, Vec) { ]; (scan, output, batches) } -fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> ExecutableDag { - let node = |id, payload| ExecutableDagNode { +fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsapDag { + let node = |id, payload| PostAsapDagNode { id: PostAsapNodeId(id), payload, output_state: state, output_schema: (**schema).clone(), guarantee: None, }; - let edge = |producer, consumer| ExecutableDagEdge { + let edge = |producer, consumer| PostAsapDagEdge { producer: PostAsapNodeId(producer), consumer: PostAsapNodeId(consumer), role: EdgeRole::Input, @@ -70,12 +70,12 @@ fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> Executab grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }; - ExecutableDag { + PostAsapDag { nodes: vec![ - node(0, ExecutableOperatorPayload::Fallback { expression: scan }), + node(0, PostAsapOperatorPayload::Fallback { expression: scan }), node( 1, - ExecutableOperatorPayload::Value { + PostAsapOperatorPayload::Value { operation: ValueOperation::Sort { keys: vec![planner_types::pre_asap::SortKey { expr: QueryExpr::Column(0), @@ -88,7 +88,7 @@ fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> Executab ), node( 2, - ExecutableOperatorPayload::Value { + PostAsapOperatorPayload::Value { operation: ValueOperation::Limit { n: 2, offset: 0, diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs index f8a2acbe..2881cf7e 100644 --- a/crates/asap-physical-operators/tests/summary_projection.rs +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -44,18 +44,18 @@ fn post_asap_summary_projection_survives_recovery() { ], time_index: None, }; - let dag = ExecutableDag { + let dag = PostAsapDag { nodes: vec![ - ExecutableDagNode { + PostAsapDagNode { id: PostAsapNodeId(0), - payload: ExecutableOperatorPayload::SummaryMerge, + payload: PostAsapOperatorPayload::SummaryMerge, output_schema: (*schema).clone(), output_state: ExecutionDataState::INGESTION_SUMMARY, guarantee: None, }, - ExecutableDagNode { + PostAsapDagNode { id: PostAsapNodeId(1), - payload: ExecutableOperatorPayload::Value { + payload: PostAsapOperatorPayload::Value { operation: ValueOperation::Project { cols: vec![1, 0] .into_iter() @@ -72,7 +72,7 @@ fn post_asap_summary_projection_survives_recovery() { guarantee: None, }, ], - edges: vec![ExecutableDagEdge { + edges: vec![PostAsapDagEdge { producer: PostAsapNodeId(0), consumer: PostAsapNodeId(1), role: EdgeRole::Input, diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index e4be66c1..139d4596 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -88,8 +88,8 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S _ => None, }) .unwrap(); - let dag = compile_executable_dag(&plan).unwrap(); - let build=dag.nodes.iter().find(|node|matches!(&node.payload,ExecutableOperatorPayload::SummaryAgg{family:SummaryFamilyType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); + let dag = compile_post_asap_dag(&plan).unwrap(); + let build=dag.nodes.iter().find(|node|matches!(&node.payload,PostAsapOperatorPayload::SummaryAgg{family:SummaryFamilyType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); let rate_id = dag .edges .iter() @@ -352,11 +352,11 @@ fn check_direct_rate_topk(dynamic: bool) { "Rate must be supplied by its exact stored-state readout" ); } - let dag = compile_executable_dag(candidate).unwrap(); + let dag = compile_post_asap_dag(candidate).unwrap(); assert!(dag.nodes.iter().any(|node| matches!(&node.payload, - ExecutableOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); let build = dag.nodes.iter().find(|node| matches!(&node.payload, - ExecutableOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); let input_id = dag .edges .iter() @@ -377,7 +377,7 @@ fn check_direct_rate_topk(dynamic: bool) { .find(|node| { matches!( &node.payload, - ExecutableOperatorPayload::Fallback { + PostAsapOperatorPayload::Fallback { expression: QueryExpr::TimeRange { .. } } ) @@ -653,14 +653,14 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { _ => None, }) .expect("signed spatial TopK must expose CountSketch with heap"); - let dag = compile_executable_dag(selected).unwrap(); + let dag = compile_post_asap_dag(selected).unwrap(); let raw = dag .nodes .iter() .find(|node| { matches!( &node.payload, - ExecutableOperatorPayload::Fallback { + PostAsapOperatorPayload::Fallback { expression: QueryExpr::TimeRange { .. } } ) @@ -779,14 +779,14 @@ fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { let Replacement::Summary(root) = candidate.replacement else { panic!() }; - let dag = compile_executable_dag(&root).unwrap(); + let dag = compile_post_asap_dag(&root).unwrap(); let state = dag .nodes .iter() .find(|node| { matches!( &node.payload, - ExecutableOperatorPayload::SummaryAgg { + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), .. } @@ -799,7 +799,7 @@ fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { .find(|node| { matches!( &node.payload, - ExecutableOperatorPayload::SummaryAgg { + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(..), .. } @@ -850,7 +850,7 @@ fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { }) }; let (family, input, grouping) = match &state.payload { - ExecutableOperatorPayload::SummaryAgg { + PostAsapOperatorPayload::SummaryAgg { family, input, grouping, diff --git a/crates/integration-tests/tests/sql_to_physical.rs b/crates/integration-tests/tests/sql_to_physical.rs index 22cb92a4..be96107e 100644 --- a/crates/integration-tests/tests/sql_to_physical.rs +++ b/crates/integration-tests/tests/sql_to_physical.rs @@ -8,7 +8,7 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use asap_types::{ - post_asap::{compile_executable_dag, ExecutableOperatorPayload, SummaryFamilyType}, + post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, pre_asap::{Column, DataType, QueryExpr, Schema}, types::AccuracyTarget, }; @@ -41,14 +41,14 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let dag = compile_executable_dag(&selected).unwrap(); + let dag = compile_post_asap_dag(&selected).unwrap(); let scan = dag .nodes .iter() .find(|node| { matches!( &node.payload, - ExecutableOperatorPayload::Fallback { + PostAsapOperatorPayload::Fallback { expression: QueryExpr::Scan { .. } } ) @@ -88,7 +88,7 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { .collect() }) .collect(); - let ExecutableOperatorPayload::Fallback { expression } = &scan.payload else { + let PostAsapOperatorPayload::Fallback { expression } = &scan.payload else { unreachable!() }; let QueryExpr::Scan { source, .. } = expression else { diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index 61003a50..17b60cb6 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -265,7 +265,7 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { values::{Batch, Value}, }; use asap_types::{ - post_asap::{compile_executable_dag, ExecutableOperatorPayload, SummaryFamilyType}, + post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, pre_asap::DataType, }; use std::{collections::BTreeMap, sync::Arc}; @@ -283,11 +283,11 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { .summary_maintenance_lifecycle, SummaryMaintenanceLifecycle::ContinuouslyMaintained ); - let dag = compile_executable_dag(&selected.root).unwrap(); + let dag = compile_post_asap_dag(&selected.root).unwrap(); let build = dag .nodes .iter() - .find(|node| matches!(node.payload, ExecutableOperatorPayload::SummaryAgg { .. })) + .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) .unwrap(); let input = dag .edges @@ -506,7 +506,7 @@ fn selected_temporal_lifecycle_compiles_panes_and_executes() { }; use asap_types::{ post_asap::{ - compile_executable_dag, plan_pane_phase, ExecutableOperatorPayload, SummaryFamilyType, + compile_post_asap_dag, plan_pane_phase, PostAsapOperatorPayload, SummaryFamilyType, SummaryWindowFramework, }, pre_asap::DataType, @@ -541,17 +541,17 @@ fn selected_temporal_lifecycle_compiles_panes_and_executes() { deployment.selected_window_framework, Some(SummaryWindowFramework::Sliding) ); - let dag = compile_executable_dag(&plan.root).unwrap(); + let dag = compile_post_asap_dag(&plan.root).unwrap(); let build = dag .nodes .iter() - .find(|node| matches!(node.payload, ExecutableOperatorPayload::SummaryAgg { .. })) + .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) .unwrap(); assert_eq!(build.id, deployment.post_asap_node_id); let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, ExecutableOperatorPayload::Fallback { .. })) + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) .unwrap(); let schema = Arc::new(raw.output_schema.clone()); let width = workload diff --git a/crates/types/src/post_asap/semantic_definition.rs b/crates/types/src/post_asap/semantic_definition.rs index 1337222f..f8014422 100644 --- a/crates/types/src/post_asap/semantic_definition.rs +++ b/crates/types/src/post_asap/semantic_definition.rs @@ -1,7 +1,7 @@ //! Persistable dependency closure using Planner's typed operation vocabulary. //! Node hashes are local semantic references, not executable or deployed IDs. use crate::post_asap::{ - EdgeRole, ExecutableDag, ExecutableOperatorPayload, PostAsapNodeId, SummarySchema, + EdgeRole, PostAsapDag, PostAsapNodeId, PostAsapOperatorPayload, SummarySchema, }; use serde::{Deserialize, Serialize}; use sha2::{Digest, Sha256}; @@ -90,13 +90,13 @@ fn role(role: EdgeRole) -> u8 { } impl SummarySemanticFragment { - pub fn from_stored_output(dag: &ExecutableDag, output: PostAsapNodeId) -> Result { + pub fn from_stored_output(dag: &PostAsapDag, output: PostAsapNodeId) -> Result { Self::export(dag, output, true) } /// All source names in this DAG resolve within this logical dataset. pub fn from_stored_output_in_dataset( - dag: &ExecutableDag, + dag: &PostAsapDag, output: PostAsapNodeId, dataset: LogicalDatasetIdentity, ) -> Result { @@ -108,12 +108,12 @@ impl SummarySemanticFragment { Ok(fragment) } - pub fn from_dag(dag: &ExecutableDag, output: PostAsapNodeId) -> Result { + pub fn from_dag(dag: &PostAsapDag, output: PostAsapNodeId) -> Result { Self::export(dag, output, false) } fn export( - dag: &ExecutableDag, + dag: &PostAsapDag, output: PostAsapNodeId, parameterize_range: bool, ) -> Result { @@ -129,7 +129,7 @@ impl SummarySemanticFragment { ); } } - let dag = ExecutableDag { + let dag = PostAsapDag { nodes: dag .nodes .iter() @@ -157,7 +157,7 @@ impl SummarySemanticFragment { .find(|n| n.id == output) .ok_or("missing output")? .payload, - ExecutableOperatorPayload::SummaryAgg { + PostAsapOperatorPayload::SummaryAgg { reduction: crate::pre_asap::Reduction::PerEntity, grouping: crate::post_asap::GroupingStrategy::PerSubpopulationInstance, input: crate::post_asap::SummaryUpdate { @@ -182,7 +182,7 @@ impl SummarySemanticFragment { if !direct.contains(&node.id) { continue; } - if let ExecutableOperatorPayload::Fallback { expression } = &mut node.payload { + if let PostAsapOperatorPayload::Fallback { expression } = &mut node.payload { let source = match expression { crate::pre_asap::QueryExpr::TimeRange { child, .. } => { std::rc::Rc::make_mut(child) @@ -281,24 +281,24 @@ impl SummarySemanticFragment { if parameterize_range && matches!( nodes[&output].payload, - ExecutableOperatorPayload::SummaryAgg { .. } + PostAsapOperatorPayload::SummaryAgg { .. } ) && dag .edges .iter() .any(|e| e.consumer == output && e.producer == id) { - if let ExecutableOperatorPayload::Fallback { + if let PostAsapOperatorPayload::Fallback { expression: crate::pre_asap::QueryExpr::TimeRange { child, .. }, } = &payload { - payload = ExecutableOperatorPayload::Fallback { + payload = PostAsapOperatorPayload::Fallback { expression: child.as_ref().clone(), }; record_range = true; } } - if let ExecutableOperatorPayload::RelationalJoin { pruning, .. } = &mut payload { + if let PostAsapOperatorPayload::RelationalJoin { pruning, .. } = &mut payload { *pruning = None; } let operation = SemanticOperation { @@ -333,17 +333,17 @@ impl SummarySemanticFragment { return Err("unsupported semantic fragment version or size".into()); } for (key, node) in &self.nodes { - let payload: ExecutableOperatorPayload = + let payload: PostAsapOperatorPayload = serde_json::from_value(node.operation.clone()).map_err(|e| e.to_string())?; if node.record_range { let root = self .nodes .get(&self.output) .ok_or("missing semantic root")?; - let root_payload: ExecutableOperatorPayload = + let root_payload: PostAsapOperatorPayload = serde_json::from_value(root.operation.clone()).map_err(|e| e.to_string())?; - if !matches!(payload, ExecutableOperatorPayload::Fallback { .. }) - || !matches!(root_payload, ExecutableOperatorPayload::SummaryAgg { .. }) + if !matches!(payload, PostAsapOperatorPayload::Fallback { .. }) + || !matches!(root_payload, PostAsapOperatorPayload::SummaryAgg { .. }) || !root.inputs.iter().any(|input| &input.node == key) { return Err("record range must belong to a direct summary input".into()); @@ -385,11 +385,11 @@ impl SummarySemanticFragment { #[cfg(test)] mod tests { use super::*; - use crate::post_asap::{compile_executable_dag, SummaryExpr, SummaryNode}; + use crate::post_asap::{compile_post_asap_dag, SummaryExpr, SummaryNode}; use crate::pre_asap::{Column, DataType, QueryExpr, Schema, Source}; use std::rc::Rc; - fn fixture(metric: &str) -> ExecutableDag { + fn fixture(metric: &str) -> PostAsapDag { let scan = QueryExpr::Scan { source: Source::TimeSeries { metric: metric.into(), @@ -405,7 +405,7 @@ mod tests { }], time_index: None, }; - compile_executable_dag(&Rc::new(SummaryNode { + compile_post_asap_dag(&Rc::new(SummaryNode { expr: SummaryExpr::KeepPreAsap(Rc::new(scan)), schema, guarantee: None, @@ -483,7 +483,7 @@ mod tests { let original = fixture("latency"); let expected = SummarySemanticFragment::from_dag(&original, original.root).unwrap(); let mut transformed = original.clone(); - let ExecutableOperatorPayload::Fallback { expression } = &mut transformed.nodes[0].payload + let PostAsapOperatorPayload::Fallback { expression } = &mut transformed.nodes[0].payload else { unreachable!() }; @@ -503,7 +503,7 @@ mod tests { canonical_bytes(&expected).unwrap(), canonical_bytes(&logged).unwrap() ); - let ExecutableOperatorPayload::Fallback { + let PostAsapOperatorPayload::Fallback { expression: QueryExpr::Project { cols, .. }, } = &mut transformed.nodes[0].payload else { @@ -520,13 +520,13 @@ mod tests { let stored = dag.root; let mut consumer = dag.nodes[0].clone(); consumer.id = PostAsapNodeId(9); - consumer.payload = ExecutableOperatorPayload::Value { + consumer.payload = PostAsapOperatorPayload::Value { operation: crate::post_asap::ValueOperation::Project { cols: vec![], qualifier: None, }, }; - dag.edges.push(crate::post_asap::ExecutableDagEdge { + dag.edges.push(crate::post_asap::PostAsapDagEdge { producer: stored, consumer: consumer.id, role: EdgeRole::Input, @@ -550,8 +550,7 @@ mod tests { use crate::pre_asap::{ColumnRef, Reduction}; let make = |seconds| { let mut dag = fixture("latency"); - let ExecutableOperatorPayload::Fallback { expression } = &mut dag.nodes[0].payload - else { + let PostAsapOperatorPayload::Fallback { expression } = &mut dag.nodes[0].payload else { unreachable!() }; *expression = QueryExpr::TimeRange { @@ -561,7 +560,7 @@ mod tests { let mut output = dag.nodes[0].clone(); output.id = PostAsapNodeId(1); let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - output.payload = ExecutableOperatorPayload::SummaryAgg { + output.payload = PostAsapOperatorPayload::SummaryAgg { family: family.clone(), input: SummaryUpdate { item: None, @@ -573,7 +572,7 @@ mod tests { }; output.output_schema.fields[0].dtype = family; output.output_state.primitive = DataPrimitive::SummaryState; - dag.edges.push(ExecutableDagEdge { + dag.edges.push(PostAsapDagEdge { producer: dag.root, consumer: output.id, role: EdgeRole::Input, @@ -599,7 +598,7 @@ mod tests { // Open entities retain their full label identity; consumer-demanded // optional labels do not change the per-entity stored computation. let mut open = one.clone(); - let ExecutableOperatorPayload::Fallback { + let PostAsapOperatorPayload::Fallback { expression: QueryExpr::TimeRange { child, .. }, } = &mut open.nodes[0].payload else { @@ -610,7 +609,7 @@ mod tests { }; schema.closed = false; let expected = SummarySemanticFragment::from_stored_output(&open, open.root).unwrap(); - let ExecutableOperatorPayload::Fallback { + let PostAsapOperatorPayload::Fallback { expression: QueryExpr::TimeRange { child, .. }, } = &mut open.nodes[0].payload else { diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 73180984..492ac283 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -28,6 +28,15 @@ implementation library. Deployment systems such as ASAPQuery and asap-fusion own deployment compilation and operation. The lifecycle is a planning contract associated with the logical DAG, not a separate computation IR. +The Logical Post-ASAP DAG is preceded by the Pre-ASAP DAG (`QueryExpr`), the +language-independent query semantics before summary selection. Both are +logical. Planning builds Post-ASAP `SummaryNode` trees; `compile_post_asap_dag` +exports the selected tree as a `PostAsapDag`, which is the Physical Plan +Compiler's input. Its per-node execution phase (ingestion or query time) is an +initial placement: compilation places ingestion-time nodes in the precompute DAG, +while frontier enumeration proposes alternative materialization splits. Which +layer owns placement is an open design question, deferred to a later change. + ### Candidate generation and deployment selection Planner exposes the supported, semantically legal **physical plan candidates**. @@ -272,7 +281,7 @@ The **Physical Plan Compiler** consumes both computation semantics and maintenan requirements: ```text -Logical Post-ASAP DAG +Logical Post-ASAP DAG (PostAsapDag) + Summary Maintenance Lifecycle + physical capabilities ↓