diff --git a/README.md b/README.md index 3c21bd7d..93c6127b 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ # ASAPPlanner -ASAPPlanner turns SQL, PromQL, and MetricsQL query workloads into legal candidate plans that may use Approximate Streaming Analytics Primitives (ASAPs), such as sketches and exact summaries. It normalizes language-specific queries into a shared representation, then enumerates and ranks semantically equivalent alternatives. Downstream systems choose, deploy, and execute a physical plan. +ASAPPlanner turns SQL, PromQL, and MetricsQL query workloads into legal candidate plans that may use Approximate Streaming Analytics Primitives (ASAPs), such as sketches and exact summaries. It normalizes language-specific queries into a shared representation, then enumerates and ranks semantically equivalent alternatives. Planner selects one physical plan using the deployment's cost model; the deployment executes it. ## Start here diff --git a/crates/asap-aware-mapping/src/accuracy/evidence.rs b/crates/asap-aware-mapping/src/accuracy/evidence.rs index e73a1fc8..5321cc9c 100644 --- a/crates/asap-aware-mapping/src/accuracy/evidence.rs +++ b/crates/asap-aware-mapping/src/accuracy/evidence.rs @@ -91,7 +91,7 @@ pub trait AccuracyEvidenceProvider { /// An observed cardinality is not an enforced population bound. fn estimator_contract( &self, - _expression: &asap_types::pre_asap::QueryExpr, + _expression: &asap_types::pre_asap::PreASAPNode, ) -> Option { None } @@ -101,7 +101,7 @@ pub trait AccuracyEvidenceProvider { /// selected candidates. Observed cardinality is not sufficient evidence. fn topk_max_distinct_items( &self, - _expression: &asap_types::pre_asap::QueryExpr, + _expression: &asap_types::pre_asap::PreASAPNode, ) -> Option { None } @@ -110,7 +110,7 @@ pub trait AccuracyEvidenceProvider { /// filters, grouping and window. `None` means unknown, including emptiness. fn quantile_input_domain( &self, - _operand: &asap_types::pre_asap::query_expr::QueryExpr, + _operand: &asap_types::pre_asap::query_expr::PreASAPNode, ) -> Option { None } diff --git a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs index 5b2688e6..fa1f1ec8 100644 --- a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs +++ b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs @@ -28,7 +28,7 @@ //! //! ## What counts as a "near-duplicate", and why //! -//! Two [`QueryExpr::Aggregate`] nodes are accuracy-near-duplicates here iff, +//! Two [`PreASAPNode::Aggregate`] nodes are accuracy-near-duplicates here iff, //! **in this order**: //! //! 1. Both are the same bindable shape [`crate::replacement::SketchAlgorithmStrategy`] @@ -141,7 +141,7 @@ //! priced with, reflecting "one more reference into a structure that's //! already being maintained" rather than "build a whole new one." //! -//! `CandidatePostASAPDAGs::global_selection` treats this rewrite as a cross-group edge: +//! `CandidateLogicalPostASAPDAGs::global_selection` treats this rewrite as a cross-group edge: //! selecting it increments `rc`'s own `effective_consumer_count`, then lets //! that sibling group propagate the uses through its selected implementation. //! Accuracy edges are directed strictly from looser to tighter budgets, so @@ -153,7 +153,7 @@ use std::cmp::Ordering; use std::rc::Rc; use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; +use asap_types::pre_asap::query_expr::{PreASAPNode, Reduction}; use asap_types::types::AccuracyTarget; use crate::replacement::{ @@ -169,7 +169,7 @@ type BindableAccuracyAggregate<'a> = ( &'a AggIntent, &'a AccuracyTarget, &'a [String], - &'a Rc, + &'a Rc, ); /// The `(reduction, intent, accuracy, output_names, child)` shape this @@ -181,8 +181,8 @@ type BindableAccuracyAggregate<'a> = ( /// `Quantile` / `Cardinality` / `TopK`). `None` for anything else, including /// a multi-measure or `HAVING` aggregate, a non-`Aggregate` node, or an /// accuracy-free intent (`Sum`, `Avg`, …). -fn bindable_accuracy_aggregate(node: &QueryExpr) -> Option> { - let QueryExpr::Aggregate { +fn bindable_accuracy_aggregate(node: &PreASAPNode) -> Option> { + let PreASAPNode::Aggregate { reduction, measures, output_names, @@ -274,7 +274,7 @@ fn strictly_tighter(a: &AccuracyTarget, b: &AccuracyTarget) -> bool { /// this strategy from the same post-CSE `Aggregate` sibling set it already /// builds for `RollupStrategy`. pub struct AccuracyReconciliationStrategy { - siblings: Vec>, + siblings: Vec>, } impl AccuracyReconciliationStrategy { @@ -282,7 +282,7 @@ impl AccuracyReconciliationStrategy { /// each as a candidate tighter-accuracy source (or looser-accuracy /// target) — typically the full set of `Aggregate` nodes a workload-wide /// discovery pass already found. - pub fn new(siblings: &[Rc]) -> Self { + pub fn new(siblings: &[Rc]) -> Self { Self { siblings: siblings.to_vec(), } @@ -307,14 +307,14 @@ impl AccuracyReconciliationStrategy { /// reports no unique key — see `cse.rs`'s "Legality" section) would get /// proposed for reconciliation even though nothing guarantees a second /// read of it lines up row-for-row with the first. - fn tighter_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { + fn tighter_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { let Some((target_reduction, target_intent, target_accuracy, target_names, target_child)) = bindable_accuracy_aggregate(target.root) else { return Vec::new(); }; - let mut sources: Vec<&Rc> = self + let mut sources: Vec<&Rc> = self .siblings .iter() .filter(|candidate| { @@ -397,8 +397,8 @@ mod tests { /// willing to hoist it — see `Schema::has_unique_key`/`cse.rs`'s own /// "Legality" section: a producer with no provable unique key is always /// inserted fresh, never hoisted, regardless of structural equality. - fn metric_scan() -> Rc { - Rc::new(QueryExpr::Scan { + fn metric_scan() -> Rc { + Rc::new(PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -413,8 +413,8 @@ mod tests { }) } - fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { + Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], @@ -423,7 +423,7 @@ mod tests { }) } - fn quantile(q: f64, accuracy: AccuracyTarget, child: &Rc) -> Rc { + fn quantile(q: f64, accuracy: AccuracyTarget, child: &Rc) -> Rc { agg( vec![2], AggIntent::Quantile { @@ -439,7 +439,11 @@ mod tests { /// reports no unique key for an empty `by` (see `query_expr.rs`'s own /// `unique_keys = if by.is_empty() || has_count_values { vec![] } else /// { .. }`). - fn global_quantile(q: f64, accuracy: AccuracyTarget, child: &Rc) -> Rc { + fn global_quantile( + q: f64, + accuracy: AccuracyTarget, + child: &Rc, + ) -> Rc { agg( vec![], AggIntent::Quantile { @@ -458,9 +462,9 @@ mod tests { q: f64, accuracy: AccuracyTarget, excluded: Vec, - child: &Rc, - ) -> Rc { - Rc::new(QueryExpr::Aggregate { + child: &Rc, + ) -> Rc { + Rc::new(PreASAPNode::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(excluded)), measures: vec![AggIntent::Quantile { col: None, @@ -678,10 +682,10 @@ mod tests { // The identical scan child, though, is still shared exactly as // before — this module changes nothing about that. - let QueryExpr::Aggregate { child: child_a, .. } = roots[0].1.as_ref() else { + let PreASAPNode::Aggregate { child: child_a, .. } = roots[0].1.as_ref() else { panic!("expected an Aggregate root"); }; - let QueryExpr::Aggregate { child: child_b, .. } = roots[1].1.as_ref() else { + let PreASAPNode::Aggregate { child: child_b, .. } = roots[1].1.as_ref() else { panic!("expected an Aggregate root"); }; assert!(Rc::ptr_eq(child_a, child_b)); diff --git a/crates/asap-aware-mapping/src/analytical_cost.rs b/crates/asap-aware-mapping/src/analytical_cost.rs index ed2a316d..cd692b41 100644 --- a/crates/asap-aware-mapping/src/analytical_cost.rs +++ b/crates/asap-aware-mapping/src/analytical_cost.rs @@ -3010,7 +3010,7 @@ mod tests { fn comparison_rejects_different_snapshot_predicate_time_or_horizon() { use std::rc::Rc; - use asap_types::pre_asap::query_expr::{Predicate, QueryExpr}; + use asap_types::pre_asap::query_expr::{PreASAPNode, Predicate}; use asap_types::workload::{DurationMs, TimestampMs}; let raw = comparison_scope(); @@ -3026,7 +3026,7 @@ mod tests { candidate = raw.clone(); candidate.sources[0] .predicates - .push(Predicate(Rc::new(QueryExpr::promql_scalar(1.0)))); + .push(Predicate(Rc::new(PreASAPNode::promql_scalar(1.0)))); assert_eq!( validate_comparison_scopes(&raw, &candidate), Err(AnalyticalCostError::ComparisonScopeMismatch("sources")) diff --git a/crates/asap-aware-mapping/src/candidate_timing.rs b/crates/asap-aware-mapping/src/candidate_timing.rs new file mode 100644 index 00000000..ec1b485c --- /dev/null +++ b/crates/asap-aware-mapping/src/candidate_timing.rs @@ -0,0 +1,325 @@ +//! Timed stage of the candidate collection; lifecycle machinery stays internal. +use std::rc::Rc; + +use crate::{ + cost_model::CostModel, + replacement::{CandidateLogicalPostASAPDAGs, RealizationError}, + summary_maintenance_lifecycle::{ + enumerate_summary_maintenance_lifecycles, SummaryMaintenanceLifecycleCandidates, + }, + Horizon, LifecyclePostASAPDAG, LifecyclePostASAPDAGError, SummaryMaintenanceDeployment, + SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecycleChoiceError, + SummaryMaintenanceTimingError, WorkloadDemand, +}; +use asap_types::post_asap::{ + index_post_asap_dag, ExecutionDataStateError, LogicalPostASAPDAG, LogicalPostASAPDAGAssignment, + LogicalPostASAPDAGIndex, PostASAPNodeId, SummaryMaintenanceLifecycle, + SummaryMaintenanceLifecycleGuarantee, +}; + +/// Demand and evidence bound to the requested workload root, not unrelated queries. +pub struct CandidateTimingContext<'a> { + pub demand: WorkloadDemand<'a>, + pub now_ms: u64, + pub horizon: Option, + pub capabilities: SummaryMaintenanceLifecycleCapabilities, + pub cost_model: &'a dyn CostModel, +} + +#[derive(Debug, thiserror::Error)] +pub enum CandidateTimingError { + #[error(transparent)] + Logical(#[from] RealizationError), + #[error(transparent)] + Lifecycle(#[from] LifecyclePostASAPDAGError), + #[error(transparent)] + Choice(#[from] SummaryMaintenanceLifecycleChoiceError), + #[error(transparent)] + Timing(#[from] SummaryMaintenanceTimingError), + #[error(transparent)] + Graph(#[from] ExecutionDataStateError), + #[error("timed candidate expansion exceeds limit {0}")] + ExpansionLimit(usize), + #[error("unknown logical candidate {0}")] + UnknownLogicalCandidate(usize), +} + +struct PreparedTiming<'a> { + index: Rc, + lifecycles: SummaryMaintenanceLifecycleCandidates<'a>, + count: usize, +} + +/// Every lifecycle assignment of one workload root's logical candidates, each a +/// [`LifecyclePostASAPDAG`]. It retains shared graph indices and factored +/// choices, never a Cartesian-product vector of timed graphs, and it does not +/// select an assignment. +pub struct CandidateLifecyclePostASAPDAGs<'a, Id> { + id: Id, + logical: Vec, Rc>>, + rejected_assemblies: Vec, + count: usize, +} + +/// Candidate identity and lifecycle evidence survive both timing and compilation +/// failures. `lifecycle` is `None` when lifecycle binding failed, not an +/// implicit default. +#[derive(Debug)] +pub struct PostASAPCandidateMetadata { + pub id: Id, + pub logical_candidate: usize, + pub assignment_candidate: usize, + pub choices: Vec<(PostASAPNodeId, SummaryMaintenanceLifecycle)>, + pub lifecycle: Option, +} + +impl CandidateLogicalPostASAPDAGs { + /// Attach every lifecycle alternative for this root's logical candidates. + /// Both budgets are checked before a collection can be iterated. Whole- + /// workload selection must still coordinate choices and shared state. + pub fn with_timing_for_root<'a>( + &self, + id: &Id, + context: CandidateTimingContext<'a>, + logical_limit: usize, + assignment_limit: usize, + ) -> Result, CandidateTimingError> { + let inventory = self + .enumerate_candidate_dags_for_root(id, logical_limit) + .map_err(|error| match error { + RealizationError::ExpansionLimit(limit) => { + CandidateTimingError::ExpansionLimit(limit) + } + error => error.into(), + })?; + let roots = inventory.candidates.into_iter().map(|mut roots| { + // The logical enumerator was explicitly scoped to this one root. + debug_assert_eq!(roots.len(), 1); + roots.remove(0).1 + }); + prepare( + id.clone(), + roots, + inventory.rejected_assemblies, + context, + assignment_limit, + ) + } +} + +impl<'a, Id> CandidateLifecyclePostASAPDAGs<'a, Id> { + /// Enter the same timed collection API when a caller already has one logical + /// graph. Explicit selection helpers do not need another lifecycle API type. + pub fn from_post_asap_dag( + id: Id, + root: LogicalPostASAPDAG, + context: CandidateTimingContext<'a>, + assignment_limit: usize, + ) -> Result { + prepare(id, [root], Vec::new(), context, assignment_limit) + } +} + +fn prepare<'a, Id>( + id: Id, + roots: impl IntoIterator, + rejected_assemblies: Vec, + context: CandidateTimingContext<'a>, + limit: usize, +) -> Result, CandidateTimingError> { + let enumerated = roots.into_iter().map(|root| { + let index = Rc::new(index_post_asap_dag(&root)?); + let lifecycles = enumerate_summary_maintenance_lifecycles( + root, + context.demand, + context.now_ms, + context.horizon, + context.capabilities, + context.cost_model, + )?; + Ok((index, lifecycles)) + }); + collect_timing(id, enumerated, rejected_assemblies, limit) +} + +/// Collect each logical candidate's enumerated lifecycles into the timed collection. +pub(crate) fn collect_timing<'a, Id>( + id: Id, + enumerated: impl IntoIterator< + Item = Result< + ( + Rc, + SummaryMaintenanceLifecycleCandidates<'a>, + ), + CandidateTimingError, + >, + >, + rejected_assemblies: Vec, + limit: usize, +) -> Result, CandidateTimingError> { + let mut logical = Vec::new(); + let mut count = 0usize; + for prepared in enumerated { + let prepared = match prepared + .and_then(|(index, lifecycles)| prepared_timing(index, lifecycles, limit)) + { + // Budget failures are collection failures, never rejected choices + // inside a deceptively complete partial collection. + Err(error @ CandidateTimingError::ExpansionLimit(_)) => return Err(error), + prepared => prepared.map_err(Rc::new), + }; + count = count + .checked_add(prepared.as_ref().map_or(1, |p| p.count)) + .filter(|count| *count <= limit) + .ok_or(CandidateTimingError::ExpansionLimit(limit))?; + logical.push(prepared); + } + Ok(CandidateLifecyclePostASAPDAGs { + id, + logical, + rejected_assemblies, + count, + }) +} + +/// A logical candidate yields its assignments, or one diagnostic entry when it +/// has none: a state with no lifecycle alternative must not vanish silently. +fn prepared_timing( + index: Rc, + lifecycles: SummaryMaintenanceLifecycleCandidates<'_>, + limit: usize, +) -> Result, CandidateTimingError> { + match lifecycles.assignment_count(limit) { + Ok(count) => Ok(PreparedTiming { + index, + lifecycles, + count, + }), + Err(SummaryMaintenanceLifecycleChoiceError::ExpansionLimit(limit)) => { + Err(CandidateTimingError::ExpansionLimit(limit)) + } + Err(error) => Err(error.into()), + } +} + +impl CandidateLifecyclePostASAPDAGs<'_, Id> { + /// Includes rejected assignments; failures retain their candidate identity. + pub fn len(&self) -> usize { + self.count + } + pub fn is_empty(&self) -> bool { + self.len() == 0 + } + pub fn logical_len(&self) -> usize { + self.logical.len() + } + pub fn rejected_assemblies(&self) -> &[String] { + &self.rejected_assemblies + } + + pub fn iter( + &self, + ) -> impl Iterator< + Item = ( + PostASAPCandidateMetadata, + Result>, + ), + > + '_ { + self.logical + .iter() + .enumerate() + .flat_map(move |(logical_candidate, prepared)| { + (0..prepared.as_ref().map_or(1, |p| p.count)).map(move |assignment_candidate| { + let mut metadata = PostASAPCandidateMetadata { + id: self.id.clone(), + logical_candidate, + assignment_candidate, + choices: Vec::new(), + lifecycle: None, + }; + let timing = match prepared { + Err(error) => Err(error.clone()), + Ok(prepared) => { + let candidate = prepared.lifecycles.assignment_at(assignment_candidate); + metadata.choices = candidate.choices; + match candidate.plan { + Err(error) => Err(Rc::new(CandidateTimingError::Choice(error))), + Ok(plan) => { + let timing = + plan.execution_assignment(prepared.index.clone()).map_err( + |error| Rc::new(CandidateTimingError::Timing(error)), + ); + metadata.lifecycle = Some(plan); + timing + } + } + } + }; + (metadata, timing) + }) + }) + } + + /// Inspect alternatives or explicitly bind a choice without exposing the + /// internal enumerator as another stage output. + pub fn lifecycle_alternatives( + &self, + logical_candidate: usize, + ) -> Result<&[SummaryMaintenanceDeployment], Rc> { + self.prepared(logical_candidate) + .map(|p| p.lifecycles.deployments()) + } + /// Guarantee that choosing `lifecycle` for `state` would attach under this + /// workload's data arrival, so a deployment can supply a price for that + /// alternative before selection. `lifecycle` must be one of the state's alternatives. + pub fn lifecycle_guarantee( + &self, + logical_candidate: usize, + state: PostASAPNodeId, + lifecycle: &SummaryMaintenanceLifecycle, + ) -> Result> { + use SummaryMaintenanceLifecycleChoiceError as E; + let prepared = self.prepared(logical_candidate)?; + let deployment = prepared + .lifecycles + .deployments() + .iter() + .find(|d| d.post_asap_node_id == state) + .ok_or_else(|| Rc::new(CandidateTimingError::Choice(E::UnknownSummary(state))))?; + if !deployment + .alternatives + .iter() + .any(|a| &a.summary_maintenance_lifecycle == lifecycle) + { + return Err(Rc::new(CandidateTimingError::Choice(E::NotAnAlternative( + state, + )))); + } + Ok(prepared.lifecycles.guarantee(lifecycle)) + } + pub fn select_lifecycles( + &self, + logical_candidate: usize, + choices: &[(PostASAPNodeId, SummaryMaintenanceLifecycle)], + ) -> Result> { + self.prepared(logical_candidate)? + .lifecycles + .clone() + .select(choices) + .map_err(|error| Rc::new(CandidateTimingError::Choice(error))) + } + fn prepared( + &self, + logical_candidate: usize, + ) -> Result<&PreparedTiming<'_>, Rc> { + self.logical + .get(logical_candidate) + .ok_or_else(|| { + Rc::new(CandidateTimingError::UnknownLogicalCandidate( + logical_candidate, + )) + })? + .as_ref() + .map_err(Rc::clone) + } +} diff --git a/crates/asap-aware-mapping/src/cost_model.rs b/crates/asap-aware-mapping/src/cost_model.rs index d8e96553..827eae67 100644 --- a/crates/asap-aware-mapping/src/cost_model.rs +++ b/crates/asap-aware-mapping/src/cost_model.rs @@ -42,20 +42,20 @@ //! `docs/design_docs/cse-cost-model-decision.md` for the full design discussion (why //! cost-based, why not a full plan-search engine, the layering constraint //! that forces detection to stay cost-agnostic). -//! [`CandidatePostASAPDAGs::cost_sorted`](crate::replacement::CandidatePostASAPDAGs::cost_sorted) +//! [`CandidateLogicalPostASAPDAGs::cost_sorted`](crate::replacement::CandidateLogicalPostASAPDAGs::cost_sorted) //! (via [`crate::replacement`]'s own `cse_preference`) and //! [`DefaultCostModel::estimate_cost`] are this crate's own callers. use std::rc::Rc; use asap_types::post_asap::{ - ExactOperation, GroupingStrategy, HydraParams, ResultGuarantee, SketchAlgorithm, SketchParams, - SketchQuery, SummaryExpr, SummaryFamilyType, SummaryMaintenanceLifecycleGuarantee, SummaryNode, - SummaryWindowFramework, + ExactOperation, GroupingStrategy, HydraParams, PostASAPNode, ResultGuarantee, SketchAlgorithm, + SketchParams, SketchQuery, SummaryExpr, SummaryFamilyType, + SummaryMaintenanceLifecycleGuarantee, SummaryWindowFramework, }; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::pre_asap::expr_ir::ColumnRef; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::pre_asap::query_expr::PreASAPNode; use asap_types::types::AccuracyTarget; use crate::exact_composition::{ExactComposition, OperationPlacement}; @@ -128,7 +128,7 @@ impl ValueOperationCapabilities { #[derive(Debug, Clone, Copy)] pub struct ExactCompositionCostRequest<'a> { /// The pre-ASAP target the composed candidate replaces. - pub target: &'a QueryExpr, + pub target: &'a PreASAPNode, /// The composition itself — placement, operator, child target. pub composition: &'a ExactComposition, /// For [`OperationPlacement::Read`]: the child target's *selected* @@ -137,9 +137,9 @@ pub struct ExactCompositionCostRequest<'a> { /// transform that consumes its output (the `SummaryAgg` this transform /// feeds). Either way, the summary whose maintenance/read cost the /// formula charges. - pub summary: &'a SummaryNode, + pub summary: &'a PostASAPNode, /// How many times this site actually runs once ancestors' own choices - /// are accounted for (see `CandidatePostASAPDAGs::global_selection`). + /// are accounted for (see `CandidateLogicalPostASAPDAGs::global_selection`). pub effective_consumer_count: usize, } @@ -268,7 +268,7 @@ fn finite_rate(units_per_second: f64) -> Option { /// A CSE-detected, legality-gated shared subtree with two or more consumers /// — the unit [`CostModel::cse_share_decision`] decides over. Built by -/// [`CandidatePostASAPDAGs::cost_sorted`](crate::replacement::CandidatePostASAPDAGs::cost_sorted) +/// [`CandidateLogicalPostASAPDAGs::cost_sorted`](crate::replacement::CandidateLogicalPostASAPDAGs::cost_sorted) /// (via [`crate::replacement`]'s own `cse_preference`) the first time it /// needs a representative bound node for a subtree that /// [`asap_types::pre_asap::cse::share_common_subtrees`] already collapsed @@ -276,11 +276,11 @@ fn finite_rate(units_per_second: f64) -> Option { /// `docs/design_docs/cse-cost-model-decision.md`. pub struct CseCandidate<'a> { /// The shared pre-ASAP subtree itself. - pub subtree: &'a QueryExpr, - /// The `SummaryNode` this subtree bound to — gives the cost model the + pub subtree: &'a PreASAPNode, + /// The `PostASAPNode` this subtree bound to — gives the cost model the /// concrete `SummaryFamilyType`/`(kind, params)` actually at stake, not /// just the pre-ASAP shape. - pub bound_summary: &'a SummaryNode, + pub bound_summary: &'a PostASAPNode, /// How many workload roots reference this exact shared subtree, counted /// once up front over the whole workload (always >= 2 — a candidate is /// only ever constructed for an actually-shared subtree). @@ -302,7 +302,7 @@ pub struct Cost(pub f64); /// node. Node identity is preserved so whole-DAG models can bind per-state /// evidence without relying on traversal order. pub struct CostedSummaryDeployment<'a> { - pub summary: &'a SummaryNode, + pub summary: &'a PostASAPNode, pub guarantee: &'a SummaryMaintenanceLifecycleGuarantee, pub selected_cost: Cost, } @@ -359,7 +359,7 @@ impl std::ops::Mul for Cost { /// [`CseCandidate`]. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum ShareDecision { - /// Reuse one bound `SummaryNode` across every consumer. + /// Reuse one bound `PostASAPNode` across every consumer. Share, /// Bind each occurrence independently — the shared-maintenance cost /// isn't worth it for this candidate. @@ -382,7 +382,7 @@ pub enum ShareDecision { /// leaf costs little to recompute, a deep multi-join subtree costs a lot. /// A deployment with real per-row/per-update cost knowledge should /// override [`CostModel::cse_recompute_cost`] instead of relying on this. -pub fn default_cse_recompute_cost(subtree: &QueryExpr) -> Cost { +pub fn default_cse_recompute_cost(subtree: &PreASAPNode) -> Cost { Cost(asap_types::pre_asap::cse::dag_node_count(subtree) as f64) } @@ -506,7 +506,7 @@ pub trait CostModel { /// Estimated number of distinct subpopulations produced by `target`'s /// grouping keys. `None` means the deployment has no cardinality estimate; /// grouping alternatives remain legal but keep their discovery order. - fn estimated_subpopulation_count(&self, _target: &QueryExpr) -> Option { + fn estimated_subpopulation_count(&self, _target: &PreASAPNode) -> Option { None } @@ -598,7 +598,7 @@ pub trait CostModel { default_cse_shared_maintenance_cost(&family) } - /// Decide whether to reuse one shared `SummaryNode` across every + /// Decide whether to reuse one shared `PostASAPNode` across every /// consumer of `candidate`, or bind each occurrence independently — a /// Volcano/Cascades-style cost comparison (issue #237, #223 stage 4; see /// `docs/design_docs/cse-cost-model-decision.md`): share iff the estimated cost of @@ -735,7 +735,7 @@ pub trait CostModel { /// [`ReplacementSubDAG`] candidate at `target` — a real `f64`, not just a /// relative rank, meant for a caller that wants to *display* "candidate A /// costs ≈ X, candidate B costs ≈ Y" (e.g. a DAG-visualization view built - /// on [`CandidatePostASAPDAGs::cost_sorted`](crate::replacement::CandidatePostASAPDAGs::cost_sorted)), + /// on [`CandidateLogicalPostASAPDAGs::cost_sorted`](crate::replacement::CandidateLogicalPostASAPDAGs::cost_sorted)), /// not just order candidates against each other — that ordering job /// already belongs to [`rank_candidates`](Self::rank_candidates) (for a /// [`SketchAlgorithmStrategy`](crate::replacement::SketchAlgorithmStrategy) @@ -745,10 +745,10 @@ pub trait CostModel { /// /// One method covers both candidate shapes this crate ships: /// `candidate.replacement`'s [`Replacement::Summary`] arm (a - /// `SketchAlgorithmStrategy` candidate — the bound `SummaryNode` is right + /// `SketchAlgorithmStrategy` candidate — the bound `PostASAPNode` is right /// there, nothing to reconstruct) and its [`Replacement::Rewrite`] arm /// (a `SharedSubtreeStrategy` share-vs-recompute candidate — no bound - /// `SummaryNode` of its own, since sharing is a decision about a target + /// `PostASAPNode` of its own, since sharing is a decision about a target /// already bound some other way; a representative binding is recovered /// from `target` itself). `target` is threaded through explicitly /// (rather than only ever the target embedded in `candidate` — there @@ -776,7 +776,7 @@ pub trait CostModel { /// optimistic zeroes. fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &PostASAPNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs::default() } @@ -786,7 +786,7 @@ pub trait CostModel { /// rate so the horizon integral equals one peak-capacity charge. fn summary_maintenance_lifecycle_cost_inputs_for_horizon( &self, - summary: &SummaryNode, + summary: &PostASAPNode, _horizon: Option, ) -> SummaryMaintenanceLifecycleCostInputs { self.summary_maintenance_lifecycle_cost_inputs(summary) @@ -796,7 +796,7 @@ pub trait CostModel { /// conservative default advertises no long-lived maintenance capability. fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &PostASAPNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities::default() } @@ -807,8 +807,8 @@ pub trait CostModel { /// must not then reuse the partial per-state sum. fn complete_summary_candidate_cost( &self, - _root: &SummaryNode, - _target: Option<&QueryExpr>, + _root: &PostASAPNode, + _target: Option<&PreASAPNode>, deployments: &[CostedSummaryDeployment<'_>], _horizon: Option, _expected_reads: Option, @@ -827,8 +827,8 @@ pub trait CostModel { /// that do not perform either decision. fn complete_summary_candidate_estimate( &self, - root: &SummaryNode, - target: Option<&QueryExpr>, + root: &PostASAPNode, + target: Option<&PreASAPNode>, deployments: &[CostedSummaryDeployment<'_>], horizon: Option, expected_reads: Option, @@ -861,7 +861,7 @@ pub trait CostModel { /// Cost of evaluating `target` directly from its logical/raw inputs once. /// When known, lifecycle-aware materialization compares this fallback with /// the aggregate cost of the selected summary deployments. - fn raw_query_recompute_cost(&self, _target: &QueryExpr) -> Option { + fn raw_query_recompute_cost(&self, _target: &PreASAPNode) -> Option { None } @@ -870,7 +870,7 @@ pub trait CostModel { /// cardinality changes between evaluations. fn raw_query_recompute_total_cost( &self, - target: &QueryExpr, + target: &PreASAPNode, expected_reads: f64, ) -> Option { self.raw_query_recompute_cost(target) @@ -879,7 +879,7 @@ pub trait CostModel { /// Physical feasibility evidence for a complete summary candidate. /// `None` defers admission to physical/deployment compilation; `Some(false)` /// excludes the candidate without changing its computation or parameters. - fn summary_support_evidence(&self, _summary: &SummaryNode) -> Option { + fn summary_support_evidence(&self, _summary: &PostASAPNode) -> Option { None } @@ -932,7 +932,7 @@ pub trait CostModel { /// /// Default: every input unknown ([`ExactCompositionCostInputs::unknown`]) /// — unknown is never zero, and with no rate derivable - /// `CandidatePostASAPDAGs::global_selection` keeps the conservative `KeepPreAsap` + /// `CandidateLogicalPostASAPDAGs::global_selection` keeps the conservative `KeepPreAsap` /// behavior for the site. A deployment that wants defaults must supply /// them here explicitly. fn exact_composition_cost_inputs( @@ -948,7 +948,7 @@ pub trait CostModel { } fn sketch_state( - node: &SummaryNode, + node: &PostASAPNode, ) -> Option<(&asap_types::post_asap::SketchKind, &GroupingStrategy)> { match &node.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_state(summary_input), @@ -1035,7 +1035,7 @@ impl CostModel for DefaultCostModel { /// [`default_cse_shared_maintenance_cost`] already orders candidates /// by). /// - [`Replacement::Rewrite`]: recovers one representative bound - /// `SummaryNode` for `target` via `realize_child` (the same + /// `PostASAPNode` for `target` via `realize_child` (the same /// rank-and-take-first helper `replacement::realize_child` reuses for the /// identical need), then charges /// `cse_shared_maintenance_cost` for the candidate that shares @@ -1103,7 +1103,7 @@ impl CostModel for DefaultCostModel { } } // A composed candidate is costed in cost-units-per-second by - // `CandidatePostASAPDAGs::global_selection` against the child decision it + // `CandidateLogicalPostASAPDAGs::global_selection` against the child decision it // is committed with — a different unit from this structural // estimate, and unknowable here without that child. `NaN` // keeps it from ever out-ranking a real estimate by accident. @@ -1352,8 +1352,8 @@ mod tests { use asap_types::pre_asap::query_expr::Source; use asap_types::pre_asap::schema::{Column, DataType, Schema}; - fn scan() -> QueryExpr { - QueryExpr::Scan { + fn scan() -> PreASAPNode { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -1367,10 +1367,10 @@ mod tests { } } - fn summary_node(family: SummaryFamilyType) -> SummaryNode { - SummaryNode { + fn summary_node(family: SummaryFamilyType) -> PostASAPNode { + PostASAPNode { expr: SummaryExpr::SummaryAgg { - child: std::rc::Rc::new(SummaryNode { + child: std::rc::Rc::new(PostASAPNode { expr: SummaryExpr::KeepPreAsap(Rc::new(scan())), schema: SummarySchema { fields: vec![], @@ -1400,7 +1400,7 @@ mod tests { #[test] fn default_recompute_cost_is_positive_and_grows_with_structural_size() { let leaf = scan(); - let nested = QueryExpr::Dedup { + let nested = PreASAPNode::Dedup { cols: vec![0], child: std::rc::Rc::new(leaf.clone()), }; @@ -1420,18 +1420,18 @@ mod tests { use asap_types::pre_asap::query_expr::{JoinKind, Predicate}; let true_pred = || { - Predicate(std::rc::Rc::new(QueryExpr::Literal(ScalarValue::Boolean( - true, - )))) + Predicate(std::rc::Rc::new(PreASAPNode::Literal( + ScalarValue::Boolean(true), + ))) }; let shared_leaf = std::rc::Rc::new(scan()); - let no_sharing = QueryExpr::Join { + let no_sharing = PreASAPNode::Join { kind: JoinKind::Inner, pred: true_pred(), left: std::rc::Rc::new(scan()), right: std::rc::Rc::new(scan()), }; - let with_sharing = QueryExpr::Join { + let with_sharing = PreASAPNode::Join { kind: JoinKind::Inner, pred: true_pred(), left: std::rc::Rc::clone(&shared_leaf), diff --git a/crates/asap-aware-mapping/src/empirical_cost.rs b/crates/asap-aware-mapping/src/empirical_cost.rs index ebfbc773..d8022cb4 100644 --- a/crates/asap-aware-mapping/src/empirical_cost.rs +++ b/crates/asap-aware-mapping/src/empirical_cost.rs @@ -3,7 +3,7 @@ //! of an accuracy guarantee. CPU quantities are nanoseconds, never CPU operations. use asap_types::post_asap::{ - GroupingStrategy, SketchAlgorithm, SketchParams, SummaryExpr, SummaryFamilyType, SummaryNode, + GroupingStrategy, PostASAPNode, SketchAlgorithm, SketchParams, SummaryExpr, SummaryFamilyType, }; use asap_types::pre_asap::AggIntent; use serde::{Deserialize, Serialize}; @@ -214,7 +214,7 @@ impl EmpiricalEvidenceProvider { /// mixed with an existing deployment's unitless or CPU-operation costs. pub fn lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &PostASAPNode, ) -> SummaryMaintenanceLifecycleCostInputs { let SummaryExpr::SummaryAgg { family: SummaryFamilyType::Sketch(kind, GroupingStrategy::PerSubpopulationInstance), @@ -280,7 +280,7 @@ impl CostModel for EmpiricalCostModel { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &PostASAPNode, ) -> SummaryMaintenanceLifecycleCostInputs { self.provider.lifecycle_cost_inputs(summary) } diff --git a/crates/asap-aware-mapping/src/exact_composition.rs b/crates/asap-aware-mapping/src/exact_composition.rs index 21e8be0f..801add4c 100644 --- a/crates/asap-aware-mapping/src/exact_composition.rs +++ b/crates/asap-aware-mapping/src/exact_composition.rs @@ -18,9 +18,9 @@ //! **not** pick that child itself (the way `construct_summary_agg`'s //! `realize_child` takes the head of the child's own ranking): a //! [`Replacement::ExactComposition`] carries only the child *target* -//! (`ExactComposition::child_target`, the same `Rc` whose -//! `TargetSubDAGCandidates` in `CandidatePostASAPDAGs` already holds every candidate for it). It is -//! [`CandidatePostASAPDAGs::global_selection`](crate::replacement::CandidatePostASAPDAGs::global_selection) +//! (`ExactComposition::child_target`, the same `Rc` whose +//! `TargetSubDAGCandidates` in `CandidateLogicalPostASAPDAGs` already holds every candidate for it). It is +//! [`CandidateLogicalPostASAPDAGs::global_selection`](crate::replacement::CandidateLogicalPostASAPDAGs::global_selection) //! that commits the compatible parent/child pair — so the child's own //! cost-model ranking, workload-wide effective consumer count, and shared //! `Rc` identity (one inner summary serving two outer folds) all stay @@ -65,11 +65,11 @@ use std::rc::Rc; use asap_types::post_asap::execution_data_state::validate_execution_data_states_at; use asap_types::post_asap::{ exact_operation_output_schema, produced_data_state, AccuracyError, ExactOperation, - ExecutionDataState, ExecutionDataStateError, ResultGuarantee, SummaryExpr, SummaryNode, + ExecutionDataState, ExecutionDataStateError, PostASAPNode, ResultGuarantee, SummaryExpr, SummarySchema, ValueOperation, }; use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; +use asap_types::pre_asap::query_expr::{PreASAPNode, Reduction}; use asap_types::types::AccuracyTarget; use crate::cost_model::CostModel; @@ -120,7 +120,7 @@ pub struct ExactComposition { pub op: ExactOperation, /// The pre-ASAP child the operator consumes; its `TargetSubDAGCandidates` holds the /// candidates `global_selection` may commit this composition with. - pub child_target: Rc, + pub child_target: Rc, /// The composed node's output schema — the target's own pre-ASAP /// output schema, lifted with every column `Plain` (an exact operator /// only ever produces plain values). @@ -132,7 +132,7 @@ impl ExactComposition { /// (the child's produced data_state — a `KeepPreAsap` leaf takes the /// phase this edge assigns) plus the plain-operand rule, checked /// through the same schema derivation [`Self::compose`] uses. - pub fn accepts_child(&self, child: &SummaryNode) -> bool { + pub fn accepts_child(&self, child: &PostASAPNode) -> bool { let phase_ok = match produced_data_state(&child.expr) { None => true, Some(avail) => avail == self.placement.data_state(), @@ -145,7 +145,7 @@ impl ExactComposition { /// `asap_types::post_asap::validate_execution_data_states`; an illegal /// placement is a typed [`RealizationError::ExecutionDataState`], never deferred to a /// runtime. - pub fn compose(&self, child: Rc) -> Result, RealizationError> { + pub fn compose(&self, child: Rc) -> Result, RealizationError> { self.compose_with_accuracy(child, &DefaultAccuracyModel) } @@ -154,9 +154,9 @@ impl ExactComposition { /// unsupported folds fail closed with a typed accuracy error. pub fn compose_with_accuracy( &self, - child: Rc, + child: Rc, accuracy_model: &dyn AccuracyModel, - ) -> Result, RealizationError> { + ) -> Result, RealizationError> { if let Some(produced) = produced_data_state(&child.expr) { if produced != self.placement.data_state() { let edge = match self.placement { @@ -209,7 +209,7 @@ impl ExactComposition { operation: ValueOperation::Exact(self.op.clone()), timing, }; - let node = Rc::new(SummaryNode { + let node = Rc::new(PostASAPNode { expr, schema, guarantee, @@ -259,10 +259,10 @@ fn needs_readout(implementation: &Realization) -> bool { /// The `(op, child)` of a read-time operation-shaped target, or `None`. fn query_time_shape( - root: &QueryExpr, + root: &PreASAPNode, cost_model: &dyn CostModel, -) -> Option<(ExactOperation, Rc, AggIntent)> { - let QueryExpr::Aggregate { +) -> Option<(ExactOperation, Rc, AggIntent)> { + let PreASAPNode::Aggregate { reduction, measures, output_names, @@ -309,10 +309,10 @@ fn query_time_shape( /// The `(op, child)` of a function-shaped target — a per-entity exact /// transform with no accumulator form — or `None`. fn ingestion_time_shape( - root: &QueryExpr, + root: &PreASAPNode, cost_model: &dyn CostModel, -) -> Option<(ExactOperation, Rc, AggIntent)> { - let QueryExpr::Aggregate { +) -> Option<(ExactOperation, Rc, AggIntent)> { + let PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures, output_names, @@ -461,21 +461,21 @@ mod tests { use asap_types::pre_asap::query_expr::Source; use asap_types::pre_asap::schema::{Column, DataType, Schema}; - fn metric_scan(labels: &[&str]) -> QueryExpr { + fn metric_scan(labels: &[&str]) -> PreASAPNode { let mut columns = vec![ Column::new("ts", DataType::Timestamp, false), Column::new("value", DataType::Float64, false), ]; columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); - QueryExpr::Scan { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index(columns, 0, vec![]), } } - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], @@ -484,8 +484,8 @@ mod tests { } } - fn per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn per_entity(intent: AggIntent, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec![], @@ -495,7 +495,7 @@ mod tests { } /// `max by (zone) (quantile by (zone, host) (m))`. - fn max_over_quantile() -> Rc { + fn max_over_quantile() -> Rc { let inner = agg( vec![2, 3], default_quantile(0.99), @@ -523,7 +523,7 @@ mod tests { candidates[0].provenance, ReplacementProvenance::ValueOperationAtQueryTime ); - let QueryExpr::Aggregate { child, .. } = root.as_ref() else { + let PreASAPNode::Aggregate { child, .. } = root.as_ref() else { unreachable!() }; assert!( diff --git a/crates/asap-aware-mapping/src/explanation.rs b/crates/asap-aware-mapping/src/explanation.rs index b2855d9d..89e6e052 100644 --- a/crates/asap-aware-mapping/src/explanation.rs +++ b/crates/asap-aware-mapping/src/explanation.rs @@ -22,7 +22,7 @@ //! (issue #252) now *already* computes, for every //! [`TargetSubDAG`](crate::replacement::TargetSubDAG) in the workload, every //! semantically valid [`crate::replacement::ReplacementSubDAG`] a registered -//! [`ReplacementStrategy`] can propose — a [`CandidatePostASAPDAGs`] of [`TargetSubDAGCandidates`]s. A +//! [`ReplacementStrategy`] can propose — a [`CandidateLogicalPostASAPDAGs`] of [`TargetSubDAGCandidates`]s. A //! rule re-deriving the same yes/no fact from scratch would be answering a //! question the search already answered, via a second, independently //! maintained traversal that has to keep agreeing with the first one. @@ -39,7 +39,7 @@ //! genuine alternative to the status quo (share this already-shared subtree //! instead of recomputing it at every consumer), *is* an applicability //! finding — [`explain_replacements`] and -//! [`explain_replacements_with`] just translate [`CandidatePostASAPDAGs`]'s +//! [`explain_replacements_with`] just translate [`CandidateLogicalPostASAPDAGs`]'s //! [`TargetSubDAGCandidates`]s into that shape: //! //! - [`ExplanationKind::SketchApproximation`] — the `TargetSubDAG`'s @@ -88,14 +88,14 @@ //! (`fn optimization(&self) -> ExplanationKind` + `fn evaluate(&self, roots) //! -> Vec`), the same shape [`crate::cost_model::CostModel`] //! and [`crate::replacement::Matcher`] use elsewhere in this crate. Once -//! findings are a *view* over [`CandidatePostASAPDAGs`] rather than an independent +//! findings are a *view* over [`CandidateLogicalPostASAPDAGs`] rather than an independent //! computation, that trait would be a second extension point answering a //! question [`ReplacementStrategy`] (issue #251) already answers: "does this //! `TargetSubDAG` have an alternative worth reporting, and why". A caller who //! wants a new optimization represented as a finding needs a new //! `impl ReplacementStrategy` wired into //! [`crate::replacement::search_workload_with`]'s strategy set *regardless* -//! (that's the only way its candidates end up in the [`CandidatePostASAPDAGs`] this +//! (that's the only way its candidates end up in the [`CandidateLogicalPostASAPDAGs`] this //! module reads) — adding an `ApplicabilityRule` too would mean maintaining //! two extension points for the same new capability, one of which (the rule) //! would just be re-describing candidates the other (the strategy) already @@ -123,7 +123,7 @@ //! [`crate::replacement`] now instead of here. //! 2. **A node reachable via more than one path is one finding, not one per //! path.** [`TargetSubDAGCandidates`]s are keyed by `Rc` pointer identity in -//! [`CandidatePostASAPDAGs`]'s internal map — there is exactly one group per distinct +//! [`CandidateLogicalPostASAPDAGs`]'s internal map — there is exactly one group per distinct //! `Rc`, full stop, so a shared `Aggregate` reached via two different //! `BinaryOp` branches (or two different workload roots) is exactly one //! group, hence at most one [`ExplanationKind::SketchApproximation`] @@ -131,16 +131,16 @@ //! [`tests::a_shared_sketchable_aggregate_is_reported_only_once`] pins //! this directly. //! -//! ## One thing [`CandidatePostASAPDAGs`] doesn't carry that this module still needs: +//! ## One thing [`CandidateLogicalPostASAPDAGs`] doesn't carry that this module still needs: //! human-readable `location` text //! -//! [`TargetSubDAGCandidates`]/[`CandidatePostASAPDAGs`] deliberately track only `Rc` +//! [`TargetSubDAGCandidates`]/[`CandidateLogicalPostASAPDAGs`] deliberately track only `Rc` //! pointer identity — the currency the search itself needs — not //! caller-facing prose. [`ReplacementExplanation::location`] is prose (a //! breadcrumb like `root "dash_a" > lhs`), so this module keeps one small, //! self-contained walk of its own, [`collect_locations`], whose *only* job //! is turning "this `Rc`" into "the human-readable place(s) it occurs" for a -//! finding already decided by [`CandidatePostASAPDAGs`]. This is not a reincarnation of +//! finding already decided by [`CandidateLogicalPostASAPDAGs`]. This is not a reincarnation of //! the deleted rule traversal: it makes no applicability decision (it runs //! the same regardless of what any strategy found), and duplicating this //! small, self-contained shape rather than threading location strings @@ -179,19 +179,19 @@ //! [`Replacement::Rewrite`]: crate::replacement::Replacement::Rewrite //! [`SketchAlgorithmStrategy`]: crate::replacement::SketchAlgorithmStrategy //! [`SharedSubtreeStrategy`]: crate::replacement::SharedSubtreeStrategy -//! [`CandidatePostASAPDAGs`]: crate::replacement::CandidatePostASAPDAGs +//! [`CandidateLogicalPostASAPDAGs`]: crate::replacement::CandidateLogicalPostASAPDAGs //! [`TargetSubDAGCandidates`]: crate::replacement::TargetSubDAGCandidates use std::collections::HashMap; use std::fmt::Display; use std::rc::Rc; -use asap_types::post_asap::{SummaryExpr, SummaryFamilyType, SummaryNode}; +use asap_types::post_asap::{PostASAPNode, SummaryExpr, SummaryFamilyType}; use asap_types::pre_asap::cse::{structural_hash, HashCache}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::pre_asap::query_expr::PreASAPNode; use crate::replacement::{ - self, CandidatePostASAPDAGs, Replacement, ReplacementStrategy, TargetSubDAGCandidates, + self, CandidateLogicalPostASAPDAGs, Replacement, ReplacementStrategy, TargetSubDAGCandidates, }; /// Which kind of replacement a [`ReplacementExplanation`] is about. @@ -235,9 +235,9 @@ pub enum ExplanationKind { /// /// `node_hash` is [`structural_hash`](asap_types::pre_asap::cse::structural_hash) /// of the `TargetSubDAG`'s own `target` subtree — the same function, on the -/// same `Rc` shape, that [`asap_types::dag_export::DagNode::hash`] +/// same `Rc` shape, that [`asap_types::dag_export::DagNode::hash`] /// is computed with. A downstream consumer that independently exported the -/// same `QueryExpr` (e.g. via `asap_types::dag_export::export`) can match +/// same `PreASAPNode` (e.g. via `asap_types::dag_export::export`) can match /// this explanation to a `DagNode` by first comparing hashes and then /// confirming structural equality with [`ReplacementExplanation::target`]. #[derive(Debug, Clone, PartialEq)] @@ -249,7 +249,7 @@ pub struct ReplacementExplanation { /// The exact target expression the explanation describes. Reporting /// integrations use this together with `node_hash`: the hash narrows the /// search, and structural equality makes the final match collision-safe. - pub target: Rc, + pub target: Rc, } /// Explain every replacement [`crate::replacement::search_workload`] finds @@ -265,7 +265,7 @@ pub struct ReplacementExplanation { /// candidate-plan space, then reads findings off it — see the module docs' /// "The reframing" section for what that translation actually checks. pub fn explain_replacements( - roots: Vec<(Id, QueryExpr)>, + roots: Vec<(Id, PreASAPNode)>, ) -> Vec { explain_replacements_with(roots, &replacement::default_strategies()) } @@ -279,10 +279,10 @@ pub fn explain_replacements( /// /// [`ReplacementStrategy`]: crate::replacement::ReplacementStrategy pub fn explain_replacements_with<'s, Id: Display>( - roots: Vec<(Id, QueryExpr)>, + roots: Vec<(Id, PreASAPNode)>, strategies: &[Box], ) -> Vec { - let ided: Vec<(String, Rc)> = roots + let ided: Vec<(String, Rc)> = roots .into_iter() .map(|(id, expr)| (id.to_string(), Rc::new(expr))) .collect(); @@ -302,7 +302,7 @@ pub fn explain_replacements_with<'s, Id: Display>( /// [`std::fmt::Debug`] for the breadcrumb text) doesn't need its own generic /// `Id` bound. fn findings_from_candidate_dags( - space: &CandidatePostASAPDAGs, + space: &CandidateLogicalPostASAPDAGs, ) -> Vec { let locations = collect_locations(&space.roots); // One cache for the whole pass, mirroring `dag_export::export`'s own @@ -409,7 +409,7 @@ fn shared_subexpr_finding_reason(group: &TargetSubDAGCandidates) -> Option bool { +fn is_sketch_realization(node: &PostASAPNode) -> bool { if node .guarantee .as_ref() @@ -427,13 +427,15 @@ fn is_sketch_realization(node: &SummaryNode) -> bool { // ── location breadcrumbs ───────────────────────────────────────────────── /// Build `location` text for every distinct `TargetSubDAG` reachable from -/// `roots` — see the module docs' "One thing `CandidatePostASAPDAGs` doesn't carry" +/// `roots` — see the module docs' "One thing `CandidateLogicalPostASAPDAGs` doesn't carry" /// section for why this module needs its own small walk for this. Returns /// every breadcrumb path that reaches a given `Rc`, not just the first: a /// shared node referenced from two workload roots (or two branches of one /// root) needs both breadcrumbs in its finding's `location`, not just one. -fn collect_locations(roots: &[(String, Rc)]) -> HashMap<*const QueryExpr, Vec> { - let mut locations: HashMap<*const QueryExpr, Vec> = HashMap::new(); +fn collect_locations( + roots: &[(String, Rc)], +) -> HashMap<*const PreASAPNode, Vec> { + let mut locations: HashMap<*const PreASAPNode, Vec> = HashMap::new(); for (id, root) in roots { visit(root, format!("root {id:?}"), &mut locations); } @@ -444,9 +446,9 @@ fn collect_locations(roots: &[(String, Rc)]) -> HashMap<*const QueryE /// through its children. A shared ancestor is intentionally traversed once /// per incoming path so every descendant receives every valid breadcrumb. fn visit( - node: &Rc, + node: &Rc, label: String, - locations: &mut HashMap<*const QueryExpr, Vec>, + locations: &mut HashMap<*const PreASAPNode, Vec>, ) { let ptr = Rc::as_ptr(node); locations.entry(ptr).or_default().push(label.clone()); @@ -456,14 +458,14 @@ fn visit( /// `node`'s own **relational-skeleton** operator children — the same scope /// `crate::replacement`'s own target-discovery `walk_children` (and /// `asap_types::pre_asap::cse::share_common_subtrees`'s `rebuild_children`) -/// use. Exhaustive over every `QueryExpr` variant: a new variant fails to +/// use. Exhaustive over every `PreASAPNode` variant: a new variant fails to /// compile here until this match is extended too. fn visit_children( - node: &QueryExpr, + node: &PreASAPNode, label: &str, - locations: &mut HashMap<*const QueryExpr, Vec>, + locations: &mut HashMap<*const PreASAPNode, Vec>, ) { - use QueryExpr::*; + use PreASAPNode::*; match node { Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => { @@ -519,21 +521,21 @@ mod tests { use asap_types::pre_asap::schema::{Column, DataType, Schema}; use asap_types::types::AccuracyTarget; - fn metric_scan(labels: &[&str]) -> QueryExpr { + fn metric_scan(labels: &[&str]) -> PreASAPNode { let mut columns = vec![ Column::new("ts", DataType::Timestamp, false), Column::new("value", DataType::Float64, false), ]; columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); - QueryExpr::Scan { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index(columns, 0, vec![]), } } - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], @@ -562,7 +564,7 @@ mod tests { } /// `node_hash` must be the literal `structural_hash` a downstream - /// consumer would compute over the *same* `QueryExpr` subtree via + /// consumer would compute over the *same* `PreASAPNode` subtree via /// `asap_types::dag_export::export` — the whole point of carrying it is /// that two independent exports of the same tree agree, with no /// string-matching against `location` required. @@ -581,7 +583,7 @@ mod tests { Some(sketch.node_hash), expected_hash, "ReplacementExplanation::node_hash must match dag_export's DagNode::hash \ - for the same QueryExpr subtree" + for the same PreASAPNode subtree" ); } @@ -643,7 +645,7 @@ mod tests { #[test] fn a_shared_sketchable_aggregate_is_reported_only_once() { let quantile = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let root = QueryExpr::BinaryOp { + let root = PreASAPNode::BinaryOp { op: asap_types::pre_asap::query_expr::BinaryOpKind::Compare( asap_types::pre_asap::expr_ir::CompareOpKind::Eq, ), @@ -737,7 +739,7 @@ mod tests { // The same shared branch appearing twice within one query (an `a/a` // shape) — single-query CSE. let branch = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let q = QueryExpr::BinaryOp { + let q = PreASAPNode::BinaryOp { op: asap_types::pre_asap::query_expr::BinaryOpKind::Arithmetic( asap_types::pre_asap::expr_ir::ArithmeticOpKind::Div, ), @@ -772,12 +774,12 @@ mod tests { use asap_types::pre_asap::query_expr::Predicate; let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let root_a = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(1)))), + let root_a = PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Int64(1)))), child: Rc::new(shared.clone()), }; - let root_b = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(2)))), + let root_b = PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Int64(2)))), child: Rc::new(shared), }; let findings = explain_replacements(vec![("a", root_a), ("b", root_b)]); diff --git a/crates/asap-aware-mapping/src/grouping.rs b/crates/asap-aware-mapping/src/grouping.rs index da4967ef..7630a691 100644 --- a/crates/asap-aware-mapping/src/grouping.rs +++ b/crates/asap-aware-mapping/src/grouping.rs @@ -73,11 +73,11 @@ use std::rc::Rc; use asap_types::post_asap::{ default_hydra_params, hydra_kind_for, AccuracyError, BoundExpr, CompositionOperator, - GroupingStrategy, GuaranteeSource, HydraKind, ProbabilityExpr, ResultGuarantee, - SketchAlgorithm, SketchParams, SummaryExpr, SummaryFamilyType, SummaryNode, + GroupingStrategy, GuaranteeSource, HydraKind, PostASAPNode, ProbabilityExpr, ResultGuarantee, + SketchAlgorithm, SketchParams, SummaryExpr, SummaryFamilyType, }; use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; +use asap_types::pre_asap::query_expr::{PreASAPNode, Reduction}; use crate::accuracy::{ AccuracyBudgetAllocator, AccuracyEvidenceProvider, AccuracyModel, PropagationStats, @@ -173,7 +173,7 @@ impl<'a> HydraGroupingStrategy<'a> { /// variant modeled. fn hydra_proposals(&self, target: &TargetSubDAG<'_>) -> Proposals { let mut proposals = Proposals::default(); - let QueryExpr::Aggregate { reduction, .. } = target.root.as_ref() else { + let PreASAPNode::Aggregate { reduction, .. } = target.root.as_ref() else { return proposals; }; if !has_subpopulations(reduction) { @@ -212,7 +212,7 @@ impl<'a> HydraGroupingStrategy<'a> { /// axis owns. fn build_candidate( &self, - root: &Rc, + root: &Rc, intent: &AggIntent, sketch_kind: SketchAlgorithm, hydra_kind: HydraKind, @@ -313,7 +313,7 @@ impl<'a> HydraGroupingStrategy<'a> { impl ReplacementStrategy for HydraGroupingStrategy<'_> { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { - let QueryExpr::Aggregate { reduction, .. } = target.root.as_ref() else { + let PreASAPNode::Aggregate { reduction, .. } = target.root.as_ref() else { return false; }; if !has_subpopulations(reduction) { @@ -357,7 +357,7 @@ impl ReplacementStrategy for HydraGroupingStrategy<'_> { /// destructures the right variant for `kind`; this function's only job is /// to find whatever `SketchParams` the bind decision already committed to /// and hand the whole thing over unchanged. -fn per_subpopulation_sketch_params(node: &SummaryNode) -> Option { +fn per_subpopulation_sketch_params(node: &PostASAPNode) -> Option { match &node.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => { per_subpopulation_sketch_params(summary_input) @@ -377,15 +377,15 @@ fn per_subpopulation_sketch_params(node: &SummaryNode) -> Option { /// sketch candidate this module builds actually has) to reach the /// `SummaryAgg` underneath. fn with_grouping( - node: Rc, + node: Rc, grouping: GroupingStrategy, stats: &PropagationStats, -) -> Rc { +) -> Rc { match &node.expr { SummaryExpr::SummaryEstimate { summary_input, query, - } => Rc::new(SummaryNode { + } => Rc::new(PostASAPNode { expr: SummaryExpr::SummaryEstimate { summary_input: with_grouping(Rc::clone(summary_input), grouping, stats), query: query.clone(), @@ -412,7 +412,7 @@ fn with_grouping( field.dtype = SummaryFamilyType::Sketch(kind.clone(), grouping.clone()); } } - Rc::new(SummaryNode { + Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child: Rc::clone(child), family: grouped_family, @@ -497,21 +497,21 @@ mod tests { use asap_types::pre_asap::schema::{Column, DataType, Schema}; use asap_types::types::AccuracyTarget; - fn metric_scan(labels: &[&str]) -> QueryExpr { + fn metric_scan(labels: &[&str]) -> PreASAPNode { let mut columns = vec![ Column::new("ts", DataType::Timestamp, false), Column::new("value", DataType::Float64, false), ]; columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); - QueryExpr::Scan { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index(columns, 0, vec![]), } } - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], @@ -520,8 +520,8 @@ mod tests { } } - fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn agg_per_entity(intent: AggIntent, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec![], @@ -836,7 +836,7 @@ mod tests { fn does_not_match_a_multi_intent_or_having_aggregate() { let strategy = HydraGroupingStrategy::default_cost_model(); - let multi = Rc::new(QueryExpr::Aggregate { + let multi = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], output_names: vec![], diff --git a/crates/asap-aware-mapping/src/lib.rs b/crates/asap-aware-mapping/src/lib.rs index 9c876a5a..5f4a3c5e 100644 --- a/crates/asap-aware-mapping/src/lib.rs +++ b/crates/asap-aware-mapping/src/lib.rs @@ -1,18 +1,18 @@ //! `asap-plan` — the cost-aware optimizer layer over the pre-ASAP intent algebra. //! //! This crate sits between the language-agnostic IR ([`asap_ir`]) and -//! any runtime: it consumes pre-ASAP [`QueryExpr`](asap_types::pre_asap::QueryExpr) +//! any runtime: it consumes pre-ASAP [`PreASAPNode`](asap_types::pre_asap::PreASAPNode) //! trees and makes the cost-aware decisions the pre-ASAP IR deliberately //! leaves open — which sketch (if any) realises each approximate intent. //! //! **Common sub-expression elimination (CSE) is not this crate's job.** -//! Detection is a primary pass over the pre-ASAP `QueryExpr` IR itself +//! Detection is a primary pass over the pre-ASAP `PreASAPNode` IR itself //! (`asap_types::pre_asap`, design tracked in issue #223), run before a //! tree ever reaches [`replacement::SketchAlgorithmStrategy`] — see issue #222 //! for why (batch query optimization needs to see shared work across a //! `QueryWorkload` before summary binding, not after). This crate may //! eventually run a second, narrower CSE pass of its own over an -//! already-bound `SummaryExpr`/`SummaryNode` DAG, recognizing sharing that's invisible +//! already-bound `SummaryExpr`/`PostASAPNode` DAG, recognizing sharing that's invisible //! at the pre-ASAP level by construction — e.g. `Quantile(x, 0.99)` and //! `Quantile(x, 0.95)` are structurally distinct `AggIntent`s but can //! still share one built sketch, read out twice. That post-ASAP pass is @@ -29,7 +29,7 @@ //! //! ## Planning workflows //! -//! Candidate search returns [`CandidatePostASAPDAGs`](replacement::CandidatePostASAPDAGs), a compact +//! Candidate search returns [`CandidateLogicalPostASAPDAGs`](replacement::CandidateLogicalPostASAPDAGs), a compact //! logical choice space with one [`TargetSubDAGCandidates`] per target sub-DAG. //! [`ReplacementStrategy`] implementations propose local alternatives; search //! applies the applicable semantic and accuracy checks. Candidate presence does @@ -37,18 +37,18 @@ //! //! Integrators choose among these workflows: //! -//! - Inspect the candidate space, optionally using [`CandidatePostASAPDAGs::cost_sorted`] -//! to obtain ranked views, and perform selection downstream. -//! - Call [`CandidatePostASAPDAGs::global_selection`] once for the workload, then +//! - Inspect the candidate space, optionally using [`CandidateLogicalPostASAPDAGs::cost_sorted`] +//! to obtain ranked views under the supplied cost model. +//! - Call [`CandidateLogicalPostASAPDAGs::global_selection`] once for the workload, then //! [`GlobalSelection::assemble_selected_dag`] for each query root. This //! coordinates logical choices and preserves shared nodes, but makes no //! summary-maintenance lifecycle decision. //! - When Planner owns maintenance-versus-recomputation decisions, use //! [`global_selection_with_summary_maintenance_lifecycles`] followed by //! [`assemble_selected_dag_with_summary_maintenance_lifecycles`] per root. -//! This alternative workflow returns [`SummaryMaintenanceLifecyclePlan`] -//! values containing DAG roots and maintenance decisions; callers do not need -//! to run ordinary selection/assembly first. +//! This alternative workflow returns [`LifecyclePostASAPDAG`] +//! values: DAG roots annotated with their lifecycle assignment; callers do not +//! need to run ordinary selection/assembly first. //! //! Models and evidence determine which choices the helpers can justify. //! Physical operator binding, placement, storage, deployment, and execution @@ -57,8 +57,8 @@ //! //! ## Supporting components //! -//! - [`cost_model`] — the [`CostModel`](cost_model::CostModel) trait every -//! deployment's cost-based sketch selection plugs into (issues #6, #33). +//! - [`cost_model`] — the [`CostModel`](cost_model::CostModel) trait through +//! which a deployment supplies the costs Planner's selection uses (issues #6, #33). //! `asap-plan` itself only ships [`DefaultCostModel`](cost_model::DefaultCostModel), //! which preserves [`replacement`]'s built-in static preference order and //! — via [`CostModel::estimate_cost`](cost_model::CostModel::estimate_cost) @@ -73,7 +73,7 @@ //! where, reusing the candidate's own rationale rather than inventing new //! prose), meant for the same downstream consumer (e.g. a //! DAG-visualization view) the crate doc's planning workflows section above -//! already names for [`replacement::CandidatePostASAPDAGs`] itself. Superseded PR +//! already names for [`replacement::CandidateLogicalPostASAPDAGs`] itself. Superseded PR //! #247's own rule-based traversal, which re-walked the tree once per //! optimization before [`replacement::search_workload`] existed to read //! from instead — see that module's docs for the full reframing. @@ -105,7 +105,7 @@ //! pair under the same grouping, re-divided back by a wrapping `Project`, //! so those *are* ordinary mergeable accumulators sharing/sketching can //! reach. It only reshapes; [`replacement::search_workload`]'s cost-based -//! ranking (or a downstream consumer reading [`replacement::CandidatePostASAPDAGs`]) +//! ranking (or a downstream consumer reading [`replacement::CandidateLogicalPostASAPDAGs`]) //! is what decides whether the reshaped form is actually worth picking, //! the same propose-don't-decide split every other strategy here keeps. //! @@ -208,11 +208,12 @@ pub use recurrence::{ }; pub use replacement::{ default_strategies, default_strategies_with, search_workload, search_workload_with, - search_workload_with_targets, summary_candidates, CandidatePostASAPDAGs, CompositionDecision, - GlobalSelection, Matcher, Proposals, RankedTargetSubDAGCandidates, Realization, - RealizationError, RecurrenceProfileMap, RejectedCandidate, Replacement, ReplacementProvenance, - ReplacementStrategy, ReplacementSubDAG, SharedSubtreeStrategy, SketchAlgorithmStrategy, - TargetSubDAG, TargetSubDAGCandidates, TargetSubDAGSelection, MAX_SEARCH_ITERATIONS, + search_workload_with_targets, summary_candidates, CandidateLogicalPostASAPDAGs, + CompositionDecision, GlobalSelection, Matcher, Proposals, RankedTargetSubDAGCandidates, + Realization, RealizationError, RecurrenceProfileMap, RejectedCandidate, Replacement, + ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, SharedSubtreeStrategy, + SketchAlgorithmStrategy, TargetSubDAG, TargetSubDAGCandidates, TargetSubDAGSelection, + MAX_SEARCH_ITERATIONS, }; pub use rewrite::{AvgToSumOverCountStrategy, SemanticEquivalentRewriteStrategy}; pub use summary_maintenance_dag_export::{ @@ -221,15 +222,20 @@ pub use summary_maintenance_dag_export::{ }; pub use summary_maintenance_lifecycle::{ assemble_selected_dag_with_summary_maintenance_lifecycles, - enumerate_summary_maintenance_lifecycles, global_selection_with_summary_maintenance_lifecycles, - plan_summary_maintenance_lifecycles, SummaryMaintenanceCapabilities, + global_selection_with_summary_maintenance_lifecycles, plan_summary_maintenance_lifecycles, + LifecyclePostASAPDAG, LifecyclePostASAPDAGError, SummaryMaintenanceCapabilities, SummaryMaintenanceDeployment, SummaryMaintenanceLifecycleAlternative, - SummaryMaintenanceLifecycleAssemblyError, SummaryMaintenanceLifecycleCandidates, - SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecycleChoiceError, - SummaryMaintenanceLifecycleCostInputs, SummaryMaintenanceLifecyclePlan, - SummaryMaintenanceLifecyclePlanError, SummaryMaintenanceLifecycleRejection, - SummaryMaintenanceLifecycleSelectionError, SummaryMaintenanceTimingError, WorkloadDemand, + SummaryMaintenanceLifecycleAssemblyError, SummaryMaintenanceLifecycleCapabilities, + SummaryMaintenanceLifecycleChoiceError, SummaryMaintenanceLifecycleCostInputs, + SummaryMaintenanceLifecycleRejection, SummaryMaintenanceLifecycleSelectionError, + SummaryMaintenanceTimingError, WorkloadDemand, }; pub use topk_reuse::TopKLimitReuseStrategy; pub mod maintained_population; + +mod candidate_timing; +pub use candidate_timing::{ + CandidateLifecyclePostASAPDAGs, CandidateTimingContext, CandidateTimingError, + PostASAPCandidateMetadata, +}; diff --git a/crates/asap-aware-mapping/src/maintained_population.rs b/crates/asap-aware-mapping/src/maintained_population.rs index 29cd3c54..81587b37 100644 --- a/crates/asap-aware-mapping/src/maintained_population.rs +++ b/crates/asap-aware-mapping/src/maintained_population.rs @@ -3,11 +3,11 @@ use crate::replacement::{ Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_types::post_asap::{ - maintained_population::*, ExecutionTiming, ResultGuarantee, SummaryExpr, SummaryFamilyType, - SummaryField, SummaryNode, SummarySchema, ValueOperation, + maintained_population::*, ExecutionTiming, PostASAPNode, ResultGuarantee, SummaryExpr, + SummaryFamilyType, SummaryField, SummarySchema, ValueOperation, }; use asap_types::pre_asap::{ - AggIntent, CompareOpKind, DataType, QueryExpr, Reduction, ScalarValue, Schema, Source, + AggIntent, CompareOpKind, DataType, PreASAPNode, Reduction, ScalarValue, Schema, Source, }; use std::rc::Rc; @@ -26,17 +26,19 @@ fn plain(schema: Schema) -> SummarySchema { } } -fn strip_projection(mut root: &QueryExpr) -> &QueryExpr { - while let QueryExpr::Project { child, .. } = root { +fn strip_projection(mut root: &PreASAPNode) -> &PreASAPNode { + while let PreASAPNode::Project { child, .. } = root { root = child; } root } -fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadout, Rc)> { +fn recognize( + root: &PreASAPNode, +) -> Option<(MaintainedPopulation, PopulationReadout, Rc)> { let root = strip_projection(root); let (source, grouping, readout, value_column) = match root { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { child, reduction: Reduction::Reduce(grouping), measures, @@ -62,12 +64,12 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou } (child, grouping, readout, col) } - QueryExpr::Limit { + PreASAPNode::Limit { n, offset: 0, child, } => { - let QueryExpr::Sort { + let PreASAPNode::Sort { child, keys, partition_by, @@ -78,7 +80,7 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou let [key] = keys.as_slice() else { return None; }; - let QueryExpr::Column(col) = &key.expr else { + let PreASAPNode::Column(col) = &key.expr else { return None; }; if key.ascending { @@ -93,7 +95,7 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou } _ => return None, }; - if let QueryExpr::Scan { + if let PreASAPNode::Scan { source: Source::Table { .. }, schema, .. @@ -123,7 +125,7 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou // temporal input scope. Membership must expire at that horizon; retain // the wrapper as the maintained input so validation can check agreement. let (series_source, lookback_ms) = match source.as_ref() { - QueryExpr::TimeRange { range, child } => { + PreASAPNode::TimeRange { range, child } => { let ms = u64::try_from(range.as_millis()).ok()?; if ms == 0 || std::time::Duration::from_millis(ms) != *range { return None; @@ -132,7 +134,7 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou } other => (other, 300_000), }; - let QueryExpr::Scan { + let PreASAPNode::Scan { source: Source::TimeSeries { metric }, predicates, schema, @@ -156,10 +158,10 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou }; let mut matchers = Vec::new(); for predicate in predicates { - let QueryExpr::Compare { left, op, right } = predicate.0.as_ref() else { + let PreASAPNode::Compare { left, op, right } = predicate.0.as_ref() else { return None; }; - let (QueryExpr::Column(col), QueryExpr::Literal(ScalarValue::Utf8(value))) = + let (PreASAPNode::Column(col), PreASAPNode::Literal(ScalarValue::Utf8(value))) = (left.as_ref(), right.as_ref()) else { return None; @@ -208,23 +210,23 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou /// population updates and price the maintenance/readout boundary. /// The population is exact; max_k bounds the shared readout cache, not its members. pub struct MaintainedPopulationStrategy { - roots: Vec>, + roots: Vec>, } impl MaintainedPopulationStrategy { - pub fn new(roots: &[Rc]) -> Self { + pub fn new(roots: &[Rc]) -> Self { Self { roots: roots.to_vec(), } } - pub fn candidate(&self, root: &Rc) -> Option> { - if let QueryExpr::Project { + pub fn candidate(&self, root: &Rc) -> Option> { + if let PreASAPNode::Project { cols, qualifier, child, } = root.as_ref() { let child = self.candidate(child)?; - return Some(Rc::new(SummaryNode { + return Some(Rc::new(PostASAPNode { guarantee: child.guarantee.clone(), schema: plain(root.output_schema().ok()?), expr: SummaryExpr::ValueOperation { @@ -253,16 +255,16 @@ impl MaintainedPopulationStrategy { } } let input_schema = plain(source.output_schema().ok()?); - let scan = Rc::new(SummaryNode { + let scan = Rc::new(PostASAPNode { expr: SummaryExpr::KeepPreAsap(source), schema: input_schema.clone(), guarantee: Some(ResultGuarantee::exact("source samples")), }); // Query time is only the initial layout: whether the population is // retained at ingestion or rebuilt per query is its lifecycle choice - // (`SummaryMaintenanceLifecyclePlan::execution_timed_dag`). The readout + // (`LifecyclePostASAPDAG::export_timed_dag`). The readout // and projection above it are query-time by construction. - let maintained = Rc::new(SummaryNode { + let maintained = Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: scan, operation: ValueOperation::MaintainPopulation { population }, @@ -273,7 +275,7 @@ impl MaintainedPopulationStrategy { "exact members under the declared population semantics", )), }); - Some(Rc::new(SummaryNode { + Some(Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: maintained, operation: ValueOperation::ReadPopulation { readout }, @@ -307,9 +309,9 @@ impl ReplacementStrategy for MaintainedPopulationStrategy { mod tests { use super::*; use crate::test_support::lower_promql; - use asap_types::post_asap::{compile_post_asap_dag, share_common_summary_subtrees}; + use asap_types::post_asap::{export_post_asap_dag, share_common_summary_subtrees}; - fn lower(q: &str) -> Rc { + fn lower(q: &str) -> Rc { Rc::new(lower_promql(q, asap_types::types::AccuracyTarget::Exact)) } @@ -331,7 +333,7 @@ mod tests { let candidate = rule .candidate(&root) .expect("current-series rule candidate"); - compile_post_asap_dag(&candidate).expect("typed post-ASAP DAG"); + export_post_asap_dag(&candidate).expect("typed post-ASAP DAG"); } } @@ -368,7 +370,7 @@ mod tests { ); let mut producers = Vec::new(); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + export_post_asap_dag(plan).unwrap(); let SummaryExpr::ValueOperation { child, operation: ValueOperation::ReadPopulation { .. }, @@ -457,7 +459,7 @@ mod tests { unreachable!() }; *timing = population; - compile_post_asap_dag(&Rc::new(node)) + export_post_asap_dag(&Rc::new(node)) }; use ExecutionTiming::{IngestionTime, QueryTime}; assert!(with_timings(IngestionTime, QueryTime).is_ok()); @@ -477,7 +479,7 @@ mod tests { *operation = ValueOperation::ReadPopulation { readout: PopulationReadout::TopK { k: 6 }, }; - assert!(compile_post_asap_dag(&Rc::new(bad.clone())).is_err()); + assert!(export_post_asap_dag(&Rc::new(bad.clone())).is_err()); let SummaryExpr::ValueOperation { child, operation, .. } = &mut bad.expr @@ -499,6 +501,6 @@ mod tests { unreachable!() }; spec.metric = "b".into(); - assert!(compile_post_asap_dag(&Rc::new(bad)).is_err()); + assert!(export_post_asap_dag(&Rc::new(bad)).is_err()); } } diff --git a/crates/asap-aware-mapping/src/pass/major.rs b/crates/asap-aware-mapping/src/pass/major.rs index a7ec9fdd..655c82f9 100644 --- a/crates/asap-aware-mapping/src/pass/major.rs +++ b/crates/asap-aware-mapping/src/pass/major.rs @@ -8,7 +8,7 @@ use std::rc::Rc; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::pre_asap::query_expr::PreASAPNode; use asap_types::types::AccuracyTarget; use super::{ @@ -40,7 +40,7 @@ impl OptimizationPass for MajorPass { // search result carries the workload binding the lifecycle stage and // the output both need. CSE may make two identical queries share one // `Rc`, but it never drops or reorders a root, so this stays aligned. - let roots: Vec<(usize, Rc, Option)> = workload + let roots: Vec<(usize, Rc, Option)> = workload .entries() .enumerate() .map(|(index, (entry, expr))| { @@ -72,7 +72,7 @@ impl OptimizationPass for MajorPass { return Ok(PlanOutput::Dag { plans }); }; - // One index per root, in `PlanSpace::roots` order — which is the order + // One index per root, in `CandidateLogicalPostASAPDAGs::roots` order — which is the order // the roots went in, which is `entries()` order. let entry_indices: Vec = (0..workload.len()).collect(); let demand = WorkloadDemand { diff --git a/crates/asap-aware-mapping/src/pass/mod.rs b/crates/asap-aware-mapping/src/pass/mod.rs index 2329ee3a..df32559b 100644 --- a/crates/asap-aware-mapping/src/pass/mod.rs +++ b/crates/asap-aware-mapping/src/pass/mod.rs @@ -2,7 +2,7 @@ //! //! An [`OptimizationPass`] is the whole optimization stage behind one //! signature: pre-ASAP IR in, post-ASAP DAG out. The trait deliberately names -//! none of this crate's two-phase vocabulary — no `PlanSpace`, no +//! none of this crate's two-phase vocabulary — no `CandidateLogicalPostASAPDAGs`, no //! `TargetSubDAGCandidates`, no `ReplacementStrategy` — so an algorithm with no //! candidate-generation phase at all (a greedy MQO loop, say) can implement it //! without pretending to have phases it does not have. The shipped algorithm is @@ -18,7 +18,7 @@ use std::collections::BTreeMap; use std::rc::Rc; use asap_types::parsed_workload::ParsedWorkload; -use asap_types::post_asap::SummaryNode; +use asap_types::post_asap::PostASAPNode; use asap_types::workload::WorkloadError; use crate::accuracy::{ @@ -28,8 +28,8 @@ use crate::cost_model::{CostModel, DefaultCostModel}; use crate::recurrence::Horizon; use crate::replacement::RealizationError; use crate::summary_maintenance_lifecycle::{ - SummaryMaintenanceLifecycleAssemblyError, SummaryMaintenanceLifecycleCapabilities, - SummaryMaintenanceLifecyclePlan, SummaryMaintenanceLifecycleSelectionError, + LifecyclePostASAPDAG, SummaryMaintenanceLifecycleAssemblyError, + SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecycleSelectionError, }; pub use major::MajorPass; @@ -170,15 +170,15 @@ pub enum OptimizationInputError { pub struct QueryPlan { /// Index into `QueryWorkload::entries()`. pub entry_index: usize, - pub dag: Rc, + pub dag: Rc, } -/// One query's DAG plus the maintenance decisions taken for it. The DAG is +/// One query's DAG annotated with its lifecycle assignment. The DAG is /// `plan.root` — this is not a representation parallel to [`QueryPlan`]. #[derive(Debug, Clone)] pub struct QueryLifecyclePlan { pub entry_index: usize, - pub plan: SummaryMaintenanceLifecyclePlan, + pub plan: LifecyclePostASAPDAG, } /// One variant per workflow. Which one comes back is decided by @@ -200,7 +200,7 @@ impl PlanOutput { } /// The selected DAG root per query, whichever variant this is. - pub fn dags(&self) -> Vec> { + pub fn dags(&self) -> Vec> { match self { Self::Dag { plans } => plans.iter().map(|p| Rc::clone(&p.dag)).collect(), Self::DagWithLifecycle { plans } => { diff --git a/crates/asap-aware-mapping/src/physical_operator_statistics.rs b/crates/asap-aware-mapping/src/physical_operator_statistics.rs index 9cf8045b..15c4f7f2 100644 --- a/crates/asap-aware-mapping/src/physical_operator_statistics.rs +++ b/crates/asap-aware-mapping/src/physical_operator_statistics.rs @@ -278,7 +278,7 @@ pub struct PartitionStatistics { /// is the authoritative operator vocabulary: every one of its variants has a /// matching statistics variant here. /// -/// This enum intentionally does not mirror either logical IR. `QueryExpr` and +/// This enum intentionally does not mirror either logical IR. `PreASAPNode` and /// `SummaryExpr` are inputs to physical lowering, and one logical node may /// expand into several physical nodes or choose among several algorithms. /// Physical configuration such as a Top-K limit or hash-join build side lives diff --git a/crates/asap-aware-mapping/src/physical_plan_cost_model.rs b/crates/asap-aware-mapping/src/physical_plan_cost_model.rs index a8cb38d7..95399872 100644 --- a/crates/asap-aware-mapping/src/physical_plan_cost_model.rs +++ b/crates/asap-aware-mapping/src/physical_plan_cost_model.rs @@ -2,14 +2,14 @@ use std::{cell::RefCell, rc::Rc}; -use asap_types::post_asap::{SketchAlgorithm, SummaryExpr, SummaryNode}; -use asap_types::pre_asap::{AggIntent, QueryExpr}; +use asap_types::post_asap::{PostASAPNode, SketchAlgorithm, SummaryExpr}; +use asap_types::pre_asap::{AggIntent, PreASAPNode}; use asap_types::resources::CacheProfile; use crate::analytical_cost::{ - estimate_physical_dag_comparison, AnalyticalCostError, - EvidenceBackedPhysicalDag as PhysicalDag, PhysicalDagComparisonEstimate, - PhysicalDagEstimateRequest, PhysicalNodeEvidence, ResourceCalibration, + estimate_physical_dag_comparison, AnalyticalCostError, EvidenceBackedPhysicalDag, + PhysicalDagComparisonEstimate, PhysicalDagEstimateRequest, PhysicalNodeEvidence, + ResourceCalibration, }; use crate::cost_model::{Cost, CostModel, DefaultCostModel}; use crate::physical_operator_statistics::ComparisonScope; @@ -57,9 +57,9 @@ pub trait PlannerPhysicalPlanProvider { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - summary: &Rc, + summary: &Rc, target: &TargetSubDAG<'_>, - ) -> Result; + ) -> Result; } /// Dimensional comparison retained for explanations and verification. @@ -91,10 +91,10 @@ pub struct PhysicalPlanCostModel<'a> { } struct CachedTargetEvidence { - root: Rc, + root: Rc, consumer_count: usize, snapshot: PhysicalEvidenceSnapshot, - raw: PhysicalDag, + raw: EvidenceBackedPhysicalDag, } impl<'a> PhysicalPlanCostModel<'a> { @@ -124,7 +124,7 @@ impl<'a> PhysicalPlanCostModel<'a> { fn target_evidence( &self, target: &TargetSubDAG<'_>, - ) -> Result<(PhysicalEvidenceSnapshot, PhysicalDag), AnalyticalCostError> { + ) -> Result<(PhysicalEvidenceSnapshot, EvidenceBackedPhysicalDag), AnalyticalCostError> { if let Some(cached) = self.target_evidence.borrow().iter().find(|cached| { Rc::ptr_eq(&cached.root, target.root) && cached.consumer_count == target.consumer_count }) { @@ -359,7 +359,7 @@ mod tests { use std::cell::Cell; use std::collections::HashMap; - use asap_types::pre_asap::{Column, DataType, QueryExpr, Reduction, Schema, Source}; + use asap_types::pre_asap::{Column, DataType, PreASAPNode, Reduction, Schema, Source}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, @@ -405,15 +405,15 @@ mod tests { } } - fn query() -> Rc { - Rc::new(QueryExpr::Aggregate { + fn query() -> Rc { + Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.01), }], output_names: vec![], having: None, - child: Rc::new(QueryExpr::Scan { + child: Rc::new(PreASAPNode::Scan { source: Source::Table { table_ref: "events".into(), }, @@ -469,7 +469,7 @@ mod tests { } } - fn summary_dag(&self, scope: &ComparisonScope) -> PhysicalDag { + fn summary_dag(&self, scope: &ComparisonScope) -> EvidenceBackedPhysicalDag { let scan_statistics = scan_statistics(self.candidate_scan_bytes, edge(100, 800)); let aggregate_statistics = aggregate_statistics(edge(100, 800), edge(1, 8)); let read_statistics = pass_through_statistics(edge(1, 8)); @@ -499,7 +499,7 @@ mod tests { }, ), ]); - PhysicalDag { + EvidenceBackedPhysicalDag { nodes: vec![ PhysicalDagNode { id: "candidate-scan".into(), @@ -582,9 +582,9 @@ mod tests { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - _summary: &Rc, + _summary: &Rc, _target: &TargetSubDAG<'_>, - ) -> Result { + ) -> Result { assert_eq!(snapshot.version, "test-snapshot-1"); self.summary_available .then(|| self.summary_dag(&snapshot.scope)) @@ -912,9 +912,9 @@ mod tests { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - summary: &Rc, + summary: &Rc, target: &TargetSubDAG<'_>, - ) -> Result { + ) -> Result { self.0.summary_physical_dag(snapshot, summary, target) } } @@ -1000,9 +1000,9 @@ mod tests { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - summary: &Rc, + summary: &Rc, target: &TargetSubDAG<'_>, - ) -> Result { + ) -> Result { let mut dag = self.0.summary_physical_dag(snapshot, summary, target)?; dag.nodes[0] .source_coverage @@ -1052,9 +1052,9 @@ mod tests { fn summary_physical_dag( &self, _snapshot: &PhysicalEvidenceSnapshot, - _summary: &Rc, + _summary: &Rc, _target: &TargetSubDAG<'_>, - ) -> Result { + ) -> Result { panic!("blank snapshot versions must fail before summary binding") } } diff --git a/crates/asap-aware-mapping/src/query_physical_lowering.rs b/crates/asap-aware-mapping/src/query_physical_lowering.rs index 1cc2b1ca..1237cd2e 100644 --- a/crates/asap-aware-mapping/src/query_physical_lowering.rs +++ b/crates/asap-aware-mapping/src/query_physical_lowering.rs @@ -13,7 +13,7 @@ use crate::physical_operator_statistics::{ }; pub struct PhysicalNodeRequest<'a> { - pub logical_node: &'a asap_types::pre_asap::QueryExpr, + pub logical_node: &'a asap_types::pre_asap::PreASAPNode, pub operator: PhysicalOperator, pub occurrence: usize, pub synthetic: bool, @@ -46,13 +46,13 @@ where /// complete query unavailable. Scalar expressions remain part of their /// containing operator's local cost. pub fn lower_query_physical_dag( - root: &Rc, + root: &Rc, scope: &ComparisonScope, evidence: &dyn PhysicalNodeEvidenceProvider, ) -> Result { use std::collections::HashMap; - use asap_types::pre_asap::{GroupKeys, QueryExpr, RelationalSetOpKind}; + use asap_types::pre_asap::{GroupKeys, PreASAPNode, RelationalSetOpKind}; scope.validate()?; @@ -65,7 +65,7 @@ pub fn lower_query_physical_dag( } impl Lowerer<'_> { - fn lower(&mut self, query: &QueryExpr) -> Result { + fn lower(&mut self, query: &PreASAPNode) -> Result { let occurrence = self.next_id; self.next_id += 1; self.lower_new(query, occurrence) @@ -73,7 +73,7 @@ pub fn lower_query_physical_dag( fn resolve( &self, - query: &QueryExpr, + query: &PreASAPNode, operator: PhysicalOperator, occurrence: usize, synthetic: bool, @@ -128,10 +128,10 @@ pub fn lower_query_physical_dag( fn lower_unary( &mut self, - query: &QueryExpr, + query: &PreASAPNode, occurrence: usize, operator: PhysicalOperator, - child: &QueryExpr, + child: &PreASAPNode, ) -> Result { let child_id = self.lower(child)?; let children = vec![child_id.clone()]; @@ -151,17 +151,17 @@ pub fn lower_query_physical_dag( fn lower_promql_unary( &mut self, - query: &QueryExpr, + query: &PreASAPNode, occurrence: usize, operator: PhysicalOperator, - child: &QueryExpr, + child: &PreASAPNode, ) -> Result { self.lower_unary(query, occurrence, operator, child) } fn lower_promql_scalar_leaf( &mut self, - query: &QueryExpr, + query: &PreASAPNode, occurrence: usize, ) -> Result { let operator = PhysicalOperator::PromqlScalarLeaf; @@ -182,11 +182,11 @@ pub fn lower_query_physical_dag( fn lower_new( &mut self, - query: &QueryExpr, + query: &PreASAPNode, occurrence: usize, ) -> Result { match query { - QueryExpr::Scan { + PreASAPNode::Scan { source, predicates, .. } => { let coverage = bind_scan_coverage( @@ -252,13 +252,13 @@ pub fn lower_query_physical_dag( require_operator_statistics(filter_operator, &filter_evidence.statistics)?; self.push(filter_evidence, filter_operator, children, None) } - QueryExpr::Filter { pred, child } => { + PreASAPNode::Filter { pred, child } => { let operator = PhysicalOperator::Filter { predicate_operations_per_row: scalar_operation_count(&pred.0)?.max(1), }; self.lower_unary(query, occurrence, operator, child) } - QueryExpr::Project { cols, child, .. } => { + PreASAPNode::Project { cols, child, .. } => { let expression_operations_per_row = cols .iter() .try_fold(0_u64, |total, item| { @@ -280,7 +280,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction, measures, having, @@ -334,7 +334,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Dedup { cols, child } => { + PreASAPNode::Dedup { cols, child } => { let key_count = if cols.is_empty() { child .output_schema() @@ -357,7 +357,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Sort { + PreASAPNode::Sort { keys, partition_by, child, @@ -377,8 +377,8 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Limit { n, offset, child } => { - if let QueryExpr::Sort { + PreASAPNode::Limit { n, offset, child } => { + if let PreASAPNode::Sort { keys, partition_by, child: sorted_child, @@ -438,7 +438,7 @@ pub fn lower_query_physical_dag( require_operator_statistics(operator, statistics)?; self.push(evidence, operator, children, None) } - QueryExpr::SQLWindowFunc { + PreASAPNode::SQLWindowFunc { func, partition_by, order_by, @@ -468,7 +468,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::TimeRange { range, child } => { + PreASAPNode::TimeRange { range, child } => { let range_millis = duration_millis(*range, "range")?; self.lower_promql_unary( query, @@ -477,7 +477,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::PromqlSubquery { + PreASAPNode::PromqlSubquery { range, resolution, child, @@ -509,7 +509,7 @@ pub fn lower_query_physical_dag( } Ok(id) } - QueryExpr::PromqlRelabel { value, child, .. } => self.lower_promql_unary( + PreASAPNode::PromqlRelabel { value, child, .. } => self.lower_promql_unary( query, occurrence, PhysicalOperator::PromqlRelabel { @@ -517,7 +517,7 @@ pub fn lower_query_physical_dag( }, child, ), - QueryExpr::PromqlSeriesSample { + PreASAPNode::PromqlSeriesSample { by, kind, child, .. } => { if by.is_without() { @@ -553,7 +553,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::PromqlInfoEnrich { selector, child } => { + PreASAPNode::PromqlInfoEnrich { selector, child } => { let left_id = self.lower(child)?; let coverage = bind_info_coverage( &format!("occurrence-{occurrence}-info"), @@ -605,7 +605,7 @@ pub fn lower_query_physical_dag( require_operator_statistics(operator, &evidence.statistics)?; self.push(evidence, operator, children, None) } - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op, lhs, rhs, @@ -666,41 +666,41 @@ pub fn lower_query_physical_dag( require_operator_statistics(operator, &evidence.statistics)?; self.push(evidence, operator, children, None) } - QueryExpr::PromqlVectorFromScalar(child) => self.lower_promql_unary( + PreASAPNode::PromqlVectorFromScalar(child) => self.lower_promql_unary( query, occurrence, PhysicalOperator::PromqlScalarToVector, child, ), - QueryExpr::PromqlScalarFromVector(child) => self.lower_promql_unary( + PreASAPNode::PromqlScalarFromVector(child) => self.lower_promql_unary( query, occurrence, PhysicalOperator::PromqlVectorToScalar, child, ), - QueryExpr::PromqlScalarBridge(inner) + PreASAPNode::PromqlScalarBridge(inner) if matches!( inner.as_ref(), - QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Float64(_)) + PreASAPNode::Literal(asap_types::pre_asap::ScalarValue::Float64(_)) ) => { self.lower_promql_scalar_leaf(query, occurrence) } - QueryExpr::EvalTimestamp => self.lower_promql_scalar_leaf(query, occurrence), - QueryExpr::TimeShift { shift, child } => { + PreASAPNode::EvalTimestamp => self.lower_promql_scalar_leaf(query, occurrence), + PreASAPNode::TimeShift { shift, child } => { if !shift.is_identity() { return Err(AnalyticalCostError::UnsupportedQueryOperator); } self.lower_unary(query, occurrence, PhysicalOperator::PassThrough, child) } - QueryExpr::Concat { children, .. } => { + PreASAPNode::Concat { children, .. } => { let child_ids = children .iter() .map(|child| self.lower(child)) .collect::, _>>()?; self.lower_concat(query, occurrence, child_ids) } - QueryExpr::SetOp { + PreASAPNode::SetOp { kind: RelationalSetOpKind::Union, all: true, left, @@ -710,7 +710,7 @@ pub fn lower_query_physical_dag( let right_id = self.lower(right)?; self.lower_concat(query, occurrence, vec![left_id, right_id]) } - QueryExpr::Join { + PreASAPNode::Join { kind, pred, left, @@ -760,7 +760,7 @@ pub fn lower_query_physical_dag( fn lower_concat( &mut self, - query: &QueryExpr, + query: &PreASAPNode, occurrence: usize, child_ids: Vec, ) -> Result { @@ -1047,11 +1047,11 @@ fn promql_vector_cardinality( } fn hash_join_key_count( - expr: &asap_types::pre_asap::QueryExpr, - left: &asap_types::pre_asap::QueryExpr, - right: &asap_types::pre_asap::QueryExpr, + expr: &asap_types::pre_asap::PreASAPNode, + left: &asap_types::pre_asap::PreASAPNode, + right: &asap_types::pre_asap::PreASAPNode, ) -> Option { - use asap_types::pre_asap::{CompareOpKind, QueryExpr}; + use asap_types::pre_asap::{CompareOpKind, PreASAPNode}; let (Ok(left_schema), Ok(right_schema)) = (left.output_schema(), right.output_schema()) else { return None; @@ -1069,14 +1069,14 @@ fn hash_join_key_count( } } - fn predicate(expr: &QueryExpr, left_width: usize, total_width: usize) -> Option { + fn predicate(expr: &PreASAPNode, left_width: usize, total_width: usize) -> Option { match expr { - QueryExpr::Compare { + PreASAPNode::Compare { left, op: CompareOpKind::Eq, right, } => match (left.as_ref(), right.as_ref()) { - (QueryExpr::Column(left), QueryExpr::Column(right)) => match ( + (PreASAPNode::Column(left), PreASAPNode::Column(right)) => match ( column_side(*left, left_width, total_width), column_side(*right, left_width, total_width), ) { @@ -1085,7 +1085,7 @@ fn hash_join_key_count( }, _ => None, }, - QueryExpr::BoolAnd(parts) if !parts.is_empty() => { + PreASAPNode::BoolAnd(parts) if !parts.is_empty() => { parts.iter().try_fold(0_u64, |count, part| { count.checked_add(predicate(part, left_width, total_width)?) }) @@ -1098,11 +1098,11 @@ fn hash_join_key_count( } fn scalar_operation_count( - expr: &asap_types::pre_asap::QueryExpr, + expr: &asap_types::pre_asap::PreASAPNode, ) -> Result { - use asap_types::pre_asap::QueryExpr; + use asap_types::pre_asap::PreASAPNode; - let add = |parts: &[&QueryExpr]| { + let add = |parts: &[&PreASAPNode]| { parts.iter().try_fold(0_u64, |total, part| { total .checked_add(scalar_operation_count(part)?) @@ -1115,14 +1115,14 @@ fn scalar_operation_count( .ok_or(AnalyticalCostError::Overflow) }; match expr { - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => Ok(0), - QueryExpr::Compare { left, right, .. } | QueryExpr::Arithmetic { left, right, .. } => { + PreASAPNode::Column(_) + | PreASAPNode::Literal(_) + | PreASAPNode::EvalTimestamp + | PreASAPNode::CurrentTimestamp => Ok(0), + PreASAPNode::Compare { left, right, .. } | PreASAPNode::Arithmetic { left, right, .. } => { with_local(&[left, right]) } - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + PreASAPNode::BoolAnd(parts) | PreASAPNode::BoolOr(parts) => { let children = parts.iter().collect::>(); add(&children)? .checked_add( @@ -1131,12 +1131,12 @@ fn scalar_operation_count( ) .ok_or(AnalyticalCostError::Overflow) } - QueryExpr::Not(child) - | QueryExpr::IsNull(child) - | QueryExpr::IsNotNull(child) - | QueryExpr::PromqlScalarBridge(child) => with_local(&[child]), - QueryExpr::Cast { expr, .. } => with_local(&[expr]), - QueryExpr::InList { expr, list, .. } => { + PreASAPNode::Not(child) + | PreASAPNode::IsNull(child) + | PreASAPNode::IsNotNull(child) + | PreASAPNode::PromqlScalarBridge(child) => with_local(&[child]), + PreASAPNode::Cast { expr, .. } => with_local(&[expr]), + PreASAPNode::InList { expr, list, .. } => { let mut children = Vec::with_capacity(list.len() + 1); children.push(expr.as_ref()); children.extend(list.iter()); @@ -1144,11 +1144,11 @@ fn scalar_operation_count( .checked_add(u64::try_from(list.len()).map_err(|_| AnalyticalCostError::Overflow)?) .ok_or(AnalyticalCostError::Overflow) } - QueryExpr::FunctionCall { args, .. } => { + PreASAPNode::FunctionCall { args, .. } => { let children = args.iter().collect::>(); with_local(&children) } - QueryExpr::Case { + PreASAPNode::Case { operand, branches, else_expr, @@ -1242,13 +1242,13 @@ fn fixed_state_per_series_intent(intent: &asap_types::pre_asap::AggIntent) -> bo ) } -fn is_promql_scalar(query: &asap_types::pre_asap::QueryExpr) -> bool { - use asap_types::pre_asap::QueryExpr; +fn is_promql_scalar(query: &asap_types::pre_asap::PreASAPNode) -> bool { + use asap_types::pre_asap::PreASAPNode; matches!( query, - QueryExpr::PromqlScalarBridge(_) - | QueryExpr::PromqlScalarFromVector(_) - | QueryExpr::EvalTimestamp + PreASAPNode::PromqlScalarBridge(_) + | PreASAPNode::PromqlScalarFromVector(_) + | PreASAPNode::EvalTimestamp ) } @@ -1448,17 +1448,17 @@ mod tests { #[test] fn correlation_lowers_to_physical_hash_aggregate() { use asap_types::pre_asap::{ - AggIntent, Column, DataType, QueryExpr, Reduction, Schema, Source, + AggIntent, Column, DataType, PreASAPNode, Reduction, Schema, Source, }; let source = Source::Table { table_ref: "pairs".into(), }; - let root = Rc::new(QueryExpr::Aggregate { + let root = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::PearsonCorr { left: 0, right: 1 }], output_names: vec!["r".into()], having: None, - child: Rc::new(QueryExpr::Scan { + child: Rc::new(PreASAPNode::Scan { source: source.clone(), predicates: vec![], schema: Schema::new(vec![ @@ -1496,39 +1496,39 @@ mod tests { #[test] fn query_lowering_recurses_and_fuses_global_sort_limit() { - use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr, Reduction, SortKey, Source}; + use asap_types::pre_asap::{AggIntent, GroupKeys, PreASAPNode, Reduction, SortKey, Source}; use asap_types::pre_asap::{Column, DataType, Schema}; use std::rc::Rc; - let scan = Rc::new(QueryExpr::Scan { + let scan = Rc::new(PreASAPNode::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![asap_types::pre_asap::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Boolean(true)), + PreASAPNode::Literal(asap_types::pre_asap::ScalarValue::Boolean(true)), ))], schema: Schema::new(vec![ Column::new("service", DataType::Utf8, false), Column::new("value", DataType::Float64, false), ]), }); - let aggregate = Rc::new(QueryExpr::Aggregate { + let aggregate = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![0]), measures: vec![AggIntent::Sum { col: Some(1) }], output_names: vec![], having: None, child: Rc::clone(&scan), }); - let sort = Rc::new(QueryExpr::Sort { + let sort = Rc::new(PreASAPNode::Sort { keys: vec![SortKey { - expr: QueryExpr::Column(0), + expr: PreASAPNode::Column(0), ascending: false, nulls_first: false, }], partition_by: GroupKeys::none(), child: aggregate, }); - let root = Rc::new(QueryExpr::Limit { + let root = Rc::new(PreASAPNode::Limit { n: 10, offset: 5, child: sort, @@ -1539,7 +1539,7 @@ mod tests { table_ref: "events".into(), }, vec![asap_types::pre_asap::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Boolean(true)), + PreASAPNode::Literal(asap_types::pre_asap::ScalarValue::Boolean(true)), ))], ); let scope = scope(vec![scan_coverage]); @@ -1635,22 +1635,22 @@ mod tests { #[test] fn query_lowering_shares_only_provider_identified_physical_nodes() { use asap_types::pre_asap::{Column, CompareOpKind, DataType, Schema}; - use asap_types::pre_asap::{JoinKind, Predicate, QueryExpr, Source}; + use asap_types::pre_asap::{JoinKind, PreASAPNode, Predicate, Source}; use std::rc::Rc; - let shared = Rc::new(QueryExpr::Scan { + let shared = Rc::new(PreASAPNode::Scan { source: Source::Table { table_ref: "dimensions".into(), }, predicates: vec![], schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), }); - let root = Rc::new(QueryExpr::Join { + let root = Rc::new(PreASAPNode::Join { kind: JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + pred: Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), + right: Rc::new(PreASAPNode::Column(1)), })), left: Rc::clone(&shared), right: Rc::clone(&shared), @@ -1778,12 +1778,12 @@ mod tests { )) ); - let invalid = Rc::new(QueryExpr::Join { + let invalid = Rc::new(PreASAPNode::Join { kind: JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + pred: Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(0)), + right: Rc::new(PreASAPNode::Column(0)), })), left: Rc::clone(&shared), right: Rc::clone(&shared), @@ -1798,36 +1798,36 @@ mod tests { fn query_lowering_covers_relational_unary_operators() { use asap_types::pre_asap::{Column, DataType, ScalarValue, Schema}; use asap_types::pre_asap::{ - GroupKeys, Predicate, QueryExpr, SortKey, Source, TimeShift, WindowFuncKind, + GroupKeys, PreASAPNode, Predicate, SortKey, Source, TimeShift, WindowFuncKind, }; use std::rc::Rc; - let scan = Rc::new(QueryExpr::Scan { + let scan = Rc::new(PreASAPNode::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), }); - let filter = Rc::new(QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + let filter = Rc::new(PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(true)))), child: scan, }); - let project = Rc::new(QueryExpr::Project { + let project = Rc::new(PreASAPNode::Project { cols: vec![], qualifier: None, child: filter, }); - let dedup = Rc::new(QueryExpr::Dedup { + let dedup = Rc::new(PreASAPNode::Dedup { cols: vec![0], child: project, }); - let window = Rc::new(QueryExpr::SQLWindowFunc { + let window = Rc::new(PreASAPNode::SQLWindowFunc { func: WindowFuncKind::RowNumber, args: vec![], partition_by: GroupKeys::none(), order_by: vec![SortKey { - expr: QueryExpr::Column(0), + expr: PreASAPNode::Column(0), ascending: true, nulls_first: false, }], @@ -1835,21 +1835,21 @@ mod tests { output_name: "rn".into(), child: dedup, }); - let sort = Rc::new(QueryExpr::Sort { + let sort = Rc::new(PreASAPNode::Sort { keys: vec![SortKey { - expr: QueryExpr::Column(0), + expr: PreASAPNode::Column(0), ascending: true, nulls_first: false, }], partition_by: GroupKeys::by(vec![0]), child: window, }); - let limit = Rc::new(QueryExpr::Limit { + let limit = Rc::new(PreASAPNode::Limit { n: 20, offset: 0, child: sort, }); - let root = Rc::new(QueryExpr::TimeShift { + let root = Rc::new(PreASAPNode::TimeShift { shift: TimeShift::default(), child: limit, }); @@ -1971,17 +1971,17 @@ mod tests { #[test] fn query_lowering_maps_concat_and_union_all_but_rejects_distinct_set_ops() { use asap_types::pre_asap::{Column, DataType, Schema}; - use asap_types::pre_asap::{QueryExpr, RelationalSetOpKind, Source}; + use asap_types::pre_asap::{PreASAPNode, RelationalSetOpKind, Source}; use std::rc::Rc; - let scan = |name: &str| QueryExpr::Scan { + let scan = |name: &str| PreASAPNode::Scan { source: Source::Table { table_ref: name.into(), }, predicates: vec![], schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), }; - let union = Rc::new(QueryExpr::SetOp { + let union = Rc::new(PreASAPNode::SetOp { kind: RelationalSetOpKind::Union, all: true, left: Rc::new(scan("a")), @@ -2030,14 +2030,14 @@ mod tests { )) ); - let concat = Rc::new(QueryExpr::Concat { + let concat = Rc::new(PreASAPNode::Concat { children: vec![scan("a"), scan("b")], discriminator_unique_key: None, }); let dag = lower_query_physical_dag(&concat, &scope, &scripted(&provided)).unwrap(); assert_eq!(dag.nodes.last().unwrap().operator, PhysicalOperator::Concat); - let distinct_union = Rc::new(QueryExpr::SetOp { + let distinct_union = Rc::new(PreASAPNode::SetOp { kind: RelationalSetOpKind::Union, all: false, left: Rc::new(scan("a")), @@ -2052,17 +2052,17 @@ mod tests { #[test] fn query_lowering_fails_closed_for_missing_or_inconsistent_statistics() { use asap_types::pre_asap::{Column, DataType, Schema}; - use asap_types::pre_asap::{QueryExpr, Source}; + use asap_types::pre_asap::{PreASAPNode, Source}; use std::rc::Rc; - let scan = Rc::new(QueryExpr::Scan { + let scan = Rc::new(PreASAPNode::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), }); - let root = Rc::new(QueryExpr::Project { + let root = Rc::new(PreASAPNode::Project { cols: vec![], qualifier: None, child: scan, @@ -2168,21 +2168,21 @@ mod tests { #[test] fn query_lowering_accepts_a_consistently_empty_edge() { use asap_types::pre_asap::{Column, DataType, ScalarValue, Schema}; - use asap_types::pre_asap::{Predicate, QueryExpr, Source}; + use asap_types::pre_asap::{PreASAPNode, Predicate, Source}; use std::rc::Rc; - let scan = Rc::new(QueryExpr::Scan { + let scan = Rc::new(PreASAPNode::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), }); - let filter = Rc::new(QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(false)))), + let filter = Rc::new(PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(false)))), child: scan, }); - let root = Rc::new(QueryExpr::Limit { + let root = Rc::new(PreASAPNode::Limit { n: 10, offset: 0, child: filter, @@ -2228,14 +2228,14 @@ mod tests { #[test] fn query_lowering_rejects_aggregates_without_a_hash_implementation() { use asap_types::pre_asap::{ - AggIntent, GroupKeys, QueryExpr, Reduction, Source, WindowFuncKind, + AggIntent, GroupKeys, PreASAPNode, Reduction, Source, WindowFuncKind, }; use asap_types::pre_asap::{Column, DataType, Schema}; use asap_types::types::AccuracyTarget; use std::rc::Rc; let scan = || { - Rc::new(QueryExpr::Scan { + Rc::new(PreASAPNode::Scan { source: Source::Table { table_ref: "events".into(), }, @@ -2243,7 +2243,7 @@ mod tests { schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), }) }; - let exact_quantile = Rc::new(QueryExpr::Aggregate { + let exact_quantile = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Quantile { col: Some(0), @@ -2254,25 +2254,25 @@ mod tests { having: None, child: scan(), }); - let empty_sort_limit = Rc::new(QueryExpr::Limit { + let empty_sort_limit = Rc::new(PreASAPNode::Limit { n: 10, offset: 0, - child: Rc::new(QueryExpr::Sort { + child: Rc::new(PreASAPNode::Sort { keys: vec![], partition_by: GroupKeys::none(), child: scan(), }), }); - let unsupported_window = Rc::new(QueryExpr::SQLWindowFunc { + let unsupported_window = Rc::new(PreASAPNode::SQLWindowFunc { func: WindowFuncKind::Lag, - args: vec![QueryExpr::Column(0)], + args: vec![PreASAPNode::Column(0)], partition_by: GroupKeys::none(), order_by: vec![], frame: None, output_name: "lag".into(), child: scan(), }); - let shifted = Rc::new(QueryExpr::TimeShift { + let shifted = Rc::new(PreASAPNode::TimeShift { shift: asap_types::pre_asap::TimeShift { offset_ms: 60_000, at: None, @@ -2301,14 +2301,14 @@ mod tests { #[test] fn scalar_work_counts_every_local_predicate_operation() { - use asap_types::pre_asap::{CompareOpKind, QueryExpr, ScalarValue}; + use asap_types::pre_asap::{CompareOpKind, PreASAPNode, ScalarValue}; - let comparison = || QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let comparison = || PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Int64(1))), }; - let predicate = QueryExpr::BoolAnd(vec![comparison(), comparison()]); + let predicate = PreASAPNode::BoolAnd(vec![comparison(), comparison()]); assert_eq!(scalar_operation_count(&predicate), Ok(3)); } @@ -2316,18 +2316,18 @@ mod tests { #[test] fn promql_presence_is_lowered_with_a_per_step_output_bound() { use asap_types::pre_asap::{ - AggIntent, Column, DataType, QueryExpr, Reduction, Schema, Source, + AggIntent, Column, DataType, PreASAPNode, Reduction, Schema, Source, }; let source = Source::TimeSeries { metric: "missing".into(), }; - let scan = Rc::new(QueryExpr::Scan { + let scan = Rc::new(PreASAPNode::Scan { source: source.clone(), predicates: vec![], schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), }); - let root = Rc::new(QueryExpr::Aggregate { + let root = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Absent], output_names: vec![], @@ -2382,20 +2382,20 @@ mod tests { #[test] fn promql_range_and_subquery_preserve_internal_steps() { - use asap_types::pre_asap::{Column, DataType, QueryExpr, Schema, Source}; + use asap_types::pre_asap::{Column, DataType, PreASAPNode, Schema, Source}; use std::time::Duration; let source = Source::TimeSeries { metric: "m".into() }; - let scan = Rc::new(QueryExpr::Scan { + let scan = Rc::new(PreASAPNode::Scan { source: source.clone(), predicates: vec![], schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), }); - let range = Rc::new(QueryExpr::TimeRange { + let range = Rc::new(PreASAPNode::TimeRange { range: Duration::from_secs(300), child: scan, }); - let root = Rc::new(QueryExpr::PromqlSubquery { + let root = Rc::new(PreASAPNode::PromqlSubquery { range: Duration::from_secs(300), resolution: Some(Duration::from_secs(60)), child: range, @@ -2457,20 +2457,20 @@ mod tests { #[test] fn promql_binary_lowering_keeps_operation_and_matching_cardinality() { use asap_types::pre_asap::{ - ArithmeticOpKind, BinaryOpKind, Column, DataType, GroupSide, QueryExpr, Schema, Source, - VectorGrouping, VectorMatch, VectorMatchKind, + ArithmeticOpKind, BinaryOpKind, Column, DataType, GroupSide, PreASAPNode, Schema, + Source, VectorGrouping, VectorMatch, VectorMatchKind, }; let left_source = Source::TimeSeries { metric: "a".into() }; let right_source = Source::TimeSeries { metric: "b".into() }; let scan = |source| { - Rc::new(QueryExpr::Scan { + Rc::new(PreASAPNode::Scan { source, predicates: vec![], schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), }) }; - let root = Rc::new(QueryExpr::BinaryOp { + let root = Rc::new(PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), lhs: scan(left_source.clone()), rhs: scan(right_source.clone()), @@ -2540,29 +2540,29 @@ mod tests { #[test] fn promql_relabel_sample_and_per_series_lower_as_a_complete_chain() { use asap_types::pre_asap::{ - AggIntent, Column, DataType, GroupKeys, QueryExpr, Reduction, SampleKind, ScalarValue, - Schema, Source, + AggIntent, Column, DataType, GroupKeys, PreASAPNode, Reduction, SampleKind, + ScalarValue, Schema, Source, }; let source = Source::TimeSeries { metric: "requests".into(), }; - let scan = Rc::new(QueryExpr::Scan { + let scan = Rc::new(PreASAPNode::Scan { source: source.clone(), predicates: vec![], schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), }); - let relabel = Rc::new(QueryExpr::PromqlRelabel { + let relabel = Rc::new(PreASAPNode::PromqlRelabel { dst: "service".into(), - value: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("api".into()))), + value: Rc::new(PreASAPNode::Literal(ScalarValue::Utf8("api".into()))), child: scan, }); - let sample = Rc::new(QueryExpr::PromqlSeriesSample { + let sample = Rc::new(PreASAPNode::PromqlSeriesSample { by: GroupKeys::none(), kind: SampleKind::LimitK(5), child: relabel, }); - let root = Rc::new(QueryExpr::Aggregate { + let root = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Sum { col: None }], output_names: vec![], diff --git a/crates/asap-aware-mapping/src/recurrence.rs b/crates/asap-aware-mapping/src/recurrence.rs index 125906ef..794d738e 100644 --- a/crates/asap-aware-mapping/src/recurrence.rs +++ b/crates/asap-aware-mapping/src/recurrence.rs @@ -75,7 +75,7 @@ //! //! - [`EvaluationRate`]: derived from [`asap_types::workload::RepeatingEntry::demand`] //! values of every repeating consumer reaching a target (via -//! [`evaluation_rate_of`], or [`crate::replacement::CandidatePostASAPDAGs::recurrence_profiles`] +//! [`evaluation_rate_of`], or [`crate::replacement::CandidateLogicalPostASAPDAGs::recurrence_profiles`] //! for a whole workload). A one-shot ([`asap_types::workload::BatchEntry`]) //! consumer contributes to [`RecurrenceProfile::one_shot_consumers`] //! instead, never to this rate. @@ -216,18 +216,18 @@ pub enum RecurrenceError { CostRate with a one-shot Cost without distorting the comparison" )] InvalidHorizon(Horizon), - /// [`crate::replacement::CandidatePostASAPDAGs::recurrence_profiles`] was called + /// [`crate::replacement::CandidateLogicalPostASAPDAGs::recurrence_profiles`] was called /// with a `root_recurrence` slice whose length doesn't match the - /// `CandidatePostASAPDAGs`'s own root count — a caller error, but recoverable + /// `CandidateLogicalPostASAPDAGs`'s own root count — a caller error, but recoverable /// (this method's whole signature promises a `Result`, so this is /// reported the same way every other input-validation failure is, /// never a panic). #[error( "recurrence_profiles: root_recurrence must have one entry per root, in the same order \ - CandidatePostASAPDAGs::roots is in (got {got} entries for {expected} roots)" + CandidateLogicalPostASAPDAGs::roots is in (got {got} entries for {expected} roots)" )] RootCountMismatch { - /// `CandidatePostASAPDAGs::roots.len()`. + /// `CandidateLogicalPostASAPDAGs::roots.len()`. expected: usize, /// `root_recurrence.len()`. got: usize, @@ -239,7 +239,7 @@ pub enum RecurrenceError { /// applied at every point an `UpdateRate` enters a [`RecurrenceProfile`] /// ([`RecurrenceProfile::with_update_rate`], /// [`update_rate_from_data_workload`], -/// [`crate::replacement::CandidatePostASAPDAGs::recurrence_profiles`]'s own parameter) +/// [`crate::replacement::CandidateLogicalPostASAPDAGs::recurrence_profiles`]'s own parameter) /// *and*, as a backstop that can't be bypassed by constructing a /// `RecurrenceProfile` via its public fields directly, inside [`decide`] /// itself before any comparison uses it. @@ -373,7 +373,7 @@ impl RecurrenceProfile { } /// How one workload root recurs — the opaque per-root tag -/// [`crate::replacement::CandidatePostASAPDAGs::recurrence_profiles`] threads down to +/// [`crate::replacement::CandidateLogicalPostASAPDAGs::recurrence_profiles`] threads down to /// every target reachable from that root. Mirrors /// [`asap_types::workload::QueryWorkload`]'s own `query_batch` (one-shot) /// vs. `repeating_queries` (an interval each) split, but at the @@ -780,16 +780,16 @@ mod tests { use crate::cost_model::CseCandidate; use asap_types::post_asap::{ - ExactKind, ExactParams, GroupingStrategy, ResultGuarantee, SummaryExpr, SummaryFamilyType, - SummaryField, SummaryNode, SummarySchema, + ExactKind, ExactParams, GroupingStrategy, PostASAPNode, ResultGuarantee, SummaryExpr, + SummaryFamilyType, SummaryField, SummarySchema, }; use asap_types::pre_asap::expr_ir::ColumnRef; - use asap_types::pre_asap::query_expr::{QueryExpr, Reduction, Source}; + use asap_types::pre_asap::query_expr::{PreASAPNode, Reduction, Source}; use asap_types::pre_asap::schema::{Column, DataType, Schema}; use std::rc::Rc; - fn scan() -> QueryExpr { - QueryExpr::Scan { + fn scan() -> PreASAPNode { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -803,10 +803,10 @@ mod tests { } } - fn summary_node(family: SummaryFamilyType) -> SummaryNode { - SummaryNode { + fn summary_node(family: SummaryFamilyType) -> PostASAPNode { + PostASAPNode { expr: SummaryExpr::SummaryAgg { - child: Rc::new(SummaryNode { + child: Rc::new(PostASAPNode { expr: SummaryExpr::KeepPreAsap(Rc::new(scan())), schema: SummarySchema { fields: vec![], @@ -1126,7 +1126,7 @@ mod tests { } } - // ── multiple roots sharing a sub-DAG, via CandidatePostASAPDAGs ────────────────── + // ── multiple roots sharing a sub-DAG, via CandidateLogicalPostASAPDAGs ────────────────── use crate::replacement::search_workload; use asap_types::pre_asap::agg_intent::AggIntent; @@ -1141,8 +1141,8 @@ mod tests { /// real one, matching the pattern /// `replacement.rs`'s own CSE fixtures already use (`metric_scan`/`agg` /// grouped by a label column). - fn labeled_scan() -> QueryExpr { - QueryExpr::Scan { + fn labeled_scan() -> PreASAPNode { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -1157,8 +1157,8 @@ mod tests { } } - fn sum_agg() -> QueryExpr { - QueryExpr::Aggregate { + fn sum_agg() -> PreASAPNode { + PreASAPNode::Aggregate { reduction: QueryReduction::by(vec![2]), measures: vec![AggIntent::Sum { col: Some(1) }], output_names: vec![], @@ -1175,9 +1175,9 @@ mod tests { /// `shared_aggregate_across_two_roots_gets_both_strategies_candidates`'s /// own doc) while letting `share_common_subtrees` unify their /// identical `sum_agg()` children onto one shared `Rc`. - fn filtered_root(distinguishing_literal: i64) -> QueryExpr { - QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64( + fn filtered_root(distinguishing_literal: i64) -> PreASAPNode { + PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Int64( distinguishing_literal, )))), child: Rc::new(sum_agg()), @@ -1186,14 +1186,14 @@ mod tests { /// Three workload roots share one underlying `sum_agg()` sub-DAG: two /// repeating consumers with different intervals, one one-shot batch - /// consumer. `CandidatePostASAPDAGs::recurrence_profiles` must aggregate all three + /// consumer. `CandidateLogicalPostASAPDAGs::recurrence_profiles` must aggregate all three /// onto the shared sub-DAG's own profile: `evaluation_rate = 1/t1 + /// 1/t2`, `one_shot_consumers = 1` — issue #287's "support a shared /// sub-DAG consumed by queries with different intervals" and "multiple /// roots sharing a sub-DAG" acceptance criteria. #[test] fn recurrence_profiles_aggregates_mixed_intervals_across_roots_sharing_a_subdag() { - let roots: Vec<(&str, Rc)> = vec![ + let roots: Vec<(&str, Rc)> = vec![ ("root_a", Rc::new(filtered_root(1))), ("root_b", Rc::new(filtered_root(2))), ("root_c", Rc::new(filtered_root(3))), @@ -1214,7 +1214,7 @@ mod tests { ); let shared_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.as_ref(), PreASAPNode::Aggregate { .. })) .expect("the shared sum_agg() is a discovered target"); assert_eq!(shared_group.consumer_count, 3, "shared by all 3 roots"); @@ -1265,7 +1265,7 @@ mod tests { let space = search_workload(roots); let shared = space .target_subdag_candidates() - .find(|group| matches!(group.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|group| matches!(group.target.as_ref(), PreASAPNode::Aggregate { .. })) .expect("the aggregate is shared by both roots"); let update_rate = Some(UpdateRate(10.0)); @@ -1324,7 +1324,7 @@ mod tests { #[test] fn recurrence_profiles_rejects_an_invalid_evaluation_rate() { let root = Rc::new(scan()); - let roots: Vec<(&str, Rc)> = vec![("only", root)]; + let roots: Vec<(&str, Rc)> = vec![("only", root)]; let space = search_workload(roots); let err = space .recurrence_profiles(&[RootRecurrence::Repeating(EvaluationRate(f64::NAN))], None) @@ -1338,7 +1338,7 @@ mod tests { #[test] fn recurrence_profiles_reports_a_root_count_mismatch_as_an_error_not_a_panic() { let root = Rc::new(scan()); - let roots: Vec<(&str, Rc)> = vec![("only", root)]; + let roots: Vec<(&str, Rc)> = vec![("only", root)]; let space = search_workload(roots); let err = space.recurrence_profiles(&[], None).unwrap_err(); assert_eq!( @@ -1353,7 +1353,7 @@ mod tests { #[test] fn recurrence_profiles_rejects_an_invalid_update_rate() { let root = Rc::new(scan()); - let roots: Vec<(&str, Rc)> = vec![("only", root)]; + let roots: Vec<(&str, Rc)> = vec![("only", root)]; let space = search_workload(roots); let err = space .recurrence_profiles( @@ -1381,14 +1381,14 @@ mod tests { /// `consumer_count`. #[test] fn recurrence_profiles_does_not_stamp_update_rate_on_a_site_unreachable_from_any_root() { - let avg_root = QueryExpr::Aggregate { + let avg_root = PreASAPNode::Aggregate { reduction: QueryReduction::by(vec![]), measures: vec![AggIntent::Avg { col: None }], output_names: vec![], having: None, child: Rc::new(scan()), }; - let roots: Vec<(&str, Rc)> = vec![("q", Rc::new(avg_root))]; + let roots: Vec<(&str, Rc)> = vec![("q", Rc::new(avg_root))]; let space = search_workload(roots); let count_group = space @@ -1396,7 +1396,7 @@ mod tests { .find(|g| { matches!( g.target.as_ref(), - QueryExpr::Aggregate { measures, .. } + PreASAPNode::Aggregate { measures, .. } if measures.iter().any(|m| matches!(m, AggIntent::Count { .. })) ) }) @@ -1435,7 +1435,7 @@ mod tests { /// reachability-set walk would (wrongly) collapse it to. #[test] fn recurrence_profiles_credits_a_direct_repeated_reference_by_its_multiplicity() { - let root = QueryExpr::BinaryOp { + let root = PreASAPNode::BinaryOp { op: asap_types::pre_asap::query_expr::BinaryOpKind::Compare( asap_types::pre_asap::expr_ir::CompareOpKind::Eq, ), @@ -1447,7 +1447,7 @@ mod tests { let shared_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.as_ref(), PreASAPNode::Aggregate { .. })) .expect("sum_agg() should merge onto one shared Rc, referenced twice from BinaryOp"); assert_eq!( shared_group.consumer_count, 2, @@ -1470,7 +1470,7 @@ mod tests { let scan_group = space .target_subdag_candidates() - .find(|group| matches!(group.target.as_ref(), QueryExpr::Scan { .. })) + .find(|group| matches!(group.target.as_ref(), PreASAPNode::Scan { .. })) .expect("the shared aggregate has a scan descendant"); assert_eq!( profiles diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 57600188..43e14a0a 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -18,7 +18,7 @@ //! `CostModel::size_params`, not a placeholder filled in later). //! 2. **Build**: for each candidate in that list, [`construct_summary`] //! mechanically turns the already-decided `(kind, params)` into a real -//! [`SummaryNode`] — derives the child schema, resolves the summarized +//! [`PostASAPNode`] — derives the child schema, resolves the summarized //! column, builds the readout query, recurses into the child (via //! [`realize_child`], so a nested aggregate gets its own //! independent enumeration, never the outer target's forced choice), and @@ -31,13 +31,13 @@ //! has to run regardless of how `(kind, params)` were chosen, so it lives //! directly inside the one method that needs it. //! -//! - [`TargetSubDAG`] — a reference to a pre-ASAP [`QueryExpr`] node that is a +//! - [`TargetSubDAG`] — a reference to a pre-ASAP [`PreASAPNode`] node that is a //! candidate for replacement, plus how many places in the workload already //! reference it (its `consumer_count`) — the one piece of cross-node //! context [`SharedSubtreeStrategy`] needs that a bare node reference alone //! doesn't carry. //! - [`ReplacementSubDAG`] — one candidate replacement for a `TargetSubDAG`: -//! either a fully bound [`SummaryNode`] or a pre-ASAP [`QueryExpr`] rewrite +//! either a fully bound [`PostASAPNode`] or a pre-ASAP [`PreASAPNode`] rewrite //! (still logical, structurally different from the target but semantically //! equivalent) — see [`Replacement`] — plus a human-readable `rationale`. //! - [`ReplacementStrategy`] — `matches` + `replacements`, the same @@ -54,7 +54,7 @@ //! //! A caller may inspect local replacements, but taking the first candidate //! does not establish a compatible workload plan or physical deployability. -//! For Planner-owned logical selection, call [`CandidatePostASAPDAGs::global_selection`] +//! For Planner-owned logical selection, call [`CandidateLogicalPostASAPDAGs::global_selection`] //! once and [`GlobalSelection::assemble_selected_dag`] for each wanted query //! root. Alternatively, use the summary-maintenance-lifecycle-aware helpers //! when Planner should also compare maintenance against raw recomputation. @@ -80,7 +80,7 @@ //! `asap_types::pre_asap::cse::share_common_subtrees`'s sharing decision. //! Wherever a [`TargetSubDAG`] already has two or more consumers (i.e. //! `share_common_subtrees` already collapsed two or more workload -//! locations onto the same `Rc` — [`discover_targets`] below +//! locations onto the same `Rc` — [`discover_targets`] below //! does the identical workload-wide discovery for [`search_workload_with`]; //! this module's own tests reuse the same dedup logic to build realistic //! fixtures), it reports the two-way candidate CSE's own detection pass @@ -107,7 +107,7 @@ //! `TargetSubDAG` discovery pass" — describing work deliberately left for a //! future Cascades/Volcano-style search engine (PR #263, //! `feat/cascades-search-252`, over the [`ReplacementStrategy`] extension -//! point above). That engine is [`CandidatePostASAPDAGs`]/[`TargetSubDAGCandidates`]/ +//! point above). That engine is [`CandidateLogicalPostASAPDAGs`]/[`TargetSubDAGCandidates`]/ //! [`search_workload`]/[`search_workload_with`] below, merged into this //! module rather than kept as a separate `search` module — the same "one //! module, one step" reasoning the top of this file already uses for @@ -143,12 +143,12 @@ //! //! 1. **Per-target candidates, not flat plans.** [`TargetSubDAGCandidates`] //! stores the alternatives for one distinct [`TargetSubDAG`] (identified by -//! its own `Rc` pointer identity — the same currency +//! its own `Rc` pointer identity — the same currency //! [`asap_types::pre_asap::cse::share_common_subtrees`] already //! established across the workload) holding every -//! [`ReplacementSubDAG`] alternative discovered for it. [`CandidatePostASAPDAGs`] is +//! [`ReplacementSubDAG`] alternative discovered for it. [`CandidateLogicalPostASAPDAGs`] is //! a collection of these groups, keyed by `TargetSubDAG` — a candidate -//! "plan" is never materialized as a distinct top-level `Rc` +//! "plan" is never materialized as a distinct top-level `Rc` //! at all; two logically-different overall choices at two different //! targets are just two different entries in two different groups, //! sharing every other node in the workload by construction (they *are* @@ -157,7 +157,7 @@ //! discipline.** [`asap_types::pre_asap::cse::structural_hash`] (made //! `pub` for exactly this reuse) is only ever a candidate-narrowing //! filter; [`TargetSubDAGCandidates::add_candidate`]'s actual duplicate check is -//! `QueryExpr`'s derived `PartialEq` — the same "hash is a filter, +//! `PreASAPNode`'s derived `PartialEq` — the same "hash is a filter, //! `PartialEq` is the decision, no exceptions" rule `cse.rs`'s own //! "Correctness" section states and this module inherits rather than //! reinvents. See [`is_duplicate_rewrite`] for the one deliberate @@ -213,7 +213,7 @@ //! own; see [`discover_new_descendant_targets`]) are scanned for pointers //! not already known, and any found become next round's frontier. Both shipped //! strategies are idempotent in exactly this sense: [`SketchAlgorithmStrategy`] -//! produces terminal [`Replacement::Summary`] candidates (no `QueryExpr` +//! produces terminal [`Replacement::Summary`] candidates (no `PreASAPNode` //! children to scan at all), and [`SharedSubtreeStrategy`]'s two //! [`Replacement::Rewrite`] candidates both reuse the target's own //! already-known child `Rc`s verbatim (`Rc::clone`/a shallow top-level @@ -235,7 +235,7 @@ //! //! ### Cost-based final selection — reusing `CostModel`, not a second interface //! -//! [`CandidatePostASAPDAGs::cost_sorted`] is the `sorted_by(cost_model)` step, and it +//! [`CandidateLogicalPostASAPDAGs::cost_sorted`] is the `sorted_by(cost_model)` step, and it //! reuses this crate's existing [`CostModel`] trait rather than inventing a //! second cost interface (`docs/design_docs/cse-cost-model-decision.md`, //! issue #237, explicitly reasoned about *why* a narrow, direct cost @@ -259,7 +259,7 @@ //! //! ## Whole-plan (cross-group) selection — issue #271 //! -//! [`CandidatePostASAPDAGs::cost_sorted`] above ranks every group's candidates +//! [`CandidateLogicalPostASAPDAGs::cost_sorted`] above ranks every group's candidates //! independently: it never lets one group's choice influence how another //! group is costed. That's the right behavior when groups genuinely don't //! interact — which both shipped strategies' one-round convergence (see @@ -277,7 +277,7 @@ //! per-group ranking has no way to see this — it only ever looks at one //! group's own `candidates`, in isolation. //! -//! [`CandidatePostASAPDAGs::global_selection`] is that missing step: a single +//! [`CandidateLogicalPostASAPDAGs::global_selection`] is that missing step: a single //! **top-down dynamic-programming pass** over the discovered sites, //! processed in the topological order [`topological_order`] computes over a //! small [`ReferenceGraph`] built for exactly this purpose (parent before @@ -305,9 +305,9 @@ //! subproblems (a site reachable through more than one parent path is //! solved once, memoized in `effective_uses`, and reused for every path //! into it) combined via a real recurrence — not just the MEMO-group -//! sharing [`CandidatePostASAPDAGs`] itself already does for *storing* candidates. That +//! sharing [`CandidateLogicalPostASAPDAGs`] itself already does for *storing* candidates. That //! distinction is exactly what issue #271 raised: this module already looks -//! like a Cascades/Volcano MEMO, but [`CandidatePostASAPDAGs::cost_sorted`] alone never +//! like a Cascades/Volcano MEMO, but [`CandidateLogicalPostASAPDAGs::cost_sorted`] alone never //! actually performed this composition step; `global_selection` is that //! step, added alongside `cost_sorted` rather than replacing it (both stay //! available — see [`RankedTargetSubDAGCandidates`] vs. [`TargetSubDAGSelection`]'s own docs for when @@ -351,9 +351,9 @@ use std::collections::{HashMap, HashSet, VecDeque}; use asap_types::post_asap::{ validate_execution_data_states_at, EntityIdentity, ExactKind, ExactOperation, ExactOperationSchemaError, ExactParams, ExecutionDataState, ExecutionDataStateError, - ExecutionTiming, GroupingStrategy, NonNegativeWeightProof, SamplingKind, SamplingParams, - SketchAlgorithm, SketchKind, SketchParams, SketchQuery as PostAsapSketchQuery, StatModelKind, - StatModelParams, SummaryExpr, SummaryFamilyType, SummaryField, SummaryInputExpr, SummaryNode, + ExecutionTiming, GroupingStrategy, NonNegativeWeightProof, PostASAPNode, SamplingKind, + SamplingParams, SketchAlgorithm, SketchKind, SketchParams, SketchQuery as PostASAPSketchQuery, + StatModelKind, StatModelParams, SummaryExpr, SummaryFamilyType, SummaryField, SummaryInputExpr, SummarySchema, SummaryUpdate, ValueOperation, WaveletKind, WaveletParams, WeightDomain, }; use asap_types::post_asap::{AccuracyError, CompositionOperator, GuaranteeSource, ResultGuarantee}; @@ -361,7 +361,7 @@ use asap_types::pre_asap::agg_intent::{agg_is_mergeable, AggIntent}; use asap_types::pre_asap::cse::{share_common_subtrees, structural_hash, HashCache}; use asap_types::pre_asap::expr_ir::{ArithmeticOpKind, ColumnRef}; use asap_types::pre_asap::query_expr::{ - BinaryOpKind, Predicate, QueryExpr, QueryExprError, Reduction, + BinaryOpKind, PreASAPNode, PreASAPNodeError, Predicate, Reduction, }; use asap_types::pre_asap::schema::{ColumnId, Schema}; use asap_types::types::AccuracyTarget; @@ -391,13 +391,13 @@ use crate::topk_reuse::TopKLimitReuseStrategy; /// ([`realize_child`] and [`keep_pre_asap`]). Moved here from the former /// `bind.rs` (issue #251): this is what a [`ReplacementStrategy`] /// implementor's own construction path can realistically fail with — -/// schema derivation over a pre-ASAP [`QueryExpr`] — not something specific +/// schema derivation over a pre-ASAP [`PreASAPNode`] — not something specific /// to workload-wide orchestration. #[derive(Debug, Error)] pub enum RealizationError { /// Schema derivation failed while lifting an edge to `SummarySchema`. #[error("schema derivation failed during pre-ASAP → post-ASAP binding: {0}")] - Schema(#[from] QueryExprError), + Schema(#[from] PreASAPNodeError), /// The candidate is accuracy-illegal (issue #172): its composed /// guarantee has no sound propagation rule, or misses the applicable /// `AccuracyTarget`. Fail-closed — the candidate is never constructed @@ -411,6 +411,10 @@ pub enum RealizationError { /// would change its semantics. #[error("unsupported physical summary realization: {0}")] PhysicalRealization(&'static str), + /// Candidate enumeration would exceed the caller's expansion limit; no + /// partial inventory is returned. + #[error("candidate expansion exceeds limit {0}")] + ExpansionLimit(usize), /// A constructed plan violates the update/readout phase contract /// (issue #171) — e.g. a summary readout placed beneath a maintained /// `SummaryAgg`. Detected at construction, never at runtime. @@ -424,10 +428,10 @@ pub enum RealizationError { /// A pre-ASAP sub-DAG a [`ReplacementStrategy`] knows how to replace. /// -/// `root` is a reference into the workload's own [`QueryExpr`] tree (an -/// `Rc`, the same currency [`search_workload`] and +/// `root` is a reference into the workload's own [`PreASAPNode`] tree (an +/// `Rc`, the same currency [`search_workload`] and /// `asap_types::pre_asap::cse::share_common_subtrees` already thread through -/// this crate's public API — not a bare `&QueryExpr` — so a strategy that +/// this crate's public API — not a bare `&PreASAPNode` — so a strategy that /// needs the node's own `Rc` identity, not just its shape, has it available /// without the caller re-deriving it). /// @@ -440,14 +444,14 @@ pub enum RealizationError { /// count; [`SharedSubtreeStrategy`] consults it directly. #[derive(Debug, Clone, Copy)] pub struct TargetSubDAG<'a> { - pub root: &'a Rc, + pub root: &'a Rc, pub consumer_count: usize, } impl<'a> TargetSubDAG<'a> { /// A target assumed to have exactly one consumer — the common case for a /// caller that isn't already tracking cross-workload sharing. - pub fn new(root: &'a Rc) -> Self { + pub fn new(root: &'a Rc) -> Self { Self { root, consumer_count: 1, @@ -456,7 +460,7 @@ impl<'a> TargetSubDAG<'a> { /// A target with an explicit `consumer_count`, used by workload discovery /// and by callers that already know how many locations reference `root`. - pub fn with_consumer_count(root: &'a Rc, consumer_count: usize) -> Self { + pub fn with_consumer_count(root: &'a Rc, consumer_count: usize) -> Self { Self { root, consumer_count, @@ -474,19 +478,19 @@ impl<'a> TargetSubDAG<'a> { pub enum Replacement { /// A fully bound post-ASAP summary decision, for one particular /// candidate realization of the target. - Summary(Rc), - /// A pre-ASAP rewrite: still a logical [`QueryExpr`], structurally + Summary(Rc), + /// A pre-ASAP rewrite: still a logical [`PreASAPNode`], structurally /// different from the target's own `root` (e.g. sharing vs. not sharing /// a subtree) but semantically equivalent to it. - Rewrite(Rc), + Rewrite(Rc), /// An exact operator composed over another target's *own* selected /// decision across an explicit update/readout boundary (issue #171): /// `ValueOperationAtQueryTime` over a child's summary readout, or /// `ValueOperationAtIngestionTime` feeding a maintained summary above. Carries only a - /// reference to the child target — [`CandidatePostASAPDAGs::global_selection`] + /// reference to the child target — [`CandidateLogicalPostASAPDAGs::global_selection`] /// commits the compatible parent/child pair and /// [`GlobalSelection::assemble_selected_dag`] links it into one validated - /// `SummaryNode`. See [`crate::exact_composition`]. + /// `PostASAPNode`. See [`crate::exact_composition`]. ExactComposition(ExactComposition), } @@ -646,7 +650,7 @@ pub trait ReplacementStrategy { /// (for example, the PromQL series identity), so /// [`search_workload_with_targets`] asks only workload roots, once each. /// They decide what to compute, never placement. Default: none. - fn propose_for_root(&self, _root: &Rc, _target: &AccuracyTarget) -> Proposals { + fn propose_for_root(&self, _root: &Rc, _target: &AccuracyTarget) -> Proposals { Proposals::default() } } @@ -665,7 +669,7 @@ pub trait ReplacementStrategy { /// no separate function that computes just "the one" `Realization` /// independently of that list. [`SketchAlgorithmStrategy`] is the sole /// consumer: it wraps every entry of this list into its own bound -/// [`SummaryNode`] and returns all of them, ranked — a caller wanting a +/// [`PostASAPNode`] and returns all of them, ranked — a caller wanting a /// single answer keeps the first one itself (see the module docs above). #[derive(Debug, Clone, PartialEq)] pub enum Realization { @@ -1303,10 +1307,10 @@ impl<'a> SketchAlgorithmStrategy<'a> { /// treats a range of historical samples as the instant vector. pub fn current_series_topk_candidates( &self, - root: &Rc, + root: &Rc, accuracy: &AccuracyTarget, ) -> Proposals { - let QueryExpr::Limit { + let PreASAPNode::Limit { n, offset: 0, child, @@ -1314,7 +1318,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { else { return Proposals::default(); }; - let QueryExpr::Sort { + let PreASAPNode::Sort { keys, partition_by, child, @@ -1325,7 +1329,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { let [key] = keys.as_slice() else { return Proposals::default(); }; - let QueryExpr::Column(value) = key.expr else { + let PreASAPNode::Column(value) = key.expr else { return Proposals::default(); }; let Ok(schema) = child.output_schema() else { @@ -1343,7 +1347,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { { return Proposals::default(); } - let ranked = Rc::new(QueryExpr::Aggregate { + let ranked = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::Reduce(partition_by.clone()), measures: vec![AggIntent::TopK { k: *n, @@ -1363,7 +1367,11 @@ impl<'a> SketchAlgorithmStrategy<'a> { /// The whole enumeration for one target, with `intent_override` /// substituting the target's own intent (only ever its `AccuracyTarget` /// differs — see [`realize_child_with`]). - fn propose_with(&self, root: &Rc, intent_override: Option<&AggIntent>) -> Proposals { + fn propose_with( + &self, + root: &Rc, + intent_override: Option<&AggIntent>, + ) -> Proposals { let mut proposals = Proposals::default(); // A selected logical rewrite otherwise remains KeepPreAsap during DAG // assembly. Also expose its concrete summary realization for selection. @@ -1470,7 +1478,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { let Some(child) = aggregate_child(root) else { continue; }; - let QueryExpr::Aggregate { reduction, .. } = root.as_ref() else { + let PreASAPNode::Aggregate { reduction, .. } = root.as_ref() else { continue; }; let Ok(input) = realize_physical_summary_input(intent, &family, reduction, child) @@ -1567,7 +1575,7 @@ impl Proposals { /// File one construction attempt: a legal node becomes a candidate, an /// [`RealizationError::Accuracy`] becomes a [`RejectedCandidate`], and a /// schema-derivation failure is skipped exactly as it always was. - fn record(&mut self, rationale: String, built: Result, RealizationError>) { + fn record(&mut self, rationale: String, built: Result, RealizationError>) { match built { Ok(node) => self.candidates.push(ReplacementSubDAG { strategy: "SketchAlgorithmStrategy", @@ -1586,16 +1594,17 @@ impl Proposals { Err( RealizationError::Schema(_) | RealizationError::ExactOperationSchema(_) - | RealizationError::PhysicalRealization(_), + | RealizationError::PhysicalRealization(_) + | RealizationError::ExpansionLimit(_), ) => {} } } } /// The `child` of a [`bindable_intent`]-shaped `Aggregate`. -fn aggregate_child(node: &QueryExpr) -> Option<&Rc> { +fn aggregate_child(node: &PreASAPNode) -> Option<&Rc> { match node { - QueryExpr::Aggregate { child, .. } => Some(child), + PreASAPNode::Aggregate { child, .. } => Some(child), _ => None, } } @@ -1619,7 +1628,7 @@ impl ReplacementStrategy for SketchAlgorithmStrategy<'_> { /// for the identity-carrying root. Placement variants (for example, /// fixed-window or query-time Rate aggregation) are not listed here: the /// lifecycle assigns timing and the physical compiler reads it. - fn propose_for_root(&self, root: &Rc, target: &AccuracyTarget) -> Proposals { + fn propose_for_root(&self, root: &Rc, target: &AccuracyTarget) -> Proposals { let Ok(typed) = asap_types::pre_asap::schema::with_promql_series_identity(root) else { return Proposals::default(); }; @@ -1711,11 +1720,11 @@ pub(crate) fn describe_intent(intent: &AggIntent) -> String { // ── realize_child / keep_pre_asap: rank-and-take-first, and its fallback ── -/// Rank-and-take-first selector for a single [`QueryExpr`] node: enumerate +/// Rank-and-take-first selector for a single [`PreASAPNode`] node: enumerate /// every candidate via [`SketchAlgorithmStrategy::replacements`], keep the /// `cost_model`-preferred (first) one, and fall back to [`keep_pre_asap`] /// when there's no candidate at all — **not** a general single-answer API -/// for a whole workload. Use [`CandidatePostASAPDAGs::global_selection`] and DAG assembly +/// for a whole workload. Use [`CandidateLogicalPostASAPDAGs::global_selection`] and DAG assembly /// for coordinated logical selection; physical deployment remains downstream. /// `root` must already be the caller's own /// `Rc`, never fabricated per call, so this never allocates beyond what the @@ -1725,16 +1734,16 @@ pub(crate) fn describe_intent(intent: &AggIntent) -> String { /// ([`construct_summary_agg`], so a nested aggregate gets its own /// independent enumeration instead of inheriting the parent's forced /// candidate), from this module's own [`realize_one`] (the representative -/// bound `SummaryNode` [`cse_preference`] needs for a +/// bound `PostASAPNode` [`cse_preference`] needs for a /// [`CostModel::cse_share_decision`] comparison), and from /// [`crate::cost_model::DefaultCostModel::estimate_cost`] (the same /// representative-node need, for a [`Replacement::Rewrite`] candidate's own /// cost estimate). Every other caller goes through /// [`SketchAlgorithmStrategy::replacements`] directly and decides for itself. pub(crate) fn realize_child( - root: &Rc, + root: &Rc, cost_model: &dyn CostModel, -) -> Result, RealizationError> { +) -> Result, RealizationError> { realize_child_with( root, CandidatePlanningInputs::with_default_accuracy(cost_model), @@ -1751,10 +1760,10 @@ pub(crate) fn realize_child( /// budget. A child whose declared target is `Exact` keeps it: an allocation /// never approximates something the caller declared exact. fn exact_topk_over_temporal_values( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, -) -> Result>, RealizationError> { - let QueryExpr::Aggregate { +) -> Result>, RealizationError> { + let PreASAPNode::Aggregate { reduction, measures, output_names: _, @@ -1767,7 +1776,7 @@ fn exact_topk_over_temporal_values( let [AggIntent::TopK { k, .. }] = measures.as_slice() else { return Ok(None); }; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction: Reduction::PerEntity, child: input, .. @@ -1775,7 +1784,7 @@ fn exact_topk_over_temporal_values( else { return Ok(None); }; - if !matches!(input.as_ref(), QueryExpr::TimeRange { .. }) { + if !matches!(input.as_ref(), PreASAPNode::TimeRange { .. }) { return Ok(None); } let values = realize_child_with(child, planning_inputs, Some(&AccuracyTarget::Exact))?; @@ -1795,14 +1804,14 @@ fn exact_topk_over_temporal_values( ))? .clone(); let score = ranking_score_index(child, &values.schema)?; - let sorted = Rc::new(SummaryNode { + let sorted = Rc::new(PostASAPNode { guarantee: values.guarantee.clone(), schema: values.schema.clone(), expr: SummaryExpr::ValueOperation { child: values, operation: ValueOperation::Sort { keys: vec![asap_types::pre_asap::SortKey { - expr: QueryExpr::Column(score), + expr: PreASAPNode::Column(score), ascending: false, nulls_first: false, }], @@ -1811,7 +1820,7 @@ fn exact_topk_over_temporal_values( timing: ExecutionTiming::QueryTime, }, }); - let node = Rc::new(SummaryNode { + let node = Rc::new(PostASAPNode { guarantee: sorted.guarantee.clone(), schema: sorted.schema.clone(), expr: SummaryExpr::ValueOperation { @@ -1829,10 +1838,10 @@ fn exact_topk_over_temporal_values( } fn realize_temporal_average( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, target: Option<&AccuracyTarget>, -) -> Result>, RealizationError> { +) -> Result>, RealizationError> { let Some(components) = crate::rewrite::temporal_average_components(root) else { return Ok(None); }; @@ -1846,10 +1855,10 @@ fn realize_temporal_average( } pub(crate) fn realize_child_with( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, end_to_end_target: Option<&AccuracyTarget>, -) -> Result, RealizationError> { +) -> Result, RealizationError> { if let Some(node) = realize_temporal_average(root, planning_inputs, end_to_end_target)? { return Ok(node); } @@ -1894,11 +1903,11 @@ pub(crate) fn realize_child_with( /// accelerated, return `None` so the caller keeps the whole query exact; /// mixed raw/summary snapshots are never constructed. fn realize_binary( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, end_to_end_target: Option<&AccuracyTarget>, -) -> Result>, RealizationError> { - let QueryExpr::BinaryOp { +) -> Result>, RealizationError> { + let PreASAPNode::BinaryOp { op, lhs, rhs, @@ -2083,7 +2092,7 @@ fn realize_binary( return Ok(None); } - Ok(Some(Rc::new(SummaryNode { + Ok(Some(Rc::new(PostASAPNode { expr: SummaryExpr::BinaryOp { timing: ExecutionTiming::QueryTime, lhs: lhs_node, @@ -2107,17 +2116,17 @@ fn realize_binary( /// query-time value consumer. Approximate summaries must already carry a /// `SummaryEstimate`, so they deliberately do not pass this predicate. pub fn finalize_query_candidate( - node: Rc, - logical_output: &QueryExpr, -) -> Result, RealizationError> { + node: Rc, + logical_output: &PreASAPNode, +) -> Result, RealizationError> { finalize_exact_accumulator_at(node, logical_output, ExecutionTiming::QueryTime) } fn finalize_exact_accumulator_at( - node: Rc, - logical_output: &QueryExpr, + node: Rc, + logical_output: &PreASAPNode, timing: ExecutionTiming, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let is_exact_state = matches!( node.expr, SummaryExpr::SummaryAgg { @@ -2134,7 +2143,7 @@ fn finalize_exact_accumulator_at( // query-time operators that follow this node. let schema = lift(&logical_output.output_schema()?); let guarantee = node.guarantee.clone(); - Ok(Rc::new(SummaryNode { + Ok(Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: node, operation: ValueOperation::FinalizeExactAccumulator, @@ -2145,10 +2154,10 @@ fn finalize_exact_accumulator_at( })) } -fn is_supported_exact_binary(root: &QueryExpr) -> bool { +fn is_supported_exact_binary(root: &PreASAPNode) -> bool { matches!( root, - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(_), vector_match: None, .. @@ -2156,17 +2165,17 @@ fn is_supported_exact_binary(root: &QueryExpr) -> bool { ) } -fn is_promql_scalar(expr: &QueryExpr) -> bool { +fn is_promql_scalar(expr: &PreASAPNode) -> bool { matches!( expr, - QueryExpr::PromqlScalarBridge(_) | QueryExpr::Literal(_) + PreASAPNode::PromqlScalarBridge(_) | PreASAPNode::Literal(_) ) } /// Quantile operands inherit one workload target. A temporal mean is exact /// on its checked finite domain and needs no approximation budget. -fn shared_quantile_target(lhs: &QueryExpr, rhs: &QueryExpr) -> Option { - let quantile_target = |expr: &QueryExpr| match bindable_intent(expr) { +fn shared_quantile_target(lhs: &PreASAPNode, rhs: &PreASAPNode) -> Option { + let quantile_target = |expr: &PreASAPNode| match bindable_intent(expr) { Some(AggIntent::Quantile { accuracy, q, .. }) if q.is_finite() && (0.0..=1.0).contains(q) => { @@ -2204,10 +2213,10 @@ fn ddsketch_ratio_operand_target(target: &AccuracyTarget) -> Option Option { +fn ddsketch_quantile_alpha(node: &PostASAPNode) -> Option { let SummaryExpr::SummaryEstimate { summary_input, - query: PostAsapSketchQuery::Quantile { .. }, + query: PostASAPSketchQuery::Quantile { .. }, } = &node.expr else { return None; @@ -2225,7 +2234,7 @@ fn ddsketch_quantile_alpha(node: &SummaryNode) -> Option { } } -fn has_missing_accuracy_evidence(node: &SummaryNode) -> bool { +fn has_missing_accuracy_evidence(node: &PostASAPNode) -> bool { node.guarantee .as_ref() .is_none_or(ResultGuarantee::has_unknown) @@ -2234,10 +2243,10 @@ fn has_missing_accuracy_evidence(node: &SummaryNode) -> bool { /// A direct ratio has an operator-specific DDSketch proof, so it must select /// DDSketch rather than the cost model's generally preferred KLL candidate. fn realize_ddsketch_quantile_operand( - operand: &Rc, + operand: &Rc, planning_inputs: CandidatePlanningInputs<'_>, target: &AccuracyTarget, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let intent = bindable_intent(operand).and_then(|intent| match intent { AggIntent::Quantile { .. } => Some(override_accuracy(intent, target)), _ => None, @@ -2256,12 +2265,12 @@ fn realize_ddsketch_quantile_operand( } fn realize_binary_operand( - operand: &Rc, + operand: &Rc, planning_inputs: CandidatePlanningInputs<'_>, end_to_end_target: Option<&AccuracyTarget>, -) -> Result, RealizationError> { +) -> Result, RealizationError> { if is_promql_scalar(operand) { - return Ok(Rc::new(SummaryNode { + return Ok(Rc::new(PostASAPNode { expr: SummaryExpr::KeepPreAsap(Rc::clone(operand)), schema: SummarySchema { fields: Vec::new(), @@ -2293,13 +2302,13 @@ fn override_accuracy(intent: &AggIntent, target: &AccuracyTarget) -> AggIntent { /// candidate for a target, or a deployment wants to force a node its own /// runtime can't actually implement — through the same fallback this /// crate's own dispatch uses, without duplicating the schema-lift logic. -pub fn keep_pre_asap(expr: &Rc) -> Result, RealizationError> { +pub fn keep_pre_asap(expr: &Rc) -> Result, RealizationError> { keep_pre_asap_rc(Rc::clone(expr)) } -fn keep_pre_asap_rc(expr: Rc) -> Result, RealizationError> { +fn keep_pre_asap_rc(expr: Rc) -> Result, RealizationError> { let schema = expr.output_schema()?; - Ok(Rc::new(SummaryNode { + Ok(Rc::new(PostASAPNode { expr: SummaryExpr::KeepPreAsap(expr), schema: lift(&schema), // A kept pre-ASAP subtree is executed exactly by the runtime @@ -2308,7 +2317,7 @@ fn keep_pre_asap_rc(expr: Rc) -> Result, RealizationE })) } -// ── Construction: turn one already-decided Realization into a SummaryNode ─ +// ── Construction: turn one already-decided Realization into a PostASAPNode ─ /// The bindable shape [`SketchAlgorithmStrategy`] targets: a single intent, no /// `HAVING`. A multi-intent node (SQL `SELECT SUM(a), AVG(b)`), or one with a @@ -2317,8 +2326,8 @@ fn keep_pre_asap_rc(expr: Rc) -> Result, RealizationE /// [`SummaryExpr::KeepPreAsap`] subtree. Composable query-time value /// operators (`Project`, `Filter`, `Sort`, and `Limit`) are retained during final /// DAG assembly so their independently planned children remain visible. -pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { - if let QueryExpr::Aggregate { +pub fn bindable_intent(node: &PreASAPNode) -> Option<&AggIntent> { + if let PreASAPNode::Aggregate { measures, having, .. } = node { @@ -2352,13 +2361,13 @@ pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { /// fail-closed answer for a composition with no sound rule or one that /// misses `intent`'s target. pub(crate) fn construct_summary_with( - expr: &QueryExpr, + expr: &PreASAPNode, intent: &AggIntent, realization: Realization, planning_inputs: CandidatePlanningInputs<'_>, child_target: Option<&AccuracyTarget>, allocation: Option, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let local_target = match allocation.as_ref() { Some(GuaranteeSource::BudgetAllocation { local_target, .. }) => Some(local_target), _ => accuracy_target(intent), @@ -2375,7 +2384,7 @@ pub(crate) fn construct_summary_with( }, other => other, }; - if let QueryExpr::Aggregate { + if let PreASAPNode::Aggregate { reduction, child, .. } = expr { @@ -2406,14 +2415,14 @@ pub(crate) fn construct_summary_with( } fn finish_weighted_topk( - candidate: Rc, - logical: &QueryExpr, + candidate: Rc, + logical: &PreASAPNode, intent: &AggIntent, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let AggIntent::TopK { k, .. } = intent else { unreachable!() }; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction: Reduction::Reduce(groups), child, .. @@ -2452,12 +2461,12 @@ fn finish_weighted_topk( }; Ok(asap_types::pre_asap::query_expr::ProjectItem { alias: Some(field.name.clone()), - expr: QueryExpr::Column(source), + expr: PreASAPNode::Column(source), }) }) .collect::, _>>()?; let guarantee = candidate.guarantee.clone(); - let projected = Rc::new(SummaryNode { + let projected = Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: candidate, operation: ValueOperation::Project { @@ -2469,12 +2478,12 @@ fn finish_weighted_topk( schema: schema.clone(), guarantee: guarantee.clone(), }); - let sorted = Rc::new(SummaryNode { + let sorted = Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: projected, operation: ValueOperation::Sort { keys: vec![asap_types::pre_asap::SortKey { - expr: QueryExpr::Column(score), + expr: PreASAPNode::Column(score), ascending: false, nulls_first: false, }], @@ -2485,7 +2494,7 @@ fn finish_weighted_topk( schema: schema.clone(), guarantee: guarantee.clone(), }); - let result = Rc::new(SummaryNode { + let result = Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: sorted, operation: ValueOperation::Limit { @@ -2502,24 +2511,24 @@ fn finish_weighted_topk( Ok(result) } -fn is_current_series_source(child: &QueryExpr) -> bool { +fn is_current_series_source(child: &PreASAPNode) -> bool { let source = match child { - QueryExpr::TimeRange { child, .. } => child.as_ref(), + PreASAPNode::TimeRange { child, .. } => child.as_ref(), source => source, }; - matches!(source, QueryExpr::Scan { + matches!(source, PreASAPNode::Scan { source: asap_types::pre_asap::Source::TimeSeries { .. }, schema, .. } if schema.has_promql_series_identity()) } -fn is_snapshot_weighted_topk(intent: &AggIntent, child: &QueryExpr) -> bool { +fn is_snapshot_weighted_topk(intent: &AggIntent, child: &PreASAPNode) -> bool { matches!(intent, AggIntent::TopK { .. }) && (is_current_series_source(child) || matches!(child, - QueryExpr::Aggregate { measures, child, .. } + PreASAPNode::Aggregate { measures, child, .. } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase]) || (matches!(measures.as_slice(), [AggIntent::Sum { .. }]) - && matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + && matches!(child.as_ref(), PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase]))))) } @@ -2553,7 +2562,7 @@ fn summary_family(realization: Realization) -> Option<(SummaryFamilyType, bool)> /// input value. Composite realizations can instead consume a larger /// logical sub-DAG and bind a different key or value. struct PhysicalSummaryInput { - child: Rc, + child: Rc, input: SummaryUpdate, } @@ -2567,7 +2576,7 @@ type PhysicalSummaryInputRule = fn( &AggIntent, &SummaryFamilyType, &Reduction, - &Rc, + &Rc, ) -> PhysicalSummaryInputRuleResult; /// Ordered physical-realization rules for realizations that consume more @@ -2585,7 +2594,7 @@ fn realize_value_frequency_summary_input( intent: &AggIntent, family: &SummaryFamilyType, _reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { // Frequency counts hash sample values as items but add one per observation. // Using the sample as a weight would turn counts into sums and admit signed CMS updates. @@ -2625,7 +2634,7 @@ fn realize_physical_summary_input( intent: &AggIntent, family: &SummaryFamilyType, reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> Result { for rule in PHYSICAL_SUMMARY_INPUT_RULES { match rule(intent, family, reduction, child) { @@ -2659,7 +2668,7 @@ fn realize_physical_summary_input( // on the update path. This is the initial layout for values feeding a summary; // lifecycle timing is authoritative. Read-time consumers keep their original // shared nodes. -fn maintenance_exact_values(node: Rc) -> Option> { +fn maintenance_exact_values(node: Rc) -> Option> { let expr = match &node.expr { // These guards can fall back at read time, but cannot recover a parent // sketch after an invalid value has entered its maintained state. @@ -2707,7 +2716,7 @@ fn maintenance_exact_values(node: Rc) -> Option> { } _ => return Some(node), }; - Some(Rc::new(SummaryNode { + Some(Rc::new(PostASAPNode { expr, schema: node.schema.clone(), guarantee: node.guarantee.clone(), @@ -2716,7 +2725,7 @@ fn maintenance_exact_values(node: Rc) -> Option> { #[allow(clippy::too_many_arguments)] fn construct_summary_agg( - node: &QueryExpr, + node: &PreASAPNode, reduction: &Reduction, intent: &AggIntent, input: PhysicalSummaryInput, @@ -2725,7 +2734,7 @@ fn construct_summary_agg( planning_inputs: CandidatePlanningInputs<'_>, child_target: Option<&AccuracyTarget>, allocation: Option, -) -> Result, RealizationError> { +) -> Result, RealizationError> { // The single canonical pre-ASAP derivation (per-series vs cross-series, // name overrides) already computes the row shape; binding only retypes // the summary state column. @@ -2735,7 +2744,7 @@ fn construct_summary_agg( SummaryFamilyType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap) ); - let snapshot_weighted = matches!(node, QueryExpr::Aggregate { child, .. } + let snapshot_weighted = matches!(node, PreASAPNode::Aggregate { child, .. } if is_snapshot_weighted_topk(intent, child)); let mut family = family; let score_population = if snapshot_weighted { @@ -2763,7 +2772,7 @@ fn construct_summary_agg( None }; let physical_reduction = if snapshot_weighted { - let QueryExpr::Aggregate { child, .. } = node else { + let PreASAPNode::Aggregate { child, .. } = node else { unreachable!() }; let source = input.child.output_schema()?; @@ -2806,13 +2815,13 @@ fn construct_summary_agg( }; let out_schema = node.output_schema()?; let measures = match node { - QueryExpr::Aggregate { measures, .. } => measures.len(), + PreASAPNode::Aggregate { measures, .. } => measures.len(), _ => 1, }; let state_idx = summary_col_index(&out_schema, reduction, measures); let readout_schema = if keyed_heap - && matches!(node, QueryExpr::Aggregate { child, .. } if is_snapshot_weighted_topk(intent, child)) + && matches!(node, PreASAPNode::Aggregate { child, .. } if is_snapshot_weighted_topk(intent, child)) { keyed_heap_readout_schema(&input, node)? } else { @@ -2828,7 +2837,7 @@ fn construct_summary_agg( | SketchParams::CountSketchWithHeap { heap_size, .. } => *heap_size, _ => unreachable!(), }; - return PostAsapSketchQuery::TopK { + return PostASAPSketchQuery::TopK { k: capacity as usize, }; } @@ -3002,7 +3011,7 @@ fn construct_summary_agg( // a genuine empty-`by` reduction apart from a per-entity shape with no // grouping concept at all (issue #163). `construct_summary_agg` is the // single place that decides this; nothing downstream re-derives it. - let agg = Rc::new(SummaryNode { + let agg = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child: bound_child, family, @@ -3019,7 +3028,7 @@ fn construct_summary_agg( // The readout: downstream of the estimate the schema is the plain // pre-ASAP row shape again (the summary-state type does not // propagate). - Some(query) => Ok(Rc::new(SummaryNode { + Some(query) => Ok(Rc::new(PostASAPNode { expr: SummaryExpr::SummaryEstimate { summary_input: agg, query, @@ -3035,11 +3044,11 @@ fn construct_summary_agg( // and an estimated score. They never inherit the exact-value producer's schema. fn keyed_heap_readout_schema( input: &PhysicalSummaryInput, - node: &QueryExpr, + node: &PreASAPNode, ) -> Result { let source = input.child.output_schema()?; let mut refs = Vec::new(); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, child, .. } = node else { @@ -3159,7 +3168,7 @@ fn keyed_heap_readout_schema( } fn ranking_score_index( - logical: &QueryExpr, + logical: &PreASAPNode, values: &SummarySchema, ) -> Result { if is_current_series_source(logical) { @@ -3175,7 +3184,7 @@ fn ranking_score_index( "snapshot ranking requires the sample value column", )); } - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, .. @@ -3228,11 +3237,11 @@ fn realize_counter_value_summary_input( intent: &AggIntent, family: &SummaryFamilyType, output_reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) || !matches!(family, SummaryFamilyType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap)) - || !matches!(child.as_ref(), QueryExpr::Aggregate { reduction: Reduction::PerEntity, measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])) + || !matches!(child.as_ref(), PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])) { return PhysicalSummaryInputRuleResult::NotApplicable; } @@ -3286,7 +3295,7 @@ fn realize_current_series_summary_input( intent: &AggIntent, family: &SummaryFamilyType, output_reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) || !is_current_series_source(child) { return PhysicalSummaryInputRuleResult::NotApplicable; @@ -3347,7 +3356,7 @@ fn realize_keyed_additive_summary_input( intent: &AggIntent, family: &SummaryFamilyType, output_reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) { return PhysicalSummaryInputRuleResult::NotApplicable; @@ -3362,7 +3371,7 @@ fn realize_keyed_additive_summary_input( ) { return PhysicalSummaryInputRuleResult::NotApplicable; } - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, having: None, @@ -3373,7 +3382,7 @@ fn realize_keyed_additive_summary_input( return PhysicalSummaryInputRuleResult::NotApplicable; }; let counter_input = matches!(measures.as_slice(), [AggIntent::Sum { .. }]) - && matches!(raw_child.as_ref(), QueryExpr::Aggregate { measures, .. } + && matches!(raw_child.as_ref(), PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])); let weight = match measures.as_slice() { [AggIntent::Count { .. }] => SummaryInputExpr::Constant(1.0), @@ -3465,7 +3474,7 @@ fn realize_keyed_additive_summary_input( }) } -fn schema_column_ref(child: &QueryExpr, index: usize) -> Option { +fn schema_column_ref(child: &PreASAPNode, index: usize) -> Option { let schema = child.output_schema().ok()?; let column = schema.columns.get(index)?; Some(match &column.table { @@ -3488,8 +3497,8 @@ fn schema_column_ref(child: &QueryExpr, index: usize) -> Option { /// exact child) — unknown, never exact. fn compose_guarantee( family: &SummaryFamilyType, - query: Option<&PostAsapSketchQuery>, - child: &SummaryNode, + query: Option<&PostASAPSketchQuery>, + child: &PostASAPNode, intent: &AggIntent, accuracy: &dyn AccuracyModel, evidence: &dyn AccuracyEvidenceProvider, @@ -3520,7 +3529,7 @@ fn compose_guarantee( ) } (_, Some(query)) => ( - if matches!(query, PostAsapSketchQuery::TopK { .. }) { + if matches!(query, PostASAPSketchQuery::TopK { .. }) { CompositionOperator::TopKSelection } else { CompositionOperator::ApproximateAggregate @@ -3644,14 +3653,14 @@ fn readout( intent: &AggIntent, input: &SummaryUpdate, cost_model: &dyn CostModel, -) -> PostAsapSketchQuery { +) -> PostASAPSketchQuery { match intent { - AggIntent::Quantile { q, .. } => PostAsapSketchQuery::Quantile { q: *q }, - AggIntent::Cardinality { .. } => PostAsapSketchQuery::Cardinality, - AggIntent::FrequencyL2 { .. } => PostAsapSketchQuery::FrequencyL2, - AggIntent::FrequencyEntropy { .. } => PostAsapSketchQuery::FrequencyEntropy, - AggIntent::TopK { k, .. } => PostAsapSketchQuery::TopK { k: *k }, - AggIntent::Count { .. } => PostAsapSketchQuery::PointCount { + AggIntent::Quantile { q, .. } => PostASAPSketchQuery::Quantile { q: *q }, + AggIntent::Cardinality { .. } => PostASAPSketchQuery::Cardinality, + AggIntent::FrequencyL2 { .. } => PostASAPSketchQuery::FrequencyL2, + AggIntent::FrequencyEntropy { .. } => PostASAPSketchQuery::FrequencyEntropy, + AggIntent::TopK { k, .. } => PostASAPSketchQuery::TopK { k: *k }, + AggIntent::Count { .. } => PostASAPSketchQuery::PointCount { key: match &input.weight { SummaryInputExpr::Column(col) => col.clone(), SummaryInputExpr::Constant(1.0) => ColumnRef::SampleValue, @@ -3754,7 +3763,7 @@ impl ReplacementStrategy for SharedSubtreeStrategy { } } -// ── Workload-wide search: TargetSubDAGCandidates / CandidatePostASAPDAGs / search_workload ────────── +// ── Workload-wide search: TargetSubDAGCandidates / CandidateLogicalPostASAPDAGs / search_workload ────────── // // Merged in from the former `search.rs` (issue #252, part of #33) — see this // file's own top-level "Workload-wide search" doc section for the full @@ -3770,37 +3779,37 @@ pub const MAX_SEARCH_ITERATIONS: usize = 1_000; // ── TargetSubDAGCandidates ────────────────────────────────────────────── /// Candidates for one distinct [`TargetSubDAG`] (its -/// own `target` `Rc`, keyed by pointer identity in -/// [`CandidatePostASAPDAGs`]'s internal map — never re-derived by value) plus every +/// own `target` `Rc`, keyed by pointer identity in +/// [`CandidateLogicalPostASAPDAGs`]'s internal map — never re-derived by value) plus every /// [`ReplacementSubDAG`] alternative any registered [`ReplacementStrategy`] /// proposed for it. /// /// `candidates` is deliberately *not* required to be non-empty — a /// `TargetSubDAG` no registered strategy has an opinion on still gets a -/// group (with an empty candidate list), so [`CandidatePostASAPDAGs`] always has +/// group (with an empty candidate list), so [`CandidateLogicalPostASAPDAGs`] always has /// exactly one group per discovered `TargetSubDAG`, not "one group per /// `TargetSubDAG` something matched". #[derive(Debug, Clone)] pub struct TargetSubDAGCandidates { /// The target sub-DAG this group is for. - pub target: Rc, + pub target: Rc, /// How many operator-child positions across the whole workload /// reference this exact `Rc` — see [`discover_targets`]. pub consumer_count: usize, /// Every distinct alternative discovered for `target`, in discovery - /// order (not ranked — see [`CandidatePostASAPDAGs::cost_sorted`] for the ranked + /// order (not ranked — see [`CandidateLogicalPostASAPDAGs::cost_sorted`] for the ranked /// view). pub candidates: Vec, /// Every candidate a strategy considered for `target` but refused on /// accuracy-legality grounds (issue #172), plus any `candidates` entry /// the root-target check ([`search_workload_with_targets`]) moved here. - /// Never ranked — [`CandidatePostASAPDAGs::cost_sorted`]/[`CandidatePostASAPDAGs::global_selection`] + /// Never ranked — [`CandidateLogicalPostASAPDAGs::cost_sorted`]/[`CandidateLogicalPostASAPDAGs::global_selection`] /// read only `candidates`, so a [`CostModel`] cannot resurrect one. pub rejected: Vec, } impl TargetSubDAGCandidates { - fn new(target: Rc, consumer_count: usize) -> Self { + fn new(target: Rc, consumer_count: usize) -> Self { Self { target, consumer_count, @@ -3844,7 +3853,7 @@ impl TargetSubDAGCandidates { /// Are `existing` and `candidate` the same [`Replacement::Rewrite`] /// candidate for a group targeting `target`? /// -/// Structural (`QueryExpr`) value equality alone is *not* enough here: this +/// Structural (`PreASAPNode`) value equality alone is *not* enough here: this /// module's one shipped multi-candidate `Replacement::Rewrite` source, /// [`SharedSubtreeStrategy`], deliberately returns **two** candidates that /// are value-equal to each other (`build once and share` vs. `build @@ -3863,7 +3872,7 @@ impl TargetSubDAGCandidates { /// So: two candidates whose "is this the target's own `Rc`?" bit disagrees /// are never duplicates of each other, full stop. Only when that bit /// *agrees* does this fall through to the real dedup discipline — -/// [`structural_hash`] as a candidate-narrowing filter, `QueryExpr`'s +/// [`structural_hash`] as a candidate-narrowing filter, `PreASAPNode`'s /// derived `PartialEq` as the actual decision — protecting against the /// (currently hypothetical, since neither shipped strategy causes it) /// case of the exact same alternative being proposed twice. A fresh @@ -3872,9 +3881,9 @@ impl TargetSubDAGCandidates { /// wider traversal to amortize the cache across the way `InternTable`'s own /// use of `structural_hash` does. fn is_duplicate_rewrite( - existing: &Rc, - candidate: &Rc, - target: &Rc, + existing: &Rc, + candidate: &Rc, + target: &Rc, ) -> bool { let existing_is_target = Rc::ptr_eq(existing, target); let candidate_is_target = Rc::ptr_eq(candidate, target); @@ -3889,9 +3898,9 @@ fn is_duplicate_rewrite( /// Are `existing` and `candidate` the same [`Replacement::Summary`] /// candidate? /// -/// [`SummaryNode`] derives neither `PartialEq` nor `Hash` (it embeds +/// [`PostASAPNode`] derives neither `PartialEq` nor `Hash` (it embeds /// `SketchParams`/`f64`-bearing accuracy targets deep inside `SummaryExpr`, -/// the same reason `QueryExpr` can't derive `Hash` either — see +/// the same reason `PreASAPNode` can't derive `Hash` either — see /// [`structural_hash`]'s own doc). Per this module's inherited "hash is a /// filter, `PartialEq` is the decision, no exceptions" rule, there is no /// real equality check to back a dedup *decision* here — and skipping the @@ -3904,43 +3913,44 @@ fn is_duplicate_rewrite( /// shipped today already return a structurally distinct candidate for every /// entry of one `replacements()` call, so this is future-proofing against a /// hypothetical repeat call, not a gap either strategy's own tests exercise. -fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc) -> bool { +fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc) -> bool { false } -// ── CandidatePostASAPDAGs ──────────────────────────────────────────────────────────── +// ── CandidateLogicalPostASAPDAGs ──────────────────────────────────────────────────────────── /// The deduped candidate space [`search_workload`]/[`search_workload_with`] /// discover: one [`TargetSubDAGCandidates`] per distinct `TargetSubDAG` in the /// (already-CSE'd) workload, plus the workload's own post-CSE roots so a -/// caller can still map a `Root`'s `Id` back to the `Rc` whose +/// caller can still map a `Root`'s `Id` back to the `Rc` whose /// group holds its alternatives. -pub struct CandidatePostASAPDAGs { +pub struct CandidateLogicalPostASAPDAGs { /// The workload's roots, after the one `share_common_subtrees` pass /// [`search_workload_with`] runs up front — the same post-CSE roots - /// every `TargetSubDAG` in `groups` was discovered from. - pub roots: Vec<(Id, Rc)>, - groups: HashMap<*const QueryExpr, TargetSubDAGCandidates>, - /// Discovery order — stable iteration for [`CandidatePostASAPDAGs::target_subdag_candidates`]/ - /// [`CandidatePostASAPDAGs::cost_sorted`], since `HashMap` iteration order isn't. - order: Vec<*const QueryExpr>, + /// every `TargetSubDAG` in `groups` was discovered from. Read through + /// [`Self::roots`]: the groups are keyed by these roots' nodes. + pub(crate) roots: asap_types::pre_asap::CandidatePreASAPDAGs, + groups: HashMap<*const PreASAPNode, TargetSubDAGCandidates>, + /// Discovery order — stable iteration for [`CandidateLogicalPostASAPDAGs::target_subdag_candidates`]/ + /// [`CandidateLogicalPostASAPDAGs::cost_sorted`], since `HashMap` iteration order isn't. + order: Vec<*const PreASAPNode>, /// Composition proofs are computed with the search model, then retained /// through costing and DAG assembly so no later default can replace it. composition_plans: Vec, } struct PreparedComposition { - target: *const QueryExpr, + target: *const PreASAPNode, operation: ExactComposition, - child: Rc, - plan: Rc, + child: Rc, + plan: Rc, } -impl CandidatePostASAPDAGs { +impl CandidateLogicalPostASAPDAGs { fn prepare_compositions( &mut self, accuracy: &dyn AccuracyModel, - targets: &HashMap<*const QueryExpr, Vec>, + targets: &HashMap<*const PreASAPNode, Vec>, ) { self.composition_plans.clear(); for group in self.groups.values() { @@ -4000,13 +4010,13 @@ impl CandidatePostASAPDAGs { /// never a silently truncated inventory presented as exhaustive. #[derive(Debug)] pub struct CandidateDagInventory { - pub candidates: Vec)>>, + pub candidates: Vec)>>, pub rejected_assemblies: Vec, } -type CandidateDagChoice<'a> = (Option<&'a ReplacementSubDAG>, Option>); +type CandidateDagChoice<'a> = (Option<&'a ReplacementSubDAG>, Option>); -impl CandidatePostASAPDAGs { +impl CandidateLogicalPostASAPDAGs { pub fn enumerate_candidate_dags( &self, expansion_limit: usize, @@ -4039,7 +4049,7 @@ impl CandidatePostASAPDAGs { fn enumerate_candidate_roots( &self, - roots: &[(Id, Rc)], + roots: &[(Id, Rc)], expansion_limit: usize, ) -> Result, RealizationError> { let mut reachable = Vec::new(); @@ -4101,9 +4111,7 @@ impl CandidatePostASAPDAGs { .iter() .try_fold(1usize, |n, choices| n.checked_mul(choices.len())) .filter(|n| *n <= expansion_limit) - .ok_or(RealizationError::PhysicalRealization( - "candidate expansion budget exceeded; no partial inventory returned", - ))?; + .ok_or(RealizationError::ExpansionLimit(expansion_limit))?; let mut inventory = CandidateDagInventory { candidates: Vec::new(), rejected_assemblies: Vec::new(), @@ -4246,25 +4254,25 @@ impl CandidatePostASAPDAGs { /// Lifecycle-aware whole-subplan costs keyed by target and candidate identity. #[derive(Default)] pub(crate) struct CandidateCostOverrides { - costs: HashMap<(*const QueryExpr, *const ReplacementSubDAG), Cost>, - raw_costs: HashMap<*const QueryExpr, Cost>, + costs: HashMap<(*const PreASAPNode, *const ReplacementSubDAG), Cost>, + raw_costs: HashMap<*const PreASAPNode, Cost>, /// Targets for which the caller requested an atomic raw-vs-summary /// decision. Other memo groups continue through ordinary CSE selection. - finalized_targets: HashSet<*const QueryExpr>, + finalized_targets: HashSet<*const PreASAPNode>, } impl CandidateCostOverrides { - pub(crate) fn finalize_target(&mut self, target: &Rc) { + pub(crate) fn finalize_target(&mut self, target: &Rc) { self.finalized_targets.insert(Rc::as_ptr(target)); } - fn finalizes(&self, target: &Rc) -> bool { + fn finalizes(&self, target: &Rc) -> bool { self.finalized_targets.contains(&Rc::as_ptr(target)) } pub(crate) fn insert( &mut self, - target: &Rc, + target: &Rc, candidate: &ReplacementSubDAG, cost: Cost, ) { @@ -4272,22 +4280,27 @@ impl CandidateCostOverrides { .insert((Rc::as_ptr(target), candidate as *const _), cost); } - fn get(&self, target: &Rc, candidate: &ReplacementSubDAG) -> Option { + fn get(&self, target: &Rc, candidate: &ReplacementSubDAG) -> Option { self.costs .get(&(Rc::as_ptr(target), candidate as *const _)) .copied() } - pub(crate) fn insert_raw(&mut self, target: &Rc, cost: Cost) { + pub(crate) fn insert_raw(&mut self, target: &Rc, cost: Cost) { self.raw_costs.insert(Rc::as_ptr(target), cost); } - fn raw(&self, target: &Rc) -> Option { + fn raw(&self, target: &Rc) -> Option { self.raw_costs.get(&Rc::as_ptr(target)).copied() } } -impl CandidatePostASAPDAGs { +impl CandidateLogicalPostASAPDAGs { + /// The workload's post-CSE roots with their entry IDs, in input order. + pub fn roots(&self) -> &asap_types::pre_asap::CandidatePreASAPDAGs { + &self.roots + } + /// One candidate set per discovered target sub-DAG, in discovery order. pub fn target_subdag_candidates(&self) -> impl Iterator { self.order.iter().map(move |ptr| &self.groups[ptr]) @@ -4299,7 +4312,7 @@ impl CandidatePostASAPDAGs { } /// Whether no targets were discovered at all (an empty workload, or one - /// with no `QueryExpr` nodes reachable from any root — never true for a + /// with no `PreASAPNode` nodes reachable from any root — never true for a /// non-empty `roots`, since every root is itself a target). pub fn is_empty(&self) -> bool { self.groups.is_empty() @@ -4308,7 +4321,10 @@ impl CandidatePostASAPDAGs { /// The candidate set for `target`, if `target`'s own `Rc` is a discovered /// `TargetSubDAG` (i.e. `Rc::ptr_eq` to some node reachable from /// `roots`). - pub fn candidates_for_target(&self, target: &Rc) -> Option<&TargetSubDAGCandidates> { + pub fn candidates_for_target( + &self, + target: &Rc, + ) -> Option<&TargetSubDAGCandidates> { self.groups.get(&Rc::as_ptr(target)) } @@ -4425,31 +4441,31 @@ impl CandidatePostASAPDAGs { // ── Recurrence-aware cost context (issue #287) ────────────────────────── /// One [`RecurrenceProfile`] per discovered [`TargetSubDAGCandidates`] target, built by -/// [`CandidatePostASAPDAGs::recurrence_profiles`] — the "carry `RepeatingEntry.demand` +/// [`CandidateLogicalPostASAPDAGs::recurrence_profiles`] — the "carry `RepeatingEntry.demand` /// and relevant `DataWorkload` into ASAP-aware search/cost context" /// half of issue #287. Looked up by `Rc` pointer identity, the same -/// currency [`CandidatePostASAPDAGs::candidates_for_target`]/[`GlobalSelection::for_target`] already +/// currency [`CandidateLogicalPostASAPDAGs::candidates_for_target`]/[`GlobalSelection::for_target`] already /// use. -/// Holds an owned `Rc` clone alongside each profile (not just its +/// Holds an owned `Rc` clone alongside each profile (not just its /// raw pointer) so this map keeps every node it describes alive for as long /// as the map itself lives — a `RecurrenceProfileMap` is safe to outlive the -/// `CandidatePostASAPDAGs` it was built from. Without this, a raw `*const QueryExpr` key -/// could, after the originating `CandidatePostASAPDAGs` (the only other owner of those +/// `CandidateLogicalPostASAPDAGs` it was built from. Without this, a raw `*const PreASAPNode` key +/// could, after the originating `CandidateLogicalPostASAPDAGs` (the only other owner of those /// `Rc`s) is dropped, collide with an unrelated, later allocation that /// happens to reuse the same freed address — silently returning a stale /// profile for the wrong node (issue #287 review, bug 4). #[derive(Debug, Clone)] pub struct RecurrenceProfileMap { - profiles: HashMap<*const QueryExpr, (Rc, RecurrenceProfile)>, + profiles: HashMap<*const PreASAPNode, (Rc, RecurrenceProfile)>, } impl RecurrenceProfileMap { /// The [`RecurrenceProfile`] for `target`, or /// [`RecurrenceProfile::EMPTY`] when `target` wasn't a discovered site - /// in the [`CandidatePostASAPDAGs`] this map was built from (or carried no + /// in the [`CandidateLogicalPostASAPDAGs`] this map was built from (or carried no /// recurring/one-shot/update-rate metadata at all) — always a valid, /// "no metadata" answer, never a panic. - pub fn for_target(&self, target: &Rc) -> RecurrenceProfile { + pub fn for_target(&self, target: &Rc) -> RecurrenceProfile { self.profiles .get(&Rc::as_ptr(target)) .map(|(_, profile)| *profile) @@ -4457,7 +4473,7 @@ impl RecurrenceProfileMap { } } -impl CandidatePostASAPDAGs { +impl CandidateLogicalPostASAPDAGs { /// Build one [`RecurrenceProfile`] per discovered site, by walking every /// root's whole reachable sub-DAG (the same relational-skeleton /// traversal [`discover_targets`] itself used to discover those sites) @@ -4503,7 +4519,7 @@ impl CandidatePostASAPDAGs { /// evaluated twice. This supplies recurrence-aware selection with the /// effective structural execution rate rather than mere reachability. /// - /// **Unreachable sites**: [`CandidatePostASAPDAGs`] can contain a site no root's own + /// **Unreachable sites**: [`CandidateLogicalPostASAPDAGs`] can contain a site no root's own /// structural tree actually reaches — e.g. one only ever produced by a /// [`Replacement::Rewrite`] candidate a [`ReplacementStrategy`] invented /// (this walk only follows [`TargetSubDAGCandidates::target`]'s own structural @@ -4546,12 +4562,12 @@ impl CandidatePostASAPDAGs { } } - let mut rates: HashMap<*const QueryExpr, f64> = HashMap::new(); - let mut one_shot_counts: HashMap<*const QueryExpr, usize> = HashMap::new(); + let mut rates: HashMap<*const PreASAPNode, f64> = HashMap::new(); + let mut one_shot_counts: HashMap<*const PreASAPNode, usize> = HashMap::new(); // Sites actually reached by at least one root's own recurrence tag // during the walk below — see this method's own "Unreachable // sites" doc. - let mut reached: HashSet<*const QueryExpr> = HashSet::new(); + let mut reached: HashSet<*const PreASAPNode> = HashSet::new(); for ((_, root), recurrence) in self.roots.iter().zip(root_recurrence) { let recurrence = *recurrence; @@ -4561,7 +4577,7 @@ impl CandidatePostASAPDAGs { // recomputed occurrence is evaluated twice as well; stopping // expansion after the first pointer visit undercounts exactly // the effective-consumer rate recurrence-aware costing needs. - let mut queue: VecDeque<(*const QueryExpr, usize)> = VecDeque::new(); + let mut queue: VecDeque<(*const PreASAPNode, usize)> = VecDeque::new(); queue.push_back((root_ptr, 1)); while let Some((ptr, path_count)) = queue.pop_front() { @@ -4629,7 +4645,7 @@ impl CandidatePostASAPDAGs { &self, workload: &QueryWorkload, data_workload: Option<&DataWorkload>, - // For each `CandidatePostASAPDAGs::roots[i]`, the explicit index of its + // For each `CandidateLogicalPostASAPDAGs::roots[i]`, the explicit index of its // corresponding normalized workload entry. root_workload_entries: &[usize], now_ms: u64, @@ -4705,7 +4721,7 @@ impl CandidatePostASAPDAGs { &self, workload: &QueryWorkload, root_workload_entries: &[usize], - ) -> Result>, RecurrenceError> { + ) -> Result>, RecurrenceError> { let entry_count = workload.entries().count(); if root_workload_entries.len() != self.roots.len() { return Err(RecurrenceError::RootCountMismatch { @@ -4713,7 +4729,7 @@ impl CandidatePostASAPDAGs { got: root_workload_entries.len(), }); } - let mut bindings: HashMap<*const QueryExpr, HashSet> = HashMap::new(); + let mut bindings: HashMap<*const PreASAPNode, HashSet> = HashMap::new(); for ((_, root), &entry_index) in self.roots.iter().zip(root_workload_entries) { if entry_index >= entry_count { return Err(RecurrenceError::InvalidWorkloadEntry { @@ -4750,17 +4766,17 @@ impl CandidatePostASAPDAGs { /// Record `times` occurrences of `recurrence` against `ptr` — `times > 1` /// when a single parent structurally references `ptr` more than once (see -/// [`CandidatePostASAPDAGs::recurrence_profiles`]'s own doc on edge multiplicity). +/// [`CandidateLogicalPostASAPDAGs::recurrence_profiles`]'s own doc on edge multiplicity). /// A no-op for `times == 0` (an `Rc` returned as a `direct_child_counts` /// child always has `edge_count >= 1` in practice, but this keeps the /// helper correct regardless). fn contribute( - ptr: *const QueryExpr, + ptr: *const PreASAPNode, times: usize, recurrence: RootRecurrence, - rates: &mut HashMap<*const QueryExpr, f64>, - one_shot_counts: &mut HashMap<*const QueryExpr, usize>, - reached: &mut HashSet<*const QueryExpr>, + rates: &mut HashMap<*const PreASAPNode, f64>, + one_shot_counts: &mut HashMap<*const PreASAPNode, usize>, + reached: &mut HashSet<*const PreASAPNode>, ) { if times == 0 { return; @@ -4778,10 +4794,10 @@ fn contribute( } /// One [`TargetSubDAGCandidates`]'s candidates, ranked best-first by -/// [`CandidatePostASAPDAGs::cost_sorted`]. +/// [`CandidateLogicalPostASAPDAGs::cost_sorted`]. #[derive(Debug)] pub struct RankedTargetSubDAGCandidates<'a> { - pub target: &'a Rc, + pub target: &'a Rc, pub consumer_count: usize, pub candidates: Vec<&'a ReplacementSubDAG>, /// `costs[i]` is `candidates[i]`'s own grouping-state cost when available, @@ -4935,7 +4951,7 @@ fn cse_preference(group: &TargetSubDAGCandidates, cost_model: &dyn CostModel) -> }) } -/// [`cse_preference`] only needs one representative bound [`SummaryNode`] +/// [`cse_preference`] only needs one representative bound [`PostASAPNode`] /// for `target` (to build a [`CseCandidate`] for /// [`CostModel::cse_share_decision`]), not the full ranked candidate list /// [`SketchAlgorithmStrategy::replacements`] returns — so this just reuses @@ -4943,7 +4959,7 @@ fn cse_preference(group: &TargetSubDAGCandidates, cost_model: &dyn CostModel) -> /// `construct_summary_agg`'s own recursion and /// [`crate::cost_model::DefaultCostModel::estimate_cost`] already use, /// wrapped to swallow the (here, uninteresting) error into `None`. -fn realize_one(target: &Rc, cost_model: &dyn CostModel) -> Option> { +fn realize_one(target: &Rc, cost_model: &dyn CostModel) -> Option> { realize_child(target, cost_model).ok() } @@ -4958,7 +4974,7 @@ fn realize_one(target: &Rc, cost_model: &dyn CostModel) -> Option Option { +fn sketch_kind_of(node: &PostASAPNode) -> Option { match &node.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_kind_of(summary_input), SummaryExpr::SummaryAgg { @@ -4971,7 +4987,7 @@ fn sketch_kind_of(node: &SummaryNode) -> Option { /// The grouping strategy used by a bound summary candidate, unwrapping its /// readout node when necessary. -fn summary_grouping(node: &SummaryNode) -> Option<&GroupingStrategy> { +fn summary_grouping(node: &PostASAPNode) -> Option<&GroupingStrategy> { match &node.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => summary_grouping(summary_input), SummaryExpr::SummaryAgg { grouping, .. } => Some(grouping), @@ -4982,12 +4998,12 @@ fn summary_grouping(node: &SummaryNode) -> Option<&GroupingStrategy> { // ── global_selection ───────────────────────────────────────────────────── /// One target sub-DAG's selected choice and usage information — the answer -/// [`CandidatePostASAPDAGs::global_selection`] commits to for one site, after folding in +/// [`CandidateLogicalPostASAPDAGs::global_selection`] commits to for one site, after folding in /// every ancestor [`SharedSubtreeStrategy`] decision on the path from a /// workload root to this site. See the module docs' "Whole-plan /// (cross-group) selection" section for the full recurrence. /// -/// Contrast with [`RankedTargetSubDAGCandidates`] ([`CandidatePostASAPDAGs::cost_sorted`]'s output): +/// Contrast with [`RankedTargetSubDAGCandidates`] ([`CandidateLogicalPostASAPDAGs::cost_sorted`]'s output): /// that ranks every candidate for one target in isolation and never commits /// to just one; this commits to exactly one (or none), and the count it /// ranks against — [`Self::effective_consumer_count`] — can differ from the @@ -4999,7 +5015,7 @@ fn summary_grouping(node: &SummaryNode) -> Option<&GroupingStrategy> { #[derive(Debug)] pub struct TargetSubDAGSelection<'a> { /// The target sub-DAG this selection is for. - pub target: &'a Rc, + pub target: &'a Rc, /// [`TargetSubDAGCandidates::consumer_count`] — how many operator-child positions /// directly reference `target`, ignoring every ancestor's own choice. pub consumer_count: usize, @@ -5024,15 +5040,15 @@ pub struct TargetSubDAGSelection<'a> { pub composition: Option>, } -/// Why [`CandidatePostASAPDAGs::global_selection`] committed an exact composition at a +/// Why [`CandidateLogicalPostASAPDAGs::global_selection`] committed an exact composition at a /// site: which child candidate it composes with, and the /// cost-units-per-second comparison against the raw fallback that it won. #[derive(Debug)] pub struct CompositionDecision<'a> { /// The exact child/operation pair validated by the search accuracy model. - pub plan: Rc, + pub plan: Rc, /// The child target the composed operator consumes. - pub child_target: &'a Rc, + pub child_target: &'a Rc, /// For a read-time operation: the child's own candidate committed alongside /// (the summary readout the operator folds). `None` for an update-path /// transform, whose input is raw update data — its cost is charged to @@ -5047,17 +5063,17 @@ pub struct CompositionDecision<'a> { pub inputs: ExactCompositionCostInputs, } -/// [`CandidatePostASAPDAGs::global_selection`]'s result: one [`TargetSubDAGSelection`] per -/// discovered site, in the same discovery order [`CandidatePostASAPDAGs::target_subdag_candidates`]/ -/// [`CandidatePostASAPDAGs::cost_sorted`] use. +/// [`CandidateLogicalPostASAPDAGs::global_selection`]'s result: one [`TargetSubDAGSelection`] per +/// discovered site, in the same discovery order [`CandidateLogicalPostASAPDAGs::target_subdag_candidates`]/ +/// [`CandidateLogicalPostASAPDAGs::cost_sorted`] use. #[derive(Debug)] pub struct GlobalSelection<'a> { - order: Vec<*const QueryExpr>, - groups: HashMap<*const QueryExpr, TargetSubDAGSelection<'a>>, + order: Vec<*const PreASAPNode>, + groups: HashMap<*const PreASAPNode, TargetSubDAGSelection<'a>>, /// [`Self::assemble_selected_dag`]'s memo — one bound node per target for the /// life of this selection, so two parents composing over one shared - /// child get the *same* `Rc`. - assembled_nodes: RefCell>>, + /// child get the *same* `Rc`. + assembled_nodes: RefCell>>, } fn normalize_cross_input_equi_predicate( @@ -5065,7 +5081,7 @@ fn normalize_cross_input_equi_predicate( left_width: usize, total_width: usize, ) -> Option { - let QueryExpr::Compare { + let PreASAPNode::Compare { left, op: asap_types::pre_asap::CompareOpKind::Eq, right, @@ -5073,7 +5089,8 @@ fn normalize_cross_input_equi_predicate( else { return None; }; - let (QueryExpr::Column(left_id), QueryExpr::Column(right_id)) = (left.as_ref(), right.as_ref()) + let (PreASAPNode::Column(left_id), PreASAPNode::Column(right_id)) = + (left.as_ref(), right.as_ref()) else { return None; }; @@ -5086,10 +5103,10 @@ fn normalize_cross_input_equi_predicate( } else { return None; }; - Some(Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(left_id)), + Some(Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(left_id)), op: asap_types::pre_asap::CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(right_id)), + right: Rc::new(PreASAPNode::Column(right_id)), }))) } @@ -5111,13 +5128,13 @@ impl<'a> GlobalSelection<'a> { /// The selection for `target`, if `target`'s own `Rc` is a discovered /// site (i.e. `Rc::ptr_eq` to some node reachable from the workload's /// roots). - pub fn for_target(&self, target: &Rc) -> Option<&TargetSubDAGSelection<'a>> { + pub fn for_target(&self, target: &Rc) -> Option<&TargetSubDAGSelection<'a>> { self.groups.get(&Rc::as_ptr(target)) } /// Link this selection's per-site decisions into one data_state-validated /// post-ASAP DAG rooted at `target` — the one place a committed - /// composition's child *reference* becomes an actual `Rc` + /// composition's child *reference* becomes an actual `Rc` /// edge (issue #171). `None` if `target` is not a discovered site. /// /// Per site: a [`Replacement::ExactComposition`] uses its validated @@ -5131,8 +5148,8 @@ impl<'a> GlobalSelection<'a> { /// shared inner summary is one `Rc` no matter how many roots reach it. pub fn assemble_selected_dag( &self, - target: &Rc, - ) -> Result>, RealizationError> { + target: &Rc, + ) -> Result>, RealizationError> { if !self.groups.contains_key(&Rc::as_ptr(target)) { return Ok(None); } @@ -5144,14 +5161,17 @@ impl<'a> GlobalSelection<'a> { /// callers exposing query results must use this boundary instead. pub fn assemble_selected_query( &self, - target: &Rc, - ) -> Result>, RealizationError> { + target: &Rc, + ) -> Result>, RealizationError> { self.assemble_selected_dag(target)? .map(|node| finalize_query_candidate(node, target)) .transpose() } - fn assemble_target(&self, target: &Rc) -> Result, RealizationError> { + fn assemble_target( + &self, + target: &Rc, + ) -> Result, RealizationError> { let ptr = Rc::as_ptr(target); if let Some(node) = self.assembled_nodes.borrow().get(&ptr) { return Ok(Rc::clone(node)); @@ -5201,9 +5221,9 @@ impl<'a> GlobalSelection<'a> { /// one opaque `KeepPreAsap` subtree. fn assemble_residual( &self, - target: &Rc, - ) -> Result, RealizationError> { - if let QueryExpr::Join { + target: &Rc, + ) -> Result, RealizationError> { + if let PreASAPNode::Join { left, right, kind, @@ -5222,7 +5242,7 @@ impl<'a> GlobalSelection<'a> { let right = finalize_query_candidate(self.assemble_target(right)?, right)?; let guarantee = relational_join_guarantee(left.guarantee.as_ref(), right.guarantee.as_ref()); - let node = Rc::new(SummaryNode { + let node = Rc::new(PostASAPNode { expr: SummaryExpr::RelationalJoin { left, right, @@ -5237,7 +5257,7 @@ impl<'a> GlobalSelection<'a> { return Ok(node); } let (child_target, operation) = match target.as_ref() { - QueryExpr::Project { + PreASAPNode::Project { cols, qualifier, child, @@ -5248,10 +5268,10 @@ impl<'a> GlobalSelection<'a> { qualifier: qualifier.clone(), }, ), - QueryExpr::Filter { pred, child } => { + PreASAPNode::Filter { pred, child } => { (child, ValueOperation::Filter { pred: pred.clone() }) } - QueryExpr::Sort { + PreASAPNode::Sort { keys, partition_by, child, @@ -5262,18 +5282,18 @@ impl<'a> GlobalSelection<'a> { partition_by: partition_by.clone(), }, ), - QueryExpr::Limit { n, offset, child } => ( + PreASAPNode::Limit { n, offset, child } => ( child, ValueOperation::Limit { n: *n, offset: *offset, partition_by: match child.as_ref() { - QueryExpr::Sort { partition_by, .. } => partition_by.clone(), + PreASAPNode::Sort { partition_by, .. } => partition_by.clone(), _ => Default::default(), }, }, ), - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction, measures, output_names, @@ -5292,7 +5312,7 @@ impl<'a> GlobalSelection<'a> { }; let child = finalize_query_candidate(self.assemble_target(child_target)?, child_target)?; let guarantee = child.guarantee.clone(); - let node = Rc::new(SummaryNode { + let node = Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child, operation, @@ -5310,10 +5330,10 @@ impl<'a> GlobalSelection<'a> { /// maintenance; otherwise keep the candidate exactly as constructed. fn relink_summary( &self, - node: &Rc, - target: &Rc, - ) -> Result, RealizationError> { - let QueryExpr::Aggregate { + node: &Rc, + target: &Rc, + ) -> Result, RealizationError> { + let PreASAPNode::Aggregate { child: pre_child, .. } = target.as_ref() else { @@ -5342,8 +5362,8 @@ impl<'a> GlobalSelection<'a> { /// reduction of the inner summary values. Maintaining the outer SUM directly /// would hide that inner temporal aggregate inside `KeepPreAsap` and lose its /// independently selected summary. -fn query_time_nested_sum(target: &QueryExpr) -> bool { - let QueryExpr::Aggregate { +fn query_time_nested_sum(target: &PreASAPNode) -> bool { + let PreASAPNode::Aggregate { measures, having: None, child, @@ -5355,13 +5375,13 @@ fn query_time_nested_sum(target: &QueryExpr) -> bool { matches!(measures.as_slice(), [AggIntent::Sum { .. }]) && contains_aggregate(child) } -fn contains_aggregate(expr: &QueryExpr) -> bool { +fn contains_aggregate(expr: &PreASAPNode) -> bool { match expr { - QueryExpr::Aggregate { .. } => true, - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => contains_aggregate(child), + PreASAPNode::Aggregate { .. } => true, + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } => contains_aggregate(child), _ => false, } } @@ -5369,7 +5389,7 @@ fn contains_aggregate(expr: &QueryExpr) -> bool { /// Rebuild `node` (a `SummaryAgg`, possibly under a `SummaryEstimate`) with /// `new_child` as the `SummaryAgg`'s child, if the result still validates /// as maintained state; otherwise return `node` unchanged. -fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc { +fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc { match &node.expr { SummaryExpr::SummaryEstimate { summary_input, @@ -5379,7 +5399,7 @@ fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc, new_child: &Rc) -> Rc, new_child: &Rc) -> Rc) -> Option<&Rc> { +fn maintained_summary(node: &Rc) -> Option<&Rc> { match &node.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => maintained_summary(summary_input), SummaryExpr::SummaryAgg { .. } => Some(node), @@ -5434,7 +5454,7 @@ fn is_composition_candidate(candidate: &ReplacementSubDAG) -> bool { matches!(candidate.replacement, Replacement::ExactComposition(_)) } -/// Everything [`CandidatePostASAPDAGs::global_selection`] threads between sites for +/// Everything [`CandidateLogicalPostASAPDAGs::global_selection`] threads between sites for /// exact compositions (issue #171): child candidates already committed by /// an earlier parent, and the maintained summary above each site. #[derive(Default)] @@ -5442,10 +5462,10 @@ struct CompositionContext { /// child target ptr → the child's candidate an ancestor's composition /// already committed to (a later parent must compose with the *same* /// one, and the child's own selection is forced to it). - committed_child: HashMap<*const QueryExpr, *const ReplacementSubDAG>, + committed_child: HashMap<*const PreASAPNode, *const ReplacementSubDAG>, /// site ptr → the maintained `SummaryAgg` directly above it, when its /// parent chose a bound `Summary` — what an `ValueOperationAtIngestionTime` here feeds. - maintaining_parent: HashMap<*const QueryExpr, Rc>, + maintaining_parent: HashMap<*const PreASAPNode, Rc>, } /// One eligible composed alternative at a site, before the cheapest wins. @@ -5456,12 +5476,12 @@ struct CompositionOption<'a> { /// Every [`Replacement::ExactComposition`] candidate of `group` whose /// composed-plan rate is *known* and beats the raw-recompute baseline — -/// costed against each compatible child candidate already in `CandidatePostASAPDAGs` +/// costed against each compatible child candidate already in `CandidateLogicalPostASAPDAGs` /// (or the one an earlier parent committed). Unknown statistics yield no /// option at all: the conservative `KeepPreAsap` path stays. fn composition_options<'a>( group: &'a TargetSubDAGCandidates, - groups: &'a HashMap<*const QueryExpr, TargetSubDAGCandidates>, + groups: &'a HashMap<*const PreASAPNode, TargetSubDAGCandidates>, effective: usize, cost_model: &dyn CostModel, context: &CompositionContext, @@ -5480,7 +5500,7 @@ fn composition_options<'a>( continue; }; let already_committed = context.committed_child.get(&child_ptr).copied(); - let cost = |summary: &SummaryNode, shared: bool| { + let cost = |summary: &PostASAPNode, shared: bool| { let request = ExactCompositionCostRequest { target: &group.target, composition, @@ -5579,7 +5599,7 @@ fn composition_options<'a>( options } -impl CandidatePostASAPDAGs { +impl CandidateLogicalPostASAPDAGs { /// The whole-plan (cross-group) selection step the module docs' /// "Whole-plan (cross-group) selection" section describes: one /// [`TargetSubDAGSelection`] per discovered site, each ranked against an @@ -5587,7 +5607,7 @@ impl CandidatePostASAPDAGs { /// [`SharedSubtreeStrategy`] decision on the path to it — unlike /// [`Self::cost_sorted`], whose per-group ranking only ever sees a /// group's own raw [`TargetSubDAGCandidates::consumer_count`]. - /// Uncertified DDSketch ratios remain in [`CandidatePostASAPDAGs`] for downstream + /// Uncertified DDSketch ratios remain in [`CandidateLogicalPostASAPDAGs`] for downstream /// inspection but are not chosen automatically by this selector. pub fn global_selection(&self, cost_model: &dyn CostModel) -> GlobalSelection<'_> { self.global_selection_impl(cost_model, None, None, None) @@ -5628,8 +5648,8 @@ impl CandidatePostASAPDAGs { let topo = topological_order(&self.order, &graph); let mut effective_uses = graph.external_root_uses.clone(); - let mut chosen_share: HashMap<*const QueryExpr, ShareDecision> = HashMap::new(); - let mut groups: HashMap<*const QueryExpr, TargetSubDAGSelection<'_>> = HashMap::new(); + let mut chosen_share: HashMap<*const PreASAPNode, ShareDecision> = HashMap::new(); + let mut groups: HashMap<*const PreASAPNode, TargetSubDAGSelection<'_>> = HashMap::new(); let mut context = CompositionContext::default(); for ptr in &topo { @@ -5856,7 +5876,7 @@ impl CandidatePostASAPDAGs { // Record the maintained summary this site's bound candidate // builds, for a child that may compose an `ValueOperationAtIngestionTime` // beneath it. - if let (Some(Replacement::Summary(node)), QueryExpr::Aggregate { child, .. }) = + if let (Some(Replacement::Summary(node)), PreASAPNode::Aggregate { child, .. }) = (chosen.map(|c| &c.replacement), group.target.as_ref()) { if let Some(summary) = maintained_summary(node) { @@ -5944,14 +5964,14 @@ fn is_automatically_selectable(candidate: &ReplacementSubDAG, cost_model: &dyn C /// /// Composing this recurrence transitively up the whole ancestor chain (not /// just the immediate parent) is exactly what makes -/// [`CandidatePostASAPDAGs::global_selection`]'s `effective_consumer_count` differ from +/// [`CandidateLogicalPostASAPDAGs::global_selection`]'s `effective_consumer_count` differ from /// [`TargetSubDAGCandidates::consumer_count`] whenever a `RecomputeIndependently` /// ancestor sits anywhere on the path from a root to a site — see the /// module docs' "Whole-plan (cross-group) selection" section. fn multiplier( - parent_ptr: *const QueryExpr, - effective_uses: &HashMap<*const QueryExpr, usize>, - chosen_share: &HashMap<*const QueryExpr, ShareDecision>, + parent_ptr: *const PreASAPNode, + effective_uses: &HashMap<*const PreASAPNode, usize>, + chosen_share: &HashMap<*const PreASAPNode, ShareDecision>, ) -> usize { let effective = *effective_uses.get(&parent_ptr).expect( "topological_order guarantees a parent is processed (and its effective_consumer_count \ @@ -6058,7 +6078,7 @@ fn pick_shared_subtree_candidate( // ── reference graph + topological order ───────────────────────────────── -/// The parent/child structure [`CandidatePostASAPDAGs::global_selection`]'s DP walks — +/// The parent/child structure [`CandidateLogicalPostASAPDAGs::global_selection`]'s DP walks — /// built separately from [`discover_targets`]'s own `order`/`nodes`/`counts` /// maps (which only track *aggregate* reference counts, not per-parent /// breakdown or direction) rather than extending that already-reviewed, @@ -6070,17 +6090,17 @@ struct ReferenceGraph { /// [`walk_children`] itself uses (an edge count above 1 happens when /// one parent references the same child from two different fields, /// e.g. a `Join`'s `left`/`right` both being the same `Rc`). - parents_of: HashMap<*const QueryExpr, Vec<(*const QueryExpr, usize)>>, + parents_of: HashMap<*const PreASAPNode, Vec<(*const PreASAPNode, usize)>>, /// parent ptr -> every distinct child ptr it directly references — the /// reverse of `parents_of`, for [`topological_order`]'s Kahn's-algorithm /// traversal. - children_of: HashMap<*const QueryExpr, Vec<*const QueryExpr>>, + children_of: HashMap<*const PreASAPNode, Vec<*const PreASAPNode>>, /// How many of the workload's own `roots` point directly at each node — /// a node's "external" use. Nothing inside the tree decides this (it /// isn't a reference from another discovered site), so it's never /// subject to any ancestor's Share/Recompute choice — it's the base - /// case [`CandidatePostASAPDAGs::global_selection`]'s recurrence starts from. - external_root_uses: HashMap<*const QueryExpr, usize>, + /// case [`CandidateLogicalPostASAPDAGs::global_selection`]'s recurrence starts from. + external_root_uses: HashMap<*const PreASAPNode, usize>, } /// Build an ordering graph containing every edge that could be selected: @@ -6090,7 +6110,7 @@ struct ReferenceGraph { /// their relational children as before. The /// graph is deliberately only used for topological ordering; effective-use /// counts are propagated through the one candidate actually selected. -fn reference_graph(space: &CandidatePostASAPDAGs) -> ReferenceGraph { +fn reference_graph(space: &CandidateLogicalPostASAPDAGs) -> ReferenceGraph { let mut graph = ReferenceGraph { parents_of: HashMap::new(), children_of: HashMap::new(), @@ -6122,8 +6142,8 @@ fn reference_graph(space: &CandidatePostASAPDAGs) -> ReferenceGraph { /// [`ReferenceGraph`]'s fields), retaining the greatest multiplicity seen /// when the target and alternative rewrites expose the same edge. fn add_edge( - parent_ptr: *const QueryExpr, - child_ptr: *const QueryExpr, + parent_ptr: *const PreASAPNode, + child_ptr: *const PreASAPNode, edge_count: usize, graph: &mut ReferenceGraph, ) { @@ -6139,8 +6159,8 @@ fn add_edge( } fn record_possible_edges( - parent_ptr: *const QueryExpr, - node: &QueryExpr, + parent_ptr: *const PreASAPNode, + node: &PreASAPNode, graph: &mut ReferenceGraph, ) { for (child_ptr, edge_count) in direct_child_counts(node) { @@ -6150,8 +6170,8 @@ fn record_possible_edges( /// Direct relational-skeleton children and their edge multiplicities. /// `Concat` is transparent, matching [`walk_children`]'s site scope. -fn direct_child_counts(node: &QueryExpr) -> Vec<(*const QueryExpr, usize)> { - fn push(children: &mut Vec<(*const QueryExpr, usize)>, child: &Rc) { +fn direct_child_counts(node: &PreASAPNode) -> Vec<(*const PreASAPNode, usize)> { + fn push(children: &mut Vec<(*const PreASAPNode, usize)>, child: &Rc) { let ptr = Rc::as_ptr(child); match children.iter_mut().find(|(existing, _)| *existing == ptr) { Some((_, count)) => *count += 1, @@ -6159,8 +6179,8 @@ fn direct_child_counts(node: &QueryExpr) -> Vec<(*const QueryExpr, usize)> { } } - fn collect(node: &QueryExpr, children: &mut Vec<(*const QueryExpr, usize)>) { - use QueryExpr::*; + fn collect(node: &PreASAPNode, children: &mut Vec<(*const PreASAPNode, usize)>) { + use PreASAPNode::*; match node { Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => { @@ -6224,15 +6244,18 @@ fn direct_child_counts(node: &QueryExpr) -> Vec<(*const QueryExpr, usize)> { /// (first-seen-first), not a valid topological one: a node reached via two /// different root paths can have a parent that's discovered *after* it (see /// this function's own test for a worked diamond example), which is exactly -/// backwards for [`CandidatePostASAPDAGs::global_selection`]'s recurrence. -fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec<*const QueryExpr> { - let mut in_degree: HashMap<*const QueryExpr, usize> = HashMap::new(); +/// backwards for [`CandidateLogicalPostASAPDAGs::global_selection`]'s recurrence. +fn topological_order( + order: &[*const PreASAPNode], + graph: &ReferenceGraph, +) -> Vec<*const PreASAPNode> { + let mut in_degree: HashMap<*const PreASAPNode, usize> = HashMap::new(); for ptr in order { let degree = graph.parents_of.get(ptr).map(Vec::len).unwrap_or(0); in_degree.insert(*ptr, degree); } - let mut queue: VecDeque<*const QueryExpr> = order + let mut queue: VecDeque<*const PreASAPNode> = order .iter() .copied() .filter(|ptr| in_degree[ptr] == 0) @@ -6256,7 +6279,7 @@ fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec< assert_eq!( topo.len(), order.len(), - "topological_order: the discovered-site reference graph has a cycle — every QueryExpr \ + "topological_order: the discovered-site reference graph has a cycle — every PreASAPNode \ node is built from Rc children, which can't form one, so this indicates a bug in \ reference_graph rather than a real cyclic workload", ); @@ -6347,14 +6370,16 @@ pub fn default_strategies_with_evidence<'a>( // ── search_workload ────────────────────────────────────────────────────── /// Search a whole workload's pre-ASAP roots for every candidate replacement -/// [`default_strategies`] can find, deduped into a [`CandidatePostASAPDAGs`]. Candidate +/// [`default_strategies`] can find, deduped into a [`CandidateLogicalPostASAPDAGs`]. Candidate /// *generation* uses the built-in [`DefaultCostModel`] (via /// [`default_strategies`], the same way [`SketchAlgorithmStrategy::default_cost_model`] -/// does); call [`CandidatePostASAPDAGs::cost_sorted`] on the result for the final +/// does); call [`CandidateLogicalPostASAPDAGs::cost_sorted`] on the result for the final /// `sorted_by(cost_model)` step. Use [`search_workload_with`] to plug in a /// custom strategy set (e.g. built via [`default_strategies_with`] for a /// deployment-specific [`CostModel`]). -pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidatePostASAPDAGs { +pub fn search_workload( + roots: asap_types::pre_asap::CandidatePreASAPDAGs, +) -> CandidateLogicalPostASAPDAGs { search_workload_with(roots, &default_strategies()) } @@ -6374,12 +6399,12 @@ pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidatePostASAP /// section). Deduping candidate plans this way needs no /// [`CostModel`] at all — that only enters at two well-defined points: each /// [`ReplacementStrategy`] in `strategies` may already carry its own (e.g. -/// [`SketchAlgorithmStrategy::new`]'s), and [`CandidatePostASAPDAGs::cost_sorted`]'s final +/// [`SketchAlgorithmStrategy::new`]'s), and [`CandidateLogicalPostASAPDAGs::cost_sorted`]'s final /// ranking step takes one explicitly. pub fn search_workload_with<'s, Id>( - roots: Vec<(Id, Rc)>, + roots: asap_types::pre_asap::CandidatePreASAPDAGs, strategies: &[Box], -) -> CandidatePostASAPDAGs { +) -> CandidateLogicalPostASAPDAGs { let mut space = search_cse_workload_with(cse_workload(roots), strategies); space.prepare_compositions(&DefaultAccuracyModel, &HashMap::new()); space @@ -6392,7 +6417,7 @@ pub fn search_workload_with<'s, Id>( /// `accuracy_model`'s [`AccuracyModel::satisfies`]: a candidate whose /// guarantee is fully known and misses the target is moved from /// [`TargetSubDAGCandidates::candidates`] to [`TargetSubDAGCandidates::rejected`] *before* -/// [`CandidatePostASAPDAGs::cost_sorted`]/[`CandidatePostASAPDAGs::global_selection`] ever rank the +/// [`CandidateLogicalPostASAPDAGs::cost_sorted`]/[`CandidateLogicalPostASAPDAGs::global_selection`] ever rank the /// group. A constructible candidate with unknown accuracy remains visible for /// downstream review under an approximate target, but default whole-plan /// selection does not commit it. An exact target cannot accept an unknown @@ -6405,10 +6430,10 @@ pub fn search_workload_with<'s, Id>( /// Precedence against per-node `AggIntent.accuracy` is documented in /// [`crate::accuracy`]'s module docs. pub fn search_workload_with_targets<'s, Id>( - roots: Vec<(Id, Rc, Option)>, + roots: Vec<(Id, Rc, Option)>, strategies: &[Box], accuracy_model: &dyn AccuracyModel, -) -> CandidatePostASAPDAGs { +) -> CandidateLogicalPostASAPDAGs { let mut targets = Vec::with_capacity(roots.len()); let roots = roots .into_iter() @@ -6419,7 +6444,7 @@ pub fn search_workload_with_targets<'s, Id>( .collect(); let mut space = search_cse_workload_with(cse_workload(roots), strategies); // `cse_workload` preserves root order, so targets zip by position. - let root_ptrs: Vec<(*const QueryExpr, AccuracyTarget)> = space + let root_ptrs: Vec<(*const PreASAPNode, AccuracyTarget)> = space .roots .iter() .zip(targets) @@ -6511,12 +6536,14 @@ pub fn search_workload_with_targets<'s, Id>( space } -fn cse_workload(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> { - // `share_common_subtrees` wants owned `QueryExpr`s, not already-`Rc` +fn cse_workload( + roots: asap_types::pre_asap::CandidatePreASAPDAGs, +) -> asap_types::pre_asap::CandidatePreASAPDAGs { + // `share_common_subtrees` wants owned `PreASAPNode`s, not already-`Rc` // roots — the same `Rc::try_unwrap`-with-clone-fallback pattern // `asap_types::pre_asap::cse::intern_child` itself uses to recover an // owned node without cloning in the common (uniquely-owned) case. - let owned_roots: Vec<(Id, QueryExpr)> = roots + let owned_roots: Vec<(Id, PreASAPNode)> = roots .into_iter() .map(|(id, rc)| { let expr = Rc::try_unwrap(rc).unwrap_or_else(|shared| (*shared).clone()); @@ -6527,32 +6554,32 @@ fn cse_workload(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> } fn search_cse_workload_with<'s, Id>( - cse_roots: Vec<(Id, Rc)>, + cse_roots: asap_types::pre_asap::CandidatePreASAPDAGs, strategies: &[Box], -) -> CandidatePostASAPDAGs { +) -> CandidateLogicalPostASAPDAGs { let mut order = Vec::new(); let mut nodes = HashMap::new(); - let mut counts: HashMap<*const QueryExpr, usize> = HashMap::new(); + let mut counts: HashMap<*const PreASAPNode, usize> = HashMap::new(); discover_targets(&cse_roots, &mut order, &mut nodes, &mut counts); - let siblings: Vec> = order + let siblings: Vec> = order .iter() .filter_map(|ptr| { let node = &nodes[ptr]; - matches!(node.as_ref(), QueryExpr::Aggregate { .. }).then(|| Rc::clone(node)) + matches!(node.as_ref(), PreASAPNode::Aggregate { .. }).then(|| Rc::clone(node)) }) .collect(); let rollup_strategy = RollupStrategy::new(&siblings); let accuracy_reconciliation_strategy = AccuracyReconciliationStrategy::new(&siblings); - let limits: Vec> = order + let limits: Vec> = order .iter() .filter_map(|ptr| { let node = &nodes[ptr]; - matches!(node.as_ref(), QueryExpr::Limit { .. }).then(|| Rc::clone(node)) + matches!(node.as_ref(), PreASAPNode::Limit { .. }).then(|| Rc::clone(node)) }) .collect(); let topk_reuse_strategy = TopKLimitReuseStrategy::new(&limits); - let mut groups: HashMap<*const QueryExpr, TargetSubDAGCandidates> = HashMap::new(); + let mut groups: HashMap<*const PreASAPNode, TargetSubDAGCandidates> = HashMap::new(); for ptr in &order { groups.insert( *ptr, @@ -6664,7 +6691,7 @@ fn search_cse_workload_with<'s, Id>( add_effective_count_cse_candidates(&order, &mut groups); - CandidatePostASAPDAGs { + CandidateLogicalPostASAPDAGs { roots: cse_roots, groups, order, @@ -6677,10 +6704,11 @@ fn search_cse_workload_with<'s, Id>( /// ancestor is recomputed. We only do this when an ordinary repeated group /// proves that `SharedSubtreeStrategy` is part of this search's strategy set. fn add_effective_count_cse_candidates( - order: &[*const QueryExpr], - groups: &mut HashMap<*const QueryExpr, TargetSubDAGCandidates>, + order: &[*const PreASAPNode], + groups: &mut HashMap<*const PreASAPNode, TargetSubDAGCandidates>, ) { - let mut possible_children: HashMap<*const QueryExpr, Vec<*const QueryExpr>> = HashMap::new(); + let mut possible_children: HashMap<*const PreASAPNode, Vec<*const PreASAPNode>> = + HashMap::new(); for ptr in order { let group = &groups[ptr]; let children = possible_children.entry(*ptr).or_default(); @@ -6748,10 +6776,10 @@ fn add_effective_count_cse_candidates( /// `Rc` and its real `consumer_count` — see the module docs' "Where /// `TargetSubDAG` discovery comes from" section for the full rationale. fn discover_targets( - roots: &[(Id, Rc)], - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, + roots: &[(Id, Rc)], + order: &mut Vec<*const PreASAPNode>, + nodes: &mut HashMap<*const PreASAPNode, Rc>, + counts: &mut HashMap<*const PreASAPNode, usize>, ) { for (_, root) in roots { walk(root, order, nodes, counts); @@ -6767,10 +6795,10 @@ fn discover_targets( /// child is already known — the case both shipped strategies always produce /// (see that section). fn discover_new_descendant_targets( - candidate: &Rc, - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, + candidate: &Rc, + order: &mut Vec<*const PreASAPNode>, + nodes: &mut HashMap<*const PreASAPNode, Rc>, + counts: &mut HashMap<*const PreASAPNode, usize>, ) { walk_children(candidate, order, nodes, counts); } @@ -6778,10 +6806,10 @@ fn discover_new_descendant_targets( /// Visit `node`: count this occurrence, and — the first time this exact /// `Rc` is seen — record it as a target and recurse into its children. fn walk( - node: &Rc, - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, + node: &Rc, + order: &mut Vec<*const PreASAPNode>, + nodes: &mut HashMap<*const PreASAPNode, Rc>, + counts: &mut HashMap<*const PreASAPNode, usize>, ) { let ptr = Rc::as_ptr(node); let already_visited = counts.contains_key(&ptr); @@ -6797,15 +6825,15 @@ fn walk( /// `asap_types::pre_asap::cse::share_common_subtrees`/`rebuild_children` /// itself uses (see that module's "Algorithm" section) and /// `tests::count_consumers` mirrors for its own fixtures. Exhaustive over -/// every `QueryExpr` variant: a new variant fails to compile here until this +/// every `PreASAPNode` variant: a new variant fails to compile here until this /// match is extended too. fn walk_children( - node: &QueryExpr, - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, + node: &PreASAPNode, + order: &mut Vec<*const PreASAPNode>, + nodes: &mut HashMap<*const PreASAPNode, Rc>, + counts: &mut HashMap<*const PreASAPNode, usize>, ) { - use QueryExpr::*; + use PreASAPNode::*; match node { Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => walk(c, order, nodes, counts), @@ -6866,7 +6894,7 @@ mod tests { use std::collections::HashMap; // Candidate shape without execution timing: what is computed, not where. - fn timing_free_shape(node: &Rc) -> serde_json::Value { + fn timing_free_shape(node: &Rc) -> serde_json::Value { fn strip(value: &mut serde_json::Value) { match value { serde_json::Value::Object(fields) => { @@ -6878,7 +6906,7 @@ mod tests { } } let mut shape = - serde_json::to_value(asap_types::post_asap::compile_post_asap_dag(node).unwrap()) + serde_json::to_value(asap_types::post_asap::export_post_asap_dag(node).unwrap()) .unwrap(); strip(&mut shape); shape @@ -6917,7 +6945,7 @@ mod tests { let inventory = search_workload(vec![(0usize, root)]) .enumerate_candidate_dags(4096) .unwrap(); - let is_exact = |node: &SummaryNode, kind: ExactKind| { + let is_exact = |node: &PostASAPNode, kind: ExactKind| { matches!(&node.expr, SummaryExpr::SummaryAgg { family: SummaryFamilyType::ExactAggregate(k, _), .. } if *k == kind) @@ -7081,10 +7109,10 @@ mod tests { } fn equi_pred(left: ColumnId, right: ColumnId) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(left)), + Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(left)), op: asap_types::pre_asap::CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(right)), + right: Rc::new(PreASAPNode::Column(right)), })) } @@ -7129,7 +7157,7 @@ mod tests { .unwrap() .expect("exact ranking is legal for an approximate request"); assert!(node.guarantee.as_ref().unwrap().is_exact()); - asap_types::post_asap::compile_post_asap_dag(&node).unwrap(); + asap_types::post_asap::export_post_asap_dag(&node).unwrap(); } // Exact Top-K consumes the Planner's maintained temporal values. @@ -7181,7 +7209,7 @@ mod tests { assert_eq!(keys.len(), 1); assert!(!keys[0].ascending); assert_eq!(node.schema, values.schema); - asap_types::post_asap::compile_post_asap_dag(&node).unwrap(); + asap_types::post_asap::export_post_asap_dag(&node).unwrap(); } } @@ -7192,7 +7220,7 @@ mod tests { impl AccuracyEvidenceProvider for Domain { fn quantile_input_domain( &self, - _: &QueryExpr, + _: &PreASAPNode, ) -> Option { Some(crate::accuracy::QuantileInputDomain { lower: 1.0, @@ -7818,21 +7846,21 @@ mod tests { // ── SketchAlgorithmStrategy / SharedSubtreeStrategy fixtures ─────────── - fn metric_scan(labels: &[&str]) -> QueryExpr { + fn metric_scan(labels: &[&str]) -> PreASAPNode { let mut columns = vec![ Column::new("ts", DataType::Timestamp, false), Column::new("value", DataType::Float64, false), ]; columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); - QueryExpr::Scan { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: SchemaTy::with_time_index(columns, 0, vec![]), } } - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: ReductionTy::by(by), measures: vec![intent], output_names: vec![], @@ -7854,7 +7882,7 @@ mod tests { fn does_not_match_a_multi_intent_or_having_aggregate() { let strategy = SketchAlgorithmStrategy::default_cost_model(); - let multi = Rc::new(QueryExpr::Aggregate { + let multi = Rc::new(PreASAPNode::Aggregate { reduction: ReductionTy::by(vec![2]), measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], output_names: vec![], @@ -7866,9 +7894,9 @@ mod tests { assert!(strategy.replacements(&target).is_empty()); let mut having_q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - if let QueryExpr::Aggregate { having, .. } = &mut having_q { + if let PreASAPNode::Aggregate { having, .. } = &mut having_q { *having = Some(asap_types::pre_asap::query_expr::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true)), + PreASAPNode::Literal(asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true)), ))); } let having_q = Rc::new(having_q); @@ -7890,7 +7918,7 @@ mod tests { #[test] fn approximate_quantile_enumerates_every_summary_candidate() { // Quantile's candidate list is [Kll, DDSketch] (summary_candidates) — - // every entry must come back as its own bound SummaryNode candidate, + // every entry must come back as its own bound PostASAPNode candidate, // not just Kll (the CostModel-ranked head realizations_for_intent commits to). let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); let target = TargetSubDAG::new(&q); @@ -8090,7 +8118,7 @@ mod tests { vec![("q", Rc::clone(&outer))], &default_strategies_with(&PreferDDSketchViaCostModel), ); - let QueryExpr::Aggregate { child, .. } = space.roots[0].1.as_ref() else { + let PreASAPNode::Aggregate { child, .. } = space.roots[0].1.as_ref() else { unreachable!() }; let inner_group = space @@ -8114,7 +8142,7 @@ mod tests { /// The `SummaryFamilyType`'s committed `SketchAlgorithm`, from the top /// `SummaryAgg` reachable under a (possibly `SummaryEstimate`-wrapped) /// bound root. - fn summary_family_algorithm(node: &SummaryNode) -> SketchAlgorithm { + fn summary_family_algorithm(node: &PostASAPNode) -> SketchAlgorithm { match &node.expr { asap_types::post_asap::SummaryExpr::SummaryEstimate { summary_input, .. } => { summary_family_algorithm(summary_input) @@ -8203,8 +8231,8 @@ mod tests { /// counted at the highest (maximal) point sharing starts. Test-only: /// this module deliberately does not ship a workload-wide discovery /// pass of its own (see the module docs' "Non-goals"). - fn count_consumers(roots: &[Rc]) -> HashMap<*const QueryExpr, usize> { - fn walk(node: &Rc, counts: &mut HashMap<*const QueryExpr, usize>) { + fn count_consumers(roots: &[Rc]) -> HashMap<*const PreASAPNode, usize> { + fn walk(node: &Rc, counts: &mut HashMap<*const PreASAPNode, usize>) { let ptr = Rc::as_ptr(node); let already_visited = counts.contains_key(&ptr); *counts.entry(ptr).or_insert(0) += 1; @@ -8212,8 +8240,8 @@ mod tests { walk_children(node, counts); } } - fn walk_children(node: &QueryExpr, counts: &mut HashMap<*const QueryExpr, usize>) { - use QueryExpr::*; + fn walk_children(node: &PreASAPNode, counts: &mut HashMap<*const PreASAPNode, usize>) { + use PreASAPNode::*; match node { Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => walk(c, counts), @@ -8279,7 +8307,7 @@ mod tests { }; assert!(Rc::ptr_eq(ra, rb), "fixture sanity: the two roots merged"); - let roots: Vec> = shared.into_iter().map(|(_, rc)| rc).collect(); + let roots: Vec> = shared.into_iter().map(|(_, rc)| rc).collect(); let counts = count_consumers(&roots); let count = counts[&Rc::as_ptr(&roots[0])]; assert_eq!(count, 2); @@ -8289,7 +8317,7 @@ mod tests { assert_eq!(SharedSubtreeStrategy.replacements(&target).len(), 2); } - // ── search_workload / CandidatePostASAPDAGs / TargetSubDAGCandidates (merged from search.rs) ── + // ── search_workload / CandidateLogicalPostASAPDAGs / TargetSubDAGCandidates (merged from search.rs) ── // // Reuses this test module's own `metric_scan`/`agg` fixture helpers // above (identical to `search.rs`'s own copies, which are dropped here @@ -8315,7 +8343,7 @@ mod tests { let agg_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.as_ref(), PreASAPNode::Aggregate { .. })) .expect("an Aggregate group must be discovered"); assert_eq!(agg_group.consumer_count, 1); assert_eq!( @@ -8368,7 +8396,7 @@ mod tests { let scan_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Scan { .. })) + .find(|g| matches!(g.target.as_ref(), PreASAPNode::Scan { .. })) .expect("a Scan group must be discovered"); assert_eq!(scan_group.consumer_count, 1); assert!( @@ -8383,7 +8411,7 @@ mod tests { let space = search_workload(vec![("q", root)]); let agg_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.as_ref(), PreASAPNode::Aggregate { .. })) .unwrap(); assert_eq!(agg_group.candidates.len(), 4); assert!(agg_group.candidates.iter().any(|candidate| matches!( @@ -8506,12 +8534,12 @@ mod tests { // Different predicates so the two Filter *parents* stay distinct // (don't themselves merge under CSE) — only their shared `child` // should collapse onto one `Rc`. - let root_a = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(1)))), + let root_a = PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Int64(1)))), child: Rc::clone(&shared), }; - let root_b = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(2)))), + let root_b = PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Int64(2)))), child: Rc::clone(&shared), }; @@ -8529,14 +8557,14 @@ mod tests { // pointer as) the pre-search `shared` variable. Recover it from the // post-CSE root's own `child` field instead of the stale `shared` // handle. - let QueryExpr::Filter { + let PreASAPNode::Filter { child: post_cse_shared_a, .. } = space.roots[0].1.as_ref() else { panic!("expected a Filter root"); }; - let QueryExpr::Filter { + let PreASAPNode::Filter { child: post_cse_shared_b, .. } = space.roots[1].1.as_ref() @@ -8565,7 +8593,7 @@ mod tests { #[test] fn add_candidate_rejects_a_true_rewrite_duplicate() { // SharedSubtreeStrategy's `Replacement::Rewrite` candidates are - // real `QueryExpr` values with `PartialEq`, so `add_candidate` can + // real `PreASAPNode` values with `PartialEq`, so `add_candidate` can // (and must) actually reject a genuine repeat — unlike the // `Replacement::Summary` case (see the test below). let root = Rc::new(agg( @@ -8603,7 +8631,7 @@ mod tests { #[test] fn add_candidate_never_dedups_summary_candidates() { - // Documented, deliberate consequence of `SummaryNode` deriving no + // Documented, deliberate consequence of `PostASAPNode` deriving no // `PartialEq` (see `is_duplicate_summary`'s own doc): re-proposing // the same `Replacement::Summary` candidates DOES grow the group — // this module refuses to guess at an equality check it can't back @@ -8731,7 +8759,7 @@ mod tests { let ranked = space.cost_sorted(&PreferDDSketch); let agg_group = ranked .iter() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.as_ref(), PreASAPNode::Aggregate { .. })) .unwrap(); assert_eq!(agg_group.candidates.len(), 2); let first_kind = match &agg_group.candidates[0].replacement { @@ -8754,7 +8782,7 @@ mod tests { candidates.to_vec() } - fn estimated_subpopulation_count(&self, _target: &QueryExpr) -> Option { + fn estimated_subpopulation_count(&self, _target: &PreASAPNode) -> Option { Some(self.0) } } @@ -8777,7 +8805,7 @@ mod tests { let ranked = space.cost_sorted(&model); let aggregate = ranked .iter() - .find(|group| matches!(group.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|group| matches!(group.target.as_ref(), PreASAPNode::Aggregate { .. })) .expect("aggregate group"); let Replacement::Summary(node) = &aggregate.candidates[0].replacement else { panic!("grouping candidate must be a summary") @@ -8808,7 +8836,7 @@ mod tests { let ranked = space.cost_sorted(&DefaultCostModel); let agg_group = ranked .iter() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.as_ref(), PreASAPNode::Aggregate { .. })) .unwrap(); assert_eq!( agg_group.costs.len(), @@ -8934,7 +8962,7 @@ mod tests { let selected = space.global_selection(&DefaultCostModel); let scan_group = selected .target_selections() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Scan { .. })) + .find(|g| matches!(g.target.as_ref(), PreASAPNode::Scan { .. })) .unwrap(); assert!(scan_group.chosen.is_none()); assert_eq!(scan_group.effective_consumer_count, 1); @@ -8972,7 +9000,7 @@ mod tests { let selected = space.global_selection(&PreferDDSketch); let agg_group = selected .target_selections() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.as_ref(), PreASAPNode::Aggregate { .. })) .unwrap(); let kind = match &agg_group.chosen.unwrap().replacement { Replacement::Summary(node) => sketch_kind_of(node), @@ -9017,7 +9045,7 @@ mod tests { }, ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::new(QueryExpr::CurrentTimestamp)), + replacement: Replacement::Rewrite(Rc::new(PreASAPNode::CurrentTimestamp)), provenance: ReplacementProvenance::LogicalRewrite, rationale: "different rewrite strategy".into(), }, @@ -9076,12 +9104,12 @@ mod tests { use asap_types::pre_asap::expr_ir::ScalarValue; use asap_types::pre_asap::query_expr::Predicate; - let c = || QueryExpr::Dedup { + let c = || PreASAPNode::Dedup { cols: vec![0], child: Rc::new(metric_scan(&["job"])), }; - let a = || QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + let a = || PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(true)))), child: Rc::new(c()), }; @@ -9097,7 +9125,7 @@ mod tests { // (non-mixed) two-candidate SharedSubtreeStrategy pairs. assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); let a_rc = &space.roots[0].1; - let QueryExpr::Filter { child: c_via_a, .. } = a_rc.as_ref() else { + let PreASAPNode::Filter { child: c_via_a, .. } = a_rc.as_ref() else { panic!("expected root1/root2 to still be a Filter"); }; assert!(Rc::ptr_eq(c_via_a, &space.roots[2].1)); @@ -9203,7 +9231,7 @@ mod tests { } } - let shared = Rc::new(QueryExpr::Dedup { + let shared = Rc::new(PreASAPNode::Dedup { cols: vec![0], child: Rc::new(metric_scan(&["job"])), }); @@ -9222,17 +9250,17 @@ mod tests { use asap_types::pre_asap::expr_ir::ScalarValue; use asap_types::pre_asap::query_expr::Predicate; - let c = || QueryExpr::Dedup { + let c = || PreASAPNode::Dedup { cols: vec![0], child: Rc::new(metric_scan(&["job"])), }; - let a = || QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + let a = || PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(true)))), child: Rc::new(c()), }; let space = search_workload(vec![("root1", Rc::new(a())), ("root2", Rc::new(a()))]); let a_rc = &space.roots[0].1; - let QueryExpr::Filter { child: c_rc, .. } = a_rc.as_ref() else { + let PreASAPNode::Filter { child: c_rc, .. } = a_rc.as_ref() else { panic!("expected Filter root"); }; @@ -9269,12 +9297,12 @@ mod tests { } } - let child = || QueryExpr::Dedup { + let child = || PreASAPNode::Dedup { cols: vec![0], child: Rc::new(metric_scan(&["job"])), }; - let parent = || QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + let parent = || PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(true)))), child: Rc::new(child()), }; let space = search_workload(vec![ @@ -9282,7 +9310,7 @@ mod tests { ("root2", Rc::new(parent())), ]); let parent_rc = &space.roots[0].1; - let QueryExpr::Filter { + let PreASAPNode::Filter { child: child_rc, .. } = parent_rc.as_ref() else { @@ -9314,13 +9342,13 @@ mod tests { struct ReplaceFilterChild; impl ReplacementStrategy for ReplaceFilterChild { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { - matches!(target.root.as_ref(), QueryExpr::Filter { .. }) + matches!(target.root.as_ref(), PreASAPNode::Filter { .. }) } fn replacements(&self, _target: &TargetSubDAG<'_>) -> Vec { vec![ReplacementSubDAG { strategy: "ReplaceFilterChild", - replacement: Replacement::Rewrite(Rc::new(QueryExpr::Dedup { + replacement: Replacement::Rewrite(Rc::new(PreASAPNode::Dedup { cols: vec![0], child: Rc::new(metric_scan(&["replacement"])), })), @@ -9331,8 +9359,8 @@ mod tests { } let original_child = Rc::new(metric_scan(&["original"])); - let root = Rc::new(QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + let root = Rc::new(PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(true)))), child: Rc::clone(&original_child), }); let strategies: Vec> = vec![Box::new(ReplaceFilterChild)]; @@ -9348,14 +9376,14 @@ mod tests { else { panic!("expected logical rewrite"); }; - let QueryExpr::Dedup { + let PreASAPNode::Dedup { child: replacement_child, .. } = rewrite.as_ref() else { panic!("expected Dedup rewrite"); }; - let QueryExpr::Filter { + let PreASAPNode::Filter { child: original_child, .. } = root.as_ref() @@ -9394,7 +9422,7 @@ mod tests { fn candidate_cost(&self, _: &ReplacementSubDAG, _: &TargetSubDAG<'_>) -> Option { Some(Cost(1.0)) } - fn summary_support_evidence(&self, _: &SummaryNode) -> Option { + fn summary_support_evidence(&self, _: &PostASAPNode) -> Option { Some(false) } } @@ -9559,12 +9587,12 @@ mod tests { use asap_types::pre_asap::query_expr::Predicate; let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let root_a = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(1)))), + let root_a = PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Int64(1)))), child: Rc::new(shared.clone()), }; - let root_b = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(2)))), + let root_b = PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Int64(2)))), child: Rc::new(shared), }; let roots = vec![("a", Rc::new(root_a)), ("b", Rc::new(root_b))]; @@ -9582,7 +9610,7 @@ mod tests { ) }) .collect(); - let space = CandidatePostASAPDAGs { + let space = CandidateLogicalPostASAPDAGs { roots, groups, order: order.clone(), @@ -9593,7 +9621,7 @@ mod tests { // Discovery-order sanity: root_b comes after the shared child in // discover_targets's own order (the exact non-topological case this // test exists to cover). - let QueryExpr::Filter { + let PreASAPNode::Filter { child: shared_via_a, .. } = space.roots[0].1.as_ref() @@ -9657,12 +9685,12 @@ mod tests { self.next.set(n + 1); use asap_types::pre_asap::expr_ir::ScalarValue; use asap_types::pre_asap::query_expr::Predicate; - let fresh_inner_layer = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(n)))), + let fresh_inner_layer = PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Int64(n)))), child: Rc::clone(target.root), }; - let outer_wrapper = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + let outer_wrapper = PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(true)))), child: Rc::new(fresh_inner_layer), }; vec![ReplacementSubDAG { @@ -9688,7 +9716,7 @@ mod tests { // Moved from the former `bind.rs` (issue #251): `bind.rs`'s own // workload-wide orchestration (`implement_workload`/ // `implement_workload_with`) was deleted. Current whole-workload logical - // selection uses `CandidatePostASAPDAGs::global_selection`; these tests exercise + // selection uses `CandidateLogicalPostASAPDAGs::global_selection`; these tests exercise // `construct_summary_agg`'s schema derivation end to end through // `realize_child` — production logic that still lives in this module — // so they move here rather than disappear. Unlike `bind.rs` (an @@ -9696,8 +9724,8 @@ mod tests { // pattern by hand since `realize_child` is `pub(crate)`), these tests // call `realize_child` directly. - fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn agg_per_entity(intent: AggIntent, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: ReductionTy::PerEntity, measures: vec![intent], output_names: vec![], @@ -9715,13 +9743,13 @@ mod tests { } fn realize_first( - expr: &QueryExpr, + expr: &PreASAPNode, cost_model: &dyn CostModel, - ) -> Result, RealizationError> { + ) -> Result, RealizationError> { realize_child(&Rc::new(expr.clone()), cost_model) } - fn realize(expr: &QueryExpr) -> Result, RealizationError> { + fn realize(expr: &PreASAPNode) -> Result, RealizationError> { realize_first(expr, &DefaultCostModel) } @@ -9739,7 +9767,7 @@ mod tests { else { panic!("expected SummaryEstimate root, got {:?}", root.expr); }; - assert!(matches!(query, PostAsapSketchQuery::Quantile { q } if *q == 0.99)); + assert!(matches!(query, PostASAPSketchQuery::Quantile { q } if *q == 0.99)); // Estimate edge: plain row shape — group key + Float64 answer. assert_eq!( field(&root.schema, "quantile_0_99").dtype, @@ -9778,7 +9806,7 @@ mod tests { ) ); assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(ref e) - if matches!(**e, QueryExpr::Scan { .. }))); + if matches!(**e, PreASAPNode::Scan { .. }))); } /// A deployment-supplied [`CostModel`] can override the default KLL @@ -9874,10 +9902,10 @@ mod tests { ext_kind: &str, payload: &serde_json::Value, _col: &ColumnRef, - ) -> PostAsapSketchQuery { + ) -> PostASAPSketchQuery { assert_eq!(ext_kind, "frequency"); let value = payload["item"].as_str().map(str::to_string); - PostAsapSketchQuery::PointCount { + PostASAPSketchQuery::PointCount { key: ColumnRef::Named("item".into()), value, } @@ -9916,7 +9944,7 @@ mod tests { }; assert!(matches!( query, - PostAsapSketchQuery::PointCount { key: ColumnRef::Named(k), value: Some(v) } + PostASAPSketchQuery::PointCount { key: ColumnRef::Named(k), value: Some(v) } if k == "item" && v == "checkout" )); @@ -9965,7 +9993,7 @@ mod tests { use std::time::Duration; let q = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { + PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(metric_scan(&["job"])), }, @@ -10003,7 +10031,7 @@ mod tests { use std::time::Duration; let q = agg_per_entity( default_quantile(0.99), - QueryExpr::TimeRange { + PreASAPNode::TimeRange { range: Duration::from_secs(10), child: Rc::new(metric_scan(&["job"])), }, @@ -10115,7 +10143,7 @@ mod tests { } /// The update expression of the first `SummaryAgg` in the tree. - fn find_summary_input(node: &SummaryNode) -> Option { + fn find_summary_input(node: &PostASAPNode) -> Option { match &node.expr { SummaryExpr::SummaryAgg { input, .. } if input.item.is_none() => { Some(input.weight.clone()) @@ -10155,11 +10183,11 @@ mod tests { // logical. use asap_types::pre_asap::expr_ir::{CompareOpKind, ScalarValue}; use asap_types::pre_asap::query_expr::Predicate; - let q = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let q = PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(0)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(0.5))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Float64(0.5))), })), child: Rc::new(agg(vec![], default_quantile(0.99), metric_scan(&[]))), }; @@ -10172,8 +10200,8 @@ mod tests { use asap_types::pre_asap::expr_ir::ScalarValue; use asap_types::pre_asap::query_expr::Predicate; let mut q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - if let QueryExpr::Aggregate { having, .. } = &mut q { - *having = Some(Predicate(Rc::new(QueryExpr::Literal( + if let PreASAPNode::Aggregate { having, .. } = &mut q { + *having = Some(Predicate(Rc::new(PreASAPNode::Literal( ScalarValue::Boolean(true), )))); } @@ -10182,7 +10210,7 @@ mod tests { SummaryExpr::KeepPreAsap(_) )); - let multi = QueryExpr::Aggregate { + let multi = PreASAPNode::Aggregate { reduction: ReductionTy::by(vec![2]), measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], output_names: vec![], @@ -10244,7 +10272,7 @@ mod tests { &self, op: &CompositionOperator, _family: &SummaryFamilyType, - _query: Option<&PostAsapSketchQuery>, + _query: Option<&PostASAPSketchQuery>, ) -> PropagationStats { if matches!(op, CompositionOperator::TopKSelection) { PropagationStats { @@ -10339,7 +10367,7 @@ mod tests { else { panic!("expected Top-K readout") }; - assert!(matches!(query, PostAsapSketchQuery::TopK { k: 10 })); + assert!(matches!(query, PostASAPSketchQuery::TopK { k: 10 })); let SummaryExpr::SummaryAgg { child, family, @@ -10415,7 +10443,7 @@ mod tests { else { panic!("expected Top-K readout") }; - assert!(matches!(query, PostAsapSketchQuery::TopK { k: 5 })); + assert!(matches!(query, PostASAPSketchQuery::TopK { k: 5 })); let SummaryExpr::SummaryAgg { child, input, .. } = &summary_input.expr else { panic!("expected fused summary aggregation") }; @@ -10432,12 +10460,12 @@ mod tests { #[test] fn temporal_per_entity_topk_uses_series_identity_and_sample_value() { - let inner = QueryExpr::Aggregate { + let inner = PreASAPNode::Aggregate { reduction: ReductionTy::PerEntity, measures: vec![AggIntent::Sum { col: None }], output_names: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: Rc::new(PreASAPNode::TimeRange { range: std::time::Duration::from_secs(60), child: Rc::new(metric_scan(&["service"])), }), @@ -10594,7 +10622,7 @@ mod tests { fn sql_reducer_resolves_named_input_column() { // SUM(bytes) over a tabular scan: `col` resolves positionally to the // named column, not the PromQL sample value. - let scan = QueryExpr::Scan { + let scan = PreASAPNode::Scan { source: Source::Table { table_ref: "t".into(), }, @@ -10635,7 +10663,7 @@ mod tests { fn local_guarantee( &self, family: &SummaryFamilyType, - query: &PostAsapSketchQuery, + query: &PostASAPSketchQuery, ) -> Option { DefaultAccuracyModel.local_guarantee(family, query) } @@ -10677,7 +10705,7 @@ mod tests { } } - fn summary_child(node: &SummaryNode) -> &Rc { + fn summary_child(node: &PostASAPNode) -> &Rc { match &node.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => summary_child(summary_input), SummaryExpr::SummaryAgg { child, .. } => child, @@ -10977,11 +11005,11 @@ mod tests { fn scoped_hll_evidence_sizes_and_certifies_without_a_deployment_model() { use crate::accuracy::EstimatorContract; struct SourceEvidence { - expression: QueryExpr, + expression: PreASAPNode, max_distinct: u32, } impl AccuracyEvidenceProvider for SourceEvidence { - fn estimator_contract(&self, expression: &QueryExpr) -> Option { + fn estimator_contract(&self, expression: &PreASAPNode) -> Option { (expression == &self.expression).then_some(EstimatorContract::ClassicHll { max_distinct_per_readout: self.max_distinct, }) @@ -11091,9 +11119,9 @@ mod tests { #[test] fn residual_projection_finalizes_selected_exact_state() { let inner = Rc::new(agg(vec![], AggIntent::Sum { col: None }, metric_scan(&[]))); - let root = Rc::new(QueryExpr::Project { + let root = Rc::new(PreASAPNode::Project { cols: vec![asap_types::pre_asap::ProjectItem { - expr: QueryExpr::Column(0), + expr: PreASAPNode::Column(0), alias: Some("result".into()), }], qualifier: None, @@ -11143,7 +11171,7 @@ mod tests { #[test] fn heap_readout_preserves_numeric_item_identity() { let mut raw = metric_scan(&["id", "description"]); - let QueryExpr::Scan { schema, .. } = &mut raw else { + let PreASAPNode::Scan { schema, .. } = &mut raw else { unreachable!() }; schema.columns[2].dtype = DataType::Int64; diff --git a/crates/asap-aware-mapping/src/rewrite.rs b/crates/asap-aware-mapping/src/rewrite.rs index 44b685f8..baa66bc8 100644 --- a/crates/asap-aware-mapping/src/rewrite.rs +++ b/crates/asap-aware-mapping/src/rewrite.rs @@ -35,7 +35,7 @@ //! - **`without(...)` grouping** leaves an `Aggregate`'s own output schema //! *open* (`closed: false`, see `without_output_schema`), while the //! `Project` this strategy always wraps the rewrite in forces -//! `closed: true` (see `QueryExpr::output_schema`'s `Project` arm). Under +//! `closed: true` (see `PreASAPNode::output_schema`'s `Project` arm). Under //! `without(...)` the rewritten form's `closed` flag would silently flip //! relative to the original — exactly the kind of schema drift this //! module exists to avoid. @@ -61,7 +61,7 @@ use std::rc::Rc; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::pre_asap::expr_ir::ArithmeticOpKind; -use asap_types::pre_asap::query_expr::{BinaryOpKind, ProjectItem, QueryExpr, Reduction}; +use asap_types::pre_asap::query_expr::{BinaryOpKind, PreASAPNode, ProjectItem, Reduction}; use asap_types::pre_asap::schema::{ColumnId, DataType}; use asap_types::types::AccuracyTarget; @@ -72,8 +72,8 @@ use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, Ta /// the module docs' "Scope" for why `without(...)`/`PerEntity` are /// excluded). Returns the grouping key count and the summed column so /// [`build_rewrite`] doesn't have to re-match. -fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { - let QueryExpr::Aggregate { +fn avg_rewrite_target(node: &PreASAPNode) -> Option<(usize, Option)> { + let PreASAPNode::Aggregate { reduction, measures, having: None, @@ -123,14 +123,14 @@ fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { /// exactly regardless of the summed column's own type (integer division /// would otherwise silently reappear whenever the input column is itself /// integer-typed: `Sum`'s output type tracks its input, `Count`'s is always -/// `Int64`, and `QueryExpr::output_schema`'s own `Arithmetic` type inference +/// `Int64`, and `PreASAPNode::output_schema`'s own `Arithmetic` type inference /// types a `Div` of two `Int64` operands as `Int64` — the explicit operand /// `Cast` is what keeps both the division and rewritten `avg` column /// `Float64` the way the original always was, not an incidental extra step). // These are conditional physical components, never an unconditional Rewrite. // The caller must attach the finite-division execution guard before admission. -pub(crate) fn temporal_average_components(root: &Rc) -> Option> { - let QueryExpr::Aggregate { +pub(crate) fn temporal_average_components(root: &Rc) -> Option> { + let PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures, child, @@ -143,7 +143,7 @@ pub(crate) fn temporal_average_components(root: &Rc) -> Option) -> Option) -> Option) -> Option) -> Option> { +fn build_rewrite(root: &Rc) -> Option> { let (group_count, col) = avg_rewrite_target(root)?; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, output_names, child, @@ -187,7 +187,7 @@ fn build_rewrite(root: &Rc) -> Option> { // The original `avg` column's own name: `output_names[0]` if the // producing front end overrode it (SQL threading DataFusion's own - // generated name — see `QueryExpr::Aggregate::output_names`'s docs), + // generated name — see `PreASAPNode::Aggregate::output_names`'s docs), // else `AggIntent::Avg`'s synthetic default. Either way this is the // *only* thing about the original output column this rewrite needs to // reproduce — `AggIntent::Avg::output_column`'s `(Float64, nullable: @@ -199,14 +199,14 @@ fn build_rewrite(root: &Rc) -> Option> { .cloned() .unwrap_or_else(|| "avg".to_string()); - let sum_agg = Rc::new(QueryExpr::Aggregate { + let sum_agg = Rc::new(PreASAPNode::Aggregate { reduction: reduction.clone(), measures: vec![AggIntent::Sum { col }], output_names: Vec::new(), having: None, child: Rc::clone(child), }); - let count_agg = Rc::new(QueryExpr::Aggregate { + let count_agg = Rc::new(PreASAPNode::Aggregate { reduction: reduction.clone(), measures: vec![AggIntent::Count { accuracy: AccuracyTarget::Exact, @@ -220,24 +220,24 @@ fn build_rewrite(root: &Rc) -> Option> { let mut cols: Vec = (0..group_count) .map(|i| ProjectItem { alias: None, - expr: QueryExpr::Column(i), + expr: PreASAPNode::Column(i), }) .collect(); cols.push(ProjectItem { alias: Some(avg_name), - expr: QueryExpr::Cast { - expr: Rc::new(QueryExpr::Column(sum_idx)), + expr: PreASAPNode::Cast { + expr: Rc::new(PreASAPNode::Column(sum_idx)), to: DataType::Float64, try_cast: false, }, }); - let float_sum = Rc::new(QueryExpr::Project { + let float_sum = Rc::new(PreASAPNode::Project { cols, qualifier: None, child: sum_agg, }); - Some(Rc::new(QueryExpr::BinaryOp { + Some(Rc::new(PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), lhs: float_sum, rhs: count_agg, @@ -247,9 +247,9 @@ fn build_rewrite(root: &Rc) -> Option> { /// Compose adjacent per-entity and cross-entity accumulators when their /// algebra, rather than a query-language spelling, proves equivalence. -pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option> { +pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option> { let original_schema = root.output_schema().ok()?; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction: outer_reduction @ Reduction::Reduce(_), measures: outer_measures, output_names, @@ -259,7 +259,7 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option) -> Option inner.clone(), _ => return None, }; - let aggregate = Rc::new(QueryExpr::Aggregate { + let aggregate = Rc::new(PreASAPNode::Aggregate { reduction: outer_reduction.clone(), measures: vec![composed], output_names: output_names.clone(), @@ -302,7 +302,7 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option = (0..by.keys().len()) .map(|i| ProjectItem { alias: None, - expr: QueryExpr::Column(i), + expr: PreASAPNode::Column(i), }) .collect(); cols.push(ProjectItem { @@ -313,13 +313,13 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option QueryExpr { + fn metric_scan(labels: &[&str]) -> PreASAPNode { let mut columns = vec![ Column::new("ts", DataType::Timestamp, false), Column::new("value", DataType::Float64, false), ]; columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); - QueryExpr::Scan { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index(columns, 0, vec![]), } } - fn avg_agg(by: Vec, col: Option, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn avg_agg(by: Vec, col: Option, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::by(by), measures: vec![AggIntent::Avg { col }], output_names: vec![], @@ -434,7 +434,7 @@ mod tests { root.output_schema().unwrap(), rewritten.output_schema().unwrap() ); - assert!(matches!(rewritten.as_ref(), QueryExpr::BinaryOp { .. })); + assert!(matches!(rewritten.as_ref(), PreASAPNode::BinaryOp { .. })); } // ── matches ────────────────────────────────────────────────────────── @@ -455,7 +455,7 @@ mod tests { #[test] fn does_not_match_a_multi_measure_aggregate() { - let q = Rc::new(QueryExpr::Aggregate { + let q = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], output_names: vec![], @@ -470,9 +470,9 @@ mod tests { #[test] fn does_not_match_a_having_bearing_avg_aggregate() { let mut q = avg_agg(vec![2], None, metric_scan(&["job"])); - if let QueryExpr::Aggregate { having, .. } = &mut q { + if let PreASAPNode::Aggregate { having, .. } = &mut q { *having = Some(asap_types::pre_asap::query_expr::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true)), + PreASAPNode::Literal(asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true)), ))); } let q = Rc::new(q); @@ -490,7 +490,7 @@ mod tests { }, AggIntent::Min { col: None }, ] { - let q = Rc::new(QueryExpr::Aggregate { + let q = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![intent.clone()], output_names: vec![], @@ -508,7 +508,7 @@ mod tests { #[test] fn does_not_match_a_without_grouped_avg_aggregate() { - let q = Rc::new(QueryExpr::Aggregate { + let q = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::Reduce(asap_types::pre_asap::query_expr::GroupKeys::without( vec![2], )), @@ -524,7 +524,7 @@ mod tests { #[test] fn does_not_match_a_per_entity_avg_aggregate() { - let q = Rc::new(QueryExpr::Aggregate { + let q = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Avg { col: None }], output_names: vec![], @@ -562,20 +562,20 @@ mod tests { Replacement::Rewrite(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let QueryExpr::BinaryOp { lhs, rhs, .. } = rewritten.as_ref() else { + let PreASAPNode::BinaryOp { lhs, rhs, .. } = rewritten.as_ref() else { panic!("expected sum/count BinaryOp, got {rewritten:?}"); }; - let QueryExpr::Project { child: sum, .. } = lhs.as_ref() else { + let PreASAPNode::Project { child: sum, .. } = lhs.as_ref() else { panic!("expected cast Project above Sum, got {lhs:?}"); }; assert!(matches!( sum.as_ref(), - QueryExpr::Aggregate { measures, .. } + PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Sum { col: None }]) )); assert!(matches!( rhs.as_ref(), - QueryExpr::Aggregate { measures, .. } + PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Count { accuracy: AccuracyTarget::Exact }]) )); @@ -593,7 +593,7 @@ mod tests { #[test] fn preserves_an_explicit_output_name_override() { let mut q = avg_agg(vec![], None, metric_scan(&[])); - if let QueryExpr::Aggregate { output_names, .. } = &mut q { + if let PreASAPNode::Aggregate { output_names, .. } = &mut q { *output_names = vec!["avg_latency".to_string()]; } let original_schema = q.output_schema().unwrap(); @@ -644,7 +644,7 @@ mod tests { let mut found_sum = false; let mut found_count = false; for group in space.target_subdag_candidates() { - let QueryExpr::Aggregate { measures, .. } = group.target.as_ref() else { + let PreASAPNode::Aggregate { measures, .. } = group.target.as_ref() else { continue; }; let expected = matches!(measures.as_slice(), [AggIntent::Sum { .. }]) @@ -682,7 +682,7 @@ mod tests { Column::new("job", DataType::Utf8, true), Column::new("bytes", DataType::Int64, false), ]; - let child = QueryExpr::Scan { + let child = PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: { @@ -713,15 +713,15 @@ mod tests { DataType::Float64 ); - let QueryExpr::BinaryOp { lhs, .. } = rewritten.as_ref() else { + let PreASAPNode::BinaryOp { lhs, .. } = rewritten.as_ref() else { panic!("expected sum/count BinaryOp"); }; - let QueryExpr::Project { cols, .. } = lhs.as_ref() else { + let PreASAPNode::Project { cols, .. } = lhs.as_ref() else { panic!("expected cast Project above Sum"); }; assert!(matches!( &cols.last().unwrap().expr, - QueryExpr::Cast { + PreASAPNode::Cast { to: DataType::Float64, .. } @@ -730,7 +730,7 @@ mod tests { #[test] fn does_not_rewrite_avg_of_a_nullable_column_via_count_star() { - let child = QueryExpr::Scan { + let child = PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -750,18 +750,18 @@ mod tests { assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); } - fn nested_aggregate(outer: AggIntent, inner: AggIntent) -> Rc { - let temporal = QueryExpr::Aggregate { + fn nested_aggregate(outer: AggIntent, inner: AggIntent) -> Rc { + let temporal = PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![inner], output_names: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: Rc::new(PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(metric_scan(&["service"])), }), }; - Rc::new(QueryExpr::Aggregate { + Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![outer], // Match the PromQL front end: an empty entry selects the intent's @@ -798,11 +798,11 @@ mod tests { rewritten.output_schema().unwrap() ); let aggregate = match rewritten.as_ref() { - QueryExpr::Aggregate { .. } => rewritten.as_ref(), - QueryExpr::Project { child, .. } => child.as_ref(), + PreASAPNode::Aggregate { .. } => rewritten.as_ref(), + PreASAPNode::Project { child, .. } => child.as_ref(), other => panic!("expected Aggregate or cast Project, got {other:?}"), }; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction: Reduction::Reduce(by), measures, child, @@ -815,9 +815,9 @@ mod tests { assert_eq!(measures, &[expected]); assert!(matches!( child.as_ref(), - QueryExpr::TimeRange { range, child } + PreASAPNode::TimeRange { range, child } if *range == Duration::from_secs(300) - && matches!(child.as_ref(), QueryExpr::Scan { .. }) + && matches!(child.as_ref(), PreASAPNode::Scan { .. }) )); } } diff --git a/crates/asap-aware-mapping/src/rollup.rs b/crates/asap-aware-mapping/src/rollup.rs index cc173672..fe98b57a 100644 --- a/crates/asap-aware-mapping/src/rollup.rs +++ b/crates/asap-aware-mapping/src/rollup.rs @@ -88,7 +88,7 @@ //! - **No materialized roll-up operator.** Actually building a pre-aggregated //! summary/scan leaf at execution time is separate, larger work outside //! `asap-aware-mapping`'s scope (see issue #254's own "Non-goal" section) -//! — this module only constructs the pre-ASAP [`QueryExpr::Aggregate`] +//! — this module only constructs the pre-ASAP [`PreASAPNode::Aggregate`] //! rewrite; a `CostModel`/search engine decides whether to prefer it. //! - **No cross-schema reconciliation** (see "`ColumnId` comparability" //! above) and **no `without(...)` grouping support** — `without`'s kept @@ -101,7 +101,7 @@ use std::collections::HashSet; use std::rc::Rc; use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{GroupKeys, QueryExpr, Reduction}; +use asap_types::pre_asap::query_expr::{GroupKeys, PreASAPNode, Reduction}; use asap_types::pre_asap::schema::{ColumnId, Schema}; use asap_types::types::AccuracyTarget; @@ -115,9 +115,9 @@ use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, Ta /// at all). `None` for anything else, including a multi-measure or `HAVING` /// aggregate, a non-`Aggregate` node, or a `PerEntity` reduction. fn bindable_grouped_aggregate( - node: &QueryExpr, -) -> Option<(&GroupKeys, &AggIntent, &Rc)> { - let QueryExpr::Aggregate { + node: &PreASAPNode, +) -> Option<(&GroupKeys, &AggIntent, &Rc)> { + let PreASAPNode::Aggregate { reduction, measures, having, @@ -259,7 +259,7 @@ fn is_strict_column_superset(finer: &[ColumnId], coarser: &[ColumnId]) -> bool { /// docs' "Non-goals" on why finding the full sibling set across a workload /// is a workload-wide traversal this strategy does not own. pub struct RollupStrategy { - siblings: Vec>, + siblings: Vec>, } impl RollupStrategy { @@ -267,7 +267,7 @@ impl RollupStrategy { /// each as a candidate roll-up source (or target) — typically the full set of `Aggregate` /// nodes a workload-wide discovery pass (issue #252) already found /// sharing at least one child `Rc` with something else. - pub fn new(siblings: &[Rc]) -> Self { + pub fn new(siblings: &[Rc]) -> Self { Self { siblings: siblings.to_vec(), } @@ -276,7 +276,7 @@ impl RollupStrategy { /// Every sibling that is a legal, strictly finer roll-up source for /// `target` — shared between `matches` and `replacements` so the two /// can never disagree about which siblings qualify. - fn finer_sources(&self, target: &TargetSubDAG<'_>) -> Vec<&Rc> { + fn finer_sources(&self, target: &TargetSubDAG<'_>) -> Vec<&Rc> { let Some((coarser_by, coarser_intent, coarser_child)) = bindable_grouped_aggregate(target.root) else { @@ -321,7 +321,7 @@ impl ReplacementStrategy for RollupStrategy { let Some((coarser_by, coarser_intent, _)) = bindable_grouped_aggregate(target.root) else { return Vec::new(); }; - let QueryExpr::Aggregate { output_names, .. } = target.root.as_ref() else { + let PreASAPNode::Aggregate { output_names, .. } = target.root.as_ref() else { unreachable!("bindable_grouped_aggregate already confirmed Aggregate"); }; self.finer_sources(target) @@ -331,7 +331,7 @@ impl ReplacementStrategy for RollupStrategy { } } -/// Build the coarser replacement: a new `QueryExpr::Aggregate` grouped by +/// Build the coarser replacement: a new `PreASAPNode::Aggregate` grouped by /// `coarser_by`'s columns (repositioned into `finer`'s own output schema — /// see below), computing `rollup_combinator(intent, ..)` over `finer`'s own /// measure column, with `child = finer` instead of the original shared @@ -346,7 +346,7 @@ impl ReplacementStrategy for RollupStrategy { /// position in the shared child to its position in `finer`'s output: the /// index its `ColumnId` occupies within `finer_by`'s own ordered list. fn build_rollup( - finer: &Rc, + finer: &Rc, coarser_by: &GroupKeys, intent: &AggIntent, output_names: &[String], @@ -363,7 +363,7 @@ fn build_rollup( .map(|id| finer_by.keys().iter().position(|f| f == id)) .collect::>>()?; - let rewritten = QueryExpr::Aggregate { + let rewritten = PreASAPNode::Aggregate { reduction: Reduction::by(remapped_by), measures: vec![combinator], output_names: output_names.to_vec(), @@ -394,8 +394,8 @@ mod tests { use asap_types::types::AccuracyTarget; /// `[ts(0), value(1), job(2), region(3)]`. - fn metric_scan() -> QueryExpr { - QueryExpr::Scan { + fn metric_scan() -> PreASAPNode { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -411,8 +411,8 @@ mod tests { } } - fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { + Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], @@ -424,9 +424,9 @@ mod tests { fn without_agg( excluded: Vec, intent: AggIntent, - child: &Rc, - ) -> Rc { - Rc::new(QueryExpr::Aggregate { + child: &Rc, + ) -> Rc { + Rc::new(PreASAPNode::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(excluded)), measures: vec![intent], output_names: vec![], @@ -577,7 +577,7 @@ mod tests { let Replacement::Rewrite(rewritten) = &replacements[0].replacement else { panic!("expected a Rewrite replacement"); }; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -638,7 +638,7 @@ mod tests { let Replacement::Rewrite(rewritten) = &replacements[0].replacement else { panic!("expected a Rewrite replacement"); }; - let QueryExpr::Aggregate { measures, .. } = rewritten.as_ref() else { + let PreASAPNode::Aggregate { measures, .. } = rewritten.as_ref() else { panic!("expected an Aggregate rewrite"); }; assert_eq!( @@ -677,7 +677,7 @@ mod tests { .find(|group| { matches!( group.target.as_ref(), - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction: Reduction::Reduce(by), .. } if by.keys() == [2] @@ -693,12 +693,12 @@ mod tests { Replacement::Summary(_) | Replacement::ExactComposition(_) => None, }) .expect("default search must include the roll-up rewrite"); - let QueryExpr::Aggregate { child, .. } = rewrite.as_ref() else { + let PreASAPNode::Aggregate { child, .. } = rewrite.as_ref() else { panic!("expected aggregate rewrite, got {rewrite:?}"); }; assert!(matches!( child.as_ref(), - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction: Reduction::Reduce(by), .. } if by.keys() == [2, 3] @@ -718,7 +718,7 @@ mod tests { .find(|group| { matches!( group.target.as_ref(), - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction: Reduction::Reduce(by), .. } if by.keys() == [2] @@ -736,7 +736,7 @@ mod tests { fn rollup_preserves_the_coarser_output_name() { let scan = Rc::new(metric_scan()); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); - let coarse = Rc::new(QueryExpr::Aggregate { + let coarse = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![AggIntent::Sum { col: Some(1) }], output_names: vec!["total_requests".into()], @@ -753,7 +753,7 @@ mod tests { }; assert_eq!(rewritten.output_schema().unwrap(), original_schema); - let QueryExpr::Aggregate { output_names, .. } = rewritten.as_ref() else { + let PreASAPNode::Aggregate { output_names, .. } = rewritten.as_ref() else { unreachable!(); }; assert_eq!(output_names, &vec!["total_requests".to_string()]); @@ -859,7 +859,7 @@ mod tests { fn does_not_match_a_multi_measure_or_having_aggregate() { let scan = Rc::new(metric_scan()); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); - let multi = Rc::new(QueryExpr::Aggregate { + let multi = Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![ AggIntent::Sum { col: Some(1) }, diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs index 4af4dabd..15f0424a 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs @@ -1,7 +1,7 @@ use super::*; pub(super) fn estimate_heterogeneous_summary( - root: &SummaryNode, + root: &PostASAPNode, deployments: &[CostedSummaryDeployment<'_>], evidence: &StreamingNodeEvidence, scope: &ComparisonScope, @@ -20,8 +20,8 @@ pub(super) fn estimate_heterogeneous_summary( .collect(); validate_summary_edges_and_physical_ids(root, evidence, &frameworks_by_node)?; fn summary_source_selections( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, + node: &PostASAPNode, + seen: &mut HashSet<*const PostASAPNode>, out: &mut Vec, ) -> Result<(), AnalyticalCostError> { if !seen.insert(node as *const _) { @@ -252,9 +252,9 @@ pub(super) fn estimate_heterogeneous_summary( #[expect(clippy::too_many_arguments, reason = "CPU and I/O traversal state")] fn visit_ops( - node: &SummaryNode, + node: &PostASAPNode, seen: &mut HashSet, - by_node: &HashMap<*const SummaryNode, &CostedSummaryDeployment<'_>>, + by_node: &HashMap<*const PostASAPNode, &CostedSummaryDeployment<'_>>, evidence: &StreamingNodeEvidence, scope: &ComparisonScope, evaluation_count: u64, @@ -405,9 +405,9 @@ pub(super) fn estimate_heterogeneous_summary( .get(&(node as *const _)) .ok_or(AnalyticalCostError::MissingOrStale("summary_delete_owner"))?; fn collect_aggs( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - out: &mut Vec<*const SummaryNode>, + node: &PostASAPNode, + seen: &mut HashSet<*const PostASAPNode>, + out: &mut Vec<*const PostASAPNode>, ) { if !seen.insert(node as *const _) { return; @@ -612,11 +612,11 @@ fn add_operator_io( } fn validate_summary_edges_and_physical_ids( - root: &SummaryNode, + root: &PostASAPNode, evidence: &StreamingNodeEvidence, - frameworks_by_node: &HashMap<*const SummaryNode, &Option>, + frameworks_by_node: &HashMap<*const PostASAPNode, &Option>, ) -> Result<(), AnalyticalCostError> { - fn children(node: &SummaryNode) -> Vec<&SummaryNode> { + fn children(node: &PostASAPNode) -> Vec<&PostASAPNode> { match &node.expr { SummaryExpr::KeepPreAsap(_) => vec![], SummaryExpr::SummaryAgg { child, .. } | SummaryExpr::ValueOperation { child, .. } => { @@ -642,7 +642,7 @@ fn validate_summary_edges_and_physical_ids( } } fn metadata( - node: &SummaryNode, + node: &PostASAPNode, evidence: &StreamingNodeEvidence, ) -> Result<(String, Vec, EdgeStatistics), AnalyticalCostError> { match &node.expr { @@ -682,10 +682,10 @@ fn validate_summary_edges_and_physical_ids( } } fn visit( - node: &SummaryNode, + node: &PostASAPNode, evidence: &StreamingNodeEvidence, - frameworks_by_node: &HashMap<*const SummaryNode, &Option>, - seen: &mut HashSet<*const SummaryNode>, + frameworks_by_node: &HashMap<*const PostASAPNode, &Option>, + seen: &mut HashSet<*const PostASAPNode>, physical: &mut HashMap, EdgeStatistics, String)>, ) -> Result { if !seen.insert(node as *const _) { @@ -756,7 +756,7 @@ fn validate_summary_edges_and_physical_ids( } fn summary_physical_id( - node: &SummaryNode, + node: &PostASAPNode, evidence: &StreamingNodeEvidence, ) -> Result { match &node.expr { @@ -785,10 +785,10 @@ fn summary_physical_id( /// child output buffers remain live until their final consumer executes; /// operator workspace and its output buffer coexist during that execution. pub(super) fn estimate_transient_liveness( - root: &SummaryNode, + root: &PostASAPNode, evidence: &StreamingNodeEvidence, ) -> Result { - fn children(node: &SummaryNode) -> Vec<&SummaryNode> { + fn children(node: &PostASAPNode) -> Vec<&PostASAPNode> { match &node.expr { SummaryExpr::KeepPreAsap(_) => vec![], SummaryExpr::SummaryAgg { child, .. } | SummaryExpr::ValueOperation { child, .. } => { @@ -814,11 +814,11 @@ pub(super) fn estimate_transient_liveness( } } fn visit<'a>( - node: &'a SummaryNode, + node: &'a PostASAPNode, evidence: &StreamingNodeEvidence, seen: &mut HashSet, uses: &mut HashMap, - order: &mut Vec<&'a SummaryNode>, + order: &mut Vec<&'a PostASAPNode>, ) -> Result<(), AnalyticalCostError> { if !seen.insert(summary_physical_id(node, evidence)?) { return Ok(()); @@ -833,7 +833,7 @@ pub(super) fn estimate_transient_liveness( Ok(()) } fn memory( - node: &SummaryNode, + node: &PostASAPNode, evidence: &StreamingNodeEvidence, ) -> Result<(u64, u64), AnalyticalCostError> { match &node.expr { @@ -901,12 +901,12 @@ pub(super) fn estimate_transient_liveness( Ok(peak) } #[cfg(test)] -pub(super) fn evidence_nodes(root: &SummaryNode) -> (Vec<&SummaryNode>, Vec<&SummaryNode>) { +pub(super) fn evidence_nodes(root: &PostASAPNode) -> (Vec<&PostASAPNode>, Vec<&PostASAPNode>) { fn visit<'a>( - node: &'a SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - aggregations: &mut Vec<&'a SummaryNode>, - joins: &mut Vec<&'a SummaryNode>, + node: &'a PostASAPNode, + seen: &mut HashSet<*const PostASAPNode>, + aggregations: &mut Vec<&'a PostASAPNode>, + joins: &mut Vec<&'a PostASAPNode>, ) { if !seen.insert(node as *const _) { return; @@ -969,7 +969,7 @@ struct SummaryOperationCounts { /// once; explicit delete frequency comes from deletion evidence. #[cfg(test)] pub(super) fn estimate_incremental_summary_maintenance( - root: &SummaryNode, + root: &PostASAPNode, guarantee: &SummaryMaintenanceLifecycleGuarantee, inputs: StreamingSummaryInputs, cpu: SummaryOperationCpuEvidence, @@ -979,7 +979,7 @@ pub(super) fn estimate_incremental_summary_maintenance( } #[cfg(test)] pub(super) fn estimate_incremental_summary_maintenance_with_join( - root: &SummaryNode, + root: &PostASAPNode, guarantee: &SummaryMaintenanceLifecycleGuarantee, inputs: StreamingSummaryInputs, cpu: SummaryOperationCpuEvidence, @@ -1236,13 +1236,13 @@ fn required_cpu_when( } #[cfg(test)] -fn count_operations(root: &SummaryNode) -> Result { +fn count_operations(root: &PostASAPNode) -> Result { fn visit( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, + node: &PostASAPNode, + seen: &mut HashSet<*const PostASAPNode>, counts: &mut SummaryOperationCounts, ) -> Result<(), AnalyticalCostError> { - if !seen.insert(node as *const SummaryNode) { + if !seen.insert(node as *const PostASAPNode) { return Ok(()); } match &node.expr { diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs index ce47f7ae..8b3225c4 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs @@ -235,29 +235,29 @@ pub struct StreamingRetainedQueryEvidence { /// structurally equal node is not silently treated as the same deployment. #[derive(Debug, Clone, Default)] pub struct StreamingNodeEvidence { - pub(super) aggregations: HashMap<*const SummaryNode, StreamingAggregateEvidence>, - pub(super) joins: HashMap<*const SummaryNode, SummaryJoinEvidence>, - pub(super) operations: HashMap<*const SummaryNode, StreamingSummaryOperatorEvidence>, - pub(super) operation_state_owners: HashMap<*const SummaryNode, *const SummaryNode>, - pub(super) retained_queries: HashMap<*const SummaryNode, StreamingRetainedQueryEvidence>, + pub(super) aggregations: HashMap<*const PostASAPNode, StreamingAggregateEvidence>, + pub(super) joins: HashMap<*const PostASAPNode, SummaryJoinEvidence>, + pub(super) operations: HashMap<*const PostASAPNode, StreamingSummaryOperatorEvidence>, + pub(super) operation_state_owners: HashMap<*const PostASAPNode, *const PostASAPNode>, + pub(super) retained_queries: HashMap<*const PostASAPNode, StreamingRetainedQueryEvidence>, } impl StreamingNodeEvidence { pub fn insert_aggregation( &mut self, - node: &Rc, + node: &Rc, evidence: StreamingAggregateEvidence, ) { self.aggregations.insert(Rc::as_ptr(node), evidence); } - pub fn insert_join(&mut self, node: &Rc, evidence: SummaryJoinEvidence) { + pub fn insert_join(&mut self, node: &Rc, evidence: SummaryJoinEvidence) { self.joins.insert(Rc::as_ptr(node), evidence); } pub fn insert_operation( &mut self, - node: &Rc, + node: &Rc, evidence: StreamingSummaryOperatorEvidence, ) { self.operations.insert(Rc::as_ptr(node), evidence); @@ -267,8 +267,8 @@ impl StreamingNodeEvidence { /// aggregation deployment whose active interval it follows. pub fn insert_state_operation( &mut self, - node: &Rc, - state: &Rc, + node: &Rc, + state: &Rc, evidence: StreamingSummaryOperatorEvidence, ) { self.operations.insert(Rc::as_ptr(node), evidence); @@ -278,19 +278,19 @@ impl StreamingNodeEvidence { pub fn insert_retained_query( &mut self, - node: &Rc, + node: &Rc, evidence: StreamingRetainedQueryEvidence, ) { self.retained_queries.insert(Rc::as_ptr(node), evidence); } - pub(super) fn aggregation(&self, node: &SummaryNode) -> Option { + pub(super) fn aggregation(&self, node: &PostASAPNode) -> Option { self.aggregations.get(&(node as *const _)).cloned() } } pub(super) fn summary_operation_evidence<'a>( - node: &SummaryNode, + node: &PostASAPNode, evidence: &'a StreamingNodeEvidence, ) -> Result<&'a StreamingSummaryOperatorEvidence, AnalyticalCostError> { let operation = evidence diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs index 23dd7e52..47f2ef96 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs @@ -8,12 +8,12 @@ use std::collections::{HashMap, HashSet}; use std::rc::Rc; use asap_types::post_asap::{ - BoundExpr, ErrorMetric, ExactKind, GuaranteeSource, ProbabilityExpr, ResultGuarantee, - SketchAlgorithm, SummaryExpr, SummaryFamilyType, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryNode, SummaryWindowFramework, + BoundExpr, ErrorMetric, ExactKind, GuaranteeSource, PostASAPNode, ProbabilityExpr, + ResultGuarantee, SketchAlgorithm, SummaryExpr, SummaryFamilyType, SummaryMaintenanceLifecycle, + SummaryMaintenanceLifecycleGuarantee, SummaryWindowFramework, }; use asap_types::pre_asap::{ - agg_intent::AggIntent, CompareOpKind, InfoMatcher, Predicate, QueryExpr, Source, + agg_intent::AggIntent, CompareOpKind, InfoMatcher, PreASAPNode, Predicate, Source, }; use asap_types::types::AccuracyTarget; use asap_types::workload::{DataArrival, DataWorkload, QueryRecurrence, RepeatedDemand}; diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs index e339994a..0d98be6f 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs @@ -7,7 +7,7 @@ pub struct SummaryMaintenanceCostModel { pub node_evidence: StreamingNodeEvidence, pub calibration: ResourceCalibration, pub capabilities: SummaryMaintenanceCapabilities, - target_comparisons: HashMap<*const QueryExpr, StreamingTargetComparison>, + target_comparisons: HashMap<*const PreASAPNode, StreamingTargetComparison>, candidate_comparisons: HashMap, physical_plan_alternatives: HashMap>, @@ -15,17 +15,17 @@ pub struct SummaryMaintenanceCostModel { HashMap>, } -type CandidateComparisonKey = (*const QueryExpr, *const SummaryNode); +type CandidateComparisonKey = (*const PreASAPNode, *const PostASAPNode); #[derive(Debug, Clone)] struct BoundCandidateIdentity { - _target: Rc, - _root: Rc, + _target: Rc, + _root: Rc, } #[derive(Debug, Clone)] struct StreamingTargetComparison { - _target: Rc, + _target: Rc, scope: ComparisonScope, raw: StreamingRawInputEvidence, } @@ -60,10 +60,10 @@ fn info_source(selector: &[InfoMatcher]) -> Result } pub(super) fn query_source_selections( - query: &QueryExpr, + query: &PreASAPNode, out: &mut Vec, ) -> Result<(), AnalyticalCostError> { - use QueryExpr::*; + use PreASAPNode::*; match query { Scan { source, predicates, .. @@ -121,7 +121,7 @@ pub(super) fn query_source_selections( } fn validate_query_scope( - target: &QueryExpr, + target: &PreASAPNode, scope: &ComparisonScope, ) -> Result<(), AnalyticalCostError> { let mut actual = Vec::new(); @@ -427,8 +427,8 @@ impl SummaryMaintenanceCostModel { /// rejected rather than silently replacing the canonical context. pub fn bind_candidate_comparison( &mut self, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, scope: ComparisonScope, raw: StreamingRawInputEvidence, ) -> Result<(), AnalyticalCostError> { @@ -473,8 +473,8 @@ impl SummaryMaintenanceCostModel { /// candidate. Duplicate or empty provider identities are rejected. pub fn bind_physical_plan_alternative( &mut self, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, alternative: StreamingPhysicalPlanAlternative, ) -> Result<(), AnalyticalCostError> { let key = (Rc::as_ptr(target), Rc::as_ptr(root)); @@ -511,8 +511,8 @@ impl SummaryMaintenanceCostModel { /// evidence keep the implementations distinct during ranking. pub fn bind_window_framework_candidate( &mut self, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, candidate: StreamingWindowFrameworkCandidate, ) -> Result<(), AnalyticalCostError> { let key = (Rc::as_ptr(target), Rc::as_ptr(root)); @@ -564,8 +564,8 @@ impl SummaryMaintenanceCostModel { fn comparison_context( &self, - root: &SummaryNode, - target: Option<&QueryExpr>, + root: &PostASAPNode, + target: Option<&PreASAPNode>, horizon: Option, expected_reads: Option, ) -> Option<(CandidateComparisonKey, &StreamingTargetComparison)> { @@ -599,7 +599,7 @@ impl SummaryMaintenanceCostModel { fn complete_cost_with_evidence( &self, - root: &SummaryNode, + root: &PostASAPNode, deployments: &[CostedSummaryDeployment<'_>], comparison: &StreamingTargetComparison, evidence: &StreamingNodeEvidence, @@ -618,7 +618,7 @@ impl SummaryMaintenanceCostModel { ) } - fn canonical_inputs(&self, summary: &SummaryNode) -> Option { + fn canonical_inputs(&self, summary: &PostASAPNode) -> Option { let evidence = self.node_evidence.aggregation(summary)?; evidence.inputs.validate().ok()?; Some(evidence) @@ -630,7 +630,7 @@ impl SummaryMaintenanceCostModel { fn lifecycle_inputs( &self, - summary: &SummaryNode, + summary: &PostASAPNode, horizon: Option, ) -> Option { let evidence = self.canonical_inputs(summary)?; @@ -699,14 +699,14 @@ impl CostModel for SummaryMaintenanceCostModel { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &PostASAPNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs::default() } fn summary_maintenance_lifecycle_cost_inputs_for_horizon( &self, - summary: &SummaryNode, + summary: &PostASAPNode, horizon: Option, ) -> SummaryMaintenanceLifecycleCostInputs { self.lifecycle_inputs(summary, horizon).unwrap_or_default() @@ -714,15 +714,15 @@ impl CostModel for SummaryMaintenanceCostModel { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &PostASAPNode, ) -> SummaryMaintenanceCapabilities { self.capabilities } fn complete_summary_candidate_cost( &self, - root: &SummaryNode, - target: Option<&QueryExpr>, + root: &PostASAPNode, + target: Option<&PreASAPNode>, deployments: &[CostedSummaryDeployment<'_>], horizon: Option, expected_reads: Option, @@ -741,8 +741,8 @@ impl CostModel for SummaryMaintenanceCostModel { fn complete_summary_candidate_estimate( &self, - root: &SummaryNode, - target: Option<&QueryExpr>, + root: &PostASAPNode, + target: Option<&PreASAPNode>, deployments: &[CostedSummaryDeployment<'_>], horizon: Option, expected_reads: Option, @@ -846,14 +846,14 @@ impl CostModel for SummaryMaintenanceCostModel { true } - fn raw_query_recompute_cost(&self, target: &QueryExpr) -> Option { + fn raw_query_recompute_cost(&self, target: &PreASAPNode) -> Option { let _ = target; None } fn raw_query_recompute_total_cost( &self, - target: &QueryExpr, + target: &PreASAPNode, expected_reads: f64, ) -> Option { let target_ptr = target as *const _; @@ -885,7 +885,7 @@ mod tests { SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, SummarySchema, }; use asap_types::pre_asap::{ - agg_intent::AggIntent, Column, ColumnRef, DataType, QueryExpr, Reduction, Schema, Source, + agg_intent::AggIntent, Column, ColumnRef, DataType, PreASAPNode, Reduction, Schema, Source, }; use asap_types::workload::{ DataWorkload, Evidence, EvidenceSource, Predictability, Query, QueryLanguage, @@ -902,7 +902,7 @@ mod tests { }; fn estimate_test( - root: &SummaryNode, + root: &PostASAPNode, guarantee: &SummaryMaintenanceLifecycleGuarantee, inputs: StreamingSummaryInputs, cpu: SummaryOperationCpuEvidence, @@ -911,7 +911,7 @@ mod tests { } fn estimate_join_test( - root: &SummaryNode, + root: &PostASAPNode, guarantee: &SummaryMaintenanceLifecycleGuarantee, inputs: StreamingSummaryInputs, cpu: SummaryOperationCpuEvidence, @@ -1560,7 +1560,7 @@ mod tests { op: CompareOpKind::Eq, value: "api".into(), }]; - let info_target = QueryExpr::PromqlInfoEnrich { + let info_target = PreASAPNode::PromqlInfoEnrich { selector: selector.clone(), child: target, }; @@ -2516,7 +2516,7 @@ mod tests { unreachable!(); }; let child = Rc::clone(child); - let nested = Rc::new(SummaryNode { + let nested = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child, family: SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count), @@ -2855,7 +2855,7 @@ mod tests { assert_eq!(estimate.peak_memory_bytes(), 64); // 4 persistent states + join memory. } - fn summary_with_operations(merge: bool, subtract: bool, delete: bool) -> Rc { + fn summary_with_operations(merge: bool, subtract: bool, delete: bool) -> Rc { let state_type = SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count); let schema = SummarySchema { fields: vec![SummaryField { @@ -2865,8 +2865,8 @@ mod tests { }], time_index: None, }; - let leaf = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(QueryExpr::Scan { + let leaf = Rc::new(PostASAPNode { + expr: SummaryExpr::KeepPreAsap(Rc::new(PreASAPNode::Scan { source: Source::TimeSeries { metric: "metrics".into(), }, @@ -2883,7 +2883,7 @@ mod tests { schema: schema.clone(), guarantee: None, }); - let agg = Rc::new(SummaryNode { + let agg = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child: leaf, family: state_type, @@ -2896,7 +2896,7 @@ mod tests { }); let mut root = Rc::clone(&agg); if merge { - root = Rc::new(SummaryNode { + root = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryMerge { timing: asap_types::post_asap::ExecutionTiming::IngestionTime, children: vec![Rc::clone(&agg), Rc::clone(&agg)], @@ -2906,7 +2906,7 @@ mod tests { }); } if subtract { - root = Rc::new(SummaryNode { + root = Rc::new(PostASAPNode { expr: SummaryExpr::SummarySubtract { left: Rc::clone(&root), right: Rc::clone(&agg), @@ -2916,7 +2916,7 @@ mod tests { }); } if delete { - root = Rc::new(SummaryNode { + root = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryDelete { summary_input: root, key: ColumnRef::Wildcard, @@ -2925,7 +2925,7 @@ mod tests { guarantee: None, }); } - Rc::new(SummaryNode { + Rc::new(PostASAPNode { expr: SummaryExpr::SummaryEstimate { summary_input: root, query: asap_types::post_asap::SketchQuery::PointCount { @@ -2938,7 +2938,7 @@ mod tests { }) } - fn summary_join() -> Rc { + fn summary_join() -> Rc { let left = summary_with_operations(false, false, false); let right = summary_with_operations(false, false, false); let SummaryExpr::SummaryEstimate { @@ -2956,7 +2956,7 @@ mod tests { unreachable!() }; let schema = left.schema.clone(); - let join = Rc::new(SummaryNode { + let join = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryJoin { outer: Rc::clone(left), inner: Rc::clone(right), @@ -2966,7 +2966,7 @@ mod tests { schema: schema.clone(), guarantee: None, }); - Rc::new(SummaryNode { + Rc::new(PostASAPNode { expr: SummaryExpr::SummaryEstimate { summary_input: join, query: asap_types::post_asap::SketchQuery::PointCount { @@ -2979,9 +2979,9 @@ mod tests { }) } - fn summary_binary() -> Rc { + fn summary_binary() -> Rc { let operand = summary_with_operations(false, false, false); - Rc::new(SummaryNode { + Rc::new(PostASAPNode { expr: SummaryExpr::BinaryOp { timing: asap_types::post_asap::ExecutionTiming::QueryTime, lhs: Rc::clone(&operand), @@ -3038,8 +3038,8 @@ mod tests { assert!(plan.summary_total_cost.is_some()); } - fn streaming_sum_query() -> Rc { - let scan = Rc::new(QueryExpr::Scan { + fn streaming_sum_query() -> Rc { + let scan = Rc::new(PreASAPNode::Scan { source: Source::TimeSeries { metric: "metrics".into(), }, @@ -3053,7 +3053,7 @@ mod tests { vec![], ), }); - Rc::new(QueryExpr::Aggregate { + Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Sum { col: None }], output_names: vec![], @@ -3156,16 +3156,16 @@ mod tests { fn bind_comparison( model: &mut SummaryMaintenanceCostModel, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, ) { model .bind_candidate_comparison(target, root, streaming_scope(), streaming_raw()) .unwrap(); fn retained( model: &mut SummaryMaintenanceCostModel, - node: &Rc, - seen: &mut HashSet<*const SummaryNode>, + node: &Rc, + seen: &mut HashSet<*const PostASAPNode>, ) { if !seen.insert(Rc::as_ptr(node)) { return; @@ -3242,8 +3242,8 @@ mod tests { fn bind_aggregations( model: &mut SummaryMaintenanceCostModel, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, inputs: StreamingSummaryInputs, cpu: SummaryOperationCpuEvidence, ) { @@ -3279,8 +3279,8 @@ mod tests { } fn bind_ops( model: &mut SummaryMaintenanceCostModel, - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, + node: &PostASAPNode, + seen: &mut HashSet<*const PostASAPNode>, inputs: StreamingSummaryInputs, cpu: SummaryOperationCpuEvidence, ) { @@ -3380,9 +3380,9 @@ mod tests { .insert(node as *const _, operation); if let SummaryExpr::SummaryDelete { summary_input, .. } = &node.expr { fn owning_aggs( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - owners: &mut Vec<*const SummaryNode>, + node: &PostASAPNode, + seen: &mut HashSet<*const PostASAPNode>, + owners: &mut Vec<*const PostASAPNode>, ) { if !seen.insert(node as *const _) { return; diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs index eff12cf3..354a08cc 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs @@ -3,7 +3,7 @@ use super::*; /// One per-state window choice within a complete Planner candidate. #[derive(Debug, Clone)] pub struct StreamingWindowFrameworkAssignment { - pub summary: Rc, + pub summary: Rc, /// `None` explicitly means that this state is not window-organized. pub framework: Option, } @@ -29,11 +29,11 @@ pub struct StreamingWindowFrameworkCandidate { pub node_evidence: StreamingNodeEvidence, } -pub(super) fn summary_aggregation_identities(root: &SummaryNode) -> HashSet<*const SummaryNode> { +pub(super) fn summary_aggregation_identities(root: &PostASAPNode) -> HashSet<*const PostASAPNode> { fn visit( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - out: &mut HashSet<*const SummaryNode>, + node: &PostASAPNode, + seen: &mut HashSet<*const PostASAPNode>, + out: &mut HashSet<*const PostASAPNode>, ) { if !seen.insert(node as *const _) { return; diff --git a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs index 129d06ed..c92fb886 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs @@ -12,12 +12,12 @@ use serde::Serialize; use asap_types::dag_export::{self, SummaryDagGraph}; use asap_types::post_asap::{ - PostAsapNodeId, ResultGuarantee, SummaryExpr, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryNode, SummaryWindowFramework, + PostASAPNode, PostASAPNodeId, ResultGuarantee, SummaryExpr, SummaryMaintenanceLifecycle, + SummaryMaintenanceLifecycleGuarantee, SummaryWindowFramework, }; use crate::summary_maintenance_lifecycle::{ - SummaryMaintenanceLifecyclePlan, SummaryMaintenanceLifecycleRejection, + LifecyclePostASAPDAG, SummaryMaintenanceLifecycleRejection, }; #[derive(Debug, Clone, Serialize)] @@ -42,7 +42,7 @@ pub struct SummaryMaintenanceDagExport { #[derive(Debug, Clone, Serialize)] pub struct SummaryMaintenanceDeploymentExport { - pub post_asap_node_id: PostAsapNodeId, + pub post_asap_node_id: PostASAPNodeId, #[serde(skip_serializing_if = "Option::is_none")] pub selected_window_framework: Option, #[serde(skip_serializing_if = "Option::is_none")] @@ -61,9 +61,7 @@ pub struct SummaryMaintenanceLifecycleAlternativeExport { pub type SummaryMaintenanceLifecycleGuaranteeExport = SummaryMaintenanceLifecycleGuarantee; -pub fn export_summary_maintenance_plan( - plan: &SummaryMaintenanceLifecyclePlan, -) -> SummaryMaintenanceDagExport { +pub fn export_summary_maintenance_plan(plan: &LifecyclePostASAPDAG) -> SummaryMaintenanceDagExport { let deployments: Vec<_> = plan .deployments .iter() @@ -121,9 +119,9 @@ pub fn export_summary_maintenance_plan( /// This makes the decision visible to graph consumers without asking them to /// reconstruct pointer identity from graph position. fn annotate_lifecycle_deployments( - node: &SummaryNode, + node: &PostASAPNode, graph: &mut SummaryDagGraph, - deployments: &HashMap<*const SummaryNode, &SummaryMaintenanceDeploymentExport>, + deployments: &HashMap<*const PostASAPNode, &SummaryMaintenanceDeploymentExport>, next_node_id: &mut usize, ) { if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) { @@ -132,14 +130,14 @@ fn annotate_lifecycle_deployments( } } let graph_node = &mut graph.nodes[*next_node_id]; - if let Some(deployment) = deployments.get(&(node as *const SummaryNode)) { + if let Some(deployment) = deployments.get(&(node as *const PostASAPNode)) { graph_node.detail["summary_maintenance"] = serde_json::to_value(deployment).expect("lifecycle export is serializable"); } *next_node_id += 1; } -fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { +fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { match expr { SummaryExpr::KeepPreAsap(_) => vec![], SummaryExpr::BinaryOp { lhs, rhs, .. } => vec![lhs, rhs], diff --git a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs index 7cafcf46..f1c2664d 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs @@ -11,7 +11,7 @@ //! //! This module enumerates and costs `Ephemeral`, `Prepared`, `Shared`, and //! `ContinuouslyMaintained` alternatives for every unique `SummaryAgg` in a -//! materialized plan, and for every maintained population (`MaintainPopulation`) +//! assembled Post-ASAP DAG, and for every maintained population (`MaintainPopulation`) //! that is not an input of a `SummaryAgg`. [`SummaryMaintenanceMode`] is an orthogonal detail of //! the selected deployment: state is either built directly or updated //! incrementally. Unknown evidence stays unknown and therefore cannot make a @@ -21,13 +21,13 @@ use std::collections::{HashMap, HashSet}; use std::rc::Rc; use asap_types::post_asap::{ - compile_post_asap_dag_with_node_ids, EvaluationSchedule, ExecutionDataStateError, - ExecutionTiming, OutputRepresentation, PostAsapDag, PostAsapDagValidationError, PostAsapNodeId, + EvaluationSchedule, ExecutionDataStateError, ExecutionTiming, LogicalPostASAPDAGTransport, + LogicalPostASAPDAGValidationError, OutputRepresentation, PostASAPNode, PostASAPNodeId, ResultGuarantee, SummaryExpr, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, SummaryNode, - SummaryWindowFramework, ValueOperation, + SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, SummaryWindowFramework, + ValueOperation, }; -use asap_types::pre_asap::QueryExpr; +use asap_types::pre_asap::PreASAPNode; use asap_types::types::AccuracyTarget; use asap_types::workload::{ DataArrival, DataWorkload, Predictability, QueryRecurrence, QueryWorkload, RepeatedDemand, @@ -43,7 +43,8 @@ use crate::recurrence::{ CostRate, EvaluationRate, Horizon, RecurrenceError, RecurrenceProfile, UpdateRate, }; use crate::replacement::{ - CandidateCostOverrides, CandidatePostASAPDAGs, GlobalSelection, RealizationError, Replacement, + CandidateCostOverrides, CandidateLogicalPostASAPDAGs, GlobalSelection, RealizationError, + Replacement, }; /// Summary-maintenance lifecycle shapes supported by the target runtime. @@ -166,11 +167,11 @@ pub struct SummaryMaintenanceDeployment { /// Identity of this summary in the exported post-ASAP semantic DAG. /// It is scoped to one plan version and is not a summary definition or /// summary instance identity. - pub post_asap_node_id: PostAsapNodeId, + pub post_asap_node_id: PostASAPNodeId, /// The unique materialized `SummaryAgg`, or maintained population /// (`MaintainPopulation`) not consumed by a `SummaryAgg`, represented by /// this deployment. Cost-model lifecycle hooks receive this node. - pub summary: Rc, + pub summary: Rc, /// Lifecycle, evaluation, and representation commitment selected for this /// state, or `None` when no alternative is selectable. pub summary_maintenance_lifecycle_guarantee: Option, @@ -181,12 +182,15 @@ pub struct SummaryMaintenanceDeployment { pub alternatives: Vec, } -/// Workload-aware lifecycle and window-framework decisions for every unique -/// summary state reachable from one materialized post-ASAP root. +/// A Post-ASAP DAG annotated with one lifecycle assignment: `root` plus, for +/// each unique summary state reachable from it, the lifecycle (including +/// retention) and window framework in `deployments`. Per-node timing is derived +/// from these annotations by [`Self::execution_assignment`], the only source of +/// timing, so no separate timed graph is kept beside this DAG. #[derive(Debug, Clone)] -pub struct SummaryMaintenanceLifecyclePlan { - /// Root of the materialized post-ASAP DAG being deployed. - pub root: Rc, +pub struct LifecyclePostASAPDAG { + /// Root of the annotated Post-ASAP DAG. + pub root: Rc, /// One entry per unique reachable `SummaryAgg`, then per unique /// maintained population outside any `SummaryAgg`'s inputs; shared `Rc` /// nodes appear only once. @@ -216,23 +220,25 @@ pub struct SummaryMaintenanceLifecyclePlan { pub raw_recompute_total_cost: Option, } -/// Why a lifecycle plan cannot assign execution timing to its DAG. +/// Why a lifecycle DAG cannot assign execution timing to its nodes. #[derive(Debug, thiserror::Error, PartialEq)] pub enum SummaryMaintenanceTimingError { #[error(transparent)] - InvalidPostAsapDag(#[from] ExecutionDataStateError), + InvalidLogicalPostASAPDAG(#[from] ExecutionDataStateError), #[error("summary {0:?} has no selected lifecycle")] - UnselectedLifecycle(PostAsapNodeId), + UnselectedLifecycle(PostASAPNodeId), /// A maintained population outside any `SummaryAgg`'s inputs has no /// deployment, so its timing would be guessed. Enumeration always emits - /// one; this arises only for a plan whose root or deployments were edited. + /// one; this arises only for a DAG whose root or deployments were edited. #[error("node {0:?} maintains state that has no summary-maintenance lifecycle")] - UnplannedMaintainedState(PostAsapNodeId), + UnplannedMaintainedState(PostASAPNodeId), + #[error("timing index belongs to a different logical graph")] + GraphMismatch, #[error(transparent)] - InvalidPhases(#[from] PostAsapDagValidationError), + InvalidPhases(#[from] LogicalPostASAPDAGValidationError), } -impl SummaryMaintenanceLifecyclePlan { +impl LifecyclePostASAPDAG { /// The post-ASAP DAG of [`Self::root`] with every node's timing derived /// from the selected lifecycles, so physical compilation places it. /// @@ -244,11 +250,29 @@ impl SummaryMaintenanceLifecyclePlan { /// maintained populations as to `SummaryAgg` states; a population feeding /// a `SummaryAgg` is one of its inputs. Timings already on the root are /// ignored. - pub fn execution_timed_dag(&self) -> Result { - let compiled = compile_post_asap_dag_with_node_ids(&self.root)?; - let dag = compiled.dag; + pub fn export_timed_dag( + &self, + ) -> Result { + let index = Rc::new(asap_types::post_asap::index_post_asap_dag(&self.root)?); + Ok(self.execution_assignment(index)?.to_transport()) + } + + /// Derive lifecycle timing over a shared index; no logical operators are + /// cloned or rewritten. Window/retention commitments remain on this plan. + pub fn execution_assignment( + &self, + index: Rc, + ) -> Result + { + if !index + .node_ids + .summary_node(index.root_id) + .is_some_and(|root| Rc::ptr_eq(root, &self.root)) + { + return Err(SummaryMaintenanceTimingError::GraphMismatch); + } for population in &standalone_populations(&self.root) { - let id = compiled + let id = index .node_ids .node_id(population) .expect("collected population belongs to the compiled DAG"); @@ -276,15 +300,16 @@ impl SummaryMaintenanceLifecyclePlan { while let Some(id) = pending.pop() { if ingestion.insert(id) { pending.extend( - dag.edges + index + .edges() .iter() .filter(|edge| edge.consumer == id) .map(|edge| edge.producer), ); } } - let phases = dag - .nodes + let phases = index + .node_views() .iter() .map(|node| { let timing = if ingestion.contains(&node.id) { @@ -295,7 +320,9 @@ impl SummaryMaintenanceLifecyclePlan { (node.id, timing) }) .collect(); - Ok(dag.with_execution_phases(&phases)?) + Ok(asap_types::post_asap::LogicalPostASAPDAGAssignment::new( + index, phases, + )?) } } @@ -342,7 +369,7 @@ impl<'a> WorkloadDemand<'a> { } #[derive(Debug, thiserror::Error)] -pub enum SummaryMaintenanceLifecyclePlanError { +pub enum LifecyclePostASAPDAGError { #[error(transparent)] InvalidWorkload(#[from] WorkloadError), #[error("optimization horizon must be finite and strictly positive")] @@ -354,7 +381,7 @@ pub enum SummaryMaintenanceLifecyclePlanError { #[error("workload entry index {index} appears more than once in one demand binding")] DuplicateWorkloadEntry { index: usize }, #[error(transparent)] - InvalidPostAsapDag(#[from] ExecutionDataStateError), + InvalidLogicalPostASAPDAG(#[from] ExecutionDataStateError), } #[derive(Debug, thiserror::Error)] @@ -362,7 +389,7 @@ pub enum SummaryMaintenanceLifecycleAssemblyError { #[error(transparent)] AssembleDag(#[from] RealizationError), #[error(transparent)] - SummaryMaintenance(#[from] SummaryMaintenanceLifecyclePlanError), + SummaryMaintenance(#[from] LifecyclePostASAPDAGError), } /// Failure while deriving workload-aware candidate costs before global @@ -372,40 +399,45 @@ pub enum SummaryMaintenanceLifecycleSelectionError { #[error(transparent)] Recurrence(#[from] RecurrenceError), #[error(transparent)] - SummaryMaintenance(#[from] SummaryMaintenanceLifecyclePlanError), + SummaryMaintenance(#[from] LifecyclePostASAPDAGError), } /// Every lifecycle alternative for each unique retained state of one fixed /// root, before any lifecycle is chosen. /// -/// Planner selection ([`plan_summary_maintenance_lifecycles`]) and a -/// deployment's explicit choice ([`Self::select`]) both finish from this value, -/// so they produce the same [`SummaryMaintenanceLifecyclePlan`] shape. -pub struct SummaryMaintenanceLifecycleCandidates<'a> { - /// Unselected plan: deployments carry alternatives but no guarantee or +/// Planner selection ([`plan_summary_maintenance_lifecycles`]) and binding an +/// explicit assignment ([`Self::select`]) both finish from this value, +/// so they produce the same [`LifecyclePostASAPDAG`] shape. +#[derive(Clone)] +pub(crate) struct SummaryMaintenanceLifecycleCandidates<'a> { + /// Unselected DAG: deployments carry alternatives but no guarantee or /// window framework. - plan: SummaryMaintenanceLifecyclePlan, + plan: LifecyclePostASAPDAG, components: Vec, arrival: DataArrival, required_accuracy: Vec, cost_model: &'a dyn CostModel, - comparison_target: Option<&'a QueryExpr>, + comparison_target: Option<&'a PreASAPNode>, } /// Why an explicit per-state lifecycle choice cannot be bound. #[derive(Debug, thiserror::Error, PartialEq)] pub enum SummaryMaintenanceLifecycleChoiceError { + #[error("lifecycle assignment expansion exceeds the requested limit {0}")] + ExpansionLimit(usize), + #[error("summary {0:?} has no lifecycle alternative")] + NoAlternatives(PostASAPNodeId), #[error("summary {0:?} is not a deployment of this root")] - UnknownSummary(PostAsapNodeId), + UnknownSummary(PostASAPNodeId), #[error("summary {0:?} is chosen more than once")] - DuplicateChoice(PostAsapNodeId), + DuplicateChoice(PostASAPNodeId), #[error("summary {0:?} has no chosen lifecycle")] - MissingChoice(PostAsapNodeId), + MissingChoice(PostASAPNodeId), #[error("chosen lifecycle is not an enumerated alternative of summary {0:?}")] - NotAnAlternative(PostAsapNodeId), + NotAnAlternative(PostASAPNodeId), #[error("chosen lifecycle of summary {post_asap_node_id:?} is rejected: {rejection:?}")] Rejected { - post_asap_node_id: PostAsapNodeId, + post_asap_node_id: PostASAPNodeId, rejection: Option, }, #[error("summary states on one maintenance path have different evaluation schedules")] @@ -414,9 +446,17 @@ pub enum SummaryMaintenanceLifecycleChoiceError { NoCompleteEstimate, } +/// One enumerated combination, including its rejection when it is infeasible. +/// A successful unpriced DAG still needs window/evidence binding before installation. +#[derive(Debug)] +pub struct LifecycleAssignmentCandidate { + pub choices: Vec<(PostASAPNodeId, SummaryMaintenanceLifecycle)>, + pub plan: Result, +} + impl SummaryMaintenanceLifecycleCandidates<'_> { /// One entry per unique retained state (see - /// [`SummaryMaintenanceLifecyclePlan::deployments`]), with every + /// [`LifecyclePostASAPDAG::deployments`]), with every /// alternative and its rejection; no lifecycle or window framework is /// selected. pub fn deployments(&self) -> &[SummaryMaintenanceDeployment] { @@ -447,7 +487,7 @@ impl SummaryMaintenanceLifecycleCandidates<'_> { fn finish( mut self, estimate: Option, - ) -> SummaryMaintenanceLifecyclePlan { + ) -> LifecyclePostASAPDAG { if let Some(estimate) = estimate { self.plan.summary_total_cost = Some(estimate.cost); self.plan.selected_window_implementation_id = estimate.physical_plan_id; @@ -458,7 +498,7 @@ impl SummaryMaintenanceLifecycleCandidates<'_> { /// Planner's choice: the cheapest complete combination of eligible /// alternatives. - fn select_cheapest(mut self) -> SummaryMaintenanceLifecyclePlan { + fn select_cheapest(mut self) -> LifecyclePostASAPDAG { let estimate = select_complete_lifecycle_combination( &self.plan.root, &mut self.plan.deployments, @@ -473,14 +513,84 @@ impl SummaryMaintenanceLifecycleCandidates<'_> { self.finish(estimate) } - /// Bind one caller-chosen lifecycle per summary state. Each choice must be + /// Bind one given lifecycle per summary state. Each choice must be /// an alternative Planner itself could select; the complete estimate is /// then obtained exactly as for Planner selection, so window framework and /// cost are the model's and unknown cost is never replaced by zero. pub fn select( + self, + choices: &[(PostASAPNodeId, SummaryMaintenanceLifecycle)], + ) -> Result { + self.bind_assignment(choices, true) + } + + /// Enumerate assignments without choosing a winner. Unknown costs remain + /// unknown; rejected combinations retain an error alongside their choices. + /// Expansion is lazy and refuses an insufficient budget before yielding. + #[cfg(test)] + pub fn assignments( + &self, + limit: usize, + ) -> Result< + impl Iterator + '_, + SummaryMaintenanceLifecycleChoiceError, + > { + let count = self.assignment_count(limit)?; + Ok((0..count).map(move |ordinal| self.assignment_at(ordinal))) + } + + pub(crate) fn assignment_count( + &self, + limit: usize, + ) -> Result { + if let Some(empty) = self + .plan + .deployments + .iter() + .find(|d| d.alternatives.is_empty()) + { + return Err(SummaryMaintenanceLifecycleChoiceError::NoAlternatives( + empty.post_asap_node_id, + )); + } + let count = self.plan.deployments.iter().try_fold(1usize, |n, d| { + n.checked_mul(d.alternatives.len()) + .filter(|n| *n <= limit) + .ok_or(SummaryMaintenanceLifecycleChoiceError::ExpansionLimit( + limit, + )) + })?; + if count > limit { + return Err(SummaryMaintenanceLifecycleChoiceError::ExpansionLimit( + limit, + )); + } + Ok(count) + } + + pub(crate) fn assignment_at(&self, mut ordinal: usize) -> LifecycleAssignmentCandidate { + let choices = self + .plan + .deployments + .iter() + .map(|deployment| { + let alternative = &deployment.alternatives[ordinal % deployment.alternatives.len()]; + ordinal /= deployment.alternatives.len(); + ( + deployment.post_asap_node_id, + alternative.summary_maintenance_lifecycle.clone(), + ) + }) + .collect::>(); + let plan = self.clone().bind_assignment(&choices, false); + LifecycleAssignmentCandidate { choices, plan } + } + + fn bind_assignment( mut self, - choices: &[(PostAsapNodeId, SummaryMaintenanceLifecycle)], - ) -> Result { + choices: &[(PostASAPNodeId, SummaryMaintenanceLifecycle)], + require_cost: bool, + ) -> Result { use SummaryMaintenanceLifecycleChoiceError as E; let deployments = &self.plan.deployments; let mut chosen: Vec> = @@ -499,7 +609,11 @@ impl SummaryMaintenanceLifecycleCandidates<'_> { .iter() .find(|alternative| alternative.summary_maintenance_lifecycle == *lifecycle) .ok_or(E::NotAnAlternative(*id))?; - if !context.eligible(alternative) { + if !(context.eligible(alternative) + || (!require_cost + && alternative.rejection + == Some(SummaryMaintenanceLifecycleRejection::MissingCostEvidence))) + { return Err(E::Rejected { post_asap_node_id: *id, rejection: alternative.rejection.clone(), @@ -507,6 +621,10 @@ impl SummaryMaintenanceLifecycleCandidates<'_> { } chosen[index] = Some(alternative); } + let all_costs_known = chosen + .iter() + .flatten() + .all(|alternative| alternative.total_cost.is_some()); let selected = chosen .into_iter() .enumerate() @@ -528,15 +646,31 @@ impl SummaryMaintenanceLifecycleCandidates<'_> { if !context.schedules_compatible(&selected) { return Err(E::IncompatibleEvaluationSchedules); } - let estimate = context - .estimate(deployments, &selected) - .ok_or(E::NoCompleteEstimate)?; + let estimate = if all_costs_known + || self + .cost_model + .complete_summary_candidate_estimate_covers_lifecycle_costs() + { + context.estimate(deployments, &selected) + } else { + None + }; + if require_cost && estimate.is_none() { + return Err(E::NoCompleteEstimate); + } let guarantees = selected .into_iter() .map(|(index, guarantee, _)| (index, guarantee)) .collect(); - apply_selection(&mut self.plan.deployments, guarantees, &estimate); - Ok(self.finish(Some(estimate))) + if let Some(estimate) = &estimate { + apply_selection(&mut self.plan.deployments, guarantees, estimate); + } else { + for (index, guarantee) in guarantees { + self.plan.deployments[index].summary_maintenance_lifecycle_guarantee = + Some(guarantee); + } + } + Ok(self.finish(estimate)) } } @@ -575,17 +709,17 @@ struct SummaryMaintenanceWorkloadFacts { requires_deletion: bool, } -/// Validate a materialized plan, enumerate lifecycle alternatives for each +/// Validate an assembled Post-ASAP DAG, enumerate lifecycle alternatives for each /// unique summary state, and select the cheapest legal alternative whose cost /// is fully known. pub fn plan_summary_maintenance_lifecycles( - root: Rc, + root: Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &dyn CostModel, -) -> Result { +) -> Result { Ok(enumerate_summary_maintenance_lifecycles( root, demand, @@ -597,18 +731,17 @@ pub fn plan_summary_maintenance_lifecycles( .select_cheapest()) } -/// Validate a materialized plan and enumerate lifecycle alternatives for each -/// unique summary state without choosing one. A deployment that prices the -/// alternatives itself binds its choice with -/// [`SummaryMaintenanceLifecycleCandidates::select`]. -pub fn enumerate_summary_maintenance_lifecycles<'a>( - root: Rc, +/// Validate a Post-ASAP DAG and enumerate lifecycle alternatives for each +/// unique summary state without choosing one. An explicit assignment is bound +/// with [`SummaryMaintenanceLifecycleCandidates::select`]. +pub(crate) fn enumerate_summary_maintenance_lifecycles<'a>( + root: Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &'a dyn CostModel, -) -> Result, SummaryMaintenanceLifecyclePlanError> { +) -> Result, LifecyclePostASAPDAGError> { enumerate_with_profile( root, demand, @@ -623,24 +756,24 @@ pub fn enumerate_summary_maintenance_lifecycles<'a>( /// Internal candidate-costing form. The workload binding supplies temporal /// eligibility and data-arrival facts; `profile` supplies effective uses after -/// DAG path multiplicity has been propagated by `CandidatePostASAPDAGs`. +/// DAG path multiplicity has been propagated by `CandidateLogicalPostASAPDAGs`. #[expect(clippy::too_many_arguments, reason = "internal bound planning context")] fn enumerate_with_profile<'a>( - root: Rc, + root: Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &'a dyn CostModel, profile: Option, - comparison_target: Option<&'a QueryExpr>, -) -> Result, SummaryMaintenanceLifecyclePlanError> { + comparison_target: Option<&'a PreASAPNode>, +) -> Result, LifecyclePostASAPDAGError> { demand.workload.validate()?; if let Some(data) = demand.data_workload { data.validate()?; } if horizon.is_some_and(|h| !h.0.is_finite() || h.0 <= 0.0) { - return Err(SummaryMaintenanceLifecyclePlanError::InvalidHorizon); + return Err(LifecyclePostASAPDAGError::InvalidHorizon); } let mut facts = workload_facts( demand.workload, @@ -672,7 +805,7 @@ fn enumerate_with_profile<'a>( StateKind::SummaryAgg, ); summaries.extend(standalone_populations(&root)); - let node_ids = compile_post_asap_dag_with_node_ids(&root)?.node_ids; + let node_ids = asap_types::post_asap::index_post_asap_dag(&root)?.node_ids; let components = summary_state_components(&summaries); let deployments: Vec = summaries .into_iter() @@ -697,7 +830,7 @@ fn enumerate_with_profile<'a>( .collect(); let selected_raw_recompute = matches!(root.expr, SummaryExpr::KeepPreAsap(_)); Ok(SummaryMaintenanceLifecycleCandidates { - plan: SummaryMaintenanceLifecyclePlan { + plan: LifecyclePostASAPDAG { root, deployments, horizon, @@ -724,7 +857,7 @@ fn enumerate_with_profile<'a>( /// attached, so shared `Rc` identity and exact-composition commitments remain /// the responsibility of `GlobalSelection`. pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( - space: &'a CandidatePostASAPDAGs, + space: &'a CandidateLogicalPostASAPDAGs, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, @@ -791,13 +924,13 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( /// summary maintenance decisions. This does not create or maintain runtime state. pub fn assemble_selected_dag_with_summary_maintenance_lifecycles( selection: &GlobalSelection<'_>, - target: &Rc, + target: &Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &dyn CostModel, -) -> Result, SummaryMaintenanceLifecycleAssemblyError> { +) -> Result, SummaryMaintenanceLifecycleAssemblyError> { selection .assemble_selected_dag(target)? .map(|root| { @@ -839,7 +972,7 @@ fn workload_facts( workload_entry_indices: &[usize], now_ms: u64, horizon: Option, -) -> Result { +) -> Result { let mut one_time_invocations = 0u64; let mut recurring_reads = 0.0; let mut recurring_known = true; @@ -853,19 +986,19 @@ fn workload_facts( let entries: Vec<_> = workload.entries().collect(); if workload_entry_indices.is_empty() { - return Err(SummaryMaintenanceLifecyclePlanError::EmptyWorkloadDemand); + return Err(LifecyclePostASAPDAGError::EmptyWorkloadDemand); } let mut seen_indices = HashSet::new(); for &index in workload_entry_indices { if !seen_indices.insert(index) { - return Err(SummaryMaintenanceLifecyclePlanError::DuplicateWorkloadEntry { index }); + return Err(LifecyclePostASAPDAGError::DuplicateWorkloadEntry { index }); } - let entry = entries.get(index).ok_or( - SummaryMaintenanceLifecyclePlanError::InvalidWorkloadEntry { + let entry = entries + .get(index) + .ok_or(LifecyclePostASAPDAGError::InvalidWorkloadEntry { index, entry_count: entries.len(), - }, - )?; + })?; required_accuracy.push(entry.requirements.accuracy.target()); requires_deletion |= entry.time_selection.lookback.is_some() && entry.time_selection.as_of.is_none() @@ -1274,9 +1407,9 @@ enum StateKind { /// Collect every unique node of `kind` reachable from `node`. fn collect_states( - node: &Rc, - seen: &mut HashSet<*const SummaryNode>, - output: &mut Vec>, + node: &Rc, + seen: &mut HashSet<*const PostASAPNode>, + output: &mut Vec>, kind: StateKind, ) { if !seen.insert(Rc::as_ptr(node)) { @@ -1333,7 +1466,7 @@ fn collect_states( /// Maintained populations that are not an input of any `SummaryAgg`. A /// population feeding summary state is on that state's maintenance path, so /// that state's lifecycle times it, even when a readout also reads it directly. -fn standalone_populations(root: &Rc) -> Vec> { +fn standalone_populations(root: &Rc) -> Vec> { let mut summaries = Vec::new(); collect_states( root, @@ -1381,7 +1514,7 @@ pub(crate) fn evaluation_schedule( /// Summary states composed on one maintenance path must be produced on the /// same schedule. Return a component id for each collected state. -fn summary_state_components(summaries: &[Rc]) -> Vec { +fn summary_state_components(summaries: &[Rc]) -> Vec { let indices: HashMap<_, _> = summaries .iter() .enumerate() @@ -1432,10 +1565,10 @@ fn summary_state_components(summaries: &[Rc]) -> Vec { /// Inputs shared by every complete lifecycle-combination evaluation of one /// root, whether Planner searches combinations or a caller supplies one. struct CompleteCostContext<'a> { - root: &'a SummaryNode, + root: &'a PostASAPNode, components: &'a [usize], cost_model: &'a dyn CostModel, - comparison_target: Option<&'a QueryExpr>, + comparison_target: Option<&'a PreASAPNode>, horizon: Option, expected_reads: Option, required_accuracy: &'a [AccuracyTarget], @@ -1524,12 +1657,12 @@ fn apply_selection( #[expect(clippy::too_many_arguments, reason = "complete combination context")] fn select_complete_lifecycle_combination( - root: &SummaryNode, + root: &PostASAPNode, deployments: &mut [SummaryMaintenanceDeployment], components: &[usize], arrival: DataArrival, cost_model: &dyn CostModel, - comparison_target: Option<&QueryExpr>, + comparison_target: Option<&PreASAPNode>, horizon: Option, expected_reads: Option, required_accuracy: &[AccuracyTarget], @@ -1666,11 +1799,13 @@ mod tests { } use super::*; use asap_types::post_asap::{ - ExactKind, ExactParams, GroupingStrategy, PostAsapOperatorPayload, ResultGuarantee, + ExactKind, ExactParams, GroupingStrategy, PostASAPOperatorPayload, ResultGuarantee, SketchAlgorithm, SummaryFamilyType, SummaryField, SummarySchema, }; use asap_types::pre_asap::AggIntent; - use asap_types::pre_asap::{Column, ColumnRef, DataType, QueryExpr, Reduction, Schema, Source}; + use asap_types::pre_asap::{ + Column, ColumnRef, DataType, PreASAPNode, Reduction, Schema, Source, + }; use asap_types::types::AccuracyTarget; use asap_types::workload::{ BatchEntry, DataWorkload, DurationMs, Evidence, EvidenceSource, Predictability, Query, @@ -1690,7 +1825,7 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &PostASAPNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(10.0)), @@ -1703,7 +1838,7 @@ mod tests { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &PostASAPNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -1726,19 +1861,19 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &PostASAPNode, ) -> SummaryMaintenanceLifecycleCostInputs { UnitCosts.summary_maintenance_lifecycle_cost_inputs(summary) } fn summary_maintenance_capabilities( &self, - summary: &SummaryNode, + summary: &PostASAPNode, ) -> SummaryMaintenanceCapabilities { UnitCosts.summary_maintenance_capabilities(summary) } - fn raw_query_recompute_cost(&self, _target: &QueryExpr) -> Option { + fn raw_query_recompute_cost(&self, _target: &PreASAPNode) -> Option { Some(Cost(1.0)) } } @@ -1756,14 +1891,14 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &PostASAPNode, ) -> SummaryMaintenanceLifecycleCostInputs { UnitCosts.summary_maintenance_lifecycle_cost_inputs(summary) } fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &PostASAPNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -1778,7 +1913,7 @@ mod tests { impl CostModel for SummaryMaintenancePrefersDdSketch { fn raw_query_recompute_total_cost( &self, - _target: &QueryExpr, + _target: &PreASAPNode, _expected_reads: f64, ) -> Option { Some(Cost(1_000.0)) @@ -1796,7 +1931,7 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &PostASAPNode, ) -> SummaryMaintenanceLifecycleCostInputs { let build = match sketch_algorithm(summary) { Some(SketchAlgorithm::Kll) => 100.0, @@ -1828,8 +1963,8 @@ mod tests { fn complete_summary_candidate_cost( &self, - _root: &SummaryNode, - _target: Option<&QueryExpr>, + _root: &PostASAPNode, + _target: Option<&PreASAPNode>, deployments: &[CostedSummaryDeployment<'_>], _horizon: Option, _expected_reads: Option, @@ -1861,7 +1996,7 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &PostASAPNode, ) -> SummaryMaintenanceLifecycleCostInputs { let is_leaf = matches!( summary.expr, @@ -1879,7 +2014,7 @@ mod tests { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &PostASAPNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -1889,7 +2024,7 @@ mod tests { } } - fn sketch_algorithm(node: &SummaryNode) -> Option { + fn sketch_algorithm(node: &PostASAPNode) -> Option { match &node.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_algorithm(summary_input), SummaryExpr::SummaryAgg { @@ -1900,12 +2035,12 @@ mod tests { } } - fn query_root() -> Rc { + fn query_root() -> Rc { query_root_for("m") } - fn query_root_for(metric: &str) -> Rc { - Rc::new(QueryExpr::Scan { + fn query_root_for(metric: &str) -> Rc { + Rc::new(PreASAPNode::Scan { source: Source::TimeSeries { metric: metric.into(), }, @@ -1921,8 +2056,8 @@ mod tests { }) } - fn sum_query() -> Rc { - Rc::new(QueryExpr::Aggregate { + fn sum_query() -> Rc { + Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Sum { col: None }], output_names: vec![], @@ -1931,8 +2066,8 @@ mod tests { }) } - fn quantile_query() -> Rc { - Rc::new(QueryExpr::Aggregate { + fn quantile_query() -> Rc { + Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Quantile { col: None, @@ -1945,8 +2080,8 @@ mod tests { }) } - fn summary() -> Rc { - let child = Rc::new(SummaryNode { + fn summary() -> Rc { + let child = Rc::new(PostASAPNode { expr: SummaryExpr::KeepPreAsap(query_root()), schema: SummarySchema { fields: vec![], @@ -1955,7 +2090,7 @@ mod tests { guarantee: Some(ResultGuarantee::exact("raw")), }); let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - Rc::new(SummaryNode { + Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child, family: family.clone(), @@ -1977,10 +2112,10 @@ mod tests { }) } - fn nested_summary() -> Rc { + fn nested_summary() -> Rc { let child = summary(); let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - Rc::new(SummaryNode { + Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child, family: family.clone(), @@ -2318,6 +2453,85 @@ mod tests { assert_eq!(plan.update_rate, None); } + /// Generation keeps unpriced legal assignments and rejected choices visible; + /// the budget is checked before iteration and every assignment shares its root. + #[test] + fn assignment_generation_preserves_unknown_costs_and_graph_identity() { + let root = summary(); + let workload = workload(vec![], vec![repeating()], continuous(1_000, 60_000)); + let candidates = enumerate_summary_maintenance_lifecycles( + root.clone(), + WorkloadDemand::new_without_data(&workload, &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &crate::cost_model::DefaultCostModel, + ) + .unwrap(); + assert!(candidates.assignments(0).is_err()); + let assignments = candidates.assignments(4096).unwrap().collect::>(); + assert_eq!( + assignments.len(), + candidates + .deployments() + .iter() + .map(|d| d.alternatives.len()) + .product::() + ); + let index = Rc::new(asap_types::post_asap::index_post_asap_dag(&root).unwrap()); + let mut legal = 0; + for candidate in assignments { + if let Ok(plan) = candidate.plan { + legal += 1; + assert!(Rc::ptr_eq(&plan.root, &root)); + assert!(plan.summary_total_cost.is_none()); + let timed = plan.execution_assignment(index.clone()).unwrap(); + assert!(Rc::ptr_eq(timed.index(), &index)); + assert_eq!(timed.to_transport(), plan.export_timed_dag().unwrap()); + } + } + assert!(legal > 0); + } + + /// A state with no lifecycle alternative is a timed-candidate diagnostic, + /// not an empty product that makes its logical candidate vanish. + #[test] + fn state_without_alternatives_is_a_timed_candidate_diagnostic() { + let root = summary(); + let workload = workload(vec![], vec![repeating()], continuous(1_000, 60_000)); + let mut candidates = enumerate_summary_maintenance_lifecycles( + root.clone(), + WorkloadDemand::new_without_data(&workload, &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &crate::cost_model::DefaultCostModel, + ) + .unwrap(); + let id = candidates.plan.deployments[0].post_asap_node_id; + candidates.plan.deployments[0].alternatives.clear(); + assert_eq!( + candidates.assignment_count(4096), + Err(SummaryMaintenanceLifecycleChoiceError::NoAlternatives(id)) + ); + let index = Rc::new(asap_types::post_asap::index_post_asap_dag(&root).unwrap()); + let timed = + crate::candidate_timing::collect_timing(7, [Ok((index, candidates))], vec![], 4096) + .unwrap(); + assert_eq!(timed.len(), 1); + let entries = timed.iter().collect::>(); + let [(metadata, Err(error))] = entries.as_slice() else { + panic!("expected one diagnostic entry"); + }; + assert_eq!((metadata.id, metadata.logical_candidate), (7, 0)); + assert!(matches!( + error.as_ref(), + crate::CandidateTimingError::Choice( + SummaryMaintenanceLifecycleChoiceError::NoAlternatives(missing) + ) if *missing == id + )); + } + #[test] fn unknown_costs_do_not_make_a_long_lived_lifecycle_win() { let plan = plan_summary_maintenance_lifecycles( @@ -2402,7 +2616,7 @@ mod tests { SummaryMaintenanceLifecycleCapabilities::ALL, &UnitCosts, ), - Err(SummaryMaintenanceLifecyclePlanError::EmptyWorkloadDemand) + Err(LifecyclePostASAPDAGError::EmptyWorkloadDemand) )); assert!(matches!( plan_summary_maintenance_lifecycles( @@ -2413,7 +2627,7 @@ mod tests { SummaryMaintenanceLifecycleCapabilities::ALL, &UnitCosts, ), - Err(SummaryMaintenanceLifecyclePlanError::DuplicateWorkloadEntry { index: 0 }) + Err(LifecyclePostASAPDAGError::DuplicateWorkloadEntry { index: 0 }) )); } @@ -2505,7 +2719,7 @@ mod tests { fn whole_candidate_cost_is_evaluated_before_selecting_a_lifecycle() { let root = summary(); let mut deployments = vec![SummaryMaintenanceDeployment { - post_asap_node_id: PostAsapNodeId(0), + post_asap_node_id: PostASAPNodeId(0), summary: Rc::clone(&root), summary_maintenance_lifecycle_guarantee: None, selected_window_framework: None, @@ -2568,7 +2782,7 @@ mod tests { ]; let mut deployments: Vec<_> = (0..13) .map(|summary_index| SummaryMaintenanceDeployment { - post_asap_node_id: PostAsapNodeId(summary_index as u32), + post_asap_node_id: PostASAPNodeId(summary_index as u32), summary: Rc::clone(&root), summary_maintenance_lifecycle_guarantee: None, selected_window_framework: None, @@ -2670,7 +2884,7 @@ mod tests { #[test] fn lifecycle_cost_counts_one_shared_summary_node_once() { let shared = summary(); - let root = Rc::new(SummaryNode { + let root = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryMerge { timing: asap_types::post_asap::ExecutionTiming::IngestionTime, children: vec![Rc::clone(&shared), Rc::clone(&shared)], @@ -2768,7 +2982,7 @@ mod tests { fn choose( candidates: &SummaryMaintenanceLifecycleCandidates<'_>, lifecycle: SummaryMaintenanceLifecycle, - ) -> Vec<(PostAsapNodeId, SummaryMaintenanceLifecycle)> { + ) -> Vec<(PostASAPNodeId, SummaryMaintenanceLifecycle)> { candidates .deployments() .iter() @@ -2878,7 +3092,7 @@ mod tests { use SummaryMaintenanceLifecycleChoiceError as E; let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); - let select = |model: &dyn CostModel, choice: &dyn Fn(PostAsapNodeId) -> Vec<_>| { + let select = |model: &dyn CostModel, choice: &dyn Fn(PostASAPNodeId) -> Vec<_>| { let candidates = continuous_candidates(&workload, &data, model); let id = candidates.deployments()[0].post_asap_node_id; (id, candidates.select(&choice(id)).unwrap_err()) @@ -2921,9 +3135,9 @@ mod tests { }); assert_eq!(error, E::DuplicateChoice(id)); let (_, error) = select(&UnitCosts, &|_| { - vec![(PostAsapNodeId(u32::MAX), continuous.clone())] + vec![(PostASAPNodeId(u32::MAX), continuous.clone())] }); - assert_eq!(error, E::UnknownSummary(PostAsapNodeId(u32::MAX))); + assert_eq!(error, E::UnknownSummary(PostASAPNodeId(u32::MAX))); } // Nested states on one maintenance path must share an evaluation schedule. @@ -2963,7 +3177,7 @@ mod tests { #[test] fn enumeration_lists_each_unique_summary_state_once() { let shared = summary(); - let root = Rc::new(SummaryNode { + let root = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryMerge { timing: asap_types::post_asap::ExecutionTiming::IngestionTime, children: vec![Rc::clone(&shared), Rc::clone(&shared), summary()], @@ -2994,8 +3208,8 @@ mod tests { .any(|deployment| Rc::ptr_eq(&deployment.summary, &shared))); } - fn readout(state: &Rc) -> Rc { - Rc::new(SummaryNode { + fn readout(state: &Rc) -> Rc { + Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: Rc::clone(state), operation: ValueOperation::FinalizeExactAccumulator, @@ -3028,12 +3242,12 @@ mod tests { /// Bind the lifecycle `choose` picks for every state of `root`, then /// derive the timed DAG. fn timed_dag( - root: Rc, + root: Rc, workload: &QueryWorkload, data: &DataWorkload, horizon: Option, choose: impl Fn(&SummaryMaintenanceDeployment) -> SummaryMaintenanceLifecycle, - ) -> PostAsapDag { + ) -> LogicalPostASAPDAGTransport { let candidates = enumerate_summary_maintenance_lifecycles( root, WorkloadDemand::new_with_data(workload, data, &[0]), @@ -3051,22 +3265,22 @@ mod tests { let dag = candidates .select(&choice) .unwrap() - .execution_timed_dag() + .export_timed_dag() .unwrap(); dag.validate().unwrap(); dag } /// Operator kinds in node-id order, each paired with its timing. - fn timings(dag: &PostAsapDag) -> Vec<(&'static str, ExecutionTiming)> { + fn timings(dag: &LogicalPostASAPDAGTransport) -> Vec<(&'static str, ExecutionTiming)> { dag.nodes .iter() .map(|node| { let kind = match node.payload { - PostAsapOperatorPayload::Fallback { .. } => "raw", - PostAsapOperatorPayload::SummaryAgg { .. } => "state", - PostAsapOperatorPayload::Value { .. } => "readout", - PostAsapOperatorPayload::Binary { .. } => "binary", + PostASAPOperatorPayload::Fallback { .. } => "raw", + PostASAPOperatorPayload::SummaryAgg { .. } => "state", + PostASAPOperatorPayload::Value { .. } => "readout", + PostASAPOperatorPayload::Binary { .. } => "binary", _ => "other", }; (kind, node.output_state.timing) @@ -3151,7 +3365,7 @@ mod tests { let state = summary(); let lhs = readout(&state); let rhs = Rc::new(lhs.as_ref().clone()); - let root = Rc::new(SummaryNode { + let root = Rc::new(PostASAPNode { expr: SummaryExpr::BinaryOp { lhs, rhs, @@ -3234,7 +3448,7 @@ mod tests { ) .unwrap(); assert_eq!( - plan.execution_timed_dag().unwrap_err(), + plan.export_timed_dag().unwrap_err(), SummaryMaintenanceTimingError::UnselectedLifecycle( plan.deployments[0].post_asap_node_id ) @@ -3248,14 +3462,11 @@ mod tests { &UnitCosts, ) .unwrap(); - assert_eq!( - timings(&raw.execution_timed_dag().unwrap()), - [("raw", QUERY)] - ); + assert_eq!(timings(&raw.export_timed_dag().unwrap()), [("raw", QUERY)]); } /// A strategy-built `sum(a)` over one maintained current-series population. - fn population_readout() -> Rc { + fn population_readout() -> Rc { let target = Rc::new(crate::test_support::lower_promql( "sum(a)", AccuracyTarget::Exact, @@ -3267,7 +3478,7 @@ mod tests { .unwrap() } - fn is_population(node: &SummaryNode) -> bool { + fn is_population(node: &PostASAPNode) -> bool { matches!( node.expr, SummaryExpr::ValueOperation { @@ -3277,12 +3488,14 @@ mod tests { ) } - fn population_timings(dag: &PostAsapDag) -> Vec<(&'static str, ExecutionTiming)> { + fn population_timings( + dag: &LogicalPostASAPDAGTransport, + ) -> Vec<(&'static str, ExecutionTiming)> { dag.nodes .iter() .zip(timings(dag)) .map(|(node, (kind, timing))| match node.payload { - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::MaintainPopulation { .. }, } => ("population", timing), _ => (kind, timing), @@ -3352,7 +3565,7 @@ mod tests { .all(|alternative| alternative.total_cost.is_none())); assert!(deployment.summary_maintenance_lifecycle_guarantee.is_none()); assert_eq!( - plan.execution_timed_dag().unwrap_err(), + plan.export_timed_dag().unwrap_err(), SummaryMaintenanceTimingError::UnselectedLifecycle(deployment.post_asap_node_id) ); } @@ -3404,7 +3617,7 @@ mod tests { Some(SummaryMaintenanceLifecycle::Shared { .. }) )); assert_eq!( - population_timings(&plan.execution_timed_dag().unwrap()), + population_timings(&plan.export_timed_dag().unwrap()), [("raw", INGEST), ("population", INGEST), ("readout", QUERY)] ); } @@ -3430,7 +3643,7 @@ mod tests { else { unreachable!() }; - let state = Rc::new(SummaryNode { + let state = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child: Rc::clone(population), family: family.clone(), @@ -3497,7 +3710,7 @@ mod tests { else { unreachable!() }; - let state = Rc::new(SummaryNode { + let state = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child: Rc::clone(population), family: family.clone(), @@ -3507,8 +3720,8 @@ mod tests { }, ..state.as_ref().clone() }); - let binary = |lhs: Rc, rhs: Rc| { - Rc::new(SummaryNode { + let binary = |lhs: Rc, rhs: Rc| { + Rc::new(PostASAPNode { schema: lhs.schema.clone(), expr: SummaryExpr::BinaryOp { lhs, @@ -3579,7 +3792,7 @@ mod tests { .unwrap(); let id = plan.deployments.remove(0).post_asap_node_id; assert_eq!( - plan.execution_timed_dag(), + plan.export_timed_dag(), Err(SummaryMaintenanceTimingError::UnplannedMaintainedState(id)) ); } diff --git a/crates/asap-aware-mapping/src/test_support.rs b/crates/asap-aware-mapping/src/test_support.rs index 612cf7c0..e74aa4e6 100644 --- a/crates/asap-aware-mapping/src/test_support.rs +++ b/crates/asap-aware-mapping/src/test_support.rs @@ -1,11 +1,11 @@ -use asap_types::pre_asap::QueryExpr; +use asap_types::pre_asap::PreASAPNode; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, }; -pub(crate) fn lower_promql(query: &str, accuracy: AccuracyTarget) -> QueryExpr { +pub(crate) fn lower_promql(query: &str, accuracy: AccuracyTarget) -> PreASAPNode { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, diff --git a/crates/asap-aware-mapping/src/topk_reuse.rs b/crates/asap-aware-mapping/src/topk_reuse.rs index 8329569d..f629faf9 100644 --- a/crates/asap-aware-mapping/src/topk_reuse.rs +++ b/crates/asap-aware-mapping/src/topk_reuse.rs @@ -7,7 +7,7 @@ use std::rc::Rc; -use asap_types::pre_asap::QueryExpr; +use asap_types::pre_asap::PreASAPNode; use crate::replacement::{ Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, @@ -15,18 +15,18 @@ use crate::replacement::{ /// Derives a smaller top-k result from a compatible larger top-k sibling. pub struct TopKLimitReuseStrategy { - limits: Vec>, + limits: Vec>, } impl TopKLimitReuseStrategy { - pub fn new(limits: &[Rc]) -> Self { + pub fn new(limits: &[Rc]) -> Self { Self { limits: limits.to_vec(), } } - fn larger_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { - let QueryExpr::Limit { + fn larger_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { + let PreASAPNode::Limit { n: target_n, offset: 0, child: target_child, @@ -42,7 +42,7 @@ impl TopKLimitReuseStrategy { if Rc::ptr_eq(candidate, target.root) { return false; } - let QueryExpr::Limit { + let PreASAPNode::Limit { n, offset: 0, child, @@ -57,7 +57,7 @@ impl TopKLimitReuseStrategy { // Prefer the smallest sufficient materialized top-k when several // larger siblings are available. sources.sort_by_key(|source| match source.as_ref() { - QueryExpr::Limit { n, .. } => *n, + PreASAPNode::Limit { n, .. } => *n, _ => unreachable!(), }); sources @@ -70,7 +70,7 @@ impl ReplacementStrategy for TopKLimitReuseStrategy { } fn replacements(&self, target: &TargetSubDAG<'_>) -> Vec { - let QueryExpr::Limit { + let PreASAPNode::Limit { n: target_n, offset: 0, .. @@ -83,12 +83,12 @@ impl ReplacementStrategy for TopKLimitReuseStrategy { .into_iter() .map(|source| { let source_n = match source.as_ref() { - QueryExpr::Limit { n, .. } => *n, + PreASAPNode::Limit { n, .. } => *n, _ => unreachable!(), }; ReplacementSubDAG { strategy: "TopKLimitReuseStrategy", - replacement: Replacement::Rewrite(Rc::new(QueryExpr::Limit { + replacement: Replacement::Rewrite(Rc::new(PreASAPNode::Limit { n: *target_n, offset: 0, child: Rc::clone(source), @@ -108,8 +108,8 @@ mod tests { use super::*; use asap_types::pre_asap::{Schema, Source}; - fn scan_named(metric: &str) -> Rc { - Rc::new(QueryExpr::Scan { + fn scan_named(metric: &str) -> Rc { + Rc::new(PreASAPNode::Scan { source: Source::TimeSeries { metric: metric.into(), }, @@ -121,12 +121,12 @@ mod tests { #[test] fn smaller_limit_reuses_larger_compatible_limit() { let child = scan_named("m"); - let small = Rc::new(QueryExpr::Limit { + let small = Rc::new(PreASAPNode::Limit { n: 5, offset: 0, child: Rc::clone(&child), }); - let large = Rc::new(QueryExpr::Limit { + let large = Rc::new(PreASAPNode::Limit { n: 10, offset: 0, child, @@ -137,7 +137,7 @@ mod tests { let Replacement::Rewrite(rewrite) = &replacements[0].replacement else { panic!() }; - let QueryExpr::Limit { n: 5, child, .. } = rewrite.as_ref() else { + let PreASAPNode::Limit { n: 5, child, .. } = rewrite.as_ref() else { panic!() }; assert!(Rc::ptr_eq(child, &large)); @@ -147,17 +147,17 @@ mod tests { fn offset_or_different_input_is_not_reused() { let a = scan_named("a"); let b = scan_named("b"); - let small = Rc::new(QueryExpr::Limit { + let small = Rc::new(PreASAPNode::Limit { n: 5, offset: 0, child: a, }); - let large = Rc::new(QueryExpr::Limit { + let large = Rc::new(PreASAPNode::Limit { n: 10, offset: 0, child: b, }); - let offset = Rc::new(QueryExpr::Limit { + let offset = Rc::new(PreASAPNode::Limit { n: 20, offset: 1, child: scan_named("a"), diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index 5f638465..25559f84 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -5,7 +5,7 @@ query time execution. The library requires neither backend engine, a server, a storage implementation, Arrow nor DataFusion. DataFusion informed the design; it is not the execution framework. -`plan::PhysicalDag` binds typed operator inputs to node IDs. Each execution starts +`plan::PhysicalExecution` binds typed operator inputs to node IDs. Each execution starts one producer per reachable node, shares output batches among its consumers, and bounds buffering. Dropping one consumer does not cancel other consumers. A `RunContext` carries query or ingestion scope, cancellation and byte accounting. @@ -24,7 +24,7 @@ use asap_physical_operators::{ expressions::Expression, operators::Operator, values::Value, - plan::PhysicalDag, + plan::PhysicalExecution, runtime::{Limits, RunContext, Scope}, }; use asap_physical_operators::planner::pre_asap::DataType; @@ -34,7 +34,7 @@ let source = Operator::scalar(Value::Int64(7), DataType::Int64)?; let negate = Operator::project(source.schema(), vec![ ("value".into(), Expression::Negate(Box::new(Expression::Column(0)))), ])?; -let mut plan = PhysicalDag::default(); +let mut plan = PhysicalExecution::default(); plan.add(0, vec![], source)?; plan.add(1, vec![0], negate)?; let run = RunContext::new( @@ -47,7 +47,7 @@ assert!(matches!(batch.rows()[0][0], Value::Int64(-7))); # Ok::<(), asap_physical_operators::dag::Error>(()) ``` -`physical_planner::compile` accepts a logical Post-ASAP DAG (`PostAsapDag`) and typed input contracts. +`physical_planner::compile` accepts a logical Post-ASAP DAG (`LogicalPostASAPDAGTransport`) and typed input contracts. The resulting candidate is instantiated with deployment readers after selection. It rejects unsupported operations and schema mismatches before starting a source. Implement `PhysicalOperator` for a deployment source, including asynchronous I/O; computation operators remain in @@ -91,7 +91,7 @@ state; operators own grouping. A source must declare `Boundedness::Bounded` to feed a blocking operator. The default for a custom raw source is `Unknown`; query or ingestion scope alone -does not promise that its cursor ends. `PhysicalDag::properties` validates these +does not promise that its cursor ends. `PhysicalExecution::properties` validates these requirements before any source starts and returns boundedness and emission mode for every reachable node. The memory connector declares finite input. Custom physical sources expose the same facts through `PhysicalOperator::properties`. @@ -104,12 +104,12 @@ There is no spill or partitioned parallel execution in this implementation. ## Physical compilation and deployment inputs -`physical_planner::compile` accepts a Planner `PostAsapDag`, typed -`InputContract`s and output roots. It returns a reusable `CompiledPhysicalDag` +`physical_planner::compile` accepts a Planner `LogicalPostASAPDAGTransport`, typed +`InputContract`s and output roots. It returns a reusable `PhysicalPostASAPDAG` containing selected native operators and no live readers. Compilation validates schemas, input ordering, sharing and boundedness before deployment source access. -A deployment calls `CompiledPhysicalDag::instantiate` with exactly the declared +A deployment calls `PhysicalPostASAPDAG::instantiate` with exactly the declared inputs. This checks source schemas and execution properties and constructs the runnable graph without repeating logical lowering. The graph executes through the shared runtime with independent per-run state. Window coverage, revision and diff --git a/crates/asap-physical-operators/src/dag/mod.rs b/crates/asap-physical-operators/src/dag/mod.rs index c7837672..02fa656a 100644 --- a/crates/asap-physical-operators/src/dag/mod.rs +++ b/crates/asap-physical-operators/src/dag/mod.rs @@ -1,5 +1,5 @@ //! Compatibility imports. New code should use plan, runtime, operators, physical_planner and sources directly. -pub use crate::plan::{NodeId, PhysicalDag, PhysicalOperator}; +pub use crate::plan::{NodeId, PhysicalExecution, PhysicalOperator}; pub use crate::runtime::batch_execution; pub use crate::runtime::{ Input, Limits, OutputStream, Reservation, RunContext, Scope, SharedValue, diff --git a/crates/asap-physical-operators/src/expressions/planner.rs b/crates/asap-physical-operators/src/expressions/planner.rs index 2130a2f7..848ce0b2 100644 --- a/crates/asap-physical-operators/src/expressions/planner.rs +++ b/crates/asap-physical-operators/src/expressions/planner.rs @@ -3,20 +3,22 @@ use crate::{ values::{Schema, Value}, Error, }; -use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind, DataType, QueryExpr, ScalarValue}; +use planner_types::pre_asap::{ + ArithmeticOpKind, CompareOpKind, DataType, PreASAPNode, ScalarValue, +}; use std::{cmp::Ordering, sync::Arc}; pub(super) fn evaluate( - expr: &QueryExpr, + expr: &PreASAPNode, row: &[Value], schema: &planner_types::pre_asap::Schema, ) -> Result { match expr { - QueryExpr::Column(index) => row.get(*index).cloned().ok_or(Error::Invalid(format!( + PreASAPNode::Column(index) => row.get(*index).cloned().ok_or(Error::Invalid(format!( "column {index} outside row width {}", row.len() ))), - QueryExpr::Literal(value) => Ok(match value { + PreASAPNode::Literal(value) => Ok(match value { ScalarValue::Interval { months, days, @@ -32,18 +34,18 @@ pub(super) fn evaluate( ScalarValue::Boolean(value) => Value::Bool(*value), ScalarValue::Null => Value::Null, }), - QueryExpr::Compare { left, op, right } => { + PreASAPNode::Compare { left, op, right } => { let left = evaluate(left, row, schema)?; let right = evaluate(right, row, schema)?; compare(op, left, right) } - QueryExpr::Arithmetic { op, left, right } => arithmetic( + PreASAPNode::Arithmetic { op, left, right } => arithmetic( op, evaluate(left, row, schema)?, evaluate(right, row, schema)?, ), - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { - let and = matches!(expr, QueryExpr::BoolAnd(_)); + PreASAPNode::BoolAnd(parts) | PreASAPNode::BoolOr(parts) => { + let and = matches!(expr, PreASAPNode::BoolAnd(_)); let mut null = false; for part in parts { match evaluate(part, row, schema)? { @@ -55,20 +57,20 @@ pub(super) fn evaluate( } Ok(if null { Value::Null } else { Value::Bool(and) }) } - QueryExpr::Not(value) => match evaluate(value, row, schema)? { + PreASAPNode::Not(value) => match evaluate(value, row, schema)? { Value::Bool(value) => Ok(Value::Bool(!value)), Value::Null => Ok(Value::Null), _ => Err(Error::Invalid("boolean predicate required".into())), }, - QueryExpr::IsNull(value) => Ok(Value::Bool(matches!( + PreASAPNode::IsNull(value) => Ok(Value::Bool(matches!( evaluate(value, row, schema)?, Value::Null ))), - QueryExpr::IsNotNull(value) => Ok(Value::Bool(!matches!( + PreASAPNode::IsNotNull(value) => Ok(Value::Bool(!matches!( evaluate(value, row, schema)?, Value::Null ))), - QueryExpr::FunctionCall { name, args } => { + PreASAPNode::FunctionCall { name, args } => { use planner_types::pre_asap::scalar_signature::MapScalarFunction; if name.eq_ignore_ascii_case("asap_struct_field") { expr.scalar_type(schema) @@ -81,10 +83,10 @@ pub(super) fn evaluate( unreachable!() }; let offset = match &args[1] { - QueryExpr::Literal(ScalarValue::Int64(index)) => { + PreASAPNode::Literal(ScalarValue::Int64(index)) => { usize::try_from(index - 1).ok() } - QueryExpr::Literal(ScalarValue::Utf8(name)) => { + PreASAPNode::Literal(ScalarValue::Utf8(name)) => { fields.iter().position(|field| &field.name == name) } _ => None, @@ -331,16 +333,16 @@ fn cell_cmp(left: &Value, right: &Value) -> Option { #[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] pub struct CompiledExpression { - expression: QueryExpr, + expression: PreASAPNode, schema: planner_types::pre_asap::Schema, output: (DataType, bool), } impl CompiledExpression { - pub(crate) fn expression(&self) -> &QueryExpr { + pub(crate) fn expression(&self) -> &PreASAPNode { &self.expression } - pub fn compile(expression: &QueryExpr, input: &Schema) -> Result { + pub fn compile(expression: &PreASAPNode, input: &Schema) -> Result { let schema = input .fields .iter() @@ -410,13 +412,13 @@ impl CompiledExpression { evaluate(&self.expression, row, &self.schema) } } -fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Result<(), Error> { +fn validate(expr: &PreASAPNode, schema: &planner_types::pre_asap::Schema) -> Result<(), Error> { let invalid = || Error::Invalid(format!("unsupported scalar expression: {expr:?}")); expr.scalar_type(schema) .map_err(|e| Error::Invalid(e.to_string()))?; match expr { - QueryExpr::Column(_) | QueryExpr::Literal(_) => Ok(()), - QueryExpr::Arithmetic { left, right, .. } => { + PreASAPNode::Column(_) | PreASAPNode::Literal(_) => Ok(()), + PreASAPNode::Arithmetic { left, right, .. } => { for value in [left, right] { validate(value, schema)?; if !matches!( @@ -431,7 +433,7 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::Compare { left, right, op } => { + PreASAPNode::Compare { left, right, op } => { if !matches!( op, CompareOpKind::Eq @@ -475,7 +477,7 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::FunctionCall { name, args } => { + PreASAPNode::FunctionCall { name, args } => { if name != "asap_struct_field" && name != "asap_element_access" && planner_types::pre_asap::scalar_signature::MapScalarFunction::from_name(name) @@ -488,7 +490,7 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + PreASAPNode::BoolAnd(parts) | PreASAPNode::BoolOr(parts) => { for part in parts { validate(part, schema)?; if !matches!( @@ -502,7 +504,7 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::Not(value) => { + PreASAPNode::Not(value) => { validate(value, schema)?; if !matches!( value @@ -515,7 +517,7 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::IsNull(value) | QueryExpr::IsNotNull(value) => validate(value, schema), + PreASAPNode::IsNull(value) | PreASAPNode::IsNotNull(value) => validate(value, schema), _ => Err(invalid()), } } diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs index f5658a9a..ba3c161a 100644 --- a/crates/asap-physical-operators/src/physical_planner/candidates.rs +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -7,8 +7,8 @@ use super::*; #[derive(Clone, serde::Serialize, serde::Deserialize)] #[serde(try_from = "UncheckedCandidate")] pub struct PhysicalCandidate { - pub precompute: Option, - pub query: CompiledPhysicalDag, + pub precompute: Option, + pub query: PhysicalPostASAPDAG, pub materialized_outputs: BTreeMap, } @@ -21,12 +21,12 @@ pub struct PhysicalCandidate { /// contract used to build each output. This API never treats a result from a /// different window or revision as interchangeable merely because types match. pub fn compile_candidate( - dag: &PostAsapDag, + dag: &LogicalPostASAPDAGTransport, inputs: BTreeMap, roots: &[NodeId], frontier: &[NodeId], ) -> Result { - cut_candidate(&compile(dag, inputs, roots)?, frontier) + cut_candidate(&compile(dag.as_view(), inputs, roots)?, frontier) } /// Derive one frontier's candidate from a complete [`compile`] result by @@ -34,7 +34,7 @@ pub fn compile_candidate( /// each query DAG once and derives every placement choice from that result. /// The candidate is identical to [`compile_candidate`] for the same frontier. pub fn cut_candidate( - compiled: &CompiledPhysicalDag, + compiled: &PhysicalPostASAPDAG, frontier: &[NodeId], ) -> Result { if frontier.is_empty() { @@ -93,18 +93,20 @@ pub fn cut_candidate( /// That holds while timing-dependent lowering (an ingestion-time `Binary` /// aligns by value column) has the same timing at compile time as here. /// A query-time node feeding an ingestion-time node has no valid placement. -pub fn frontier_from_timing(dag: &PostAsapDag) -> Result, Error> { +pub fn frontier_from_timing( + dag: planner_types::post_asap::LogicalPostASAPDAGView<'_>, +) -> Result, Error> { use planner_types::post_asap::ExecutionTiming::IngestionTime; let timing = dag - .nodes + .nodes() .iter() - .map(|node| (node.id, node.output_state.timing)) + .map(|node| (node.id, dag.timing(node))) .collect::>(); let mut frontier = BTreeSet::new(); - if timing.get(&dag.root) == Some(&IngestionTime) { - frontier.insert(u64::from(dag.root.0)); + if timing.get(&dag.root()) == Some(&IngestionTime) { + frontier.insert(u64::from(dag.root().0)); } - for edge in &dag.edges { + for edge in dag.edges() { let (Some(&producer), Some(&consumer)) = (timing.get(&edge.producer), timing.get(&edge.consumer)) else { @@ -127,16 +129,19 @@ pub fn frontier_from_timing(dag: &PostAsapDag) -> Result, Error> { /// and deployment feasibility are evaluated separately before cost selection. /// Exceeding the search budget returns an error, never a partial inventory. pub fn enumerate_frontiers( - dag: &PostAsapDag, + dag: &LogicalPostASAPDAGTransport, inputs: &BTreeMap, roots: &[NodeId], max_candidates: usize, ) -> Result>, Error> { - enumerate_compiled_frontiers(&compile(dag, inputs.clone(), roots)?, max_candidates) + enumerate_compiled_frontiers( + &compile(dag.as_view(), inputs.clone(), roots)?, + max_candidates, + ) } fn enumerate_compiled_frontiers( - compiled: &CompiledPhysicalDag, + compiled: &PhysicalPostASAPDAG, max_candidates: usize, ) -> Result>, Error> { if max_candidates == 0 { @@ -192,12 +197,12 @@ fn enumerate_compiled_frontiers( /// individual failures visible; do not substitute another computation on error. /// The DAG is lowered once; each frontier is a [`cut_candidate`] of it. pub fn compile_candidates( - dag: &PostAsapDag, + dag: &LogicalPostASAPDAGTransport, inputs: BTreeMap, roots: &[NodeId], frontiers: &[Vec], ) -> Vec> { - match compile(dag, inputs, roots) { + match compile(dag.as_view(), inputs, roots) { Ok(compiled) => frontiers .iter() .map(|frontier| cut_candidate(&compiled, frontier)) @@ -206,6 +211,162 @@ pub fn compile_candidates( } } +// The cut descriptor is an implementation detail of the collection. A public +// entry exposes its shared PhysicalPostASAPDAG, metadata and diagnostics directly. +struct CompiledCut { + compiled: Arc, + frontier: Vec, +} + +/// Why one timed candidate has no physical realization. Timing failures keep +/// the caller's typed lifecycle error; compilation failures are this crate's. +#[derive(Debug, Clone, thiserror::Error)] +pub enum PhysicalCandidateError { + #[error("lifecycle timing failed: {0}")] + Timing(E), + #[error(transparent)] + Compile(Error), +} + +/// All physical alternatives with their caller-supplied candidate metadata `M` +/// (for example `asap_aware_mapping::PostASAPCandidateMetadata`). The collection +/// owns shared compilation and cut descriptors; callers do not assemble a second +/// candidate wrapper or lose generation errors in a filter. +pub struct CandidatePhysicalPostASAPDAGs { + entries: Vec<(M, Result>)>, + rejected_assemblies: Vec, +} +impl CandidatePhysicalPostASAPDAGs { + pub fn len(&self) -> usize { + self.entries.len() + } + pub fn is_empty(&self) -> bool { + self.entries.is_empty() + } + pub fn rejected_assemblies(&self) -> &[String] { + &self.rejected_assemblies + } + pub fn iter( + &self, + ) -> impl Iterator< + Item = ( + &M, + Result<&Arc, &PhysicalCandidateError>, + ), + > { + self.entries + .iter() + .map(|(metadata, result)| (metadata, result.as_ref().map(|cut| &cut.compiled))) + } + pub fn frontier(&self, candidate: usize) -> Result<&[NodeId], PhysicalCandidateError> { + Ok(&self.cut(candidate)?.frontier) + } + pub fn materialize( + &self, + candidate: usize, + ) -> Result> { + let cut = self.cut(candidate)?; + cut_candidate(&cut.compiled, &cut.frontier).map_err(PhysicalCandidateError::Compile) + } + fn cut(&self, candidate: usize) -> Result<&CompiledCut, PhysicalCandidateError> { + self.entries + .get(candidate) + .ok_or_else(|| PhysicalCandidateError::Compile(invalid("unknown physical candidate")))? + .1 + .as_ref() + .map_err(Clone::clone) + } +} + +/// Compile timed logical candidates (e.g. the iterator of +/// `CandidateLifecyclePostASAPDAGs`), preserving every assignment and +/// rejection. Input contracts and requested roots can differ between logical +/// realizations. The resolver supplies contracts, never live runtime readers. +/// The metadata and timing error types are the caller's, so this layer does +/// not depend on lifecycle planning. +pub fn compile_physical_dag_candidates( + candidates: impl IntoIterator< + Item = ( + M, + Result, + ), + >, + rejected_assemblies: Vec, + mut resolve_inputs: impl FnMut( + &M, + &planner_types::post_asap::LogicalPostASAPDAGAssignment, + ) -> Result<(BTreeMap, Vec), Error>, +) -> CandidatePhysicalPostASAPDAGs { + use planner_types::post_asap::ExecutionTiming; + // Keep graph identities alive, and include contracts/roots in reuse checks: + // different candidate input boundaries must never share an invalid lowering. + let mut compilations: Vec = Vec::new(); + let entries = candidates + .into_iter() + .map(|(metadata, assignment)| { + let result = + assignment + .map_err(PhysicalCandidateError::Timing) + .and_then(|assignment| { + (|| { + let (inputs, roots) = resolve_inputs(&metadata, &assignment)?; + // Only an ingestion-time Binary lowers differently, so + // other timing differences share one compilation. + let binary_timing = assignment + .index() + .node_views() + .iter() + .filter(|node| matches!(node.payload, Payload::Binary { .. })) + .map(|node| { + ( + node.id.0, + assignment.phases()[&node.id] + == ExecutionTiming::IngestionTime, + ) + }) + .collect::>(); + let compiled = if let Some(existing) = + compilations.iter().find(|existing| { + std::rc::Rc::ptr_eq(&existing.index, assignment.index()) + && existing.binary_timing == binary_timing + && existing.inputs == inputs + && existing.roots == roots + }) { + existing.compiled.clone()? + } else { + let compiled = compile(assignment.view(), inputs.clone(), &roots) + .map(Arc::new); + compilations.push(SharedCompilation { + index: assignment.index().clone(), + binary_timing, + inputs, + roots, + compiled: compiled.clone(), + }); + compiled? + }; + let frontier = frontier_from_timing(assignment.view())?; + Ok(CompiledCut { compiled, frontier }) + })() + .map_err(PhysicalCandidateError::Compile) + }); + (metadata, result) + }) + .collect(); + CandidatePhysicalPostASAPDAGs { + entries, + rejected_assemblies, + } +} + +struct SharedCompilation { + index: std::rc::Rc, + binary_timing: Vec<(u32, bool)>, + inputs: BTreeMap, + roots: Vec, + compiled: Result, Error>, +} + /// Complete workload cost supplied by scoped optimizer/deployment evidence. /// The evaluator includes build/update work, retained state, shared producers /// and recurrent reads over the same horizon; these are not per-query timings. @@ -272,8 +433,8 @@ pub fn select_candidate( #[derive(serde::Deserialize)] #[serde(deny_unknown_fields)] struct UncheckedCandidate { - precompute: Option, - query: CompiledPhysicalDag, + precompute: Option, + query: PhysicalPostASAPDAG, materialized_outputs: BTreeMap, } impl TryFrom for PhysicalCandidate { @@ -331,16 +492,44 @@ mod tests { use super::*; use planner_types::workload::*; - fn grouped_rate() -> (PostAsapDag, BTreeMap, NodeId) { + fn compile_test_candidates( + candidates: impl IntoIterator< + Item = ( + usize, + planner_types::post_asap::LogicalPostASAPDAGAssignment, + ), + >, + inputs: BTreeMap, + roots: &[NodeId], + ) -> CandidatePhysicalPostASAPDAGs { + compile_physical_dag_candidates( + candidates + .into_iter() + .map(|(id, timing)| (id, Ok::<_, String>(timing))), + Vec::new(), + |_, _| Ok((inputs.clone(), roots.to_vec())), + ) + } + + fn grouped_root() -> planner_types::post_asap::LogicalPostASAPDAG { + logical_root("sum by(job)(rate(m[1m]))") + } + + fn logical_root(query: &str) -> planner_types::post_asap::LogicalPostASAPDAG { + logical_root_at(query, planner_types::types::AccuracyTarget::Exact) + } + + fn logical_root_at( + query: &str, + accuracy: planner_types::types::AccuracyTarget, + ) -> planner_types::post_asap::LogicalPostASAPDAG { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, query_batch: Some(vec![BatchEntry { - query: Query("sum by(job)(rate(m[1m]))".into()), + query: Query(query.into()), requirements: QueryRequirements { - accuracy: AccuracyRequirement::Explicit( - planner_types::types::AccuracyTarget::Exact, - ), + accuracy: AccuracyRequirement::Explicit(accuracy.clone()), ..Default::default() }, predictability: Predictability::Unknown, @@ -365,10 +554,19 @@ mod tests { let space = asap_aware_mapping::search_workload(vec![("q", root)]); let selected = space .global_selection(&asap_aware_mapping::cost_model::DefaultCostModel) - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .unwrap() .unwrap(); - let dag = planner_types::post_asap::compile_post_asap_dag(&selected).unwrap(); + selected + } + + fn grouped_rate() -> ( + LogicalPostASAPDAGTransport, + BTreeMap, + NodeId, + ) { + let selected = grouped_root(); + let dag = planner_types::post_asap::export_post_asap_dag(&selected).unwrap(); let state = dag .nodes .iter() @@ -381,13 +579,329 @@ mod tests { (dag.clone(), inputs, u64::from(dag.root.0)) } + /// Assignments share the logical index and physical lowering; on-demand cuts + /// have exactly the same contracts and operators as independent compilation. + #[test] + fn named_candidates_share_graphs_and_preserve_cuts() { + use planner_types::post_asap::{ + index_post_asap_dag, ExecutionTiming, LogicalPostASAPDAGAssignment, + }; + let root = grouped_root(); + let index = std::rc::Rc::new(index_post_asap_dag(&root).unwrap()); + assert!(std::rc::Rc::ptr_eq( + index.node_ids.summary_node(index.root_id).unwrap(), + &root + )); + let inputs = raw_input(&index.to_transport()); + let assignments = + [ExecutionTiming::QueryTime, ExecutionTiming::IngestionTime].map(|phase| { + LogicalPostASAPDAGAssignment::new( + index.clone(), + index.node_views().iter().map(|n| (n.id, phase)).collect(), + ) + .unwrap() + }); + let lowered = || crate::physical_planner::LOWERED_NODES.with(|count| count.get()); + let before = lowered(); + let candidates = compile_test_candidates( + assignments.iter().cloned().enumerate(), + inputs.clone(), + &[u64::from(index.root_id.0)], + ); + let count = lowered() - before; + let a = candidates.iter().next().unwrap().1.unwrap(); + let b = candidates.iter().nth(1).unwrap().1.unwrap(); + assert!(Arc::ptr_eq(a, b)); + assert!(count > 0); + for (i, assignment) in assignments.iter().enumerate() { + let actual = candidates.materialize(i).unwrap(); + assert_eq!(lowered() - before, count); + actual.validate().unwrap(); + let expected = + cut_candidate(a, &frontier_from_timing(assignment.view()).unwrap()).unwrap(); + assert_eq!( + serde_json::to_vec(&actual).unwrap(), + serde_json::to_vec(&expected).unwrap() + ); + } + let direct = compile(index.view(), inputs.clone(), &[u64::from(index.root_id.0)]).unwrap(); + let imported = compile( + index.to_transport().as_view(), + inputs, + &[u64::from(index.root_id.0)], + ) + .unwrap(); + assert_eq!( + serde_json::to_vec(&direct).unwrap(), + serde_json::to_vec(&imported).unwrap() + ); + let mut illegal = assignments[1].phases().clone(); + let first = index.edges().first().unwrap(); + illegal.insert(first.producer, ExecutionTiming::QueryTime); + assert!(LogicalPostASAPDAGAssignment::new(index.clone(), illegal).is_err()); + assert!(LogicalPostASAPDAGAssignment::new(index, BTreeMap::new()).is_err()); + } + + /// Binary placement changes lowering, so these candidates must not share + /// a compilation even though the logical graph and input contracts agree. + #[test] + fn binary_timing_gets_distinct_compilations_and_errors_keep_identity() { + use planner_types::post_asap::{ + index_post_asap_dag, ExecutionTiming, LogicalPostASAPDAGAssignment, + }; + let lhs = logical_root("m"); + let rhs = logical_root("n"); + let root = std::rc::Rc::new(planner_types::post_asap::PostASAPNode { + schema: lhs.schema.clone(), + guarantee: lhs.guarantee.clone(), + expr: planner_types::post_asap::SummaryExpr::BinaryOp { + lhs, + rhs, + timing: ExecutionTiming::QueryTime, + operator: planner_types::post_asap::BinaryOperator { + kind: planner_types::pre_asap::BinaryOpKind::Arithmetic( + planner_types::pre_asap::ArithmeticOpKind::Add, + ), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + }, + }); + let index = std::rc::Rc::new(index_post_asap_dag(&root).unwrap()); + assert!(index + .node_views() + .iter() + .any(|n| matches!(n.payload, Payload::Binary { .. }))); + let inputs = index + .node_views() + .iter() + .filter(|n| matches!(n.payload, Payload::Fallback { .. })) + .map(|n| { + ( + u64::from(n.id.0), + InputContract::bounded(Arc::new(n.output_schema.clone())), + ) + }) + .collect(); + let assignments = + [ExecutionTiming::QueryTime, ExecutionTiming::IngestionTime].map(|phase| { + LogicalPostASAPDAGAssignment::new( + index.clone(), + index.node_views().iter().map(|n| (n.id, phase)).collect(), + ) + .unwrap() + }); + let candidates = compile_test_candidates( + assignments.iter().cloned().enumerate(), + inputs, + &[u64::from(index.root_id.0)], + ); + let a = candidates.iter().next().unwrap().1.unwrap(); + let b = candidates.iter().nth(1).unwrap().1.unwrap(); + assert!(!Arc::ptr_eq(a, b)); + candidates.materialize(0).unwrap().validate().unwrap(); + candidates.materialize(1).unwrap().validate().unwrap(); + let rejected = + compile_test_candidates(assignments.into_iter().enumerate(), BTreeMap::new(), &[999]); + assert_eq!( + rejected + .iter() + .map(|(metadata, _)| *metadata) + .collect::>(), + vec![0, 1] + ); + assert!(rejected + .iter() + .all(|(_, result)| matches!(result, Err(PhysicalCandidateError::Compile(_))))); + } + + /// Candidates share one compilation unless a Binary's timing differs, so + /// every other payload must lower identically at ingestion and query time. + #[test] + fn only_binary_lowering_depends_on_timing() { + use planner_types::post_asap::{ + index_post_asap_dag, ExecutionTiming, LogicalPostASAPDAGAssignment, + }; + use planner_types::types::AccuracyTarget::{Epsilon, Exact}; + let queries = [ + ("sum by(job)(rate(m[1m]))", Exact), + ("m", Exact), + ("count(m)", Exact), + ("topk(2, m)", Exact), + ("max_over_time(m[5m])", Exact), + ("sum(increase(m[1m]))", Exact), + ("quantile(0.5, m)", Epsilon(0.01)), + ("quantile_over_time(0.9, m[5m])", Epsilon(0.01)), + ("count(count by(job)(m))", Epsilon(0.05)), + ]; + let mut payloads = BTreeSet::new(); + for (query, accuracy) in queries { + let index = + std::rc::Rc::new(index_post_asap_dag(&logical_root_at(query, accuracy)).unwrap()); + assert!(!index + .node_views() + .iter() + .any(|n| matches!(n.payload, Payload::Binary { .. }))); + let inputs = raw_input(&index.to_transport()); + let roots = [u64::from(index.root_id.0)]; + let lowered = + [ExecutionTiming::QueryTime, ExecutionTiming::IngestionTime].map(|phase| { + let assignment = LogicalPostASAPDAGAssignment::new( + index.clone(), + index.node_views().iter().map(|n| (n.id, phase)).collect(), + ) + .unwrap(); + compile(assignment.view(), inputs.clone(), &roots) + .map(|dag| serde_json::to_vec(&dag).unwrap()) + .map_err(|error| error.to_string()) + }); + // Unsupported shapes must also fail identically at either timing. + assert_eq!(lowered[0], lowered[1], "{query}"); + if lowered[0].is_err() { + continue; + } + payloads.extend(index.node_views().iter().map(|n| { + format!("{:?}", n.payload) + .split([' ', '{', '(']) + .next() + .unwrap() + .to_string() + })); + } + // The fixtures exercise every non-Binary payload kind the compiler lowers natively. + for kind in ["Fallback", "SummaryAgg", "SummaryEstimate", "Value"] { + assert!(payloads.contains(kind), "{kind} not covered: {payloads:?}"); + } + } + + /// Each collection candidate equals an independent compilation of its own + /// assignment after a transport round trip; an ingestion-time Binary lowers + /// differently, so the collection must compile from the assignment's timing. + #[test] + fn candidates_match_independent_compilation_of_their_assignments() { + use planner_types::post_asap::{ + index_post_asap_dag, ExecutionTiming, LogicalPostASAPDAGAssignment, + LogicalPostASAPDAGDocument, + }; + let lhs = logical_root("m"); + let rhs = logical_root("n"); + let root = std::rc::Rc::new(planner_types::post_asap::PostASAPNode { + schema: lhs.schema.clone(), + guarantee: lhs.guarantee.clone(), + expr: planner_types::post_asap::SummaryExpr::BinaryOp { + lhs, + rhs, + timing: ExecutionTiming::QueryTime, + operator: planner_types::post_asap::BinaryOperator { + kind: planner_types::pre_asap::BinaryOpKind::Arithmetic( + planner_types::pre_asap::ArithmeticOpKind::Add, + ), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + }, + }); + let index = std::rc::Rc::new(index_post_asap_dag(&root).unwrap()); + let inputs: BTreeMap<_, _> = index + .node_views() + .iter() + .filter(|n| matches!(n.payload, Payload::Fallback { .. })) + .map(|n| { + ( + u64::from(n.id.0), + InputContract::bounded(Arc::new(n.output_schema.clone())), + ) + }) + .collect(); + let roots = [u64::from(index.root_id.0)]; + let assignments = + [ExecutionTiming::QueryTime, ExecutionTiming::IngestionTime].map(|phase| { + LogicalPostASAPDAGAssignment::new( + index.clone(), + index.node_views().iter().map(|n| (n.id, phase)).collect(), + ) + .unwrap() + }); + let candidates = compile_test_candidates( + assignments.iter().cloned().enumerate(), + inputs.clone(), + &roots, + ); + for (i, assignment) in assignments.iter().enumerate() { + let bytes = + serde_json::to_vec(&LogicalPostASAPDAGDocument::new(assignment.to_transport())) + .unwrap(); + let document: LogicalPostASAPDAGDocument = serde_json::from_slice(&bytes).unwrap(); + document.validate().unwrap(); + let frontier = frontier_from_timing(document.dag.as_view()).unwrap(); + let expected = + compile_candidate(&document.dag, inputs.clone(), &roots, &frontier).unwrap(); + assert_eq!( + serde_json::to_vec(&candidates.materialize(i).unwrap()).unwrap(), + serde_json::to_vec(&expected).unwrap(), + "assignment {i}" + ); + } + let lowered = assignments.each_ref().map(|a| { + serde_json::to_vec(&compile(a.view(), inputs.clone(), &roots).unwrap()).unwrap() + }); + assert_ne!(lowered[0], lowered[1]); + } + + /// Identical logical timing is insufficient for reuse when the deployment + /// supplies different input contracts or requests different output roots. + #[test] + fn shared_compilation_respects_contracts_and_roots() { + use planner_types::post_asap::{ + index_post_asap_dag, ExecutionTiming, LogicalPostASAPDAGAssignment, + }; + let index = std::rc::Rc::new(index_post_asap_dag(&grouped_root()).unwrap()); + let assignment = LogicalPostASAPDAGAssignment::new( + index.clone(), + index + .node_views() + .iter() + .map(|n| (n.id, ExecutionTiming::QueryTime)) + .collect(), + ) + .unwrap(); + let inputs = raw_input(&index.to_transport()); + let raw = *inputs.keys().next().unwrap(); + let candidates = compile_physical_dag_candidates( + (0..3usize).map(|id| (id, Ok::<_, String>(assignment.clone()))), + Vec::new(), + |metadata, _| { + let mut inputs = inputs.clone(); + if *metadata == 1 { + inputs + .values_mut() + .for_each(|input| input.properties.emission = Emission::AfterInput); + } + let root = if *metadata == 2 { + raw + } else { + u64::from(index.root_id.0) + }; + Ok((inputs, vec![root])) + }, + ); + let dags = candidates + .iter() + .map(|(_, dag)| dag.unwrap()) + .collect::>(); + assert!(!Arc::ptr_eq(dags[0], dags[1])); + assert!(!Arc::ptr_eq(dags[0], dags[2])); + } + /// Enumerating and cutting every frontier lowers each Planner node once. #[test] fn candidates_for_all_frontiers_share_one_lowering() { let (dag, inputs, root) = grouped_rate(); let lowered = || crate::physical_planner::LOWERED_NODES.with(|count| count.get()); let before = lowered(); - let compiled = compile(&dag, inputs, &[root]).unwrap(); + let compiled = compile(dag.as_view(), inputs, &[root]).unwrap(); let once = lowered() - before; let frontiers = enumerate_compiled_frontiers(&compiled, 4096).unwrap(); assert!(frontiers.len() >= 3, "{frontiers:?}"); @@ -399,9 +913,9 @@ mod tests { } fn with_timing( - dag: &PostAsapDag, - timing: impl Fn(&PostAsapDagNode) -> planner_types::post_asap::ExecutionTiming, - ) -> PostAsapDag { + dag: &LogicalPostASAPDAGTransport, + timing: impl Fn(&LogicalPostASAPDAGNode) -> planner_types::post_asap::ExecutionTiming, + ) -> LogicalPostASAPDAGTransport { let mut timed = dag.clone(); for node in &mut timed.nodes { node.output_state.timing = timing(node); @@ -413,7 +927,7 @@ mod tests { timed } - fn raw_input(dag: &PostAsapDag) -> BTreeMap { + fn raw_input(dag: &LogicalPostASAPDAGTransport) -> BTreeMap { let raw = dag .nodes .iter() @@ -436,10 +950,10 @@ mod tests { let inputs = raw_input(&retained); let lowered = || crate::physical_planner::LOWERED_NODES.with(|count| count.get()); let before = lowered(); - let compiled = compile(&ephemeral, inputs.clone(), &[root]).unwrap(); + let compiled = compile(ephemeral.as_view(), inputs.clone(), &[root]).unwrap(); let once = lowered() - before; let cuts = [&retained, &ephemeral].map(|timed| { - let frontier = frontier_from_timing(timed).unwrap(); + let frontier = frontier_from_timing(timed.as_view()).unwrap(); let cut = cut_candidate(&compiled, &frontier).unwrap(); (timed, frontier, cut) }); @@ -463,7 +977,7 @@ mod tests { use planner_types::post_asap::ExecutionTiming::IngestionTime; let (dag, _, root) = grouped_rate(); let timed = with_timing(&dag, |_| IngestionTime); - assert_eq!(frontier_from_timing(&timed).unwrap(), [root]); + assert_eq!(frontier_from_timing(timed.as_view()).unwrap(), [root]); } /// A query-time node feeding an ingestion-time node is rejected. @@ -478,6 +992,6 @@ mod tests { QueryTime } }); - assert!(frontier_from_timing(&timed).is_err()); + assert!(frontier_from_timing(timed.as_view()).is_err()); } } diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs index af0bbe39..9b6680f9 100644 --- a/crates/asap-physical-operators/src/physical_planner/compiled.rs +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -2,7 +2,7 @@ use super::*; /// A typed execution boundary, without storage identity or a live reader. -#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)] +#[derive(Clone, Debug, PartialEq, serde::Serialize, serde::Deserialize)] pub struct InputContract { pub schema: Schema, pub properties: PlanProperties, @@ -38,7 +38,7 @@ enum Node { /// Deserialization validates the graph before it is usable. #[derive(Clone, serde::Serialize, serde::Deserialize)] #[serde(try_from = "UncheckedDag")] -pub struct CompiledPhysicalDag { +pub struct PhysicalPostASAPDAG { nodes: BTreeMap, roots: Vec, } @@ -48,7 +48,7 @@ struct UncheckedDag { nodes: BTreeMap, roots: Vec, } -impl TryFrom for CompiledPhysicalDag { +impl TryFrom for PhysicalPostASAPDAG { type Error = Error; fn try_from(dag: UncheckedDag) -> Result { let result = Self { @@ -60,7 +60,7 @@ impl TryFrom for CompiledPhysicalDag { } } -impl CompiledPhysicalDag { +impl PhysicalPostASAPDAG { /// Link already-selected physical fragments without lowering operators again. /// Fragment keys and source keys share a namespace; repeated dependency IDs /// therefore remain one producer in the composed graph. @@ -285,8 +285,8 @@ impl CompiledPhysicalDag { pub fn instantiate<'a>( &self, mut sources: BTreeMap>, - ) -> Result, Error> { - let mut graph = PhysicalDag::default(); + ) -> Result, Error> { + let mut graph = PhysicalExecution::default(); for (&id, node) in &self.nodes { match node { Node::Input(contract) => { diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index edcfb833..29afc4f1 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -4,17 +4,18 @@ use crate::operators::ReadoutQuery; use crate::summary_kernels::exact::ExactReadout; use crate::{ operators::{Expression, Operator, Reduction, SortKey}, - plan::{Boundedness, Emission, NodeId, PhysicalDag, PhysicalOperator, PlanProperties}, + plan::{Boundedness, Emission, NodeId, PhysicalExecution, PhysicalOperator, PlanProperties}, values::{Batch, Schema}, Error, }; use planner_types::{ post_asap::{ - ExactOperation, PostAsapDag, PostAsapDagNode, PostAsapOperatorPayload as Payload, - SketchQuery, SummaryFamilyType, SummaryInputExpr, ValueOperation, + ExactOperation, LogicalPostASAPDAGNode, LogicalPostASAPDAGTransport, + PostASAPOperatorPayload as Payload, SketchQuery, SummaryFamilyType, SummaryInputExpr, + ValueOperation, }, pre_asap::{ - AggIntent, ColumnRef, CompareOpKind, DataType, GroupKeys, QueryExpr, + AggIntent, ColumnRef, CompareOpKind, DataType, GroupKeys, PreASAPNode, Reduction as PlannerReduction, }, }; @@ -38,46 +39,49 @@ pub mod promql_values; mod candidates; pub use candidates::{ - compile_candidate, compile_candidates, cut_candidate, enumerate_frontiers, - frontier_from_timing, select_candidate, CandidateCost, CandidateSelection, PhysicalCandidate, + compile_candidate, compile_candidates, compile_physical_dag_candidates, cut_candidate, + enumerate_frontiers, frontier_from_timing, select_candidate, CandidateCost, + CandidatePhysicalPostASAPDAGs, CandidateSelection, PhysicalCandidate, PhysicalCandidateError, }; mod compiled; -pub use compiled::{CompiledPhysicalDag, InputContract}; +pub use compiled::{InputContract, PhysicalPostASAPDAG}; mod row_values; /// Compile computation without opening or retaining deployment readers. /// Input contracts identify explicit boundaries selected by maintenance planning. +/// The view is a transport document's ([`LogicalPostASAPDAGTransport::as_view`]), a +/// shared index's, or a lifecycle assignment's timing over that index. pub fn compile( - dag: &PostAsapDag, + dag: planner_types::post_asap::LogicalPostASAPDAGView<'_>, inputs: BTreeMap, roots: &[NodeId], -) -> Result { - compile_internal(dag, inputs, roots) +) -> Result { + compile_internal(&dag, inputs, roots) } /// Convenience for callers that already resolved inputs. Lowering still uses /// only their contracts, and instantiation checks those contracts again. pub fn bind<'a>( - dag: &PostAsapDag, + dag: &LogicalPostASAPDAGTransport, sources: BTreeMap>, roots: &[NodeId], -) -> Result, Error> { +) -> Result, Error> { let inputs = sources .iter() .map(|(&id, source)| (id, InputContract::from_source(source.as_ref()))) .collect(); - compile(dag, inputs, roots)?.instantiate(sources) + compile(dag.as_view(), inputs, roots)?.instantiate(sources) } /// Resolve raw scan connectors before invoking the reader-independent compiler. pub fn bind_with_data_sources<'a>( - dag: &PostAsapDag, + dag: &LogicalPostASAPDAGTransport, mut sources: BTreeMap>, roots: &[NodeId], data_sources: &crate::sources::DataSources, -) -> Result, Error> { +) -> Result, Error> { // Only resolve scans reachable below the selected input boundaries. let mut pending = roots.to_vec(); let mut seen = BTreeSet::new(); @@ -91,7 +95,7 @@ pub fn bind_with_data_sources<'a>( .find(|n| u64::from(n.id.0) == id) .ok_or_else(|| invalid(format!("missing node {id}")))?; if let Payload::Fallback { - expression: expression @ QueryExpr::Scan { .. }, + expression: expression @ PreASAPNode::Scan { .. }, } = &node.payload { sources.insert(id, Box::new(data_sources.bind(expression)?)); @@ -123,20 +127,20 @@ fn helper_id(node: NodeId, index: u64) -> NodeId { } fn compile_internal( - dag: &PostAsapDag, + dag: &planner_types::post_asap::LogicalPostASAPDAGView<'_>, mut sources: BTreeMap, roots: &[NodeId], -) -> Result { +) -> Result { preflight_depth(dag)?; dag.validate().map_err(|e| invalid(e.to_string()))?; let nodes = dag - .nodes + .nodes() .iter() .map(|node| (u64::from(node.id.0), node)) .collect::>(); let mut dependencies = BTreeMap::>::new(); // Binary input order is semantic; serialized edge order is not. - let mut edges = dag.edges.iter().collect::>(); + let mut edges = dag.edges().iter().collect::>(); edges.sort_by_key(|edge| { ( edge.consumer.0, @@ -153,7 +157,7 @@ fn compile_internal( let consumer = u64::from(edge.consumer.0); if let ( Payload::Fallback { expression }, - Some(PostAsapDagNode { + Some(LogicalPostASAPDAGNode { payload: Payload::Binary { .. }, .. }), @@ -179,7 +183,7 @@ fn compile_internal( || promql_fallback::raw_series_owner(*id).is_some_and(|owner| { matches!( nodes.get(&owner), - Some(PostAsapDagNode { + Some(LogicalPostASAPDAGNode { payload: Payload::Fallback { .. }, .. }) @@ -210,7 +214,7 @@ fn compile_internal( } } } - let mut graph = CompiledPhysicalDag::new(roots.to_vec()); + let mut graph = PhysicalPostASAPDAG::new(roots.to_vec()); for id in ordered { let node = nodes[&id]; let mut auxiliary = helper_id(id, 0); @@ -246,9 +250,9 @@ fn compile_internal( let raw_rows = matches!( &node.payload, Payload::Fallback { - expression: QueryExpr::TimeRange { .. } + expression: PreASAPNode::TimeRange { .. } } - ) && dag.edges.iter().any(|e| u64::from(e.producer.0) == id); + ) && dag.edges().iter().any(|e| u64::from(e.producer.0) == id); if let (Payload::Fallback { expression }, false) = (&node.payload, raw_rows) { let promql_fallback::Lowering { selectors, @@ -415,14 +419,14 @@ fn compile_internal( return Err(invalid("per-entity summary requires one input")); }; let Payload::Fallback { - expression: QueryExpr::TimeRange { child, .. }, + expression: PreASAPNode::TimeRange { child, .. }, } = &nodes[input_id].payload else { return Err(invalid( "per-entity summary requires a resolved raw time range", )); }; - let QueryExpr::Scan { schema, .. } = child.as_ref() else { + let PreASAPNode::Scan { schema, .. } = child.as_ref() else { return Err(invalid("per-entity summary requires a resolved source")); }; if !schema.closed || update.item.is_some() { @@ -462,8 +466,8 @@ fn compile_internal( continue; } if let Payload::Binary { operator } = &node.payload { - let query_time = node.output_state.timing - == planner_types::post_asap::ExecutionTiming::QueryTime; + let query_time = + dag.timing(node) == planner_types::post_asap::ExecutionTiming::QueryTime; if let Some(&(value, left)) = literals.get(&id) { let [input] = schemas.as_slice() else { return Err(invalid("scalar binary requires one row input")); @@ -529,7 +533,7 @@ fn compile_internal( } = &node.payload { // Exact counts read out as Int64; PromQL declares a Float64 sample. - let readout = bind_operation(node, &schemas) + let readout = bind_operation(node, dag.timing(node), &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; let actual = readout.schema(); let converted = actual.fields.iter().zip(&output.fields).position(|(a, d)| { @@ -569,7 +573,7 @@ fn compile_internal( continue; } } - let mut operator = compile_node(node, &schemas) + let mut operator = compile_timed_node(node, dag.timing(node), &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; if operator.is_counter_readout() { let mut pending = vec![id]; @@ -580,7 +584,7 @@ fn compile_internal( continue; } if let Payload::Fallback { - expression: QueryExpr::TimeRange { range, .. }, + expression: PreASAPNode::TimeRange { range, .. }, } = &nodes[&ancestor].payload { ranges.insert( @@ -612,7 +616,7 @@ fn compile_internal( // Temporal summary readouts produce PromQL vectors, whose range functions drop // the metric name before matching/filtering. Stored state retains its full identity. -fn temporal_readout_drops_name(node: &PostAsapDagNode) -> bool { +fn temporal_readout_drops_name(node: &LogicalPostASAPDAGNode) -> bool { node.output_schema .fields .iter() @@ -633,19 +637,32 @@ fn temporal_readout_drops_name(node: &PostAsapDagNode) -> bool { /// Bind a Planner node against the schemas supplied by its deployment edges. /// This is the same checked path used by complete DAG binding. -pub fn compile_node(node: &PostAsapDagNode, inputs: &[Schema]) -> Result { +pub fn compile_node(node: &LogicalPostASAPDAGNode, inputs: &[Schema]) -> Result { + compile_timed_node(node, node.output_state.timing, inputs) +} + +/// Lower `node` under `timing`, which a lifecycle assignment may overlay. +fn compile_timed_node( + node: &LogicalPostASAPDAGNode, + timing: planner_types::post_asap::ExecutionTiming, + inputs: &[Schema], +) -> Result { for schema in inputs { crate::values::validate_schema(schema)?; } - bind_operation(node, inputs)?.with_output_schema(Arc::new(node.output_schema.clone())) + bind_operation(node, timing, inputs)?.with_output_schema(Arc::new(node.output_schema.clone())) } -fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result { +fn bind_operation( + node: &LogicalPostASAPDAGNode, + timing: planner_types::post_asap::ExecutionTiming, + inputs: &[Schema], +) -> Result { if let Payload::Binary { operator } = &node.payload { let [left, right] = inputs else { return Err(invalid("binary requires two inputs")); }; - if node.output_state.timing == planner_types::post_asap::ExecutionTiming::IngestionTime { + if timing == planner_types::post_asap::ExecutionTiming::IngestionTime { let value = |schema: &Schema| -> Result { let columns = schema .fields @@ -747,7 +764,7 @@ fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result Expression::Column(*index), + PreASAPNode::Column(index) => Expression::Column(*index), expr => expression(expr, input)?, }, )) @@ -761,7 +778,7 @@ fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result Result, Error> { } Ok(groups.keys().to_vec()) } -fn expression(expr: &QueryExpr, input: &Schema) -> Result { +fn expression(expr: &PreASAPNode, input: &Schema) -> Result { Ok(Expression::planner( crate::expressions::CompiledExpression::compile(expr, input)?, )) @@ -1047,17 +1064,19 @@ impl PhysicalOperator for CheckedSource<'_> { } // Bound recursion before invoking the upstream recursive provenance validator. -fn preflight_depth(dag: &PostAsapDag) -> Result<(), Error> { +fn preflight_depth( + dag: &planner_types::post_asap::LogicalPostASAPDAGView<'_>, +) -> Result<(), Error> { let mut remaining = dag - .nodes + .nodes() .iter() .map(|node| (node.id, 0usize)) .collect::>(); - if remaining.len() != dag.nodes.len() { + if remaining.len() != dag.nodes().len() { return Err(invalid("duplicate Planner node")); } let mut consumers = BTreeMap::<_, Vec<_>>::new(); - for edge in &dag.edges { + for edge in dag.edges() { if !remaining.contains_key(&edge.producer) { return Err(invalid("missing Planner edge producer")); } @@ -1092,7 +1111,7 @@ fn preflight_depth(dag: &PostAsapDag) -> Result<(), Error> { } } } - if visited != dag.nodes.len() { + if visited != dag.nodes().len() { return Err(invalid("Planner DAG contains a cycle")); } Ok(()) @@ -1100,23 +1119,23 @@ fn preflight_depth(dag: &PostAsapDag) -> Result<(), Error> { /// Join predicates address the concatenated left/right schema. fn semi_join_keys( - expr: &QueryExpr, + expr: &PreASAPNode, left: usize, right: usize, keys: &mut Vec<(usize, usize)>, ) -> Result<(), Error> { match expr { - QueryExpr::BoolAnd(parts) => { + PreASAPNode::BoolAnd(parts) => { for part in parts { semi_join_keys(part, left, right, keys)?; } } - QueryExpr::Compare { + PreASAPNode::Compare { left: a, op: CompareOpKind::Eq, right: b, } => { - let (QueryExpr::Column(a), QueryExpr::Column(b)) = (a.as_ref(), b.as_ref()) else { + let (PreASAPNode::Column(a), PreASAPNode::Column(b)) = (a.as_ref(), b.as_ref()) else { return Err(invalid("semi-join requires column equality keys")); }; let (a, b) = if a < b { (*a, *b) } else { (*b, *a) }; diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index aee87d74..b6986e14 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -77,17 +77,17 @@ pub fn raw_sample_row( /// Input contract of a precompute boundary: raw sample rows for a raw time /// series scan, otherwise the stored population of its summary state. -pub fn boundary_schema(node: &PostAsapDagNode) -> Result { +pub fn boundary_schema(node: &LogicalPostASAPDAGNode) -> Result { let Payload::Fallback { expression } = &node.payload else { return source_schema(&node.output_schema); }; let scan = match expression { - planner_types::pre_asap::QueryExpr::TimeRange { child, .. } => child.as_ref(), + planner_types::pre_asap::PreASAPNode::TimeRange { child, .. } => child.as_ref(), expression => expression, }; if !matches!( scan, - planner_types::pre_asap::QueryExpr::Scan { + planner_types::pre_asap::PreASAPNode::Scan { source: planner_types::pre_asap::Source::TimeSeries { .. }, .. } @@ -154,11 +154,11 @@ pub fn is_population_schema(schema: &Schema) -> bool { /// Compile a complete selected precompute sub-DAG. Inputs are already-computed /// state boundaries; the deployment supplies groups, panes and states, never operations. pub fn compile( - dag: &PostAsapDag, + dag: &LogicalPostASAPDAGTransport, frontiers: &[NodeId], roots: &[NodeId], -) -> Result { - preflight_depth(dag)?; +) -> Result { + preflight_depth(&dag.as_view())?; dag.validate().map_err(|e| invalid(e.to_string()))?; let nodes = dag .nodes @@ -246,10 +246,10 @@ pub fn compile( outputs.insert(id, graph.output_contract(graph.roots()[0])?.schema); fragments.insert(id, (inputs, graph)); } - CompiledPhysicalDag::compose(sources, fragments, roots.to_vec()) + PhysicalPostASAPDAG::compose(sources, fragments, roots.to_vec()) } -fn validate_value_output(node: &PostAsapDagNode) -> Result<(), Error> { +fn validate_value_output(node: &LogicalPostASAPDAGNode) -> Result<(), Error> { let schema = &node.output_schema; // Physical population rows already carry the complete identity in `$population`. // Typed logical plans may expose its opaque series-identity column as metadata. @@ -289,10 +289,10 @@ fn validate_value_output(node: &PostAsapDagNode) -> Result<(), Error> { } fn fragment( - node: &PostAsapDagNode, + node: &LogicalPostASAPDAGNode, schemas: &[Schema], - parents: &[&PostAsapDagNode], -) -> Result { + parents: &[&LogicalPostASAPDAGNode], +) -> Result { let sources = schemas .iter() .enumerate() @@ -547,7 +547,7 @@ fn fragment( )) } }; - CompiledPhysicalDag::from_operators(sources, operators, vec![root]) + PhysicalPostASAPDAG::from_operators(sources, operators, vec![root]) } /// Resolve keyed item identities over raw sample rows: labels (absent labels diff --git a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs index 9323e8e0..65b2a38b 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs @@ -19,14 +19,14 @@ pub(super) fn raw_series_owner(slot: NodeId) -> Option { } /// A selector expression and its raw-series row schema. -pub type Selector = (QueryExpr, Schema); +pub type Selector = (PreASAPNode, Schema); /// The selectors a Fallback expression reads, left to right, and the row /// schema of the raw series the deployment supplies for each at /// [`raw_series_input`]. The rows must cover the selector's window at every /// evaluation instant `T`, or at its `@` time: `(T - offset - range, T - offset]`; /// under a subquery `[R:S] offset O` that is `(T - O - R - offset - range, T - O - offset]`. -pub fn raw_series(expression: &QueryExpr) -> Result, Error> { +pub fn raw_series(expression: &PreASAPNode) -> Result, Error> { Ok(lower(expression)?.selectors) } @@ -43,13 +43,13 @@ pub(super) struct Lowering { pub steps: Vec<(Operator, Vec)>, } -pub(super) fn lower(expression: &QueryExpr) -> Result { +pub(super) fn lower(expression: &PreASAPNode) -> Result { let mut lowering = Lowering::default(); lowering.value(expression)?; Ok(lowering) } -fn declared(expression: &QueryExpr) -> Result { +fn declared(expression: &PreASAPNode) -> Result { let schema = expression .output_schema() .map_err(|error| invalid(error.to_string()))?; @@ -69,10 +69,10 @@ fn at(shift: &planner_types::pre_asap::TimeShift) -> Result, Error> } } -fn range_anchor(expression: &QueryExpr) -> Option { +fn range_anchor(expression: &PreASAPNode) -> Option { match expression { - QueryExpr::TimeRange { child, .. } => range_anchor(child), - QueryExpr::TimeShift { shift, .. } => shift + PreASAPNode::TimeRange { child, .. } => range_anchor(child), + PreASAPNode::TimeShift { shift, .. } => shift .at .filter(|at| matches!(at, AtModifier::Start | AtModifier::End)), _ => None, @@ -80,15 +80,15 @@ fn range_anchor(expression: &QueryExpr) -> Option { } /// `TimeRange { range, [TimeShift { offset, @ }], Scan }`: range, offset, `@`. -fn selector(expression: &QueryExpr) -> Result<(i64, i64, Option), Error> { - let QueryExpr::TimeRange { range, child } = expression else { +fn selector(expression: &PreASAPNode) -> Result<(i64, i64, Option), Error> { + let PreASAPNode::TimeRange { range, child } = expression else { return Err(invalid("PromQL operand must be a series selector")); }; let (offset, at, scan) = match child.as_ref() { - QueryExpr::TimeShift { shift, child } => (shift.offset_ms, at(shift)?, child.as_ref()), + PreASAPNode::TimeShift { shift, child } => (shift.offset_ms, at(shift)?, child.as_ref()), scan => (0, None, scan), }; - if !matches!(scan, QueryExpr::Scan { .. }) { + if !matches!(scan, PreASAPNode::Scan { .. }) { return Err(invalid("PromQL selector must read one scan")); } Ok((millis(range)?, offset, at)) @@ -96,12 +96,12 @@ fn selector(expression: &QueryExpr) -> Result<(i64, i64, Option), Error> { /// PromQL scalar-valued expressions have no labels to match. A binary /// operator is scalar-valued when both operands are. -pub(super) fn scalar(expression: &QueryExpr) -> bool { +pub(super) fn scalar(expression: &PreASAPNode) -> bool { match expression { - QueryExpr::PromqlScalarBridge(_) - | QueryExpr::PromqlScalarFromVector(_) - | QueryExpr::EvalTimestamp => true, - QueryExpr::BinaryOp { lhs, rhs, .. } => scalar(lhs) && scalar(rhs), + PreASAPNode::PromqlScalarBridge(_) + | PreASAPNode::PromqlScalarFromVector(_) + | PreASAPNode::EvalTimestamp => true, + PreASAPNode::BinaryOp { lhs, rhs, .. } => scalar(lhs) && scalar(rhs), _ => false, } } @@ -124,12 +124,12 @@ impl Lowering { &mut self, operator: Operator, inputs: Vec, - logical: &QueryExpr, + logical: &PreASAPNode, ) -> Result { Ok(self.add(operator.with_output_schema(declared(logical)?)?, inputs)) } - fn read(&mut self, selector: &QueryExpr) -> Result { + fn read(&mut self, selector: &PreASAPNode) -> Result { let schema = declared(selector)?; if !schema .fields @@ -145,12 +145,12 @@ impl Lowering { } /// An instant vector, or a scalar for scalar-valued expressions. - fn value(&mut self, expression: &QueryExpr) -> Result { + fn value(&mut self, expression: &PreASAPNode) -> Result { match expression { - QueryExpr::Concat { children, .. } => { + PreASAPNode::Concat { children, .. } => { if !children.iter().all(|branch| matches!(branch, - QueryExpr::PromqlRelabel { child, .. } if matches!(child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::HistogramQuantile { .. }])))) { + PreASAPNode::PromqlRelabel { child, .. } if matches!(child.as_ref(), + PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::HistogramQuantile { .. }])))) { return Err(invalid("PromQL concatenation requires classic histogram quantile branches")); } let inputs = children @@ -171,15 +171,17 @@ impl Lowering { expression, ) } - QueryExpr::PromqlRelabel { dst, value, child } => { + PreASAPNode::PromqlRelabel { dst, value, child } => { let step = self.value(child)?; let input = self.schema(&step); let (replacement, source_regex) = match value.as_ref() { - QueryExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8(value)) => { + PreASAPNode::Literal(planner_types::pre_asap::ScalarValue::Utf8(value)) => { (value.clone(), None) } - QueryExpr::FunctionCall { name, args } if name == "label_replace" => { - let [QueryExpr::Column(source), QueryExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8(pattern)), QueryExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8( + PreASAPNode::FunctionCall { name, args } if name == "label_replace" => { + let [PreASAPNode::Column(source), PreASAPNode::Literal(planner_types::pre_asap::ScalarValue::Utf8( + pattern, + )), PreASAPNode::Literal(planner_types::pre_asap::ScalarValue::Utf8( replacement, ))] = args.as_slice() else { @@ -204,7 +206,7 @@ impl Lowering { )?; self.push(operator, vec![step], expression) } - QueryExpr::TimeRange { .. } => { + PreASAPNode::TimeRange { .. } => { let (range, offset, at) = selector(expression)?; let input = self.read(expression)?; let schema = self.schema(&input); @@ -215,7 +217,7 @@ impl Lowering { expression, ) } - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction: planner_types::pre_asap::Reduction::PerEntity, measures, having: None, @@ -233,7 +235,7 @@ impl Lowering { let input = self.schema(&step); Ok(self.add(Operator::series_without_name(input)?, vec![step])) } - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction: planner_types::pre_asap::Reduction::Reduce(keys), measures, having: None, @@ -246,7 +248,7 @@ impl Lowering { let input = self.value(child)?; self.aggregate(input, measure, keys, expression) } - QueryExpr::Sort { + PreASAPNode::Sort { keys, partition_by, child, @@ -256,7 +258,7 @@ impl Lowering { let keys = keys .iter() .map(|key| match key.expr { - QueryExpr::Column(column) => Ok(SortKey { + PreASAPNode::Column(column) => Ok(SortKey { column, descending: !key.ascending, nulls_first: key.nulls_first, @@ -267,12 +269,12 @@ impl Lowering { let groups = groups(&input, partition_by)?; self.push(Operator::sort(input, keys, groups)?, vec![step], expression) } - QueryExpr::Limit { n, offset, child } => { + PreASAPNode::Limit { n, offset, child } => { let step = self.value(child)?; let input = self.schema(&step); // `topk by (...)` partitions through the Sort it limits. let groups = match child.as_ref() { - QueryExpr::Sort { partition_by, .. } => groups(&input, partition_by)?, + PreASAPNode::Sort { partition_by, .. } => groups(&input, partition_by)?, _ => vec![], }; self.push( @@ -281,7 +283,7 @@ impl Lowering { expression, ) } - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op, lhs, rhs, @@ -302,7 +304,7 @@ impl Lowering { )?; self.push(binary, sides, expression) } - QueryExpr::PromqlScalarFromVector(child) => { + PreASAPNode::PromqlScalarFromVector(child) => { let step = self.value(child)?; let input = self.schema(&step); let value = named_column(&input, &ColumnRef::SampleValue)?; @@ -312,7 +314,7 @@ impl Lowering { expression, ) } - QueryExpr::PromqlVectorFromScalar(child) => { + PreASAPNode::PromqlVectorFromScalar(child) => { let step = self.value(child)?; let input = self.schema(&step); Ok(self.add( @@ -320,8 +322,10 @@ impl Lowering { vec![step], )) } - QueryExpr::EvalTimestamp => self.push(Operator::evaluation_time(), vec![], expression), - QueryExpr::PromqlScalarBridge(_) => { + PreASAPNode::EvalTimestamp => { + self.push(Operator::evaluation_time(), vec![], expression) + } + PreASAPNode::PromqlScalarBridge(_) => { let value = row_values::scalar_literal(expression) .ok_or_else(|| invalid("PromQL scalar must be a literal"))?; self.push( @@ -338,15 +342,17 @@ impl Lowering { fn range_function( &mut self, function: &AggIntent, - matrix: &QueryExpr, - logical: &QueryExpr, + matrix: &PreASAPNode, + logical: &PreASAPNode, ) -> Result { let function = unbound(function)?; let (subquery, offset, at_ms) = match matrix { - QueryExpr::TimeShift { shift, child } => (child.as_ref(), shift.offset_ms, at(shift)?), + PreASAPNode::TimeShift { shift, child } => { + (child.as_ref(), shift.offset_ms, at(shift)?) + } other => (other, 0, None), }; - let QueryExpr::PromqlSubquery { + let PreASAPNode::PromqlSubquery { range: outer, resolution, child, @@ -373,7 +379,7 @@ impl Lowering { }; // Each step evaluates a per-series selection or range function. let (inner, selected) = match child.as_ref() { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction: planner_types::pre_asap::Reduction::PerEntity, measures, having: None, @@ -423,7 +429,7 @@ impl Lowering { mut step: Input, measure: &AggIntent, keys: &GroupKeys, - logical: &QueryExpr, + logical: &PreASAPNode, ) -> Result { let mut input = self.schema(&step); if let AggIntent::HistogramQuantile { q, le } = measure { diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index d2c13328..149a0249 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -24,7 +24,7 @@ pub fn decode_series_identity(encoded: &str) -> Result, /// Resolve the row representation before candidate search; see /// [`planner_types::pre_asap::schema::with_promql_series_identity`]. -pub fn with_series_identity(root: &QueryExpr) -> Result { +pub fn with_series_identity(root: &PreASAPNode) -> Result { planner_types::pre_asap::schema::with_promql_series_identity(root).map_err(invalid) } @@ -79,12 +79,12 @@ pub fn series_row( /// source. The boundary supplies the complete eligible vector, not a truncated /// TopK result; ranking remains a native physical operator. pub fn compile_current_series_readout( - selected: &Rc, -) -> Result { + selected: &Rc, +) -> Result { use planner_types::post_asap::{ - compile_post_asap_dag, maintained_population::PopulationReadout, SummaryField, + export_post_asap_dag, maintained_population::PopulationReadout, SummaryField, }; - let mut dag = compile_post_asap_dag(selected).map_err(|error| invalid(error.to_string()))?; + let mut dag = export_post_asap_dag(selected).map_err(|error| invalid(error.to_string()))?; // Typed snapshot candidates already carry full identity throughout the DAG. // Cut at the population output, preserving all selected heap/readout nodes. let populations = dag.nodes.iter().filter(|node| matches!(&node.payload, @@ -99,7 +99,7 @@ pub fn compile_current_series_readout( .any(|field| field.name == SERIES_IDENTITY_COLUMN) { return compile( - &dag, + dag.as_view(), BTreeMap::from([( u64::from(population.id.0), InputContract::bounded(Arc::new(population.output_schema.clone())), @@ -179,7 +179,7 @@ pub fn compile_current_series_readout( .clone(), ); compile( - &dag, + dag.as_view(), BTreeMap::from([(frontier, InputContract::bounded(schema))]), &[u64::from(dag.root.0)], ) @@ -190,18 +190,16 @@ pub fn compile_current_series_readout( /// the heap is rebuilt independently for each evaluation. This does not move /// that frontier to ingestion time or authorize combining finalized rates. pub fn compile_rate_ranking( - selected: &Rc, + selected: &Rc, ) -> Result< ( - Rc, - CompiledPhysicalDag, + Rc, + PhysicalPostASAPDAG, ), Error, > { - use planner_types::post_asap::{ - compile_post_asap_dag_with_node_ids, ExactKind, SummaryExpr, SummaryNode, - }; - fn frontier(node: &Rc) -> Option> { + use planner_types::post_asap::{index_post_asap_dag, ExactKind, PostASAPNode, SummaryExpr}; + fn frontier(node: &Rc) -> Option> { match &node.expr { SummaryExpr::ValueOperation { child, @@ -211,7 +209,7 @@ pub fn compile_rate_ranking( family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), reduction: planner_types::pre_asap::Reduction::PerEntity, child: raw, .. - } if matches!(&raw.expr, SummaryExpr::KeepPreAsap(expr) if matches!(expr.as_ref(), QueryExpr::TimeRange { .. }))) => + } if matches!(&raw.expr, SummaryExpr::KeepPreAsap(expr) if matches!(expr.as_ref(), PreASAPNode::TimeRange { .. }))) => { Some(Rc::clone(node)) } @@ -232,8 +230,7 @@ pub fn compile_rate_ranking( { return Err(invalid("Rate ranking requires complete series identity")); } - let compiled = compile_post_asap_dag_with_node_ids(selected) - .map_err(|error| invalid(error.to_string()))?; + let compiled = index_post_asap_dag(selected).map_err(|error| invalid(error.to_string()))?; let id = u64::from( compiled .node_ids @@ -242,9 +239,9 @@ pub fn compile_rate_ranking( .0, ); let program = compile( - &compiled.dag, + compiled.view(), BTreeMap::from([(id, InputContract::bounded(Arc::new(source.schema.clone())))]), - &[u64::from(compiled.dag.root.0)], + &[u64::from(compiled.root_id.0)], )?; Ok((source, program)) } @@ -253,7 +250,7 @@ pub fn compile_rate_ranking( /// Rate readouts runs at ingestion time: fresh aggregate state per closed /// window. The input is the complete collection of per-series counter states. pub fn compile_fixed_window_rate_aggregation( - dag: &planner_types::post_asap::PostAsapDag, + dag: &planner_types::post_asap::LogicalPostASAPDAGTransport, ) -> Result { use planner_types::post_asap::{ExactKind, ExecutionTiming, SketchAlgorithm}; let sources = dag diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/asap-physical-operators/src/physical_planner/promql_values.rs index 98032505..45345349 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_values.rs @@ -12,13 +12,13 @@ pub fn matrix_schema() -> Schema { crate::operators::vector_window::matrix_schema() } -pub fn compile_scalar(value: f64) -> Result { +pub fn compile_scalar(value: f64) -> Result { let operator = Operator::scalar( crate::values::Value::Float64(value), planner_types::pre_asap::DataType::Float64, )? .with_output_schema(scalar_schema())?; - CompiledPhysicalDag::from_operators( + PhysicalPostASAPDAG::from_operators( BTreeMap::new(), BTreeMap::from([(0, (vec![], operator))]), vec![0], @@ -28,7 +28,7 @@ pub fn compile_scalar(value: f64) -> Result { pub fn compile_temporal( intent: &AggIntent, preserve_metric_name: bool, -) -> Result { +) -> Result { let operator = Operator::range_window(intent.clone())?; let mut operators = vec![operator]; if !preserve_metric_name { @@ -50,8 +50,8 @@ pub fn compile_temporal( unary(operators, matrix_schema()) } -pub fn compile_histogram_quantile() -> Result { - CompiledPhysicalDag::from_operators( +pub fn compile_histogram_quantile() -> Result { + PhysicalPostASAPDAG::from_operators( BTreeMap::from([ (0, InputContract::bounded(scalar_schema())), (1, InputContract::bounded(vector_schema())), @@ -67,11 +67,11 @@ pub fn compile_binary( return_bool: bool, left_scalar: bool, right_scalar: bool, -) -> Result { +) -> Result { let left = crate::operators::vector_binary::value_schema(left_scalar); let right = crate::operators::vector_binary::value_schema(right_scalar); let op = Operator::vector_binary(left.clone(), right.clone(), operator.clone(), return_bool)?; - CompiledPhysicalDag::from_operators( + PhysicalPostASAPDAG::from_operators( BTreeMap::from([ (0, InputContract::bounded(left)), (1, InputContract::bounded(right)), @@ -81,9 +81,9 @@ pub fn compile_binary( ) } -fn unary(operators: Vec, input: Schema) -> Result { +fn unary(operators: Vec, input: Schema) -> Result { let root = operators.len() as u64; - CompiledPhysicalDag::from_operators( + PhysicalPostASAPDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(input))]), operators .into_iter() @@ -134,7 +134,7 @@ fn vector_output(input: Schema, labels: usize, value: usize) -> Result, grouping: &GroupKeys, -) -> Result { +) -> Result { let project = grouped(grouping)?; let reduction = match intent { AggIntent::Sum { .. } => Reduction::Sum(1), @@ -153,7 +153,7 @@ pub fn compile_aggregate( pub fn compile_sort( descending: bool, grouping: &GroupKeys, -) -> Result { +) -> Result { let project = grouped(grouping)?; let sort = Operator::sort( project.schema(), @@ -172,14 +172,14 @@ pub fn compile_limit( n: u64, offset: u64, grouping: &GroupKeys, -) -> Result { +) -> Result { let project = grouped(grouping)?; let limit = Operator::limit(project.schema(), n, offset, vec![2])?; let output = vector_output(limit.schema(), 0, 1)?; unary(vec![project, limit, output], vector_schema()) } -pub fn compile_negate(scalar: bool) -> Result { +pub fn compile_negate(scalar: bool) -> Result { let input = if scalar { scalar_schema() } else { @@ -200,7 +200,7 @@ pub fn compile_negate(scalar: bool) -> Result { unary(vec![Operator::project(input.clone(), columns)?], input) } -pub fn compile_vector_to_scalar() -> Result { +pub fn compile_vector_to_scalar() -> Result { unary( vec![Operator::vector_to_scalar(vector_schema(), 1)?.with_output_schema(scalar_schema())?], vector_schema(), @@ -224,7 +224,7 @@ pub fn compile_exact_readout( family: SummaryFamilyType, lookback_ms: u64, preserve_metric_name: bool, -) -> Result { +) -> Result { use planner_types::post_asap::ExactKind; let statistic = match &family { SummaryFamilyType::ExactAggregate(kind, _) => match kind { diff --git a/crates/asap-physical-operators/src/physical_planner/row_values.rs b/crates/asap-physical-operators/src/physical_planner/row_values.rs index 8763437b..a248b1bc 100644 --- a/crates/asap-physical-operators/src/physical_planner/row_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/row_values.rs @@ -4,10 +4,10 @@ use planner_types::post_asap::maintained_population::PopulationReadout; use planner_types::pre_asap::{DataType, ScalarValue}; /// A PromQL number literal has no row schema; its consumer folds it in. -pub(super) fn scalar_literal(expression: &QueryExpr) -> Option { +pub(super) fn scalar_literal(expression: &PreASAPNode) -> Option { match expression { - QueryExpr::PromqlScalarBridge(child) => scalar_literal(child), - QueryExpr::Literal(ScalarValue::Float64(value)) => Some(*value), + PreASAPNode::PromqlScalarBridge(child) => scalar_literal(child), + PreASAPNode::Literal(ScalarValue::Float64(value)) => Some(*value), _ => None, } } diff --git a/crates/asap-physical-operators/src/plan/mod.rs b/crates/asap-physical-operators/src/plan/mod.rs index 0dce4dd8..d3f6694a 100644 --- a/crates/asap-physical-operators/src/plan/mod.rs +++ b/crates/asap-physical-operators/src/plan/mod.rs @@ -42,17 +42,25 @@ pub(crate) struct Node<'a, V, S> { pub(crate) inputs: Vec, pub(crate) operator: Box + 'a>, } -pub struct PhysicalDag<'a, V, S> { +/// Execution handle: a graph of runnable operators with their inputs connected. +/// [`PhysicalPostASAPDAG::instantiate`](crate::physical_planner::PhysicalPostASAPDAG::instantiate) +/// derives one from a compiled `PhysicalPostASAPDAG` by binding runtime sources, as do +/// the [`bind`](crate::physical_planner::bind) and +/// [`bind_with_data_sources`](crate::physical_planner::bind_with_data_sources) +/// conveniences. It holds those bound sources, so it lives no longer than they +/// do; each [`Self::execute`] call is a separate run. It is never persisted or +/// compared as a plan. +pub struct PhysicalExecution<'a, V, S> { pub(crate) nodes: BTreeMap>, } -impl Default for PhysicalDag<'_, V, S> { +impl Default for PhysicalExecution<'_, V, S> { fn default() -> Self { Self { nodes: BTreeMap::new(), } } } -impl<'a, V: 'a, S: Clone + PartialEq + Debug + 'a> PhysicalDag<'a, V, S> { +impl<'a, V: 'a, S: Clone + PartialEq + Debug + 'a> PhysicalExecution<'a, V, S> { pub fn add( &mut self, id: NodeId, @@ -79,7 +87,7 @@ impl<'a, V: 'a, S: Clone + PartialEq + Debug + 'a> PhysicalDag<'a, V, S> { /// Derive properties while checking topology and schemas, before starting sources. pub fn properties(&self, roots: &[NodeId]) -> Result, Error> { fn visit( - dag: &PhysicalDag<'_, V, S>, + dag: &PhysicalExecution<'_, V, S>, id: NodeId, active: &mut BTreeSet, done: &mut BTreeMap, diff --git a/crates/asap-physical-operators/src/runtime/batch_execution.rs b/crates/asap-physical-operators/src/runtime/batch_execution.rs index f6300440..32f97613 100644 --- a/crates/asap-physical-operators/src/runtime/batch_execution.rs +++ b/crates/asap-physical-operators/src/runtime/batch_execution.rs @@ -2,7 +2,7 @@ //! bridge for deployments whose boundary values are not yet streaming batches. use crate::{ operators::Operator, - plan::PhysicalDag, + plan::PhysicalExecution, runtime::{RunContext, SharedValue}, values::Batch, Error, @@ -17,7 +17,7 @@ pub fn evaluate_batch( operators: Vec, context: RunContext, ) -> Result>, Error> { - let mut graph = PhysicalDag::default(); + let mut graph = PhysicalExecution::default(); graph.add( 0, vec![], @@ -37,7 +37,7 @@ pub fn evaluate_inputs( operator: Operator, context: RunContext, ) -> Result>, Error> { - let mut graph = PhysicalDag::default(); + let mut graph = PhysicalExecution::default(); let root = inputs.len() as u64; for (id, input) in inputs.into_iter().enumerate() { graph.add( @@ -55,13 +55,13 @@ pub fn evaluate_source( source: Operator, context: RunContext, ) -> Result>, Error> { - let mut graph = PhysicalDag::default(); + let mut graph = PhysicalExecution::default(); graph.add(0, vec![], source)?; evaluate_graph(graph, 0, context) } fn evaluate_graph( - graph: PhysicalDag<'_, Batch, crate::values::Schema>, + graph: PhysicalExecution<'_, Batch, crate::values::Schema>, root: crate::plan::NodeId, context: RunContext, ) -> Result>, Error> { diff --git a/crates/asap-physical-operators/src/runtime/mod.rs b/crates/asap-physical-operators/src/runtime/mod.rs index f72a57ef..afa02893 100644 --- a/crates/asap-physical-operators/src/runtime/mod.rs +++ b/crates/asap-physical-operators/src/runtime/mod.rs @@ -1,6 +1,6 @@ //! Per-run producer sharing, streams, backpressure and resource ownership. use crate::{ - plan::{NodeId, PhysicalDag}, + plan::{NodeId, PhysicalExecution}, Error, }; use futures::{stream::LocalBoxStream, Stream}; @@ -42,7 +42,7 @@ impl SharedValue { } pub(crate) fn execute<'r, V: 'r, S: Clone + PartialEq + Debug + 'r>( - dag: &'r PhysicalDag<'_, V, S>, + dag: &'r PhysicalExecution<'_, V, S>, roots: &[NodeId], context: RunContext, ) -> Result>, Error> { @@ -60,7 +60,7 @@ pub(crate) fn execute<'r, V: 'r, S: Clone + PartialEq + Debug + 'r>( } } fn build<'r, V: 'r, S: 'r>( - dag: &'r PhysicalDag<'_, V, S>, + dag: &'r PhysicalExecution<'_, V, S>, id: NodeId, context: &RunContext, states: &mut BTreeMap>>>, diff --git a/crates/asap-physical-operators/src/runtime/tests.rs b/crates/asap-physical-operators/src/runtime/tests.rs index f683041c..a7a5fb33 100644 --- a/crates/asap-physical-operators/src/runtime/tests.rs +++ b/crates/asap-physical-operators/src/runtime/tests.rs @@ -93,7 +93,7 @@ fn source(fail: bool) -> (Source, Rc>, Rc>) { #[test] fn shared_source_backpressure_and_reader_drop() { let (source, starts, polls) = source(false); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], source).unwrap(); let context = context(); let mut readers = dag.execute(&[0, 0], context.clone()).unwrap(); @@ -124,7 +124,7 @@ fn shared_source_backpressure_and_reader_drop() { #[test] fn diamond_and_run_isolation() { let (source, starts, polls) = source(false); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], source).unwrap(); dag.add(1, vec![0], Identity).unwrap(); dag.add(2, vec![0], Identity).unwrap(); @@ -151,7 +151,7 @@ fn diamond_and_run_isolation() { #[test] fn broadcast_error_and_cancel() { let (source, _, polls) = source(true); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], source).unwrap(); let mut outputs = dag.execute(&[0, 0], context()).unwrap(); let a = outputs.pop().unwrap(); @@ -177,7 +177,7 @@ fn broadcast_error_and_cancel() { #[test] fn retained_outputs_count_against_budget() { let (source, _, _) = source(false); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], source).unwrap(); let run = RunContext::new( Scope::Query { @@ -207,16 +207,16 @@ fn retained_outputs_count_against_budget() { #[test] fn invalid_graphs_do_not_start_sources() { let (source, starts, _) = source(false); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], source).unwrap(); dag.add(1, vec![2], Identity).unwrap(); dag.add(2, vec![1], Identity).unwrap(); assert!(dag.execute(&[0, 1], context()).is_err()); assert_eq!(starts.get(), 0); - let mut missing = PhysicalDag::default(); + let mut missing = PhysicalExecution::default(); missing.add(1, vec![9], Identity).unwrap(); assert!(missing.validate(&[1]).is_err()); - let mut arity = PhysicalDag::default(); + let mut arity = PhysicalExecution::default(); arity.add(1, vec![], Identity).unwrap(); assert!(arity.validate(&[1]).is_err()); } @@ -226,7 +226,7 @@ fn invalid_graphs_do_not_start_sources() { fn ready_sources_cooperate_with_cancellation() { let (mut source, _, polls) = source(false); source.end = 10_000; - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], source).unwrap(); let context = context(); let mut input = dag.execute(&[0], context.clone()).unwrap().remove(0); @@ -253,7 +253,7 @@ fn ready_sources_cooperate_with_cancellation() { #[test] fn depth_limit_covers_shared_paths() { let (source, _, _) = source(false); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], source).unwrap(); for id in 1..129 { dag.add(id, vec![id - 1], Identity).unwrap(); diff --git a/crates/asap-physical-operators/src/sources/mod.rs b/crates/asap-physical-operators/src/sources/mod.rs index 4da778b0..04c46fe8 100644 --- a/crates/asap-physical-operators/src/sources/mod.rs +++ b/crates/asap-physical-operators/src/sources/mod.rs @@ -9,7 +9,7 @@ use crate::{ use futures::{stream, StreamExt}; use planner_types::{ post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, - pre_asap::{DataType, QueryExpr, Source}, + pre_asap::{DataType, PreASAPNode, Source}, }; use std::sync::Arc; @@ -40,8 +40,8 @@ impl DataSources { self.sources.push((identity, source)); Ok(()) } - pub fn bind(&self, expression: &QueryExpr) -> Result { - let QueryExpr::Scan { + pub fn bind(&self, expression: &PreASAPNode) -> Result { + let PreASAPNode::Scan { source, predicates, schema, diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/asap-physical-operators/tests/blocking_resources.rs index 7a312892..5c608643 100644 --- a/crates/asap-physical-operators/tests/blocking_resources.rs +++ b/crates/asap-physical-operators/tests/blocking_resources.rs @@ -1,7 +1,7 @@ //! Blocking operators enforce resources before returning their first batch. use asap_physical_operators::{ operators::Operator, - plan::{PhysicalDag, PhysicalOperator}, + plan::{PhysicalExecution, PhysicalOperator}, runtime::{Limits, RunContext, Scope}, values::{Batch, Schema, Value}, Error, @@ -9,7 +9,7 @@ use asap_physical_operators::{ use futures::{executor::block_on, FutureExt, StreamExt}; use planner_types::{ post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, - pre_asap::{DataType, JoinKind, Predicate, QueryExpr, ScalarValue}, + pre_asap::{DataType, JoinKind, PreASAPNode, Predicate, ScalarValue}, }; use std::sync::Arc; @@ -38,8 +38,8 @@ fn context(max_bytes: usize) -> RunContext { ) .unwrap() } -fn source(n: usize) -> PhysicalDag<'static, Batch, Schema> { - let mut dag = PhysicalDag::default(); +fn source(n: usize) -> PhysicalExecution<'static, Batch, Schema> { + let mut dag = PhysicalExecution::default(); dag.add( 0, vec![], @@ -57,9 +57,9 @@ fn cross_join() -> Operator { schema(1), schema(1), JoinKind::Cross, - &Predicate(std::rc::Rc::new(QueryExpr::Literal(ScalarValue::Boolean( - true, - )))), + &Predicate(std::rc::Rc::new(PreASAPNode::Literal( + ScalarValue::Boolean(true), + ))), schema(2), ) .unwrap() @@ -150,7 +150,7 @@ fn cooperative_sort_preserves_ties_across_chunks() { .collect(), ) .unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], Operator::source(schema(2), vec![batch]).unwrap()) .unwrap(); dag.add( @@ -201,7 +201,7 @@ fn weighted_summary_build_yields_within_a_batch() { ], time_index: None, }); - let mut sources = PhysicalDag::default(); + let mut sources = PhysicalExecution::default(); let batch = Batch::try_new( input.clone(), (0..1500) diff --git a/crates/asap-physical-operators/tests/candidate_post_asap_dags_series_identity_heap.rs b/crates/asap-physical-operators/tests/candidate_post_asap_dags_series_identity_heap.rs index 2f8d5249..1a6ea2cb 100644 --- a/crates/asap-physical-operators/tests/candidate_post_asap_dags_series_identity_heap.rs +++ b/crates/asap-physical-operators/tests/candidate_post_asap_dags_series_identity_heap.rs @@ -13,7 +13,7 @@ use asap_physical_operators::physical_planner::promql_rows::{ }; use planner_types::{ post_asap::*, - pre_asap::QueryExpr, + pre_asap::PreASAPNode, types::AccuracyTarget, workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence as WorkloadEvidence, @@ -25,7 +25,7 @@ use std::rc::Rc; struct Evidence; impl AccuracyEvidenceProvider for Evidence { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &PreASAPNode) -> Option { Some(1000) } fn propagation_stats( @@ -64,7 +64,7 @@ impl ReplacementStrategy for LogicalOnly { } } -fn lower(query: &str, accuracy: &AccuracyTarget) -> Rc { +fn lower(query: &str, accuracy: &AccuracyTarget) -> Rc { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -96,7 +96,7 @@ fn lower(query: &str, accuracy: &AccuracyTarget) -> Rc { ) } -type Dag = Vec<(usize, Rc)>; +type Dag = Vec<(usize, Rc)>; /// Candidate DAGs for query 1 of a two-query workload, with and without /// whole-root proposals. Query 0 is a bystander that must not multiply them. @@ -126,7 +126,7 @@ fn inventories(query: &str, accuracy: AccuracyTarget) -> (Vec, Vec) { fn carries_identity(dag: &Dag) -> bool { dag.iter().any(|(_, root)| { - compile_post_asap_dag(root) + export_post_asap_dag(root) .unwrap() .nodes .iter() @@ -140,7 +140,7 @@ fn carries_identity(dag: &Dag) -> bool { } /// Shared acceptance checks; returns the added identity-carrying alternatives. -fn added_alternatives(query: &str, accuracy: AccuracyTarget) -> Vec> { +fn added_alternatives(query: &str, accuracy: AccuracyTarget) -> Vec> { let (full, logical) = inventories(query, accuracy); for (index, dag) in full.iter().enumerate() { assert_eq!(dag.len(), 1, "one root per candidate, no workload product"); @@ -213,7 +213,7 @@ fn global_selection_never_commits_a_series_identity_heap() { ); let selected = space .global_selection(&DefaultCostModel) - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .unwrap() .unwrap(); assert!(!carries_identity(&vec![(0, selected)])); @@ -236,7 +236,7 @@ fn repeated_roots_do_not_duplicate_alternatives() { .collect(); let space = search_workload_with_targets(roots, &strategies, &DefaultAccuracyModel); space - .candidates_for_target(&space.roots[0].1) + .candidates_for_target(&space.roots()[0].1) .unwrap() .candidates .iter() diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs index 322bd6fe..24d47b25 100644 --- a/crates/asap-physical-operators/tests/current_series_heap.rs +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -3,7 +3,7 @@ use asap_physical_operators::{ operators::Operator, physical_planner::{ promql_rows::{decode_series_identity, series_row, SERIES_IDENTITY_COLUMN}, - CompiledPhysicalDag, InputContract, Source, + InputContract, PhysicalPostASAPDAG, Source, }, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, @@ -30,8 +30,8 @@ fn schema() -> Arc { time_index: Some(0), }) } -fn run(program: &CompiledPhysicalDag, data: Batch, end: i64) -> Result, String> { - let recovered = serde_json::from_slice::( +fn run(program: &PhysicalPostASAPDAG, data: Batch, end: i64) -> Result, String> { + let recovered = serde_json::from_slice::( &serde_json::to_vec(&program).map_err(|e| e.to_string())?, ) .map_err(|e| e.to_string())?; @@ -78,8 +78,8 @@ fn input(samples: &[(&str, i64, f64)]) -> Batch { .collect(); Batch::try_new(schema, rows).unwrap() } -fn snapshot_plan() -> CompiledPhysicalDag { - CompiledPhysicalDag::from_operators( +fn snapshot_plan() -> PhysicalPostASAPDAG { + PhysicalPostASAPDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(schema()))]), BTreeMap::from([( 1, @@ -175,7 +175,7 @@ fn spatial_heap_ranks_latest_values_in_independent_runs() { time_index: None, }); let read = Operator::keyed_readout(build.schema(), 1, 1, output).unwrap(); - let plan = CompiledPhysicalDag::from_operators( + let plan = PhysicalPostASAPDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(schema()))]), BTreeMap::from([ ( @@ -343,15 +343,15 @@ fn planner_current_series_candidate_compiles_with_dynamic_identity() { ) .candidate(&root) .unwrap(); - let logical = compile_post_asap_dag(&selected).unwrap(); + let logical = export_post_asap_dag(&selected).unwrap(); let raw = logical .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| matches!(node.payload, PostASAPOperatorPayload::Fallback { .. })) .unwrap(); let raw_schema = Arc::new(raw.output_schema.clone()); let physical = compile( - &logical, + logical.as_view(), BTreeMap::from([( u64::from(raw.id.0), InputContract::bounded(raw_schema.clone()), diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index 3a92e4d9..d3de2b35 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -2,19 +2,19 @@ //! the deployment supplies only raw rows at the ingestion frontier. use asap_physical_operators::{ operators::Operator, - physical_planner::{compile, promql_rows, CompiledPhysicalDag, InputContract, Source}, + physical_planner::{compile, promql_rows, InputContract, PhysicalPostASAPDAG, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; -use planner_types::{post_asap::*, pre_asap::QueryExpr, types::AccuracyTarget, workload::*}; +use planner_types::{post_asap::*, pre_asap::PreASAPNode, types::AccuracyTarget, workload::*}; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; -fn lower(query: &str) -> QueryExpr { +fn lower(query: &str) -> PreASAPNode { lower_with(query, AccuracyTarget::Exact) } -fn lower_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { +fn lower_with(query: &str, accuracy: AccuracyTarget) -> PreASAPNode { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -45,7 +45,7 @@ fn lower_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { } /// The first exact summary candidate, as Planner selection would hand it over. -fn exact_dag(query: &str) -> PostAsapDag { +fn exact_dag(query: &str) -> LogicalPostASAPDAGTransport { use asap_aware_mapping::{Replacement, ReplacementStrategy, TargetSubDAG}; let expression = lower(query); let root = Rc::new(promql_rows::with_series_identity(&expression).unwrap_or(expression)); @@ -54,10 +54,10 @@ fn exact_dag(query: &str) -> PostAsapDag { .into_iter() .find_map(|candidate| match candidate.replacement { Replacement::Summary(node) => { - let dag = compile_post_asap_dag(&node).ok()?; + let dag = export_post_asap_dag(&node).ok()?; dag.nodes .iter() - .all(|n| !matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { family, .. } if !matches!(family, SummaryFamilyType::ExactAggregate(..)))) + .all(|n| !matches!(&n.payload, PostASAPOperatorPayload::SummaryAgg { family, .. } if !matches!(family, SummaryFamilyType::ExactAggregate(..)))) .then_some(dag) } _ => None, @@ -65,25 +65,25 @@ fn exact_dag(query: &str) -> PostAsapDag { .unwrap() } -fn population_dag(query: &str) -> PostAsapDag { +fn population_dag(query: &str) -> LogicalPostASAPDAGTransport { let root = Rc::new(promql_rows::with_series_identity(&lower(query)).unwrap()); let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( std::slice::from_ref(&root), ) .candidate(&root) .unwrap(); - compile_post_asap_dag(&selected).unwrap() + export_post_asap_dag(&selected).unwrap() } /// Raw scan nodes are the frontier; everything above them is compiled. -fn raw_inputs(dag: &PostAsapDag) -> Vec<(u64, Arc, String)> { +fn raw_inputs(dag: &LogicalPostASAPDAGTransport) -> Vec<(u64, Arc, String)> { dag.nodes .iter() .filter_map(|node| match &node.payload { - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::TimeRange { child, .. }, + PostASAPOperatorPayload::Fallback { + expression: PreASAPNode::TimeRange { child, .. }, } => match child.as_ref() { - QueryExpr::Scan { + PreASAPNode::Scan { source: planner_types::pre_asap::Source::TimeSeries { metric }, .. } => Some(( @@ -103,7 +103,7 @@ type Sample = (&'static str, &'static str, &'static str, i64, f64); /// Compile, round-trip, bind raw `(metric, job, instance, ts, value)` samples, /// and return the root's batches. fn execute( - dag: &PostAsapDag, + dag: &LogicalPostASAPDAGTransport, samples: &[Sample], end: i64, ) -> Result>, String> { @@ -113,14 +113,14 @@ fn execute( /// [`execute`], supplying samples of each instance in `relabel` under its /// `(__name__, instance)` instead. fn execute_relabeled( - dag: &PostAsapDag, + dag: &LogicalPostASAPDAGTransport, samples: &[Sample], end: i64, relabel: &BTreeMap<&str, (&str, &str)>, ) -> Result>, String> { let inputs = raw_inputs(dag); let program = compile( - dag, + dag.as_view(), inputs .iter() .map(|(id, schema, _)| (*id, InputContract::bounded(schema.clone()))) @@ -128,7 +128,7 @@ fn execute_relabeled( &[u64::from(dag.root.0)], ) .map_err(|e| e.to_string())?; - let program: CompiledPhysicalDag = + let program: PhysicalPostASAPDAG = serde_json::from_slice(&serde_json::to_vec(&program).unwrap()).unwrap(); let sources = inputs .iter() @@ -194,7 +194,11 @@ fn execute_relabeled( } /// [`execute`], returning `(job, value)` rows of the root. -fn run(dag: &PostAsapDag, samples: &[Sample], end: i64) -> Result, String> { +fn run( + dag: &LogicalPostASAPDAGTransport, + samples: &[Sample], + end: i64, +) -> Result, String> { let mut rows = BTreeMap::new(); for batch in execute(dag, samples, end)? { let job = batch.schema().fields.iter().position(|f| f.name == "job"); @@ -319,7 +323,7 @@ fn exact_count_finalizes_to_declared_float_value() { .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::FinalizeExactAccumulator } ) @@ -335,7 +339,7 @@ fn exact_count_finalizes_to_declared_float_value() { .clone(); // Read the rolled-up exact state the same way the query path does. let mut read = finalize.clone(); - read.id = PostAsapNodeId(root.id.0 + 1); + read.id = PostASAPNodeId(root.id.0 + 1); read.output_schema = root.output_schema.clone(); read.output_schema.fields.last_mut().unwrap().dtype = SummaryFamilyType::Plain(planner_types::pre_asap::DataType::Float64); @@ -353,9 +357,12 @@ fn exact_count_finalizes_to_declared_float_value() { } /// `dag` with its Binary operator replaced by `kind`. -fn with_kind(mut dag: PostAsapDag, kind: planner_types::pre_asap::BinaryOpKind) -> PostAsapDag { +fn with_kind( + mut dag: LogicalPostASAPDAGTransport, + kind: planner_types::pre_asap::BinaryOpKind, +) -> LogicalPostASAPDAGTransport { for node in &mut dag.nodes { - if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { + if let PostASAPOperatorPayload::Binary { operator } = &mut node.payload { operator.kind = kind.clone(); } } @@ -409,7 +416,11 @@ fn per_series_comparisons_filter_or_return_bool() { /// [`execute`], returning per-series `(identity, value)` rows of the root, /// with NaN-aware formatting for comparison. -fn run_series(dag: &PostAsapDag, samples: &[Sample], end: i64) -> Result { +fn run_series( + dag: &LogicalPostASAPDAGTransport, + samples: &[Sample], + end: i64, +) -> Result { let mut rows = BTreeMap::new(); for batch in execute(dag, samples, end)? { let schema = batch.schema(); @@ -544,12 +555,12 @@ fn per_series_scalar_arithmetic_rejects_label_sets_equal_without_the_name() { } fn with_vector_match( - mut dag: PostAsapDag, + mut dag: LogicalPostASAPDAGTransport, kind: planner_types::pre_asap::VectorMatchKind, labels: &[&str], -) -> PostAsapDag { +) -> LogicalPostASAPDAGTransport { for node in &mut dag.nodes { - if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { + if let PostASAPOperatorPayload::Binary { operator } = &mut node.payload { operator.vector_match = Some(planner_types::pre_asap::VectorMatch { kind: kind.clone(), labels: labels.iter().map(|l| l.to_string()).collect(), @@ -633,17 +644,17 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { .into_iter() .find_map(|candidate| match candidate.replacement { Replacement::Summary(node) => { - let dag = compile_post_asap_dag(&node).ok()?; + let dag = export_post_asap_dag(&node).ok()?; let bare_count = dag.nodes.iter().any(|n| { matches!( &n.payload, - PostAsapOperatorPayload::SummaryEstimate { + PostASAPOperatorPayload::SummaryEstimate { query: SketchQuery::PointCount { value: None, .. } } ) }); let count_min = dag.nodes.iter().any(|n| { - matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { + matches!(&n.payload, PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::Cms) }); @@ -655,9 +666,9 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { let state = dag .nodes .iter() - .find(|n| matches!(n.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .find(|n| matches!(n.payload, PostASAPOperatorPayload::SummaryAgg { .. })) .unwrap(); - let PostAsapOperatorPayload::SummaryAgg { + let PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } = &state.payload @@ -669,7 +680,7 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { }; let schema = Arc::new(state.output_schema.clone()); let program = compile( - &dag, + dag.as_view(), BTreeMap::from([( u64::from(state.id.0), InputContract::bounded(schema.clone()), @@ -677,7 +688,7 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { &[u64::from(dag.root.0)], ) .unwrap(); - let program: CompiledPhysicalDag = + let program: PhysicalPostASAPDAG = serde_json::from_slice(&serde_json::to_vec(&program).unwrap()).unwrap(); let mut sketch = CountMinSketchAccumulator::new(*depth as usize, *width as usize); sketch.inner.update("a", 3.0); diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 652880e0..e4af6620 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -3,7 +3,7 @@ use asap_physical_operators::{ dag::{ operators::{Expression, Operator, Reduction, SortKey}, values::{Batch, Schema, Value}, - Limits, PhysicalDag, RunContext, Scope, + Limits, PhysicalExecution, RunContext, Scope, }, Statistic, }; @@ -26,7 +26,7 @@ fn schema(fields: &[(&str, DataType, bool)]) -> Schema { time_index: None, }) } -fn run(dag: &PhysicalDag<'_, Batch, Schema>, root: u64, scope: Scope) -> Vec> { +fn run(dag: &PhysicalExecution<'_, Batch, Schema>, root: u64, scope: Scope) -> Vec> { let context = RunContext::new( scope, Limits { @@ -85,7 +85,7 @@ fn grouped_sort_limit_across_batches() { .unwrap() }) .collect(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add( 0, vec![], @@ -122,7 +122,7 @@ fn summary_construction_merge_and_readout_at_both_phases() { let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let build = Operator::summary_build(schema.clone(), family, 0, None, vec![]).unwrap(); let state = build.schema(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], Operator::source(schema, batches).unwrap()) .unwrap(); dag.add(1, vec![0], build).unwrap(); @@ -181,7 +181,7 @@ fn diamond_semijoin_preserves_left_values_and_multiplicity() { ), ) .unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add( 0, vec![], @@ -210,7 +210,7 @@ fn exact_integer_and_empty_extrema() { vec![("sum".into(), Reduction::Sum(0))], ) .unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); let value = 9_007_199_254_740_993; dag.add( 0, @@ -228,7 +228,7 @@ fn exact_integer_and_empty_extrema() { .unwrap(); dag.add(1, vec![0], aggregate).unwrap(); assert!(matches!(run(&dag,1,query())[0][0],Value::Int64(v) if v==value+2)); - let mut empty = PhysicalDag::default(); + let mut empty = PhysicalExecution::default(); empty .add(0, vec![], Operator::source(schema.clone(), vec![]).unwrap()) .unwrap(); @@ -255,7 +255,7 @@ fn scalar_negation_and_vector_conversion() { ) .unwrap(); let convert = Operator::vector_to_scalar(project.schema(), 0).unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], scalar).unwrap(); dag.add(1, vec![0], project).unwrap(); dag.add(2, vec![1], convert).unwrap(); @@ -266,7 +266,7 @@ fn scalar_negation_and_vector_conversion() { Box::new(Expression::Column(0)), ); let filter = Operator::filter(scalar.schema(), predicate).unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], scalar).unwrap(); dag.add(1, vec![0], filter).unwrap(); assert!(run(&dag, 1, query()).is_empty()); @@ -315,7 +315,7 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { let build = Operator::summary_build(input.clone(), family, 0, None, vec![]).unwrap(); let state = build.schema(); let build_range = |start: u32, end: u32| { - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); let batch = Batch::try_new( input.clone(), (start..end) @@ -343,7 +343,7 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { let prefix = build_range(0, 64); let complete = build_range(0, 128); let query_plan = |stored: Option>>, raw_start: Option| { - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); let mut states = vec![]; if let Some(rows) = stored { dag.add( @@ -432,7 +432,7 @@ fn exact_state_and_family_validation() { family: family.clone(), state: Arc::new(acc), }; - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add( 0, vec![], @@ -481,39 +481,39 @@ fn bind_post_asap_before_execution() { use asap_physical_operators::dag::planner::bind; use planner_types::{ post_asap::{ - EdgeRole, ExecutionDataState, GroupingEdgeCompatibility, PostAsapDag, PostAsapDagEdge, - PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, ValueOperation, - WindowEdgeCompatibility, + EdgeRole, ExecutionDataState, GroupingEdgeCompatibility, LogicalPostASAPDAGEdge, + LogicalPostASAPDAGNode, LogicalPostASAPDAGTransport, PostASAPNodeId, + PostASAPOperatorPayload, ValueOperation, WindowEdgeCompatibility, }, - pre_asap::{ArithmeticOpKind, ProjectItem, QueryExpr, ScalarValue}, + pre_asap::{ArithmeticOpKind, PreASAPNode, ProjectItem, ScalarValue}, }; use std::{collections::BTreeMap, rc::Rc}; let schema = schema(&[("value", DataType::Float64, false)]); - let node = |id, payload| PostAsapDagNode { - id: PostAsapNodeId(id), + let node = |id, payload| LogicalPostASAPDAGNode { + id: PostASAPNodeId(id), payload, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*schema).clone(), guarantee: None, }; - let mut dag = PostAsapDag { + let mut dag = LogicalPostASAPDAGTransport { nodes: vec![ node( 0, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(1.), + PostASAPOperatorPayload::Fallback { + expression: PreASAPNode::promql_scalar(1.), }, ), node( 1, - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::Project { cols: vec![ProjectItem { alias: None, - expr: QueryExpr::Arithmetic { + expr: PreASAPNode::Arithmetic { op: ArithmeticOpKind::Add, - left: Rc::new(QueryExpr::Column(0)), - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.))), + left: Rc::new(PreASAPNode::Column(0)), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Float64(2.))), }, }], qualifier: None, @@ -521,16 +521,16 @@ fn bind_post_asap_before_execution() { }, ), ], - edges: vec![PostAsapDagEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), + edges: vec![LogicalPostASAPDAGEdge { + producer: PostASAPNodeId(0), + consumer: PostASAPNodeId(1), role: EdgeRole::Input, intermediate_schema: (*schema).clone(), data_state: ExecutionDataState::QUERY_ROWS, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }], - root: PostAsapNodeId(1), + root: PostASAPNodeId(1), }; let sources = || -> BTreeMap> { BTreeMap::from([( @@ -549,7 +549,7 @@ fn bind_post_asap_before_execution() { // A literal Fallback needs no deployment input. let literal = bind(&dag, BTreeMap::new(), &[1]).unwrap(); assert_eq!(floats(&run(&literal, 1, query()), 0), vec![3.]); - dag.nodes[1].payload = PostAsapOperatorPayload::Value { + dag.nodes[1].payload = PostASAPOperatorPayload::Value { operation: ValueOperation::Extension { name: "unknown".into(), }, @@ -580,7 +580,7 @@ fn empty_exact_count_is_an_integer_state_readout() { ), ) .unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], Operator::source(input, vec![]).unwrap()) .unwrap(); dag.add(1, vec![0], build).unwrap(); @@ -594,10 +594,10 @@ fn source_batches_must_match_the_bound_schema() { use asap_physical_operators::dag::{self, PhysicalOperator}; use planner_types::{ post_asap::{ - ExecutionDataState, PostAsapDag, PostAsapDagNode, PostAsapNodeId, - PostAsapOperatorPayload, + ExecutionDataState, LogicalPostASAPDAGNode, LogicalPostASAPDAGTransport, + PostASAPNodeId, PostASAPOperatorPayload, }, - pre_asap::QueryExpr, + pre_asap::PreASAPNode, }; use std::{cell::Cell, collections::BTreeMap, rc::Rc}; struct WrongSource { @@ -631,18 +631,18 @@ fn source_batches_must_match_the_bound_schema() { } let expected = schema(&[("value", DataType::Float64, false)]); let starts = Rc::new(Cell::new(0)); - let plan = PostAsapDag { - nodes: vec![PostAsapDagNode { - id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(1.), + let plan = LogicalPostASAPDAGTransport { + nodes: vec![LogicalPostASAPDAGNode { + id: PostASAPNodeId(0), + payload: PostASAPOperatorPayload::Fallback { + expression: PreASAPNode::promql_scalar(1.), }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*expected).clone(), guarantee: None, }], edges: vec![], - root: PostAsapNodeId(0), + root: PostASAPNodeId(0), }; let source = Box::new(WrongSource { schema: expected, @@ -663,7 +663,7 @@ fn source_batches_must_match_the_bound_schema() { #[test] fn extrema_preserve_numeric_values_in_the_presence_of_nan() { let input = schema(&[("v", DataType::Float64, false)]); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add( 0, vec![], @@ -703,7 +703,7 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { use asap_physical_operators::dag::planner::{bind, Source}; use planner_types::{ post_asap::*, - pre_asap::{CompareOpKind, GroupKeys, JoinKind, Predicate, QueryExpr, SortKey}, + pre_asap::{CompareOpKind, GroupKeys, JoinKind, PreASAPNode, Predicate, SortKey}, }; use std::{collections::BTreeMap, rc::Rc}; let rows_schema = schema(&[ @@ -712,16 +712,16 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { ("score", DataType::Float64, false), ]); let keys_schema = schema(&[("key", DataType::Utf8, false)]); - let node = |id, payload, schema: &Schema| PostAsapDagNode { - id: PostAsapNodeId(id), + let node = |id, payload, schema: &Schema| LogicalPostASAPDAGNode { + id: PostASAPNodeId(id), payload, output_schema: (**schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, guarantee: None, }; - let edge = |producer, consumer, role, schema: &Schema| PostAsapDagEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), + let edge = |producer, consumer, role, schema: &Schema| LogicalPostASAPDAGEdge { + producer: PostASAPNodeId(producer), + consumer: PostASAPNodeId(consumer), role, intermediate_schema: (**schema).clone(), data_state: ExecutionDataState::QUERY_ROWS, @@ -729,41 +729,41 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { window: WindowEdgeCompatibility::NotApplicable, }; let groups = GroupKeys::by(vec![0]); - let dag = PostAsapDag { + let dag = LogicalPostASAPDAGTransport { nodes: vec![ node( 0, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(0.), + PostASAPOperatorPayload::Fallback { + expression: PreASAPNode::promql_scalar(0.), }, &rows_schema, ), node( 1, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(0.), + PostASAPOperatorPayload::Fallback { + expression: PreASAPNode::promql_scalar(0.), }, &keys_schema, ), node( 2, - PostAsapOperatorPayload::RelationalJoin { + PostASAPOperatorPayload::RelationalJoin { join_kind: JoinKind::Semi, pruning: None, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(1)), + pred: Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(1)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(3)), + right: Rc::new(PreASAPNode::Column(3)), })), }, &rows_schema, ), node( 3, - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::Sort { keys: vec![SortKey { - expr: QueryExpr::Column(2), + expr: PreASAPNode::Column(2), ascending: false, nulls_first: false, }], @@ -774,7 +774,7 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { ), node( 4, - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::Limit { n: 1, offset: 0, @@ -791,7 +791,7 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { edge(2, 3, EdgeRole::Input, &rows_schema), edge(3, 4, EdgeRole::Input, &rows_schema), ], - root: PostAsapNodeId(4), + root: PostASAPNodeId(4), }; let text = |v: &str| Value::Utf8(v.into()); for (phase, scope) in [ @@ -854,7 +854,7 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { #[test] fn planner_expressions_preserve_collection_and_nullable_types() { use asap_physical_operators::dag::expressions::CompiledExpression; - use planner_types::pre_asap::{CompareOpKind, QueryExpr, ScalarValue}; + use planner_types::pre_asap::{CompareOpKind, PreASAPNode, ScalarValue}; use std::rc::Rc; let input_schema = schema(&[( "items", @@ -865,11 +865,11 @@ fn planner_expressions_preserve_collection_and_nullable_types() { }, false, )]); - let access = QueryExpr::FunctionCall { + let access = PreASAPNode::FunctionCall { name: "asap_element_access".into(), args: vec![ - QueryExpr::Column(0), - QueryExpr::Literal(ScalarValue::Utf8("count".into())), + PreASAPNode::Column(0), + PreASAPNode::Literal(ScalarValue::Utf8("count".into())), ], }; let project = Operator::project( @@ -880,7 +880,7 @@ fn planner_expressions_preserve_collection_and_nullable_types() { )], ) .unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add( 0, vec![], @@ -902,10 +902,10 @@ fn planner_expressions_preserve_collection_and_nullable_types() { .unwrap(); let projected = project.schema(); dag.add(1, vec![0], project).unwrap(); - let predicate = QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let predicate = PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(0)), op: CompareOpKind::Ge, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Int64(1))), }; dag.add( 2, @@ -919,9 +919,9 @@ fn planner_expressions_preserve_collection_and_nullable_types() { .unwrap(); let rows = run(&dag, 2, query()); assert!(matches!(rows.as_slice(),[row] if matches!(row.as_slice(),[Value::Int64(7)]))); - let unknown = QueryExpr::FunctionCall { + let unknown = PreASAPNode::FunctionCall { name: "unregistered_function".into(), - args: vec![QueryExpr::Column(0)], + args: vec![PreASAPNode::Column(0)], }; assert!(CompiledExpression::compile(&unknown, &input_schema).is_err()); } @@ -929,13 +929,13 @@ fn planner_expressions_preserve_collection_and_nullable_types() { // Outer, semi and anti joins share Planner predicates and preserve SQL null behavior. #[test] fn native_relational_join_kinds_preserve_unmatched_rows() { - use planner_types::pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}; + use planner_types::pre_asap::{CompareOpKind, JoinKind, PreASAPNode, Predicate}; use std::rc::Rc; let input = schema(&[("key", DataType::Int64, true)]); - let predicate = Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let predicate = Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), + right: Rc::new(PreASAPNode::Column(1)), })); for (kind, count) in [ (JoinKind::Inner, 1), @@ -954,7 +954,7 @@ fn native_relational_join_kinds_preserve_unmatched_rows() { ("right", DataType::Int64, true), ]) }; - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); for (id, rows) in [ ( 0, @@ -1077,7 +1077,7 @@ fn assert_weighted_rate_topk(count_sketch: bool) { ("score", DataType::Float64, false), ]); let readout = Operator::keyed_readout(build.schema(), 1, 8, output.clone()).unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add( 0, vec![], @@ -1131,15 +1131,15 @@ fn assert_weighted_rate_topk(count_sketch: bool) { #[test] fn grouped_temporal_schema_compiles_and_executes_topk() { use asap_physical_operators::physical_planner::{ - compile_node, CompiledPhysicalDag, InputContract, Source, + compile_node, InputContract, PhysicalPostASAPDAG, Source, }; use planner_types::post_asap::{ - ExecutionDataState, PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, + ExecutionDataState, LogicalPostASAPDAGNode, PostASAPNodeId, PostASAPOperatorPayload, ValueOperation, }; use planner_types::pre_asap::{ - aggregate_output_schema, AggIntent, Column, GroupKeys, QueryExpr, Reduction as IrReduction, - Schema as IrSchema, + aggregate_output_schema, AggIntent, Column, GroupKeys, PreASAPNode, + Reduction as IrReduction, Schema as IrSchema, }; let grouped = IrSchema::new(vec![ Column::new("job", DataType::Utf8, false), @@ -1159,9 +1159,9 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { .map(|c| (c.name.as_str(), c.dtype.clone(), c.nullable)) .collect::>(), ); - let node = |id, operation| PostAsapDagNode { - id: PostAsapNodeId(id), - payload: PostAsapOperatorPayload::Value { operation }, + let node = |id, operation| LogicalPostASAPDAGNode { + id: PostASAPNodeId(id), + payload: PostASAPOperatorPayload::Value { operation }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*input).clone(), guarantee: None, @@ -1171,7 +1171,7 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { 1, ValueOperation::Sort { keys: vec![planner_types::pre_asap::SortKey { - expr: QueryExpr::Column(1), + expr: PreASAPNode::Column(1), ascending: false, nulls_first: false, }], @@ -1193,14 +1193,14 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { std::slice::from_ref(&input), ) .unwrap(); - let compiled = CompiledPhysicalDag::from_operators( + let compiled = PhysicalPostASAPDAG::from_operators( [(0, InputContract::bounded(input.clone()))].into(), [(1, (vec![0], sort)), (2, (vec![1], limit))].into(), vec![2], ) .unwrap(); let recovered = - serde_json::from_slice::(&serde_json::to_vec(&compiled).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&compiled).unwrap()) .unwrap(); assert_eq!(recovered.row_source(2), Some(0)); assert_eq!(recovered.operator_name(2), Some("Limit")); @@ -1235,26 +1235,26 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { #[test] fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { use asap_physical_operators::physical_planner::{ - compile_node, CompiledPhysicalDag, InputContract, Source, + compile_node, InputContract, PhysicalPostASAPDAG, Source, }; use planner_types::{ post_asap::*, - pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}, + pre_asap::{CompareOpKind, JoinKind, PreASAPNode, Predicate}, }; use std::{collections::BTreeMap, rc::Rc}; let schema = schema(&[("key", DataType::Utf8, false)]); for certified in [false, true] { - let node = PostAsapDagNode { - id: PostAsapNodeId(2), + let node = LogicalPostASAPDAGNode { + id: PostASAPNodeId(2), output_schema: (*schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, guarantee: None, - payload: PostAsapOperatorPayload::RelationalJoin { + payload: PostASAPOperatorPayload::RelationalJoin { join_kind: JoinKind::Semi, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + pred: Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), + right: Rc::new(PreASAPNode::Column(1)), })), pruning: certified.then_some(CandidateCompleteness::Certified { guarantee: ResultGuarantee { @@ -1266,7 +1266,7 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { }), }, }; - let graph = CompiledPhysicalDag::from_operators( + let graph = PhysicalPostASAPDAG::from_operators( [ (0, InputContract::bounded(schema.clone())), (1, InputContract::bounded(schema.clone())), @@ -1284,7 +1284,7 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { ) .unwrap(); let graph = - serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) .unwrap(); assert_eq!( graph.certified_pruning_keys(2), @@ -1348,7 +1348,7 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { #[test] fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { use asap_physical_operators::physical_planner::{ - compile_node, CompiledPhysicalDag, InputContract, Source, + compile_node, InputContract, PhysicalPostASAPDAG, Source, }; use planner_types::{ post_asap::*, @@ -1360,12 +1360,12 @@ fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { ("time", DataType::Timestamp, false), ("value", DataType::Float64, false), ]); - let node = PostAsapDagNode { - id: PostAsapNodeId(2), + let node = LogicalPostASAPDAGNode { + id: PostASAPNodeId(2), output_schema: (*input).clone(), output_state: ExecutionDataState::INGESTION_ROWS, guarantee: None, - payload: PostAsapOperatorPayload::Binary { + payload: PostASAPOperatorPayload::Binary { operator: BinaryOperator { kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), vector_match: None, @@ -1374,7 +1374,7 @@ fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { }, }, }; - let program = CompiledPhysicalDag::from_operators( + let program = PhysicalPostASAPDAG::from_operators( [ (0, InputContract::bounded(input.clone())), (1, InputContract::bounded(input.clone())), @@ -1392,7 +1392,7 @@ fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { ) .unwrap(); let program = - serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) .unwrap(); for (right, expected) in [ (vec![("b", 2, 3.), ("a", 1, 2.)], Some(vec![8., 17.])), diff --git a/crates/asap-physical-operators/tests/physical_plan_recovery.rs b/crates/asap-physical-operators/tests/physical_plan_recovery.rs index 5a920a8d..c1ad720a 100644 --- a/crates/asap-physical-operators/tests/physical_plan_recovery.rs +++ b/crates/asap-physical-operators/tests/physical_plan_recovery.rs @@ -2,7 +2,7 @@ //! Deployments choose the encoding; JSON is used here only as a test format. use asap_physical_operators::{ operators::{Operator, SortKey}, - physical_planner::{CompiledPhysicalDag, InputContract}, + physical_planner::{InputContract, PhysicalPostASAPDAG}, }; use planner_types::{ post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, @@ -10,7 +10,7 @@ use planner_types::{ }; use std::{collections::BTreeMap, sync::Arc}; -fn sorted() -> CompiledPhysicalDag { +fn sorted() -> PhysicalPostASAPDAG { let schema = Arc::new(SummarySchema { fields: vec![SummaryField { name: "value".into(), @@ -19,7 +19,7 @@ fn sorted() -> CompiledPhysicalDag { }], time_index: None, }); - CompiledPhysicalDag::from_operators( + PhysicalPostASAPDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), BTreeMap::from([( 1, @@ -45,7 +45,7 @@ fn sorted() -> CompiledPhysicalDag { #[test] fn recovery_retains_selected_operator_and_rejects_invalid_contracts() { let bytes = serde_json::to_vec(&sorted()).unwrap(); - let recovered = serde_json::from_slice::(&bytes).unwrap(); + let recovered = serde_json::from_slice::(&bytes).unwrap(); assert_eq!(serde_json::to_vec(&recovered).unwrap(), bytes); for mutation in ["column", "edge", "output"] { let mut wire: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); @@ -62,7 +62,7 @@ fn recovery_retains_selected_operator_and_rejects_invalid_contracts() { _ => unreachable!(), } assert!( - serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()) .is_err(), "accepted {mutation}" ); @@ -74,7 +74,7 @@ fn candidate_recovery_preserves_materialization_boundary() { use asap_physical_operators::physical_planner::PhysicalCandidate; let precompute = sorted(); let output = InputContract::bounded(precompute.output_contract(1).unwrap().schema); - let query = CompiledPhysicalDag::from_operators( + let query = PhysicalPostASAPDAG::from_operators( BTreeMap::from([(1, output.clone())]), BTreeMap::from([( 2, diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs index 5770d7d6..72235344 100644 --- a/crates/asap-physical-operators/tests/physical_semantics.rs +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -4,14 +4,14 @@ use asap_physical_operators::{ expressions::CompiledExpression, operators::{Expression, Operator, Reduction, SortKey}, - plan::PhysicalDag, + plan::PhysicalExecution, runtime::{Limits, RunContext, Scope}, values::{Batch, Schema, Value}, }; use futures::{executor::block_on, StreamExt}; use planner_types::{ post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, - pre_asap::{CompareOpKind, DataType, JoinKind, Predicate, QueryExpr}, + pre_asap::{CompareOpKind, DataType, JoinKind, PreASAPNode, Predicate}, }; use std::{rc::Rc, sync::Arc}; @@ -41,7 +41,7 @@ fn context() -> RunContext { ) .unwrap() } -fn collect(dag: &PhysicalDag<'_, Batch, Schema>, root: u64) -> Vec> { +fn collect(dag: &PhysicalExecution<'_, Batch, Schema>, root: u64) -> Vec> { let run = context(); let rows = block_on(async { let mut stream = dag.execute(&[root], run.clone()).unwrap().remove(0); @@ -55,7 +55,7 @@ fn collect(dag: &PhysicalDag<'_, Batch, Schema>, root: u64) -> Vec> { rows } fn unary(input: Schema, batches: Vec>>, op: Operator) -> Vec> { - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); let batches = batches .into_iter() .map(|rows| Batch::try_new(input.clone(), rows).unwrap()) @@ -71,10 +71,10 @@ fn keys(rows: &[Vec]) -> Vec>> { .collect() } fn eq_predicate() -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), + right: Rc::new(PreASAPNode::Column(1)), })) } fn join(left: Vec, right: Vec, kind: JoinKind, keyed: bool) -> Vec> { @@ -93,7 +93,7 @@ fn join(left: Vec, right: Vec, kind: JoinKind, keyed: bool) -> Vec Operator::relational_join(input.clone(), input.clone(), kind, &eq_predicate(), output) .unwrap() }; - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); for (id, values) in [(0, left), (1, right)] { let batches = values .into_iter() @@ -332,7 +332,7 @@ fn aggregate_empty_and_all_null_follow_asap_contract() { fn projection_rejects_expression_bound_to_another_schema() { let original = schema(&[("a", DataType::Int64, false), ("b", DataType::Int64, false)]); let current = schema(&[("a", DataType::Int64, false)]); - let expr = CompiledExpression::compile(&QueryExpr::Column(1), &original).unwrap(); + let expr = CompiledExpression::compile(&PreASAPNode::Column(1), &original).unwrap(); assert!(Operator::project(current, vec![("b".into(), Expression::planner(expr))]).is_err()); } @@ -360,9 +360,9 @@ fn global_extrema_bind_with_planner_derived_schema() { .unwrap(); let result = derived.columns[0].clone(); let output = schema(&[(&result.name, result.dtype, result.nullable)]); - let node = PostAsapDagNode { - id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::Value { + let node = LogicalPostASAPDAGNode { + id: PostASAPNodeId(1), + payload: PostASAPOperatorPayload::Value { operation: ValueOperation::Exact(ExactOperation::Aggregate { reduction: PlanReduction::Reduce(GroupKeys::by(vec![])), measures: vec![measure], @@ -399,10 +399,10 @@ fn planner_comparisons_handle_nan_without_execution_errors() { CompareOpKind::Gt, CompareOpKind::Ge, ] { - let expression = QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let expression = PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(0)), op: op.clone(), - right: Rc::new(QueryExpr::Column(1)), + right: Rc::new(PreASAPNode::Column(1)), }; let compiled = CompiledExpression::compile(&expression, &input).unwrap(); for row in [ @@ -420,7 +420,7 @@ fn planner_comparisons_handle_nan_without_execution_errors() { #[test] fn limit_branch_finishes_without_blocking_shared_sibling() { let input = schema(&[("v", DataType::Int64, false)]); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); let batches = (0..100) .map(|v| Batch::try_new(input.clone(), vec![vec![Value::Int64(v)]]).unwrap()) .collect(); @@ -466,10 +466,10 @@ fn mixed_numeric_comparisons_preserve_large_integer_precision() { ("a", DataType::Int64, false), ("b", DataType::Float64, false), ]); - let expr = QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let expr = PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(0)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Column(1)), + right: Rc::new(PreASAPNode::Column(1)), }; let compiled = CompiledExpression::compile(&expr, &input).unwrap(); for (a, b, expected) in [ @@ -490,11 +490,11 @@ fn boolean_truth_tables_agree_between_expression_paths() { for and in [true, false] { for a in [None, Some(false), Some(true)] { for b in [None, Some(false), Some(true)] { - let parts = vec![QueryExpr::Column(0), QueryExpr::Column(1)]; + let parts = vec![PreASAPNode::Column(0), PreASAPNode::Column(1)]; let planner = if and { - QueryExpr::BoolAnd(parts) + PreASAPNode::BoolAnd(parts) } else { - QueryExpr::BoolOr(parts) + PreASAPNode::BoolOr(parts) }; let native = if and { Expression::And( @@ -542,7 +542,7 @@ fn kll_partial_merge_and_multiple_readouts_preserve_population() { SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), Default::default(), ); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); for (id, range) in [(0, 0..64), (1, 64..128), (2, 0..128)] { let rows = range.map(|n| vec![Value::Float64(n as f64)]).collect(); dag.add( @@ -626,7 +626,7 @@ fn zero_column_output_obeys_memory_limit() { use asap_physical_operators::Error; let input = schema(&[]); let batch = Batch::try_new(input.clone(), vec![vec![]; 200]).unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], Operator::source(input, vec![batch]).unwrap()) .unwrap(); let run = RunContext::new( @@ -668,7 +668,7 @@ fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { ) .unwrap(); let state = build.schema(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], Operator::source(input.clone(), vec![]).unwrap()) .unwrap(); dag.add(1, vec![0], build).unwrap(); diff --git a/crates/asap-physical-operators/tests/plan_properties.rs b/crates/asap-physical-operators/tests/plan_properties.rs index f7a468ab..3a3573d5 100644 --- a/crates/asap-physical-operators/tests/plan_properties.rs +++ b/crates/asap-physical-operators/tests/plan_properties.rs @@ -1,7 +1,7 @@ //! Finite-input contracts are validated before source execution. use asap_physical_operators::{ operators::{Operator, SortKey}, - plan::{Boundedness, Emission, PhysicalDag}, + plan::{Boundedness, Emission, PhysicalExecution}, runtime::{Limits, OutputStream, RunContext, Scope}, sources::{DataSources, RawSource}, values::{Batch, Schema}, @@ -9,7 +9,7 @@ use asap_physical_operators::{ }; use planner_types::{ post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, - pre_asap::{Column, DataType, QueryExpr, Schema as LogicalSchema, Source}, + pre_asap::{Column, DataType, PreASAPNode, Schema as LogicalSchema, Source}, }; use std::sync::{ atomic::{AtomicUsize, Ordering}, @@ -64,13 +64,13 @@ fn blocking_inputs_require_an_explicit_finite_source() { ) .unwrap(); let scan = registry - .bind(&QueryExpr::Scan { + .bind(&PreASAPNode::Scan { source: identity, schema: LogicalSchema::new(vec![Column::new("v", DataType::Int64, false)]), predicates: vec![], }) .unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add(0, vec![], scan).unwrap(); dag.add( 1, diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index 91bd5fdb..8344cf93 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -5,7 +5,7 @@ use asap_physical_operators::{ operators::Operator, physical_planner::{ compile, compile_candidate, compile_candidates, cut_candidate, enumerate_frontiers, - select_candidate, CandidateCost, CompiledPhysicalDag, InputContract, PhysicalCandidate, + select_candidate, CandidateCost, InputContract, PhysicalCandidate, PhysicalPostASAPDAG, Source, }, runtime::{Limits, RunContext, Scope}, @@ -15,7 +15,7 @@ use futures::{executor::block_on, StreamExt}; use planner_types::{post_asap::*, pre_asap::DataType, types::AccuracyTarget, workload::*}; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; -fn grouped_rate_space() -> asap_aware_mapping::CandidatePostASAPDAGs<&'static str> { +fn grouped_rate_space() -> asap_aware_mapping::CandidateLogicalPostASAPDAGs<&'static str> { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -52,16 +52,16 @@ fn grouped_rate_space() -> asap_aware_mapping::CandidatePostASAPDAGs<&'static st search_workload(vec![("grouped-rate", root)]) } -fn grouped_rate() -> PostAsapDag { +fn grouped_rate() -> LogicalPostASAPDAGTransport { let space = grouped_rate_space(); let selected = space .global_selection(&DefaultCostModel) - .assemble_selected_query(&space.roots[0].1) + .assemble_selected_query(&space.roots()[0].1) .unwrap() .unwrap(); - compile_post_asap_dag(&selected).unwrap() + export_post_asap_dag(&selected).unwrap() } -fn run(plan: &CompiledPhysicalDag, inputs: BTreeMap, scope: Scope) -> Vec { +fn run(plan: &PhysicalPostASAPDAG, inputs: BTreeMap, scope: Scope) -> Vec { let sources = inputs .into_iter() .map(|(id, batch)| { @@ -92,7 +92,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::SummaryAgg { + PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), .. } @@ -105,7 +105,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::FinalizeExactAccumulator } ) && dag @@ -116,7 +116,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .unwrap(); let input_schema = Arc::new(state.output_schema.clone()); let (family, update, grouping) = match &state.payload { - PostAsapOperatorPayload::SummaryAgg { + PostASAPOperatorPayload::SummaryAgg { family, input, grouping, @@ -390,7 +390,7 @@ fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::SummaryAgg { + PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), .. } @@ -424,11 +424,11 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { let mut executed = 0; for forest in inventory.candidates { let root = &forest[0].1; - let dag = compile_post_asap_dag(root).unwrap(); + let dag = export_post_asap_dag(root).unwrap(); let Some(state) = dag.nodes.iter().find(|node| { matches!( node.payload, - PostAsapOperatorPayload::SummaryAgg { + PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), .. } @@ -442,7 +442,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::SummaryAgg { + PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), .. } @@ -460,7 +460,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { &[vec![], vec![boundary]], ); let (family, input, grouping) = match &state.payload { - PostAsapOperatorPayload::SummaryAgg { + PostASAPOperatorPayload::SummaryAgg { family, input, grouping, @@ -560,7 +560,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { /// The per-frontier lowering used before compile-once cuts: each boundary /// choice lowers the precompute and query DAGs from the logical DAG again. fn recompiled_candidate( - dag: &PostAsapDag, + dag: &LogicalPostASAPDAGTransport, inputs: &BTreeMap, roots: &[u64], frontier: &[u64], @@ -569,11 +569,11 @@ fn recompiled_candidate( if frontier.is_empty() { return Ok(PhysicalCandidate { precompute: None, - query: compile(dag, inputs.clone(), roots)?, + query: compile(dag.as_view(), inputs.clone(), roots)?, materialized_outputs: BTreeMap::new(), }); } - let precompute = compile(dag, inputs.clone(), frontier)?; + let precompute = compile(dag.as_view(), inputs.clone(), frontier)?; let mut materialized_outputs = BTreeMap::new(); for &id in frontier { let mut output = precompute.output_contract(id)?; @@ -584,18 +584,18 @@ fn recompiled_candidate( query_inputs.extend(materialized_outputs.clone()); Ok(PhysicalCandidate { precompute: Some(precompute), - query: compile(dag, query_inputs, roots)?, + query: compile(dag.as_view(), query_inputs, roots)?, materialized_outputs, }) } fn assert_cuts_match_recompilation( - dag: &PostAsapDag, + dag: &LogicalPostASAPDAGTransport, inputs: BTreeMap, roots: &[u64], min_frontiers: usize, ) { - let compiled = compile(dag, inputs.clone(), roots).unwrap(); + let compiled = compile(dag.as_view(), inputs.clone(), roots).unwrap(); let frontiers = enumerate_frontiers(dag, &inputs, roots, 4096).unwrap(); assert!(frontiers.len() >= min_frontiers, "{frontiers:?}"); for frontier in &frontiers { @@ -617,7 +617,7 @@ fn grouped_rate_cuts_equal_per_frontier_compilation() { let state = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .find(|node| matches!(node.payload, PostASAPOperatorPayload::SummaryAgg { .. })) .unwrap(); let inputs = BTreeMap::from([( u64::from(state.id.0), @@ -666,18 +666,18 @@ fn population_topk_cuts_equal_per_frontier_compilation() { ) .candidate(&root) .unwrap(); - let dag = compile_post_asap_dag(&selected).unwrap(); + let dag = export_post_asap_dag(&selected).unwrap(); let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| matches!(node.payload, PostASAPOperatorPayload::Fallback { .. })) .unwrap(); let inputs = BTreeMap::from([( u64::from(raw.id.0), InputContract::bounded(Arc::new(raw.output_schema.clone())), )]); let roots = [u64::from(dag.root.0)]; - let compiled = compile(&dag, inputs.clone(), &roots).unwrap(); + let compiled = compile(dag.as_view(), inputs.clone(), &roots).unwrap(); // The root reads its population through a Sort helper numbered by the root. let helper = u64::MAX - (roots[0] << 16); assert_eq!(compiled.operator_name(helper), Some("Sort")); @@ -696,7 +696,7 @@ fn cut_candidate_rejects_invalid_frontiers() { let state = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .find(|node| matches!(node.payload, PostASAPOperatorPayload::SummaryAgg { .. })) .unwrap(); let readout = dag .nodes @@ -704,7 +704,7 @@ fn cut_candidate_rejects_invalid_frontiers() { .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::FinalizeExactAccumulator } ) @@ -719,7 +719,7 @@ fn cut_candidate_rejects_invalid_frontiers() { state_id, InputContract::bounded(Arc::new(state.output_schema.clone())), )]); - let compiled = compile(&dag, inputs.clone(), &[root]).unwrap(); + let compiled = compile(dag.as_view(), inputs.clone(), &[root]).unwrap(); for frontier in [ vec![rate_id, rate_id], vec![state_id], diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs index 3021cf11..f0637a8f 100644 --- a/crates/asap-physical-operators/tests/precompute_population.rs +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -2,7 +2,7 @@ use asap_physical_operators::{ factory::create_planner_accumulator, operators::Operator, - physical_planner::{precompute, CompiledPhysicalDag, Source}, + physical_planner::{precompute, PhysicalPostASAPDAG, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, Statistic, @@ -44,25 +44,25 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { (SummaryInputExpr::Constant(1.), 4.), ] { let nodes = vec![ - PostAsapDagNode { - id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::SummaryMerge, + LogicalPostASAPDAGNode { + id: PostASAPNodeId(0), + payload: PostASAPOperatorPayload::SummaryMerge, output_state: ExecutionDataState::INGESTION_SUMMARY, output_schema: state_schema.clone(), guarantee: None, }, - PostAsapDagNode { - id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::Value { + LogicalPostASAPDAGNode { + id: PostASAPNodeId(1), + payload: PostASAPOperatorPayload::Value { operation: ValueOperation::FinalizeExactAccumulator, }, output_state: ExecutionDataState::INGESTION_ROWS, output_schema: value_schema.clone(), guarantee: None, }, - PostAsapDagNode { - id: PostAsapNodeId(2), - payload: PostAsapOperatorPayload::Binary { + LogicalPostASAPDAGNode { + id: PostASAPNodeId(2), + payload: PostASAPOperatorPayload::Binary { operator: BinaryOperator { kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), vector_match: None, @@ -74,9 +74,9 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { output_schema: value_schema.clone(), guarantee: None, }, - PostAsapDagNode { - id: PostAsapNodeId(3), - payload: PostAsapOperatorPayload::SummaryAgg { + LogicalPostASAPDAGNode { + id: PostASAPNodeId(3), + payload: PostASAPOperatorPayload::SummaryAgg { family: family.clone(), input: SummaryUpdate { weight, @@ -97,9 +97,9 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { (2, 3, EdgeRole::Input), ] .into_iter() - .map(|(producer, consumer, role)| PostAsapDagEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), + .map(|(producer, consumer, role)| LogicalPostASAPDAGEdge { + producer: PostASAPNodeId(producer), + consumer: PostASAPNodeId(consumer), role, intermediate_schema: nodes[producer as usize].output_schema.clone(), data_state: nodes[producer as usize].output_state, @@ -107,10 +107,10 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { window: WindowEdgeCompatibility::NotApplicable, }) .collect(); - let dag = PostAsapDag { + let dag = LogicalPostASAPDAGTransport { nodes, edges, - root: PostAsapNodeId(3), + root: PostASAPNodeId(3), }; // Identity metadata must remain one non-null Utf8 column. for mutation in 0..3 { @@ -124,7 +124,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { assert!(precompute::compile(&invalid_identity, &[0], &[3]).is_err()); } let mut invalid_grouping = dag.clone(); - let PostAsapOperatorPayload::SummaryAgg { reduction, .. } = + let PostASAPOperatorPayload::SummaryAgg { reduction, .. } = &mut invalid_grouping.nodes[3].payload else { unreachable!() @@ -136,7 +136,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { ); let program = precompute::compile(&dag, &[0], &[3]).unwrap(); let program = - serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) .unwrap(); assert_eq!(program.input_contracts().count(), 1); for revision in [1, 2] { @@ -224,25 +224,25 @@ fn state_graph( family: SummaryFamilyType, target: Option, merge: bool, -) -> CompiledPhysicalDag { - let mut nodes = vec![PostAsapDagNode { - id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::SummaryMerge, +) -> PhysicalPostASAPDAG { + let mut nodes = vec![LogicalPostASAPDAGNode { + id: PostASAPNodeId(0), + payload: PostASAPOperatorPayload::SummaryMerge, output_state: ExecutionDataState::INGESTION_SUMMARY, output_schema: logical_schema(family.clone()), guarantee: None, }]; if merge { - nodes.push(PostAsapDagNode { - id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::SummaryMerge, + nodes.push(LogicalPostASAPDAGNode { + id: PostASAPNodeId(1), + payload: PostASAPOperatorPayload::SummaryMerge, ..nodes[0].clone() }); } let read_id = nodes.len() as u32; - nodes.push(PostAsapDagNode { - id: PostAsapNodeId(read_id), - payload: PostAsapOperatorPayload::Value { + nodes.push(LogicalPostASAPDAGNode { + id: PostASAPNodeId(read_id), + payload: PostASAPOperatorPayload::Value { operation: ValueOperation::FinalizeExactAccumulator, }, output_state: ExecutionDataState::INGESTION_ROWS, @@ -250,9 +250,9 @@ fn state_graph( guarantee: None, }); if let Some(target) = target { - nodes.push(PostAsapDagNode { - id: PostAsapNodeId(nodes.len() as u32), - payload: PostAsapOperatorPayload::SummaryAgg { + nodes.push(LogicalPostASAPDAGNode { + id: PostASAPNodeId(nodes.len() as u32), + payload: PostASAPOperatorPayload::SummaryAgg { family: target.clone(), input: SummaryUpdate::column(ColumnRef::SampleValue), reduction: Reduction::by(vec![]), @@ -264,7 +264,7 @@ fn state_graph( }); } let edges = (1..nodes.len()) - .map(|i| PostAsapDagEdge { + .map(|i| LogicalPostASAPDAGEdge { producer: nodes[i - 1].id, consumer: nodes[i].id, role: EdgeRole::Input, @@ -276,20 +276,20 @@ fn state_graph( .collect(); let root = nodes.last().unwrap().id; precompute::compile( - &PostAsapDag { nodes, edges, root }, + &LogicalPostASAPDAGTransport { nodes, edges, root }, &[0], &[u64::from(root.0)], ) .unwrap() } fn native_run( - program: &CompiledPhysicalDag, + program: &PhysicalPostASAPDAG, family: SummaryFamilyType, states: Vec>, context: RunContext, ) -> Result>, asap_physical_operators::Error> { let program = - serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) .unwrap(); let rows = states .into_iter() diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs index 822bde4e..e2481bc1 100644 --- a/crates/asap-physical-operators/tests/promql_binary.rs +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -1,15 +1,15 @@ //! Binary computation must be fully compiled before deployment binds values. use asap_physical_operators::{ operators::Operator, - physical_planner::{compile_node, CompiledPhysicalDag, InputContract, Source}, + physical_planner::{compile_node, InputContract, PhysicalPostASAPDAG, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Schema, Value}, }; use futures::{executor::block_on, StreamExt}; use planner_types::{ post_asap::{ - BinaryOperator, ExecutionDataState, PostAsapDagNode, PostAsapNodeId, - PostAsapOperatorPayload, SummaryFamilyType, SummaryField, SummarySchema, + BinaryOperator, ExecutionDataState, LogicalPostASAPDAGNode, PostASAPNodeId, + PostASAPOperatorPayload, SummaryFamilyType, SummaryField, SummarySchema, }, pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, }; @@ -48,7 +48,7 @@ fn row(name: &str, job: &str, value: f64) -> Vec { Value::Float64(value), ] } -fn program() -> CompiledPhysicalDag { +fn program() -> PhysicalPostASAPDAG { program_for(BinaryOperator { kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), vector_match: None, @@ -56,17 +56,17 @@ fn program() -> CompiledPhysicalDag { checked_finite_division: false, }) } -fn program_for(operator: BinaryOperator) -> CompiledPhysicalDag { +fn program_for(operator: BinaryOperator) -> PhysicalPostASAPDAG { let schema = schema(); - let node = PostAsapDagNode { - id: PostAsapNodeId(2), - payload: PostAsapOperatorPayload::Binary { operator }, + let node = LogicalPostASAPDAGNode { + id: PostASAPNodeId(2), + payload: PostASAPOperatorPayload::Binary { operator }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*schema).clone(), guarantee: None, }; let operator = compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(); - let graph = CompiledPhysicalDag::from_operators( + let graph = PhysicalPostASAPDAG::from_operators( BTreeMap::from([ (0, InputContract::bounded(schema.clone())), (1, InputContract::bounded(schema)), @@ -75,7 +75,7 @@ fn program_for(operator: BinaryOperator) -> CompiledPhysicalDag { vec![2], ) .unwrap(); - serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()).unwrap() + serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()).unwrap() } fn evaluate( left: Vec>, @@ -84,7 +84,7 @@ fn evaluate( evaluate_with(program(), left, right) } fn evaluate_with( - graph: CompiledPhysicalDag, + graph: PhysicalPostASAPDAG, left: Vec>, right: Vec>, ) -> Result>, asap_physical_operators::Error> { @@ -163,7 +163,7 @@ fn scalar_broadcast_and_bool_comparison_are_distinct() { ) .unwrap(); let graph = - serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) .unwrap(); let scalar = promql_values::scalar_schema(); let vector = promql_values::vector_schema(); @@ -341,14 +341,14 @@ fn stored_series_readouts_support_filters_and_sets() { BinaryOpKind::Set(PromQLVectorSetOpKind::Or), ] { let nodes = (0..5) - .map(|id| PostAsapDagNode { - id: PostAsapNodeId(id), + .map(|id| LogicalPostASAPDAGNode { + id: PostASAPNodeId(id), payload: match id { - 0 | 1 => PostAsapOperatorPayload::SummaryMerge, - 2 | 3 => PostAsapOperatorPayload::Value { + 0 | 1 => PostASAPOperatorPayload::SummaryMerge, + 2 | 3 => PostASAPOperatorPayload::Value { operation: ValueOperation::FinalizeExactAccumulator, }, - _ => PostAsapOperatorPayload::Binary { + _ => PostASAPOperatorPayload::Binary { operator: BinaryOperator { kind: kind.clone(), vector_match: None, @@ -377,9 +377,9 @@ fn stored_series_readouts_support_filters_and_sets() { (3, 4, EdgeRole::Right), ] .into_iter() - .map(|(producer, consumer, role)| PostAsapDagEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), + .map(|(producer, consumer, role)| LogicalPostASAPDAGEdge { + producer: PostASAPNodeId(producer), + consumer: PostASAPNodeId(consumer), role, intermediate_schema: nodes[producer as usize].output_schema.clone(), data_state: nodes[producer as usize].output_state, @@ -387,13 +387,13 @@ fn stored_series_readouts_support_filters_and_sets() { window: WindowEdgeCompatibility::NotApplicable, }) .collect(); - let dag = PostAsapDag { + let dag = LogicalPostASAPDAGTransport { nodes, edges, - root: PostAsapNodeId(4), + root: PostASAPNodeId(4), }; let graph = compile( - &dag, + dag.as_view(), BTreeMap::from([ (0, InputContract::bounded(state_schema.clone())), (1, InputContract::bounded(state_schema.clone())), @@ -401,7 +401,7 @@ fn stored_series_readouts_support_filters_and_sets() { &[4], ) .unwrap(); - let graph: CompiledPhysicalDag = + let graph: PhysicalPostASAPDAG = serde_json::from_slice(&serde_json::to_vec(&graph).unwrap()).unwrap(); let sources = [(0, "a", 6.), (1, "b", 2.)] .into_iter() diff --git a/crates/asap-physical-operators/tests/promql_fallback.rs b/crates/asap-physical-operators/tests/promql_fallback.rs index 88fe3690..bffc6eb6 100644 --- a/crates/asap-physical-operators/tests/promql_fallback.rs +++ b/crates/asap-physical-operators/tests/promql_fallback.rs @@ -3,25 +3,25 @@ //! hand-computed with Prometheus semantics. use asap_physical_operators::{ operators::Operator, - physical_planner::{compile, promql_fallback, promql_rows, CompiledPhysicalDag, InputContract}, + physical_planner::{compile, promql_fallback, promql_rows, InputContract, PhysicalPostASAPDAG}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; use planner_types::{ post_asap::{execution_data_state::lift_plain, *}, - pre_asap::QueryExpr, + pre_asap::PreASAPNode, types::AccuracyTarget, workload::*, }; use std::{collections::BTreeMap, rc::Rc}; /// Bare selectors look back one ingestion interval: 60s. -fn parse(query: &str) -> QueryExpr { +fn parse(query: &str) -> PreASAPNode { parse_with(query, AccuracyTarget::Exact) } -fn parse_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { +fn parse_with(query: &str, accuracy: AccuracyTarget) -> PreASAPNode { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -51,14 +51,14 @@ fn parse_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { .remove(0) } -fn lower(query: &str) -> QueryExpr { +fn lower(query: &str) -> PreASAPNode { promql_rows::with_series_identity(&parse(query)).unwrap() } /// The whole query retained as one pre-ASAP node. -fn fallback_dag(expression: QueryExpr) -> PostAsapDag { +fn fallback_dag(expression: PreASAPNode) -> LogicalPostASAPDAGTransport { let schema = lift_plain(&expression.output_schema().unwrap()); - compile_post_asap_dag(&Rc::new(SummaryNode { + export_post_asap_dag(&Rc::new(PostASAPNode { expr: SummaryExpr::KeepPreAsap(Rc::new(expression)), schema, guarantee: None, @@ -82,24 +82,29 @@ fn labels(spec: &str) -> BTreeMap { } /// The metric a selector reads. -fn metric(selector: &QueryExpr) -> String { +fn metric(selector: &PreASAPNode) -> String { match selector { - QueryExpr::Scan { + PreASAPNode::Scan { source: planner_types::pre_asap::Source::TimeSeries { metric }, .. } => metric.clone(), - QueryExpr::TimeRange { child, .. } | QueryExpr::TimeShift { child, .. } => metric(child), + PreASAPNode::TimeRange { child, .. } | PreASAPNode::TimeShift { child, .. } => { + metric(child) + } other => panic!("not a selector: {other:?}"), } } -fn compile_query(query: &str) -> Result { +fn compile_query(query: &str) -> Result { let expression = lower(query); compile_dag(&expression, &fallback_dag(expression.clone())) } /// Compile a DAG whose root is the Fallback computing `expression`. -fn compile_dag(expression: &QueryExpr, dag: &PostAsapDag) -> Result { +fn compile_dag( + expression: &PreASAPNode, + dag: &LogicalPostASAPDAGTransport, +) -> Result { let root = u64::from(dag.root.0); let inputs = promql_fallback::raw_series(expression) .map_err(|e| e.to_string())? @@ -112,7 +117,7 @@ fn compile_dag(expression: &QueryExpr, dag: &PostAsapDag) -> Result Result, i64, f64)>, String> { @@ -140,8 +145,8 @@ fn evaluate_dag( #[allow(clippy::type_complexity)] fn evaluate_dag_with_range( - expression: &QueryExpr, - dag: &PostAsapDag, + expression: &PreASAPNode, + dag: &LogicalPostASAPDAGTransport, metrics: &[(&str, &[Sample])], at: i64, bounds: Option<(i64, i64)>, @@ -428,13 +433,15 @@ fn raw_series_contract_is_explicit() { .unwrap() .try_into() .unwrap(); - assert!(matches!(selector, QueryExpr::TimeRange { .. })); - let missing = compile(&dag, BTreeMap::new(), &[root]).err().unwrap(); + assert!(matches!(selector, PreASAPNode::TimeRange { .. })); + let missing = compile(dag.as_view(), BTreeMap::new(), &[root]) + .err() + .unwrap(); assert!(missing.to_string().contains("raw series input")); let mut wrong = (*schema).clone(); wrong.fields.pop(); let wrong = compile( - &dag, + dag.as_view(), BTreeMap::from([( promql_fallback::raw_series_input(root, 0), InputContract::bounded(std::sync::Arc::new(wrong)), @@ -446,24 +453,24 @@ fn raw_series_contract_is_explicit() { // turned into instant selection. let selector = lower("m"); let schema = lift_plain(&selector.output_schema().unwrap()); - let node = |id, payload| PostAsapDagNode { - id: PostAsapNodeId(id), + let node = |id, payload| LogicalPostASAPDAGNode { + id: PostASAPNodeId(id), payload, output_state: ExecutionDataState::QUERY_ROWS, output_schema: schema.clone(), guarantee: None, }; - let consumed = PostAsapDag { + let consumed = LogicalPostASAPDAGTransport { nodes: vec![ node( 0, - PostAsapOperatorPayload::Fallback { + PostASAPOperatorPayload::Fallback { expression: selector.clone(), }, ), node( 1, - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::Limit { n: 1, offset: 0, @@ -472,20 +479,20 @@ fn raw_series_contract_is_explicit() { }, ), ], - edges: vec![PostAsapDagEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), + edges: vec![LogicalPostASAPDAGEdge { + producer: PostASAPNodeId(0), + consumer: PostASAPNodeId(1), role: EdgeRole::Input, intermediate_schema: schema.clone(), data_state: ExecutionDataState::QUERY_ROWS, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }], - root: PostAsapNodeId(1), + root: PostASAPNodeId(1), }; let raw = promql_fallback::raw_series(&selector).unwrap().remove(0).1; assert!(compile( - &consumed, + consumed.as_view(), BTreeMap::from([( promql_fallback::raw_series_input(0, 0), InputContract::bounded(raw) @@ -1239,7 +1246,7 @@ fn histogram_quantile_selection_keeps_the_exact_fallback() { &default_strategies(), &DefaultAccuracyModel, ); - let planned = &space.roots[0].1; + let planned = &space.roots()[0].1; let candidates = &space.candidates_for_target(planned).unwrap().candidates; assert!( candidates.iter().all(|c| matches!(&c.replacement, @@ -1252,7 +1259,7 @@ fn histogram_quantile_selection_keeps_the_exact_fallback() { .assemble_selected_dag(planned) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&selected).unwrap(); + let dag = export_post_asap_dag(&selected).unwrap(); let rows = evaluate_dag(&root, &dag, &[("x_bucket", &samples)], 60).unwrap(); let values: Vec<_> = rows.iter().map(|(_, _, v)| *v).collect(); assert_eq!(values, vec![1.75], "{query} {target:?}"); @@ -1285,7 +1292,7 @@ fn nonfinite_literals_round_trip_in_plans() { ] { let expression = lower(query); let json = serde_json::to_vec(&expression).unwrap(); - let restored: QueryExpr = serde_json::from_slice(&json).unwrap(); + let restored: PreASAPNode = serde_json::from_slice(&json).unwrap(); let result = evaluate_dag(&restored, &fallback_dag(restored.clone()), &[], 60).unwrap(); assert_eq!(result.len(), 1); if expected.is_nan() { @@ -1579,7 +1586,7 @@ fn subquery_label_uniqueness_is_checked_per_evaluation_step() { #[test] fn logical_nonfinite_quantile_parameter_round_trips() { let expression = lower("histogram_quantile(NaN, x_bucket)"); - let restored: QueryExpr = + let restored: PreASAPNode = serde_json::from_slice(&serde_json::to_vec(&expression).unwrap()).unwrap(); let samples = buckets(&[("job=a", HISTOGRAM)]); let result = evaluate_dag( diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs index 886b7d00..4923e48d 100644 --- a/crates/asap-physical-operators/tests/promql_values.rs +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -1,7 +1,7 @@ //! Compile, persist and rebind dynamic-label computation without deployment lowering. use asap_physical_operators::{ operators::Operator, - physical_planner::{promql_values::*, CompiledPhysicalDag, Source}, + physical_planner::{promql_values::*, PhysicalPostASAPDAG, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; @@ -21,14 +21,14 @@ fn row(labels: &[(&str, &str)], value: f64) -> Vec { Value::Float64(value), ] } -fn run(graph: CompiledPhysicalDag, rows: Vec>) -> Vec> { +fn run(graph: PhysicalPostASAPDAG, rows: Vec>) -> Vec> { run_inputs(graph, vec![Batch::try_new(vector_schema(), rows).unwrap()]).unwrap() } fn run_inputs( - graph: CompiledPhysicalDag, + graph: PhysicalPostASAPDAG, batches: Vec, ) -> Result>, asap_physical_operators::Error> { - let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) .unwrap(); let sources = batches .into_iter() @@ -263,7 +263,7 @@ fn composed_ensemble_shares_a_producer_across_roots() { false, ) .unwrap(); - let graph = CompiledPhysicalDag::compose( + let graph = PhysicalPostASAPDAG::compose( BTreeMap::from([(0, InputContract::bounded(vector_schema()))]), BTreeMap::from([ (10, (vec![0], aggregate)), @@ -273,7 +273,7 @@ fn composed_ensemble_shares_a_producer_across_roots() { vec![20, 30], ) .unwrap(); - let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) .unwrap(); assert_eq!(graph.input_contracts().count(), 1); let starts = std::rc::Rc::new(std::cell::Cell::new(0)); diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs index 78e4c8c0..1f2b408f 100644 --- a/crates/asap-physical-operators/tests/raw_scan.rs +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -8,7 +8,7 @@ use asap_physical_operators::dag::{ use futures::{executor::block_on, stream, StreamExt}; use planner_types::{ post_asap::*, - pre_asap::{Column, DataType, GroupKeys, Predicate, QueryExpr, Source}, + pre_asap::{Column, DataType, GroupKeys, PreASAPNode, Predicate, Source}, }; use std::{ collections::BTreeMap, @@ -19,7 +19,7 @@ use std::{ }, }; -fn fixture() -> (QueryExpr, Schema, Vec) { +fn fixture() -> (PreASAPNode, Schema, Vec) { let schema = planner_types::pre_asap::Schema::new(vec![Column::new("value", DataType::Int64, true)]); let output = Arc::new(SummarySchema { @@ -30,12 +30,12 @@ fn fixture() -> (QueryExpr, Schema, Vec) { }], time_index: None, }); - let scan = QueryExpr::Scan { + let scan = PreASAPNode::Scan { source: Source::Table { table_ref: "numbers".into(), }, - predicates: vec![Predicate(Rc::new(QueryExpr::IsNotNull(Rc::new( - QueryExpr::Column(0), + predicates: vec![Predicate(Rc::new(PreASAPNode::IsNotNull(Rc::new( + PreASAPNode::Column(0), ))))], schema, }; @@ -53,32 +53,36 @@ fn fixture() -> (QueryExpr, Schema, Vec) { ]; (scan, output, batches) } -fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsapDag { - let node = |id, payload| PostAsapDagNode { - id: PostAsapNodeId(id), +fn plan( + scan: PreASAPNode, + schema: &Schema, + state: ExecutionDataState, +) -> LogicalPostASAPDAGTransport { + let node = |id, payload| LogicalPostASAPDAGNode { + id: PostASAPNodeId(id), payload, output_state: state, output_schema: (**schema).clone(), guarantee: None, }; - let edge = |producer, consumer| PostAsapDagEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), + let edge = |producer, consumer| LogicalPostASAPDAGEdge { + producer: PostASAPNodeId(producer), + consumer: PostASAPNodeId(consumer), role: EdgeRole::Input, intermediate_schema: (**schema).clone(), data_state: state, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }; - PostAsapDag { + LogicalPostASAPDAGTransport { nodes: vec![ - node(0, PostAsapOperatorPayload::Fallback { expression: scan }), + node(0, PostASAPOperatorPayload::Fallback { expression: scan }), node( 1, - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::Sort { keys: vec![planner_types::pre_asap::SortKey { - expr: QueryExpr::Column(0), + expr: PreASAPNode::Column(0), ascending: false, nulls_first: false, }], @@ -88,7 +92,7 @@ fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsap ), node( 2, - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::Limit { n: 2, offset: 0, @@ -98,7 +102,7 @@ fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsap ), ], edges: vec![edge(0, 1), edge(1, 2)], - root: PostAsapNodeId(2), + root: PostASAPNodeId(2), } } fn registry(source: Arc) -> DataSources { @@ -226,8 +230,8 @@ fn binding_errors_and_reader_errors_are_not_empty_results() { opened: opened.clone(), fail: true, })); - if let QueryExpr::Scan { predicates, .. } = &mut scan { - predicates.push(Predicate(Rc::new(QueryExpr::Column(0)))); + if let PreASAPNode::Scan { predicates, .. } = &mut scan { + predicates.push(Predicate(Rc::new(PreASAPNode::Column(0)))); } assert!(sources.bind(&scan).is_err()); assert_eq!(opened.load(Ordering::SeqCst), 0); @@ -294,17 +298,17 @@ fn schema_drift_and_memory_limits_fail_the_scan() { fn empty_sources_and_three_valued_predicates() { use planner_types::pre_asap::{CompareOpKind, ScalarValue}; let (mut scan, schema, batches) = fixture(); - if let QueryExpr::Scan { + if let PreASAPNode::Scan { predicates, source, .. } = &mut scan { *source = Source::TimeSeries { metric: "samples".into(), }; - *predicates = vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + *predicates = vec![Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(0)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(2))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Int64(2))), }))]; } for (batches, expected) in [(vec![], 0), (batches, 2)] { @@ -340,7 +344,7 @@ fn compile_without_readers_and_rebind_inputs() { let (scan, schema, batches) = fixture(); let dag = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); let compiled = compile( - &dag, + dag.as_view(), BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), &[2], ) @@ -382,5 +386,5 @@ fn compilation_rejects_unknown_boundedness_for_sort() { emission: Emission::Unknown, }, }; - assert!(compile(&dag, BTreeMap::from([(0, input)]), &[2]).is_err()); + assert!(compile(dag.as_view(), BTreeMap::from([(0, input)]), &[2]).is_err()); } diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs index a61a596c..6dbfd29a 100644 --- a/crates/asap-physical-operators/tests/summary_projection.rs +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -3,14 +3,14 @@ use asap_physical_operators::{ expressions::Expression, factory::create_planner_accumulator, operators::Operator, - physical_planner::{compile, CompiledPhysicalDag, InputContract, Source}, + physical_planner::{compile, InputContract, PhysicalPostASAPDAG, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; use planner_types::{ post_asap::*, - pre_asap::{ColumnRef, DataType, ProjectItem, QueryExpr}, + pre_asap::{ColumnRef, DataType, PreASAPNode, ProjectItem}, }; use std::{collections::BTreeMap, sync::Arc}; @@ -44,24 +44,24 @@ fn post_asap_summary_projection_survives_recovery() { ], time_index: None, }; - let dag = PostAsapDag { + let dag = LogicalPostASAPDAGTransport { nodes: vec![ - PostAsapDagNode { - id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::SummaryMerge, + LogicalPostASAPDAGNode { + id: PostASAPNodeId(0), + payload: PostASAPOperatorPayload::SummaryMerge, output_schema: (*schema).clone(), output_state: ExecutionDataState::INGESTION_SUMMARY, guarantee: None, }, - PostAsapDagNode { - id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::Value { + LogicalPostASAPDAGNode { + id: PostASAPNodeId(1), + payload: PostASAPOperatorPayload::Value { operation: ValueOperation::Project { cols: vec![1, 0] .into_iter() .map(|index| ProjectItem { alias: None, - expr: QueryExpr::Column(index), + expr: PreASAPNode::Column(index), }) .collect(), qualifier: None, @@ -72,30 +72,30 @@ fn post_asap_summary_projection_survives_recovery() { guarantee: None, }, ], - edges: vec![PostAsapDagEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), + edges: vec![LogicalPostASAPDAGEdge { + producer: PostASAPNodeId(0), + consumer: PostASAPNodeId(1), role: EdgeRole::Input, intermediate_schema: (*schema).clone(), data_state: ExecutionDataState::INGESTION_SUMMARY, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }], - root: PostAsapNodeId(1), + root: PostASAPNodeId(1), }; let program = compile( - &dag, + dag.as_view(), BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), &[1], ) .unwrap(); let encoded = serde_json::to_vec(&program).unwrap(); - let program = serde_json::from_slice::(&encoded).unwrap(); + let program = serde_json::from_slice::(&encoded).unwrap(); let mut forged: serde_json::Value = serde_json::from_slice(&encoded).unwrap(); forged["nodes"]["1"]["Operator"]["operator"]["output"]["fields"][1]["dtype"] = serde_json::json!({"Plain": "float64"}); assert!( - serde_json::from_slice::(&serde_json::to_vec(&forged).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&forged).unwrap()) .is_err() ); assert!(Operator::project( diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index ef086b1b..680f610a 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -15,13 +15,13 @@ use asap_physical_operators::dag::{ use futures::{executor::block_on, StreamExt}; use planner_types::{ post_asap::*, - pre_asap::{DataType, QueryExpr}, + pre_asap::{DataType, PreASAPNode}, types::AccuracyTarget, }; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; struct Evidence; impl AccuracyEvidenceProvider for Evidence { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &PreASAPNode) -> Option { Some(1000) } fn propagation_stats( @@ -88,8 +88,8 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S _ => None, }) .unwrap(); - let dag = compile_post_asap_dag(&plan).unwrap(); - let build=dag.nodes.iter().find(|node|matches!(&node.payload,PostAsapOperatorPayload::SummaryAgg{family:SummaryFamilyType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); + let dag = export_post_asap_dag(&plan).unwrap(); + let build=dag.nodes.iter().find(|node|matches!(&node.payload,PostASAPOperatorPayload::SummaryAgg{family:SummaryFamilyType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); let rate_id = dag .edges .iter() @@ -154,7 +154,7 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S let source = Box::new(Operator::source(rates.clone(), vec![batch.clone()]).unwrap()) as Source<'static>; let compiled = compile( - &placed, + placed.as_view(), BTreeMap::from([(rate_id.0 as u64, InputContract::bounded(rates.clone()))]), &[dag.root.0 as u64], ) @@ -203,7 +203,7 @@ use planner_types::workload::{ pub fn lower_promql( query: &str, accuracy: AccuracyTarget, -) -> Result { +) -> Result { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -284,12 +284,12 @@ fn check_direct_rate_topk(dynamic: bool) { }; let mut logical = lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(); - fn resolve_catalog(node: &mut QueryExpr) { + fn resolve_catalog(node: &mut PreASAPNode) { match node { - QueryExpr::Aggregate { child, .. } | QueryExpr::TimeRange { child, .. } => { + PreASAPNode::Aggregate { child, .. } | PreASAPNode::TimeRange { child, .. } => { resolve_catalog(Rc::make_mut(child)) } - QueryExpr::Scan { schema, .. } => { + PreASAPNode::Scan { schema, .. } => { schema.closed = true; schema .columns @@ -352,11 +352,11 @@ fn check_direct_rate_topk(dynamic: bool) { "Rate must be supplied by its exact stored-state readout" ); } - let dag = compile_post_asap_dag(candidate).unwrap(); + let dag = export_post_asap_dag(candidate).unwrap(); assert!(dag.nodes.iter().any(|node| matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); + PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); let build = dag.nodes.iter().find(|node| matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); + PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); let input_id = dag .edges .iter() @@ -377,15 +377,15 @@ fn check_direct_rate_topk(dynamic: bool) { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::TimeRange { .. } + PostASAPOperatorPayload::Fallback { + expression: PreASAPNode::TimeRange { .. } } ) }) .unwrap_or_else(|| panic!("no raw counter source: {dag:?}")); let raw_schema = Arc::new(raw.output_schema.clone()); let raw_compiled = compile( - &dag, + dag.as_view(), BTreeMap::from([( u64::from(raw.id.0), InputContract::bounded(raw_schema.clone()), @@ -395,7 +395,7 @@ fn check_direct_rate_topk(dynamic: bool) { .unwrap(); let bytes = serde_json::to_vec(&raw_compiled).unwrap(); let raw_compiled = serde_json::from_slice::< - asap_physical_operators::physical_planner::CompiledPhysicalDag, + asap_physical_operators::physical_planner::PhysicalPostASAPDAG, >(&bytes) .unwrap(); // Each evaluation receives a complete raw window. A reset, a stopped @@ -520,7 +520,7 @@ fn check_direct_rate_topk(dynamic: bool) { } } let compiled = compile( - &dag, + dag.as_view(), BTreeMap::from([( u64::from(input_id.0), InputContract::bounded(schema.clone()), @@ -655,22 +655,22 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { _ => None, }) .expect("signed spatial TopK must expose CountSketch with heap"); - let dag = compile_post_asap_dag(selected).unwrap(); + let dag = export_post_asap_dag(selected).unwrap(); let raw = dag .nodes .iter() .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::TimeRange { .. } + PostASAPOperatorPayload::Fallback { + expression: PreASAPNode::TimeRange { .. } } ) }) .unwrap(); let schema = Arc::new(raw.output_schema.clone()); let program = compile( - &dag, + dag.as_view(), BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), &[u64::from(dag.root.0)], ) @@ -758,11 +758,10 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { /// Deployment-side lifecycle choice: every summary state of `candidate` is /// continuously maintained, and the chosen lifecycles set execution timing. -fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDag { +fn continuously_maintained_dag(candidate: &Rc) -> LogicalPostASAPDAGTransport { use asap_aware_mapping::{ cost_model::{Cost, CostModel}, - enumerate_summary_maintenance_lifecycles, CostRate, Horizon, - SummaryMaintenanceCapabilities, SummaryMaintenanceLifecycleCapabilities, + CostRate, Horizon, SummaryMaintenanceCapabilities, SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecycleCostInputs, WorkloadDemand, }; use planner_types::workload::{ @@ -779,7 +778,7 @@ fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDag { } fn summary_maintenance_lifecycle_cost_inputs( &self, - _: &SummaryNode, + _: &PostASAPNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(10.)), @@ -791,7 +790,7 @@ fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDag { } fn summary_maintenance_capabilities( &self, - _: &SummaryNode, + _: &PostASAPNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -822,17 +821,22 @@ fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDag { }, ..Default::default() }; - let lifecycles = enumerate_summary_maintenance_lifecycles( + let lifecycles = asap_aware_mapping::CandidateLifecyclePostASAPDAGs::from_post_asap_dag( + (), Rc::clone(candidate), - WorkloadDemand::new_with_data(&queries, &data, &[0]), - NOW_MS, - Some(Horizon(100.)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &Costed, + asap_aware_mapping::CandidateTimingContext { + demand: WorkloadDemand::new_with_data(&queries, &data, &[0]), + now_ms: NOW_MS, + horizon: Some(Horizon(100.)), + capabilities: SummaryMaintenanceLifecycleCapabilities::ALL, + cost_model: &Costed, + }, + 4096, ) .unwrap(); let choices = lifecycles - .deployments() + .lifecycle_alternatives(0) + .unwrap() .iter() .map(|deployment| { ( @@ -842,9 +846,9 @@ fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDag { }) .collect::>(); lifecycles - .select(&choices) + .select_lifecycles(0, &choices) .unwrap() - .execution_timed_dag() + .export_timed_dag() .unwrap() } @@ -884,7 +888,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::SummaryAgg { + PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), .. } @@ -897,7 +901,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::SummaryAgg { + PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(..), .. } @@ -922,7 +926,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { ); // Execute the selected split across a state serialization boundary. // Each run builds fresh weights from that window's counters. - let execute = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDag, + let execute = |plan: &asap_physical_operators::physical_planner::PhysicalPostASAPDAG, input: Batch, scope: Scope| { let id = plan.input_contracts().next().unwrap().0; @@ -946,7 +950,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { }) }; let (family, input, grouping) = match &state.payload { - PostAsapOperatorPayload::SummaryAgg { + PostASAPOperatorPayload::SummaryAgg { family, input, grouping, diff --git a/crates/devtools/examples/canonical_examples.rs b/crates/devtools/examples/canonical_examples.rs index c7d08924..67112cc7 100644 --- a/crates/devtools/examples/canonical_examples.rs +++ b/crates/devtools/examples/canonical_examples.rs @@ -1,6 +1,6 @@ // cargo run -p asap-lower --example canonical_examples // -// One-off: pretty-print the QueryExpr for one canonical query per variant, +// One-off: pretty-print the PreASAPNode for one canonical query per variant, // plus custom Join/SetOp/Dedup/CTE probes, to eyeball the actual shape. use asap_devtools::lower_promql_with_data_ingestion_interval; diff --git a/crates/devtools/src/bin/analyze_corpora.rs b/crates/devtools/src/bin/analyze_corpora.rs index 503fb1f0..ee0082c6 100644 --- a/crates/devtools/src/bin/analyze_corpora.rs +++ b/crates/devtools/src/bin/analyze_corpora.rs @@ -218,7 +218,7 @@ fn run_corpus(name: &str, source: &str, interval_ms: u64) -> CorpusResult { normalized_expression, structural_shape, lowered: true, - ir: Some(serde_json::to_value(&ir).expect("QueryExpr must serialize")), + ir: Some(serde_json::to_value(&ir).expect("PreASAPNode must serialize")), ir_debug: Some(format!("{ir:#?}")), error: None, }), @@ -493,7 +493,7 @@ async fn run_sql_corpora(out_dir: PathBuf) { normalized_expression, structural_shape, lowered: true, - ir: Some(serde_json::to_value(&ir).expect("QueryExpr must serialize")), + ir: Some(serde_json::to_value(&ir).expect("PreASAPNode must serialize")), ir_debug: Some(format!("{ir:#?}")), error: None, }), diff --git a/crates/devtools/src/bin/dag_export.rs b/crates/devtools/src/bin/dag_export.rs index 8e6d3eca..e0ec02bb 100644 --- a/crates/devtools/src/bin/dag_export.rs +++ b/crates/devtools/src/bin/dag_export.rs @@ -22,7 +22,7 @@ // additionally runs `asap_aware_mapping::replacement::search_workload` (this // binary took no strategies of its own — `default_strategies()` already // includes `AvgToSumOverCountStrategy` as of #282) over every lowered query -// and ranks each discovered `TargetSubDAGCandidates` via `CandidatePostASAPDAGs::cost_sorted`. The +// and ranks each discovered `TargetSubDAGCandidates` via `CandidateLogicalPostASAPDAGs::cost_sorted`. The // best-ranked // candidate per group feeds two additive outputs: // @@ -55,7 +55,7 @@ use std::rc::Rc; use std::time::Instant; use asap_aware_mapping::analytical_cost::{ - cache_hit_ratios, AnalyticalCostError, EvidenceBackedPhysicalDag as PhysicalDag, + cache_hit_ratios, AnalyticalCostError, EvidenceBackedPhysicalDag as PhysicalExecution, PhysicalNodeEvidence, ResourceCalibration, ANALYTICAL_COST_MODEL_VERSION, }; #[cfg(test)] @@ -73,14 +73,14 @@ use asap_aware_mapping::replacement::{ use asap_aware_mapping::{AccuracyEvidenceProvider, PropagationStats}; use asap_types::cost::{BaselineRef, CostAnnotation, CostInput, CostSource, CostUnit}; use asap_types::dag_export::{ - self, DagDecision, DagGraph, DagNote, NamedGraph, PostAsapSubstitution, TargetRejection, + self, DagDecision, DagGraph, DagNote, NamedGraph, PostASAPSubstitution, TargetRejection, TargetReplacement, TargetReplacementAfter, WorkloadGraph, }; +use asap_types::post_asap::PostASAPNode; use asap_types::post_asap::SummaryExpr; -use asap_types::post_asap::SummaryNode; use asap_types::post_asap::{CompositionOperator, SketchQuery, SummaryFamilyType}; use asap_types::pre_asap::cse::{structural_hash, HashCache}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::pre_asap::query_expr::PreASAPNode; use asap_types::pre_asap::schema::{Column, DataType, Schema}; use asap_types::resources::CacheProfile; use asap_types::types::AccuracyTarget; @@ -116,7 +116,7 @@ fn parse_planner_cost_document(raw: &str) -> Result #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] struct TargetPhysicalEvidence { - target: QueryExpr, + target: PreASAPNode, scope: ComparisonScopeEvidence, candidates: Vec, } @@ -171,7 +171,7 @@ impl ComparisonScopeEvidence { #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] struct QueryNodePhysicalEvidence { - logical_node: QueryExpr, + logical_node: PreASAPNode, operator: asap_aware_mapping::analytical_cost::PhysicalOperator, occurrence: usize, synthetic: bool, @@ -188,7 +188,7 @@ enum CandidatePhysicalEvidence { Summary { plan: serde_json::Value, query_nodes: Vec, - physical_dag: PhysicalDag, + physical_dag: PhysicalExecution, }, } @@ -218,7 +218,7 @@ impl CandidatePhysicalEvidence { actual.is_ok_and(|actual| plan_values_match(&actual, self.plan())) } - fn summary_dag(&self) -> Option<&PhysicalDag> { + fn summary_dag(&self) -> Option<&PhysicalExecution> { match self { Self::Summary { physical_dag, .. } => Some(physical_dag), Self::Rewrite { .. } => None, @@ -348,9 +348,9 @@ impl PlannerPhysicalPlanProvider for ExportPhysicalProvider<'_> { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - _summary: &Rc, + _summary: &Rc, _target: &asap_aware_mapping::replacement::TargetSubDAG<'_>, - ) -> Result { + ) -> Result { if snapshot.scope != self.target.scope.resolve()? { return Err(AnalyticalCostError::ComparisonScopeMismatch( "planner evidence snapshot", @@ -408,7 +408,7 @@ impl ExportPlannerCostModel<'_> { fn annotations( &self, candidate: &ReplacementSubDAG, - target: &Rc, + target: &Rc, ) -> (CostAnnotation, CostAnnotation, CostAnnotation) { let target = asap_aware_mapping::replacement::TargetSubDAG::new(target); let Some((provider, calibration)) = self.bound(candidate, &target) else { @@ -941,12 +941,12 @@ fn annotate_with_explanations( } /// One `TargetSubDAGCandidates`'s best-ranked candidate, kept alongside its own `target` -/// — the unit both [`PostAsapResults::replacements`] and -/// [`PostAsapResults::post_graphs`] are built from, so the two outputs can +/// — the unit both [`PostASAPResults::replacements`] and +/// [`PostASAPResults::post_graphs`] are built from, so the two outputs can /// never disagree about which candidate won for a given target. #[allow(dead_code)] struct Winner<'a> { - target: &'a Rc, + target: &'a Rc, candidate: &'a ReplacementSubDAG, costs: (CostAnnotation, CostAnnotation, CostAnnotation), } @@ -1010,7 +1010,7 @@ fn lookup_winner( by_hash: &HashMap>, winners: &[Winner<'_>], cache: &mut HashCache, - expr: &QueryExpr, + expr: &PreASAPNode, ) -> Option { let hash = structural_hash(expr, cache); by_hash @@ -1087,7 +1087,7 @@ fn target_replacement( /// The two additive `--post-asap` outputs — see this file's top-of-file /// usage doc for what each is for. -struct PostAsapResults { +struct PostASAPResults { /// One `(query_name, TargetReplacement)` pair per discovered replacement /// site whose target node is found in that query's own exported graph. A /// target can in principle be reachable from more than one query's root @@ -1104,8 +1104,8 @@ struct PostAsapResults { rejections: Vec<(String, TargetRejection)>, } -fn raw_only_post_asap_results() -> PostAsapResults { - PostAsapResults { +fn raw_only_post_asap_results() -> PostASAPResults { + PostASAPResults { replacements: Vec::new(), post_graphs: Vec::new(), rejections: Vec::new(), @@ -1158,23 +1158,23 @@ fn assign_workload_node_ids(graphs: &mut [&mut DagGraph]) { /// `default_strategies()` — which includes `AvgToSumOverCountStrategy` as of /// #282 — is exactly the strategy set this binary wants; no custom list /// needed) over every lowered query, rank each discovered `TargetSubDAGCandidates` via -/// `CandidatePostASAPDAGs::global_selection`, and build both `--post-asap` outputs from the +/// `CandidateLogicalPostASAPDAGs::global_selection`, and build both `--post-asap` outputs from the /// exact same set of winning candidates (see [`Winner`]), so the flat /// `replacements` list and the merged `post_graph` can never disagree about /// which candidate won for a given target. #[allow(dead_code)] fn run_post_asap_with_progress( - lowered_queries: &[(String, String, QueryExpr)], + lowered_queries: &[(String, String, PreASAPNode)], progress: bool, cost_model: &dyn CostModel, export_model: Option<&ExportPlannerCostModel<'_>>, evidence: Option<&dyn AccuracyEvidenceProvider>, -) -> PostAsapResults { +) -> PostASAPResults { let mapping_started = Instant::now(); if progress { eprintln!("[3/4] ASAP-aware mapping is running…"); } - let roots: Vec<(String, Rc)> = lowered_queries + let roots: Vec<(String, Rc)> = lowered_queries .iter() .map(|(name, _, qe)| (name.clone(), Rc::new(qe.clone()))) .collect(); @@ -1188,7 +1188,7 @@ fn run_post_asap_with_progress( let selection = space.global_selection(cost_model); // A group's top candidate can be `keep_pre_asap`'s own conservative - // fallback — `Replacement::Summary(SummaryNode { expr: + // fallback — `Replacement::Summary(PostASAPNode { expr: // KeepPreAsap(Rc::new(target.clone())), .. })` — the *whole target* // wrapped as unbound, e.g. for a multi-measure/`HAVING`-bearing // aggregate, or (the case that actually surfaces this: `STDDEV_POP`/ @@ -1272,7 +1272,7 @@ fn run_post_asap_with_progress( } let post_started = Instant::now(); let mut post_graph_cache = HashCache::new(); - let mut find_winner = |expr: &QueryExpr| -> Option { + let mut find_winner = |expr: &PreASAPNode| -> Option { let i = lookup_winner(&by_hash, &winners, &mut post_graph_cache, expr)?; let winner = &winners[i]; let (baseline_cost, selected_cost, benefit) = winner.costs.clone(); @@ -1291,11 +1291,11 @@ fn run_post_asap_with_progress( benefit: Some(benefit), }; Some(match &winners[i].candidate.replacement { - Replacement::Rewrite(rc) => PostAsapSubstitution::Rewrite { + Replacement::Rewrite(rc) => PostASAPSubstitution::Rewrite { replacement: Rc::clone(rc), decision, }, - Replacement::Summary(rc) => PostAsapSubstitution::Summary { + Replacement::Summary(rc) => PostASAPSubstitution::Summary { replacement: Rc::clone(rc), decision, }, @@ -1411,7 +1411,7 @@ fn run_post_asap_with_progress( ); } - PostAsapResults { + PostASAPResults { replacements, post_graphs, rejections, @@ -1419,7 +1419,7 @@ fn run_post_asap_with_progress( } #[cfg(test)] -fn run_post_asap(lowered_queries: &[(String, String, QueryExpr)]) -> PostAsapResults { +fn run_post_asap(lowered_queries: &[(String, String, PreASAPNode)]) -> PostASAPResults { run_post_asap_with_progress(lowered_queries, false, &DefaultCostModel, None, None) } @@ -1639,19 +1639,19 @@ mod tests { use asap_devtools::PromqlError; use asap_types::pre_asap::{Column, DataType, Reduction, Schema, Source}; - fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Result { + fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Result { lower_promql_with_data_ingestion_interval(query, accuracy, 1_000) } - fn non_topk_query() -> QueryExpr { - QueryExpr::Aggregate { + fn non_topk_query() -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::by(vec![]), measures: vec![asap_types::pre_asap::AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.1), }], output_names: vec![], having: None, - child: Rc::new(QueryExpr::Scan { + child: Rc::new(PreASAPNode::Scan { source: Source::Table { table_ref: "events".into(), }, @@ -1662,10 +1662,10 @@ mod tests { } fn fixture_raw_dag( - query: &QueryExpr, + query: &PreASAPNode, candidate: &ReplacementSubDAG, document: &PlannerCostDocument, - ) -> PhysicalDag { + ) -> PhysicalExecution { let model = ExportPlannerCostModel { document }; let root = Rc::new(query.clone()); let target = asap_aware_mapping::replacement::TargetSubDAG::new(&root); @@ -2126,7 +2126,7 @@ mod tests { EdgeStatistics { rows, bytes } } - fn query_evidence(query: &QueryExpr) -> Vec { + fn query_evidence(query: &PreASAPNode) -> Vec { let entries = RefCell::new(Vec::new()); let scope = test_scope().resolve().unwrap(); let provider = |request: PhysicalNodeRequest<'_>| { @@ -2178,7 +2178,7 @@ mod tests { entries.into_inner() } - fn cheap_candidate_dag() -> PhysicalDag { + fn cheap_candidate_dag() -> PhysicalExecution { let coverage = test_scope().sources[0].clone(); let statistics = OperatorStatistics::Scan { edges: UnaryEdgeStatistics { @@ -2188,7 +2188,7 @@ mod tests { }, source_read_bytes: 64, }; - PhysicalDag { + PhysicalExecution { nodes: vec![PhysicalDagNode { id: "summary-read".into(), operator: PhysicalOperator::Scan, @@ -2226,7 +2226,7 @@ mod tests { } } - fn cost_fixture() -> (QueryExpr, ReplacementSubDAG, PlannerCostDocument) { + fn cost_fixture() -> (PreASAPNode, ReplacementSubDAG, PlannerCostDocument) { let query = non_topk_query(); let root = Rc::new(query.clone()); let space = search_workload(vec![(String::from("q"), Rc::clone(&root))]); diff --git a/crates/devtools/src/bin/show_post_asap_ir.rs b/crates/devtools/src/bin/show_post_asap_ir.rs index 18260b42..5f9e03c6 100644 --- a/crates/devtools/src/bin/show_post_asap_ir.rs +++ b/crates/devtools/src/bin/show_post_asap_ir.rs @@ -3,7 +3,7 @@ // // Lowers a batch of ad-hoc SQL/PromQL queries to pre-ASAP IR, then runs the // `asap-aware-mapping` pre-ASAP → post-ASAP binding pass and prints the -// resulting **post-ASAP IR** (the sketch-bound IR: `SummaryExpr`/`SummaryNode` +// resulting **post-ASAP IR** (the sketch-bound IR: `SummaryExpr`/`PostASAPNode` // — the concrete `SummaryKind`/`SummaryParams` committed per aggregate, or // `KeepPreAsap` for whatever the pass left untouched). See `show_pre_asap_ir` // for the sketch-agnostic IR one layer upstream. @@ -25,7 +25,7 @@ use asap_aware_mapping::{ Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, }; use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::pre_asap::query_expr::PreASAPNode; use asap_types::pre_asap::schema::{Column, DataType, Schema}; use asap_types::types::AccuracyTarget; use std::io::Read; @@ -36,7 +36,7 @@ const ACCURACY: AccuracyTarget = AccuracyTarget::Epsilon(0.01); /// `SketchAlgorithmStrategy::replacements` returns every candidate. This /// debug tool prints all of them so callers can inspect the planner's choices. /// If the strategy has none, preserve the single pre-ASAP fallback output. -fn bind_all(expr: &QueryExpr) -> Result>, String> { +fn bind_all(expr: &PreASAPNode) -> Result>, String> { let root = Rc::new(expr.clone()); let target = TargetSubDAG::new(&root); let candidates = SketchAlgorithmStrategy::default_cost_model() @@ -177,7 +177,7 @@ mod tests { candidates[0].guarantee.is_none(), "missing evidence must not claim a certified ratio bound" ); - asap_types::post_asap::compile_post_asap_dag(&candidates[0]) + asap_types::post_asap::export_post_asap_dag(&candidates[0]) .expect("the demo candidate remains executable"); } diff --git a/crates/devtools/src/bin/show_pre_asap_ir.rs b/crates/devtools/src/bin/show_pre_asap_ir.rs index b2f60463..806c1c1b 100644 --- a/crates/devtools/src/bin/show_pre_asap_ir.rs +++ b/crates/devtools/src/bin/show_pre_asap_ir.rs @@ -2,7 +2,7 @@ // (or pipe via stdin: cargo run -p asap-devtools --bin show_pre_asap_ir < queries.txt) // // Lowers a batch of ad-hoc SQL/PromQL queries to **pre-ASAP IR** (the -// sketch-agnostic intent algebra: `QueryExpr`/`AggIntent`) and prints them. +// sketch-agnostic intent algebra: `PreASAPNode`/`AggIntent`) and prints them. // See `show_post_asap_ir` for the post-ASAP sketch-bound IR one layer // downstream — this tool never picks a sketch, it only shows what a query // means. diff --git a/crates/devtools/src/bin/sketch_coverage.rs b/crates/devtools/src/bin/sketch_coverage.rs index e9f0ccb1..fa24a950 100644 --- a/crates/devtools/src/bin/sketch_coverage.rs +++ b/crates/devtools/src/bin/sketch_coverage.rs @@ -29,7 +29,7 @@ use asap_aware_mapping::{explain_replacements, ExplanationKind}; use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::QueryExpr; +use asap_types::pre_asap::PreASAPNode; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use std::collections::BTreeSet; @@ -157,7 +157,7 @@ fn root_label(id: &str) -> String { /// reachable from. fn analyze_corpus( name: &'static str, - roots: Vec<(String, QueryExpr)>, + roots: Vec<(String, PreASAPNode)>, failed: usize, ) -> CorpusCoverage { let lowered = roots.len(); diff --git a/crates/devtools/src/bin/variant_coverage.rs b/crates/devtools/src/bin/variant_coverage.rs index a96042a2..a354130a 100644 --- a/crates/devtools/src/bin/variant_coverage.rs +++ b/crates/devtools/src/bin/variant_coverage.rs @@ -1,13 +1,13 @@ // cargo run -p asap-lower --bin variant_coverage // // Lowers every query in every corpus we have (PromQL + SQL), walks the -// resulting QueryExpr trees, and reports which enum variants show up — per -// corpus, then rolled up globally. Used to find the minimal QueryExpr node set. +// resulting PreASAPNode trees, and reports which enum variants show up — per +// corpus, then rolled up globally. Used to find the minimal PreASAPNode node set. use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::QueryExpr; +use asap_types::pre_asap::PreASAPNode; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use std::collections::BTreeSet; @@ -38,114 +38,114 @@ const ALL_VARIANTS: &[&str] = &[ "BinaryOp", ]; -fn walk(e: &QueryExpr, seen: &mut BTreeSet<&'static str>) { +fn walk(e: &PreASAPNode, seen: &mut BTreeSet<&'static str>) { match e { - QueryExpr::Scan { .. } => { + PreASAPNode::Scan { .. } => { seen.insert("Scan"); } - QueryExpr::PromqlScalarBridge(_) => { + PreASAPNode::PromqlScalarBridge(_) => { seen.insert("PromqlScalarBridge"); } - QueryExpr::EvalTimestamp => { + PreASAPNode::EvalTimestamp => { seen.insert("EvalTimestamp"); } - QueryExpr::CurrentTimestamp => { + PreASAPNode::CurrentTimestamp => { seen.insert("CurrentTimestamp"); } - QueryExpr::PromqlVectorFromScalar(inner) => { + PreASAPNode::PromqlVectorFromScalar(inner) => { seen.insert("PromqlVectorFromScalar"); walk(inner, seen); } - QueryExpr::PromqlScalarFromVector(inner) => { + PreASAPNode::PromqlScalarFromVector(inner) => { seen.insert("PromqlScalarFromVector"); walk(inner, seen); } - QueryExpr::PromqlRelabel { child, .. } => { + PreASAPNode::PromqlRelabel { child, .. } => { seen.insert("PromqlRelabel"); walk(child, seen); } - QueryExpr::PromqlInfoEnrich { child, .. } => { + PreASAPNode::PromqlInfoEnrich { child, .. } => { seen.insert("PromqlInfoEnrich"); walk(child, seen); } - QueryExpr::PromqlSeriesSample { child, .. } => { + PreASAPNode::PromqlSeriesSample { child, .. } => { seen.insert("PromqlSeriesSample"); walk(child, seen); } - QueryExpr::Filter { child, .. } => { + PreASAPNode::Filter { child, .. } => { seen.insert("Filter"); walk(child, seen); } - QueryExpr::Project { child, .. } => { + PreASAPNode::Project { child, .. } => { seen.insert("Project"); walk(child, seen); } - QueryExpr::Aggregate { child, .. } => { + PreASAPNode::Aggregate { child, .. } => { seen.insert("Aggregate"); walk(child, seen); } - QueryExpr::Dedup { child, .. } => { + PreASAPNode::Dedup { child, .. } => { seen.insert("Dedup"); walk(child, seen); } - QueryExpr::Concat { children, .. } => { + PreASAPNode::Concat { children, .. } => { seen.insert("Concat"); children.iter().for_each(|c| walk(c, seen)); } - QueryExpr::Join { left, right, .. } => { + PreASAPNode::Join { left, right, .. } => { seen.insert("Join"); walk(left, seen); walk(right, seen); } - QueryExpr::SetOp { left, right, .. } => { + PreASAPNode::SetOp { left, right, .. } => { seen.insert("SetOp"); walk(left, seen); walk(right, seen); } - QueryExpr::Sort { child, .. } => { + PreASAPNode::Sort { child, .. } => { seen.insert("Sort"); walk(child, seen); } - QueryExpr::Limit { child, .. } => { + PreASAPNode::Limit { child, .. } => { seen.insert("Limit"); walk(child, seen); } - QueryExpr::PromqlSubquery { child, .. } => { + PreASAPNode::PromqlSubquery { child, .. } => { seen.insert("PromqlSubquery"); walk(child, seen); } - QueryExpr::TimeRange { child, .. } => { + PreASAPNode::TimeRange { child, .. } => { seen.insert("TimeRange"); walk(child, seen); } - QueryExpr::TimeShift { child, .. } => { + PreASAPNode::TimeShift { child, .. } => { seen.insert("TimeShift"); walk(child, seen); } - QueryExpr::SQLWindowFunc { child, .. } => { + PreASAPNode::SQLWindowFunc { child, .. } => { seen.insert("SQLWindowFunc"); walk(child, seen); } - QueryExpr::BinaryOp { lhs, rhs, .. } => { + PreASAPNode::BinaryOp { lhs, rhs, .. } => { seen.insert("BinaryOp"); walk(lhs, seen); walk(rhs, seen); } // Scalar expression variants (issue #205) aren't relational nodes; // this walk only reports on the relational skeleton, so stop here. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} + PreASAPNode::Column(_) + | PreASAPNode::Literal(_) + | PreASAPNode::Compare { .. } + | PreASAPNode::BoolAnd(_) + | PreASAPNode::BoolOr(_) + | PreASAPNode::Not(_) + | PreASAPNode::IsNull(_) + | PreASAPNode::IsNotNull(_) + | PreASAPNode::Cast { .. } + | PreASAPNode::InList { .. } + | PreASAPNode::FunctionCall { .. } + | PreASAPNode::Arithmetic { .. } + | PreASAPNode::Case { .. } => {} } } diff --git a/crates/devtools/src/lib.rs b/crates/devtools/src/lib.rs index 5e6a0208..f212493d 100644 --- a/crates/devtools/src/lib.rs +++ b/crates/devtools/src/lib.rs @@ -23,7 +23,7 @@ pub fn lower_promql_with_data_ingestion_interval( query: &str, accuracy: asap_types::types::AccuracyTarget, interval_ms: u64, -) -> Result { +) -> Result { use asap_types::workload::{ BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryRequirements, QueryWorkload, TimeSelection, diff --git a/crates/devtools/tests/cross_language.rs b/crates/devtools/tests/cross_language.rs index 0518f62b..051bc413 100644 --- a/crates/devtools/tests/cross_language.rs +++ b/crates/devtools/tests/cross_language.rs @@ -15,7 +15,7 @@ use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::pre_asap::{AggIntent, GroupKeys, PreASAPNode}; use asap_types::types::AccuracyTarget; fn col(name: &str, dtype: DataType) -> Column { @@ -39,21 +39,21 @@ fn catalog() -> SqlCatalog { ) } -async fn sql(q: &str) -> QueryExpr { +async fn sql(q: &str) -> PreASAPNode { lower_sql(q, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|e| panic!("SQL {q:?} failed to lower: {e:?}")) } -fn promql(q: &str) -> QueryExpr { +fn promql(q: &str) -> PreASAPNode { lower_promql_with_data_ingestion_interval(q, AccuracyTarget::Exact, 1_000) .unwrap_or_else(|e| panic!("PromQL {q:?} failed to lower: {e:?}")) } /// The canonical heavy-hitter shape: an outer `Aggregate([TopK{k}])` (grouped by /// `by`) over an inner `Aggregate([Count])`. Returns `(k, outer_by)`. -fn heavy_hitter(qe: &QueryExpr) -> Option<(usize, GroupKeys)> { - let QueryExpr::Aggregate { +fn heavy_hitter(qe: &PreASAPNode) -> Option<(usize, GroupKeys)> { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -67,7 +67,7 @@ fn heavy_hitter(qe: &QueryExpr) -> Option<(usize, GroupKeys)> { }; // The child must be the explicit inner Count (not a raw Scan) — this is the // structural unification #25 asked for. - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures: inner, .. } = child.as_ref() else { @@ -151,19 +151,19 @@ async fn ascending_count_ranked_topk_stays_generic_in_both_languages() { ); // Both are the generic order-by-value + limit shape. assert!( - matches!(&s, QueryExpr::Limit { .. }), + matches!(&s, PreASAPNode::Limit { .. }), "SQL stays a Limit: {s:?}" ); assert!( - matches!(&p, QueryExpr::Limit { .. }), + matches!(&p, PreASAPNode::Limit { .. }), "PromQL stays a Limit: {p:?}" ); } /// Descend through a leading `Project` (the derived-table SELECT list). -fn strip_project(qe: &QueryExpr) -> &QueryExpr { +fn strip_project(qe: &PreASAPNode) -> &PreASAPNode { match qe { - QueryExpr::Project { child, .. } => strip_project(child), + PreASAPNode::Project { child, .. } => strip_project(child), other => other, } } @@ -202,10 +202,10 @@ async fn sql_rownumber_avg_topk_is_a_generic_partitioned_sort_limit() { heavy_hitter(strip_project(&s9)).is_none(), "AVG-ranked is not a heavy-hitter" ); - let QueryExpr::Limit { child, .. } = strip_project(&s9) else { + let PreASAPNode::Limit { child, .. } = strip_project(&s9) else { panic!("expected a Limit, got {:?}", strip_project(&s9)); }; - let QueryExpr::Sort { partition_by, .. } = child.as_ref() else { + let PreASAPNode::Sort { partition_by, .. } = child.as_ref() else { panic!("expected a Sort under the Limit"); }; assert!(!partition_by.is_empty(), "partitioned by region"); diff --git a/crates/frontend-metricsql/src/lib.rs b/crates/frontend-metricsql/src/lib.rs index 4d07f64d..609200dd 100644 --- a/crates/frontend-metricsql/src/lib.rs +++ b/crates/frontend-metricsql/src/lib.rs @@ -1,11 +1,11 @@ -//! MetricsQL AST to canonical `QueryExpr` frontend. +//! MetricsQL AST to canonical `PreASAPNode` frontend. use std::{rc::Rc, time::Duration}; use asap_types::pre_asap::{ resolve_root, AggIntent, ArithmeticOpKind, BinaryOpKind, ColumnRef, CompareOpKind, GroupKeys, - Predicate, PromQLVectorSetOpKind, QueryExpr, Reduction, ScalarValue, Source, - UnresolvedQueryExpr as U, + PreASAPNode, Predicate, PromQLVectorSetOpKind, Reduction, ScalarValue, Source, + UnresolvedPreASAPNode as U, }; use asap_types::types::AccuracyTarget; use metricsql_parser::ast::{AggregateModifier, DurationExpr, Expr, MetricExpr, RollupExpr}; @@ -33,7 +33,10 @@ pub fn canonical_metricsql(query: &str) -> Result { Ok(parse_metricsql(query)?.to_string()) } -pub fn lower_metricsql(query: &str, accuracy: AccuracyTarget) -> Result { +pub fn lower_metricsql( + query: &str, + accuracy: AccuracyTarget, +) -> Result { let ast = parse_metricsql(query)?; let unresolved = Lowerer { accuracy }.lower(&ast)?; resolve_root(&unresolved).map_err(|e| MetricsqlError::Resolve(e.to_string())) diff --git a/crates/frontend-metricsql/tests/lowering.rs b/crates/frontend-metricsql/tests/lowering.rs index 133fc328..8de0018f 100644 --- a/crates/frontend-metricsql/tests/lowering.rs +++ b/crates/frontend-metricsql/tests/lowering.rs @@ -3,10 +3,10 @@ use std::time::Duration; use asap_frontend_metricsql::{ canonical_metricsql, lower_metricsql, parse_metricsql, MetricsqlError, }; -use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction, Source}; +use asap_types::pre_asap::{AggIntent, PreASAPNode, Reduction, Source}; use asap_types::types::AccuracyTarget; -fn lower(query: &str) -> QueryExpr { +fn lower(query: &str) -> PreASAPNode { lower_metricsql(query, AccuracyTarget::Epsilon(0.01)).unwrap() } @@ -14,7 +14,7 @@ fn lower(query: &str) -> QueryExpr { fn selector_range_aggregate_and_call_share_the_canonical_shape() { let query = r#"sum by (job) (rate(http_requests_total{status=~"5.."}[5m]))"#; let tree = lower(query); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -25,26 +25,26 @@ fn selector_range_aggregate_and_call_share_the_canonical_shape() { }; assert_eq!(reduction, Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = child.as_ref() else { panic!("expected rate aggregate"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let PreASAPNode::TimeRange { range, child } = child.as_ref() else { panic!("expected range"); }; assert_eq!(*range, Duration::from_secs(300)); assert!( - matches!(child.as_ref(), QueryExpr::Scan { source: Source::TimeSeries { metric }, predicates, .. } if metric == "http_requests_total" && predicates.len() == 1) + matches!(child.as_ref(), PreASAPNode::Scan { source: Source::TimeSeries { metric }, predicates, .. } if metric == "http_requests_total" && predicates.len() == 1) ); } #[test] fn default_rollup_with_explicit_range_is_last_over_time() { let tree = lower("default_rollup(cpu_usage[5m])"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -56,7 +56,7 @@ fn default_rollup_with_explicit_range_is_last_over_time() { assert_eq!(reduction, Reduction::PerEntity); assert!(matches!(measures.as_slice(), [AggIntent::LastOverTime])); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, .. } if *range == Duration::from_secs(300)) + matches!(child.as_ref(), PreASAPNode::TimeRange { range, .. } if *range == Duration::from_secs(300)) ); } @@ -147,7 +147,7 @@ fn metricsql_multi_argument_aggregates_fail_closed() { #[test] fn supported_parameterized_functions_require_their_exact_arity() { let quantile = lower("quantile(0.9, requests_total)"); - assert!(matches!(quantile, QueryExpr::Aggregate { .. })); + assert!(matches!(quantile, PreASAPNode::Aggregate { .. })); let rollup = lower("quantile_over_time(0.9, requests_total[5m])"); - assert!(matches!(rollup, QueryExpr::Aggregate { .. })); + assert!(matches!(rollup, PreASAPNode::Aggregate { .. })); } diff --git a/crates/frontend-promql/src/lib.rs b/crates/frontend-promql/src/lib.rs index 60254d3d..8ffa25e4 100644 --- a/crates/frontend-promql/src/lib.rs +++ b/crates/frontend-promql/src/lib.rs @@ -1,8 +1,8 @@ //! PromQL front end: parse (via `promql-parser`) → the canonical, unresolved //! shape, built directly (issue #179) → [`resolve_root`]. //! -//! Emits [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) itself — the -//! canonical `QueryExpr`, generic over an unresolved +//! Emits [`UnresolvedPreASAPNode`](asap_types::pre_asap::UnresolvedPreASAPNode) itself — the +//! canonical `PreASAPNode`, generic over an unresolved //! [`ColumnRef`](asap_types::pre_asap::ColumnRef) — directly, rather than a //! separate per-language relational tree; `resolve_root` runs the //! [`SchemaResolver`](asap_types::pre_asap::SchemaResolver) for positional name resolution. @@ -13,13 +13,13 @@ pub mod histogram; pub mod promql; use asap_types::pre_asap::resolve_root; -use asap_types::pre_asap::QueryExpr; +use asap_types::pre_asap::PreASAPNode; use asap_types::workload::{DurationMs, PlanningWorkload, QueryLanguage, WorkloadError}; pub use error::PromqlError; pub use histogram::{HistogramCatalog, HistogramKind}; -/// Lower every normalized PromQL workload entry to a plan-ready `QueryExpr`. +/// Lower every normalized PromQL workload entry to a plan-ready `PreASAPNode`. /// /// PromQL workloads must declare a non-zero `data_ingestion_interval`; it is /// injected around each bare instant selector. Explicit range selectors keep @@ -29,7 +29,7 @@ pub use histogram::{HistogramCatalog, HistogramKind}; pub fn lower_promql_workload( workload: &PlanningWorkload, now_ms: u64, -) -> Result, PromqlError> { +) -> Result, PromqlError> { lower_promql_workload_inner(workload, now_ms) } @@ -39,7 +39,7 @@ pub fn lower_promql_workload_with_histograms( workload: &PlanningWorkload, histograms: HistogramCatalog, now_ms: u64, -) -> Result, PromqlError> { +) -> Result, PromqlError> { let _guard = histogram::CatalogGuard::install(histograms); lower_promql_workload_inner(workload, now_ms) } @@ -47,7 +47,7 @@ pub fn lower_promql_workload_with_histograms( fn lower_promql_workload_inner( workload: &PlanningWorkload, now_ms: u64, -) -> Result, PromqlError> { +) -> Result, PromqlError> { if !matches!(workload.query_workload.language, QueryLanguage::PromQL) { return Err(PromqlError::WrongLanguage(format!( "{:?}", @@ -123,7 +123,7 @@ mod tests { } use std::time::Duration; - use asap_types::pre_asap::QueryExpr; + use asap_types::pre_asap::PreASAPNode; use asap_types::workload::{ BatchEntry, DataWorkload, Evidence, PlanningWorkload, Query, QueryRequirements, QueryWorkload, TimeSelection, @@ -158,24 +158,24 @@ mod tests { #[test] fn instant_selector_uses_declared_ingestion_interval() { let query = lower_promql_workload(&workload("sum by (job) (data)"), 0).unwrap(); - let QueryExpr::Aggregate { child, .. } = &query[0] else { + let PreASAPNode::Aggregate { child, .. } = &query[0] else { panic!("expected aggregate") }; assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, child } - if *range == Duration::from_secs(1) && matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.as_ref(), PreASAPNode::TimeRange { range, child } + if *range == Duration::from_secs(1) && matches!(child.as_ref(), PreASAPNode::Scan { .. })) ); } #[test] fn explicit_range_selector_keeps_its_query_range() { let query = lower_promql_workload(&workload("sum_over_time(data[5m])"), 0).unwrap(); - let QueryExpr::Aggregate { child, .. } = &query[0] else { + let PreASAPNode::Aggregate { child, .. } = &query[0] else { panic!("expected aggregate") }; assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, child } - if *range == Duration::from_secs(300) && matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.as_ref(), PreASAPNode::TimeRange { range, child } + if *range == Duration::from_secs(300) && matches!(child.as_ref(), PreASAPNode::Scan { .. })) ); } diff --git a/crates/frontend-promql/src/promql.rs b/crates/frontend-promql/src/promql.rs index 920acfa7..7564fe5e 100644 --- a/crates/frontend-promql/src/promql.rs +++ b/crates/frontend-promql/src/promql.rs @@ -1,14 +1,14 @@ //! PromQL string → the canonical, unresolved -//! [`UnresolvedQueryExpr`](asap_types::pre_asap::query_expr::UnresolvedQueryExpr) -//! (`QueryExpr`). +//! [`UnresolvedPreASAPNode`](asap_types::pre_asap::query_expr::UnresolvedPreASAPNode) +//! (`PreASAPNode`). //! //! - **Parsing** is delegated to `promql-parser` 0.8. //! - **Lowering** builds *directly in canonical shape* here (issue #179): the //! walk interprets PromQL semantics (range vectors, aggregate operators, -//! label matchers) and emits `UnresolvedQueryExpr` nodes with unresolved +//! label matchers) and emits `UnresolvedPreASAPNode` nodes with unresolved //! `ColumnRef`s — the same tree shape //! [`resolve_root`](asap_types::pre_asap::resolve_root) later binds to -//! canonical, positional `QueryExpr`. The structural decisions a +//! canonical, positional `PreASAPNode`. The structural decisions a //! separate converter stage would otherwise have to make (heavy-hitter //! `topk` recognition, the `PerEntity`/`Reduce` reduction choice, //! `without(...)` grouping) are made right here, since a front end @@ -67,7 +67,7 @@ use promql_parser::parser::{ use asap_types::pre_asap::agg_intent::{topk, AggIntent, MathFunc, TimeFunc}; use asap_types::pre_asap::query_expr::{ AtModifier, BinaryOpKind, GroupKeys, GroupSide, Predicate, PromQLVectorSetOpKind, Reduction, - SortKey, Source, TimeShift, UnresolvedQueryExpr as Unresolved, VectorGrouping, VectorMatch, + SortKey, Source, TimeShift, UnresolvedPreASAPNode as Unresolved, VectorGrouping, VectorMatch, VectorMatchKind, }; use asap_types::pre_asap::{ @@ -576,7 +576,7 @@ fn outer_kind(agg: &AggregateExpr) -> Result { /// build this node) decides `PerEntity` vs `Reduce(by)` *without* knowing /// about `without` yet — it only ever sees `by`-mode keys, since `without`'s /// excluded-labels list is applied here, after the fact, exactly like the -/// pre-#179 legacy `relational::QueryExpr` tree's own `mark_without` did (its +/// pre-#179 legacy `relational::PreASAPNode` tree's own `mark_without` did (its /// converter read `without` only after this front-end step had already set /// it). Whether /// `reduction_for` picked `PerEntity` (only possible when `keys` was empty) diff --git a/crates/frontend-promql/tests/count_planning.rs b/crates/frontend-promql/tests/count_planning.rs index fded3ceb..6e7088d1 100644 --- a/crates/frontend-promql/tests/count_planning.rs +++ b/crates/frontend-promql/tests/count_planning.rs @@ -9,7 +9,7 @@ use asap_aware_mapping::{ }; mod support; use asap_types::post_asap::{ - compile_post_asap_dag, ExactKind, NonNegativeWeightProof, PostAsapOperatorPayload, + export_post_asap_dag, ExactKind, NonNegativeWeightProof, PostASAPOperatorPayload, SketchAlgorithm, SummaryExpr, SummaryFamilyType, SummaryInputExpr, WeightDomain, }; use asap_types::types::AccuracyTarget; @@ -27,7 +27,7 @@ fn grouped_count_keeps_uncertified_hydra_candidates_for_backend_review() { &default_strategies(), &DefaultAccuracyModel, ); - let planned = &space.roots[0].1; + let planned = &space.roots()[0].1; let hydra: Vec<_> = space .candidates_for_target(planned) .unwrap() @@ -98,11 +98,11 @@ fn frequency_count_candidates_use_unit_weights() { ) { continue; } - let dag = compile_post_asap_dag(node).unwrap(); + let dag = export_post_asap_dag(node).unwrap(); assert!( dag.nodes.iter().any(|node| matches!( &node.payload, - PostAsapOperatorPayload::SummaryAgg { input: actual, .. } if actual == input + PostASAPOperatorPayload::SummaryAgg { input: actual, .. } if actual == input )), "post-ASAP DAG must preserve the count update contract" ); @@ -135,9 +135,9 @@ fn frequency_count_candidates_use_unit_weights() { // This narrow test oracle interprets the emitted aggregate, not Prometheus ingestion, // staleness, or scrape scheduling. Unsupported plan shapes fail explicitly. fn aggregate_fixture(query: &str, series: &[Vec]) -> Vec { - use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction}; + use asap_types::pre_asap::{AggIntent, PreASAPNode, Reduction}; let root = lower_promql(query, AccuracyTarget::Exact).unwrap(); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -147,13 +147,13 @@ fn aggregate_fixture(query: &str, series: &[Vec]) -> Vec { panic!("expected aggregate: {root:?}"); }; match child.as_ref() { - QueryExpr::Scan { .. } => assert!(series.iter().all(|samples| samples.len() == 1)), - QueryExpr::TimeRange { range, child } => { + PreASAPNode::Scan { .. } => assert!(series.iter().all(|samples| samples.len() == 1)), + PreASAPNode::TimeRange { range, child } => { assert!(matches!(range.as_secs(), 1 | 300)); if range.as_secs() == 1 { assert!(series.iter().all(|samples| samples.len() == 1)); } - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::Scan { .. })); } other => panic!("unsupported fixture input: {other:?}"), } @@ -235,12 +235,12 @@ fn cms_count_updates_total_ten_for_zero_positive_and_negative_samples() { let Replacement::Summary(node) = &candidate.replacement else { return None; }; - let dag = compile_post_asap_dag(node).unwrap(); + let dag = export_post_asap_dag(node).unwrap(); dag.nodes .iter() .any(|node| { matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } + PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::Cms) }) .then_some(dag) @@ -250,7 +250,7 @@ fn cms_count_updates_total_ten_for_zero_positive_and_negative_samples() { .nodes .iter() .find_map(|node| match &node.payload { - PostAsapOperatorPayload::SummaryAgg { input, .. } => Some(input), + PostASAPOperatorPayload::SummaryAgg { input, .. } => Some(input), _ => None, }) .unwrap(); diff --git a/crates/frontend-promql/tests/histogram_metadata.rs b/crates/frontend-promql/tests/histogram_metadata.rs index 60ff5a60..d2b8a68b 100644 --- a/crates/frontend-promql/tests/histogram_metadata.rs +++ b/crates/frontend-promql/tests/histogram_metadata.rs @@ -7,16 +7,16 @@ use asap_frontend_promql::{HistogramCatalog, HistogramKind}; mod support; -use asap_types::pre_asap::{AggIntent, QueryExpr}; +use asap_types::pre_asap::{AggIntent, PreASAPNode}; use asap_types::types::AccuracyTarget; use support::{lower_promql, lower_promql_with_histograms}; /// The histogram/quantile intent kind in the lowered tree: `"HQ"` for the /// classic-bucket `HistogramQuantile`, `"Q"` for the sketch-able `Quantile`. -fn quantile_kind(qe: &QueryExpr) -> &'static str { - fn walk(e: &QueryExpr) -> Option<&'static str> { +fn quantile_kind(qe: &PreASAPNode) -> &'static str { + fn walk(e: &PreASAPNode) -> Option<&'static str> { match e { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { measures, child, .. } => measures .iter() @@ -26,12 +26,12 @@ fn quantile_kind(qe: &QueryExpr) -> &'static str { _ => None, }) .or_else(|| walk(child)), - QueryExpr::TimeRange { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Project { child, .. } => walk(child), + PreASAPNode::TimeRange { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } + | PreASAPNode::Project { child, .. } => walk(child), _ => None, } } diff --git a/crates/frontend-promql/tests/maintained_population_horizon.rs b/crates/frontend-promql/tests/maintained_population_horizon.rs index 88b1c88f..27a4ddfe 100644 --- a/crates/frontend-promql/tests/maintained_population_horizon.rs +++ b/crates/frontend-promql/tests/maintained_population_horizon.rs @@ -26,8 +26,8 @@ fn population_preserves_selector_horizon() { panic!() }; assert_eq!(spec.lookback_ms, 1_000); - asap_types::post_asap::compile_post_asap_dag(&candidate).unwrap(); - let asap_types::pre_asap::QueryExpr::Aggregate { child: source, .. } = root.as_ref() else { + asap_types::post_asap::export_post_asap_dag(&candidate).unwrap(); + let asap_types::pre_asap::PreASAPNode::Aggregate { child: source, .. } = root.as_ref() else { panic!() }; assert!(spec.matches_input(source)); diff --git a/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs b/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs index 12afb94d..ab179465 100644 --- a/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs +++ b/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs @@ -29,7 +29,7 @@ use asap_frontend_promql::PromqlError as LoweringError; #[path = "../support.rs"] mod support; -use asap_types::pre_asap::{AggIntent, BinaryOpKind, CompareOpKind, QueryExpr, Reduction}; +use asap_types::pre_asap::{AggIntent, BinaryOpKind, CompareOpKind, PreASAPNode, Reduction}; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -44,41 +44,41 @@ fn queries() -> impl Iterator { } /// Lower, expecting success. -fn ok(q: &str) -> QueryExpr { +fn ok(q: &str) -> PreASAPNode { lower_promql(q, AccuracyTarget::Exact) .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) } /// Every `AggIntent` in the tree. -fn intents(e: &QueryExpr) -> Vec { +fn intents(e: &PreASAPNode) -> Vec { let mut out = Vec::new(); - fn go(e: &QueryExpr, out: &mut Vec) { + fn go(e: &PreASAPNode, out: &mut Vec) { match e { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { measures, child, .. } => { out.extend(measures.iter().cloned()); go(child, out); } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::Project { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => go(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { + PreASAPNode::TimeRange { child, .. } + | PreASAPNode::TimeShift { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::SQLWindowFunc { child, .. } + | PreASAPNode::Project { child, .. } + | PreASAPNode::PromqlRelabel { child, .. } + | PreASAPNode::PromqlSeriesSample { child, .. } + | PreASAPNode::PromqlInfoEnrich { child, .. } => go(child, out), + PreASAPNode::BinaryOp { lhs, rhs, .. } + | PreASAPNode::Join { left: lhs, right: rhs, .. } - | QueryExpr::SetOp { + | PreASAPNode::SetOp { left: lhs, right: rhs, .. @@ -86,36 +86,35 @@ fn intents(e: &QueryExpr) -> Vec { go(lhs, out); go(rhs, out); } - QueryExpr::Concat { children, .. } => children.iter().for_each(|c| go(c, out)), - QueryExpr::PromqlVectorFromScalar(inner) | QueryExpr::PromqlScalarFromVector(inner) => { - go(inner, out) - } + PreASAPNode::Concat { children, .. } => children.iter().for_each(|c| go(c, out)), + PreASAPNode::PromqlVectorFromScalar(inner) + | PreASAPNode::PromqlScalarFromVector(inner) => go(inner, out), // `AggIntent` only ever lives in `Aggregate.measures`, never in a // scalar position (issue #205) — nothing to collect there. - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} + PreASAPNode::Scan { .. } + | PreASAPNode::PromqlScalarBridge(_) + | PreASAPNode::EvalTimestamp + | PreASAPNode::CurrentTimestamp => {} + PreASAPNode::Column(_) + | PreASAPNode::Literal(_) + | PreASAPNode::Compare { .. } + | PreASAPNode::BoolAnd(_) + | PreASAPNode::BoolOr(_) + | PreASAPNode::Not(_) + | PreASAPNode::IsNull(_) + | PreASAPNode::IsNotNull(_) + | PreASAPNode::Cast { .. } + | PreASAPNode::InList { .. } + | PreASAPNode::FunctionCall { .. } + | PreASAPNode::Arithmetic { .. } + | PreASAPNode::Case { .. } => {} } } go(e, &mut out); out } -fn has bool>(e: &QueryExpr, p: F) -> bool { +fn has bool>(e: &PreASAPNode, p: F) -> bool { intents(e).iter().any(p) } @@ -180,15 +179,15 @@ fn vector_vs_vector_comparison_lowers_to_binaryop() { // Both operands are instant vectors → a `BinaryOp{Compare}` of two // ingestion-interval-bounded scans. let qe = ok("node_hwmon_temp_celsius > node_hwmon_temp_max_celsius"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &qe else { + let PreASAPNode::BinaryOp { op, lhs, rhs, .. } = &qe else { panic!("expected BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Compare(CompareOpKind::Gt)); assert!( - matches!(lhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(lhs.as_ref(), PreASAPNode::TimeRange { child, .. } if matches!(child.as_ref(), PreASAPNode::Scan { .. })) ); assert!( - matches!(rhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(rhs.as_ref(), PreASAPNode::TimeRange { child, .. } if matches!(child.as_ref(), PreASAPNode::Scan { .. })) ); } @@ -198,7 +197,7 @@ fn kube_replica_mismatch_comparison_lowers() { let qe = ok("kube_replicaset_spec_replicas != kube_replicaset_status_ready_replicas"); assert!(matches!( &qe, - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Ne) + PreASAPNode::BinaryOp { op, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Ne) )); } @@ -225,7 +224,7 @@ fn error_ratio_core_lowers() { // threshold: `sum(rate(failed[5m])) / sum(rate(total[5m]))` → a `BinaryOp(Div)` // of two cross-series sums over per-series rates. let qe = ok("sum(rate(litellm_proxy_failed_requests_metric_total[5m])) / sum(rate(litellm_proxy_total_requests_metric_total[5m]))"); - let QueryExpr::BinaryOp { op, .. } = &qe else { + let PreASAPNode::BinaryOp { op, .. } = &qe else { panic!("expected BinaryOp, got {qe:?}"); }; assert!(matches!(op, BinaryOpKind::Arithmetic(_))); @@ -250,7 +249,7 @@ fn all_targets_missing_core_lowers() { // Prometheus self-monitoring `sum by (job) (up)` (the corpus query is // `… == 0`). Cross-series sum grouped positionally on `job`. let qe = ok("sum by (job) (up)"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, .. @@ -280,11 +279,11 @@ fn scalar_threshold_comparisons_lower_to_binaryop_scalar() { "increase(prometheus_tsdb_compactions_failed_total[1m]) > 0", "rate(alertmanager_notifications_failed_total[3m]) > 0.05", ] { - let QueryExpr::BinaryOp { rhs, .. } = ok(q) else { + let PreASAPNode::BinaryOp { rhs, .. } = ok(q) else { panic!("expected a BinaryOp for {q:?}"); }; assert!( - matches!(rhs.as_ref(), QueryExpr::PromqlScalarBridge(_)), + matches!(rhs.as_ref(), PreASAPNode::PromqlScalarBridge(_)), "scalar threshold operand for {q:?}, got {rhs:?}" ); } @@ -336,7 +335,7 @@ fn vector_literal_lowers_to_a_labelless_vector() { // `vector(1)` — used in dead-man's-switch ("always firing") alerts. Now // lowers to a `PromqlVectorFromScalar` over the scalar `1` (issue #48). let qe = ok("vector(1)"); - let QueryExpr::PromqlVectorFromScalar(inner) = &qe else { + let PreASAPNode::PromqlVectorFromScalar(inner) = &qe else { panic!("expected PromqlVectorFromScalar, got {qe:?}"); }; assert_eq!(inner.as_promql_scalar(), Some(1.0)); @@ -352,10 +351,10 @@ fn without_grouping_lowers_to_the_exclusion_form() { // labels are stored and the kept set is runtime-resolved (issue #39). let qe = ok(r#"(min without (cpu) (rate(node_cpu_seconds_total{mode="idle"}[1h]))) > 0.8"#); // Top level is the `> 0.8` comparison; the `min without (cpu)` is its LHS. - let QueryExpr::BinaryOp { lhs, .. } = &qe else { + let PreASAPNode::BinaryOp { lhs, .. } = &qe else { panic!("expected a comparison BinaryOp, got {qe:?}"); }; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, .. diff --git a/crates/frontend-promql/tests/observability/metrics_observability.rs b/crates/frontend-promql/tests/observability/metrics_observability.rs index 87d653ef..de08cdc4 100644 --- a/crates/frontend-promql/tests/observability/metrics_observability.rs +++ b/crates/frontend-promql/tests/observability/metrics_observability.rs @@ -13,8 +13,8 @@ use asap_aware_mapping::{ use asap_frontend_promql::PromqlError; #[path = "../support.rs"] mod support; -use asap_types::post_asap::{SummaryExpr, SummaryNode}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::post_asap::{PostASAPNode, SummaryExpr}; +use asap_types::pre_asap::query_expr::PreASAPNode; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -64,7 +64,7 @@ fn queries(corpus: &str) -> impl Iterator { .filter(|line| !line.is_empty() && !line.starts_with('#')) } -fn post_asap_candidate(expr: &QueryExpr) -> Result, RealizationError> { +fn post_asap_candidate(expr: &PreASAPNode) -> Result, RealizationError> { let root = Rc::new(expr.clone()); let target = TargetSubDAG::new(&root); match SketchAlgorithmStrategy::default_cost_model() diff --git a/crates/frontend-promql/tests/observability/promql_corpus.rs b/crates/frontend-promql/tests/observability/promql_corpus.rs index f265edb4..deb3a7e9 100644 --- a/crates/frontend-promql/tests/observability/promql_corpus.rs +++ b/crates/frontend-promql/tests/observability/promql_corpus.rs @@ -22,8 +22,8 @@ use asap_aware_mapping::{ use asap_frontend_promql::PromqlError as LoweringError; #[path = "../support.rs"] mod support; -use asap_types::post_asap::{SummaryExpr, SummaryNode}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::post_asap::{PostASAPNode, SummaryExpr}; +use asap_types::pre_asap::query_expr::PreASAPNode; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -33,7 +33,7 @@ use support::lower_promql; /// take-the-first-(`cost_model`-preferred)-candidate pattern so [`bind_tally`] /// gets one representative `Result` per query, matching what a totality /// check over the whole corpus wants. -fn bind(expr: &QueryExpr) -> Result, RealizationError> { +fn bind(expr: &PreASAPNode) -> Result, RealizationError> { let root = Rc::new(expr.clone()); let target = TargetSubDAG::new(&root); match SketchAlgorithmStrategy::default_cost_model() diff --git a/crates/frontend-promql/tests/promql_binding_regressions.rs b/crates/frontend-promql/tests/promql_binding_regressions.rs index b5edf604..01d78ee5 100644 --- a/crates/frontend-promql/tests/promql_binding_regressions.rs +++ b/crates/frontend-promql/tests/promql_binding_regressions.rs @@ -33,9 +33,9 @@ fn irate_and_rate_have_distinct_canonical_intents() { /// PromQL count counts series even when two sample values are equal. #[test] fn count_is_row_count_not_distinct_sample_value_count() { - use asap_types::pre_asap::{AggIntent, QueryExpr}; + use asap_types::pre_asap::{AggIntent, PreASAPNode}; let tree = lower_promql("count(smoke_gauge)", AccuracyTarget::Exact).unwrap(); - let QueryExpr::Aggregate { measures, .. } = tree else { + let PreASAPNode::Aggregate { measures, .. } = tree else { panic!("expected aggregate") }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); diff --git a/crates/frontend-promql/tests/promql_conformance.rs b/crates/frontend-promql/tests/promql_conformance.rs index 69fbb223..ab9f69f8 100644 --- a/crates/frontend-promql/tests/promql_conformance.rs +++ b/crates/frontend-promql/tests/promql_conformance.rs @@ -37,8 +37,8 @@ use asap_frontend_promql::PromqlError as LoweringError; mod support; use asap_types::pre_asap::schema::DataType; use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, MathFunc, - PromQLVectorSetOpKind, QueryExpr, Reduction, SampleKind, Source, TimeFunc, + AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, MathFunc, PreASAPNode, + PromQLVectorSetOpKind, Reduction, SampleKind, Source, TimeFunc, }; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -46,7 +46,7 @@ use support::lower_promql; // ── harness helpers ───────────────────────────────────────────────────────────── /// Lower, expecting success. -fn ok(q: &str) -> QueryExpr { +fn ok(q: &str) -> PreASAPNode { lower_promql(q, AccuracyTarget::Exact) .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) } @@ -60,71 +60,71 @@ fn rejected(q: &str) -> LoweringError { } /// Every `AggIntent` anywhere in the tree, root-to-leaf. -fn intents(e: &QueryExpr) -> Vec { +fn intents(e: &PreASAPNode) -> Vec { let mut out = Vec::new(); collect(e, &mut out); out } -fn collect(e: &QueryExpr, out: &mut Vec) { +fn collect(e: &PreASAPNode, out: &mut Vec) { match e { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { measures, child, .. } => { out.extend(measures.iter().cloned()); collect(child, out); } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::Project { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => collect(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } => { + PreASAPNode::TimeRange { child, .. } + | PreASAPNode::TimeShift { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::SQLWindowFunc { child, .. } + | PreASAPNode::Project { child, .. } + | PreASAPNode::PromqlRelabel { child, .. } + | PreASAPNode::PromqlSeriesSample { child, .. } + | PreASAPNode::PromqlInfoEnrich { child, .. } => collect(child, out), + PreASAPNode::BinaryOp { lhs, rhs, .. } => { collect(lhs, out); collect(rhs, out); } - QueryExpr::Join { left, right, .. } | QueryExpr::SetOp { left, right, .. } => { + PreASAPNode::Join { left, right, .. } | PreASAPNode::SetOp { left, right, .. } => { collect(left, out); collect(right, out); } - QueryExpr::Concat { children, .. } => children.iter().for_each(|c| collect(c, out)), - QueryExpr::PromqlVectorFromScalar(inner) | QueryExpr::PromqlScalarFromVector(inner) => { + PreASAPNode::Concat { children, .. } => children.iter().for_each(|c| collect(c, out)), + PreASAPNode::PromqlVectorFromScalar(inner) | PreASAPNode::PromqlScalarFromVector(inner) => { collect(inner, out) } // `AggIntent` only ever lives in `Aggregate.measures`, never in a // scalar position (issue #205) — nothing to collect there. - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} + PreASAPNode::Scan { .. } + | PreASAPNode::PromqlScalarBridge(_) + | PreASAPNode::EvalTimestamp + | PreASAPNode::CurrentTimestamp => {} + PreASAPNode::Column(_) + | PreASAPNode::Literal(_) + | PreASAPNode::Compare { .. } + | PreASAPNode::BoolAnd(_) + | PreASAPNode::BoolOr(_) + | PreASAPNode::Not(_) + | PreASAPNode::IsNull(_) + | PreASAPNode::IsNotNull(_) + | PreASAPNode::Cast { .. } + | PreASAPNode::InList { .. } + | PreASAPNode::FunctionCall { .. } + | PreASAPNode::Arithmetic { .. } + | PreASAPNode::Case { .. } => {} } } /// The first `Scan` reached by descending single-child nodes, with its metric /// name and predicate count. -fn first_scan(e: &QueryExpr) -> (String, usize) { +fn first_scan(e: &PreASAPNode) -> (String, usize) { match e { - QueryExpr::Scan { + PreASAPNode::Scan { source, predicates, .. } => { let name = match source { @@ -133,43 +133,43 @@ fn first_scan(e: &QueryExpr) -> (String, usize) { }; (name, predicates.len()) } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_scan(child), + PreASAPNode::TimeRange { child, .. } + | PreASAPNode::TimeShift { child, .. } + | PreASAPNode::Aggregate { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } => first_scan(child), other => panic!("no Scan reachable from {other:?}"), } } -fn has bool>(e: &QueryExpr, pred: F) -> bool { +fn has bool>(e: &PreASAPNode, pred: F) -> bool { intents(e).iter().any(pred) } /// Whether the tree contains a `Mul`-by-`PromqlScalarBridge(-1)` anywhere — the shape unary /// negation lowers to (issue #36). -fn negates_via_scalar(e: &QueryExpr) -> bool { - let is_neg_one = |q: &QueryExpr| { +fn negates_via_scalar(e: &PreASAPNode) -> bool { + let is_neg_one = |q: &PreASAPNode| { q.as_promql_scalar() .is_some_and(|v| (v + 1.0).abs() < 1e-12) }; match e { - QueryExpr::BinaryOp { op, lhs, rhs, .. } => { + PreASAPNode::BinaryOp { op, lhs, rhs, .. } => { (*op == BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul) && (is_neg_one(lhs) || is_neg_one(rhs))) || negates_via_scalar(lhs) || negates_via_scalar(rhs) } - QueryExpr::Aggregate { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Project { child, .. } => negates_via_scalar(child), + PreASAPNode::Aggregate { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::TimeRange { child, .. } + | PreASAPNode::TimeShift { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Project { child, .. } => negates_via_scalar(child), _ => false, } } @@ -193,10 +193,10 @@ fn promql_scan_schema_is_open() { // runtime-only, so the binding schema lists only the (ts, value) floor + // referenced labels and may be a subset of the runtime row. let qe = ok("node_cpu_seconds_total"); - let QueryExpr::TimeRange { child, .. } = &qe else { + let PreASAPNode::TimeRange { child, .. } = &qe else { panic!("expected a TimeRange for a bare selector, got {qe:?}"); }; - let QueryExpr::Scan { schema, .. } = child.as_ref() else { + let PreASAPNode::Scan { schema, .. } = child.as_ref() else { panic!("expected a Scan inside the TimeRange, got {qe:?}"); }; assert!( @@ -241,7 +241,7 @@ fn range_vector_selector_is_time_range() { // SEMANTICS: `[5m]` turns an instant vector into a range vector, // represented in the canonical tree as a dedicated `TimeRange` node. let qe = ok("node_cpu_seconds_total[5m]"); - let QueryExpr::TimeRange { range, .. } = &qe else { + let PreASAPNode::TimeRange { range, .. } = &qe else { panic!("expected TimeRange for a range-vector selector, got {qe:?}"); }; assert_eq!(*range, Duration::from_secs(300)); @@ -257,14 +257,14 @@ fn rate_range_lives_in_time_range_node() { // SEMANTICS: per-second average rate; the temporal range lives on the // enclosing `TimeRange` node, not inside the intent. let qe = ok("rate(http_requests_total[5m])"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { panic!("expected Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let PreASAPNode::TimeRange { range, .. } = child.as_ref() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); @@ -281,14 +281,14 @@ fn irate_maps_to_its_own_intent() { #[test] fn increase_range_lives_in_time_range_node() { let qe = ok("increase(http_requests_total[1h])"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { panic!("expected Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Increase])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let PreASAPNode::TimeRange { range, .. } = child.as_ref() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(3600)); @@ -303,7 +303,7 @@ fn increase_range_lives_in_time_range_node() { fn sum_collapses_all_series() { // SEMANTICS: `sum(v)` → one output series. No grouping → no Partition. let qe = ok("sum(node_filesystem_size_bytes)"); - assert!(matches!(&qe, QueryExpr::Aggregate { .. })); + assert!(matches!(&qe, PreASAPNode::Aggregate { .. })); assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); } @@ -314,7 +314,7 @@ fn sum_by_groups_via_positional_aggregate() { // name-based Partition). SchemaResolver leaf = [ts, value, instance, job] (referenced // keys appended sorted), so the keys resolve to columns [2, 3]. let qe = ok("sum by(job, instance) (node_filesystem_size_bytes)"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -330,7 +330,7 @@ fn sum_by_groups_via_positional_aggregate() { ); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.as_ref(), PreASAPNode::TimeRange { child, .. } if matches!(child.as_ref(), PreASAPNode::Scan { .. })) ); } @@ -369,7 +369,7 @@ fn sum_without_groups_by_the_complement() { // the runtime: the grouping is the exclusion form and the output schema // stays OPEN (unlike `by`, which freezes to closed). let qe = ok("sum without(instance) (node_filesystem_size_bytes)"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, .. @@ -418,7 +418,7 @@ fn group_aggregator_lowers_to_a_distinct_intent() { fn sum_of_rate_is_two_levels() { // SEMANTICS: per-series rate, THEN cross-series sum. Both must survive. let qe = ok("sum(rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { @@ -427,7 +427,7 @@ fn sum_of_rate_is_two_levels() { assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!(matches!( child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -436,7 +436,7 @@ fn sum_by_of_rate_groups_outer_level() { // Outer cross-series Sum grouped on positional `Aggregate.by` over the // label-preserving inner Rate. Leaf = [ts, value, instance] → by = [2]. let qe = ok("sum by(instance) (rate(node_network_receive_bytes_total[5m]))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -450,7 +450,7 @@ fn sum_by_of_rate_groups_outer_level() { // child is the inner per-series Rate aggregate. assert!(matches!( child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -461,7 +461,7 @@ fn sum_by_of_over_time_groups_outer_level() { // preserving, so the key resolves positionally just like the rate case (no // name-based Partition). Leaf = [ts, value, instance] → by = [2]. let qe = ok("sum by(instance) (avg_over_time(node_cpu_seconds_total[5m]))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -473,14 +473,14 @@ fn sum_by_of_over_time_groups_outer_level() { assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // child is the inner per-series reduction: Aggregate{Avg} over TimeRange. - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = child.as_ref() else { panic!("expected Aggregate (per-series avg_over_time) under the Sum, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::TimeRange { .. })); } // ───────────────────────────────────────────────────────────────────────────── @@ -501,7 +501,7 @@ fn over_time_functions_reduce_over_time_range() { ] { let qe = ok(q); assert!( - matches!(&qe, QueryExpr::Aggregate { .. }), + matches!(&qe, PreASAPNode::Aggregate { .. }), "{q}: expected Aggregate" ); let matched = intents(&qe).iter().any(|i| match want { @@ -519,7 +519,7 @@ fn over_time_functions_reduce_over_time_range() { #[test] fn quantile_over_time_is_aggregate_over_time_range() { let qe = ok("quantile_over_time(0.9, request_latency_seconds[5m])"); - assert!(matches!(&qe, QueryExpr::Aggregate { .. })); + assert!(matches!(&qe, PreASAPNode::Aggregate { .. })); assert!(has( &qe, |i| matches!(i, AggIntent::Quantile { q, .. } if (*q - 0.9).abs() < 1e-9) @@ -536,7 +536,7 @@ fn histogram_quantile_over_rate() { // φ-quantile from bucket rates. The `_bucket` metric marks the classic // cumulative-bucket form → `HistogramQuantile` (even without `sum by (le)`). let qe = ok("histogram_quantile(0.9, rate(demo_api_request_duration_seconds_bucket[5m]))"); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let PreASAPNode::Aggregate { measures, .. } = &qe else { panic!("expected Aggregate{{HistogramQuantile}}, got {qe:?}"); }; assert!( @@ -553,7 +553,7 @@ fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { let qe = ok( "histogram_quantile(0.99, sum by(le) (rate(demo_api_request_duration_seconds_bucket[5m])))", ); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { @@ -566,7 +566,7 @@ fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { )); // `sum by(le)` now survives as a positional Aggregate (by = [2], `le`), over // the inner Rate — no name-based Partition. - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, .. @@ -586,7 +586,7 @@ fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { #[test] fn vector_arithmetic() { let qe = ok("node_memory_MemFree_bytes + node_memory_Cached_bytes"); - let QueryExpr::BinaryOp { op, .. } = &qe else { + let PreASAPNode::BinaryOp { op, .. } = &qe else { panic!("expected BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Add)); @@ -597,7 +597,7 @@ fn on_matching_with_group_left() { // SEMANTICS: many-to-one matching on a label subset. let qe = ok("rate(demo_cpu_usage_seconds_total[1m]) / on(instance, job) group_left demo_num_cpus"); - let QueryExpr::BinaryOp { + let PreASAPNode::BinaryOp { op, vector_match, .. } = &qe else { @@ -617,7 +617,7 @@ fn vector_comparison_filters() { // SEMANTICS: `>` between two vectors keeps the LHS series where it holds. let qe = ok("go_goroutines > go_threads"); assert!( - matches!(&qe, QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Gt)) + matches!(&qe, PreASAPNode::BinaryOp { op, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Gt)) ); } @@ -643,7 +643,7 @@ fn unary_negation_lowers_as_multiply_by_minus_one() { } // `-some_metric` at the root: `Scan * PromqlScalarBridge(-1)`, schema follows the vector. - let QueryExpr::BinaryOp { + let PreASAPNode::BinaryOp { op, lhs, rhs, @@ -654,7 +654,7 @@ fn unary_negation_lowers_as_multiply_by_minus_one() { }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul)); assert!( - matches!(lhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })), + matches!(lhs.as_ref(), PreASAPNode::TimeRange { child, .. } if matches!(child.as_ref(), PreASAPNode::Scan { .. })), "vector on the left" ); assert!( @@ -679,7 +679,7 @@ fn unary_negation_lowers_as_multiply_by_minus_one() { // `sum(-m)` — the negation lowers inside the aggregate argument (issue #27 // nesting), so the outer node is the `Sum` aggregate over the `Mul`. - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &ok("sum(-node_cpu_seconds_total)") else { @@ -688,7 +688,7 @@ fn unary_negation_lowers_as_multiply_by_minus_one() { assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!(matches!( child.as_ref(), - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), .. } @@ -708,14 +708,14 @@ fn unary_negation_of_constant_folds_to_scalar() { fn double_unary_negation_nests() { // `- -some_metric` — negation of a negation: `(m * -1) * -1`. Both levels // lower; the value is unchanged but the structure is faithfully nested. - let QueryExpr::BinaryOp { op, lhs, .. } = &ok("- -some_metric") else { + let PreASAPNode::BinaryOp { op, lhs, .. } = &ok("- -some_metric") else { panic!("expected outer BinaryOp for `- -some_metric`"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul)); assert!( matches!( lhs.as_ref(), - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), .. } @@ -758,12 +758,12 @@ fn scalar_literal_operand_lowers_as_binaryop_scalar() { // `PromqlScalarBridge` operand of the `BinaryOp`, and constant arithmetic // (`10*1024*1024`) is folded. The output schema is the vector side's. let qe = ok("node_filesystem_avail_bytes > 10*1024*1024"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &qe else { + let PreASAPNode::BinaryOp { op, lhs, rhs, .. } = &qe else { panic!("expected a BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Compare(CompareOpKind::Gt)); assert!( - matches!(lhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })), + matches!(lhs.as_ref(), PreASAPNode::TimeRange { child, .. } if matches!(child.as_ref(), PreASAPNode::Scan { .. })), "vector on the left" ); assert!( @@ -780,7 +780,7 @@ fn scalar_arithmetic_scales_the_vector() { // `rate(m[5m]) * 100` — a unit conversion. Arithmetic BinaryOp of the vector // with a `PromqlScalarBridge(100)`. let qe = ok("rate(m[5m]) * 100"); - let QueryExpr::BinaryOp { op, rhs, .. } = &qe else { + let PreASAPNode::BinaryOp { op, rhs, .. } = &qe else { panic!("expected a BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul)); @@ -798,11 +798,11 @@ fn scalar_arithmetic_scales_the_vector() { fn set_ops_lower_to_binaryop() { // SEMANTICS: or = union of label sets; and = intersection; unless = difference. assert!(matches!(&ok("up{job=\"a\"} or up{job=\"b\"}"), - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::Or))); + PreASAPNode::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::Or))); assert!(matches!(&ok("node_network_mtu_bytes and node_up"), - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::And))); + PreASAPNode::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::And))); assert!(matches!(&ok("node_network_mtu_bytes unless node_down"), - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::Unless))); + PreASAPNode::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::Unless))); } // ───────────────────────────────────────────────────────────────────────────── @@ -824,7 +824,7 @@ fn topk_over_count_is_heavy_hitter() { fn bottomk_is_generic_sort_limit() { // SEMANTICS: bottom-k → generic ascending order + limit (no sketch). let qe = ok("bottomk(3, count_over_time(http_requests_total[5m]))"); - assert!(matches!(&qe, QueryExpr::Limit { .. })); + assert!(matches!(&qe, PreASAPNode::Limit { .. })); } #[test] @@ -833,7 +833,7 @@ fn topk_over_nested_sum_preserves_weighted_topk_accuracy() { // The final rates are query-time values. Their ordering does not establish // frequency-sketch membership semantics. let qe = ok("topk(3, sum by(instance) (rate(node_cpu_seconds_total[5m])))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { @@ -861,14 +861,14 @@ fn outer_aggregate_over_nested_aggregate_nests() { // flat two-level template rejected. Each level survives into the // canonical tree (issue #27). let qe = ok("max(sum by (job) (rate(http_requests_total[5m])))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { panic!("expected outer Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, .. @@ -895,7 +895,7 @@ fn outer_group_key_absent_from_nested_aggregate_is_dropped() { // the query lowers with the provably-absent key dropped, exactly // `sum(sum by (group)(…))`. let qe = ok(r#"sum(sum by (group)(http_requests{job="api-server"})) by (job)"#); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -910,7 +910,7 @@ fn outer_group_key_absent_from_nested_aggregate_is_dropped() { "absent `job` key dropped → global aggregate" ); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let QueryExpr::Aggregate { reduction, .. } = child.as_ref() else { + let PreASAPNode::Aggregate { reduction, .. } = child.as_ref() else { panic!("expected inner `sum by (group)` Aggregate, got {child:?}"); }; assert_eq!( @@ -927,13 +927,13 @@ fn outer_group_key_present_after_inner_aggregate_still_resolves() { // resolving positionally — the absent-key drop only fires on provable // absence, never on a resolvable key. let qe = ok("sum(sum by (job, group)(http_requests)) by (job)"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, child, .. } = &qe else { panic!("expected outer Aggregate, got {qe:?}"); }; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction: inner_reduction, .. } = child.as_ref() @@ -957,7 +957,7 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { // still resolve. Each `or` side is bound independently against its own // sub-tree, so the key is seeded as an inherited column on both sides. let qe = ok(r#"sum by (__name__)(metric_a{env="1"} or metric_b{env="2"})"#); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, child, .. } = &qe else { @@ -969,7 +969,7 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { 1, "grouped by the one `__name__` key" ); - let QueryExpr::BinaryOp { lhs, rhs, .. } = child.as_ref() else { + let PreASAPNode::BinaryOp { lhs, rhs, .. } = child.as_ref() else { panic!("expected a BinaryOp child, got {child:?}"); }; // Both independently-bound sides carry `__name__` at the same position, so @@ -984,7 +984,7 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { // The general case (a plain label, not just `__name__`) also lowers. assert!(matches!( ok("sum by (job)(metric_a or metric_b)"), - QueryExpr::Aggregate { .. } + PreASAPNode::Aggregate { .. } )); } @@ -994,7 +994,7 @@ fn aggregate_over_binary_op_nests() { // op over two range vectors. The old template only accepted a single inner // selector/call; now the binary op lowers and the outer sum wraps it. let qe = ok("sum(rate(a[5m]) + rate(b[5m]))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { @@ -1002,7 +1002,7 @@ fn aggregate_over_binary_op_nests() { }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!( - matches!(child.as_ref(), QueryExpr::BinaryOp { .. }), + matches!(child.as_ref(), PreASAPNode::BinaryOp { .. }), "argument lowers as a BinaryOp, got {child:?}" ); } @@ -1015,7 +1015,7 @@ fn aggregate_over_binary_op_nests() { fn subquery_wraps_inner_query() { // SEMANTICS: `[range:res]` evaluates the inner query across a range. let qe = ok("rate(demo_api_request_duration_seconds_count[5m])[1h:]"); - assert!(matches!(&qe, QueryExpr::PromqlSubquery { .. })); + assert!(matches!(&qe, PreASAPNode::PromqlSubquery { .. })); assert!(has(&qe, |i| matches!(i, AggIntent::Rate))); } @@ -1026,7 +1026,7 @@ fn over_time_of_subquery_reduces_per_series() { // then `max_over_time` takes the max of those samples *per series*. It lowers // to a per-series `Max` reduction over a `PromqlSubquery` (issue #27). let qe = ok("max_over_time(rate(demo_api_request_duration_seconds_count[5m])[1h:])"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -1044,7 +1044,7 @@ fn over_time_of_subquery_reduces_per_series() { // The reduction rides directly on the sub-query (the structural range marker // that keeps it label-preserving), which wraps the inner `rate`. assert!( - matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. }), + matches!(child.as_ref(), PreASAPNode::PromqlSubquery { .. }), "the `Max` reduces over a PromqlSubquery, got {child:?}" ); assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Rate))); @@ -1055,7 +1055,7 @@ fn quantile_over_time_of_subquery_carries_phi() { // The `quantile_over_time` φ parameter is read from arg 0; the sub-query is // arg 1. It lowers to a per-series `Quantile(φ)` over the `PromqlSubquery`. let qe = ok("quantile_over_time(0.9, rate(demo[5m])[1h:])"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { @@ -1064,7 +1064,7 @@ fn quantile_over_time_of_subquery_carries_phi() { assert!( matches!(measures.as_slice(), [AggIntent::Quantile { q, .. }] if (*q - 0.9).abs() < 1e-9) ); - assert!(matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::PromqlSubquery { .. })); } #[test] @@ -1074,7 +1074,7 @@ fn aggregation_over_over_time_of_subquery_keeps_labels() { // survives for the OUTER cross-series `sum by (job)` to group on. If the // inner `Max` collapsed labels, `job` would not resolve here. let qe = ok("sum by (job) (max_over_time(rate(demo{job=\"api\"}[5m])[1h:]))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -1089,7 +1089,7 @@ fn aggregation_over_over_time_of_subquery_keeps_labels() { ); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // Inner node is the per-series `max_over_time` reduction over the subquery. - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction: inner_reduction, measures: inner_measures, child: inner_child, @@ -1102,7 +1102,7 @@ fn aggregation_over_over_time_of_subquery_keeps_labels() { assert!(matches!(inner_measures.as_slice(), [AggIntent::Max { .. }])); assert!(matches!( inner_child.as_ref(), - QueryExpr::PromqlSubquery { .. } + PreASAPNode::PromqlSubquery { .. } )); } @@ -1124,7 +1124,7 @@ fn nested_subquery_from_prometheus_docs() { // the label-preserving `[ts, value]`. let qe = ok("max_over_time(deriv(rate(distance_covered_total[5s])[30s:5s])[10m:])"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -1136,7 +1136,7 @@ fn nested_subquery_from_prometheus_docs() { assert_eq!(reduction, &Reduction::PerEntity); assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let QueryExpr::PromqlSubquery { + let PreASAPNode::PromqlSubquery { range, resolution, child, @@ -1147,7 +1147,7 @@ fn nested_subquery_from_prometheus_docs() { assert_eq!(*range, Duration::from_secs(600)); assert_eq!(*resolution, None, "`[10m:]` keeps the default resolution"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -1159,7 +1159,7 @@ fn nested_subquery_from_prometheus_docs() { assert_eq!(reduction, &Reduction::PerEntity); assert!(matches!(measures.as_slice(), [AggIntent::Deriv])); - let QueryExpr::PromqlSubquery { + let PreASAPNode::PromqlSubquery { range, resolution, child, @@ -1170,14 +1170,14 @@ fn nested_subquery_from_prometheus_docs() { assert_eq!(*range, Duration::from_secs(30)); assert_eq!(*resolution, Some(Duration::from_secs(5))); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = child.as_ref() else { panic!("expected the `rate` Aggregate, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let PreASAPNode::TimeRange { range, .. } = child.as_ref() else { panic!("expected the `[5s]` TimeRange under rate, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(5)); @@ -1205,21 +1205,21 @@ fn offset_modifier_lowers_to_a_time_shift() { // past — a `TimeShift` wrapper over the selector (signed ms; a negative // offset shifts forward). Schema is unchanged (the shift only moves *when*). let qe = ok("http_requests_total offset 5m"); - let QueryExpr::TimeRange { child, .. } = &qe else { + let PreASAPNode::TimeRange { child, .. } = &qe else { panic!("expected an ingestion TimeRange, got {qe:?}"); }; - let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + let PreASAPNode::TimeShift { shift, child } = child.as_ref() else { panic!("expected a TimeShift, got {qe:?}"); }; assert_eq!(shift.offset_ms, 300_000); assert!(shift.at.is_none()); - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::Scan { .. })); // `offset -5m` shifts forward → negative ms. - let QueryExpr::TimeRange { child, .. } = &ok("http_requests_total offset -5m") else { + let PreASAPNode::TimeRange { child, .. } = &ok("http_requests_total offset -5m") else { panic!("expected an ingestion TimeRange"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let PreASAPNode::TimeShift { shift, .. } = child.as_ref() else { panic!("expected a TimeShift"); }; assert_eq!(shift.offset_ms, -300_000); @@ -1230,29 +1230,29 @@ fn at_modifier_lowers_to_a_time_shift() { // SEMANTICS (PromQL, issue #40): `@ ` pins the evaluation to an absolute // instant (PromQL seconds → IR milliseconds); `@ start()` / `@ end()` anchor // to the query range bounds. - let QueryExpr::TimeRange { child, .. } = &ok("http_requests_total @ 1609746000") else { + let PreASAPNode::TimeRange { child, .. } = &ok("http_requests_total @ 1609746000") else { panic!("expected an ingestion TimeRange"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let PreASAPNode::TimeShift { shift, .. } = child.as_ref() else { panic!("expected a TimeShift for `@ `"); }; assert_eq!(shift.at, Some(AtModifier::Timestamp(1_609_746_000_000))); assert_eq!(shift.offset_ms, 0); - let QueryExpr::TimeRange { child, .. } = &ok("http_requests_total @ start()") else { + let PreASAPNode::TimeRange { child, .. } = &ok("http_requests_total @ start()") else { panic!("expected an ingestion TimeRange"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let PreASAPNode::TimeShift { shift, .. } = child.as_ref() else { panic!("expected a TimeShift for `@ start()`"); }; assert_eq!(shift.at, Some(AtModifier::Start)); // Offset and `@` compose: `@ end() offset 5m` carries both. let qe = ok("http_requests_total @ end() offset 5m"); - let QueryExpr::TimeRange { child, .. } = &qe else { + let PreASAPNode::TimeRange { child, .. } = &qe else { panic!("expected an ingestion TimeRange, got {qe:?}"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let PreASAPNode::TimeShift { shift, .. } = child.as_ref() else { panic!("expected a TimeShift, got {qe:?}"); }; assert_eq!(shift.at, Some(AtModifier::End)); @@ -1265,21 +1265,21 @@ fn offset_on_a_ranged_selector_wraps_inside_the_time_range() { // `TimeShift` sits *under* the `TimeRange` (the 5m window is taken at the // shifted time), and the whole thing under the per-series `Rate` (#40). let qe = ok("rate(http_requests_total[5m] offset 1h)"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { panic!("expected the rate Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { child, .. } = child.as_ref() else { + let PreASAPNode::TimeRange { child, .. } = child.as_ref() else { panic!("expected a TimeRange under rate, got {child:?}"); }; - let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + let PreASAPNode::TimeShift { shift, child } = child.as_ref() else { panic!("expected a TimeShift under the TimeRange, got {child:?}"); }; assert_eq!(shift.offset_ms, 3_600_000); - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::Scan { .. })); } // ───────────────────────────────────────────────────────────────────────────── @@ -1335,7 +1335,7 @@ fn counter_derivative_functions_lower_to_distinct_intents() { ("resets(m[1h])", AggIntent::Resets), ] { let qe = ok(q); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -1355,7 +1355,7 @@ fn counter_derivative_functions_lower_to_distinct_intents() { "{q}: wrong intent" ); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { .. }), + matches!(child.as_ref(), PreASAPNode::TimeRange { .. }), "{q}: reduction rides on a TimeRange, got {child:?}" ); } @@ -1366,7 +1366,7 @@ fn predict_linear_carries_horizon_seconds() { // `predict_linear(v[w], t)` — the 2nd (scalar) arg is the prediction horizon // in seconds; it must be carried in the intent (it changes the result). let qe = ok("predict_linear(node_filesystem_avail_bytes[3h], 86400)"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { @@ -1376,7 +1376,7 @@ fn predict_linear_carries_horizon_seconds() { measures.as_slice(), &[AggIntent::PredictLinear { seconds: 86400.0 }] ); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::TimeRange { .. })); } #[test] @@ -1394,7 +1394,7 @@ fn aggregation_over_counter_derivative_keeps_labels() { // A counter-derivative is per-series (label-preserving), so an outer // `sum by (job)` can group on a label the inner `changes` preserved. let qe = ok(r#"sum by (job) (changes(m{job="api"}[15m]))"#); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -1420,7 +1420,7 @@ fn outer_stat_over_counter_derivative_nests_two_levels() { // grouped outer (`avg by (dc)`) must resolve its key against the labels the // inner reduction preserved, threading any scalar param (predict horizon). let qe = ok("avg by (dc) (predict_linear(m[3h], 3600))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -1434,7 +1434,7 @@ fn outer_stat_over_counter_derivative_nests_two_levels() { "outer `avg by (dc)` groups on a label" ); assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction: inner_reduction, measures: inner_measures, .. @@ -1458,11 +1458,11 @@ fn topk_over_counter_derivative_is_generic_sort_limit() { // `topk(k, deriv(...))` ranks the per-series derivative values — a generic // `Sort + Limit`, NOT a heavy-hitter `TopK` (that's only `count_over_time`). let qe = ok("topk(3, deriv(m[5m]))"); - let QueryExpr::Limit { n, child, .. } = &qe else { + let PreASAPNode::Limit { n, child, .. } = &qe else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 3); - assert!(matches!(child.as_ref(), QueryExpr::Sort { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::Sort { .. })); assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Deriv))); assert!( !intents(&qe) @@ -1477,28 +1477,28 @@ fn counter_derivative_composes_in_binary_ops() { // As a vector operand: `delta(a[5m]) / delta(b[5m])` is a BinaryOp of two // per-series Delta reductions. let ratio = ok("delta(a[5m]) / delta(b[5m])"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &ratio else { + let PreASAPNode::BinaryOp { op, lhs, rhs, .. } = &ratio else { panic!("expected BinaryOp, got {ratio:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); assert!( - matches!(lhs.as_ref(), QueryExpr::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) + matches!(lhs.as_ref(), PreASAPNode::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) ); assert!( - matches!(rhs.as_ref(), QueryExpr::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) + matches!(rhs.as_ref(), PreASAPNode::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) ); // Under an aggregate over a binary op mixing a counter-derivative with // another per-series function: `sum(rate(m[5m]) + changes(m[5m]))`. let mixed = ok("sum(rate(m[5m]) + changes(m[5m]))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &mixed else { panic!("expected Aggregate, got {mixed:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::BinaryOp { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::BinaryOp { .. })); assert!(intents(&mixed).iter().any(|i| matches!(i, AggIntent::Rate))); assert!(intents(&mixed) .iter() @@ -1522,7 +1522,7 @@ fn range_functions_over_a_subquery_reduce_per_series() { ("resets(sum(m)[5m:])", AggIntent::Resets), ] { let qe = ok(q); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -1542,7 +1542,7 @@ fn range_functions_over_a_subquery_reduce_per_series() { "{q}: wrong intent" ); assert!( - matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. }), + matches!(child.as_ref(), PreASAPNode::PromqlSubquery { .. }), "{q}: reduces directly over the PromqlSubquery (no TimeRange), got {child:?}" ); } @@ -1624,7 +1624,7 @@ fn histogram_accessors_lower_to_per_series_intents() { ("histogram_stdvar(v)", AggIntent::HistogramStdVar), ] { let qe = ok(q); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, .. @@ -1678,7 +1678,7 @@ fn math_functions_lower_to_per_series_math_intents() { ("rad(v)", MathFunc::Rad), ] { let qe = ok(q); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, .. @@ -1763,7 +1763,7 @@ fn absent_keeps_matcher_labels_for_the_synthesized_output() { fn time_lowers_to_the_eval_time_scalar() { // SEMANTICS: `time()` is the query evaluation timestamp as a scalar — a leaf, // not an aggregate over any series. - assert!(matches!(ok("time()"), QueryExpr::EvalTimestamp)); + assert!(matches!(ok("time()"), PreASAPNode::EvalTimestamp)); // …and it is scalar-shaped: a single float `value`, no time index. let sch = ok("time()").output_schema().unwrap(); assert_eq!(sch.columns.len(), 1); @@ -1777,10 +1777,10 @@ fn time_minus_vector_is_the_uptime_pattern() { // The scalar `time()` broadcasts against the vector; the result takes the // vector's schema. let qe = ok("time() - process_start_time_seconds"); - let QueryExpr::BinaryOp { lhs, op, .. } = &qe else { + let PreASAPNode::BinaryOp { lhs, op, .. } = &qe else { panic!("expected a BinaryOp, got {qe:?}"); }; - assert!(matches!(lhs.as_ref(), QueryExpr::EvalTimestamp)); + assert!(matches!(lhs.as_ref(), PreASAPNode::EvalTimestamp)); assert!(matches!( op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub) @@ -1817,7 +1817,7 @@ fn no_arg_calendar_function_reads_the_eval_time() { // `day_of_week()` with no argument computes over the evaluation time itself, // so it is a `TimeFn` aggregate whose child is the `EvalTimestamp` scalar. let qe = ok("day_of_week()"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { @@ -1827,7 +1827,7 @@ fn no_arg_calendar_function_reads_the_eval_time() { measures.as_slice(), [AggIntent::TimeFn(TimeFunc::DayOfWeek)] )); - assert!(matches!(child.as_ref(), QueryExpr::EvalTimestamp)); + assert!(matches!(child.as_ref(), PreASAPNode::EvalTimestamp)); } #[test] @@ -1848,7 +1848,7 @@ fn vector_promotes_a_scalar_to_a_vector() { // SEMANTICS: `vector(s)` is the scalar→instant-vector bridge — a label-less // single series carrying the scalar's value. let qe = ok("vector(1)"); - let QueryExpr::PromqlVectorFromScalar(inner) = &qe else { + let PreASAPNode::PromqlVectorFromScalar(inner) = &qe else { panic!("expected PromqlVectorFromScalar, got {qe:?}"); }; assert_eq!(inner.as_promql_scalar(), Some(1.0)); @@ -1862,7 +1862,7 @@ fn vector_promotes_a_scalar_to_a_vector() { fn scalar_collapses_a_vector_to_a_scalar() { // SEMANTICS: `scalar(v)` is the instant-vector→scalar bridge. let qe = ok("scalar(node_load1)"); - let QueryExpr::PromqlScalarFromVector(inner) = &qe else { + let PreASAPNode::PromqlScalarFromVector(inner) = &qe else { panic!("expected PromqlScalarFromVector, got {qe:?}"); }; let (metric, _) = first_scan(inner); @@ -1880,11 +1880,14 @@ fn vector_zero_is_a_vector_operand_of_a_set_op() { // vectors, so `vector(0)` must be a vector (a `PromqlVectorFromScalar`), never a // folded scalar operand. let qe = ok("up or vector(0)"); - let QueryExpr::BinaryOp { rhs, op, .. } = &qe else { + let PreASAPNode::BinaryOp { rhs, op, .. } = &qe else { panic!("expected a BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Set(PromQLVectorSetOpKind::Or)); - assert!(matches!(rhs.as_ref(), QueryExpr::PromqlVectorFromScalar(_))); + assert!(matches!( + rhs.as_ref(), + PreASAPNode::PromqlVectorFromScalar(_) + )); } #[test] @@ -1892,10 +1895,13 @@ fn scalar_of_a_vector_feeds_a_threshold_comparison() { // `node_load1 > scalar(node_cpu_count)` — `scalar(...)` is a scalar operand, // so the BinaryOp output takes the vector (lhs) side's schema. let qe = ok("node_load1 > scalar(node_cpu_count)"); - let QueryExpr::BinaryOp { lhs, rhs, .. } = &qe else { + let PreASAPNode::BinaryOp { lhs, rhs, .. } = &qe else { panic!("expected a BinaryOp, got {qe:?}"); }; - assert!(matches!(rhs.as_ref(), QueryExpr::PromqlScalarFromVector(_))); + assert!(matches!( + rhs.as_ref(), + PreASAPNode::PromqlScalarFromVector(_) + )); // The BinaryOp output schema follows the vector (lhs) side, not the scalar. let (metric, _) = first_scan(lhs); assert_eq!(metric, "node_load1"); @@ -1909,7 +1915,7 @@ fn info_lowers_to_a_label_enrichment_join() { // (issue #84). The value/time axis pass through; the enriched labels are // runtime, so the schema stays the child's. let qe = ok("info(rate(http_requests_total[5m]))"); - let QueryExpr::PromqlInfoEnrich { selector, child } = &qe else { + let PreASAPNode::PromqlInfoEnrich { selector, child } = &qe else { panic!("expected an PromqlInfoEnrich, got {qe:?}"); }; assert!(selector.is_empty(), "no selector → default target_info"); @@ -1925,7 +1931,7 @@ fn info_selector_carries_the_info_side_matchers() { // matchers are kept symbolically (not run through the single-metric selector // path). let qe = ok(r#"info(build_info, {__name__=~".+_info", another_data=~".+"})"#); - let QueryExpr::PromqlInfoEnrich { selector, .. } = &qe else { + let PreASAPNode::PromqlInfoEnrich { selector, .. } = &qe else { panic!("expected an PromqlInfoEnrich, got {qe:?}"); }; assert_eq!( @@ -1950,11 +1956,11 @@ fn info_composes_under_an_aggregation_and_over_a_time_shift() { // (issue #40) — the enrichment composes over the shifted selector. assert!(matches!( ok("info(metric @ 60)"), - QueryExpr::PromqlInfoEnrich { .. } + PreASAPNode::PromqlInfoEnrich { .. } )); assert!(matches!( ok("info(metric offset 1m)"), - QueryExpr::PromqlInfoEnrich { .. } + PreASAPNode::PromqlInfoEnrich { .. } )); } @@ -1967,7 +1973,7 @@ fn group_lowers_to_a_constant_group_intent() { // SEMANTICS: `group(v)` yields a constant 1 per group — a distinct intent, // NOT folded onto `sum` (which would return the value sum instead of 1). let qe = ok("group(up)"); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let PreASAPNode::Aggregate { measures, .. } = &qe else { panic!("expected an Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Group])); @@ -1991,7 +1997,7 @@ fn count_values_groups_by_value_and_synthesizes_a_label() { // value, counts each distinct value, and emits that value as a new label // `l`. The intent carries the label; schema gains a `Utf8` `l` column. let qe = ok(r#"count_values("version", build_version)"#); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let PreASAPNode::Aggregate { measures, .. } = &qe else { panic!("expected an Aggregate, got {qe:?}"); }; assert!( @@ -2047,14 +2053,14 @@ fn limitk_and_limit_ratio_lower_to_series_sampling() { // `PromqlSeriesSample` node, never `topk`'s `Sort → Limit` (issue #86). assert!(matches!( ok("limitk(2, http_requests)"), - QueryExpr::PromqlSeriesSample { + PreASAPNode::PromqlSeriesSample { kind: SampleKind::LimitK(2), .. } )); assert!(matches!( ok("limit_ratio(0.1, http_requests)"), - QueryExpr::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 0.1).abs() < 1e-9 + PreASAPNode::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 0.1).abs() < 1e-9 )); // Series-preserving: the output schema equals the input's (ts, value). let sch = ok("limitk(2, http_requests)").output_schema().unwrap(); @@ -2068,11 +2074,11 @@ fn limit_ratio_keeps_a_negative_ratio_and_clamps_out_of_range() { // be normalised away. Out-of-range magnitudes clamp to [-1, 1] (Prometheus). assert!(matches!( ok("limit_ratio(-0.5, http_requests)"), - QueryExpr::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r + 0.5).abs() < 1e-9 + PreASAPNode::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r + 0.5).abs() < 1e-9 )); assert!(matches!( ok("limit_ratio(1.1, http_requests)"), - QueryExpr::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 1.0).abs() < 1e-9 + PreASAPNode::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 1.0).abs() < 1e-9 )); } @@ -2080,7 +2086,7 @@ fn limit_ratio_keeps_a_negative_ratio_and_clamps_out_of_range() { fn limitk_by_carries_the_grouping_and_composes_in_a_set_op() { // `limitk by (group)` samples per group; the grouping label is seeded. let qe = ok("limitk by (group) (2, http_requests)"); - let QueryExpr::PromqlSeriesSample { by, .. } = &qe else { + let PreASAPNode::PromqlSeriesSample { by, .. } = &qe else { panic!("expected a PromqlSeriesSample, got {qe:?}"); }; assert!(!by.is_empty(), "grouped sampling keeps its `by` keys"); @@ -2106,20 +2112,20 @@ fn dynamic_and_non_finite_sample_params_are_rejected() { // ───────────────────────────────────────────────────────────────────────────── /// Descend single-child nodes to the first `PromqlRelabel`. -fn first_relabel(e: &QueryExpr) -> &QueryExpr { +fn first_relabel(e: &PreASAPNode) -> &PreASAPNode { match e { - QueryExpr::PromqlRelabel { .. } => e, - QueryExpr::Aggregate { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } => first_relabel(child), + PreASAPNode::PromqlRelabel { .. } => e, + PreASAPNode::Aggregate { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::TimeRange { child, .. } + | PreASAPNode::TimeShift { child, .. } => first_relabel(child), other => panic!("no PromqlRelabel reachable from {other:?}"), } } /// True when `value` is a `FunctionCall` with the given name. -fn is_fn_named(value: &QueryExpr, name: &str) -> bool { - matches!(value, QueryExpr::FunctionCall { name: n, .. } if n == name) +fn is_fn_named(value: &PreASAPNode, name: &str) -> bool { + matches!(value, PreASAPNode::FunctionCall { name: n, .. } if n == name) } #[test] @@ -2127,7 +2133,7 @@ fn label_replace_is_a_relabel_over_the_vector() { // SEMANTICS: `label_replace(v, dst, repl, src, regex)` rewrites the `dst` // label per series from a regex over `src`; the sample value is untouched. let qe = ok(r#"label_replace(up, "host", "$1", "instance", "(.+):.*")"#); - let QueryExpr::PromqlRelabel { dst, value, child } = &qe else { + let PreASAPNode::PromqlRelabel { dst, value, child } = &qe else { panic!("expected a PromqlRelabel, got {qe:?}"); }; assert_eq!(dst, "host"); @@ -2148,7 +2154,7 @@ fn label_join_concatenates_source_labels() { // SEMANTICS: `label_join(v, dst, sep, src…)` joins the source labels with // `sep` into `dst`. let qe = ok(r#"label_join(up, "combined", "-", "job", "instance")"#); - let QueryExpr::PromqlRelabel { dst, value, .. } = &qe else { + let PreASAPNode::PromqlRelabel { dst, value, .. } = &qe else { panic!("expected a PromqlRelabel, got {qe:?}"); }; assert_eq!(dst, "combined"); @@ -2164,7 +2170,7 @@ fn label_replace_composes_under_an_aggregation() { let qe = ok(r#"sum by (host) (label_replace(up, "host", "$1", "instance", "(.+):.*"))"#); // A PromqlRelabel sits below the outer Sum. let relabel = first_relabel(&qe); - assert!(matches!(relabel, QueryExpr::PromqlRelabel { dst, .. } if dst == "host")); + assert!(matches!(relabel, PreASAPNode::PromqlRelabel { dst, .. } if dst == "host")); assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); let sch = qe.output_schema().unwrap(); assert!(sch.columns.iter().any(|c| c.name == "host")); @@ -2191,7 +2197,7 @@ fn extra_over_time_reducers_lower_to_per_series_intents() { assert!(has(&qe, |i| *i == want), "{q}: {:?}", intents(&qe)); // Per-series: the range window survives as a `TimeRange`. assert!( - matches!(&qe, QueryExpr::Aggregate { child, .. } if matches!(child.as_ref(), QueryExpr::TimeRange { .. })), + matches!(&qe, PreASAPNode::Aggregate { child, .. } if matches!(child.as_ref(), PreASAPNode::TimeRange { .. })), "{q} keeps its range as a TimeRange" ); } @@ -2215,13 +2221,13 @@ fn sort_and_sort_desc_reorder_by_value_without_a_limit() { ("sort_desc(http_requests)", false), ] { let qe = ok(q); - let QueryExpr::Sort { keys, child, .. } = &qe else { + let PreASAPNode::Sort { keys, child, .. } = &qe else { panic!("{q}: expected a Sort, got {qe:?}"); }; assert_eq!(keys.len(), 1); assert_eq!(keys[0].ascending, ascending, "{q}"); // No Limit above the Sort — every series is preserved. - assert!(!matches!(&qe, QueryExpr::Limit { .. })); + assert!(!matches!(&qe, PreASAPNode::Limit { .. })); // The value column is what it ranks on: descend to the scan. let (metric, _) = first_scan(child); assert_eq!(metric, "http_requests"); @@ -2233,7 +2239,7 @@ fn sort_by_label_orders_on_each_label_in_turn() { // `sort_by_label(v, "group", "instance", "job")` — one ascending sort key per // label, in argument order; the labels are seeded into the schema. let qe = ok(r#"sort_by_label(http_requests, "group", "instance", "job")"#); - let QueryExpr::Sort { keys, .. } = &qe else { + let PreASAPNode::Sort { keys, .. } = &qe else { panic!("expected a Sort, got {qe:?}"); }; assert_eq!(keys.len(), 3, "one key per label"); @@ -2250,7 +2256,7 @@ fn sort_by_label_orders_on_each_label_in_turn() { #[test] fn sort_by_label_desc_is_descending() { let qe = ok(r#"sort_by_label_desc(http_requests, "instance")"#); - let QueryExpr::Sort { keys, .. } = &qe else { + let PreASAPNode::Sort { keys, .. } = &qe else { panic!("expected a Sort, got {qe:?}"); }; assert!(keys.iter().all(|k| !k.ascending)); @@ -2270,7 +2276,7 @@ fn min_of_max_of_fold_constant_scalars() { Some(10.0) ); let qe = ok("up > max_of(1, 2)"); - let QueryExpr::BinaryOp { rhs, .. } = &qe else { + let PreASAPNode::BinaryOp { rhs, .. } = &qe else { panic!("{qe:?}") }; assert_eq!(rhs.as_promql_scalar(), Some(2.0)); diff --git a/crates/frontend-promql/tests/promql_equivalence.rs b/crates/frontend-promql/tests/promql_equivalence.rs index d1177cc5..bb5a05af 100644 --- a/crates/frontend-promql/tests/promql_equivalence.rs +++ b/crates/frontend-promql/tests/promql_equivalence.rs @@ -18,11 +18,11 @@ #![allow(non_snake_case)] mod support; -use asap_types::pre_asap::QueryExpr; +use asap_types::pre_asap::PreASAPNode; use asap_types::types::AccuracyTarget; use support::lower_promql; -fn lo(q: &str) -> QueryExpr { +fn lo(q: &str) -> PreASAPNode { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("{q:?} should lower: {e}")) } diff --git a/crates/frontend-promql/tests/promql_lowering.rs b/crates/frontend-promql/tests/promql_lowering.rs index 9d3b80de..0cb95abf 100644 --- a/crates/frontend-promql/tests/promql_lowering.rs +++ b/crates/frontend-promql/tests/promql_lowering.rs @@ -3,7 +3,7 @@ use std::time::Duration; use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, QueryExpr, Reduction, ScalarValue, + AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, PreASAPNode, Reduction, ScalarValue, Source, }; use asap_types::types::AccuracyTarget; @@ -16,7 +16,7 @@ use asap_frontend_promql::{lower_promql_workload, PromqlError as LoweringError}; mod support; use support::lower_promql; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> PreASAPNode { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } @@ -77,10 +77,10 @@ fn distinct_over_time_preserves_cardinality_accuracy_and_nested_windows() { #[test] fn bare_selector_is_scan_with_predicates() { let qe = lower(r#"http_requests_total{env="prod",status!="500"}"#); - let QueryExpr::TimeRange { child, .. } = &qe else { + let PreASAPNode::TimeRange { child, .. } = &qe else { panic!("expected TimeRange, got {qe:?}"); }; - let QueryExpr::Scan { + let PreASAPNode::Scan { source, predicates, .. } = child.as_ref() else { @@ -92,29 +92,29 @@ fn bare_selector_is_scan_with_predicates() { assert_eq!(predicates.len(), 2); assert!(predicates .iter() - .all(|p| matches!(p.0.as_ref(), QueryExpr::Compare { .. }))); + .all(|p| matches!(p.0.as_ref(), PreASAPNode::Compare { .. }))); } #[test] fn regex_matcher_lowers_to_regex_compareop() { let qe = lower(r#"http_requests_total{path=~"/api/.*"}"#); - let QueryExpr::TimeRange { child, .. } = &qe else { + let PreASAPNode::TimeRange { child, .. } = &qe else { panic!("expected TimeRange, got {qe:?}"); }; - let QueryExpr::Scan { + let PreASAPNode::Scan { predicates, schema, .. } = child.as_ref() else { panic!("expected Scan, got {qe:?}"); }; - let QueryExpr::Compare { left, op, right } = predicates[0].0.as_ref() else { + let PreASAPNode::Compare { left, op, right } = predicates[0].0.as_ref() else { panic!("expected Compare, got {:?}", predicates[0].0); }; assert_eq!(*op, CompareOpKind::Regex); // The label matcher's column is resolved positionally against the scan schema. let path_id = schema.column_id("path").expect("path in scan schema"); - assert!(matches!(left.as_ref(), QueryExpr::Column(id) if *id == path_id)); - assert!(matches!(right.as_ref(), QueryExpr::Literal(ScalarValue::Utf8(v)) if v == "/api/.*")); + assert!(matches!(left.as_ref(), PreASAPNode::Column(id) if *id == path_id)); + assert!(matches!(right.as_ref(), PreASAPNode::Literal(ScalarValue::Utf8(v)) if v == "/api/.*")); } // ── *_over_time → Aggregate over TimeRange ────────────────────────────────────── @@ -122,7 +122,7 @@ fn regex_matcher_lowers_to_regex_compareop() { #[test] fn quantile_over_time_is_time_range_aggregate() { let qe = lower(r#"quantile_over_time(0.99, http_request_duration{env="prod"}[5m])"#); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -135,12 +135,14 @@ fn quantile_over_time_is_time_range_aggregate() { assert!( matches!(measures.as_slice(), [AggIntent::Quantile { q, .. }] if (*q - 0.99).abs() < 1e-9) ); - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let PreASAPNode::TimeRange { range, child } = child.as_ref() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); // The label matcher folded onto the Scan. - assert!(matches!(child.as_ref(), QueryExpr::Scan { predicates, .. } if predicates.len() == 1)); + assert!( + matches!(child.as_ref(), PreASAPNode::Scan { predicates, .. } if predicates.len() == 1) + ); } #[test] @@ -151,7 +153,7 @@ fn outer_sum_by_over_quantile_over_time_groups_positionally() { // a name-based Partition. Leaf = [ts, value, host, service] (referenced // names appended sorted) → host = col 2. let qe = lower(r#"sum by (host) (quantile_over_time(0.99, latency{service="web"}[5m]))"#); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -163,27 +165,27 @@ fn outer_sum_by_over_quantile_over_time_groups_positionally() { assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // Inner: Aggregate{Quantile} over TimeRange (per-series over_time reduction). - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = child.as_ref() else { panic!("expected Aggregate (quantile_over_time) under the outer Sum, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Quantile { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::TimeRange { .. })); } #[test] fn avg_over_time_maps_to_avg_intent() { let qe = lower("avg_over_time(cpu_seconds_total[10m])"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { panic!("expected Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let PreASAPNode::TimeRange { range, .. } = child.as_ref() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(600)); @@ -192,7 +194,7 @@ fn avg_over_time_maps_to_avg_intent() { #[test] fn stddev_and_stdvar_over_time() { let qe = lower("stddev_over_time(m[5m])"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { @@ -205,10 +207,10 @@ fn stddev_and_stdvar_over_time() { .. }] )); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::TimeRange { .. })); let qe = lower("stdvar_over_time(m[5m])"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { @@ -221,7 +223,7 @@ fn stddev_and_stdvar_over_time() { .. }] )); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::TimeRange { .. })); } #[test] @@ -230,7 +232,7 @@ fn histogram_quantile_wraps_inner_in_quantile() { // not squashed away. The `_bucket` metric + `le` matcher mark the classic // form → `HistogramQuantile` over `Aggregate{Rate}` over Scan. let qe = lower(r#"histogram_quantile(0.95, rate(http_duration_seconds_bucket{le="0.5"}[5m]))"#); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { @@ -239,14 +241,14 @@ fn histogram_quantile_wraps_inner_in_quantile() { assert!( matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if (*q - 0.95).abs() < 1e-9) ); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = child.as_ref() else { panic!("expected inner Aggregate{{Rate}}, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { + let PreASAPNode::TimeRange { range, child: tr_child, } = child.as_ref() @@ -255,7 +257,7 @@ fn histogram_quantile_wraps_inner_in_quantile() { }; assert_eq!(*range, Duration::from_secs(300)); assert!( - matches!(tr_child.as_ref(), QueryExpr::Scan { predicates, .. } if predicates.len() == 1) + matches!(tr_child.as_ref(), PreASAPNode::Scan { predicates, .. } if predicates.len() == 1) ); } @@ -266,7 +268,7 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { // `sum by (le)` aggregate; now the `le` grouping survives into the // canonical tree. let qe = lower(r#"histogram_quantile(0.99, sum by (le) (rate(http_requests_bucket[5m])))"#); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { @@ -278,7 +280,7 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { ); // `sum by (le)` survives as a positional Aggregate (by = [2], `le`) over the // inner Rate — no name-based Partition. - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, .. @@ -292,8 +294,8 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { /// The classic `histogram_quantile` aggregate: its `without` keys, `le` /// column, and output column names. -fn classic_histogram(qe: &QueryExpr) -> (Vec, usize, Vec) { - let QueryExpr::Aggregate { +fn classic_histogram(qe: &PreASAPNode) -> (Vec, usize, Vec) { + let PreASAPNode::Aggregate { reduction: Reduction::Reduce(by), measures, .. @@ -321,7 +323,7 @@ fn classic_histogram(qe: &QueryExpr) -> (Vec, usize, Vec) { fn classic_histogram_quantile_groups_without_le() { let qe = lower("histogram_quantile(0.9, rate(http_duration_seconds_bucket[5m]))"); let (keys, le, names) = classic_histogram(&qe); - let QueryExpr::Aggregate { child, .. } = &qe else { + let PreASAPNode::Aggregate { child, .. } = &qe else { unreachable!() }; let child = child.output_schema().unwrap(); @@ -349,14 +351,14 @@ fn classic_histogram_quantile_keeps_out_of_range_quantiles() { ("histogram_quantile(-1, x_bucket)", -1.), ("histogram_quantile(2, x_bucket)", 2.), ] { - let QueryExpr::Aggregate { measures, .. } = lower(query) else { + let PreASAPNode::Aggregate { measures, .. } = lower(query) else { panic!("{query}"); }; assert!( matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if *q == expected) ); } - let QueryExpr::Aggregate { measures, .. } = lower("histogram_quantile(NaN, x_bucket)") else { + let PreASAPNode::Aggregate { measures, .. } = lower("histogram_quantile(NaN, x_bucket)") else { panic!("NaN"); }; assert!(matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if q.is_nan())); @@ -379,14 +381,14 @@ fn classic_histogram_quantile_rejects_an_argument_without_le() { #[test] fn rate_has_time_range_child_not_window() { let qe = lower("rate(http_requests_total[5m])"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { panic!("expected Aggregate for rate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let PreASAPNode::TimeRange { range, .. } = child.as_ref() else { panic!("expected TimeRange child (not Window), got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); @@ -395,14 +397,14 @@ fn rate_has_time_range_child_not_window() { #[test] fn increase_maps_to_increase_intent() { let qe = lower("increase(errors_total[1h])"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { panic!("expected Aggregate for increase, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Increase])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let PreASAPNode::TimeRange { range, .. } = child.as_ref() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(3600)); @@ -415,21 +417,21 @@ fn sum_over_rate_keeps_both_levels() { // Regression: `sum(rate(m[w]))` — the most common PromQL shape — must keep // the cross-series Sum, not collapse to a bare per-series Rate. let qe = lower("sum(rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { panic!("expected outer Aggregate{{Sum}}, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = child.as_ref() else { panic!("expected inner Aggregate{{Rate}}, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::TimeRange { .. })); } #[test] @@ -438,7 +440,7 @@ fn sum_by_over_rate_groups_the_outer_sum() { // on a positional `Aggregate.by` (the same shape SQL produces) over the // label-preserving inner Rate. Leaf = [ts, value, job] → by = [2]. let qe = lower("sum by (job) (rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -451,7 +453,7 @@ fn sum_by_over_rate_groups_the_outer_sum() { assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!(matches!( child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -459,7 +461,7 @@ fn sum_by_over_rate_groups_the_outer_sum() { fn count_over_rate_keeps_both_levels() { // The `Outer::Count` sibling of the `sum(rate(...))` bug. let qe = lower("count(rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { @@ -468,7 +470,7 @@ fn count_over_rate_keeps_both_levels() { assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); assert!(matches!( child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -487,7 +489,7 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { ), ] { let tree = lower(query); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, reduction: actual, child, @@ -501,7 +503,7 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { "{query}: {tree:?}" ); assert_eq!(actual, &reduction, "{query}"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, reduction, child, @@ -516,7 +518,7 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { ); assert_eq!(reduction, &Reduction::PerEntity, "{query}"); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, .. } if range.as_secs() == 300) + matches!(child.as_ref(), PreASAPNode::TimeRange { range, .. } if range.as_secs() == 300) ); } } @@ -552,14 +554,14 @@ fn count_never_lowers_to_distinct_sample_values() { #[test] fn count_over_time_is_count_intent() { let qe = lower("count_over_time(m[5m])"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = &qe else { panic!("expected Aggregate"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::TimeRange { .. })); } #[test] @@ -568,7 +570,7 @@ fn outer_count_counts_series() { // over the window (label-preserving), outer cross-series row count grouped // on a positional `Aggregate.by`. Leaf = [ts, value, symbol] → symbol = col 2. let qe = lower("count by (symbol) (count_over_time(financial_last_trade_price[5m]))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -580,14 +582,14 @@ fn outer_count_counts_series() { assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); // Inner: Aggregate{Count} over TimeRange (per-series count_over_time). - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = child.as_ref() else { panic!("expected Aggregate (count_over_time) under the outer count, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::TimeRange { .. })); } // ── topk / bottomk ──────────────────────────────────────────────────────────── @@ -596,7 +598,7 @@ fn outer_count_counts_series() { fn topk_over_count_is_heavy_hitter_topk() { let qe = lower(r#"topk by (service) (10, count_over_time(requests{env="prod"}[1m]))"#); // Heavy-hitter: Aggregate{TopK} with grouping resolved to positional ids. - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -612,24 +614,24 @@ fn topk_over_count_is_heavy_hitter_topk() { [AggIntent::TopK { k: 10, .. }] )); // The count_over_time under the TopK is a TimeRange-backed aggregate. - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = child.as_ref() else { panic!("expected Aggregate (count_over_time) under TopK, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let PreASAPNode::TimeRange { range, child } = child.as_ref() else { panic!("expected TimeRange under Count aggregate, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(60)); - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::Scan { .. })); } #[test] fn topk_over_sum_is_value_weighted_heavy_hitter_topk() { let qe = lower(r#"topk by (service) (5, sum_over_time(requests{env="prod"}[1m]))"#); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -643,25 +645,25 @@ fn topk_over_sum_is_value_weighted_heavy_hitter_topk() { measures.as_slice(), [AggIntent::TopK { k: 5, .. }] )); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = child.as_ref() else { panic!("expected Aggregate (sum_over_time) under TopK, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::TimeRange { .. })); } #[test] fn topk_over_avg_is_generic_sort_limit() { let qe = lower("topk by (host) (5, avg_over_time(cpu[5m]))"); - let QueryExpr::Limit { n, offset, child } = &qe else { + let PreASAPNode::Limit { n, offset, child } = &qe else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 5); assert_eq!(*offset, 0); - let QueryExpr::Sort { + let PreASAPNode::Sort { keys, partition_by, child, @@ -678,7 +680,7 @@ fn topk_over_avg_is_generic_sort_limit() { // Underneath: the label-preserving windowed avg aggregate (by: []), no // intervening Partition. assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { reduction, measures, .. } + matches!(child.as_ref(), PreASAPNode::Aggregate { reduction, measures, .. } if reduction == &Reduction::PerEntity && matches!(measures.as_slice(), [AggIntent::Avg { .. }])), "expected bare per-series Avg aggregate under Sort, got {child:?}" ); @@ -687,7 +689,7 @@ fn topk_over_avg_is_generic_sort_limit() { #[test] fn ungrouped_topk_over_sum_is_heavy_hitter() { let qe = lower("topk(5, sum_over_time(m[5m]))"); - assert!(matches!(&qe, QueryExpr::Aggregate { .. })); + assert!(matches!(&qe, PreASAPNode::Aggregate { .. })); assert!(has_intent(&qe, |i| matches!(i, AggIntent::Sum { .. }))); assert!(has_intent(&qe, |i| matches!( i, @@ -699,11 +701,11 @@ fn ungrouped_topk_over_sum_is_heavy_hitter() { fn bottomk_over_count_is_generic_sort_ascending() { // `bottomk` is never a heavy-hitter (descending=false), even over count. let qe = lower("bottomk(3, count_over_time(m[5m]))"); - let QueryExpr::Limit { n, child, .. } = &qe else { + let PreASAPNode::Limit { n, child, .. } = &qe else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 3); - let QueryExpr::Sort { keys, .. } = child.as_ref() else { + let PreASAPNode::Sort { keys, .. } = child.as_ref() else { panic!("expected Sort"); }; assert!(keys[0].ascending, "bottomk ranks ascending"); @@ -715,11 +717,11 @@ fn bottomk_over_count_is_generic_sort_ascending() { #[test] fn bottomk_is_always_generic_sort_ascending() { let qe = lower("bottomk(3, count_over_time(m[5m]))"); - let QueryExpr::Limit { n, child, .. } = &qe else { + let PreASAPNode::Limit { n, child, .. } = &qe else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 3); - let QueryExpr::Sort { keys, .. } = child.as_ref() else { + let PreASAPNode::Sort { keys, .. } = child.as_ref() else { panic!("expected Sort"); }; assert!(keys[0].ascending, "bottomk ranks ascending"); @@ -731,7 +733,7 @@ fn topk_count_output_schema_carries_group_key() { // (`service`) flows through to the outer TopK's `by` column. Leaf schema = // [ts, value, service] → TopK groups on service (col 2). let qe = lower("topk by (service) (5, count_over_time(m[1m]))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -750,14 +752,14 @@ fn topk_count_output_schema_carries_group_key() { [AggIntent::TopK { k: 5, .. }] )); // Inner Count aggregate is visible with its TimeRange child. - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = child.as_ref() else { panic!("expected inner Aggregate{{Count}}, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::TimeRange { .. })); } // ── binary ops ──────────────────────────────────────────────────────────────── @@ -765,22 +767,22 @@ fn topk_count_output_schema_carries_group_key() { #[test] fn binary_op_division() { let qe = lower("rate(a[5m]) / rate(b[5m])"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &qe else { + let PreASAPNode::BinaryOp { op, lhs, rhs, .. } = &qe else { panic!("expected BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); assert!( - matches!(lhs.as_ref(), QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) + matches!(lhs.as_ref(), PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) ); assert!( - matches!(rhs.as_ref(), QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) + matches!(rhs.as_ref(), PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) ); } #[test] fn binary_op_with_on_grouping() { let qe = lower("a / on(host) b"); - let QueryExpr::BinaryOp { vector_match, .. } = &qe else { + let PreASAPNode::BinaryOp { vector_match, .. } = &qe else { panic!("expected BinaryOp, got {qe:?}"); }; let vm = vector_match.as_ref().expect("vector_match present"); @@ -794,7 +796,7 @@ fn binary_op_with_on_grouping() { #[test] fn bool_comparisons_are_distinct() { let op = |q: &str| match lower(q) { - QueryExpr::BinaryOp { op, .. } => op, + PreASAPNode::BinaryOp { op, .. } => op, other => panic!("expected BinaryOp, got {other:?}"), }; assert_eq!(op("a > 1"), BinaryOpKind::Compare(CompareOpKind::Gt)); @@ -814,7 +816,7 @@ fn binary_op_binds_each_branch_against_its_own_schema() { // single root schema threaded to both branches, the left scan would leak the // right's group key (and vice-versa). Per-branch binding keeps them separate. let qe = lower("count by (job) (a) / count by (region) (b)"); - let QueryExpr::BinaryOp { lhs, rhs, .. } = &qe else { + let PreASAPNode::BinaryOp { lhs, rhs, .. } = &qe else { panic!("expected BinaryOp, got {qe:?}"); }; let lcols = scan_columns(lhs); @@ -830,25 +832,25 @@ fn binary_op_binds_each_branch_against_its_own_schema() { } /// Collect every `AggIntent` in the tree, root-to-leaf. -fn all_intents(e: &QueryExpr) -> Vec { +fn all_intents(e: &PreASAPNode) -> Vec { let mut out = Vec::new(); collect_intents(e, &mut out); out } -fn collect_intents(e: &QueryExpr, out: &mut Vec) { +fn collect_intents(e: &PreASAPNode, out: &mut Vec) { match e { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { measures, child, .. } => { out.extend(measures.iter().cloned()); collect_intents(child, out); } - QueryExpr::TimeRange { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => collect_intents(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } => { + PreASAPNode::TimeRange { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } => collect_intents(child, out), + PreASAPNode::BinaryOp { lhs, rhs, .. } => { collect_intents(lhs, out); collect_intents(rhs, out); } @@ -857,19 +859,19 @@ fn collect_intents(e: &QueryExpr, out: &mut Vec) { } /// True if any `AggIntent` anywhere in the tree satisfies `pred`. -fn has_intent bool>(e: &QueryExpr, pred: F) -> bool { +fn has_intent bool>(e: &PreASAPNode, pred: F) -> bool { all_intents(e).iter().any(pred) } /// Column names on the first `Scan` reachable by descending single-child nodes. -fn scan_columns(e: &QueryExpr) -> Vec { +fn scan_columns(e: &PreASAPNode) -> Vec { match e { - QueryExpr::Scan { schema, .. } => schema.columns.iter().map(|c| c.name.clone()).collect(), - QueryExpr::Aggregate { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => scan_columns(child), + PreASAPNode::Scan { schema, .. } => schema.columns.iter().map(|c| c.name.clone()).collect(), + PreASAPNode::Aggregate { child, .. } + | PreASAPNode::TimeRange { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } => scan_columns(child), _ => vec![], } } @@ -883,7 +885,7 @@ fn without_grouping_lowers_to_the_exclusion_form() { // label is stored positionally (the SchemaResolver seeds it), the grouping is the // `without` form, and the output schema stays open. let qe = lower("sum without (instance) (rate(m[5m]))"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -899,7 +901,7 @@ fn without_grouping_lowers_to_the_exclusion_form() { // The inner per-series rate is preserved (label-preserving) under the outer // cross-series `without` reduction. assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + matches!(child.as_ref(), PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) ); assert!(!qe.output_schema().unwrap().closed); @@ -979,7 +981,7 @@ fn accuracy_target_flows_into_quantile_intent() { AccuracyTarget::Epsilon(0.01), ) .unwrap(); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let PreASAPNode::Aggregate { measures, .. } = &qe else { panic!("expected Aggregate"); }; assert!(matches!( @@ -998,7 +1000,7 @@ fn aggregate_output_schema_preserves_time_axis_and_labels() { // predicate columns) to the scan schema, so `env` appears as a column // even though it is only used as a filter. // per_series_reduction_schema preserves the time axis and all label columns. - let QueryExpr::Aggregate { .. } = &qe else { + let PreASAPNode::Aggregate { .. } = &qe else { panic!("expected Aggregate, got {qe:?}"); }; let schema = qe.output_schema().expect("aggregate schema"); @@ -1016,16 +1018,16 @@ fn scan_schema_carries_ts_value_and_group_keys() { // `service` is a group key → the SchemaResolver lands it in the self-contained // Scan schema (positional). `env` is only a filter, so it is not a column. let qe = lower("count by (service) (count_over_time(requests[1m]))"); - fn find_scan(n: &QueryExpr) -> &QueryExpr { + fn find_scan(n: &PreASAPNode) -> &PreASAPNode { match n { - QueryExpr::Scan { .. } => n, - QueryExpr::TimeRange { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Filter { child, .. } => find_scan(child), + PreASAPNode::Scan { .. } => n, + PreASAPNode::TimeRange { child, .. } + | PreASAPNode::Aggregate { child, .. } + | PreASAPNode::Filter { child, .. } => find_scan(child), other => panic!("unexpected node {other:?}"), } } - let QueryExpr::Scan { schema, .. } = find_scan(&qe) else { + let PreASAPNode::Scan { schema, .. } = find_scan(&qe) else { unreachable!() }; let mut names: Vec<&str> = schema.columns.iter().map(|c| c.name.as_str()).collect(); @@ -1113,18 +1115,18 @@ fn reducing_group_by_lowers_to_aggregate_by() { // Cross-series reduce, no keys → bare `Aggregate { reduction: Reduce([]) }`. let q = lower("sum(http_requests_total)"); assert!( - matches!(q, QueryExpr::Aggregate { ref reduction, .. } if reduction == &Reduction::by(vec![])) + matches!(q, PreASAPNode::Aggregate { ref reduction, .. } if reduction == &Reduction::by(vec![])) ); // Cross-series reduce grouped by a label → `Aggregate.reduction`. let q = lower("sum by (job) (http_requests_total)"); - assert!(matches!(q, QueryExpr::Aggregate { ref reduction, .. } + assert!(matches!(q, PreASAPNode::Aggregate { ref reduction, .. } if reduction.expect_reduce().len() == 1)); // Reduce over a label-preserving `rate` grouped by a label → still // `Aggregate.reduction` (the keys resolve against rate's preserved schema). let q = lower("sum by (job) (rate(http_requests_total[5m]))"); - assert!(matches!(q, QueryExpr::Aggregate { ref reduction, .. } + assert!(matches!(q, PreASAPNode::Aggregate { ref reduction, .. } if reduction.expect_reduce().len() == 1)); } @@ -1134,10 +1136,10 @@ fn generic_topk_grouping_lowers_to_sort_partition_by() { // reducing → the grouping rides on `Sort.partition_by`, and the windowed // reduction beneath stays label-preserving (`by: []`). No `Partition` node. let q = lower("topk by (host) (5, avg_over_time(cpu[5m]))"); - let QueryExpr::Limit { child, .. } = &q else { + let PreASAPNode::Limit { child, .. } = &q else { panic!("expected Limit, got {q:?}"); }; - let QueryExpr::Sort { + let PreASAPNode::Sort { partition_by, child, .. @@ -1147,7 +1149,7 @@ fn generic_topk_grouping_lowers_to_sort_partition_by() { }; assert_eq!(partition_by, &vec![2], "host is col 2 in [ts, value, host]"); assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { reduction, .. } if reduction == &Reduction::PerEntity) + matches!(child.as_ref(), PreASAPNode::Aggregate { reduction, .. } if reduction == &Reduction::PerEntity) ); } @@ -1160,11 +1162,11 @@ fn topk_over_bare_selector_by_label_ranks_per_group() { // Partition→Sort.partition_by reframe in #12). Expected: // Limit{3} → Sort{value desc, partition_by:[job]} → Scan let q = lower("topk(3, http_requests_total) by (job)"); - let QueryExpr::Limit { n, child, .. } = &q else { + let PreASAPNode::Limit { n, child, .. } = &q else { panic!("expected Limit, got {q:?}"); }; assert_eq!(*n, 3); - let QueryExpr::Sort { + let PreASAPNode::Sort { keys, partition_by, child, @@ -1177,7 +1179,7 @@ fn topk_over_bare_selector_by_label_ranks_per_group() { // No implicit reducing aggregate — the selector is label-preserving, so the // sort is directly over the selector horizon (the `job` label survives to partition by). assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })), + matches!(child.as_ref(), PreASAPNode::TimeRange { child, .. } if matches!(child.as_ref(), PreASAPNode::Scan { .. })), "ranking is over the bare selector horizon, not a reducing Aggregate, got {child:?}" ); assert!( @@ -1191,10 +1193,10 @@ fn topk_over_bare_selector_ranks_raw_samples() { // Even without `by`, `topk(3, m)` ranks the raw instant-vector samples — it // does not sum them. The sort sits directly over the Scan, partition empty. let q = lower("topk(3, http_requests_total)"); - let QueryExpr::Limit { child, .. } = &q else { + let PreASAPNode::Limit { child, .. } = &q else { panic!("expected Limit, got {q:?}"); }; - let QueryExpr::Sort { + let PreASAPNode::Sort { partition_by, child, .. @@ -1204,7 +1206,7 @@ fn topk_over_bare_selector_ranks_raw_samples() { }; assert!(partition_by.is_empty(), "no `by` → global ranking"); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.as_ref(), PreASAPNode::TimeRange { child, .. } if matches!(child.as_ref(), PreASAPNode::Scan { .. })) ); assert!(!has_intent(&q, |i| matches!(i, AggIntent::Sum { .. }))); } @@ -1212,20 +1214,20 @@ fn topk_over_bare_selector_ranks_raw_samples() { // ── Issue #109: histogram_quantiles fans out into one branch per φ ────────── /// The `(label value, intent)` of each `histogram_quantiles` branch. -fn quantile_branches(q: &QueryExpr) -> Vec<(String, AggIntent)> { - let QueryExpr::Concat { children, .. } = q else { +fn quantile_branches(q: &PreASAPNode) -> Vec<(String, AggIntent)> { + let PreASAPNode::Concat { children, .. } = q else { panic!("expected a Concat at the root, got {q:?}"); }; children .iter() .map(|c| { - let QueryExpr::PromqlRelabel { value, child, .. } = c else { + let PreASAPNode::PromqlRelabel { value, child, .. } = c else { panic!("expected PromqlRelabel per branch, got {c:?}"); }; - let QueryExpr::Literal(ScalarValue::Utf8(v)) = value.as_ref() else { + let PreASAPNode::Literal(ScalarValue::Utf8(v)) = value.as_ref() else { panic!("expected a literal label value, got {value:?}"); }; - let QueryExpr::Aggregate { measures, .. } = child.as_ref() else { + let PreASAPNode::Aggregate { measures, .. } = child.as_ref() else { panic!("expected an Aggregate under the PromqlRelabel, got {child:?}"); }; (v.clone(), measures[0].clone()) @@ -1271,7 +1273,7 @@ fn histogram_quantiles_branches_are_union_compatible() { // `Concat` derives its schema from the first child, so every branch must // agree on column names — the φ lives in the label, not the column name. let q = lower(r#"histogram_quantiles(testhistogram3, "q", 0.5, 0.9)"#); - let QueryExpr::Concat { children, .. } = &q else { + let PreASAPNode::Concat { children, .. } = &q else { panic!("expected Concat"); }; let shapes: Vec> = children @@ -1297,10 +1299,10 @@ fn histogram_quantiles_branches_are_union_compatible() { #[test] fn histogram_quantiles_uses_the_given_label_name() { let q = lower(r#"histogram_quantiles(h, "phi", 0.5)"#); - let QueryExpr::Concat { children, .. } = &q else { + let PreASAPNode::Concat { children, .. } = &q else { panic!("expected Concat"); }; - let QueryExpr::PromqlRelabel { dst, .. } = &children[0] else { + let PreASAPNode::PromqlRelabel { dst, .. } = &children[0] else { panic!("expected PromqlRelabel"); }; assert_eq!(dst, "phi"); @@ -1331,16 +1333,16 @@ fn histogram_quantiles_rejects_an_out_of_range_quantile() { // A subquery's `offset`/`@` shift the whole subquery, so the tree keeps them. #[test] fn subquery_time_shift_is_retained() { - let QueryExpr::Aggregate { child, .. } = lower("max_over_time(m[5m:1m] offset 1m)") else { + let PreASAPNode::Aggregate { child, .. } = lower("max_over_time(m[5m:1m] offset 1m)") else { panic!("expected a range function"); }; - let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + let PreASAPNode::TimeShift { shift, child } = child.as_ref() else { panic!("subquery offset was dropped: {child:?}"); }; assert_eq!(shift.offset_ms, 60_000); - assert!(matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::PromqlSubquery { .. })); assert!(matches!( lower("max_over_time(m[5m:1m] @ 100)"), - QueryExpr::Aggregate { child, .. } if matches!(child.as_ref(), QueryExpr::TimeShift { .. }) + PreASAPNode::Aggregate { child, .. } if matches!(child.as_ref(), PreASAPNode::TimeShift { .. }) )); } diff --git a/crates/frontend-promql/tests/support.rs b/crates/frontend-promql/tests/support.rs index 1f15b1ca..f2ec862b 100644 --- a/crates/frontend-promql/tests/support.rs +++ b/crates/frontend-promql/tests/support.rs @@ -1,7 +1,7 @@ use asap_frontend_promql::{ lower_promql_workload, lower_promql_workload_with_histograms, HistogramCatalog, PromqlError, }; -use asap_types::pre_asap::QueryExpr; +use asap_types::pre_asap::PreASAPNode; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, @@ -35,7 +35,7 @@ fn workload(query: &str, accuracy: AccuracyTarget) -> PlanningWorkload { } } -pub fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Result { +pub fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Result { let mut lowered = lower_promql_workload(&workload(query, accuracy), 0)?; Ok(lowered.remove(0)) } @@ -45,7 +45,7 @@ pub fn lower_promql_with_histograms( query: &str, accuracy: AccuracyTarget, histograms: HistogramCatalog, -) -> Result { +) -> Result { let mut lowered = lower_promql_workload_with_histograms(&workload(query, accuracy), histograms, 0)?; Ok(lowered.remove(0)) diff --git a/crates/frontend-promql/tests/univmon_candidates.rs b/crates/frontend-promql/tests/univmon_candidates.rs index d34768e8..c598d84c 100644 --- a/crates/frontend-promql/tests/univmon_candidates.rs +++ b/crates/frontend-promql/tests/univmon_candidates.rs @@ -8,9 +8,9 @@ use asap_aware_mapping::replacement::{default_strategies, search_workload_with_t use asap_aware_mapping::{Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG}; mod support; use asap_types::post_asap::{ - compile_post_asap_dag, cse::share_common_summary_subtrees, AccuracyError, BoundExpr, - CompositionOperator, ErrorMetric, ProbabilityExpr, ResultGuarantee, SketchAlgorithm, - SketchQuery, SummaryExpr, SummaryFamilyType, SummaryInputExpr, SummaryNode, + cse::share_common_summary_subtrees, export_post_asap_dag, AccuracyError, BoundExpr, + CompositionOperator, ErrorMetric, PostASAPNode, ProbabilityExpr, ResultGuarantee, + SketchAlgorithm, SketchQuery, SummaryExpr, SummaryFamilyType, SummaryInputExpr, }; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -49,7 +49,7 @@ impl AccuracyModel for TestEvidence { } } -fn candidate(query: &str, accuracy: AccuracyTarget) -> Rc { +fn candidate(query: &str, accuracy: AccuracyTarget) -> Rc { let root = lower_promql(query, accuracy).unwrap(); SketchAlgorithmStrategy::new_with_planning_inputs(&DefaultCostModel, &TestEvidence, &EqualSplitAllocator) .replacements(&TargetSubDAG::new(&Rc::new(root))) @@ -115,7 +115,7 @@ fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { "production has no calibrated error bound" ); } - compile_post_asap_dag(root).unwrap(); + export_post_asap_dag(root).unwrap(); } } @@ -157,14 +157,14 @@ fn uncalibrated_frequency_readouts_do_not_bypass_accuracy_targets() { &DefaultAccuracyModel, ); assert!(space - .candidates_for_target(&space.roots[0].1) + .candidates_for_target(&space.roots()[0].1) .unwrap() .candidates .iter() .any(|candidate| candidate.has_missing_accuracy_evidence())); assert!(!space .global_selection(&DefaultCostModel) - .for_target(&space.roots[0].1) + .for_target(&space.roots()[0].1) .unwrap() .chosen .is_some_and(|candidate| candidate.has_missing_accuracy_evidence())); diff --git a/crates/frontend-sql/src/lib.rs b/crates/frontend-sql/src/lib.rs index 58474e2e..ea2c6833 100644 --- a/crates/frontend-sql/src/lib.rs +++ b/crates/frontend-sql/src/lib.rs @@ -1,8 +1,8 @@ //! SQL front end: parse + plan (via DataFusion) → the canonical, unresolved //! shape, built directly (issue #179) → [`resolve_root`]. //! -//! Emits [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) itself — the -//! canonical `QueryExpr`, generic over an unresolved +//! Emits [`UnresolvedPreASAPNode`](asap_types::pre_asap::UnresolvedPreASAPNode) itself — the +//! canonical `PreASAPNode`, generic over an unresolved //! [`ColumnRef`](asap_types::pre_asap::ColumnRef) — directly, rather than a //! separate per-language relational tree; `resolve_root` runs the //! [`SchemaResolver`](asap_types::pre_asap::SchemaResolver) for positional name resolution. @@ -12,14 +12,14 @@ pub mod error; pub mod sql; use asap_types::pre_asap::resolve_root; -use asap_types::pre_asap::QueryExpr; +use asap_types::pre_asap::PreASAPNode; use asap_types::types::AccuracyTarget; use asap_types::workload::{QueryLanguage, QueryWorkload, SqlDialect}; pub use error::SqlError; pub use sql::{SqlCatalog, SqlLowerer}; -/// Lower a single SQL query string to the canonical, resolved `QueryExpr`, +/// Lower a single SQL query string to the canonical, resolved `PreASAPNode`, /// parsed as `SqlDialect::DataFusionSQL`. /// /// The `catalog` supplies table schemas (used both to plan the SQL with @@ -29,7 +29,7 @@ pub async fn lower_sql( query: &str, catalog: &SqlCatalog, accuracy: AccuracyTarget, -) -> Result { +) -> Result { lower_sql_dialect(query, catalog, SqlDialect::DataFusionSQL, accuracy).await } @@ -46,7 +46,7 @@ pub async fn lower_sql_dialect( catalog: &SqlCatalog, dialect: SqlDialect, accuracy: AccuracyTarget, -) -> Result { +) -> Result { let unresolved = SqlLowerer::with_dialect(catalog, dialect) .lower(query, &accuracy) .await?; @@ -59,7 +59,7 @@ pub async fn lower_sql_dialect( Ok(resolved) } -/// Lower every SQL batch entry in `workload` to a `QueryExpr`. +/// Lower every SQL batch entry in `workload` to a `PreASAPNode`. /// /// One `Result` per entry — errors are per-query, not fatal for the batch. /// Returns `WrongLanguage` for every entry if the workload is not SQL, and @@ -67,7 +67,7 @@ pub async fn lower_sql_dialect( pub async fn lower_sql_batch( workload: &QueryWorkload, catalog: &SqlCatalog, -) -> Vec> { +) -> Vec> { let entries = match &workload.query_batch { Some(e) if !e.is_empty() => e, _ => return vec![], diff --git a/crates/frontend-sql/src/sql/collection_planning.rs b/crates/frontend-sql/src/sql/collection_planning.rs index 92fda701..3c902fff 100644 --- a/crates/frontend-sql/src/sql/collection_planning.rs +++ b/crates/frontend-sql/src/sql/collection_planning.rs @@ -4,7 +4,7 @@ use super::types::{arrow_to_dtype, dtype_to_arrow, scalar_value_to_asap}; use asap_types::pre_asap::scalar_signature::{ element_access_type, struct_field_type, MapScalarFunction, }; -use asap_types::pre_asap::{Column, QueryExpr, Schema}; +use asap_types::pre_asap::{Column, PreASAPNode, Schema}; use datafusion::arrow::datatypes::DataType; use datafusion::common::{DataFusionError, ExprSchema, Result}; use datafusion::logical_expr::{ @@ -91,10 +91,10 @@ impl CollectionPlanningFunction { if let Some(Expr::Literal(value)) = expressions.and_then(|args| args.get(index)) { scalar_value_to_asap(value) - .map(QueryExpr::Literal) + .map(PreASAPNode::Literal) .map_err(|error| DataFusionError::Plan(error.to_string())) } else { - Ok(QueryExpr::Column(index)) + Ok(PreASAPNode::Column(index)) } }) .collect::>>()?; diff --git a/crates/frontend-sql/src/sql/mod.rs b/crates/frontend-sql/src/sql/mod.rs index d18e6821..75b6ac20 100644 --- a/crates/frontend-sql/src/sql/mod.rs +++ b/crates/frontend-sql/src/sql/mod.rs @@ -1,12 +1,12 @@ //! SQL → the canonical, unresolved -//! [`UnresolvedQueryExpr`](asap_types::pre_asap::query_expr::UnresolvedQueryExpr) -//! (`QueryExpr`). +//! [`UnresolvedPreASAPNode`](asap_types::pre_asap::query_expr::UnresolvedPreASAPNode) +//! (`PreASAPNode`). //! //! Parses SQL via DataFusion (over the catalog's registered tables), then -//! walks the unoptimized `LogicalPlan` and emits `UnresolvedQueryExpr` nodes with +//! walks the unoptimized `LogicalPlan` and emits `UnresolvedPreASAPNode` nodes with //! unresolved `ColumnRef`s directly (issue #179) — the same tree shape //! [`resolve_root`](asap_types::pre_asap::resolve_root) binds to canonical, -//! positional `QueryExpr`. Unlike PromQL's front end, SQL's +//! positional `PreASAPNode`. Unlike PromQL's front end, SQL's //! Ordinary SQL `Aggregate` nodes are `Reduction::Reduce`. The explicit //! `asap_rate`/`asap_increase` bridge is the narrow exception: it //! spells a time-series range reducer with an explicit value, time-index, and @@ -54,7 +54,7 @@ use asap_sql_function_catalog::{AggSemantic, Arity, RewriteKind}; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::pre_asap::query_expr::{ GroupKeys, Predicate, ProjectItem, Reduction, SortKey, Source, - UnresolvedQueryExpr as Unresolved, WindowFrame, WindowFrameBound, WindowFrameOffset, + UnresolvedPreASAPNode as Unresolved, WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, }; use asap_types::pre_asap::schema::{DataType, Schema}; @@ -108,7 +108,7 @@ fn current_accuracy() -> AccuracyTarget { ACCURACY.with(|a| a.borrow().clone()) } -/// Lowers SQL strings to the canonical [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) +/// Lowers SQL strings to the canonical [`UnresolvedPreASAPNode`](asap_types::pre_asap::UnresolvedPreASAPNode) /// over a table [`SqlCatalog`]. Call /// [`resolve_root`](asap_types::pre_asap::resolve_root) on the result for /// the canonical, resolved tree. diff --git a/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs b/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs index ae27e31e..ac52dda1 100644 --- a/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs +++ b/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs @@ -37,7 +37,7 @@ use asap_frontend_sql::{lower_sql_dialect, SqlCatalog, SqlError as LoweringError}; use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::pre_asap::{AggIntent, GroupKeys, PreASAPNode}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use datafusion::error::DataFusionError; @@ -92,7 +92,7 @@ fn queries() -> Vec { .collect() } -async fn lower(q: &str) -> Result { +async fn lower(q: &str) -> Result { lower_sql_dialect( q, &catalog(), @@ -218,19 +218,19 @@ async fn corpus_lowering_matches_the_pinned_per_query_outcome() { ); } -fn first_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { +fn first_aggregate(qe: &PreASAPNode) -> Option<(&GroupKeys, &Vec)> { match qe { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_aggregate(child), + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } => first_aggregate(child), _ => None, } } @@ -260,7 +260,7 @@ async fn top_k_queries_are_count_grouped_by_prefix() { idx + 1 ); assert!( - matches!(qe, QueryExpr::Limit { .. }), + matches!(qe, PreASAPNode::Limit { .. }), "q{} ({label}) top-k shape keeps the LIMIT at the root: {qe:?}", idx + 1 ); diff --git a/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs b/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs index 302a349a..1c115859 100644 --- a/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs +++ b/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs @@ -70,7 +70,7 @@ fn catalog() -> SqlCatalog { .with_table("bgp.bgp_updates", updates) } -async fn lower(q: &str) -> Result { +async fn lower(q: &str) -> Result { lower_sql_dialect( q, &catalog(), diff --git a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs index 273597e7..f0ff7679 100644 --- a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs +++ b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs @@ -21,7 +21,7 @@ use asap_frontend_sql::{lower_sql, SqlCatalog, SqlError as LoweringError}; use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::pre_asap::{AggIntent, GroupKeys, PreASAPNode}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/synthetic_packet_trace_queries.sql"); @@ -68,35 +68,35 @@ fn queries() -> Vec { // ── tree helpers ────────────────────────────────────────────────────────────── /// Every `AggIntent` in the tree, root-to-leaf. -fn intents(e: &QueryExpr) -> Vec { +fn intents(e: &PreASAPNode) -> Vec { let mut out = Vec::new(); - fn go(e: &QueryExpr, out: &mut Vec) { + fn go(e: &PreASAPNode, out: &mut Vec) { match e { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { measures, child, .. } => { out.extend(measures.iter().cloned()); go(child, out); } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::Project { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => go(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { + PreASAPNode::TimeRange { child, .. } + | PreASAPNode::TimeShift { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::SQLWindowFunc { child, .. } + | PreASAPNode::Project { child, .. } + | PreASAPNode::PromqlRelabel { child, .. } + | PreASAPNode::PromqlSeriesSample { child, .. } + | PreASAPNode::PromqlInfoEnrich { child, .. } => go(child, out), + PreASAPNode::BinaryOp { lhs, rhs, .. } + | PreASAPNode::Join { left: lhs, right: rhs, .. } - | QueryExpr::SetOp { + | PreASAPNode::SetOp { left: lhs, right: rhs, .. @@ -104,30 +104,29 @@ fn intents(e: &QueryExpr) -> Vec { go(lhs, out); go(rhs, out); } - QueryExpr::Concat { children, .. } => children.iter().for_each(|c| go(c, out)), - QueryExpr::PromqlVectorFromScalar(inner) | QueryExpr::PromqlScalarFromVector(inner) => { - go(inner, out) - } - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} + PreASAPNode::Concat { children, .. } => children.iter().for_each(|c| go(c, out)), + PreASAPNode::PromqlVectorFromScalar(inner) + | PreASAPNode::PromqlScalarFromVector(inner) => go(inner, out), + PreASAPNode::Scan { .. } + | PreASAPNode::PromqlScalarBridge(_) + | PreASAPNode::EvalTimestamp + | PreASAPNode::CurrentTimestamp => {} // Scalar expression variants (issue #205): `AggIntent` only ever // lives in `Aggregate.measures`, never nested inside a scalar // expression tree, so there's nothing to recurse into here. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} + PreASAPNode::Column(_) + | PreASAPNode::Literal(_) + | PreASAPNode::Compare { .. } + | PreASAPNode::BoolAnd(_) + | PreASAPNode::BoolOr(_) + | PreASAPNode::Not(_) + | PreASAPNode::IsNull(_) + | PreASAPNode::IsNotNull(_) + | PreASAPNode::Cast { .. } + | PreASAPNode::InList { .. } + | PreASAPNode::FunctionCall { .. } + | PreASAPNode::Arithmetic { .. } + | PreASAPNode::Case { .. } => {} } } go(e, &mut out); @@ -137,40 +136,40 @@ fn intents(e: &QueryExpr) -> Vec { /// The first `Aggregate`'s `(by, measures)` along the single-child spine. SQL /// never lowers to `Reduction::PerEntity` (it has no per-series concept), so /// `expect_reduce()` here is a safe, load-bearing assumption for these tests. -fn first_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { +fn first_aggregate(qe: &PreASAPNode) -> Option<(&GroupKeys, &Vec)> { match qe { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_aggregate(child), + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::SQLWindowFunc { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } => first_aggregate(child), _ => None, } } /// Whether a `SQLWindowFunc` (analytic `OVER (…)`) node appears anywhere. -fn has_window_func(qe: &QueryExpr) -> bool { +fn has_window_func(qe: &PreASAPNode) -> bool { match qe { - QueryExpr::SQLWindowFunc { .. } => true, - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => has_window_func(child), + PreASAPNode::SQLWindowFunc { .. } => true, + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Aggregate { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } => has_window_func(child), _ => false, } } -async fn lower(q: &str) -> QueryExpr { +async fn lower(q: &str) -> PreASAPNode { lower_sql(q, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) diff --git a/crates/frontend-sql/tests/maintained_population.rs b/crates/frontend-sql/tests/maintained_population.rs index 0019b3b8..c945f812 100644 --- a/crates/frontend-sql/tests/maintained_population.rs +++ b/crates/frontend-sql/tests/maintained_population.rs @@ -3,16 +3,16 @@ use asap_aware_mapping::maintained_population::MaintainedPopulationStrategy; use asap_frontend_sql::{lower_sql, SqlCatalog}; use asap_types::{ post_asap::{ - compile_post_asap_dag, + export_post_asap_dag, maintained_population::{MaintainedPopulation, PopulationInput}, share_common_summary_subtrees, SummaryExpr, ValueOperation, }, - pre_asap::{Column, DataType, QueryExpr, Schema}, + pre_asap::{Column, DataType, PreASAPNode, Schema}, types::AccuracyTarget, }; use std::rc::Rc; -async fn aggregate(q: &str) -> Rc { +async fn aggregate(q: &str) -> Rc { let catalog = SqlCatalog::new().with_table( "samples", Schema::new(vec![ @@ -25,9 +25,9 @@ async fn aggregate(q: &str) -> Rc { } fn population( - mut node: &asap_types::post_asap::SummaryNode, + mut node: &asap_types::post_asap::PostASAPNode, ) -> ( - &Rc, + &Rc, &MaintainedPopulation, ) { while let SummaryExpr::ValueOperation { @@ -67,7 +67,7 @@ async fn sql_quantiles_share_rows_without_promql_lookback() { .collect(), ); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + export_post_asap_dag(plan).unwrap(); } let (a, spec) = population(&plans[0].1); let (b, _) = population(&plans[1].1); @@ -128,7 +128,7 @@ async fn sql_scalar_readouts_share_membership() { .collect(), ); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + export_post_asap_dag(plan).unwrap(); assert!(Rc::ptr_eq(population(&plans[0].1).0, population(plan).0)); } } @@ -156,7 +156,7 @@ async fn malformed_table_population_fails_validation() { unreachable!() }; *value_column = 1; - assert!(compile_post_asap_dag(&candidate).is_err()); + assert!(export_post_asap_dag(&candidate).is_err()); } // SQL ORDER BY value DESC LIMIT k uses the same maximum-k state contract. @@ -175,7 +175,7 @@ async fn sql_topk_limits_share_maximum_k() { .collect(), ); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + export_post_asap_dag(plan).unwrap(); assert_eq!(population(plan).1.max_k, 5); assert!(Rc::ptr_eq(population(&plans[0].1).0, population(plan).0)); } diff --git a/crates/frontend-sql/tests/netflow/netflow.rs b/crates/frontend-sql/tests/netflow/netflow.rs index 09d742b8..1dccb04a 100644 --- a/crates/frontend-sql/tests/netflow/netflow.rs +++ b/crates/frontend-sql/tests/netflow/netflow.rs @@ -6,7 +6,7 @@ use asap_frontend_sql::{lower_sql, SqlCatalog}; use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::pre_asap::{AggIntent, GroupKeys, PreASAPNode}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/netflow.sql"); @@ -123,7 +123,7 @@ async fn netflow_sql_corpus_lowers_to_expected_intents() { } } -fn assert_expected(qe: &QueryExpr, expected: Expected, case_no: usize) { +fn assert_expected(qe: &PreASAPNode, expected: Expected, case_no: usize) { match expected { Expected::Quantile { q, by } => { let (actual_by, measures) = first_aggregate(qe).expect("expected Aggregate"); @@ -190,49 +190,49 @@ impl AggKind { } } -fn first_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { +fn first_aggregate(qe: &PreASAPNode) -> Option<(&GroupKeys, &Vec)> { match qe { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_aggregate(child), + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } => first_aggregate(child), _ => None, } } -fn has_scan_predicate(qe: &QueryExpr) -> bool { +fn has_scan_predicate(qe: &PreASAPNode) -> bool { any_node( qe, - |node| matches!(node, QueryExpr::Scan { predicates, .. } if !predicates.is_empty()), + |node| matches!(node, PreASAPNode::Scan { predicates, .. } if !predicates.is_empty()), ) } -fn has_topk(qe: &QueryExpr, k: usize) -> bool { +fn has_topk(qe: &PreASAPNode, k: usize) -> bool { any_node(qe, |node| { matches!( node, - QueryExpr::Aggregate { measures, .. } + PreASAPNode::Aggregate { measures, .. } if measures.iter().any(|agg| matches!(agg, AggIntent::TopK { k: actual, .. } if *actual == k)) ) }) } fn aggregate_by_with( - qe: &QueryExpr, + qe: &PreASAPNode, by: &'static [usize], pred: impl Fn(&AggIntent) -> bool, ) -> bool { let expected_by = GroupKeys::by(by.to_vec()); let mut found = false; visit(qe, &mut |node| { - if let QueryExpr::Aggregate { + if let PreASAPNode::Aggregate { reduction, measures, .. @@ -244,45 +244,45 @@ fn aggregate_by_with( found } -fn all_intents(qe: &QueryExpr) -> Vec { +fn all_intents(qe: &PreASAPNode) -> Vec { let mut intents = Vec::new(); visit(qe, &mut |node| { - if let QueryExpr::Aggregate { measures, .. } = node { + if let PreASAPNode::Aggregate { measures, .. } = node { intents.extend(measures.iter().cloned()); } }); intents } -fn any_node(qe: &QueryExpr, pred: impl Fn(&QueryExpr) -> bool) -> bool { +fn any_node(qe: &PreASAPNode, pred: impl Fn(&PreASAPNode) -> bool) -> bool { let mut found = false; visit(qe, &mut |node| found |= pred(node)); found } -fn visit(qe: &QueryExpr, f: &mut impl FnMut(&QueryExpr)) { +fn visit(qe: &PreASAPNode, f: &mut impl FnMut(&PreASAPNode)) { f(qe); match qe { - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => visit(child, f), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Aggregate { child, .. } + | PreASAPNode::TimeRange { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::SQLWindowFunc { child, .. } + | PreASAPNode::PromqlRelabel { child, .. } + | PreASAPNode::PromqlSeriesSample { child, .. } + | PreASAPNode::TimeShift { child, .. } + | PreASAPNode::PromqlInfoEnrich { child, .. } => visit(child, f), + PreASAPNode::BinaryOp { lhs, rhs, .. } + | PreASAPNode::Join { left: lhs, right: rhs, .. } - | QueryExpr::SetOp { + | PreASAPNode::SetOp { left: lhs, right: rhs, .. @@ -290,32 +290,32 @@ fn visit(qe: &QueryExpr, f: &mut impl FnMut(&QueryExpr)) { visit(lhs, f); visit(rhs, f); } - QueryExpr::Concat { children, .. } => { + PreASAPNode::Concat { children, .. } => { for child in children { visit(child, f); } } - QueryExpr::PromqlVectorFromScalar(child) | QueryExpr::PromqlScalarFromVector(child) => { + PreASAPNode::PromqlVectorFromScalar(child) | PreASAPNode::PromqlScalarFromVector(child) => { visit(child, f) } - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} + PreASAPNode::Scan { .. } + | PreASAPNode::PromqlScalarBridge(_) + | PreASAPNode::EvalTimestamp + | PreASAPNode::CurrentTimestamp => {} // Scalar expression variants (issue #205) aren't relational nodes; // this visitor only walks the relational tree, so stop here. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} + PreASAPNode::Column(_) + | PreASAPNode::Literal(_) + | PreASAPNode::Compare { .. } + | PreASAPNode::BoolAnd(_) + | PreASAPNode::BoolOr(_) + | PreASAPNode::Not(_) + | PreASAPNode::IsNull(_) + | PreASAPNode::IsNotNull(_) + | PreASAPNode::Cast { .. } + | PreASAPNode::InList { .. } + | PreASAPNode::FunctionCall { .. } + | PreASAPNode::Arithmetic { .. } + | PreASAPNode::Case { .. } => {} } } diff --git a/crates/frontend-sql/tests/pearson_corr.rs b/crates/frontend-sql/tests/pearson_corr.rs index ae728e49..64eaa4e4 100644 --- a/crates/frontend-sql/tests/pearson_corr.rs +++ b/crates/frontend-sql/tests/pearson_corr.rs @@ -2,7 +2,7 @@ use std::rc::Rc; use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_types::pre_asap::{AggIntent, Column, DataType, QueryExpr, Schema}; +use asap_types::pre_asap::{AggIntent, Column, DataType, PreASAPNode, Schema}; use asap_types::types::AccuracyTarget; fn catalog() -> SqlCatalog { @@ -16,20 +16,20 @@ fn catalog() -> SqlCatalog { .with_table("b", schema) } -async fn lower(sql: &str) -> QueryExpr { +async fn lower(sql: &str) -> PreASAPNode { lower_sql(sql, &catalog(), AccuracyTarget::Exact) .await .unwrap() } -fn aggregate(query: &QueryExpr) -> (&[AggIntent], &QueryExpr) { +fn aggregate(query: &PreASAPNode) -> (&[AggIntent], &PreASAPNode) { match query { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { measures, child, .. } => (measures, child), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } => aggregate(child), + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } => aggregate(child), other => panic!("expected aggregate, got {other:?}"), } } @@ -46,13 +46,13 @@ async fn corr_materializes_both_arguments() { let query = lower(sql).await; let (measures, child) = aggregate(&query); assert_eq!(measures, &[AggIntent::PearsonCorr { left: 0, right: 1 }]); - let QueryExpr::Project { cols, .. } = child else { + let PreASAPNode::Project { cols, .. } = child else { panic!("derived inputs") }; assert_eq!(cols.len(), 2); assert!(cols .iter() - .any(|col| !matches!(col.expr, QueryExpr::Column(_)))); + .any(|col| !matches!(col.expr, PreASAPNode::Column(_)))); assert_eq!( query.output_schema().unwrap().columns[0].dtype, DataType::Float64 @@ -66,11 +66,11 @@ async fn corr_preserves_qualified_join_inputs() { let query = lower("SELECT corr(a.x, b.x) FROM a JOIN b ON a.g = b.g").await; let (measures, child) = aggregate(&query); assert_eq!(measures[0].input_cols(), vec![0, 1]); - let QueryExpr::Project { cols, .. } = child else { + let PreASAPNode::Project { cols, .. } = child else { panic!("paired projection") }; - assert_eq!(cols[0].expr, QueryExpr::Column(0)); - assert_eq!(cols[1].expr, QueryExpr::Column(3)); + assert_eq!(cols[0].expr, PreASAPNode::Column(0)); + assert_eq!(cols[1].expr, PreASAPNode::Column(3)); } // Grouping and sibling reducers cannot drop either correlation argument. @@ -99,7 +99,7 @@ async fn corr_repeated_input_and_serialization() { let query = lower("SELECT corr(x, x) FROM a").await; assert_eq!(aggregate(&query).0[0].input_cols(), vec![0, 0]); let encoded = serde_json::to_string(&query).unwrap(); - let decoded: QueryExpr = serde_json::from_str(&encoded).unwrap(); + let decoded: PreASAPNode = serde_json::from_str(&encoded).unwrap(); assert_eq!(query, decoded); } @@ -132,5 +132,5 @@ async fn corr_survives_exact_plan_compilation() { panic!("expected exact fallback"); }; assert_eq!(aggregate(retained).0, aggregate(&query).0); - asap_types::post_asap::compile_post_asap_dag(&plan).unwrap(); + asap_types::post_asap::export_post_asap_dag(&plan).unwrap(); } diff --git a/crates/frontend-sql/tests/sql_lowering.rs b/crates/frontend-sql/tests/sql_lowering.rs index c61ddcd4..6a3ef655 100644 --- a/crates/frontend-sql/tests/sql_lowering.rs +++ b/crates/frontend-sql/tests/sql_lowering.rs @@ -1,14 +1,14 @@ //! End-to-end SQL → unresolved → canonical tree lowering tests (positional IR). //! //! Validates the DataFusion front end: SQL parses + plans, lowers directly to -//! the canonical, unresolved shape (`QueryExpr`, issue #179), and +//! the canonical, unresolved shape (`PreASAPNode`, issue #179), and //! the shared `resolve_root` produces the positional, resolved canonical //! tree (the same resolver the PromQL path uses). use asap_frontend_sql::{lower_sql, lower_sql_dialect, SqlCatalog, SqlError as LoweringError}; use asap_types::pre_asap::schema::{Column, DataType, Schema}; use asap_types::pre_asap::{ - AggIntent, CompareOpKind, GroupKeys, JoinKind, QueryExpr, Reduction, ScalarValue, Source, + AggIntent, CompareOpKind, GroupKeys, JoinKind, PreASAPNode, Reduction, ScalarValue, Source, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, }; use asap_types::types::AccuracyTarget; @@ -43,7 +43,7 @@ fn catalog() -> SqlCatalog { ) } -async fn lower(sql: &str) -> QueryExpr { +async fn lower(sql: &str) -> PreASAPNode { lower_sql(sql, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) @@ -57,13 +57,13 @@ async fn planning_subquery_bridge_reuses_canonical_promql_subquery() { SELECT sum(bytes) AS value FROM metrics))", ) .await; - let QueryExpr::Project { child, .. } = query else { + let PreASAPNode::Project { child, .. } = query else { panic!("expected outer SQL projection"); }; - let QueryExpr::Aggregate { child, .. } = child.as_ref() else { + let PreASAPNode::Aggregate { child, .. } = child.as_ref() else { panic!("expected outer max aggregate, got {child:?}"); }; - let QueryExpr::PromqlSubquery { + let PreASAPNode::PromqlSubquery { range, resolution, child, @@ -73,7 +73,7 @@ async fn planning_subquery_bridge_reuses_canonical_promql_subquery() { }; assert_eq!(*range, std::time::Duration::from_secs(6 * 60 * 60)); assert_eq!(*resolution, Some(std::time::Duration::from_secs(60))); - assert!(matches!(child.as_ref(), QueryExpr::Project { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::Project { .. })); } #[tokio::test] @@ -83,7 +83,7 @@ async fn planning_histogram_bridge_reuses_classic_bucket_intent() { SELECT service AS le, sum(bytes) AS value FROM metrics GROUP BY service)", ) .await; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -98,7 +98,7 @@ async fn planning_histogram_bridge_reuses_classic_bucket_intent() { measures.as_slice(), [AggIntent::HistogramQuantile { q, le: 0 }] if (*q - 0.95).abs() < 1e-12 )); - assert!(matches!(child.as_ref(), QueryExpr::Project { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::Project { .. })); } #[tokio::test] @@ -134,31 +134,31 @@ async fn planning_relation_bridges_reject_ambiguous_shapes() { } /// Find the first `Aggregate` node along the single-child spine. -fn find_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { +fn find_aggregate(qe: &PreASAPNode) -> Option<(&GroupKeys, &Vec)> { match qe { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_aggregate(child), + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } => find_aggregate(child), _ => None, } } /// The first `Aggregate` node itself, for tests that need its child. -fn find_aggregate_node(qe: &QueryExpr) -> Option<&QueryExpr> { +fn find_aggregate_node(qe: &PreASAPNode) -> Option<&PreASAPNode> { match qe { - QueryExpr::Aggregate { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => find_aggregate_node(child), + PreASAPNode::Aggregate { .. } => Some(qe), + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } => find_aggregate_node(child), _ => None, } } @@ -166,8 +166,8 @@ fn find_aggregate_node(qe: &QueryExpr) -> Option<&QueryExpr> { /// The names of the columns the first `Aggregate`'s reducers read, resolved /// against its child's schema, plus whether that child is a materializing /// `Project` (issue #110). -fn reducer_input_names(qe: &QueryExpr) -> (Vec, bool) { - let QueryExpr::Aggregate { +fn reducer_input_names(qe: &PreASAPNode) -> (Vec, bool) { + let PreASAPNode::Aggregate { measures, child, .. } = find_aggregate_node(qe).expect("expected an Aggregate") else { @@ -179,34 +179,34 @@ fn reducer_input_names(qe: &QueryExpr) -> (Vec, bool) { .flat_map(|a| a.input_cols()) .map(|id| schema.columns[id].name.clone()) .collect(); - (names, matches!(**child, QueryExpr::Project { .. })) + (names, matches!(**child, PreASAPNode::Project { .. })) } /// Find the first `Join` node along the single-child spine. -fn find_join(qe: &QueryExpr) -> Option<&QueryExpr> { +fn find_join(qe: &PreASAPNode) -> Option<&PreASAPNode> { match qe { - QueryExpr::Join { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_join(child), + PreASAPNode::Join { .. } => Some(qe), + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Aggregate { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } => find_join(child), _ => None, } } /// The first `Filter` node along the single-child spine. -fn find_filter(qe: &QueryExpr) -> Option<&QueryExpr> { +fn find_filter(qe: &PreASAPNode) -> Option<&PreASAPNode> { match qe { - QueryExpr::Filter { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_filter(child), + PreASAPNode::Filter { .. } => Some(qe), + PreASAPNode::Project { child, .. } + | PreASAPNode::Aggregate { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } => find_filter(child), _ => None, } } @@ -215,7 +215,7 @@ fn find_filter(qe: &QueryExpr) -> Option<&QueryExpr> { async fn select_star_with_where_folds_predicate_onto_scan() { // SELECT * elides the projection; WHERE folds onto the Scan predicates. let qe = lower("SELECT * FROM metrics WHERE service = 'api'").await; - let QueryExpr::Scan { + let PreASAPNode::Scan { source, predicates, schema, @@ -315,7 +315,7 @@ async fn count_ranked_topk_is_heavy_hitter() { "count-ranked topk → heavy-hitter TopK, got {measures:?}" ); // The inner child is the explicit Count, grouped by service (col 1). - let QueryExpr::Aggregate { child, .. } = &qe else { + let PreASAPNode::Aggregate { child, .. } = &qe else { panic!("expected outer Aggregate, got {qe:?}"); }; let (inner_by, inner_measures) = find_aggregate(child).expect("expected inner Count aggregate"); @@ -426,7 +426,7 @@ async fn select_distinct_lowers_to_distinct_with_positional_cols() { // (not name-based ColumnRefs). DataFusion's `Distinct::All` dedups on every // column, so `cols` is empty here — but the field type is now `Vec`. let qe = lower("SELECT DISTINCT service FROM metrics").await; - let QueryExpr::Dedup { cols, .. } = &qe else { + let PreASAPNode::Dedup { cols, .. } = &qe else { panic!("expected a Dedup at the root, got {qe:?}"); }; let _: &Vec = cols; // compile-time: positional ids, not ColumnRefs @@ -442,24 +442,24 @@ async fn inner_join_lowers_to_join_over_two_scans() { ) .await; let join = find_join(&qe).expect("expected a Join in the tree"); - let QueryExpr::Join { + let PreASAPNode::Join { kind, left, right, .. } = join else { unreachable!("find_join only returns Join"); }; assert_eq!(*kind, JoinKind::Inner); - assert!(matches!(left.as_ref(), QueryExpr::Scan { .. })); - assert!(matches!(right.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(left.as_ref(), PreASAPNode::Scan { .. })); + assert!(matches!(right.as_ref(), PreASAPNode::Scan { .. })); } /// The two `ColumnId`s an equijoin predicate `Column(l) = Column(r)` binds to, /// returned sorted so the assertion is independent of left/right ordering. -fn join_eq_columns(join: &QueryExpr) -> [usize; 2] { - let QueryExpr::Join { pred, .. } = join else { +fn join_eq_columns(join: &PreASAPNode) -> [usize; 2] { + let PreASAPNode::Join { pred, .. } = join else { unreachable!("expected a Join"); }; - let QueryExpr::Compare { + let PreASAPNode::Compare { left, op: CompareOpKind::Eq, right, @@ -468,7 +468,7 @@ fn join_eq_columns(join: &QueryExpr) -> [usize; 2] { panic!("expected an equijoin Compare, got {:?}", pred.0); }; match (left.as_ref(), right.as_ref()) { - (QueryExpr::Column(l), QueryExpr::Column(r)) => { + (PreASAPNode::Column(l), PreASAPNode::Column(r)) => { let mut cols = [*l, *r]; cols.sort_unstable(); cols @@ -567,12 +567,12 @@ async fn qualified_where_over_join_resolves_to_right_side() { ) .await; let filter = find_filter(&qe).expect("expected a Filter over the join"); - let QueryExpr::Filter { pred, .. } = filter else { + let PreASAPNode::Filter { pred, .. } = filter else { unreachable!("find_filter only returns Filter"); }; assert!( - matches!(pred.0.as_ref(), QueryExpr::Compare { left, op: CompareOpKind::Eq, .. } - if matches!(left.as_ref(), QueryExpr::Column(4))), + matches!(pred.0.as_ref(), PreASAPNode::Compare { left, op: CompareOpKind::Eq, .. } + if matches!(left.as_ref(), PreASAPNode::Column(4))), "hosts.service must bind to concatenated position 4 (not the first `service`), got {:?}", pred.0 ); @@ -638,8 +638,8 @@ async fn aggregate_over_join_binds_against_concatenated_schema() { // ── Issue #111: IN / EXISTS subquery predicates become semi / anti joins ──── /// The first `Join` node's `(kind, predicate, left column count)`. -fn join_parts(qe: &QueryExpr) -> (&JoinKind, &QueryExpr, usize) { - let QueryExpr::Join { +fn join_parts(qe: &PreASAPNode) -> (&JoinKind, &PreASAPNode, usize) { + let PreASAPNode::Join { kind, pred, left, @@ -665,13 +665,13 @@ async fn in_subquery_lowers_to_a_semi_join() { // `service` column, so a name-based lookup would bind *both* sides to the // left's — silently making this `service = service`, always true. The key is // projected under a synthetic name to make that impossible. - let QueryExpr::Compare { left, right, .. } = pred else { + let PreASAPNode::Compare { left, right, .. } = pred else { panic!("expected a comparison, got {pred:?}"); }; - assert_eq!(**left, QueryExpr::Column(1), "outer service"); + assert_eq!(**left, PreASAPNode::Column(1), "outer service"); assert_eq!( **right, - QueryExpr::Column(left_len), + PreASAPNode::Column(left_len), "the subquery key, not the outer column again" ); } @@ -723,13 +723,13 @@ async fn an_ordinary_conjunct_still_folds_onto_the_scan() { AND service IN (SELECT service FROM hosts)", ) .await; - fn scan_has_predicate(qe: &QueryExpr) -> bool { + fn scan_has_predicate(qe: &PreASAPNode) -> bool { match qe { - QueryExpr::Scan { predicates, .. } => !predicates.is_empty(), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } => scan_has_predicate(child), - QueryExpr::Join { left, right, .. } => { + PreASAPNode::Scan { predicates, .. } => !predicates.is_empty(), + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Aggregate { child, .. } => scan_has_predicate(child), + PreASAPNode::Join { left, right, .. } => { scan_has_predicate(left) || scan_has_predicate(right) } _ => false, @@ -743,16 +743,16 @@ async fn an_ordinary_conjunct_still_folds_onto_the_scan() { } /// Find the first `SQLWindowFunc` node along the single-child spine. -fn find_windowfunc(qe: &QueryExpr) -> Option<&QueryExpr> { +fn find_windowfunc(qe: &PreASAPNode) -> Option<&PreASAPNode> { match qe { - QueryExpr::SQLWindowFunc { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_windowfunc(child), + PreASAPNode::SQLWindowFunc { .. } => Some(qe), + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Aggregate { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } => find_windowfunc(child), _ => None, } } @@ -766,7 +766,7 @@ async fn window_function_lowers_to_positional_windowfunc() { ) .await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { + let PreASAPNode::SQLWindowFunc { func, partition_by, order_by, @@ -780,7 +780,7 @@ async fn window_function_lowers_to_positional_windowfunc() { assert_eq!(order_by.len(), 1); assert_eq!( order_by[0].expr, - QueryExpr::Column(3), + PreASAPNode::Column(3), "ORDER BY bytes → col 3" ); assert!(!order_by[0].ascending, "DESC"); @@ -799,11 +799,15 @@ async fn window_function_lowers_to_positional_windowfunc() { async fn window_aggregate_lowers_to_windowfunc() { let qe = lower("SELECT service, SUM(bytes) OVER (PARTITION BY service) FROM metrics").await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { func, args, .. } = win else { + let PreASAPNode::SQLWindowFunc { func, args, .. } = win else { unreachable!(); }; assert_eq!(*func, WindowFuncKind::Sum); - assert_eq!(args, &vec![QueryExpr::Column(3)], "SUM(bytes) → arg col 3"); + assert_eq!( + args, + &vec![PreASAPNode::Column(3)], + "SUM(bytes) → arg col 3" + ); } // ── Window frames (issue #268) ─────────────────────────────────────────────── @@ -827,8 +831,8 @@ async fn window_frame_is_captured_not_dropped() { ) .await; - let frame_of = |qe: &QueryExpr| { - let QueryExpr::SQLWindowFunc { frame, .. } = find_windowfunc(qe).unwrap() else { + let frame_of = |qe: &PreASAPNode| { + let PreASAPNode::SQLWindowFunc { frame, .. } = find_windowfunc(qe).unwrap() else { unreachable!(); }; frame @@ -868,7 +872,7 @@ async fn range_interval_frame_is_preserved() { RANGE BETWEEN INTERVAL '1' HOUR PRECEDING AND CURRENT ROW) FROM metrics", ) .await; - let QueryExpr::SQLWindowFunc { + let PreASAPNode::SQLWindowFunc { frame: Some(frame), .. } = find_windowfunc(&qe).unwrap() else { @@ -900,8 +904,8 @@ async fn range_numeric_frames_remain_scalar_offsets() { ) .await; - let start_bound = |qe: &QueryExpr| { - let QueryExpr::SQLWindowFunc { + let start_bound = |qe: &PreASAPNode| { + let PreASAPNode::SQLWindowFunc { frame: Some(frame), .. } = find_windowfunc(qe).unwrap() else { @@ -939,30 +943,30 @@ async fn groups_frame_is_rejected() { // ── Nested query functions: derived tables / inline views (issue #27) ─────────── /// Collect every `AggIntent` in the tree, root-to-leaf. -fn all_intents(qe: &QueryExpr) -> Vec { +fn all_intents(qe: &PreASAPNode) -> Vec { let mut out = Vec::new(); - fn go(qe: &QueryExpr, out: &mut Vec) { + fn go(qe: &PreASAPNode, out: &mut Vec) { match qe { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { measures, child, .. } => { out.extend(measures.iter().cloned()); go(child, out); } - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => go(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Dedup { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::SQLWindowFunc { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } => go(child, out), + PreASAPNode::BinaryOp { lhs, rhs, .. } + | PreASAPNode::Join { left: lhs, right: rhs, .. } - | QueryExpr::SetOp { + | PreASAPNode::SetOp { left: lhs, right: rhs, .. @@ -1072,15 +1076,15 @@ async fn correlated_exists_lifts_its_correlation_into_the_join() { .await; let (kind, pred, left_len) = join_parts(&qe); assert_eq!(kind, &JoinKind::Semi); - let QueryExpr::Compare { left, right, .. } = pred else { + let PreASAPNode::Compare { left, right, .. } = pred else { panic!("expected the correlation as a comparison, got {pred:?}"); }; assert_eq!( **left, - QueryExpr::Column(left_len), + PreASAPNode::Column(left_len), "h.service (right side)" ); - assert_eq!(**right, QueryExpr::Column(1), "m.service (left side)"); + assert_eq!(**right, PreASAPNode::Column(1), "m.service (left side)"); } #[tokio::test] @@ -1099,7 +1103,7 @@ async fn an_uncorrelated_exists_is_an_unconditional_semi_join() { let qe = lower("SELECT service FROM metrics WHERE EXISTS (SELECT 1 FROM hosts)").await; let (kind, pred, _) = join_parts(&qe); assert_eq!(kind, &JoinKind::Semi); - assert_eq!(*pred, QueryExpr::Literal(ScalarValue::Boolean(true))); + assert_eq!(*pred, PreASAPNode::Literal(ScalarValue::Boolean(true))); } #[tokio::test] @@ -1281,7 +1285,7 @@ async fn time_bucketing_group_by_lowers_to_a_derived_key() { let qe = lower("SELECT date_trunc('minute', ts) AS m, SUM(bytes) FROM metrics GROUP BY m").await; let node = find_aggregate_node(&qe).expect("expected an Aggregate"); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -1291,7 +1295,7 @@ async fn time_bucketing_group_by_lowers_to_a_derived_key() { unreachable!() }; assert!( - matches!(**child, QueryExpr::Project { .. }), + matches!(**child, PreASAPNode::Project { .. }), "expected a materializing Project beneath the Aggregate" ); let schema = child.output_schema().expect("child schema"); @@ -1317,14 +1321,14 @@ async fn time_bucketing_keeps_the_scan_predicate() { WHERE bytes > 10 GROUP BY m", ) .await; - fn scan_has_predicate(qe: &QueryExpr) -> bool { + fn scan_has_predicate(qe: &PreASAPNode) -> bool { match qe { - QueryExpr::Scan { predicates, .. } => !predicates.is_empty(), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => scan_has_predicate(child), + PreASAPNode::Scan { predicates, .. } => !predicates.is_empty(), + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Aggregate { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } => scan_has_predicate(child), _ => false, } } @@ -1341,13 +1345,13 @@ async fn a_plain_group_by_inserts_no_projection() { "SELECT COUNT(*) FROM metrics", ] { let qe = lower(q).await; - let QueryExpr::Aggregate { child, .. } = + let PreASAPNode::Aggregate { child, .. } = find_aggregate_node(&qe).expect("expected an Aggregate") else { unreachable!() }; assert!( - !matches!(**child, QueryExpr::Project { .. }), + !matches!(**child, PreASAPNode::Project { .. }), "{q} should not gain a projection" ); } @@ -1356,7 +1360,7 @@ async fn a_plain_group_by_inserts_no_projection() { #[tokio::test] async fn a_shared_expression_is_materialized_once() { let qe = lower("SELECT SUM(bytes * 2), MIN(bytes * 2) FROM metrics").await; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = find_aggregate_node(&qe).expect("expected an Aggregate") else { @@ -1373,14 +1377,14 @@ async fn a_shared_expression_is_materialized_once() { // ── Issue #118: multi-level grouping expands into one Aggregate per level ─── /// The branches of the first `Concat` along the single-child spine. -fn merge_branches(qe: &QueryExpr) -> &Vec { - fn find(qe: &QueryExpr) -> Option<&Vec> { +fn merge_branches(qe: &PreASAPNode) -> &Vec { + fn find(qe: &PreASAPNode) -> Option<&Vec> { match qe { - QueryExpr::Concat { children, .. } => Some(children), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => find(child), + PreASAPNode::Concat { children, .. } => Some(children), + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } => find(child), _ => None, } } @@ -1388,14 +1392,14 @@ fn merge_branches(qe: &QueryExpr) -> &Vec { } /// `(group keys, column names)` of each merged grouping level. -fn grouping_levels(qe: &QueryExpr) -> Vec<(GroupKeys, Vec)> { +fn grouping_levels(qe: &PreASAPNode) -> Vec<(GroupKeys, Vec)> { merge_branches(qe) .iter() .map(|b| { - let QueryExpr::Project { child, .. } = b else { + let PreASAPNode::Project { child, .. } = b else { panic!("expected a Project per level, got {b:?}"); }; - let QueryExpr::Aggregate { reduction, .. } = child.as_ref() else { + let PreASAPNode::Aggregate { reduction, .. } = child.as_ref() else { panic!("expected an Aggregate under the Project, got {child:?}"); }; let names = b @@ -1534,10 +1538,10 @@ async fn multi_level_grouping_composes_with_a_derived_reducer_argument() { // #110's materializing Project sits beneath every level's Aggregate. let qe = lower("SELECT service, SUM(bytes * 8) FROM metrics GROUP BY ROLLUP(service)").await; for b in merge_branches(&qe) { - let QueryExpr::Project { child, .. } = b else { + let PreASAPNode::Project { child, .. } = b else { panic!("expected a Project per level"); }; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = child.as_ref() else { @@ -1548,7 +1552,7 @@ async fn multi_level_grouping_composes_with_a_derived_reducer_argument() { [AggIntent::Sum { col: Some(_) }] )); assert!( - matches!(**child, QueryExpr::Project { .. }), + matches!(**child, PreASAPNode::Project { .. }), "the derived-column projection should sit under each level" ); } @@ -1611,7 +1615,7 @@ async fn array_agg_is_deliberately_rejected() { // ── Issue #225: catalog-driven ClickHouse builtins (countIf, generalizing // uniqExact from #221) ─────────────────────────────────────────────────── -async fn lower_clickhouse(sql: &str) -> QueryExpr { +async fn lower_clickhouse(sql: &str) -> PreASAPNode { lower_sql_dialect( sql, &catalog(), @@ -1622,20 +1626,20 @@ async fn lower_clickhouse(sql: &str) -> QueryExpr { .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) } -fn temporal_aggregate(qe: &QueryExpr) -> (&AggIntent, std::time::Duration, &QueryExpr) { +fn temporal_aggregate(qe: &PreASAPNode) -> (&AggIntent, std::time::Duration, &PreASAPNode) { match qe { - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures, child, .. } => { - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let PreASAPNode::TimeRange { range, child } = child.as_ref() else { panic!("temporal Aggregate must directly wrap TimeRange, got {child:?}"); }; (&measures[0], *range, child) } - QueryExpr::Project { child, .. } | QueryExpr::Filter { child, .. } => { + PreASAPNode::Project { child, .. } | PreASAPNode::Filter { child, .. } => { temporal_aggregate(child) } other => panic!("expected temporal Aggregate, got {other:?}"), @@ -1656,15 +1660,15 @@ async fn explicit_temporal_aggregates_share_promql_intents_and_timerange() { let (intent, range, child) = temporal_aggregate(&qe); assert_eq!(intent, &expected); assert_eq!(range, std::time::Duration::from_secs(300)); - assert!(matches!(child, QueryExpr::Project { child, .. } - if matches!(child.as_ref(), QueryExpr::Scan { predicates, .. } if predicates.len() == 1))); + assert!(matches!(child, PreASAPNode::Project { child, .. } + if matches!(child.as_ref(), PreASAPNode::Scan { predicates, .. } if predicates.len() == 1))); - let QueryExpr::Project { cols, .. } = &qe else { + let PreASAPNode::Project { cols, .. } = &qe else { panic!("SELECT list must remain a Project, got {qe:?}"); }; - assert!(matches!(cols[0].expr, QueryExpr::Column(2))); + assert!(matches!(cols[0].expr, PreASAPNode::Column(2))); assert_eq!(cols[1].alias.as_deref(), Some("v")); - assert!(matches!(cols[1].expr, QueryExpr::Column(1))); + assert!(matches!(cols[1].expr, PreASAPNode::Column(1))); } } @@ -1825,10 +1829,10 @@ async fn project_filter_and_outer_aggregate_preserve_temporal_child() { ) r WHERE v >= 0", ) .await; - let QueryExpr::Project { child, .. } = &qe else { + let PreASAPNode::Project { child, .. } = &qe else { panic!("expected outer SELECT Project, got {qe:?}"); }; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction: Reduction::Reduce(_), measures, child, @@ -1838,7 +1842,7 @@ async fn project_filter_and_outer_aggregate_preserve_temporal_child() { panic!("expected outer Aggregate, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let QueryExpr::Filter { child, .. } = child.as_ref() else { + let PreASAPNode::Filter { child, .. } = child.as_ref() else { panic!("derived-table WHERE must remain above the inner query, got {child:?}"); }; let (intent, range, _) = temporal_aggregate(child); @@ -2004,13 +2008,13 @@ async fn lag_in_frame_lowers_to_its_own_kind_not_lag() { ) .await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { func, args, .. } = win else { + let PreASAPNode::SQLWindowFunc { func, args, .. } = win else { unreachable!(); }; assert_eq!(*func, WindowFuncKind::LagInFrame); assert_eq!( args, - &vec![QueryExpr::Column(3)], + &vec![PreASAPNode::Column(3)], "lagInFrame(bytes) → arg col 3" ); } @@ -2023,7 +2027,7 @@ async fn lead_in_frame_lowers_to_its_own_kind_not_lead() { ) .await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { func, .. } = win else { + let PreASAPNode::SQLWindowFunc { func, .. } = win else { unreachable!(); }; assert_eq!(*func, WindowFuncKind::LeadInFrame); @@ -2036,13 +2040,13 @@ async fn lead_in_frame_lowers_to_its_own_kind_not_lead() { async fn now_in_predicate_lowers_to_current_timestamp() { // SELECT * folds WHERE onto Scan.predicates (no explicit Filter node). let qe = lower("SELECT * FROM metrics WHERE ts < NOW()").await; - let QueryExpr::Scan { predicates, .. } = &qe else { + let PreASAPNode::Scan { predicates, .. } = &qe else { panic!("expected Scan at root, got {qe:?}"); }; assert_eq!(predicates.len(), 1); assert!( - matches!(predicates[0].0.as_ref(), QueryExpr::Compare { right, .. } - if matches!(right.as_ref(), QueryExpr::CurrentTimestamp)), + matches!(predicates[0].0.as_ref(), PreASAPNode::Compare { right, .. } + if matches!(right.as_ref(), PreASAPNode::CurrentTimestamp)), "NOW() must lower to CurrentTimestamp, got {:?}", predicates[0].0 ); @@ -2053,13 +2057,13 @@ async fn now_in_predicate_lowers_to_current_timestamp() { #[tokio::test] async fn clickhouse_now_in_predicate_lowers_to_current_timestamp() { let qe = lower_clickhouse("SELECT * FROM metrics WHERE ts < now()").await; - let QueryExpr::Scan { predicates, .. } = &qe else { + let PreASAPNode::Scan { predicates, .. } = &qe else { panic!("expected Scan at root, got {qe:?}"); }; assert_eq!(predicates.len(), 1); assert!( - matches!(predicates[0].0.as_ref(), QueryExpr::Compare { right, .. } - if matches!(right.as_ref(), QueryExpr::CurrentTimestamp)), + matches!(predicates[0].0.as_ref(), PreASAPNode::Compare { right, .. } + if matches!(right.as_ref(), PreASAPNode::CurrentTimestamp)), "now() must lower to CurrentTimestamp, got {:?}", predicates[0].0 ); @@ -2068,10 +2072,10 @@ async fn clickhouse_now_in_predicate_lowers_to_current_timestamp() { #[tokio::test] async fn current_timestamp_lowers_to_typed_current_timestamp_leaf() { let qe = lower("SELECT CURRENT_TIMESTAMP FROM metrics").await; - let QueryExpr::Project { cols, .. } = &qe else { + let PreASAPNode::Project { cols, .. } = &qe else { panic!("expected Project at root, got {qe:?}"); }; - assert!(matches!(&cols[0].expr, QueryExpr::CurrentTimestamp)); + assert!(matches!(&cols[0].expr, PreASAPNode::CurrentTimestamp)); let schema = cols[0].expr.output_schema().expect("timestamp schema"); assert_eq!(schema.columns[0].dtype, DataType::Timestamp); } @@ -2484,7 +2488,7 @@ async fn composite_distinct_counts_tuples() { ) .await .unwrap(); - let QueryExpr::Aggregate { measures, .. } = + let PreASAPNode::Aggregate { measures, .. } = find_aggregate_node(&composite).expect("expected an Aggregate") else { unreachable!() @@ -2501,7 +2505,7 @@ async fn composite_distinct_counts_tuples() { ) .await .unwrap(); - let QueryExpr::Aggregate { measures, .. } = + let PreASAPNode::Aggregate { measures, .. } = find_aggregate_node(&single).expect("expected an Aggregate") else { unreachable!() diff --git a/crates/integration-tests/src/lib.rs b/crates/integration-tests/src/lib.rs index 99fb2422..4e139068 100644 --- a/crates/integration-tests/src/lib.rs +++ b/crates/integration-tests/src/lib.rs @@ -14,7 +14,7 @@ pub mod fixtures { use asap_frontend_promql::lower_promql_workload; use asap_types::pre_asap::schema::{Column, DataType, Schema}; - use asap_types::pre_asap::QueryExpr; + use asap_types::pre_asap::PreASAPNode; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, @@ -26,7 +26,7 @@ pub mod fixtures { pub fn lower_promql( query: &str, accuracy: AccuracyTarget, - ) -> Result { + ) -> Result { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, diff --git a/crates/integration-tests/tests/aggregate.rs b/crates/integration-tests/tests/aggregate.rs index dd0bef4e..3ccb6fd6 100644 --- a/crates/integration-tests/tests/aggregate.rs +++ b/crates/integration-tests/tests/aggregate.rs @@ -1,4 +1,4 @@ -//! `QueryExpr::Aggregate` — cross-series aggregation tests. +//! `PreASAPNode::Aggregate` — cross-series aggregation tests. //! //! topk/bottomk are omitted — dispatch is deferred. //! @@ -13,15 +13,15 @@ use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction, Source}; +use asap_types::pre_asap::{AggIntent, PreASAPNode, Reduction, Source}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> PreASAPNode { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::Scan { +fn scan(metric: &str, labels: &[&str]) -> PreASAPNode { + PreASAPNode::Scan { source: Source::TimeSeries { metric: metric.into(), }, @@ -30,13 +30,13 @@ fn scan(metric: &str, labels: &[&str]) -> QueryExpr { } } -fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { +fn agg(by: Vec, intent: AggIntent, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec!["".into()], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: Rc::new(PreASAPNode::TimeRange { range: Duration::from_secs(1), child: Rc::new(child), }), diff --git a/crates/integration-tests/tests/binary_op.rs b/crates/integration-tests/tests/binary_op.rs index 63ba458f..798dc74d 100644 --- a/crates/integration-tests/tests/binary_op.rs +++ b/crates/integration-tests/tests/binary_op.rs @@ -1,4 +1,4 @@ -//! `QueryExpr::BinaryOp` — arithmetic, comparison, and vector-match tests. +//! `PreASAPNode::BinaryOp` — arithmetic, comparison, and vector-match tests. //! //! Each side of a `BinaryOp` is bound independently by the SchemaResolver, so each //! gets its own scan schema derived from the labels it references. @@ -11,24 +11,24 @@ use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, GroupSide, QueryExpr, Reduction, + AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, GroupSide, PreASAPNode, Reduction, Source, VectorGrouping, VectorMatch, VectorMatchKind, }; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> PreASAPNode { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::TimeRange { +fn scan(metric: &str, labels: &[&str]) -> PreASAPNode { + PreASAPNode::TimeRange { range: Duration::from_secs(1), child: Rc::new(source_scan(metric, labels)), } } -fn source_scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::Scan { +fn source_scan(metric: &str, labels: &[&str]) -> PreASAPNode { + PreASAPNode::Scan { source: Source::TimeSeries { metric: metric.into(), }, @@ -37,21 +37,21 @@ fn source_scan(metric: &str, labels: &[&str]) -> QueryExpr { } } -fn rate_agg(metric: &str) -> QueryExpr { - QueryExpr::Aggregate { +fn rate_agg(metric: &str) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Rate], output_names: vec!["".into()], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: Rc::new(PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(source_scan(metric, &[])), }), } } -fn sum_by_job(metric: &str) -> QueryExpr { - QueryExpr::Aggregate { +fn sum_by_job(metric: &str) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![AggIntent::Sum { col: None }], output_names: vec!["".into()], @@ -63,7 +63,7 @@ fn sum_by_job(metric: &str) -> QueryExpr { // #18 — arithmetic binary op between two bare scans; no vector match #[test] fn q18_div_bare_scans() { - let expected = QueryExpr::BinaryOp { + let expected = PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), lhs: Rc::new(scan("http_requests_total", &[])), rhs: Rc::new(scan("http_requests_total", &[])), @@ -75,7 +75,7 @@ fn q18_div_bare_scans() { // #19 — add with on(job) vector match; match labels are strings, not column ids #[test] fn q19_add_with_on_match() { - let expected = QueryExpr::BinaryOp { + let expected = PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), lhs: Rc::new(scan("http_requests_total", &[])), rhs: Rc::new(scan("http_requests_total", &[])), @@ -94,7 +94,7 @@ fn q19_add_with_on_match() { // #20 — divide two rate aggregates over different metrics #[test] fn q20_div_two_rates() { - let expected = QueryExpr::BinaryOp { + let expected = PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), lhs: Rc::new(rate_agg("http_requests_total")), rhs: Rc::new(rate_agg("http_errors_total")), @@ -111,7 +111,7 @@ fn q20_div_two_rates() { fn q_gt_comparison() { assert_eq!( lower("http_requests_total > http_errors_total"), - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op: BinaryOpKind::Compare(CompareOpKind::Gt), lhs: Rc::new(scan("http_requests_total", &[])), rhs: Rc::new(scan("http_errors_total", &[])), @@ -124,7 +124,7 @@ fn q_gt_comparison() { fn q_lt_comparison() { assert_eq!( lower("http_requests_total < http_errors_total"), - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op: BinaryOpKind::Compare(CompareOpKind::Lt), lhs: Rc::new(scan("http_requests_total", &[])), rhs: Rc::new(scan("http_errors_total", &[])), @@ -137,7 +137,7 @@ fn q_lt_comparison() { fn q_ge_comparison() { assert_eq!( lower("http_requests_total >= http_errors_total"), - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op: BinaryOpKind::Compare(CompareOpKind::Ge), lhs: Rc::new(scan("http_requests_total", &[])), rhs: Rc::new(scan("http_errors_total", &[])), @@ -150,7 +150,7 @@ fn q_ge_comparison() { fn q_le_comparison() { assert_eq!( lower("http_requests_total <= http_errors_total"), - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op: BinaryOpKind::Compare(CompareOpKind::Le), lhs: Rc::new(scan("http_requests_total", &[])), rhs: Rc::new(scan("http_errors_total", &[])), @@ -164,7 +164,7 @@ fn q_le_comparison() { fn q_add_with_ignoring() { assert_eq!( lower("http_requests_total + ignoring(job) http_errors_total"), - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), lhs: Rc::new(scan("http_requests_total", &[])), rhs: Rc::new(scan("http_errors_total", &[])), @@ -182,7 +182,7 @@ fn q_add_with_ignoring() { fn q_mul_group_left() { assert_eq!( lower("http_requests_total * on(job) group_left() node_info"), - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), lhs: Rc::new(scan("http_requests_total", &[])), rhs: Rc::new(scan("node_info", &[])), @@ -203,7 +203,7 @@ fn q_mul_group_left() { fn q_mul_group_right() { assert_eq!( lower("node_info * on(job) group_right() http_requests_total"), - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), lhs: Rc::new(scan("node_info", &[])), rhs: Rc::new(scan("http_requests_total", &[])), @@ -223,7 +223,7 @@ fn q_mul_group_right() { // each side: Aggregate{Sum, by=[2]} over Scan([ts, value, job]) #[test] fn q21_div_two_sum_by_job() { - let expected = QueryExpr::BinaryOp { + let expected = PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), lhs: Rc::new(sum_by_job("http_requests_total")), rhs: Rc::new(sum_by_job("http_errors_total")), @@ -239,10 +239,10 @@ fn q21_div_two_sum_by_job() { // against PromqlScalarBridge(-1), no vector match. The vector side keeps its schema. #[test] fn q36_unary_negation_is_multiply_by_minus_one() { - let expected = QueryExpr::BinaryOp { + let expected = PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), lhs: Rc::new(scan("some_metric", &[])), - rhs: Rc::new(QueryExpr::promql_scalar(-1.0)), + rhs: Rc::new(PreASAPNode::promql_scalar(-1.0)), vector_match: None, }; assert_eq!(lower("-some_metric"), expected); @@ -252,15 +252,15 @@ fn q36_unary_negation_is_multiply_by_minus_one() { // `sum(-m)` → Aggregate{Sum} over the `m * -1` BinaryOp. #[test] fn q36_sum_of_negation_nests() { - let expected = QueryExpr::Aggregate { + let expected = PreASAPNode::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Sum { col: None }], output_names: vec!["".into()], having: None, - child: Rc::new(QueryExpr::BinaryOp { + child: Rc::new(PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), lhs: Rc::new(scan("node_cpu_seconds_total", &[])), - rhs: Rc::new(QueryExpr::promql_scalar(-1.0)), + rhs: Rc::new(PreASAPNode::promql_scalar(-1.0)), vector_match: None, }), }; diff --git a/crates/integration-tests/tests/cse.rs b/crates/integration-tests/tests/cse.rs index c5c10578..38351da2 100644 --- a/crates/integration-tests/tests/cse.rs +++ b/crates/integration-tests/tests/cse.rs @@ -2,12 +2,12 @@ //! #223). //! //! Drives the full staged pipeline this issue lands: two independently -//! lowered `QueryExpr` trees → `share_common_subtrees` (stage 1, +//! lowered `PreASAPNode` trees → `share_common_subtrees` (stage 1, //! `asap-types::pre_asap::cse`, run internally by `search_workload`) → //! `search_workload` (stage 2, `asap-aware-mapping`) — and asserts the //! sharing that stage 1 decides survives into stage 2's discovered -//! `CandidatePostASAPDAGs` as one genuinely shared `TargetSubDAGCandidates`, not just one shared -//! `Rc`. This is the "real caller" the issue's landing plan +//! `CandidateLogicalPostASAPDAGs` as one genuinely shared `TargetSubDAGCandidates`, not just one shared +//! `Rc`. This is the "real caller" the issue's landing plan //! requires before `share_common_subtrees` is allowed to exist at all (its //! predecessor, `asap-plan::cse::dedupe_subtrees`, was deleted in #192 for //! being unwired dead code). @@ -16,7 +16,7 @@ //! workload (the former `implement_workload`/`implement_workload_with`, //! which this test file used to drive instead of `search_workload`) is out //! of `asap-aware-mapping`'s scope — see that crate's `lib.rs` `## Status` -//! section — so these tests assert on the discovered `CandidatePostASAPDAGs` shape +//! section — so these tests assert on the discovered `CandidateLogicalPostASAPDAGs` shape //! directly, the same way `asap-aware-mapping::replacement`'s own //! `shared_aggregate_across_two_roots_gets_both_strategies_candidates` test //! does, just exercised through the crate's public API from this external @@ -26,12 +26,12 @@ use std::rc::Rc; use asap_aware_mapping::{search_workload, Replacement}; use asap_integration_tests::fixtures::lower_promql; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::pre_asap::query_expr::PreASAPNode; use asap_types::types::AccuracyTarget; /// Two workload entries that happen to submit the exact same query (a /// realistic case — two dashboards, or a query fired both standalone and as -/// part of a larger batch) collapse onto one shared `Rc` after +/// part of a larger batch) collapse onto one shared `Rc` after /// `search_workload`'s internal `share_common_subtrees` pass, and onto one /// genuinely-shared [`TargetSubDAGCandidates`](asap_aware_mapping::TargetSubDAGCandidates) — carrying /// every candidate discovered for it exactly once, not once per root — no @@ -62,8 +62,8 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { // roots[0] and roots[1] must have merged onto the same Rc — the // `share_common_subtrees` pass `search_workload` runs internally. assert!( - Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1), - "search_workload must collapse the two identical roots onto one Rc" + Rc::ptr_eq(&space.roots()[0].1, &space.roots()[1].1), + "search_workload must collapse the two identical roots onto one Rc" ); // The single shared root is one discovered TargetSubDAG, holding one @@ -73,7 +73,7 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { // (asap-aware-mapping::replacement's own equivalent, internal test) // pins for the same fixture shape. let group = space - .candidates_for_target(&space.roots[0].1) + .candidates_for_target(&space.roots()[0].1) .expect("shared root must be a discovered target"); assert_eq!(group.consumer_count, 2); assert_eq!( @@ -122,13 +122,13 @@ fn distinct_workload_queries_get_independent_memo_groups() { assert_ne!(a, b, "fixture sanity: the two queries differ"); let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); - assert!(!Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); + assert!(!Rc::ptr_eq(&space.roots()[0].1, &space.roots()[1].1)); let group_a = space - .candidates_for_target(&space.roots[0].1) + .candidates_for_target(&space.roots()[0].1) .expect("root a must be a discovered target"); let group_b = space - .candidates_for_target(&space.roots[1].1) + .candidates_for_target(&space.roots()[1].1) .expect("root b must be a discovered target"); assert!( !Rc::ptr_eq(&group_a.target, &group_b.target), @@ -140,7 +140,7 @@ fn distinct_workload_queries_get_independent_memo_groups() { /// Single-query CSE (a repeated sub-expression within one query) also /// survives through `search_workload`: the two grouped-`Aggregate` branches -/// of a `BinaryOp` collapse to one shared `Rc` in the internal +/// of a `BinaryOp` collapse to one shared `Rc` in the internal /// `share_common_subtrees` pass, and to one shared `TargetSubDAGCandidates` (with /// `consumer_count == 2`, one per branch) here. #[test] @@ -149,15 +149,15 @@ fn single_query_repeated_subexpression_shares_one_memo_group() { let expr = lower_promql(query, AccuracyTarget::Exact).expect("query failed to lower"); let space = search_workload(vec![("q", Rc::new(expr))]); - let [(_, root)] = space.roots.as_slice() else { + let [(_, root)] = space.roots().as_slice() else { panic!("expected 1 root"); }; - let QueryExpr::BinaryOp { lhs, rhs, .. } = root.as_ref() else { + let PreASAPNode::BinaryOp { lhs, rhs, .. } = root.as_ref() else { panic!("expected a BinaryOp root, got {root:?}"); }; assert!( Rc::ptr_eq(lhs, rhs), - "the two identical sum-by-job branches must collapse onto one Rc" + "the two identical sum-by-job branches must collapse onto one Rc" ); let group = space diff --git a/crates/integration-tests/tests/exact_composition.rs b/crates/integration-tests/tests/exact_composition.rs index 3f87904d..8d98e17b 100644 --- a/crates/integration-tests/tests/exact_composition.rs +++ b/crates/integration-tests/tests/exact_composition.rs @@ -1,6 +1,6 @@ //! Issue #171 — composing exact operators with summary plans across //! explicit update/readout boundaries, end to end through -//! `search_workload_with` → `CandidatePostASAPDAGs::global_selection` → +//! `search_workload_with` → `CandidateLogicalPostASAPDAGs::global_selection` → //! `GlobalSelection::assemble_selected_dag` → `dag_export`. //! //! Covers the issue's integration matrix: both nesting directions, grouped @@ -27,22 +27,22 @@ use asap_integration_tests::fixtures::lower_promql; use asap_types::dag_export; use asap_types::post_asap::{ validate_execution_data_states, ExactKind, ExactOperation, ExecutionDataState, ExecutionTiming, - SketchAlgorithm, SummaryExpr, SummaryFamilyType, SummaryNode, SummaryUpdate, + PostASAPNode, SketchAlgorithm, SummaryExpr, SummaryFamilyType, SummaryUpdate, }; use asap_types::pre_asap::agg_intent::{default_quantile, AggIntent}; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction, Source}; +use asap_types::pre_asap::query_expr::{PreASAPNode, Reduction, Source}; use asap_types::pre_asap::schema::{Column, DataType, Schema}; use asap_types::types::AccuracyTarget; // ── fixtures ──────────────────────────────────────────────────────────── -fn metric_scan(labels: &[&str]) -> QueryExpr { +fn metric_scan(labels: &[&str]) -> PreASAPNode { let mut columns = vec![ Column::new("ts", DataType::Timestamp, false), Column::new("value", DataType::Float64, false), ]; columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); - QueryExpr::Scan { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "latency".into(), }, @@ -51,8 +51,8 @@ fn metric_scan(labels: &[&str]) -> QueryExpr { } } -fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { +fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + Rc::new(PreASAPNode::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], @@ -61,8 +61,8 @@ fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc }) } -fn per_entity(intent: AggIntent, child: Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { +fn per_entity(intent: AggIntent, child: Rc) -> Rc { + Rc::new(PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec![], @@ -72,7 +72,7 @@ fn per_entity(intent: AggIntent, child: Rc) -> Rc { } /// `quantile by (zone, host) (latency)` — the fine-grained inner summary. -fn fine_quantile() -> Rc { +fn fine_quantile() -> Rc { agg( vec![2, 3], default_quantile(0.99), @@ -126,12 +126,12 @@ fn custom_accuracy_rule_survives_root_target_and_materialization() { ); let selection = space.global_selection(&StatsModel); assert!(selection - .for_target(&space.roots[0].1) + .for_target(&space.roots()[0].1) .unwrap() .composition .is_some()); let node = selection - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .unwrap() .unwrap(); let guarantee = node.guarantee.as_ref().unwrap(); @@ -150,7 +150,7 @@ fn root_target_rejects_unproven_composition() { ); let selection = space.global_selection(&StatsModel); assert!(selection - .for_target(&space.roots[0].1) + .for_target(&space.roots()[0].1) .unwrap() .composition .is_none()); @@ -246,7 +246,7 @@ impl CostModel for UnknownCapabilityModel { fn unknown_runtime_capability_keeps_candidate_but_prevents_selection() { let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); let space = plan(vec![("q", root)], &UnknownCapabilityModel); - let group = space.candidates_for_target(&space.roots[0].1).unwrap(); + let group = space.candidates_for_target(&space.roots()[0].1).unwrap(); assert!(group.candidates.iter().any(|candidate| { matches!(candidate.replacement, Replacement::ExactComposition(_)) && candidate @@ -255,37 +255,37 @@ fn unknown_runtime_capability_keeps_candidate_but_prevents_selection() { && UnknownCapabilityModel .candidate_cost( candidate, - &asap_aware_mapping::TargetSubDAG::new(&space.roots[0].1), + &asap_aware_mapping::TargetSubDAG::new(&space.roots()[0].1), ) .is_none() })); let selection = space.global_selection(&UnknownCapabilityModel); assert!(selection - .for_target(&space.roots[0].1) + .for_target(&space.roots()[0].1) .unwrap() .composition .is_none()); assert!(selection - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .unwrap() .is_some()); } fn plan( - roots: Vec<(&'static str, Rc)>, + roots: Vec<(&'static str, Rc)>, cost_model: &dyn CostModel, -) -> asap_aware_mapping::CandidatePostASAPDAGs<&'static str> { +) -> asap_aware_mapping::CandidateLogicalPostASAPDAGs<&'static str> { search_workload_with(roots, &default_strategies_with(cost_model)) } -fn is_plain(node: &SummaryNode) -> bool { +fn is_plain(node: &PostASAPNode) -> bool { node.schema .fields .iter() .all(|f| matches!(f.dtype, SummaryFamilyType::Plain(_))) } -fn names(node: &SummaryNode) -> Vec<&str> { +fn names(node: &PostASAPNode) -> Vec<&str> { node.schema.fields.iter().map(|f| f.name.as_str()).collect() } @@ -294,7 +294,7 @@ fn names(node: &SummaryNode) -> Vec<&str> { #[test] fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { use std::time::Duration; - let cases: Vec<(Rc, ExactKind)> = vec![ + let cases: Vec<(Rc, ExactKind)> = vec![ ( agg( vec![2], @@ -332,7 +332,7 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { ( per_entity( AggIntent::Rate, - Rc::new(QueryExpr::TimeRange { + Rc::new(PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(metric_scan(&["zone"])), }), @@ -342,7 +342,7 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { ( per_entity( AggIntent::Increase, - Rc::new(QueryExpr::TimeRange { + Rc::new(PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(metric_scan(&["zone"])), }), @@ -395,8 +395,8 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { for intent in [AggIntent::Max { col: None }, AggIntent::Avg { col: None }] { let root = agg(vec![0], intent.clone(), fine_quantile()); let space = plan(vec![("q", Rc::clone(&root))], &StatsModel); - let root = Rc::clone(&space.roots[0].1); - let QueryExpr::Aggregate { child: inner, .. } = root.as_ref() else { + let root = Rc::clone(&space.roots()[0].1); + let PreASAPNode::Aggregate { child: inner, .. } = root.as_ref() else { unreachable!() }; @@ -491,7 +491,7 @@ fn avg_over_quantile_keeps_the_sum_over_count_rewrite_as_a_competitor() { ); let root = agg(vec![0], AggIntent::Avg { col: None }, inner); let space = plan(vec![("q", root)], &StatsModel); - let group = space.candidates_for_target(&space.roots[0].1).unwrap(); + let group = space.candidates_for_target(&space.roots()[0].1).unwrap(); let provenances: Vec<_> = group.candidates.iter().map(|c| c.provenance).collect(); assert!(provenances.contains(&ReplacementProvenance::LogicalRewrite)); assert!(provenances.contains(&ReplacementProvenance::ValueOperationAtQueryTime)); @@ -513,7 +513,7 @@ fn identity_and_genuine_multi_row_folds_both_compose() { ] { let root = agg(vec![0], AggIntent::Max { col: None }, inner); let space = plan(vec![("q", root)], &StatsModel); - let root = &space.roots[0].1; + let root = &space.roots()[0].1; let composed = space .global_selection(&StatsModel) .assemble_selected_dag(root) @@ -537,7 +537,7 @@ fn identity_and_genuine_multi_row_folds_both_compose() { /// One inner quantile consumed by two outer folds in two queries: CSE /// collapses the inner target onto one `Rc`, both compositions commit to /// the *same* child candidate, and both materializations share one -/// `Rc` for it — the summary is maintained once. +/// `Rc` for it — the summary is maintained once. #[test] fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { let max = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); @@ -545,9 +545,9 @@ fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { let space = plan(vec![("max", max), ("min", min)], &StatsModel); let selection = space.global_selection(&StatsModel); - let roots: Vec> = space.roots.iter().map(|(_, r)| Rc::clone(r)).collect(); - let inner_of = |r: &Rc| match r.as_ref() { - QueryExpr::Aggregate { child, .. } => Rc::clone(child), + let roots: Vec> = space.roots().iter().map(|(_, r)| Rc::clone(r)).collect(); + let inner_of = |r: &Rc| match r.as_ref() { + PreASAPNode::Aggregate { child, .. } => Rc::clone(child), _ => unreachable!(), }; assert!( @@ -583,7 +583,7 @@ fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { .iter() .map(|r| selection.assemble_selected_dag(r).unwrap().unwrap()) .collect(); - let child_of = |n: &Rc| match &n.expr { + let child_of = |n: &Rc| match &n.expr { SummaryExpr::ValueOperation { child, timing: ExecutionTiming::QueryTime, @@ -593,7 +593,7 @@ fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { }; assert!( Rc::ptr_eq(&child_of(&composed[0]), &child_of(&composed[1])), - "both folds compose over the same Rc" + "both folds compose over the same Rc" ); } @@ -608,15 +608,15 @@ fn outer_summary_over_an_exact_function_composes_at_ingestion_time() { use std::time::Duration; let deriv = per_entity( AggIntent::Deriv, - Rc::new(QueryExpr::TimeRange { + Rc::new(PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(metric_scan(&["zone"])), }), ); let root = agg(vec![], default_quantile(0.99), deriv); let space = plan(vec![("q", root)], &StatsModel); - let root = Rc::clone(&space.roots[0].1); - let QueryExpr::Aggregate { child: deriv, .. } = root.as_ref() else { + let root = Rc::clone(&space.roots()[0].1); + let PreASAPNode::Aggregate { child: deriv, .. } = root.as_ref() else { unreachable!() }; assert!(space @@ -676,10 +676,10 @@ fn summary_construction_follows_its_value_input_phase() { let space = plan(vec![("q", Rc::clone(&root))], &StatsModel); let post = space .global_selection(&StatsModel) - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .unwrap() .unwrap(); - let illegal = Rc::new(SummaryNode { + let illegal = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child: post, family: SummaryFamilyType::ExactAggregate( @@ -705,7 +705,7 @@ fn summary_construction_follows_its_value_input_phase() { fn a_runtime_without_mixed_execution_gets_no_composition_candidates() { let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); let space = plan(vec![("q", root)], &NoCapabilityModel); - let root = Rc::clone(&space.roots[0].1); + let root = Rc::clone(&space.roots()[0].1); let group = space.candidates_for_target(&root).unwrap(); assert!(group .candidates @@ -716,21 +716,21 @@ fn a_runtime_without_mixed_execution_gets_no_composition_candidates() { let node = selection.assemble_selected_dag(&root).unwrap().unwrap(); assert!(!matches!(node.expr, SummaryExpr::ValueOperation { .. })); // The inner quantile is still independently selectable. - let QueryExpr::Aggregate { child, .. } = root.as_ref() else { + let PreASAPNode::Aggregate { child, .. } = root.as_ref() else { unreachable!() }; assert!(selection.for_target(child).unwrap().chosen.is_some()); } /// Without statistics (the built-in model) the composition is *proposed* -/// — visible in `CandidatePostASAPDAGs` and explanations — but never *selected*: the +/// — visible in `CandidateLogicalPostASAPDAGs` and explanations — but never *selected*: the /// site keeps a non-composed alternative, and the inner summary stays /// independently selectable. #[test] fn missing_cost_statistics_preserve_the_conservative_keep_pre_asap() { let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); let space = plan(vec![("q", root)], &DefaultCostModel); - let root = Rc::clone(&space.roots[0].1); + let root = Rc::clone(&space.roots()[0].1); assert!(space .candidates_for_target(&root) .unwrap() @@ -759,7 +759,7 @@ fn missing_cost_statistics_preserve_the_conservative_keep_pre_asap() { fn dag_export_carries_explicit_stage_and_plain_schema_for_a_composed_plan() { let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); let space = plan(vec![("q", root)], &StatsModel); - let root = &space.roots[0].1; + let root = &space.roots()[0].1; let composed = space .global_selection(&StatsModel) .assemble_selected_dag(root) @@ -795,7 +795,7 @@ fn promql_max_by_zone_over_quantile_over_time_composes() { ) .unwrap(); let space = plan(vec![("q", Rc::new(expr))], &StatsModel); - let root = &space.roots[0].1; + let root = &space.roots()[0].1; let selection = space.global_selection(&StatsModel); let selected = selection.for_target(root).unwrap(); assert_eq!( diff --git a/crates/integration-tests/tests/frontend_timestamps.rs b/crates/integration-tests/tests/frontend_timestamps.rs index 67c5418c..c0f8d538 100644 --- a/crates/integration-tests/tests/frontend_timestamps.rs +++ b/crates/integration-tests/tests/frontend_timestamps.rs @@ -3,7 +3,7 @@ use asap_frontend_sql::{lower_sql, SqlCatalog}; use asap_integration_tests::fixtures::lower_promql; use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::QueryExpr; +use asap_types::pre_asap::PreASAPNode; use asap_types::types::AccuracyTarget; /// PromQL exposes its evaluation time as Unix seconds, whereas SQL exposes @@ -12,7 +12,7 @@ use asap_types::types::AccuracyTarget; #[tokio::test] async fn promql_eval_time_and_sql_current_timestamp_remain_distinct() { let promql = lower_promql("time()", AccuracyTarget::Exact).expect("lower PromQL time()"); - assert!(matches!(promql, QueryExpr::EvalTimestamp)); + assert!(matches!(promql, PreASAPNode::EvalTimestamp)); let promql_schema = promql.output_schema().expect("PromQL time() schema"); assert_eq!(promql_schema.columns[0].dtype, DataType::Float64); @@ -27,10 +27,10 @@ async fn promql_eval_time_and_sql_current_timestamp_remain_distinct() { ) .await .expect("lower SQL CURRENT_TIMESTAMP"); - let QueryExpr::Project { cols, .. } = sql else { + let PreASAPNode::Project { cols, .. } = sql else { panic!("expected SQL projection, got {sql:?}"); }; - assert!(matches!(&cols[0].expr, QueryExpr::CurrentTimestamp)); + assert!(matches!(&cols[0].expr, PreASAPNode::CurrentTimestamp)); let sql_schema = cols[0] .expr .output_schema() diff --git a/crates/integration-tests/tests/kll_pane_execution.rs b/crates/integration-tests/tests/kll_pane_execution.rs index e818796f..3dfb4b41 100644 --- a/crates/integration-tests/tests/kll_pane_execution.rs +++ b/crates/integration-tests/tests/kll_pane_execution.rs @@ -2,8 +2,8 @@ mod physical_common; use asap_physical_operators::{ operators::{Operator, ReadoutQuery}, - physical_planner::{CompiledPhysicalDag, InputContract, Source}, - plan::{PhysicalDag, PhysicalOperator, PlanProperties}, + physical_planner::{InputContract, PhysicalPostASAPDAG, Source}, + plan::{PhysicalExecution, PhysicalOperator, PlanProperties}, runtime::{Input, Limits, OutputStream, RunContext, Scope}, summary_kernels::datasketches_kll::DatasketchesKLLAccumulator, values::{Batch, Schema, Value}, @@ -47,11 +47,11 @@ fn query_scope() -> Scope { revision: 1, } } -fn fixture() -> (CompiledPhysicalDag, Operator, Schema) { +fn fixture() -> (PhysicalPostASAPDAG, Operator, Schema) { let raw = raw_schema(); let build = Operator::summary_build(raw.clone(), family(200), 0, None, vec![]).unwrap(); let state = build.schema(); - let maintenance = CompiledPhysicalDag::from_operators( + let maintenance = PhysicalPostASAPDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(raw))]), BTreeMap::from([(1, (vec![0], build))]), vec![1], @@ -60,7 +60,7 @@ fn fixture() -> (CompiledPhysicalDag, Operator, Schema) { let merge = Operator::summary_merge(state.clone(), 0, vec![]).unwrap(); (maintenance, merge, state) } -fn pane_state(maintenance: &CompiledPhysicalDag, pane: i64) -> Arc { +fn pane_state(maintenance: &PhysicalPostASAPDAG, pane: i64) -> Arc { // Twenty samples in each (start,end] one-minute pane; k=200 avoids // compaction so quantiles and sample counts have deterministic oracles. let raw = raw_schema(); @@ -139,7 +139,7 @@ fn five_panes_roundtrip_and_shared_merge_runs_once() { let (maintenance, merge, schema) = fixture(); let panes: Vec<_> = (0..6).map(|pane| pane_state(&maintenance, pane)).collect(); drop(maintenance); - let compiled = CompiledPhysicalDag::from_operators( + let compiled = PhysicalPostASAPDAG::from_operators( (0..5) .map(|id| (id, InputContract::bounded(schema.clone()))) .collect(), @@ -224,7 +224,7 @@ fn five_panes_roundtrip_and_shared_merge_runs_once() { assert!((value(1) - (50 + offset * 20) as f64).abs() <= 1.); assert!((value(2) - (99 + offset * 20) as f64).abs() <= 1.); let starts = Arc::new(AtomicUsize::new(0)); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalExecution::default(); dag.add( 0, vec![], @@ -267,7 +267,7 @@ fn panes_reject_parameters_schema_and_missing_binding() { state: Arc::new(DatasketchesKLLAccumulator::new(128)), }; assert!(Batch::try_new(schema.clone(), vec![vec![wrong]]).is_err()); - let compiled = CompiledPhysicalDag::from_operators( + let compiled = PhysicalPostASAPDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(schema))]), BTreeMap::from([(1, (vec![0], merge))]), vec![1], diff --git a/crates/integration-tests/tests/nested.rs b/crates/integration-tests/tests/nested.rs index 2d937082..3e5b0a62 100644 --- a/crates/integration-tests/tests/nested.rs +++ b/crates/integration-tests/tests/nested.rs @@ -14,18 +14,18 @@ use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, GroupKeys, Predicate, - PromQLVectorSetOpKind, QueryExpr, Reduction, ScalarValue, Source, TimeShift, VectorMatch, + AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, GroupKeys, PreASAPNode, + Predicate, PromQLVectorSetOpKind, Reduction, ScalarValue, Source, TimeShift, VectorMatch, VectorMatchKind, }; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> PreASAPNode { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { +fn agg(by: Vec, intent: AggIntent, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec!["".into()], @@ -34,8 +34,8 @@ fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { } } -fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { +fn agg_per_entity(intent: AggIntent, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec!["".into()], @@ -48,7 +48,7 @@ fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { // label-preserving output schema [ts, value, job] #[test] fn q22_sum_by_job_over_rate() { - let scan = QueryExpr::Scan { + let scan = PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, @@ -57,7 +57,7 @@ fn q22_sum_by_job_over_rate() { }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { + PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(scan), }, @@ -74,21 +74,21 @@ fn q22_sum_by_job_over_rate() { // predicate on status (col 3); group key job (col 2) #[test] fn q23_sum_by_job_over_filtered_scan() { - let scan = QueryExpr::Scan { + let scan = PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), + predicates: vec![Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(3)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("200".into()))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Utf8("200".into()))), }))], schema: metric_schema(&["job", "status"]), }; let expected = agg( vec![2], AggIntent::Sum { col: None }, - QueryExpr::TimeRange { + PreASAPNode::TimeRange { range: Duration::from_secs(1), child: Rc::new(scan), }, @@ -106,14 +106,14 @@ fn q23_sum_by_job_over_filtered_scan() { // schema [ts, value, job]; outer by=[2] (job) #[test] fn q25_div_over_complex_subtrees() { - let lhs_scan = QueryExpr::Scan { + let lhs_scan = PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), + predicates: vec![Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(3)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("200".into()))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Utf8("200".into()))), }))], schema: metric_schema(&["job", "status"]), }; @@ -122,14 +122,14 @@ fn q25_div_over_complex_subtrees() { AggIntent::Sum { col: None }, agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { + PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(lhs_scan), }, ), ); - let rhs_scan = QueryExpr::Scan { + let rhs_scan = PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_errors_total".into(), }, @@ -141,14 +141,14 @@ fn q25_div_over_complex_subtrees() { AggIntent::Sum { col: None }, agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { + PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(rhs_scan), }, ), ); - let expected = QueryExpr::BinaryOp { + let expected = PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), lhs: Rc::new(lhs), rhs: Rc::new(rhs), @@ -169,7 +169,7 @@ fn q25_div_over_complex_subtrees() { // label-preserving output schema; the outer `max` has no grouping. #[test] fn q27_max_over_sum_by_job_over_rate() { - let scan = QueryExpr::Scan { + let scan = PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, @@ -178,7 +178,7 @@ fn q27_max_over_sum_by_job_over_rate() { }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { + PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(scan), }, @@ -199,21 +199,21 @@ fn q27_max_over_sum_by_job_over_rate() { // Scan schema: [ts(0), value(1), group(2), job(3)] (labels alphabetical). #[test] fn q53_outer_group_key_absent_from_nested_aggregate() { - let scan = QueryExpr::Scan { + let scan = PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_requests".into(), }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), + predicates: vec![Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(3)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("api-server".into()))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Utf8("api-server".into()))), }))], schema: metric_schema(&["group", "job"]), }; let inner = agg( vec![2], AggIntent::Sum { col: None }, - QueryExpr::TimeRange { + PreASAPNode::TimeRange { range: Duration::from_secs(1), child: Rc::new(scan), }, @@ -234,16 +234,16 @@ fn q53_outer_group_key_absent_from_nested_aggregate() { // parser's default `ignoring([])` match modifier. #[test] fn q52_outer_name_label_over_binary_op() { - let side = |metric: &str, env: &str| QueryExpr::TimeRange { + let side = |metric: &str, env: &str| PreASAPNode::TimeRange { range: Duration::from_secs(1), - child: Rc::new(QueryExpr::Scan { + child: Rc::new(PreASAPNode::Scan { source: Source::TimeSeries { metric: metric.into(), }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(2)), // env + predicates: vec![Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(2)), // env op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(env.into()))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Utf8(env.into()))), }))], schema: metric_schema(&["env", "__name__"]), }), @@ -251,7 +251,7 @@ fn q52_outer_name_label_over_binary_op() { let expected = agg( vec![3], // __name__ AggIntent::Sum { col: None }, - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op: BinaryOpKind::Set(PromQLVectorSetOpKind::Or), lhs: Rc::new(side("metric_a", "1")), rhs: Rc::new(side("metric_b", "2")), @@ -274,7 +274,7 @@ fn q52_outer_name_label_over_binary_op() { // the inner rate is label-preserving. Scan schema [ts(0), value(1), instance(2)]. #[test] fn q39_sum_without_instance_over_rate() { - let scan = QueryExpr::Scan { + let scan = PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, @@ -283,12 +283,12 @@ fn q39_sum_without_instance_over_rate() { }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { + PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(scan), }, ); - let expected = QueryExpr::Aggregate { + let expected = PreASAPNode::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(vec![2])), // exclude `instance` measures: vec![AggIntent::Sum { col: None }], output_names: vec!["".into()], @@ -307,13 +307,13 @@ fn q39_sum_without_instance_over_rate() { #[test] fn q40_week_over_week_offset() { let rate_over = |shift: Option| { - let scan = QueryExpr::Scan { + let scan = PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: metric_schema(&[]), }; let ranged = match shift { - Some(ms) => QueryExpr::TimeShift { + Some(ms) => PreASAPNode::TimeShift { shift: TimeShift { offset_ms: ms, at: None, @@ -324,13 +324,13 @@ fn q40_week_over_week_offset() { }; agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { + PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(ranged), }, ) }; - let expected = QueryExpr::BinaryOp { + let expected = PreASAPNode::BinaryOp { op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), lhs: Rc::new(rate_over(None)), rhs: Rc::new(rate_over(Some(604_800_000))), // 1w @@ -343,14 +343,14 @@ fn q40_week_over_week_offset() { // (seconds → ms); a bare selector wrapped in a `TimeShift` carrying the anchor. #[test] fn q40_at_modifier_absolute() { - let expected = QueryExpr::TimeRange { + let expected = PreASAPNode::TimeRange { range: Duration::from_secs(1), - child: Rc::new(QueryExpr::TimeShift { + child: Rc::new(PreASAPNode::TimeShift { shift: TimeShift { offset_ms: 0, at: Some(AtModifier::Timestamp(1_609_746_000_000)), }, - child: Rc::new(QueryExpr::Scan { + child: Rc::new(PreASAPNode::Scan { source: Source::TimeSeries { metric: "up".into(), }, @@ -367,20 +367,20 @@ fn q40_at_modifier_absolute() { // so outer sum by job still finds job at col 2 #[test] fn q24_sum_by_job_over_rate_over_filtered_scan() { - let scan = QueryExpr::Scan { + let scan = PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), + predicates: vec![Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(3)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("200".into()))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Utf8("200".into()))), }))], schema: metric_schema(&["job", "status"]), }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { + PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(scan), }, @@ -400,7 +400,7 @@ fn q24_sum_by_job_over_rate_over_filtered_scan() { // the whole spine survives verbatim and the schema stays label-preserving. #[test] fn q27_nested_subquery_prometheus_docs_example() { - let scan = QueryExpr::Scan { + let scan = PreASAPNode::Scan { source: Source::TimeSeries { metric: "distance_covered_total".into(), }, @@ -409,14 +409,14 @@ fn q27_nested_subquery_prometheus_docs_example() { }; let rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { + PreASAPNode::TimeRange { range: Duration::from_secs(5), child: Rc::new(scan), }, ); let deriv = agg_per_entity( AggIntent::Deriv, - QueryExpr::PromqlSubquery { + PreASAPNode::PromqlSubquery { range: Duration::from_secs(30), resolution: Some(Duration::from_secs(5)), child: Rc::new(rate), @@ -424,7 +424,7 @@ fn q27_nested_subquery_prometheus_docs_example() { ); let expected = agg_per_entity( AggIntent::Max { col: None }, - QueryExpr::PromqlSubquery { + PreASAPNode::PromqlSubquery { range: Duration::from_secs(600), resolution: None, child: Rc::new(deriv), diff --git a/crates/integration-tests/tests/physical_common/mod.rs b/crates/integration-tests/tests/physical_common/mod.rs index aca4d4ef..4c5232ac 100644 --- a/crates/integration-tests/tests/physical_common/mod.rs +++ b/crates/integration-tests/tests/physical_common/mod.rs @@ -1,6 +1,6 @@ use asap_physical_operators::{ operators::Operator, - physical_planner::{CompiledPhysicalDag, Source}, + physical_planner::{PhysicalPostASAPDAG, Source}, runtime::{Limits, RunContext, Scope}, values::Batch, }; @@ -8,7 +8,7 @@ use futures::{executor::block_on, StreamExt}; use std::collections::BTreeMap; pub fn execute( - plan: &CompiledPhysicalDag, + plan: &PhysicalPostASAPDAG, inputs: BTreeMap, scope: Scope, ) -> Vec> { diff --git a/crates/integration-tests/tests/precompute_raw_samples.rs b/crates/integration-tests/tests/precompute_raw_samples.rs index 7fef578d..da33b25e 100644 --- a/crates/integration-tests/tests/precompute_raw_samples.rs +++ b/crates/integration-tests/tests/precompute_raw_samples.rs @@ -18,8 +18,9 @@ use asap_physical_operators::{ AggregateCore, KeyByLabelValues, Statistic, }; use asap_types::post_asap::{ - compile_post_asap_dag, EntityIdentity, ExactKind, PostAsapDag, PostAsapOperatorPayload, - SketchAlgorithm, SketchQuery, SummaryFamilyType, SummaryInputExpr, SummaryNode, SummaryUpdate, + export_post_asap_dag, EntityIdentity, ExactKind, LogicalPostASAPDAGTransport, PostASAPNode, + PostASAPOperatorPayload, SketchAlgorithm, SketchQuery, SummaryFamilyType, SummaryInputExpr, + SummaryUpdate, }; use asap_types::pre_asap::{expr_ir::ColumnRef, query_expr::Reduction}; use asap_types::types::AccuracyTarget; @@ -50,7 +51,7 @@ fn canonical(labels: &Series) -> Series { /// Every Planner candidate for `query`: the searched selection plus each /// summary replacement of the root. -fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { +fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { let root = Rc::new(lower_promql(query, accuracy).expect("lowering failed")); let mut result = SketchAlgorithmStrategy::default_cost_model() .replacements(&TargetSubDAG::new(&root)) @@ -66,7 +67,7 @@ fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { let space = search_workload(vec![("query", root)]); if let Ok(Some(selected)) = space .global_selection(&DefaultCostModel) - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) { result.push(selected); } @@ -74,10 +75,10 @@ fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { } /// Raw-input summary nodes: `(dag, raw source id, summary id)`. -fn raw_summaries(dag: &PostAsapDag) -> Vec<(u64, u64)> { +fn raw_summaries(dag: &LogicalPostASAPDAGTransport) -> Vec<(u64, u64)> { dag.nodes .iter() - .filter(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .filter(|node| matches!(node.payload, PostASAPOperatorPayload::SummaryAgg { .. })) .filter_map(|node| { let inputs = dag .edges @@ -88,7 +89,7 @@ fn raw_summaries(dag: &PostAsapDag) -> Vec<(u64, u64)> { return None; }; let source = dag.nodes.iter().find(|n| n.id == edge.producer)?; - matches!(source.payload, PostAsapOperatorPayload::Fallback { .. }) + matches!(source.payload, PostASAPOperatorPayload::Fallback { .. }) .then_some((u64::from(source.id.0), u64::from(node.id.0))) }) .collect() @@ -113,7 +114,7 @@ fn samples() -> Vec<(Series, i64, f64)> { } fn execute( - dag: &PostAsapDag, + dag: &LogicalPostASAPDAGTransport, source: u64, root: u64, rows: &[(Series, i64, f64)], @@ -128,7 +129,7 @@ fn execute( ); }); let program = serde_json::from_slice::< - asap_physical_operators::physical_planner::CompiledPhysicalDag, + asap_physical_operators::physical_planner::PhysicalPostASAPDAG, >(&serde_json::to_vec(&program).unwrap()) .unwrap(); let schema = precompute::raw_sample_schema(); @@ -273,7 +274,7 @@ fn readouts(state: &dyn AggregateCore, family: &SummaryFamilyType) -> Vec { /// or the family when it has no native state. fn check( query: &str, - dag: &PostAsapDag, + dag: &LogicalPostASAPDAGTransport, source: u64, root: u64, rows: &[(Series, i64, f64)], @@ -283,7 +284,7 @@ fn check( .iter() .find(|n| u64::from(n.id.0) == root) .unwrap(); - let PostAsapOperatorPayload::SummaryAgg { + let PostASAPOperatorPayload::SummaryAgg { family, input, reduction, @@ -408,7 +409,7 @@ fn raw_sample_summaries_compile_and_match_their_kernels() { let mut checked = BTreeMap::new(); for (query, accuracy) in queries { for candidate in candidates(query, accuracy.clone()) { - let dag = compile_post_asap_dag(&candidate).unwrap(); + let dag = export_post_asap_dag(&candidate).unwrap(); for (source, root) in raw_summaries(&dag) { match check(query, &dag, source, root, &rows) { Ok(family) => { @@ -454,21 +455,24 @@ fn raw_sample_summaries_compile_and_match_their_kernels() { /// Replace the raw summary of `sum by (service) (sum_over_time(m[5m]))` with /// another update, keeping its raw input and reduction. -fn grouped_raw_summary(family: SummaryFamilyType, input: SummaryUpdate) -> (PostAsapDag, u64, u64) { +fn grouped_raw_summary( + family: SummaryFamilyType, + input: SummaryUpdate, +) -> (LogicalPostASAPDAGTransport, u64, u64) { let candidate = candidates( "sum by (service) (sum_over_time(m[5m]))", AccuracyTarget::Exact, ) .pop() .unwrap(); - let mut dag = compile_post_asap_dag(&candidate).unwrap(); + let mut dag = export_post_asap_dag(&candidate).unwrap(); let (source, root) = raw_summaries(&dag)[0]; let node = dag .nodes .iter_mut() .find(|n| u64::from(n.id.0) == root) .unwrap(); - let PostAsapOperatorPayload::SummaryAgg { + let PostASAPOperatorPayload::SummaryAgg { family: old, input: update, .. @@ -602,7 +606,7 @@ fn raw_sample_without_grouping_drops_labels_and_name() { .iter_mut() .find(|n| u64::from(n.id.0) == root) .unwrap(); - let PostAsapOperatorPayload::SummaryAgg { reduction, .. } = &mut node.payload else { + let PostASAPOperatorPayload::SummaryAgg { reduction, .. } = &mut node.payload else { unreachable!() }; *reduction = Reduction::Reduce(GroupKeys::without(vec![service])); diff --git a/crates/integration-tests/tests/promql_numeric_regressions.rs b/crates/integration-tests/tests/promql_numeric_regressions.rs index bec86289..3982e45b 100644 --- a/crates/integration-tests/tests/promql_numeric_regressions.rs +++ b/crates/integration-tests/tests/promql_numeric_regressions.rs @@ -3,14 +3,14 @@ use asap_aware_mapping::{Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG}; use asap_integration_tests::fixtures::lower_promql; use asap_types::post_asap::{ - compile_post_asap_dag, ExactKind, SummaryExpr, SummaryFamilyType, SummaryInputExpr, - SummaryNode, SummaryUpdate, + export_post_asap_dag, ExactKind, PostASAPNode, SummaryExpr, SummaryFamilyType, + SummaryInputExpr, SummaryUpdate, }; use asap_types::pre_asap::{ColumnRef, Reduction}; use asap_types::types::AccuracyTarget; use std::rc::Rc; -fn plan(query: &str, accuracy: AccuracyTarget) -> Rc { +fn plan(query: &str, accuracy: AccuracyTarget) -> Rc { let pre = Rc::new(lower_promql(query, accuracy).unwrap()); SketchAlgorithmStrategy::default_cost_model() .replacements(&TargetSubDAG::new(&pre)) @@ -21,7 +21,7 @@ fn plan(query: &str, accuracy: AccuracyTarget) -> Rc { }) .unwrap_or_else(|| asap_aware_mapping::replacement::keep_pre_asap(&pre).unwrap()) } -fn aggregate(node: &SummaryNode) -> (&SummaryFamilyType, &SummaryUpdate, &Reduction) { +fn aggregate(node: &PostASAPNode) -> (&SummaryFamilyType, &SummaryUpdate, &Reduction) { match &node.expr { SummaryExpr::SummaryAgg { family, @@ -66,7 +66,7 @@ fn count_up_counts_targets_even_when_values_repeat_or_change_sign() { 3. ); } - compile_post_asap_dag(&node).unwrap(); + export_post_asap_dag(&node).unwrap(); } /// Ten samples give count ten, whereas sum retains the signed sample values. @@ -84,7 +84,7 @@ fn window_counts_and_sums_distinguish_one_zero_three_and_negative_values() { let got: f64 = (0..10).map(|_| contribution(family, update, value)).sum(); assert_eq!(got, if is_count { 10. } else { value * 10. }); } - compile_post_asap_dag(&node).unwrap(); + export_post_asap_dag(&node).unwrap(); } } @@ -100,7 +100,7 @@ fn sum_rate_and_increase_have_real_exact_accumulator_nodes() { let (family, _, _) = aggregate(&node); assert!(matches!(family, SummaryFamilyType::ExactAggregate(k, _) if *k == kind)); assert!(node.guarantee.as_ref().unwrap().is_exact()); - compile_post_asap_dag(&node).unwrap(); + export_post_asap_dag(&node).unwrap(); } } @@ -116,7 +116,7 @@ fn checked_ratio_must_not_certify_cross_zero_interpolation() { "direct quantile ratio should remain an available candidate" ); assert!(node.guarantee.is_none()); - compile_post_asap_dag(&node).unwrap(); + export_post_asap_dag(&node).unwrap(); // Keep the actual signed-sketch counterexample: division guards alone pass // even though the quantile interpolation does not preserve relative error. let alpha = (0.01 - 8.0 * f64::EPSILON) / 2.01; @@ -164,7 +164,7 @@ fn quantile_over_temporal_average_keeps_a_legal_candidate() { matches!(child.expr, SummaryExpr::KeepPreAsap(_)), "guarded expression must retain native maintenance input" ); - compile_post_asap_dag(&node).unwrap(); + export_post_asap_dag(&node).unwrap(); } } @@ -274,6 +274,6 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { }; assert_eq!(got, if is_count { 10. } else { 10. * value }); } - compile_post_asap_dag(node).unwrap(); + export_post_asap_dag(node).unwrap(); } } diff --git a/crates/integration-tests/tests/promql_to_post_asap.rs b/crates/integration-tests/tests/promql_to_post_asap.rs index 18ebd9c7..24f2e6a1 100644 --- a/crates/integration-tests/tests/promql_to_post_asap.rs +++ b/crates/integration-tests/tests/promql_to_post_asap.rs @@ -1,6 +1,6 @@ //! End-to-end query-string → post-ASAP IR pin (issue #98). //! -//! Drives the full pipeline — PromQL text → pre-ASAP `QueryExpr` +//! Drives the full pipeline — PromQL text → pre-ASAP `PreASAPNode` //! (`lower_promql`) → post-ASAP `SummaryExpr` DAG (via //! `SketchAlgorithmStrategy::replacements`, see [`realize`] below) — and pins //! the summary-bound shape node by node, including the family `(Kind, @@ -20,12 +20,12 @@ use asap_aware_mapping::{ }; use asap_integration_tests::fixtures::lower_promql; use asap_types::post_asap::{ - compile_post_asap_dag, CompositionOperator, EntityIdentity, ExactKind, ExactParams, - GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, SummaryExpr, - SummaryFamilyType, SummaryInputExpr, SummaryNode, SummarySchema, SummaryUpdate, ValueOperation, + export_post_asap_dag, CompositionOperator, EntityIdentity, ExactKind, ExactParams, + GroupingStrategy, PostASAPNode, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, + SummaryExpr, SummaryFamilyType, SummaryInputExpr, SummarySchema, SummaryUpdate, ValueOperation, }; use asap_types::pre_asap::expr_ir::ColumnRef; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; +use asap_types::pre_asap::query_expr::{PreASAPNode, Reduction}; use asap_types::pre_asap::schema::DataType; use asap_types::types::AccuracyTarget; @@ -34,7 +34,7 @@ use asap_types::types::AccuracyTarget; /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so the /// single-answer pins below don't all repeat it by hand. -fn realize(expr: &QueryExpr) -> Result, RealizationError> { +fn realize(expr: &PreASAPNode) -> Result, RealizationError> { let root = Rc::new(expr.clone()); let target = TargetSubDAG::new(&root); match SketchAlgorithmStrategy::default_cost_model() @@ -72,12 +72,12 @@ fn distinct_over_time_offers_hll_cardinality_readout() { }), "no HLL cardinality candidate: {candidates:?}"); } -fn lower_search_and_materialize(query: &str) -> Rc { +fn lower_search_and_materialize(query: &str) -> Rc { let pre = Rc::new(lower_promql(query, AccuracyTarget::Exact).expect("lowering failed")); let space = search_workload(vec![("query", pre)]); let selection = space.global_selection(&DefaultCostModel); selection - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .expect("materialization failed") .expect("root must be discovered") } @@ -177,7 +177,7 @@ fn dtype<'a>(schema: &'a SummarySchema, name: &str) -> &'a SummaryFamilyType { .dtype } -fn lower_and_realize(query: &str) -> Rc { +fn lower_and_realize(query: &str) -> Rc { let pre = lower_promql(query, AccuracyTarget::Exact).expect("lowering failed"); realize(&pre).expect("binding failed") } @@ -247,7 +247,7 @@ fn value_ranked_topk_over_binary_ratio_finalizes_both_summary_operands() { struct SeparatedTopK; impl AccuracyEvidenceProvider for SeparatedTopK { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &PreASAPNode) -> Option { Some(1000) } @@ -292,19 +292,19 @@ fn grouped_rate_topk_consumes_finalized_rate_values() { _ => None, }) .expect("rate-weighted CMS plan"); - let dag = compile_post_asap_dag(&plan).unwrap(); + let dag = export_post_asap_dag(&plan).unwrap(); assert!(!dag.nodes.iter().any(|node| matches!( node.payload, - asap_types::post_asap::PostAsapOperatorPayload::RelationalJoin { .. } + asap_types::post_asap::PostASAPOperatorPayload::RelationalJoin { .. } ))); let node = dag.nodes.iter().find(|node| matches!(&node.payload, - asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } + asap_types::post_asap::PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::CmsWithHeap)).unwrap(); assert_eq!( node.output_state.timing, asap_types::post_asap::ExecutionTiming::QueryTime ); - let asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { input, .. } = &node.payload + let asap_types::post_asap::PostASAPOperatorPayload::SummaryAgg { input, .. } = &node.payload else { unreachable!() }; @@ -375,7 +375,7 @@ fn weighted_topk_exports_symbolic_evidence_requirements() { let Replacement::Summary(node) = &candidate.replacement else { panic!("summary candidate") }; - let dag = compile_post_asap_dag(node).unwrap(); + let dag = export_post_asap_dag(node).unwrap(); let exported = serde_json::to_string(&dag).unwrap(); assert!(exported.contains("topk_max_distinct_items")); assert!(exported.contains("topk_membership_margin")); @@ -387,7 +387,7 @@ fn weighted_topk_exports_symbolic_evidence_requirements() { fn weighted_topk_rejects_invalid_population_evidence() { struct InvalidPopulation; impl AccuracyEvidenceProvider for InvalidPopulation { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &PreASAPNode) -> Option { Some(0) } } @@ -503,7 +503,7 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { .. } )); - let dag = compile_post_asap_dag(&plan).unwrap(); + let dag = export_post_asap_dag(&plan).unwrap(); for phase in [ asap_types::post_asap::ExecutionTiming::IngestionTime, asap_types::post_asap::ExecutionTiming::QueryTime, @@ -525,7 +525,7 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { #[test] fn promql_binary_arithmetic_preserves_both_scalar_operand_orders() { - fn is_exact_readout_or_scalar(node: &SummaryNode) -> bool { + fn is_exact_readout_or_scalar(node: &PostASAPNode) -> bool { matches!(node.expr, SummaryExpr::KeepPreAsap(_)) || matches!( node.expr, @@ -626,7 +626,7 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { ), &DefaultAccuracyModel, ); - let root = &space.roots[0].1; + let root = &space.roots()[0].1; let selected = space.global_selection(&DefaultCostModel); let chosen = selected .for_target(root) @@ -647,7 +647,7 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { let SummaryExpr::BinaryOp { lhs, rhs, .. } = &shared[0].1.expr else { panic!("expected binary ratio") }; - let producer = |readout: &Rc| match &readout.expr { + let producer = |readout: &Rc| match &readout.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => Rc::clone(summary_input), other => panic!("expected DDSketch readout, got {other:?}"), }; @@ -748,7 +748,7 @@ fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() /// Execute the ungrouped temporal TopK subset with exact state. This tests /// the emitted update contract, not sketch approximation or backend execution. -fn execute_topk_reference(plan: &SummaryNode) -> Vec<(String, f64)> { +fn execute_topk_reference(plan: &PostASAPNode) -> Vec<(String, f64)> { use std::collections::BTreeMap; let SummaryExpr::SummaryEstimate { summary_input, @@ -770,10 +770,10 @@ fn execute_topk_reference(plan: &SummaryNode) -> Vec<(String, f64)> { let SummaryExpr::KeepPreAsap(raw) = &child.expr else { panic!("expected fused raw input") }; - let QueryExpr::TimeRange { range, child } = raw.as_ref() else { + let PreASAPNode::TimeRange { range, child } = raw.as_ref() else { panic!("expected temporal input") }; - let QueryExpr::Scan { + let PreASAPNode::Scan { source: asap_types::pre_asap::Source::TimeSeries { metric }, predicates, .. @@ -995,11 +995,11 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { let SummaryExpr::KeepPreAsap(kept_leaf) = &leaf.expr else { panic!("expected KeepPreAsap leaf, got {:?}", leaf.expr); }; - let QueryExpr::TimeRange { range, child: scan } = kept_leaf.as_ref() else { + let PreASAPNode::TimeRange { range, child: scan } = kept_leaf.as_ref() else { panic!("expected TimeRange leaf, got {kept_leaf:?}"); }; assert_eq!(range.as_secs(), 300); - assert!(matches!(scan.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(scan.as_ref(), PreASAPNode::Scan { .. })); assert!( leaf.schema .fields @@ -1058,7 +1058,7 @@ fn promql_sum_of_count_over_time_is_composed_by_default_search() { ); let original_schema = original.output_schema().unwrap(); let space = search_workload(vec![("query", original)]); - let root = &space.roots[0].1; + let root = &space.roots()[0].1; let group = space.candidates_for_target(root).expect("root memo group"); let candidate = group .candidates @@ -1070,10 +1070,10 @@ fn promql_sum_of_count_over_time_is_composed_by_default_search() { }; assert_eq!(rewritten.output_schema().unwrap(), original_schema); - let QueryExpr::Project { child, .. } = rewritten.as_ref() else { + let PreASAPNode::Project { child, .. } = rewritten.as_ref() else { panic!("sum(count_over_time) needs a Float64 cast Project") }; - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction: Reduction::Reduce(by), measures, child, @@ -1091,8 +1091,8 @@ fn promql_sum_of_count_over_time_is_composed_by_default_search() { )); assert!(matches!( child.as_ref(), - QueryExpr::TimeRange { range, child } - if range.as_secs() == 300 && matches!(child.as_ref(), QueryExpr::Scan { .. }) + PreASAPNode::TimeRange { range, child } + if range.as_secs() == 300 && matches!(child.as_ref(), PreASAPNode::Scan { .. }) )); } @@ -1110,7 +1110,7 @@ fn nested_summary_explicitly_finalizes_exact_child_at_ingestion_time() { let space = search_workload(vec![("query", pre)]); let selected = space.global_selection(&DefaultCostModel); let plan = selected - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .unwrap() .unwrap(); let SummaryExpr::SummaryEstimate { summary_input, .. } = &plan.expr else { @@ -1155,12 +1155,12 @@ fn nested_summary_explicitly_finalizes_exact_child_at_ingestion_time() { .fields .iter() .any(|field| matches!(field.dtype, SummaryFamilyType::Plain(DataType::Float64)))); - compile_post_asap_dag(&plan).expect("explicit boundary is a valid post-ASAP DAG"); + export_post_asap_dag(&plan).expect("explicit boundary is a valid post-ASAP DAG"); } #[test] fn physical_node_owns_phase_independently_of_binary_payload() { - use asap_types::post_asap::{ExecutionTiming, PostAsapOperatorPayload}; + use asap_types::post_asap::{ExecutionTiming, PostASAPOperatorPayload}; for (query, expected) in [ ( // One selector: both operands cover the same series. @@ -1178,22 +1178,22 @@ fn physical_node_owns_phase_independently_of_binary_payload() { let search = search_workload(vec![("q", Rc::new(input))]); let choice = search.global_selection(&DefaultCostModel); let plan = choice - .assemble_selected_dag(&search.roots[0].1) + .assemble_selected_dag(&search.roots()[0].1) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&plan).unwrap(); + let dag = export_post_asap_dag(&plan).unwrap(); let node = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Binary { .. })) + .find(|node| matches!(node.payload, PostASAPOperatorPayload::Binary { .. })) .unwrap(); assert_eq!(node.output_state.timing, expected); let wire = serde_json::to_value(&node.payload).unwrap(); assert!(wire.get("timing").is_none()); let mut obsolete = wire.clone(); obsolete["timing"] = serde_json::json!(expected.as_str()); - assert!(serde_json::from_value::(obsolete).is_err()); - let restored: PostAsapOperatorPayload = serde_json::from_value(wire).unwrap(); + assert!(serde_json::from_value::(obsolete).is_err()); + let restored: PostASAPOperatorPayload = serde_json::from_value(wire).unwrap(); assert_eq!(restored, node.payload); } } @@ -1220,7 +1220,7 @@ fn ddsketch_ratio_without_domain_proof_is_uncertified() { ); let root_group = space .target_subdag_candidates() - .find(|group| Rc::ptr_eq(&group.target, &space.roots[0].1)) + .find(|group| Rc::ptr_eq(&group.target, &space.roots()[0].1)) .expect("root memo group"); assert!( root_group.candidates.iter().any(|candidate| { @@ -1231,20 +1231,20 @@ fn ddsketch_ratio_without_domain_proof_is_uncertified() { && node.guarantee.is_none() ) }), - "backend must receive the uncertified ratio candidate for its own selection" + "selection must receive the uncertified ratio candidate" ); let selection = space.global_selection(&DefaultCostModel); assert!( selection - .for_target(&space.roots[0].1) + .for_target(&space.roots()[0].1) .expect("selected root group") .chosen .is_none(), "Planner must not automatically select an uncertified ratio" ); let materialized = selection - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .unwrap() .expect("materialized root"); assert!(matches!(materialized.expr, SummaryExpr::KeepPreAsap(_))); @@ -1255,7 +1255,7 @@ struct FixtureQuantileDomain { upper: f64, } impl AccuracyEvidenceProvider for FixtureQuantileDomain { - fn quantile_input_domain(&self, _: &QueryExpr) -> Option { + fn quantile_input_domain(&self, _: &PreASAPNode) -> Option { Some(QuantileInputDomain { lower: self.lower, upper: self.upper, @@ -1304,8 +1304,8 @@ fn ddsketch_ratio_rejects_unsafe_domains() { fn ddsketch_ratio_rejects_one_invalid_domain_when_the_other_is_missing() { struct PartialUnsafeDomain; impl AccuracyEvidenceProvider for PartialUnsafeDomain { - fn quantile_input_domain(&self, operand: &QueryExpr) -> Option { - let QueryExpr::Aggregate { measures, .. } = operand else { + fn quantile_input_domain(&self, operand: &PreASAPNode) -> Option { + let PreASAPNode::Aggregate { measures, .. } = operand else { return None; }; matches!( @@ -1365,7 +1365,7 @@ fn ddsketch_ratio_bound_holds_for_signed_pinned_sketch_readouts() { let SummaryExpr::BinaryOp { lhs, rhs, .. } = &node.expr else { panic!("ratio") }; - let alpha = |node: &SummaryNode| { + let alpha = |node: &PostASAPNode| { let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { panic!("readout") }; @@ -1412,7 +1412,7 @@ fn ddsketch_ratio_bound_holds_for_signed_pinned_sketch_readouts() { fn ddsketch_ratio_requires_a_supported_population_size() { struct PopulationEvidence(u64); impl AccuracyEvidenceProvider for PopulationEvidence { - fn quantile_input_domain(&self, _: &QueryExpr) -> Option { + fn quantile_input_domain(&self, _: &PreASAPNode) -> Option { Some(QuantileInputDomain { lower: 1., upper: 10., @@ -1461,7 +1461,7 @@ fn without_aggregation_candidates_export_valid_dags() { let inventory = space.enumerate_candidate_dags_for_root(&0, 65_536).unwrap(); assert!(!inventory.candidates.is_empty(), "{query}"); for (_, node) in inventory.candidates.iter().flatten() { - compile_post_asap_dag(node).unwrap_or_else(|e| panic!("{query}: {e}")); + export_post_asap_dag(node).unwrap_or_else(|e| panic!("{query}: {e}")); } } } diff --git a/crates/integration-tests/tests/scan.rs b/crates/integration-tests/tests/scan.rs index bb7988d7..fa4a4f85 100644 --- a/crates/integration-tests/tests/scan.rs +++ b/crates/integration-tests/tests/scan.rs @@ -1,4 +1,4 @@ -//! `QueryExpr::Scan` — label matcher / predicate tests. +//! `PreASAPNode::Scan` — label matcher / predicate tests. //! //! The Scan schema is always [ts(0), value(1), label_a(2), label_b(3), …] //! where labels are appended alphabetically after dedup by the SchemaResolver. @@ -11,15 +11,15 @@ use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{CompareOpKind, Predicate, QueryExpr, ScalarValue, Source}; +use asap_types::pre_asap::{CompareOpKind, PreASAPNode, Predicate, ScalarValue, Source}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> PreASAPNode { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn bare_scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::Scan { +fn bare_scan(metric: &str, labels: &[&str]) -> PreASAPNode { + PreASAPNode::Scan { source: Source::TimeSeries { metric: metric.into(), }, @@ -28,42 +28,42 @@ fn bare_scan(metric: &str, labels: &[&str]) -> QueryExpr { } } -fn instant(child: QueryExpr) -> QueryExpr { - QueryExpr::TimeRange { +fn instant(child: PreASAPNode) -> PreASAPNode { + PreASAPNode::TimeRange { range: Duration::from_secs(1), child: Rc::new(child), } } fn eq_pred(col_id: usize, value: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), + Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(col_id)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(value.into()))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Utf8(value.into()))), })) } fn ne_pred(col_id: usize, value: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), + Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(col_id)), op: CompareOpKind::Ne, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(value.into()))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Utf8(value.into()))), })) } fn regex_pred(col_id: usize, pattern: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), + Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(col_id)), op: CompareOpKind::Regex, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(pattern.into()))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Utf8(pattern.into()))), })) } fn notregex_pred(col_id: usize, pattern: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), + Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(col_id)), op: CompareOpKind::NotRegex, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(pattern.into()))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Utf8(pattern.into()))), })) } @@ -80,7 +80,7 @@ fn q01_bare_scan() { // schema: [ts(0), value(1), job(2)] #[test] fn q02_equality_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, @@ -94,7 +94,7 @@ fn q02_equality_predicate() { // schema: [ts(0), value(1), status(2)] #[test] fn q03_inequality_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, @@ -108,7 +108,7 @@ fn q03_inequality_predicate() { // schema: [ts(0), value(1), job(2)] #[test] fn q04_regex_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, @@ -122,7 +122,7 @@ fn q04_regex_predicate() { // schema: [ts(0), value(1), job(2)] #[test] fn q_notregex_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, @@ -137,7 +137,7 @@ fn q_notregex_predicate() { // predicates in same alphabetical order: job first, then status #[test] fn q_multi_two_predicates() { - let expected = instant(QueryExpr::Scan { + let expected = instant(PreASAPNode::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, diff --git a/crates/integration-tests/tests/schema.rs b/crates/integration-tests/tests/schema.rs index 6024619c..ef727c3f 100644 --- a/crates/integration-tests/tests/schema.rs +++ b/crates/integration-tests/tests/schema.rs @@ -1,6 +1,6 @@ //! `Schema::closed` propagation — open/closed invariant tests. //! -//! Verifies that `QueryExpr::output_schema()` propagates the open/closed +//! Verifies that `PreASAPNode::output_schema()` propagates the open/closed //! completeness flag correctly through a lowered query tree. //! //! Key invariant: a PromQL scan is always `closed: false` (open) because its @@ -12,7 +12,7 @@ use asap_integration_tests::fixtures::lower_promql; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> asap_types::pre_asap::QueryExpr { +fn lower(q: &str) -> asap_types::pre_asap::PreASAPNode { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } diff --git a/crates/integration-tests/tests/sql_to_physical.rs b/crates/integration-tests/tests/sql_to_physical.rs index be96107e..0e31ca8b 100644 --- a/crates/integration-tests/tests/sql_to_physical.rs +++ b/crates/integration-tests/tests/sql_to_physical.rs @@ -8,8 +8,8 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use asap_types::{ - post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, - pre_asap::{Column, DataType, QueryExpr, Schema}, + post_asap::{export_post_asap_dag, PostASAPOperatorPayload, SummaryFamilyType}, + pre_asap::{Column, DataType, PreASAPNode, Schema}, types::AccuracyTarget, }; use futures::StreamExt; @@ -38,18 +38,18 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { let space = search_workload(vec![("sql", logical)]); let selected = space .global_selection(&DefaultCostModel) - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&selected).unwrap(); + let dag = export_post_asap_dag(&selected).unwrap(); let scan = dag .nodes .iter() .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::Scan { .. } + PostASAPOperatorPayload::Fallback { + expression: PreASAPNode::Scan { .. } } ) }) @@ -60,7 +60,7 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { .iter() .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); let plan = compile( - &dag, + dag.as_view(), BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), &[u64::from(dag.root.0)], ) @@ -88,10 +88,10 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { .collect() }) .collect(); - let PostAsapOperatorPayload::Fallback { expression } = &scan.payload else { + let PostASAPOperatorPayload::Fallback { expression } = &scan.payload else { unreachable!() }; - let QueryExpr::Scan { source, .. } = expression else { + let PreASAPNode::Scan { source, .. } = expression else { unreachable!() }; let mut sources = DataSources::default(); diff --git a/crates/integration-tests/tests/sql_to_post_asap.rs b/crates/integration-tests/tests/sql_to_post_asap.rs index d2d41822..7d73017d 100644 --- a/crates/integration-tests/tests/sql_to_post_asap.rs +++ b/crates/integration-tests/tests/sql_to_post_asap.rs @@ -1,7 +1,7 @@ //! End-to-end SQL query-string → post-ASAP IR pin (issue #191). //! //! The SQL counterpart of `promql_to_post_asap.rs`: drives SQL text — -//! `lower_sql` (text → pre-ASAP `QueryExpr`) → +//! `lower_sql` (text → pre-ASAP `PreASAPNode`) → //! `SketchAlgorithmStrategy::replacements` (pre-ASAP → post-ASAP //! `SummaryExpr`, see [`realize`] below) — and pins the resulting //! sketch-vs-exact-accumulator shape node by node, the way @@ -9,7 +9,7 @@ //! //! ## A structural wrinkle PromQL doesn't have //! -//! `lower_promql` returns a *bare* `QueryExpr::Aggregate` for a top-level +//! `lower_promql` returns a *bare* `PreASAPNode::Aggregate` for a top-level //! aggregation (`sum by (job) (m)`, `quantile(0.99, …)`), so [`realize`] can //! bind it directly at the tree root. `lower_sql` never does: DataFusion's //! planner always wraps even a single, unaliased aggregate in an identity @@ -27,12 +27,12 @@ use asap_aware_mapping::{ }; use asap_frontend_sql::{lower_sql, lower_sql_dialect, SqlCatalog}; use asap_types::post_asap::{ - compile_post_asap_dag, EdgeRole, ExactKind, ExactParams, GroupingStrategy, - PostAsapOperatorPayload, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, SummaryExpr, - SummaryFamilyType, SummaryNode, SummarySchema, SummaryUpdate, ValueOperation, + export_post_asap_dag, EdgeRole, ExactKind, ExactParams, GroupingStrategy, PostASAPNode, + PostASAPOperatorPayload, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, SummaryExpr, + SummaryFamilyType, SummarySchema, SummaryUpdate, ValueOperation, }; use asap_types::pre_asap::expr_ir::ColumnRef; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; +use asap_types::pre_asap::query_expr::{PreASAPNode, Reduction}; use asap_types::pre_asap::schema::{Column, DataType, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; @@ -42,7 +42,7 @@ use asap_types::workload::SqlDialect; /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so the /// single-answer pins below don't all repeat it by hand. -fn realize(expr: &QueryExpr) -> Result, RealizationError> { +fn realize(expr: &PreASAPNode) -> Result, RealizationError> { let root = Rc::new(expr.clone()); let target = TargetSubDAG::new(&root); match SketchAlgorithmStrategy::default_cost_model() @@ -89,7 +89,7 @@ fn catalog() -> SqlCatalog { ) } -async fn lower(sql: &str, accuracy: AccuracyTarget) -> QueryExpr { +async fn lower(sql: &str, accuracy: AccuracyTarget) -> PreASAPNode { lower_sql(sql, &catalog(), accuracy) .await .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) @@ -138,9 +138,11 @@ async fn clickhouse_temporal_sql_reuses_rate_and_increase_physical_summaries() { child.expr ); }; - assert!(matches!(raw.as_ref(), QueryExpr::TimeRange { range, child } + assert!( + matches!(raw.as_ref(), PreASAPNode::TimeRange { range, child } if *range == std::time::Duration::from_secs(300) - && matches!(child.as_ref(), QueryExpr::Project { .. }))); + && matches!(child.as_ref(), PreASAPNode::Project { .. })) + ); } } @@ -169,11 +171,11 @@ async fn clickhouse_outer_sum_recursively_binds_inner_temporal_aggregate() { let space = search_workload(vec![("nested", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .expect("materialization failed") .expect("root must be discovered"); - fn has_temporal_summary(node: &SummaryNode) -> bool { + fn has_temporal_summary(node: &PostASAPNode) -> bool { match &node.expr { SummaryExpr::SummaryAgg { family: @@ -192,10 +194,10 @@ async fn clickhouse_outer_sum_recursively_binds_inner_temporal_aggregate() { has_temporal_summary(&root), "inner {function} was hidden: {root:?}" ); - let dag = compile_post_asap_dag(&root).expect("nested SQL DAG must compile"); + let dag = export_post_asap_dag(&root).expect("nested SQL DAG must compile"); assert!(dag.nodes.iter().any(|node| matches!( node.payload, - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::Exact(_), .. } @@ -205,10 +207,10 @@ async fn clickhouse_outer_sum_recursively_binds_inner_temporal_aggregate() { /// The `Aggregate` node beneath the identity `Project` DataFusion's planner /// always wraps a top-level aggregate in — see the module docs above. -fn inner_aggregate(qe: &QueryExpr) -> &QueryExpr { +fn inner_aggregate(qe: &PreASAPNode) -> &PreASAPNode { match qe { - QueryExpr::Project { child, .. } => inner_aggregate(child), - QueryExpr::Aggregate { .. } => qe, + PreASAPNode::Project { child, .. } => inner_aggregate(child), + PreASAPNode::Aggregate { .. } => qe, other => panic!("expected a Project{{Aggregate}} shape, got {other:?}"), } } @@ -223,17 +225,17 @@ async fn sql_full_query_retains_project_and_binds_inner_aggregate() { ) .await; assert!( - matches!(pre_asap, QueryExpr::Project { .. }), + matches!(pre_asap, PreASAPNode::Project { .. }), "sanity: a SQL root is a Project, unlike lower_promql's bare Aggregate" ); let pre_asap = Rc::new(pre_asap); let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .expect("materialization failed") .expect("root must be discovered"); - let QueryExpr::Project { + let PreASAPNode::Project { cols: expected_cols, qualifier: expected_qualifier, .. @@ -283,7 +285,7 @@ async fn sql_join_recursively_binds_both_temporal_aggregate_children() { let space = search_workload(vec![("ratio", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .expect("materialization failed") .expect("root must be discovered"); let SummaryExpr::ValueOperation { @@ -299,7 +301,7 @@ async fn sql_join_recursively_binds_both_temporal_aggregate_children() { }; assert!(matches!( &cols[1].expr, - QueryExpr::Arithmetic { + PreASAPNode::Arithmetic { op: asap_types::pre_asap::ArithmeticOpKind::Div, .. } @@ -317,12 +319,12 @@ async fn sql_join_recursively_binds_both_temporal_aggregate_children() { assert_eq!(kind, &asap_types::pre_asap::JoinKind::Inner); assert!(matches!( pred.0.as_ref(), - QueryExpr::Compare { + PreASAPNode::Compare { left, op: asap_types::pre_asap::CompareOpKind::Eq, right, - } if matches!(left.as_ref(), QueryExpr::Column(0)) - && matches!(right.as_ref(), QueryExpr::Column(2)) + } if matches!(left.as_ref(), PreASAPNode::Column(0)) + && matches!(right.as_ref(), PreASAPNode::Column(2)) )); assert_eq!( join.schema @@ -361,11 +363,11 @@ async fn sql_join_recursively_binds_both_temporal_aggregate_children() { .guarantee .as_ref() .is_some_and(|value| value.is_exact())); - let dag = compile_post_asap_dag(&root).expect("join DAG must compile"); + let dag = export_post_asap_dag(&root).expect("join DAG must compile"); let join_id = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::RelationalJoin { .. })) + .find(|node| matches!(node.payload, PostASAPOperatorPayload::RelationalJoin { .. })) .expect("relational join node") .id; let roles = dag @@ -397,7 +399,7 @@ async fn unsupported_sql_join_shapes_remain_fail_closed() { let space = search_workload(vec![("unsupported-join", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .expect("materialization failed") .expect("root must be discovered"); let SummaryExpr::ValueOperation { child, .. } = &root.expr else { @@ -428,7 +430,7 @@ async fn sql_relational_parents_retain_summary_bound_aggregate() { let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .expect("materialization failed") .expect("root must be discovered"); @@ -483,10 +485,10 @@ async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { let mut node = pre_asap.as_ref(); loop { match node { - QueryExpr::Filter { pred, .. } => break pred.clone(), - QueryExpr::Project { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => node = child, + PreASAPNode::Filter { pred, .. } => break pred.clone(), + PreASAPNode::Project { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } => node = child, other => panic!("expected a Filter above the aggregate, got {other:?}"), } } @@ -495,12 +497,12 @@ async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { let mut node = pre_asap.as_ref(); loop { match node { - QueryExpr::Scan { predicates, .. } => break predicates.clone(), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => node = child, + PreASAPNode::Scan { predicates, .. } => break predicates.clone(), + PreASAPNode::Project { child, .. } + | PreASAPNode::Filter { child, .. } + | PreASAPNode::Aggregate { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } => node = child, other => panic!("expected a unary SQL plan over Scan, got {other:?}"), } } @@ -510,7 +512,7 @@ async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .expect("materialization failed") .expect("root must be discovered"); @@ -534,7 +536,7 @@ async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { let SummaryExpr::KeepPreAsap(raw_input) = &child.expr else { panic!("expected raw summary population below SummaryAgg"); }; - let QueryExpr::Scan { predicates, .. } = raw_input.as_ref() else { + let PreASAPNode::Scan { predicates, .. } = raw_input.as_ref() else { panic!("expected source selection to remain a Scan"); }; assert_eq!(predicates, &expected_source_predicates); @@ -545,10 +547,10 @@ async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { } assert_eq!(retained_read_predicate, Some(expected_read_predicate)); - let dag = compile_post_asap_dag(&root).expect("typed DAG compilation failed"); + let dag = export_post_asap_dag(&root).expect("typed DAG compilation failed"); assert!(dag.nodes.iter().any(|node| matches!( &node.payload, - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::Filter { pred }, .. } if pred == retained_read_predicate.as_ref().unwrap() @@ -572,7 +574,7 @@ async fn sql_filter_preserves_local_fallback_boundary_for_unsupported_child() { let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .expect("materialization failed") .expect("root must be discovered"); @@ -588,7 +590,7 @@ async fn sql_filter_preserves_local_fallback_boundary_for_unsupported_child() { } SummaryExpr::KeepPreAsap(fallback) => { assert!( - matches!(fallback.as_ref(), QueryExpr::BinaryOp { .. }), + matches!(fallback.as_ref(), PreASAPNode::BinaryOp { .. }), "AVG's unsupported rewritten child should be opaque, got {fallback:?}" ); break; @@ -683,7 +685,7 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { let SummaryExpr::KeepPreAsap(kept_leaf) = &child.expr else { panic!("expected KeepPreAsap leaf, got {:?}", child.expr); }; - assert!(matches!(kept_leaf.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(kept_leaf.as_ref(), PreASAPNode::Scan { .. })); assert!( child .schema @@ -803,20 +805,20 @@ async fn map_projection_export_preserves_unsupported_child_boundary() { let space = search_workload(vec![("map_query", pre)]); let root = space .global_selection(&DefaultCostModel) - .assemble_selected_dag(&space.roots[0].1) + .assemble_selected_dag(&space.roots()[0].1) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&root).unwrap(); + let dag = export_post_asap_dag(&root).unwrap(); assert!(dag.nodes.iter().any(|node| matches!(&node.payload, - PostAsapOperatorPayload::Value { operation: ValueOperation::Project { cols, .. }, .. } - if cols.iter().any(|item| matches!(&item.expr, QueryExpr::FunctionCall { name, .. } if name == "map")) + PostASAPOperatorPayload::Value { operation: ValueOperation::Project { cols, .. }, .. } + if cols.iter().any(|item| matches!(&item.expr, PreASAPNode::FunctionCall { name, .. } if name == "map")) ))); let mut node = root.as_ref(); loop { match &node.expr { SummaryExpr::ValueOperation { child, .. } => node = child, SummaryExpr::KeepPreAsap(child) => { - assert!(matches!(child.as_ref(), QueryExpr::BinaryOp { .. })); + assert!(matches!(child.as_ref(), PreASAPNode::BinaryOp { .. })); break; } other => panic!("unexpected map/fallback composition: {other:?}"), diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index d3d73121..5c1346e5 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -14,7 +14,7 @@ use asap_aware_mapping::{ }; use asap_frontend_promql::lower_promql_workload; use asap_types::post_asap::{ - EvaluationSchedule, SummaryMaintenanceLifecycle, SummaryMaintenanceMode, SummaryNode, + EvaluationSchedule, PostASAPNode, SummaryMaintenanceLifecycle, SummaryMaintenanceMode, }; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::types::AccuracyTarget; @@ -32,7 +32,7 @@ struct FullyCostedRuntime; impl CostModel for FullyCostedRuntime { fn raw_query_recompute_total_cost( &self, - _target: &asap_types::pre_asap::QueryExpr, + _target: &asap_types::pre_asap::PreASAPNode, _expected_reads: f64, ) -> Option { Some(Cost(1_000.0)) @@ -48,7 +48,7 @@ impl CostModel for FullyCostedRuntime { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &PostASAPNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(10.0)), @@ -61,7 +61,7 @@ impl CostModel for FullyCostedRuntime { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &PostASAPNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -187,16 +187,14 @@ fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() ); } -fn selected_plan( - workload: &PlanningWorkload, -) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { +fn selected_plan(workload: &PlanningWorkload) -> asap_aware_mapping::LifecyclePostASAPDAG { selected_plan_with_model(workload, &FullyCostedRuntime) } fn selected_plan_with_model( workload: &PlanningWorkload, model: &dyn CostModel, -) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { +) -> asap_aware_mapping::LifecyclePostASAPDAG { selected_plan_with_horizon(workload, model, Horizon(100.)) } @@ -204,7 +202,7 @@ fn selected_plan_with_horizon( workload: &PlanningWorkload, model: &dyn CostModel, horizon: Horizon, -) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { +) -> asap_aware_mapping::LifecyclePostASAPDAG { workload.validate().unwrap(); let lowered = lower_promql_workload(workload, 0) @@ -217,14 +215,14 @@ fn selected_plan_with_horizon( fn selected_plan_for_lowered( workload: &PlanningWorkload, - lowered: asap_types::pre_asap::QueryExpr, + lowered: asap_types::pre_asap::PreASAPNode, model: &dyn CostModel, horizon: Horizon, -) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { +) -> asap_aware_mapping::LifecyclePostASAPDAG { let root = Rc::new(lowered); let strategies = asap_aware_mapping::default_strategies_with(model); let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); - let target = Rc::clone(&space.roots[0].1); + let target = Rc::clone(&space.roots()[0].1); let capabilities = SummaryMaintenanceLifecycleCapabilities { supports_ephemeral: true, supports_prepared: false, @@ -274,7 +272,7 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { values::{Batch, Value}, }; use asap_types::{ - post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, + post_asap::{export_post_asap_dag, PostASAPOperatorPayload, SummaryFamilyType}, pre_asap::DataType, }; use std::{collections::BTreeMap, sync::Arc}; @@ -292,11 +290,11 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { .summary_maintenance_lifecycle, SummaryMaintenanceLifecycle::ContinuouslyMaintained ); - let dag = compile_post_asap_dag(&selected.root).unwrap(); + let dag = export_post_asap_dag(&selected.root).unwrap(); let build = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .find(|node| matches!(node.payload, PostASAPOperatorPayload::SummaryAgg { .. })) .unwrap(); let input = dag .edges @@ -448,8 +446,7 @@ fn quantile_workload(query: &str) -> PlanningWorkload { fn lifecycle_timed_dag( query: &str, lifecycle: &SummaryMaintenanceLifecycle, -) -> (asap_types::post_asap::PostAsapDag, Vec) { - use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; +) -> (asap_types::post_asap::LogicalPostASAPDAGTransport, Vec) { let workload = quantile_workload(query); let mut lowered = lower_promql_workload(&workload, 0).unwrap().remove(0); if query.contains(" by(") { @@ -459,37 +456,42 @@ fn lifecycle_timed_dag( } let root = selected_plan_for_lowered(&workload, lowered, &FullyCostedRuntime, Horizon(100.)).root; - let candidates = enumerate_summary_maintenance_lifecycles( + let candidates = asap_aware_mapping::CandidateLifecyclePostASAPDAGs::from_post_asap_dag( + (), root, - WorkloadDemand::new_with_data( - &workload.query_workload, - workload.data_workload.as_ref().unwrap(), - &[1], - ), - NOW_MS, - Some(Horizon(100.)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &FullyCostedRuntime, + asap_aware_mapping::CandidateTimingContext { + demand: WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + now_ms: NOW_MS, + horizon: Some(Horizon(100.)), + capabilities: SummaryMaintenanceLifecycleCapabilities::ALL, + cost_model: &FullyCostedRuntime, + }, + 4096, ) .unwrap(); let choices: Vec<_> = candidates - .deployments() + .lifecycle_alternatives(0) + .unwrap() .iter() .map(|deployment| (deployment.post_asap_node_id, lifecycle.clone())) .collect(); let mut states: Vec<_> = choices.iter().map(|(id, _)| u64::from(id.0)).collect(); states.sort_unstable(); let dag = candidates - .select(&choices) + .select_lifecycles(0, &choices) .unwrap() - .execution_timed_dag() + .export_timed_dag() .unwrap(); (dag, states) } /// Compile inputs for a timed DAG: its raw source, available at either phase. fn raw_inputs( - dag: &asap_types::post_asap::PostAsapDag, + dag: &asap_types::post_asap::LogicalPostASAPDAGTransport, ) -> std::collections::BTreeMap { let raw = dag .nodes @@ -497,7 +499,7 @@ fn raw_inputs( .find(|node| { matches!( node.payload, - asap_types::post_asap::PostAsapOperatorPayload::Fallback { .. } + asap_types::post_asap::PostASAPOperatorPayload::Fallback { .. } ) }) .unwrap(); @@ -529,8 +531,8 @@ fn planner_lifecycle_selection_reproduces_strategy_timing() { != SummaryMaintenanceLifecycle::Ephemeral }) })); - let strategy = asap_types::post_asap::compile_post_asap_dag(&plan.root).unwrap(); - assert_eq!(plan.execution_timed_dag().unwrap(), strategy, "{query}"); + let strategy = asap_types::post_asap::export_post_asap_dag(&plan.root).unwrap(); + assert_eq!(plan.export_timed_dag().unwrap(), strategy, "{query}"); } } @@ -559,7 +561,7 @@ fn chosen_lifecycle_timing_decides_precompute_contents() { let inputs = raw_inputs(&dag); let (&raw_id, contract) = inputs.iter().next().unwrap(); let schema = contract.schema.clone(); - let frontier = frontier_from_timing(&dag).unwrap(); + let frontier = frontier_from_timing(dag.as_view()).unwrap(); let candidate = compile_candidate(&dag, inputs, &[u64::from(dag.root.0)], &frontier).unwrap(); let rows = (1..=100) @@ -645,13 +647,13 @@ fn lifecycle_timing_cuts_one_compilation() { let (compiled_dag, _) = lifecycle_timed_dag(query, &ephemeral); let inputs = raw_inputs(&compiled_dag); let roots = [u64::from(compiled_dag.root.0)]; - let compiled = compile(&compiled_dag, inputs.clone(), &roots).unwrap(); + let compiled = compile(compiled_dag.as_view(), inputs.clone(), &roots).unwrap(); for lifecycle in [ SummaryMaintenanceLifecycle::ContinuouslyMaintained, ephemeral, ] { let (dag, states) = lifecycle_timed_dag(query, &lifecycle); - let frontier = frontier_from_timing(&dag).unwrap(); + let frontier = frontier_from_timing(dag.as_view()).unwrap(); // Retained states read by a query-time consumer, or the root itself. let query_time = |id: u64| { dag.nodes.iter().any(|node| { @@ -692,10 +694,7 @@ fn lifecycle_timing_cuts_one_compilation() { /// rebuilds it from the raw source at query time; both rank alike. #[test] fn chosen_population_lifecycle_decides_precompute_contents() { - use asap_aware_mapping::{ - enumerate_summary_maintenance_lifecycles, - maintained_population::MaintainedPopulationStrategy, - }; + use asap_aware_mapping::maintained_population::MaintainedPopulationStrategy; use asap_physical_operators::{ physical_planner::{ compile_candidate, @@ -706,7 +705,7 @@ fn chosen_population_lifecycle_decides_precompute_contents() { values::{Batch, Value}, }; use asap_types::post_asap::{ - maintained_population::PopulationInput, PostAsapOperatorPayload, ValueOperation, + maintained_population::PopulationInput, PostASAPOperatorPayload, ValueOperation, }; use std::{collections::BTreeMap, sync::Arc}; @@ -722,30 +721,34 @@ fn chosen_population_lifecycle_decides_precompute_contents() { SummaryMaintenanceLifecycle::ContinuouslyMaintained, SummaryMaintenanceLifecycle::Ephemeral, ] { - let candidates = enumerate_summary_maintenance_lifecycles( + let candidates = asap_aware_mapping::CandidateLifecyclePostASAPDAGs::from_post_asap_dag( + (), Rc::clone(&root), - WorkloadDemand::new_with_data( - &workload.query_workload, - workload.data_workload.as_ref().unwrap(), - &[1], - ), - NOW_MS, - Some(Horizon(100.)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &FullyCostedRuntime, + asap_aware_mapping::CandidateTimingContext { + demand: WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + now_ms: NOW_MS, + horizon: Some(Horizon(100.)), + capabilities: SummaryMaintenanceLifecycleCapabilities::ALL, + cost_model: &FullyCostedRuntime, + }, + 4096, ) .unwrap(); - let [deployment] = candidates.deployments() else { + let [deployment] = candidates.lifecycle_alternatives(0).unwrap() else { panic!("one population state"); }; let id = deployment.post_asap_node_id; let dag = candidates - .select(&[(id, lifecycle.clone())]) + .select_lifecycles(0, &[(id, lifecycle.clone())]) .unwrap() - .execution_timed_dag() + .export_timed_dag() .unwrap(); let population = dag.nodes.iter().find(|node| node.id == id).unwrap(); - let PostAsapOperatorPayload::Value { + let PostASAPOperatorPayload::Value { operation: ValueOperation::MaintainPopulation { population }, } = &population.payload else { @@ -758,11 +761,11 @@ fn chosen_population_lifecycle_decides_precompute_contents() { let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| matches!(node.payload, PostASAPOperatorPayload::Fallback { .. })) .unwrap(); let (raw_id, schema) = (u64::from(raw.id.0), Arc::new(raw.output_schema.clone())); let frontier = - asap_physical_operators::physical_planner::frontier_from_timing(&dag).unwrap(); + asap_physical_operators::physical_planner::frontier_from_timing(dag.as_view()).unwrap(); let candidate = compile_candidate( &dag, BTreeMap::from([(raw_id, InputContract::bounded(schema.clone()))]), @@ -838,10 +841,9 @@ fn chosen_population_lifecycle_decides_precompute_contents() { /// state leaves Sum in the query DAG. #[test] fn grouped_rate_sum_placement_is_a_lifecycle_choice() { - use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; use asap_physical_operators::physical_planner::{compile_candidate, InputContract}; use asap_types::post_asap::{ - ExactKind, PostAsapOperatorPayload, SummaryExpr, SummaryFamilyType, + ExactKind, PostASAPOperatorPayload, SummaryExpr, SummaryFamilyType, }; use std::{collections::BTreeMap, sync::Arc}; @@ -852,7 +854,7 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { ) .unwrap(), ); - let is_exact = |node: &SummaryNode, kind: ExactKind| { + let is_exact = |node: &PostASAPNode, kind: ExactKind| { matches!(&node.expr, SummaryExpr::SummaryAgg { family: SummaryFamilyType::ExactAggregate(k, _), .. } if *k == kind) @@ -877,21 +879,26 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { SummaryMaintenanceLifecycle::ContinuouslyMaintained, SummaryMaintenanceLifecycle::Ephemeral, ] { - let lifecycles = enumerate_summary_maintenance_lifecycles( + let lifecycles = asap_aware_mapping::CandidateLifecyclePostASAPDAGs::from_post_asap_dag( + (), Rc::clone(candidate), - WorkloadDemand::new_with_data( - &workload.query_workload, - workload.data_workload.as_ref().unwrap(), - &[1], - ), - NOW_MS, - Some(Horizon(100.)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &FullyCostedRuntime, + asap_aware_mapping::CandidateTimingContext { + demand: WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + now_ms: NOW_MS, + horizon: Some(Horizon(100.)), + capabilities: SummaryMaintenanceLifecycleCapabilities::ALL, + cost_model: &FullyCostedRuntime, + }, + 4096, ) .unwrap(); let choices = lifecycles - .deployments() + .lifecycle_alternatives(0) + .unwrap() .iter() .map(|deployment| { let lifecycle = if is_exact(&deployment.summary, ExactKind::Sum) { @@ -904,17 +911,17 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { .collect::>(); assert_eq!(choices.len(), 2, "Rate and Sum states"); let dag = lifecycles - .select(&choices) + .select_lifecycles(0, &choices) .unwrap() - .execution_timed_dag() + .export_timed_dag() .unwrap(); let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| matches!(node.payload, PostASAPOperatorPayload::Fallback { .. })) .unwrap(); let frontier = - asap_physical_operators::physical_planner::frontier_from_timing(&dag).unwrap(); + asap_physical_operators::physical_planner::frontier_from_timing(dag.as_view()).unwrap(); let [boundary] = frontier.as_slice() else { panic!("one precompute output, got {frontier:?}"); }; @@ -950,8 +957,8 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { else { unreachable!() }; - let state = |payload: &PostAsapOperatorPayload, kind: ExactKind| { - matches!(payload, PostAsapOperatorPayload::SummaryAgg { + let state = |payload: &PostASAPOperatorPayload, kind: ExactKind| { + matches!(payload, PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(k, _), .. } if *k == kind) }; @@ -965,18 +972,18 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { /// The lifecycle-timed DAG Planner selects for `query` with upfront series /// typing, and whether it keeps an ingestion-time Binary. -fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDag, bool) { - use asap_types::post_asap::{ExecutionTiming, PostAsapOperatorPayload}; +fn typed_selection(query: &str) -> (asap_types::post_asap::LogicalPostASAPDAGTransport, bool) { + use asap_types::post_asap::{ExecutionTiming, PostASAPOperatorPayload}; let workload = quantile_workload(query); let lowered = asap_types::pre_asap::schema::with_promql_series_identity( &lower_promql_workload(&workload, 0).unwrap().remove(0), ) .unwrap(); let dag = selected_plan_for_lowered(&workload, lowered, &FullyCostedRuntime, Horizon(100.)) - .execution_timed_dag() + .export_timed_dag() .unwrap(); let ingestion_binary = dag.nodes.iter().any(|node| { - matches!(node.payload, PostAsapOperatorPayload::Binary { .. }) + matches!(node.payload, PostASAPOperatorPayload::Binary { .. }) && node.output_state.timing == ExecutionTiming::IngestionTime }); (dag, ingestion_binary) @@ -985,7 +992,7 @@ fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDag, bool) { /// Execute a timed DAG's precompute and query graphs over `samples` /// (`(metric, job, seconds, value)`) at 300s; returns the root's values. fn execute_timed( - dag: &asap_types::post_asap::PostAsapDag, + dag: &asap_types::post_asap::LogicalPostASAPDAGTransport, samples: &[(&str, &str, i64, f64)], ) -> Vec { use asap_physical_operators::{ @@ -996,26 +1003,26 @@ fn execute_timed( values::{Batch, Value}, }; use asap_types::{ - post_asap::PostAsapOperatorPayload, - pre_asap::{QueryExpr, Source}, + post_asap::PostASAPOperatorPayload, + pre_asap::{PreASAPNode, Source}, }; use std::{collections::BTreeMap, sync::Arc}; // Raw inputs: a selector Fallback is itself the input; a retained // expression reads each of its selectors through its raw-series slots. let mut raw = BTreeMap::new(); for node in &dag.nodes { - let PostAsapOperatorPayload::Fallback { expression } = &node.payload else { + let PostASAPOperatorPayload::Fallback { expression } = &node.payload else { continue; }; - let metric = |selector: &QueryExpr| match selector { - QueryExpr::TimeRange { child, .. } => match child.as_ref() { - QueryExpr::Scan { + let metric = |selector: &PreASAPNode| match selector { + PreASAPNode::TimeRange { child, .. } => match child.as_ref() { + PreASAPNode::Scan { source: Source::TimeSeries { metric }, .. } => Some(metric.clone()), _ => None, }, - QueryExpr::Scan { + PreASAPNode::Scan { source: Source::TimeSeries { metric }, .. } => Some(metric.clone()), @@ -1053,7 +1060,7 @@ fn execute_timed( .collect(); Batch::try_new(schema.clone(), rows).unwrap() }; - let frontier = frontier_from_timing(dag).unwrap(); + let frontier = frontier_from_timing(dag.as_view()).unwrap(); let candidate = compile_candidate( dag, raw.iter() @@ -1063,7 +1070,7 @@ fn execute_timed( &frontier, ) .unwrap(); - let raw_sources = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDag| { + let raw_sources = |plan: &asap_physical_operators::physical_planner::PhysicalPostASAPDAG| { plan.input_contracts() .filter_map(|(id, _)| raw.get(&id).map(|(schema, name)| (id, batch(schema, name)))) .collect::>() @@ -1151,3 +1158,219 @@ fn maintained_arithmetic_over_one_selector_executes() { // two values only the larger is. assert_eq!(values, [10.0]); } + +/// The named layer collections connect directly: IDs, rejections, lifecycle +/// metadata and shared logical identity survive through physical compilation. +#[test] +fn named_candidate_collections_preserve_timing_and_compile_errors() { + use asap_aware_mapping::{ + CandidateLifecyclePostASAPDAGs, CandidateTimingContext, CandidateTimingError, + }; + use asap_physical_operators::physical_planner::{ + compile_physical_dag_candidates, PhysicalCandidateError, + }; + let workload = quantile_workload("sum by(job)(rate(m[1m]))"); + let lowered = lower_promql_workload(&workload, 0).unwrap().remove(0); + let lowered = + asap_physical_operators::physical_planner::promql_rows::with_series_identity(&lowered) + .unwrap(); + let logical = asap_aware_mapping::search_workload(vec![(17usize, Rc::new(lowered))]); + let context = || CandidateTimingContext { + demand: WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + now_ms: NOW_MS, + horizon: Some(Horizon(100.)), + capabilities: SummaryMaintenanceLifecycleCapabilities::ALL, + cost_model: &FullyCostedRuntime, + }; + assert!(matches!( + logical.with_timing_for_root(&17, context(), 4096, 0), + Err(CandidateTimingError::ExpansionLimit(0)) + )); + // The logical budget is reported the same way as the assignment budget. + assert!(matches!( + logical.with_timing_for_root(&17, context(), 0, 4096), + Err(CandidateTimingError::ExpansionLimit(0)) + )); + assert!(logical + .with_timing_for_root(&18, context(), 4096, 4096) + .is_err()); + let timed = logical + .with_timing_for_root(&17, context(), 4096, 65536) + .unwrap(); + assert!(!timed.is_empty()); + let mut indices = std::collections::BTreeMap::new(); + let mut valid = 0; + let mut rejected = 0; + for (metadata, assignment) in timed.iter() { + assert_eq!(metadata.id, 17); + match assignment { + Ok(assignment) => { + valid += 1; + let plan = metadata.lifecycle.as_ref().unwrap(); + assert!(Rc::ptr_eq( + &plan.root, + assignment + .index() + .node_ids + .summary_node(assignment.index().root_id) + .unwrap() + )); + if let Some(previous) = + indices.insert(metadata.logical_candidate, assignment.index().clone()) + { + assert!(Rc::ptr_eq(&previous, assignment.index())); + } + } + Err(_) => rejected += 1, + } + } + assert!(valid > 1); + assert!(rejected > 0); + assert_eq!(valid + rejected, timed.len()); + // The total assignment budget is exact: the collection's size fits, one less does not. + assert_eq!( + logical + .with_timing_for_root(&17, context(), 4096, timed.len()) + .unwrap() + .len(), + timed.len() + ); + let short = timed.len() - 1; + assert!(matches!( + logical.with_timing_for_root(&17, context(), 4096, short), + Err(CandidateTimingError::ExpansionLimit(limit)) if limit == short + )); + let physical = compile_physical_dag_candidates( + timed.iter(), + timed.rejected_assemblies().to_vec(), + |_, assignment| { + Ok(( + raw_inputs(&assignment.to_transport()), + vec![u64::from(assignment.index().root_id.0)], + )) + }, + ); + assert_eq!(physical.len(), timed.len()); + assert_eq!(physical.rejected_assemblies(), timed.rejected_assemblies()); + let mut compiled = 0; + for (i, ((before, timing), (after, dag))) in timed.iter().zip(physical.iter()).enumerate() { + assert_eq!( + ( + before.id, + before.logical_candidate, + before.assignment_candidate + ), + ( + after.id, + after.logical_candidate, + after.assignment_candidate + ) + ); + assert_eq!(before.choices, after.choices); + if timing.is_err() { + assert!(matches!(dag, Err(PhysicalCandidateError::Timing(_)))); + } + if dag.is_ok() { + compiled += 1; + assert!(after.lifecycle.is_some()); + physical.materialize(i).unwrap().validate().unwrap(); + } + } + assert!(compiled > 1); + let failed = compile_physical_dag_candidates(timed.iter(), Vec::new(), |_, _| { + Err(asap_physical_operators::Error::Invalid( + "missing deployment input evidence".into(), + )) + }); + assert_eq!(failed.len(), timed.len()); + // A resolver failure is a compile error; timing failures stay timing errors. + for ((_, timing), (metadata, result)) in timed.iter().zip(failed.iter()) { + assert_eq!(metadata.id, 17); + if timing.is_ok() { + assert!(matches!(result, Err(PhysicalCandidateError::Compile(_)))); + } else { + assert!(matches!(result, Err(PhysicalCandidateError::Timing(_)))); + } + } + assert!(matches!( + physical.materialize(physical.len()), + Err(PhysicalCandidateError::Compile(_)) + )); + + // The single-DAG entry point is the same collection, not a public lifecycle helper. + let first = timed + .iter() + .find_map(|(metadata, timing)| { + timing.ok().and_then(|_| { + metadata + .lifecycle + .filter(|plan| !plan.deployments.is_empty()) + .map(|plan| plan.root) + }) + }) + .unwrap(); + let one = + CandidateLifecyclePostASAPDAGs::from_post_asap_dag(17usize, first.clone(), context(), 4096) + .unwrap(); + assert_eq!(one.logical_len(), 1); + assert!(!one.lifecycle_alternatives(0).unwrap().is_empty()); + + // A listed alternative can be priced before binding; a lifecycle or state + // the candidate does not offer is rejected rather than priced. + use asap_aware_mapping::SummaryMaintenanceLifecycleChoiceError as Choice; + let deployment = &one.lifecycle_alternatives(0).unwrap()[0]; + let state = deployment.post_asap_node_id; + for alternative in &deployment.alternatives { + let lifecycle = &alternative.summary_maintenance_lifecycle; + let guarantee = one.lifecycle_guarantee(0, state, lifecycle).unwrap(); + assert_eq!(&guarantee.summary_maintenance_lifecycle, lifecycle); + } + let unoffered = SummaryMaintenanceLifecycle::Shared { + retention: asap_types::workload::DurationMs(u64::MAX), + }; + assert!(!deployment + .alternatives + .iter() + .any(|a| a.summary_maintenance_lifecycle == unoffered)); + assert!(matches!( + one.lifecycle_guarantee(0, state, &unoffered).unwrap_err().as_ref(), + CandidateTimingError::Choice(Choice::NotAnAlternative(id)) if *id == state + )); + let unknown = asap_types::post_asap::PostASAPNodeId(u32::MAX); + assert!(matches!( + one.lifecycle_guarantee(0, unknown, &SummaryMaintenanceLifecycle::Ephemeral) + .unwrap_err() + .as_ref(), + CandidateTimingError::Choice(Choice::UnknownSummary(id)) if *id == unknown + )); + let mut invalid = context(); + invalid.horizon = Some(Horizon(-1.)); + let rejected = + CandidateLifecyclePostASAPDAGs::from_post_asap_dag(17usize, first, invalid, 4096).unwrap(); + assert_eq!(rejected.len(), 1); + let physical = compile_physical_dag_candidates(rejected.iter(), Vec::new(), |_, _| { + panic!("invalid lifecycle context must not reach compilation") + }); + assert_eq!(physical.len(), 1); + let (metadata, error) = physical.iter().next().unwrap(); + assert_eq!(metadata.id, 17); + assert!(matches!( + error, + Err(PhysicalCandidateError::Timing(timing)) + if matches!(timing.as_ref(), CandidateTimingError::Lifecycle(_)) + )); + let mut counts = std::collections::BTreeMap::::new(); + for (metadata, _) in timed.iter() { + *counts.entry(metadata.logical_candidate).or_default() += 1; + } + let largest = *counts.values().max().unwrap(); + assert!(largest < timed.len()); + assert!(matches!( + logical.with_timing_for_root(&17, context(), 4096, largest), + Err(CandidateTimingError::ExpansionLimit(limit)) if limit == largest + )); +} diff --git a/crates/integration-tests/tests/time_range.rs b/crates/integration-tests/tests/time_range.rs index 940facc3..8c0ed531 100644 --- a/crates/integration-tests/tests/time_range.rs +++ b/crates/integration-tests/tests/time_range.rs @@ -1,4 +1,4 @@ -//! `QueryExpr::TimeRange` — range / streaming function tests. +//! `PreASAPNode::TimeRange` — range / streaming function tests. //! //! All range functions lower to `Aggregate { child: TimeRange { range, child: Scan } }`. //! The temporal range lives on the `TimeRange` node, not in the `AggIntent`. @@ -12,15 +12,15 @@ use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction, Source}; +use asap_types::pre_asap::{AggIntent, PreASAPNode, Reduction, Source}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> PreASAPNode { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn scan(metric: &str) -> QueryExpr { - QueryExpr::Scan { +fn scan(metric: &str) -> PreASAPNode { + PreASAPNode::Scan { source: Source::TimeSeries { metric: metric.into(), }, @@ -29,13 +29,13 @@ fn scan(metric: &str) -> QueryExpr { } } -fn range_agg(range_secs: u64, intent: AggIntent, metric: &str) -> QueryExpr { - QueryExpr::Aggregate { +fn range_agg(range_secs: u64, intent: AggIntent, metric: &str) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec!["".into()], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: Rc::new(PreASAPNode::TimeRange { range: Duration::from_secs(range_secs), child: Rc::new(scan(metric)), }), diff --git a/crates/planner/src/lib.rs b/crates/planner/src/lib.rs index 837b988a..881b5113 100644 --- a/crates/planner/src/lib.rs +++ b/crates/planner/src/lib.rs @@ -14,7 +14,7 @@ use std::rc::Rc; use asap_types::parsed_workload::{ParsedWorkload, ParsedWorkloadError}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::pre_asap::query_expr::PreASAPNode; use asap_types::workload::{PlanningWorkload, QueryLanguage, SqlDialect, WorkloadError}; use asap_frontend_metricsql::{lower_metricsql, MetricsqlError}; @@ -196,7 +196,11 @@ pub enum PlanError { pub async fn e2e_plan(input: UserInput<'_>) -> Result { input.validate()?; - let exprs = lower(&input).await?; + let exprs = lower_pre_asap_dag_candidates(&input) + .await? + .into_iter() + .map(|(_, dag)| dag) + .collect(); let parsed = ParsedWorkload::new(input.workload.clone(), exprs)?; let fallback = MajorPass; @@ -209,13 +213,23 @@ pub async fn e2e_plan(input: UserInput<'_>) -> Result { optimize(pass, optimization).map_err(PlanError::Optimize) } +/// Generate frontend candidates without logical or lifecycle selection. +/// IDs are normalized workload entry indices, including repeating entries. +/// Current frontends lower deterministically: one candidate for each entry. +pub async fn lower_pre_asap_dag_candidates( + input: &UserInput<'_>, +) -> Result, PlanError> { + input.validate()?; + Ok(lower(input).await?.into_iter().enumerate().collect()) +} + /// One expression per normalized entry, in `QueryWorkload::entries()` order. /// /// The SQL and MetricsQL frontends are driven one entry at a time rather than /// through `lower_sql_batch`, which walks `query_batch` alone and would drop /// every repeating query — exactly the entries whose recurrence the lifecycle /// stage needs. -async fn lower(input: &UserInput<'_>) -> Result>, PlanError> { +async fn lower(input: &UserInput<'_>) -> Result>, PlanError> { let entries = || input.workload.query_workload.entries(); match &input.frontend_specific { diff --git a/crates/planner/tests/e2e_plan.rs b/crates/planner/tests/e2e_plan.rs index 063527a0..e60b4d40 100644 --- a/crates/planner/tests/e2e_plan.rs +++ b/crates/planner/tests/e2e_plan.rs @@ -65,7 +65,7 @@ fn sql_workload( } /// The facade turns a prepared workload into one selected DAG per query, in -/// `QueryWorkload::entries()` order, without the caller touching `PlanSpace`. +/// `QueryWorkload::entries()` order, without the caller touching `CandidateLogicalPostASAPDAGs`. #[tokio::test] async fn plans_every_query_in_entry_order() { let workload = sql_workload( @@ -82,6 +82,13 @@ async fn plans_every_query_in_entry_order() { PlanningModels::builtin(), ); + let frontend = asap_planner::lower_pre_asap_dag_candidates(&input) + .await + .expect("frontend candidates"); + assert_eq!( + frontend.iter().map(|(id, _)| *id).collect::>(), + vec![0, 1] + ); let output = e2e_plan(input).await.expect("workload plans"); let PlanOutput::Dag { plans } = &output else { panic!("no lifecycle input was supplied, so the DAG-only variant is expected"); @@ -114,6 +121,13 @@ async fn lowers_repeating_sql_entries_too() { PlanningModels::builtin(), ); + let frontend = asap_planner::lower_pre_asap_dag_candidates(&input) + .await + .expect("frontend candidates"); + assert_eq!( + frontend.iter().map(|(id, _)| *id).collect::>(), + vec![0, 1] + ); let output = e2e_plan(input).await.expect("workload plans"); assert_eq!( output.len(), diff --git a/crates/sql-function-catalog/src/lib.rs b/crates/sql-function-catalog/src/lib.rs index bf7dd143..01702c24 100644 --- a/crates/sql-function-catalog/src/lib.rs +++ b/crates/sql-function-catalog/src/lib.rs @@ -280,7 +280,7 @@ pub struct ClickHouseBuiltin { pub const CLICKHOUSE_BUILTINS: &[ClickHouseBuiltin] = &[ // Explicit time-series reducers. These deliberately survive under their // own names: the SQL frontend validates (value, timestamp, window_ms) and - // lowers the window to QueryExpr::TimeRange rather than pretending these + // lowers the window to PreASAPNode::TimeRange rather than pretending these // are ordinary tabular aggregates. ClickHouseBuiltin { name: "asap_rate", diff --git a/crates/types/Cargo.toml b/crates/types/Cargo.toml index 6b353657..2d8d18ca 100644 --- a/crates/types/Cargo.toml +++ b/crates/types/Cargo.toml @@ -8,7 +8,7 @@ edition = "2021" # execution logic — removed, no real implementor existed; see issue #190). # No internal deps. [dependencies] -# "rc" — QueryExpr's child fields are Rc> (issue #212, #222: +# "rc" — PreASAPNode's child fields are Rc> (issue #212, #222: # shared sub-expressions), and Rc's Serialize/Deserialize impls live behind # this feature flag. dag_export.rs / DagNode already flatten the tree to a # node+edge list for JSON export, so this does not change wire format — a diff --git a/crates/types/src/cost.rs b/crates/types/src/cost.rs index 5a7a5a4d..d77ec10f 100644 --- a/crates/types/src/cost.rs +++ b/crates/types/src/cost.rs @@ -73,7 +73,7 @@ pub enum BaselineRef { /// — "do nothing" (never apply ASAP-aware replacement at all). PreAsapRecomputation, /// The best-ranked *non-selected* legal candidate for the same target - /// (`rank` into that target's own `CandidatePostASAPDAGs::cost_sorted` ordering, + /// (`rank` into that target's own `CandidateLogicalPostASAPDAGs::cost_sorted` ordering, /// `0` = best; a baseline referencing this variant is always `rank >= /// 1`, since `rank 0` is what got selected). HighestRankedNonSelectedCandidate { rank: usize }, diff --git a/crates/types/src/dag_export.rs b/crates/types/src/dag_export.rs index 439e8f25..3792d465 100644 --- a/crates/types/src/dag_export.rs +++ b/crates/types/src/dag_export.rs @@ -1,15 +1,15 @@ -//! Export the pre-ASAP [`QueryExpr`] tree as a generic node/edge graph, for tools +//! Export the pre-ASAP [`PreASAPNode`] tree as a generic node/edge graph, for tools //! that need to render or diff the IR (the `dag_export` example + the //! `tools/dag-viewer` viewer — see issue #133) rather than walk it in Rust. //! -//! `QueryExpr` already derives `Serialize`, but as a Rust-shaped tagged tree +//! `PreASAPNode` already derives `Serialize`, but as a Rust-shaped tagged tree //! (`Rc` children nested inside each variant's own field). This module //! flattens that into an explicit node list + child-id edges — the shape a //! generic graph renderer wants — and additionally tags each node with //! [`structural_hash`](crate::pre_asap::cse::structural_hash), so a caller //! with several exported queries can spot identical subtrees (a //! shared `Scan`, a repeated `Aggregate` shape, …) by comparing hashes -//! rather than re-implementing `QueryExpr: PartialEq` structural comparison +//! rather than re-implementing `PreASAPNode: PartialEq` structural comparison //! client-side. //! //! This is literally the same hashing @@ -33,7 +33,7 @@ //! `asap_types`, never the reverse — can annotate an already-exported graph //! after the fact without this module needing to know anything about that //! layer's concepts. Concretely: `asap-aware-mapping`'s `explanation` module -//! (issue #257) computes `structural_hash` over the same `QueryExpr` +//! (issue #257) computes `structural_hash` over the same `PreASAPNode` //! subtrees this module does (via the identical function). The devtools //! exporter uses that hash to narrow candidates, then compares //! `ReplacementExplanation::target` with [`DagNode::source_expr`] for a @@ -47,9 +47,9 @@ use std::rc::Rc; use serde::Serialize; use crate::cost::CostAnnotation; -use crate::post_asap::{AccuracyError, ResultGuarantee, SummaryExpr, SummaryNode}; +use crate::post_asap::{AccuracyError, PostASAPNode, ResultGuarantee, SummaryExpr}; use crate::pre_asap::cse::{structural_hash, HashCache}; -use crate::pre_asap::query_expr::{QueryExpr, Source}; +use crate::pre_asap::query_expr::{PreASAPNode, Source}; /// One flattened IR node. `detail` holds this node's own scalar fields /// (predicates, aggregate funcs, schema, sort keys, …) — everything except @@ -57,7 +57,7 @@ use crate::pre_asap::query_expr::{QueryExpr, Source}; #[derive(Debug, Clone, Serialize)] pub struct DagNode { pub id: u32, - /// The `QueryExpr` variant name (e.g. `"Aggregate"`). + /// The `PreASAPNode` variant name (e.g. `"Aggregate"`). pub kind: &'static str, /// Short human-readable summary for a node's collapsed on-graph label. pub label: String, @@ -83,7 +83,7 @@ pub struct DagNode { /// /// `None` for the same reason `source_expr` is `None` — a post-ASAP- /// originated node in an [`export_post_asap`] merged graph has no - /// `QueryExpr` to hash. Omitted from JSON entirely (rather than, say, + /// `PreASAPNode` to hash. Omitted from JSON entirely (rather than, say, /// serialized as `0`) so a consumer's shared-subtree-by-hash pass can /// tell "no hash" apart from a real hash that happens to collide with a /// placeholder — `0` is a legal `structural_hash` output, not a safe @@ -95,14 +95,14 @@ pub struct DagNode { /// this value structurally to avoid treating a hash collision as node /// identity. /// - /// `None` for a node with no corresponding pre-ASAP `QueryExpr` at all — + /// `None` for a node with no corresponding pre-ASAP `PreASAPNode` at all — /// only possible for a post-ASAP-originated node inside a merged /// [`export_post_asap`] graph (a `SummaryAgg`/`SummaryJoin`/… node has no - /// single `QueryExpr` it corresponds to). Every node [`export`] itself + /// single `PreASAPNode` it corresponds to). Every node [`export`] itself /// produces is pre-ASAP by construction and always carries `Some`. #[serde(skip)] - pub source_expr: Option, - /// In-process identity of the source `QueryExpr`. Unlike `source_expr`'s + pub source_expr: Option, + /// In-process identity of the source `PreASAPNode`. Unlike `source_expr`'s /// structural value, this preserves an `Rc` child reached from multiple /// parents so post-ASAP flattening can retain true DAG sharing. #[serde(skip)] @@ -207,7 +207,7 @@ pub struct NamedGraph { pub name: String, /// The original query text (SQL or PromQL) this graph was lowered from, /// for display alongside the graph — not used by `export` itself, since - /// that only sees the already-lowered `QueryExpr`. Optional because not + /// that only sees the already-lowered `PreASAPNode`. Optional because not /// every producer of a `NamedGraph` has the source text on hand. #[serde(skip_serializing_if = "Option::is_none")] pub source: Option, @@ -231,8 +231,8 @@ pub struct NamedGraph { /// query: every node that has no winning replacement renders as an /// ordinary pre-ASAP [`DagNode`] (same shape [`export`] itself /// produces), and every node that does splices in its winning - /// candidate's shape instead — a rewritten [`QueryExpr`] subtree, or a - /// bound `SummaryNode` subtree, rendered inline in the very same node + /// candidate's shape instead — a rewritten [`PreASAPNode`] subtree, or a + /// bound `PostASAPNode` subtree, rendered inline in the very same node /// list. `None` unless a higher layer explicitly built one (e.g. the /// `dag_export` devtools binary's `--post-asap` flag); omitted from the /// JSON entirely when absent, so every existing producer/consumer of @@ -285,8 +285,8 @@ pub struct WorkloadGraph { // ── Post-ASAP replacement export — a second, layering-seam-shaped feature ── // // Everything below this point is the post-ASAP counterpart of the pre-ASAP -// flattening above: [`export_summary`] flattens a `SummaryNode` the same way -// [`export`] flattens a `QueryExpr`, and [`TargetReplacement`] is the +// flattening above: [`export_summary`] flattens a `PostASAPNode` the same way +// [`export`] flattens a `PreASAPNode`, and [`TargetReplacement`] is the // generic, crate-agnostic "one replacement site, before and after" shape a // higher layer (`asap-aware-mapping`, via the `dag_export` devtools binary's // `--post-asap` flag) populates after running its own search — the exact @@ -300,28 +300,28 @@ pub struct WorkloadGraph { // // A single whole-query "post-ASAP tree" isn't attempted here, and isn't // representable in the current type system either: `SummaryExpr` has no -// variant letting a `SummaryNode` be embedded back inside a plain -// `QueryExpr`'s child slot (`QueryExpr`'s own children are always -// `Rc`, never `Rc`), so there is no way to splice a +// variant letting a `PostASAPNode` be embedded back inside a plain +// `PreASAPNode`'s child slot (`PreASAPNode`'s own children are always +// `Rc`, never `Rc`), so there is no way to splice a // post-ASAP binding back into its original pre-ASAP tree in place. Inventing // a bridge type for that is a real `asap_types`/`asap-aware-mapping` IR // design decision, well beyond what a devtools visualization export should // decide unilaterally. Instead, each independently-discovered replacement // target gets its own small, self-contained `before`/`after` pair — the -// target's own pre-ASAP subtree, and either the winning `SummaryNode` or the -// winning rewritten `QueryExpr`, both of which *are* fully representable +// target's own pre-ASAP subtree, and either the winning `PostASAPNode` or the +// winning rewritten `PreASAPNode`, both of which *are* fully representable // today via [`export`]/[`export_summary`] as-is. /// One flattened post-ASAP node — the [`SummaryExpr`] analogue of /// [`DagNode`]. `detail` holds this node's own scalar fields (the summarized /// column, the summary family, grouping strategy, sketch-query kind, …) — -/// everything except its `SummaryNode` children, which live in `children` +/// everything except its `PostASAPNode` children, which live in `children` /// instead. /// /// Unlike [`DagNode`], this carries no `hash`/`source_expr` pair: nothing in /// this module ever needs to re-identify a particular `SummaryDagNode` the /// way `DagNode::hash` lets a higher layer re-identify a pre-ASAP node (a -/// `SummaryNode` is always freshly exported for exactly one +/// `PostASAPNode` is always freshly exported for exactly one /// [`TargetReplacementAfter::Summary`] site, never matched back against a /// separately-exported graph the way pre-ASAP notes are). /// @@ -350,7 +350,7 @@ pub struct SummaryDagNode { /// `[outer, inner]`). pub children: Vec, /// The value's machine-readable accuracy guarantee (issue #172) — - /// [`SummaryNode::guarantee`] serialized structurally (metric, symbolic + /// [`PostASAPNode::guarantee`] serialized structurally (metric, symbolic /// bound, failure probability, provenance including any budget /// allocation), not as prose. Omitted when the node carries none (raw /// summary state, or a family with no error model), so every consumer @@ -378,21 +378,21 @@ pub struct TargetRejection { pub error: AccuracyError, } -/// One post-ASAP `SummaryNode` tree, flattened the same way [`DagGraph`] -/// flattens a pre-ASAP `QueryExpr` tree. +/// One post-ASAP `PostASAPNode` tree, flattened the same way [`DagGraph`] +/// flattens a pre-ASAP `PreASAPNode` tree. #[derive(Debug, Clone, Serialize)] pub struct SummaryDagGraph { pub nodes: Vec, pub root: u32, } -/// Flatten a [`SummaryNode`] the same way [`export`] flattens a `QueryExpr` +/// Flatten a [`PostASAPNode`] the same way [`export`] flattens a `PreASAPNode` /// — post-order, one [`SummaryDagNode`] per [`SummaryExpr`] variant, no -/// memoization of repeated `Rc` references (a shared +/// memoization of repeated `Rc` references (a shared /// sub-expression reachable through two parents is flattened twice, into two /// separate node entries — the same "this is a flattened tree view, not a /// pointer-identity-preserving graph" behavior [`build`] already has for -/// `QueryExpr`). +/// `PreASAPNode`). /// /// A `KeepPreAsap(inner)` leaf embeds the *whole* pre-ASAP subtree beneath it /// as a nested [`DagGraph`] (via [`export(inner)`](export)) inside its own @@ -402,7 +402,7 @@ pub struct SummaryDagGraph { /// `Vec` isn't type-safe; nesting is. `label` for a `KeepPreAsap` node is /// `format!("KeepPreAsap({kind})")`, where `kind` is the inner subtree's own /// top-level `DagNode::kind`. -pub fn export_summary(node: &SummaryNode) -> SummaryDagGraph { +pub fn export_summary(node: &PostASAPNode) -> SummaryDagGraph { let mut nodes = Vec::new(); let root = build_summary(node, &mut nodes); SummaryDagGraph { nodes, root } @@ -563,12 +563,12 @@ fn summary_shape(expr: &SummaryExpr) -> (&'static str, String, serde_json::Value } } -/// `expr`'s own `Rc` children, in the variant's field order +/// `expr`'s own `Rc` children, in the variant's field order /// (e.g. `SummaryJoin` is `[outer, inner]`) — empty for -/// [`SummaryExpr::KeepPreAsap`], which has no `SummaryNode` children at all -/// (only a boxed pre-ASAP `QueryExpr`). Shared by [`build_summary`] and +/// [`SummaryExpr::KeepPreAsap`], which has no `PostASAPNode` children at all +/// (only a boxed pre-ASAP `PreASAPNode`). Shared by [`build_summary`] and /// [`build_summary_hybrid`] for the same reason [`summary_shape`] is. -fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { +fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { match expr { SummaryExpr::KeepPreAsap(_) => vec![], SummaryExpr::BinaryOp { lhs, rhs, .. } => vec![lhs, rhs], @@ -587,8 +587,8 @@ fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { /// Recursively flatten `node`, appending [`SummaryDagNode`]s to `nodes` in /// post-order (children pushed before their parent), and return the pushed /// root's id. Exhaustive over every [`SummaryExpr`] variant, matching this -/// file's own exhaustive style for `QueryExpr` in [`build`]. -fn build_summary(node: &SummaryNode, nodes: &mut Vec) -> u32 { +/// file's own exhaustive style for `PreASAPNode` in [`build`]. +fn build_summary(node: &PostASAPNode, nodes: &mut Vec) -> u32 { if let SummaryExpr::KeepPreAsap(inner) = &node.expr { let pre_asap_subgraph = export(inner); let inner_kind = pre_asap_subgraph.nodes[pre_asap_subgraph.root as usize].kind; @@ -613,7 +613,7 @@ fn build_summary(node: &SummaryNode, nodes: &mut Vec) -> u32 { /// One replacement site a higher layer (the `dag_export` binary) found by /// running `asap_aware_mapping::replacement::search_workload_with` + -/// `CandidatePostASAPDAGs::cost_sorted` and picking the best-ranked candidate for one +/// `CandidateLogicalPostASAPDAGs::cost_sorted` and picking the best-ranked candidate for one /// `TargetSubDAGCandidates` — `asap_types` never runs that search itself (same layering /// rule as [`DagNote`]: this crate defines the shape, a higher crate /// populates it). @@ -640,7 +640,7 @@ pub struct TargetReplacement { /// not re-derived here). pub rationale: String, /// This candidate's rank among its `TargetSubDAGCandidates`'s alternatives after - /// `CandidatePostASAPDAGs::cost_sorted` (`0` = best). Exposed so a renderer can show + /// `CandidateLogicalPostASAPDAGs::cost_sorted` (`0` = best). Exposed so a renderer can show /// "this was the best of N candidates" without re-deriving the ranking. pub rank: usize, /// This candidate's own estimated cost, straight off @@ -689,7 +689,7 @@ pub enum TargetReplacementAfter { } /// Flatten `expr` into a [`DagGraph`]. -pub fn export(expr: &QueryExpr) -> DagGraph { +pub fn export(expr: &PreASAPNode) -> DagGraph { let mut nodes = Vec::new(); // One cache for the whole export — persisted across every `build`/ // `push_node` call, not reset per node, so `structural_hash` memoizes @@ -712,25 +712,25 @@ pub fn export(expr: &QueryExpr) -> DagGraph { /// merged post-ASAP graph via [`export_post_asap`] — see that function's own /// doc for the full design. `asap_types` has no opinion on *how* this is /// decided (that's `asap_aware_mapping::replacement::search_workload_with` + -/// `CandidatePostASAPDAGs::cost_sorted`'s job, a higher layer, exactly the layering rule +/// `CandidateLogicalPostASAPDAGs::cost_sorted`'s job, a higher layer, exactly the layering rule /// [`DagNode::notes`] already states); it only defines the shape a decision /// comes back in. #[derive(Debug, Clone)] -pub enum PostAsapSubstitution { +pub enum PostASAPSubstitution { /// This exact node has a winning `Replacement::Rewrite` — keep building /// from `.0` instead of the original node. Still pre-ASAP shaped, so /// [`build`] renders it via the same ordinary `DagNode` path — see /// [`build`]'s own doc for why `.0`'s own top level is rendered without /// re-querying `find_winner` on it (its descendants still are). Rewrite { - replacement: Rc, + replacement: Rc, decision: DagDecision, }, /// This exact node has a winning `Replacement::Summary` — switch to - /// rendering `.0`'s bound `SummaryNode` shape from here down, via + /// rendering `.0`'s bound `PostASAPNode` shape from here down, via /// [`build_summary_hybrid`]. Summary { - replacement: Rc, + replacement: Rc, decision: DagDecision, }, } @@ -745,12 +745,12 @@ pub enum PostAsapSubstitution { /// export" section doc for why *that* design doesn't attempt a single /// whole-query composite, and why this one can: this is a synthetic /// id/edge list, the same kind of thing [`DagGraph`] already is for the -/// pre-ASAP side, not a real `QueryExpr`/`SummaryNode` value with a type +/// pre-ASAP side, not a real `PreASAPNode`/`PostASAPNode` value with a type /// system to satisfy). /// /// `find_winner` is the whole layering seam: `asap_types` never runs /// `asap_aware_mapping::replacement::search_workload_with` or -/// `CandidatePostASAPDAGs::cost_sorted` itself, and has no idea what a `TargetSubDAGCandidates` or a +/// `CandidateLogicalPostASAPDAGs::cost_sorted` itself, and has no idea what a `TargetSubDAGCandidates` or a /// `ReplacementProvenance` is — it only asks, for one node at a time, "did a /// higher layer already decide something for you?" A caller (e.g. the /// `dag_export` devtools binary) builds this closure once per workload @@ -771,8 +771,8 @@ pub enum PostAsapSubstitution { /// for every registered strategy, not just the ones that happen not to /// return the target itself as a candidate. pub fn export_post_asap( - root: &QueryExpr, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, + root: &PreASAPNode, + find_winner: &mut dyn FnMut(&PreASAPNode) -> Option, ) -> DagGraph { let mut nodes = Vec::new(); let mut cache = HashCache::new(); @@ -819,23 +819,23 @@ macro_rules! define_query_kind_tags { #[cfg(test)] const QUERY_KIND_TAGS: &[&str] = &[$($tag),+]; - fn kind_tag(expr: &QueryExpr) -> &'static str { + fn kind_tag(expr: &PreASAPNode) -> &'static str { match expr { $($pattern => $tag),+, - other @ (QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. }) => unreachable!( - "kind_tag reached a scalar QueryExpr variant directly: {other:?}" + other @ (PreASAPNode::Column(_) + | PreASAPNode::Literal(_) + | PreASAPNode::Compare { .. } + | PreASAPNode::BoolAnd(_) + | PreASAPNode::BoolOr(_) + | PreASAPNode::Not(_) + | PreASAPNode::IsNull(_) + | PreASAPNode::IsNotNull(_) + | PreASAPNode::Cast { .. } + | PreASAPNode::InList { .. } + | PreASAPNode::FunctionCall { .. } + | PreASAPNode::Arithmetic { .. } + | PreASAPNode::Case { .. }) => unreachable!( + "kind_tag reached a scalar PreASAPNode variant directly: {other:?}" ), } } @@ -843,29 +843,29 @@ macro_rules! define_query_kind_tags { } define_query_kind_tags! { - QueryExpr::Scan { .. } => "Scan", - QueryExpr::PromqlScalarBridge(_) => "PromqlScalarBridge", - QueryExpr::EvalTimestamp => "EvalTimestamp", - QueryExpr::CurrentTimestamp => "CurrentTimestamp", - QueryExpr::PromqlVectorFromScalar(_) => "PromqlVectorFromScalar", - QueryExpr::PromqlScalarFromVector(_) => "PromqlScalarFromVector", - QueryExpr::PromqlRelabel { .. } => "PromqlRelabel", - QueryExpr::PromqlInfoEnrich { .. } => "PromqlInfoEnrich", - QueryExpr::PromqlSeriesSample { .. } => "PromqlSeriesSample", - QueryExpr::Filter { .. } => "Filter", - QueryExpr::Project { .. } => "Project", - QueryExpr::Aggregate { .. } => "Aggregate", - QueryExpr::Dedup { .. } => "Dedup", - QueryExpr::Concat { .. } => "Concat", - QueryExpr::Join { .. } => "Join", - QueryExpr::SetOp { .. } => "SetOp", - QueryExpr::Sort { .. } => "Sort", - QueryExpr::Limit { .. } => "Limit", - QueryExpr::PromqlSubquery { .. } => "PromqlSubquery", - QueryExpr::TimeRange { .. } => "TimeRange", - QueryExpr::TimeShift { .. } => "TimeShift", - QueryExpr::SQLWindowFunc { .. } => "SQLWindowFunc", - QueryExpr::BinaryOp { .. } => "BinaryOp", + PreASAPNode::Scan { .. } => "Scan", + PreASAPNode::PromqlScalarBridge(_) => "PromqlScalarBridge", + PreASAPNode::EvalTimestamp => "EvalTimestamp", + PreASAPNode::CurrentTimestamp => "CurrentTimestamp", + PreASAPNode::PromqlVectorFromScalar(_) => "PromqlVectorFromScalar", + PreASAPNode::PromqlScalarFromVector(_) => "PromqlScalarFromVector", + PreASAPNode::PromqlRelabel { .. } => "PromqlRelabel", + PreASAPNode::PromqlInfoEnrich { .. } => "PromqlInfoEnrich", + PreASAPNode::PromqlSeriesSample { .. } => "PromqlSeriesSample", + PreASAPNode::Filter { .. } => "Filter", + PreASAPNode::Project { .. } => "Project", + PreASAPNode::Aggregate { .. } => "Aggregate", + PreASAPNode::Dedup { .. } => "Dedup", + PreASAPNode::Concat { .. } => "Concat", + PreASAPNode::Join { .. } => "Join", + PreASAPNode::SetOp { .. } => "SetOp", + PreASAPNode::Sort { .. } => "Sort", + PreASAPNode::Limit { .. } => "Limit", + PreASAPNode::PromqlSubquery { .. } => "PromqlSubquery", + PreASAPNode::TimeRange { .. } => "TimeRange", + PreASAPNode::TimeShift { .. } => "TimeShift", + PreASAPNode::SQLWindowFunc { .. } => "SQLWindowFunc", + PreASAPNode::BinaryOp { .. } => "BinaryOp", } /// Push one flattened node for `expr`. `expr` is the *whole* subtree this @@ -877,7 +877,7 @@ define_query_kind_tags! { /// caller-supplied argument — see that function's doc for why. fn push_node( nodes: &mut Vec, - expr: &QueryExpr, + expr: &PreASAPNode, cache: &mut HashCache, label: String, detail: serde_json::Value, @@ -898,20 +898,20 @@ fn push_node( workload_node_id: None, hash, source_expr: Some(expr.clone()), - source_ptr: Some(expr as *const QueryExpr as usize), + source_ptr: Some(expr as *const PreASAPNode as usize), notes: Vec::new(), decision: None, }); id } -/// Push one flattened node with no corresponding pre-ASAP `QueryExpr` at +/// Push one flattened node with no corresponding pre-ASAP `PreASAPNode` at /// all — a post-ASAP-originated node inside [`export_post_asap`]'s merged /// graph (a `SummaryAgg`/`SummaryJoin`/… node, via [`build_summary_hybrid`]). /// `hash`/`source_expr`-based re-identification (see [`DagNode::hash`]'s own -/// doc) has no meaning for a node with no `QueryExpr` behind it, so this +/// doc) has no meaning for a node with no `PreASAPNode` behind it, so this /// pushes a fixed placeholder hash (`0`) and `source_expr: None` rather than -/// inventing a hash over `SummaryExpr` (which, unlike `QueryExpr`, has no +/// inventing a hash over `SummaryExpr` (which, unlike `PreASAPNode`, has no /// [`structural_hash`]-equivalent function at all — see [`SummaryDagNode`]'s /// own doc on why `SummaryExpr`'s fields don't even derive `Hash`/`PartialEq` /// consistently enough to build one). @@ -941,7 +941,7 @@ fn push_summary_originated_node( } /// The [`build_summary`]/[`build_summary_hybrid`] counterpart of [`build`] -/// for a bound [`SummaryNode`] reached while building +/// for a bound [`PostASAPNode`] reached while building /// [`export_post_asap`]'s merged graph: appends into the *same* `nodes: /// Vec` list `build` itself is filling, instead of a separate /// [`SummaryDagGraph`]. A `KeepPreAsap(inner)` leaf recurses back into @@ -952,10 +952,10 @@ fn push_summary_originated_node( /// wrapper (a nested aggregate a strategy independently found a /// replacement for, say) still gets spliced in correctly. fn build_summary_hybrid( - node: &SummaryNode, + node: &PostASAPNode, nodes: &mut Vec, cache: &mut HashCache, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, + find_winner: &mut dyn FnMut(&PreASAPNode) -> Option, ) -> u32 { if let SummaryExpr::KeepPreAsap(inner) = &node.expr { return build(inner, nodes, cache, find_winner); @@ -1000,7 +1000,7 @@ fn source_label(source: &Source) -> String { /// Recursively flatten `expr`, appending nodes to `nodes` in post-order /// (children pushed before their parent), and return the id of the pushed -/// root node. Exhaustive over every **operator** `QueryExpr` variant — a new +/// root node. Exhaustive over every **operator** `PreASAPNode` variant — a new /// one fails to compile here until this match is extended, matching the rest /// of the IR's exhaustive-match style (e.g. `output_schema`). The scalar /// variants (issue #205) are never passed to `build` directly: every operator @@ -1020,13 +1020,13 @@ fn source_label(source: &Source) -> String { /// through `build` again, so *they* still get a fresh `find_winner` call) /// rather than by looping back through this check a second time. fn build( - expr: &QueryExpr, + expr: &PreASAPNode, nodes: &mut Vec, cache: &mut HashCache, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, + find_winner: &mut dyn FnMut(&PreASAPNode) -> Option, ) -> u32 { match find_winner(expr) { - Some(PostAsapSubstitution::Rewrite { + Some(PostASAPSubstitution::Rewrite { replacement, decision, }) => { @@ -1045,7 +1045,7 @@ fn build( } return root; } - Some(PostAsapSubstitution::Summary { + Some(PostASAPSubstitution::Summary { replacement, decision, }) => { @@ -1070,19 +1070,19 @@ fn build( } /// The actual per-variant match [`build`] dispatches to once it has decided -/// (by consulting `find_winner` exactly once) which `QueryExpr` value to +/// (by consulting `find_winner` exactly once) which `PreASAPNode` value to /// render at this position — either `expr` itself (unchanged), or a winning /// `Replacement::Rewrite`'s own target. Every recursive call here goes back /// through [`build`] (not this function), so every child gets its own fresh /// `find_winner` query. fn build_no_recheck( - expr: &QueryExpr, + expr: &PreASAPNode, nodes: &mut Vec, cache: &mut HashCache, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, + find_winner: &mut dyn FnMut(&PreASAPNode) -> Option, ) -> u32 { match expr { - QueryExpr::Scan { + PreASAPNode::Scan { source, predicates, schema, @@ -1100,7 +1100,7 @@ fn build_no_recheck( // `detail` JSON, same as every other scalar-typed field // (`Filter.pred`, `Project.cols`, …) rather than pushing it as a // separate DAG node. - QueryExpr::PromqlScalarBridge(inner) => { + PreASAPNode::PromqlScalarBridge(inner) => { let detail = serde_json::json!({ "value": inner }); push_node( nodes, @@ -1111,7 +1111,7 @@ fn build_no_recheck( vec![], ) } - QueryExpr::EvalTimestamp => push_node( + PreASAPNode::EvalTimestamp => push_node( nodes, expr, cache, @@ -1119,7 +1119,7 @@ fn build_no_recheck( serde_json::json!({}), vec![], ), - QueryExpr::CurrentTimestamp => push_node( + PreASAPNode::CurrentTimestamp => push_node( nodes, expr, cache, @@ -1127,7 +1127,7 @@ fn build_no_recheck( serde_json::json!({}), vec![], ), - QueryExpr::PromqlVectorFromScalar(child) => { + PreASAPNode::PromqlVectorFromScalar(child) => { let c = build(child, nodes, cache, find_winner); push_node( nodes, @@ -1138,7 +1138,7 @@ fn build_no_recheck( vec![c], ) } - QueryExpr::PromqlScalarFromVector(child) => { + PreASAPNode::PromqlScalarFromVector(child) => { let c = build(child, nodes, cache, find_winner); push_node( nodes, @@ -1149,7 +1149,7 @@ fn build_no_recheck( vec![c], ) } - QueryExpr::PromqlRelabel { dst, value, child } => { + PreASAPNode::PromqlRelabel { dst, value, child } => { let c = build(child, nodes, cache, find_winner); let detail = serde_json::json!({ "dst": dst, "value": value }); push_node( @@ -1161,7 +1161,7 @@ fn build_no_recheck( vec![c], ) } - QueryExpr::PromqlInfoEnrich { selector, child } => { + PreASAPNode::PromqlInfoEnrich { selector, child } => { let c = build(child, nodes, cache, find_winner); let detail = serde_json::json!({ "selector": selector }); push_node( @@ -1173,7 +1173,7 @@ fn build_no_recheck( vec![c], ) } - QueryExpr::PromqlSeriesSample { by, kind, child } => { + PreASAPNode::PromqlSeriesSample { by, kind, child } => { let c = build(child, nodes, cache, find_winner); let detail = serde_json::json!({ "by": by, "kind": kind }); push_node( @@ -1185,12 +1185,12 @@ fn build_no_recheck( vec![c], ) } - QueryExpr::Filter { pred, child } => { + PreASAPNode::Filter { pred, child } => { let c = build(child, nodes, cache, find_winner); let detail = serde_json::json!({ "pred": pred }); push_node(nodes, expr, cache, "Filter".into(), detail, vec![c]) } - QueryExpr::Project { + PreASAPNode::Project { cols, qualifier, child, @@ -1206,7 +1206,7 @@ fn build_no_recheck( vec![c], ) } - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction, measures, output_names, @@ -1229,7 +1229,7 @@ fn build_no_recheck( vec![c], ) } - QueryExpr::Dedup { cols, child } => { + PreASAPNode::Dedup { cols, child } => { let c = build(child, nodes, cache, find_winner); let detail = serde_json::json!({ "cols": cols }); push_node( @@ -1241,7 +1241,7 @@ fn build_no_recheck( vec![c], ) } - QueryExpr::Concat { + PreASAPNode::Concat { children, discriminator_unique_key, } => { @@ -1254,7 +1254,7 @@ fn build_no_recheck( serde_json::json!({ "discriminator_unique_key": discriminator_unique_key }); push_node(nodes, expr, cache, label, detail, ids) } - QueryExpr::Join { + PreASAPNode::Join { kind, pred, left, @@ -1272,7 +1272,7 @@ fn build_no_recheck( vec![l, r], ) } - QueryExpr::SetOp { + PreASAPNode::SetOp { kind, all, left, @@ -1290,7 +1290,7 @@ fn build_no_recheck( vec![l, r], ) } - QueryExpr::Sort { + PreASAPNode::Sort { keys, partition_by, child, @@ -1306,12 +1306,12 @@ fn build_no_recheck( vec![c], ) } - QueryExpr::Limit { n, offset, child } => { + PreASAPNode::Limit { n, offset, child } => { let c = build(child, nodes, cache, find_winner); let detail = serde_json::json!({ "n": n, "offset": offset }); push_node(nodes, expr, cache, format!("Limit({n})"), detail, vec![c]) } - QueryExpr::PromqlSubquery { + PreASAPNode::PromqlSubquery { range, resolution, child, @@ -1320,7 +1320,7 @@ fn build_no_recheck( let detail = serde_json::json!({ "range": range, "resolution": resolution }); push_node(nodes, expr, cache, "PromqlSubquery".into(), detail, vec![c]) } - QueryExpr::TimeRange { range, child } => { + PreASAPNode::TimeRange { range, child } => { let c = build(child, nodes, cache, find_winner); let detail = serde_json::json!({ "range": range }); push_node( @@ -1332,12 +1332,12 @@ fn build_no_recheck( vec![c], ) } - QueryExpr::TimeShift { shift, child } => { + PreASAPNode::TimeShift { shift, child } => { let c = build(child, nodes, cache, find_winner); let detail = serde_json::json!({ "shift": shift }); push_node(nodes, expr, cache, "TimeShift".into(), detail, vec![c]) } - QueryExpr::SQLWindowFunc { + PreASAPNode::SQLWindowFunc { func, args, partition_by, @@ -1364,7 +1364,7 @@ fn build_no_recheck( vec![c], ) } - QueryExpr::BinaryOp { + PreASAPNode::BinaryOp { op, lhs, rhs, @@ -1382,20 +1382,22 @@ fn build_no_recheck( vec![l, r], ) } - other @ (QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. }) => { - unreachable!("dag_export::build reached a scalar QueryExpr variant directly: {other:?}") + other @ (PreASAPNode::Column(_) + | PreASAPNode::Literal(_) + | PreASAPNode::Compare { .. } + | PreASAPNode::BoolAnd(_) + | PreASAPNode::BoolOr(_) + | PreASAPNode::Not(_) + | PreASAPNode::IsNull(_) + | PreASAPNode::IsNotNull(_) + | PreASAPNode::Cast { .. } + | PreASAPNode::InList { .. } + | PreASAPNode::FunctionCall { .. } + | PreASAPNode::Arithmetic { .. } + | PreASAPNode::Case { .. }) => { + unreachable!( + "dag_export::build reached a scalar PreASAPNode variant directly: {other:?}" + ) } } } @@ -1411,8 +1413,8 @@ mod tests { use crate::pre_asap::schema::{Column, DataType, Schema}; use crate::types::AccuracyTarget; - fn scan(table: &str, columns: Vec) -> QueryExpr { - QueryExpr::Scan { + fn scan(table: &str, columns: Vec) -> PreASAPNode { + PreASAPNode::Scan { source: Source::Table { table_ref: table.into(), }, @@ -1472,16 +1474,16 @@ mod tests { // `export_post_asap` must merge them onto one node id. Sharing alone // is not physical cost evidence, so no edge cost may be fabricated. let shared_scan = Rc::new(scan("metrics", value_col())); - let left_branch = QueryExpr::Dedup { + let left_branch = PreASAPNode::Dedup { cols: vec![0], child: Rc::clone(&shared_scan), }; - let right_branch = QueryExpr::Limit { + let right_branch = PreASAPNode::Limit { n: 5, offset: 0, child: Rc::clone(&shared_scan), }; - let root = QueryExpr::Concat { + let root = PreASAPNode::Concat { children: vec![left_branch, right_branch], discriminator_unique_key: None, }; @@ -1503,9 +1505,9 @@ mod tests { #[test] fn a_single_parent_referencing_a_shared_child_twice_is_one_consumer_not_two() { let shared_scan = Rc::new(scan("metrics", value_col())); - let root = QueryExpr::Join { + let root = PreASAPNode::Join { kind: crate::pre_asap::query_expr::JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(true)))), left: Rc::clone(&shared_scan), right: Rc::clone(&shared_scan), }; @@ -1530,9 +1532,9 @@ mod tests { // pointer identity — even a workload-level shared subtree renders as // two independent tree nodes here, so there is nothing to annotate. let shared_scan = Rc::new(scan("metrics", value_col())); - let root = QueryExpr::Join { + let root = PreASAPNode::Join { kind: crate::pre_asap::query_expr::JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(true)))), left: Rc::clone(&shared_scan), right: Rc::clone(&shared_scan), }; @@ -1543,9 +1545,9 @@ mod tests { #[test] fn chain_preserves_shape_and_child_links() { - let expr = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(QueryExpr::Aggregate { + let expr = PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(true)))), + child: Rc::new(PreASAPNode::Aggregate { reduction: Reduction::Reduce(GroupKeys::none()), measures: vec![AggIntent::Count { accuracy: AccuracyTarget::Exact, @@ -1573,7 +1575,7 @@ mod tests { #[test] fn merge_keeps_every_branch_as_a_child() { - let expr = QueryExpr::concat(vec![ + let expr = PreASAPNode::concat(vec![ scan("a", value_col()), scan("b", value_col()), scan("c", value_col()), @@ -1614,12 +1616,12 @@ mod tests { // though it's embedded at different depths / under different parents. let shared_shape = || scan("metrics", value_col()); - let q1 = QueryExpr::Limit { + let q1 = PreASAPNode::Limit { n: 10, offset: 0, child: Rc::new(shared_shape()), }; - let q2 = QueryExpr::Dedup { + let q2 = PreASAPNode::Dedup { cols: vec![0], child: Rc::new(shared_shape()), }; @@ -1660,9 +1662,9 @@ mod tests { fn every_node_hash_matches_cse_structural_hash_on_its_own_subtree() { // A multi-level tree: check the parity holds at every depth, not // just the root — each `DagNode::hash` must equal - // `structural_hash` applied to the actual `QueryExpr` subtree that + // `structural_hash` applied to the actual `PreASAPNode` subtree that // node represents. - let agg = QueryExpr::Aggregate { + let agg = PreASAPNode::Aggregate { reduction: Reduction::Reduce(GroupKeys::none()), measures: vec![AggIntent::Count { accuracy: AccuracyTarget::Exact, @@ -1671,8 +1673,8 @@ mod tests { having: None, child: Rc::new(scan("metrics", value_col())), }; - let root = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + let root = PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(true)))), child: Rc::new(agg.clone()), }; @@ -1704,7 +1706,7 @@ mod tests { SummaryFamilyType, SummarySchema, }; let leaf = Rc::new(scan("t", vec![Column::new("v", DataType::Float64, false)])); - let kept = Rc::new(SummaryNode { + let kept = Rc::new(PostASAPNode { expr: SummaryExpr::KeepPreAsap(Rc::clone(&leaf)), schema: SummarySchema { fields: vec![], @@ -1712,7 +1714,7 @@ mod tests { }, guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), }); - let agg = Rc::new(SummaryNode { + let agg = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child: kept, family: SummaryFamilyType::Sketch( @@ -1756,7 +1758,7 @@ mod tests { }, ], }; - let root = SummaryNode { + let root = PostASAPNode { expr: SummaryExpr::SummaryEstimate { summary_input: agg, query: SketchQuery::Quantile { q: 0.99 }, diff --git a/crates/types/src/parsed_workload.rs b/crates/types/src/parsed_workload.rs index ff955e6a..21b73b34 100644 --- a/crates/types/src/parsed_workload.rs +++ b/crates/types/src/parsed_workload.rs @@ -10,7 +10,7 @@ use std::rc::Rc; -use crate::pre_asap::query_expr::QueryExpr; +use crate::pre_asap::query_expr::PreASAPNode; use crate::workload::{ DataWorkload, PlanningWorkload, QueryWorkload, QueryWorkloadEntry, WorkloadError, }; @@ -33,7 +33,7 @@ pub enum ParsedWorkloadError { #[derive(Debug, Clone)] pub struct ParsedWorkload { workload: PlanningWorkload, - exprs: Vec>, + exprs: Vec>, } impl ParsedWorkload { @@ -41,7 +41,7 @@ impl ParsedWorkload { /// `i`-th entry. pub fn new( workload: PlanningWorkload, - exprs: Vec>, + exprs: Vec>, ) -> Result { let entries = workload.query_workload.entries().count(); if entries != exprs.len() { @@ -65,7 +65,7 @@ impl ParsedWorkload { self.workload.data_workload.as_ref() } - pub fn exprs(&self) -> &[Rc] { + pub fn exprs(&self) -> &[Rc] { &self.exprs } @@ -78,7 +78,7 @@ impl ParsedWorkload { } /// Normalized entries paired with their lowered expression. - pub fn entries(&self) -> impl Iterator)> + '_ { + pub fn entries(&self) -> impl Iterator)> + '_ { self.workload .query_workload .entries() diff --git a/crates/types/src/post_asap/cse.rs b/crates/types/src/post_asap/cse.rs index 6d1d0dc9..6dc106a1 100644 --- a/crates/types/src/post_asap/cse.rs +++ b/crates/types/src/post_asap/cse.rs @@ -7,7 +7,7 @@ use std::collections::HashMap; use std::rc::Rc; -use super::{SummaryExpr, SummaryNode}; +use super::{PostASAPNode, SummaryExpr}; /// Numeric PartialEq alone conflates signed zeros. The serialized check is /// additional evidence, never a replacement for typed equality (JSON maps @@ -22,7 +22,7 @@ fn same_value(left: &T, right: &T) -> bool { /// Children have already been interned. Comparing their identities avoids /// recursively expanding a shared DAG once for every path to each descendant. -fn same_node(left: &SummaryNode, right: &SummaryNode) -> bool { +fn same_node(left: &PostASAPNode, right: &PostASAPNode) -> bool { use SummaryExpr::*; let expression_equal = match (&left.expr, &right.expr) { (KeepPreAsap(a), KeepPreAsap(b)) => Rc::ptr_eq(a, b) || same_value(a, b), @@ -169,13 +169,13 @@ fn same_node(left: &SummaryNode, right: &SummaryNode) -> bool { /// maintenance scope. Downstream realization must still check physical /// implementation compatibility. Use separate calls for independent executions. pub fn share_common_summary_subtrees( - roots: Vec<(Id, Rc)>, -) -> Vec<(Id, Rc)> { + roots: Vec<(Id, Rc)>, +) -> Vec<(Id, Rc)> { fn visit( - node: &Rc, - seen: &mut HashMap>, - pool: &mut Vec>, - ) -> Rc { + node: &Rc, + seen: &mut HashMap>, + pool: &mut Vec>, + ) -> Rc { let identity = Rc::as_ptr(node) as usize; if let Some(node) = seen.get(&identity) { return Rc::clone(node); @@ -235,11 +235,11 @@ pub fn share_common_summary_subtrees( mod tests { use super::*; use crate::post_asap::{ResultGuarantee, SummarySchema}; - use crate::pre_asap::{QueryExpr, ScalarValue}; + use crate::pre_asap::{PreASAPNode, ScalarValue}; - fn leaf(value: f64) -> Rc { - Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(QueryExpr::Literal(ScalarValue::Float64( + fn leaf(value: f64) -> Rc { + Rc::new(PostASAPNode { + expr: SummaryExpr::KeepPreAsap(Rc::new(PreASAPNode::Literal(ScalarValue::Float64( value, )))), schema: SummarySchema { @@ -262,7 +262,7 @@ mod tests { // A diamond is retained across the returned roots, not copied per consumer. #[test] fn shares_children_across_distinct_roots() { - let merge = Rc::new(SummaryNode { + let merge = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryMerge { timing: crate::post_asap::ExecutionTiming::IngestionTime, children: vec![leaf(1.0), leaf(2.0)], @@ -307,7 +307,7 @@ mod tests { let SummaryExpr::KeepPreAsap(expr) = &root.expr else { panic!() }; - let QueryExpr::Literal(ScalarValue::Float64(actual)) = expr.as_ref() else { + let PreASAPNode::Literal(ScalarValue::Float64(actual)) = expr.as_ref() else { panic!() }; assert_eq!(actual.to_bits(), expected.to_bits()); @@ -321,8 +321,8 @@ mod tests { #[test] fn nested_values_and_nonfinite_values_remain_distinct() { let wrapped = |value| { - Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(QueryExpr::promql_scalar(value))), + Rc::new(PostASAPNode { + expr: SummaryExpr::KeepPreAsap(Rc::new(PreASAPNode::promql_scalar(value))), ..leaf(1.0).as_ref().clone() }) }; @@ -349,8 +349,8 @@ mod tests { SummaryFamilyType, SummaryUpdate, }; use crate::pre_asap::{ColumnRef, Reduction}; - fn readout(q: f64, alpha: f64) -> Rc { - let producer = Rc::new(SummaryNode { + fn readout(q: f64, alpha: f64) -> Rc { + let producer = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child: leaf(1.0), family: SummaryFamilyType::Sketch( @@ -370,7 +370,7 @@ mod tests { }, guarantee: None, }); - Rc::new(SummaryNode { + Rc::new(PostASAPNode { expr: SummaryExpr::SummaryEstimate { summary_input: producer, query: SketchQuery::Quantile { q }, @@ -387,7 +387,7 @@ mod tests { ("p99", readout(0.99, 0.01)), ("strict", readout(0.95, 0.001)), ]); - let producer = |root: &Rc| match &root.expr { + let producer = |root: &Rc| match &root.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => Rc::clone(summary_input), _ => panic!(), }; @@ -402,10 +402,10 @@ mod tests { fn shared_diamond_does_not_expand_during_comparison() { let (done, completion) = std::sync::mpsc::channel(); let worker = std::thread::spawn(move || { - fn diamond() -> Rc { + fn diamond() -> Rc { let mut current = leaf(1.0); for _ in 0..24 { - current = Rc::new(SummaryNode { + current = Rc::new(PostASAPNode { expr: SummaryExpr::BinaryOp { timing: super::super::ExecutionTiming::QueryTime, lhs: Rc::clone(¤t), diff --git a/crates/types/src/post_asap/execution_data_state.rs b/crates/types/src/post_asap/execution_data_state.rs index 696c84e5..cfb0c090 100644 --- a/crates/types/src/post_asap/execution_data_state.rs +++ b/crates/types/src/post_asap/execution_data_state.rs @@ -37,7 +37,7 @@ //! — every existing consumer pattern-matches the one-field shape — so its //! data_state is *assigned* by [`validate_execution_data_states`] from the edge that //! reaches it and reported in the returned [`ExecutionDataStateAssignment`]. What it may -//! not do is stay ambiguous inside one mixed plan: the same `Rc` +//! not do is stay ambiguous inside one mixed plan: the same `Rc` //! reached once as update input and once as query-time fallback is //! [`ExecutionDataStateError::AmbiguousKeepPreAsap`], because no single execution of that //! subtree can serve both roles. @@ -47,9 +47,9 @@ use std::rc::Rc; use thiserror::Error; -use super::expr::{ExactOperation, SummaryExpr, SummaryNode, ValueOperation}; +use super::expr::{ExactOperation, PostASAPNode, SummaryExpr, ValueOperation}; use super::schema::{SummaryFamilyType, SummaryField, SummarySchema}; -use crate::pre_asap::query_expr::{aggregate_output_schema, QueryExprError}; +use crate::pre_asap::query_expr::{aggregate_output_schema, PreASAPNodeError}; use crate::pre_asap::schema::{Column, Schema}; /// When a post-ASAP value is produced. @@ -209,30 +209,30 @@ pub enum ExecutionDataStateError { } /// The data_state assigned to every node of a validated plan, keyed by -/// `Rc` pointer identity — the explicit per-node "execution_data_state" a +/// `Rc` pointer identity — the explicit per-node "execution_data_state" a /// runtime or a DAG export reads instead of re-deriving it. For every /// non-`KeepPreAsap` node this equals [`produced_data_state`]; for a /// `KeepPreAsap` leaf it is the data_state the reaching edge assigned. #[derive(Debug, Clone, Default)] pub struct ExecutionDataStateAssignment { - domains: HashMap<*const SummaryNode, ExecutionDataState>, + domains: HashMap<*const PostASAPNode, ExecutionDataState>, } impl ExecutionDataStateAssignment { /// The data_state assigned to `node`, if it was part of the validated plan. - pub fn data_state_of(&self, node: &Rc) -> Option { + pub fn data_state_of(&self, node: &Rc) -> Option { self.domains.get(&Rc::as_ptr(node)).copied() } /// The data_state assigned to the node at `ptr` — for callers walking a plan /// by reference rather than by `Rc`. - pub fn data_state_of_ptr(&self, ptr: *const SummaryNode) -> Option { + pub fn data_state_of_ptr(&self, ptr: *const PostASAPNode) -> Option { self.domains.get(&ptr).copied() } } /// Initial layout proposed by semantic realization, not a restriction on physical -/// operator placement. `PostAsapDag::with_execution_phases` assigns the final +/// operator placement. `LogicalPostASAPDAGTransport::with_execution_phases` assigns the final /// phase independently of payload kind. Returns `None` for /// [`SummaryExpr::KeepPreAsap`], whose data_state is assigned by the edge reaching /// it (see the module docs). @@ -282,11 +282,11 @@ fn is_exact_accumulator_state(schema: &SummarySchema) -> Result<(), ExecutionDat /// Validate every edge of the DAG rooted at `root` against the module-level /// rules, returning each node's assigned data_state on success. Shared -/// `Rc`s are visited once per reaching edge (the assignment is +/// `Rc`s are visited once per reaching edge (the assignment is /// per node, so a conflict between two edges is what /// [`ExecutionDataStateError::AmbiguousKeepPreAsap`] detects). pub fn validate_execution_data_states( - root: &Rc, + root: &Rc, ) -> Result { // The root may be a readable value or bare maintained state (a // deployment may hand an `ExactAggregate` accumulator straight to a @@ -307,7 +307,7 @@ pub fn validate_execution_data_states( /// legal update-path input. Validates every edge beneath `root` exactly /// as the whole-plan entry point does. pub fn validate_execution_data_states_at( - root: &Rc, + root: &Rc, data_state: ExecutionDataState, ) -> Result { let mut assignment = ExecutionDataStateAssignment::default(); @@ -318,7 +318,7 @@ pub fn validate_execution_data_states_at( /// The source rows whose series a maintenance operand has one row for: a /// finalized per-series Sum or Count of those rows, or aligned arithmetic of /// operands over the same rows. Each emits exactly the series with a sample. -fn per_series_rows(node: &SummaryNode) -> Option<&crate::pre_asap::QueryExpr> { +fn per_series_rows(node: &PostASAPNode) -> Option<&crate::pre_asap::PreASAPNode> { use crate::post_asap::ExactKind; match &node.expr { SummaryExpr::ValueOperation { @@ -353,7 +353,7 @@ fn per_series_rows(node: &SummaryNode) -> Option<&crate::pre_asap::QueryExpr> { /// Record `data_state` for `node` (detecting a conflicting earlier assignment /// for a `KeepPreAsap`), then check and recurse into every child edge. fn visit( - node: &Rc, + node: &Rc, data_state: ExecutionDataState, assignment: &mut ExecutionDataStateAssignment, ) -> Result<(), ExecutionDataStateError> { @@ -589,7 +589,7 @@ fn visit( /// under a state-only edge). For DAG export and other reporting that needs /// an explicit per-node data_state even on a plan that /// [`validate_execution_data_states`] would reject. -pub fn assigned_child_data_state(parent: &SummaryExpr, child: &SummaryNode) -> ExecutionDataState { +pub fn assigned_child_data_state(parent: &SummaryExpr, child: &PostASAPNode) -> ExecutionDataState { if let Some(avail) = produced_data_state(&child.expr) { return avail; } @@ -625,7 +625,7 @@ pub fn assigned_child_data_state(parent: &SummaryExpr, child: &SummaryNode) -> E /// (checked via `accept`), or — for a `KeepPreAsap` leaf — the data_state the /// edge assigns it, derived from what that edge accepts. fn child_domain( - child: &Rc, + child: &Rc, edge: ExecutionDataStateEdge, accept: impl Fn(ExecutionDataState) -> Result<(), ExecutionDataStateError>, ) -> Result { @@ -660,7 +660,7 @@ fn child_domain( } fn state_only( - child: &Rc, + child: &Rc, edge: ExecutionDataStateEdge, ) -> Result { child_domain(child, edge, |avail| match avail { @@ -815,7 +815,7 @@ pub enum ExactOperationSchemaError { #[error("exact operator input carries summary state, not plain columns")] NonPlainInput, #[error("schema derivation failed: {0}")] - Schema(#[from] QueryExprError), + Schema(#[from] PreASAPNodeError), } #[cfg(test)] @@ -824,7 +824,7 @@ mod tests { use crate::post_asap::{ExactKind, ExactParams, GroupingStrategy, SketchQuery}; use crate::pre_asap::agg_intent::AggIntent; use crate::pre_asap::expr_ir::ColumnRef; - use crate::pre_asap::query_expr::{QueryExpr, Reduction, Source}; + use crate::pre_asap::query_expr::{PreASAPNode, Reduction, Source}; use crate::pre_asap::schema::DataType; /// Both execution phases use raw values, distinct from maintained state. @@ -843,8 +843,8 @@ mod tests { assert_eq!(DataPrimitive::SummaryState.as_str(), "summary_state"); } - fn scan() -> Rc { - Rc::new(QueryExpr::Scan { + fn scan() -> Rc { + Rc::new(PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -859,10 +859,10 @@ mod tests { }) } - fn keep() -> Rc { + fn keep() -> Rc { let s = scan(); let schema = lift_plain(&s.output_schema().unwrap()); - Rc::new(SummaryNode { + Rc::new(PostASAPNode { expr: SummaryExpr::KeepPreAsap(s), schema, guarantee: None, @@ -883,8 +883,8 @@ mod tests { } } - fn agg(child: Rc, family: SummaryFamilyType) -> Rc { - Rc::new(SummaryNode { + fn agg(child: Rc, family: SummaryFamilyType) -> Rc { + Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child, family: family.clone(), @@ -912,8 +912,8 @@ mod tests { ) } - fn estimate(child: Rc) -> Rc { - Rc::new(SummaryNode { + fn estimate(child: Rc) -> Rc { + Rc::new(PostASAPNode { expr: SummaryExpr::SummaryEstimate { summary_input: child, query: SketchQuery::Quantile { q: 0.99 }, @@ -955,7 +955,7 @@ mod tests { let identity = crate::pre_asap::schema::PROMQL_SERIES_IDENTITY; // A finalized per-series Sum of `metric`'s rows. let operand = |metric: &str, schema: &SummarySchema| { - let rows = Rc::new(QueryExpr::Scan { + let rows = Rc::new(PreASAPNode::Scan { source: Source::TimeSeries { metric: metric.into(), }, @@ -972,9 +972,9 @@ mod tests { let mut state = schema.clone(); state.fields[0].dtype = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let sum = Rc::new(SummaryNode { + let sum = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { - child: Rc::new(SummaryNode { + child: Rc::new(PostASAPNode { expr: SummaryExpr::KeepPreAsap(rows), schema: schema.clone(), guarantee: None, @@ -987,7 +987,7 @@ mod tests { schema: state, guarantee: None, }); - Rc::new(SummaryNode { + Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: sum, operation: ValueOperation::FinalizeExactAccumulator, @@ -998,7 +998,7 @@ mod tests { }) }; let validate = |schema: SummarySchema, rhs: &str| { - let binary = Rc::new(SummaryNode { + let binary = Rc::new(PostASAPNode { expr: SummaryExpr::BinaryOp { lhs: operand("m", &schema), rhs: operand(rhs, &schema), @@ -1078,7 +1078,7 @@ mod tests { #[test] fn query_time_operation_over_readout_is_legal_and_root_is_readout() { let inner = estimate(agg(keep(), kll())); - let root = Rc::new(SummaryNode { + let root = Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: inner, operation: ValueOperation::Exact(max_op()), @@ -1097,7 +1097,7 @@ mod tests { #[test] fn non_exact_operator_uses_the_same_read_domain_contract() { let inner = estimate(agg(keep(), kll())); - let root = Rc::new(SummaryNode { + let root = Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: inner, operation: ValueOperation::Extension { @@ -1119,7 +1119,7 @@ mod tests { #[test] fn query_time_values_can_feed_query_time_summary_construction() { let inner = estimate(agg(keep(), kll())); - let post = Rc::new(SummaryNode { + let post = Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: inner, operation: ValueOperation::Exact(max_op()), @@ -1135,7 +1135,7 @@ mod tests { #[test] fn function_under_summary_agg_is_legal_but_not_at_root() { - let operation = Rc::new(SummaryNode { + let operation = Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: keep(), operation: ValueOperation::Exact(max_op()), @@ -1159,7 +1159,7 @@ mod tests { #[test] fn function_over_readout_is_rejected() { let inner = estimate(agg(keep(), kll())); - let operation = Rc::new(SummaryNode { + let operation = Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: inner, operation: ValueOperation::Exact(max_op()), @@ -1199,7 +1199,7 @@ mod tests { fn summary_merge_runs_at_ingestion_or_query_time() { for timing in [ExecutionTiming::IngestionTime, ExecutionTiming::QueryTime] { let input = agg(keep(), kll()); - let merged = Rc::new(SummaryNode { + let merged = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryMerge { children: vec![input.clone()], timing, @@ -1216,9 +1216,9 @@ mod tests { primitive: DataPrimitive::SummaryState, }) ); - let exported = crate::post_asap::compile_post_asap_dag(&root).unwrap(); + let exported = crate::post_asap::export_post_asap_dag(&root).unwrap(); assert!(exported.nodes.iter().any(|node| matches!(node.payload, - crate::post_asap::PostAsapOperatorPayload::SummaryMerge + crate::post_asap::PostASAPOperatorPayload::SummaryMerge if node.output_state.timing == timing))); } } @@ -1226,7 +1226,7 @@ mod tests { #[test] fn ingestion_merge_cannot_depend_on_query_execution() { let input = agg(keep(), kll()); - let query_merge = Rc::new(SummaryNode { + let query_merge = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryMerge { children: vec![input.clone()], timing: ExecutionTiming::QueryTime, @@ -1234,7 +1234,7 @@ mod tests { schema: input.schema.clone(), guarantee: None, }); - let ingestion_merge = Rc::new(SummaryNode { + let ingestion_merge = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryMerge { children: vec![query_merge], timing: ExecutionTiming::IngestionTime, @@ -1252,7 +1252,7 @@ mod tests { // execution can serve both, so the plan is rejected. let shared = keep(); let maintained = estimate(agg(Rc::clone(&shared), kll())); - let post_over_raw = Rc::new(SummaryNode { + let post_over_raw = Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: Rc::clone(&shared), operation: ValueOperation::Exact(max_op()), @@ -1261,11 +1261,11 @@ mod tests { schema: plain(&["max"]), guarantee: None, }); - let root = Rc::new(SummaryNode { + let root = Rc::new(PostASAPNode { expr: SummaryExpr::SummaryMerge { timing: ExecutionTiming::IngestionTime, children: vec![ - Rc::new(SummaryNode { + Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: maintained, operation: ValueOperation::Exact(max_op()), diff --git a/crates/types/src/post_asap/expr.rs b/crates/types/src/post_asap/expr.rs index 12e2c21c..a5d1efdb 100644 --- a/crates/types/src/post_asap/expr.rs +++ b/crates/types/src/post_asap/expr.rs @@ -7,7 +7,7 @@ use super::sketch::{GroupingStrategy, SketchQuery, SummaryUpdate}; use crate::pre_asap::agg_intent::AggIntent; use crate::pre_asap::query_expr::Predicate; use crate::pre_asap::{ - BinaryOpKind, ColumnRef, GroupKeys, JoinKind, ProjectItem, QueryExpr, Reduction, SortKey, + BinaryOpKind, ColumnRef, GroupKeys, JoinKind, PreASAPNode, ProjectItem, Reduction, SortKey, VectorMatch, }; @@ -91,7 +91,7 @@ pub enum CandidateCompleteness { /// summary-state-typed columns (`SummaryFamilyType`'s non-`Plain` variants); /// the pre-ASAP `Schema` cannot. #[derive(Debug, Clone, PartialEq)] -pub struct SummaryNode { +pub struct PostASAPNode { pub expr: SummaryExpr, /// Output schema of `expr` — the schema of the data flowing on the edge /// leading *from* this node to its parent(s). @@ -114,34 +114,34 @@ pub struct SummaryNode { /// Sketch-bound IR produced by post-ASAP binding and final selection. Binding /// rules selectively replace logical aggregates and joins in the pre-ASAP -/// `QueryExpr` with summary-bound counterparts. Final selection can retain +/// `PreASAPNode` with summary-bound counterparts. Final selection can retain /// supported read-time value operations around independently planned children; -/// other unsupported subtrees pass through as `KeepPreAsap(Rc)`. +/// other unsupported subtrees pass through as `KeepPreAsap(Rc)`. /// /// Traversing from the root node yields a DAG; shared sub-expressions appear -/// as multiple `Rc` references to the same `SummaryNode`. +/// as multiple `Rc` references to the same `PostASAPNode`. #[derive(Debug, Clone, PartialEq)] pub enum SummaryExpr { /// A pre-ASAP subtree kept as-is because it has no selected implementation /// or supported residual decomposition. Output schema is the inner node's /// schema, lifted to `SummarySchema` with all fields as /// `SummaryFamilyType::Plain`. - KeepPreAsap(Rc), + KeepPreAsap(Rc), /// A PromQL binary operation whose operands were planned independently. /// This keeps realizable summary/readout leaves visible instead of /// hiding the complete expression inside `KeepPreAsap`. BinaryOp { timing: ExecutionTiming, - lhs: Rc, - rhs: Rc, + lhs: Rc, + rhs: Rc, operator: BinaryOperator, }, /// Plain-row semantics composed with a post-ASAP child. Timing is an /// independent physical choice, not part of the operation's identity. ValueOperation { - child: Rc, + child: Rc, operation: ValueOperation, timing: super::execution_data_state::ExecutionTiming, }, @@ -150,8 +150,8 @@ pub enum SummaryExpr { /// distinct from [`SummaryJoin`](Self::SummaryJoin), which combines /// summary states for join estimation during maintenance. RelationalJoin { - left: Rc, - right: Rc, + left: Rc, + right: Rc, kind: JoinKind, pred: Predicate, /// Optional proof for candidate pruning; ranking remains a separate operation. @@ -165,7 +165,7 @@ pub enum SummaryExpr { /// Output schema: grouping columns (verbatim) + one field carrying /// partial summary state per group, typed `family`. SummaryAgg { - child: Rc, + child: Rc, /// Which summary family realizes this aggregation, and that /// family's own `(kind, params)`. Never `SummaryFamilyType::Plain` /// — this node always produces summary state, not a plain value. @@ -206,8 +206,8 @@ pub enum SummaryExpr { /// Output schema: one field typed `family`, read by a downstream /// `SummaryEstimate`. SummaryJoin { - outer: Rc, - inner: Rc, + outer: Rc, + inner: Rc, key: ColumnRef, /// Never `SummaryFamilyType::Plain` — see [`SummaryAgg::family`](SummaryExpr::SummaryAgg). family: SummaryFamilyType, @@ -218,15 +218,15 @@ pub enum SummaryExpr { /// `subtractable` must be true for the family. /// Output schema: one field (same family + params as inputs). SummarySubtract { - left: Rc, - right: Rc, + left: Rc, + right: Rc, }, /// Delete a key from a summary (CMS update with −1, deletable Bloom /// filter). Catalog flag `deletable` must be true. Output schema = /// input schema unchanged in type (same field type as input). SummaryDelete { - summary_input: Rc, + summary_input: Rc, key: ColumnRef, }, @@ -235,7 +235,7 @@ pub enum SummaryExpr { /// schema is a regular row-shaped schema (Float64 for quantile, Int64 /// for count/cardinality, `[(key, count)]` for top-k). SummaryEstimate { - summary_input: Rc, + summary_input: Rc, query: SketchQuery, }, @@ -246,7 +246,7 @@ pub enum SummaryExpr { /// allocator (not modeled in this crate) on cut edges. /// Output schema: one field (same family + params as inputs). SummaryMerge { - children: Vec>, + children: Vec>, timing: ExecutionTiming, }, } diff --git a/crates/types/src/post_asap/guarantee.rs b/crates/types/src/post_asap/guarantee.rs index 88e070ce..d3cdc8f6 100644 --- a/crates/types/src/post_asap/guarantee.rs +++ b/crates/types/src/post_asap/guarantee.rs @@ -16,7 +16,7 @@ //! ## What a guarantee says //! //! [`ResultGuarantee`] is attached to a finalized, caller-visible value — -//! [`super::SummaryNode::guarantee`] on a `SummaryEstimate` readout, an +//! [`super::PostASAPNode::guarantee`] on a `SummaryEstimate` readout, an //! exact accumulator, or a kept pre-ASAP subtree — never to raw summary //! state (a `SummaryAgg` sketch node carries `None`; its readout carries the //! guarantee). Its statement is: diff --git a/crates/types/src/post_asap/maintained_population.rs b/crates/types/src/post_asap/maintained_population.rs index 3939c7c1..181c4983 100644 --- a/crates/types/src/post_asap/maintained_population.rs +++ b/crates/types/src/post_asap/maintained_population.rs @@ -37,23 +37,23 @@ pub enum PopulationReadout { impl CurrentSeriesInput { /// Verify the named contract against the canonical maintenance input. - pub fn matches_input(&self, input: &crate::pre_asap::QueryExpr) -> bool { - use crate::pre_asap::{CompareOpKind, DataType, QueryExpr, ScalarValue, Source}; + pub fn matches_input(&self, input: &crate::pre_asap::PreASAPNode) -> bool { + use crate::pre_asap::{CompareOpKind, DataType, PreASAPNode, ScalarValue, Source}; // PromQL instant selectors carry an ingestion-interval `TimeRange` as // their input scope. The population must use the same expiry horizon; // shifted and otherwise transformed inputs still fail below. let input = match input { - QueryExpr::TimeRange { range, child } + PreASAPNode::TimeRange { range, child } if self.lookback_ms > 0 && *range == std::time::Duration::from_millis(self.lookback_ms) => { child.as_ref() } - QueryExpr::TimeRange { .. } => return false, + PreASAPNode::TimeRange { .. } => return false, other if self.lookback_ms == 300_000 => other, _ => return false, }; - let QueryExpr::Scan { + let PreASAPNode::Scan { source: Source::TimeSeries { metric }, predicates, schema, @@ -78,10 +78,10 @@ impl CurrentSeriesInput { } let mut matchers = Vec::new(); for predicate in predicates { - let QueryExpr::Compare { left, op, right } = predicate.0.as_ref() else { + let PreASAPNode::Compare { left, op, right } = predicate.0.as_ref() else { return false; }; - let (QueryExpr::Column(col), QueryExpr::Literal(ScalarValue::Utf8(value))) = + let (PreASAPNode::Column(col), PreASAPNode::Literal(ScalarValue::Utf8(value))) = (left.as_ref(), right.as_ref()) else { return false; @@ -117,7 +117,7 @@ impl CurrentSeriesInput { pub enum PopulationInput { CurrentSeries(CurrentSeriesInput), Rows { - input: std::rc::Rc, + input: std::rc::Rc, value_column: usize, grouping: crate::pre_asap::GroupKeys, }, @@ -131,7 +131,7 @@ pub struct MaintainedPopulation { } impl MaintainedPopulation { - pub fn matches_input(&self, input: &crate::pre_asap::QueryExpr) -> bool { + pub fn matches_input(&self, input: &crate::pre_asap::PreASAPNode) -> bool { match &self.input { PopulationInput::CurrentSeries(spec) => spec.matches_input(input), PopulationInput::Rows { @@ -139,9 +139,9 @@ impl MaintainedPopulation { value_column, grouping, } => { - use crate::pre_asap::{DataType, QueryExpr, Source}; + use crate::pre_asap::{DataType, PreASAPNode, Source}; expected.as_ref() == input - && matches!(input, QueryExpr::Scan { source: Source::Table { .. }, schema, .. } + && matches!(input, PreASAPNode::Scan { source: Source::Table { .. }, schema, .. } if schema.closed && schema.columns.get(*value_column).is_some_and(|c| c.dtype == DataType::Float64 && !c.nullable) && !grouping.is_without() && grouping.keys().iter().all(|k| *k < schema.columns.len())) } diff --git a/crates/types/src/post_asap/mod.rs b/crates/types/src/post_asap/mod.rs index 6bf31742..93acdc17 100644 --- a/crates/types/src/post_asap/mod.rs +++ b/crates/types/src/post_asap/mod.rs @@ -8,7 +8,7 @@ //! [`sketch::SamplingKind`]/[`sketch::SamplingParams`], //! [`sketch::WaveletKind`]/[`sketch::WaveletParams`], //! [`sketch::StatModelKind`]/[`sketch::StatModelParams`]), and -//! [`expr::SummaryNode`] / [`expr::SummaryExpr`] describe the summary +//! [`expr::PostASAPNode`] / [`expr::SummaryExpr`] describe the summary //! computation. The `Sketch` family is the one exception to that //! one-pair-per-family shape: it nests a third level, [`sketch::SketchKind`] //! (quantile/cardinality/frequency/top-k), which itself carries the @@ -48,17 +48,19 @@ pub use execution_data_state::{ ExecutionDataStateError, ExecutionTiming, }; pub use expr::{ - BinaryOperator, CandidateCompleteness, ExactOperation, SummaryExpr, SummaryNode, ValueOperation, + BinaryOperator, CandidateCompleteness, ExactOperation, PostASAPNode, SummaryExpr, + ValueOperation, }; pub use guarantee::{ AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, GuaranteeSource, ProbabilityExpr, ResultGuarantee, }; pub use post_asap_dag::{ - compile_post_asap_dag, compile_post_asap_dag_with_node_ids, EdgeRole, - GroupingEdgeCompatibility, PostAsapDag, PostAsapDagCompilation, PostAsapDagDocument, - PostAsapDagEdge, PostAsapDagNode, PostAsapDagValidationError, PostAsapNodeId, - PostAsapNodeIdentityMap, PostAsapOperatorPayload, WindowEdgeCompatibility, + export_post_asap_dag, index_post_asap_dag, EdgeRole, GroupingEdgeCompatibility, + LogicalPostASAPDAGAssignment, LogicalPostASAPDAGDocument, LogicalPostASAPDAGEdge, + LogicalPostASAPDAGIndex, LogicalPostASAPDAGNode, LogicalPostASAPDAGTransport, + LogicalPostASAPDAGValidationError, LogicalPostASAPDAGView, PostASAPNodeId, + PostASAPNodeIdentityMap, PostASAPOperatorPayload, WindowEdgeCompatibility, POST_ASAP_DAG_WIRE_VERSION, }; pub use query_time::{ @@ -81,3 +83,7 @@ pub use summary_window::{ plan_pane_phase, validate_pane_coverage, PaneCoverageError, PaneLayout, SummaryWindowFramework, WindowEdgeCoverage, }; + +/// Authoritative shared logical graph; lifecycle timing is an assignment over +/// its nodes, not a second computation graph. +pub type LogicalPostASAPDAG = std::rc::Rc; diff --git a/crates/types/src/post_asap/post_asap_dag.rs b/crates/types/src/post_asap/post_asap_dag.rs index 93b7806f..1d8af1cf 100644 --- a/crates/types/src/post_asap/post_asap_dag.rs +++ b/crates/types/src/post_asap/post_asap_dag.rs @@ -4,14 +4,14 @@ use std::collections::HashMap; use std::rc::Rc; use super::{ - validate_execution_data_states, ExecutionDataState, ExecutionDataStateError, ResultGuarantee, - SummaryExpr, SummaryNode, SummarySchema, + validate_execution_data_states, ExecutionDataState, ExecutionDataStateError, PostASAPNode, + ResultGuarantee, SummaryExpr, SummarySchema, }; use super::{ BinaryOperator, CandidateCompleteness, ExecutionTiming, GroupingStrategy, SketchQuery, SummaryFamilyType, SummaryUpdate, ValueOperation, }; -use crate::pre_asap::{ColumnRef, JoinKind, Predicate, QueryExpr, Reduction}; +use crate::pre_asap::{ColumnRef, JoinKind, PreASAPNode, Predicate, Reduction}; use thiserror::Error; pub const POST_ASAP_DAG_WIRE_VERSION: u32 = 5; @@ -45,13 +45,13 @@ pub enum WindowEdgeCompatibility { Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, serde::Serialize, serde::Deserialize, )] #[serde(transparent)] -pub struct PostAsapNodeId(pub u32); +pub struct PostASAPNodeId(pub u32); #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(tag = "kind", rename_all = "snake_case", deny_unknown_fields)] -pub enum PostAsapOperatorPayload { +pub enum PostASAPOperatorPayload { Fallback { - expression: QueryExpr, + expression: PreASAPNode, }, Binary { operator: BinaryOperator, @@ -86,10 +86,10 @@ pub enum PostAsapOperatorPayload { #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] -pub struct PostAsapDagNode { - pub id: PostAsapNodeId, +pub struct LogicalPostASAPDAGNode { + pub id: PostASAPNodeId, /// The payload variant is the sole operator identity (`payload.kind` in JSON). - pub payload: PostAsapOperatorPayload, + pub payload: PostASAPOperatorPayload, /// Phase is a placement choice for every operator, independent of payload kind. pub output_state: ExecutionDataState, pub output_schema: SummarySchema, @@ -98,9 +98,9 @@ pub struct PostAsapDagNode { #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] -pub struct PostAsapDagEdge { - pub producer: PostAsapNodeId, - pub consumer: PostAsapNodeId, +pub struct LogicalPostASAPDAGEdge { + pub producer: PostASAPNodeId, + pub consumer: PostASAPNodeId, pub role: EdgeRole, pub intermediate_schema: SummarySchema, pub data_state: ExecutionDataState, @@ -110,12 +110,12 @@ pub struct PostAsapDagEdge { #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] -pub struct PostAsapDag { - pub nodes: Vec, - pub edges: Vec, +pub struct LogicalPostASAPDAGTransport { + pub nodes: Vec, + pub edges: Vec, /// Semantic workload root. Physical query/precompute sinks are selected /// downstream by the control plane. - pub root: PostAsapNodeId, + pub root: PostASAPNodeId, } /// Versioned transport envelope for a post-ASAP semantic DAG. @@ -123,61 +123,61 @@ pub struct PostAsapDag { /// Process boundaries exchange this envelope and call [`Self::validate`]. #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] -pub struct PostAsapDagDocument { +pub struct LogicalPostASAPDAGDocument { pub schema_version: u32, - pub dag: PostAsapDag, + pub dag: LogicalPostASAPDAGTransport, } #[derive(Debug, Clone, PartialEq, Eq, Error)] -pub enum PostAsapDagValidationError { +pub enum LogicalPostASAPDAGValidationError { #[error("phase assignment must name every DAG node exactly once")] IncompletePhaseAssignment, #[error("ingestion node {consumer:?} depends on query node {producer:?}")] QueryDependencyInIngestion { - producer: PostAsapNodeId, - consumer: PostAsapNodeId, + producer: PostASAPNodeId, + consumer: PostASAPNodeId, }, #[error("unsupported post-ASAP DAG schema version {0}")] UnsupportedVersion(u32), #[error("duplicate post-ASAP node id {0:?}")] - DuplicateNodeId(PostAsapNodeId), + DuplicateNodeId(PostASAPNodeId), #[error("post-ASAP DAG root {0:?} does not name a node")] - MissingRoot(PostAsapNodeId), + MissingRoot(PostASAPNodeId), #[error("edge endpoint {0:?} does not name a node")] - MissingEdgeEndpoint(PostAsapNodeId), + MissingEdgeEndpoint(PostASAPNodeId), #[error("edge {producer:?}->{consumer:?} schema differs from producer output")] EdgeSchemaMismatch { - producer: PostAsapNodeId, - consumer: PostAsapNodeId, + producer: PostASAPNodeId, + consumer: PostASAPNodeId, }, #[error("edge {producer:?}->{consumer:?} data state differs from producer output")] EdgeDataStateMismatch { - producer: PostAsapNodeId, - consumer: PostAsapNodeId, + producer: PostASAPNodeId, + consumer: PostASAPNodeId, }, #[error("post-ASAP DAG contains a cycle")] Cycle, #[error("post-ASAP node {0:?} is not reachable from the root")] - UnreachableNode(PostAsapNodeId), + UnreachableNode(PostASAPNodeId), #[error("summary aggregate node {node:?} output schema does not contain its declared family")] - SummaryFamilySchemaMismatch { node: PostAsapNodeId }, + SummaryFamilySchemaMismatch { node: PostASAPNodeId }, #[error( "summary aggregate node {node:?} declares grouping inconsistent with its sketch state" )] - SummaryGroupingMismatch { node: PostAsapNodeId }, + SummaryGroupingMismatch { node: PostASAPNodeId }, } -impl PostAsapDagDocument { - pub fn new(dag: PostAsapDag) -> Self { +impl LogicalPostASAPDAGDocument { + pub fn new(dag: LogicalPostASAPDAGTransport) -> Self { Self { schema_version: POST_ASAP_DAG_WIRE_VERSION, dag, } } - pub fn validate(&self) -> Result<(), PostAsapDagValidationError> { + pub fn validate(&self) -> Result<(), LogicalPostASAPDAGValidationError> { if self.schema_version != POST_ASAP_DAG_WIRE_VERSION { - return Err(PostAsapDagValidationError::UnsupportedVersion( + return Err(LogicalPostASAPDAGValidationError::UnsupportedVersion( self.schema_version, )); } @@ -185,40 +185,90 @@ impl PostAsapDagDocument { } } -impl PostAsapDag { +impl LogicalPostASAPDAGTransport { /// Assign execution phases without changing operator semantics. Phase choices /// do not prove deployment support: callers must bind concrete implementations /// and storage boundaries before installing this plan. pub fn with_execution_phases( &self, - phases: &std::collections::BTreeMap, - ) -> Result { + phases: &std::collections::BTreeMap, + ) -> Result { self.validate()?; - if phases.len() != self.nodes.len() - || self.nodes.iter().any(|node| !phases.contains_key(&node.id)) - { - return Err(PostAsapDagValidationError::IncompletePhaseAssignment); - } let mut dag = self.clone(); - for node in &mut dag.nodes { - node.output_state.timing = phases[&node.id]; - } - let states: HashMap<_, _> = dag.nodes.iter().map(|n| (n.id, n.output_state)).collect(); - for edge in &mut dag.edges { - edge.data_state = states[&edge.producer]; - } + assign_phases(&mut dag.nodes, &mut dag.edges, phases)?; dag.validate()?; Ok(dag) } - pub fn validate(&self) -> Result<(), PostAsapDagValidationError> { + pub fn as_view(&self) -> LogicalPostASAPDAGView<'_> { + LogicalPostASAPDAGView { + nodes: &self.nodes, + edges: &self.edges, + root: self.root, + phases: None, + } + } + pub fn validate(&self) -> Result<(), LogicalPostASAPDAGValidationError> { + self.as_view().validate() + } +} + +/// Borrowed compilation/validation projection. Owns no logical computation. +/// A lifecycle assignment overlays its timing on the shared index's records, +/// so node and edge states must be read through [`Self::timing`], +/// [`Self::output_state`] and [`Self::edge_state`], not from the records. +#[derive(Clone, Copy)] +pub struct LogicalPostASAPDAGView<'a> { + nodes: &'a [LogicalPostASAPDAGNode], + edges: &'a [LogicalPostASAPDAGEdge], + root: PostASAPNodeId, + phases: Option<&'a std::collections::BTreeMap>, +} +impl<'a> LogicalPostASAPDAGView<'a> { + pub fn nodes(&self) -> &'a [LogicalPostASAPDAGNode] { + self.nodes + } + pub fn edges(&self) -> &'a [LogicalPostASAPDAGEdge] { + self.edges + } + pub fn root(&self) -> PostASAPNodeId { + self.root + } + /// Execution timing of `node` under this view's assignment. + pub fn timing(&self, node: &LogicalPostASAPDAGNode) -> ExecutionTiming { + self.phases + .and_then(|phases| phases.get(&node.id).copied()) + .unwrap_or(node.output_state.timing) + } + pub fn output_state(&self, node: &LogicalPostASAPDAGNode) -> ExecutionDataState { + ExecutionDataState { + timing: self.timing(node), + ..node.output_state + } + } + /// An assigned edge carries its producer's assigned state. + pub fn edge_state(&self, edge: &LogicalPostASAPDAGEdge) -> ExecutionDataState { + match self.phases { + None => edge.data_state, + Some(phases) => ExecutionDataState { + timing: phases + .get(&edge.producer) + .copied() + .unwrap_or(edge.data_state.timing), + ..edge.data_state + }, + } + } +} +impl LogicalPostASAPDAGView<'_> { + pub fn validate(&self) -> Result<(), LogicalPostASAPDAGValidationError> { use std::collections::{HashMap, HashSet}; let mut nodes = HashMap::new(); - for node in &self.nodes { + for node in self.nodes { if nodes.insert(node.id, node).is_some() { - return Err(PostAsapDagValidationError::DuplicateNodeId(node.id)); + return Err(LogicalPostASAPDAGValidationError::DuplicateNodeId(node.id)); } - if let PostAsapOperatorPayload::SummaryAgg { + if let PostASAPOperatorPayload::SummaryAgg { family, grouping, .. } = &node.payload { @@ -229,48 +279,54 @@ impl PostAsapDag { } if let SummaryFamilyType::Sketch(_, schema_grouping) = &field.dtype { if schema_grouping != grouping { - return Err(PostAsapDagValidationError::SummaryGroupingMismatch { - node: node.id, - }); + return Err( + LogicalPostASAPDAGValidationError::SummaryGroupingMismatch { + node: node.id, + }, + ); } } } if !found_family { - return Err(PostAsapDagValidationError::SummaryFamilySchemaMismatch { - node: node.id, - }); + return Err( + LogicalPostASAPDAGValidationError::SummaryFamilySchemaMismatch { + node: node.id, + }, + ); } } } if !nodes.contains_key(&self.root) { - return Err(PostAsapDagValidationError::MissingRoot(self.root)); + return Err(LogicalPostASAPDAGValidationError::MissingRoot(self.root)); } - let mut children: HashMap> = HashMap::new(); - for edge in &self.edges { + let mut children: HashMap> = HashMap::new(); + for edge in self.edges { let producer = nodes.get(&edge.producer).ok_or( - PostAsapDagValidationError::MissingEdgeEndpoint(edge.producer), + LogicalPostASAPDAGValidationError::MissingEdgeEndpoint(edge.producer), )?; if !nodes.contains_key(&edge.consumer) { - return Err(PostAsapDagValidationError::MissingEdgeEndpoint( + return Err(LogicalPostASAPDAGValidationError::MissingEdgeEndpoint( edge.consumer, )); } - if producer.output_state.timing == ExecutionTiming::QueryTime - && nodes[&edge.consumer].output_state.timing == ExecutionTiming::IngestionTime + if self.timing(producer) == ExecutionTiming::QueryTime + && self.timing(nodes[&edge.consumer]) == ExecutionTiming::IngestionTime { - return Err(PostAsapDagValidationError::QueryDependencyInIngestion { - producer: edge.producer, - consumer: edge.consumer, - }); + return Err( + LogicalPostASAPDAGValidationError::QueryDependencyInIngestion { + producer: edge.producer, + consumer: edge.consumer, + }, + ); } if edge.intermediate_schema != producer.output_schema { - return Err(PostAsapDagValidationError::EdgeSchemaMismatch { + return Err(LogicalPostASAPDAGValidationError::EdgeSchemaMismatch { producer: edge.producer, consumer: edge.consumer, }); } - if edge.data_state != producer.output_state { - return Err(PostAsapDagValidationError::EdgeDataStateMismatch { + if self.edge_state(edge) != self.output_state(producer) { + return Err(LogicalPostASAPDAGValidationError::EdgeDataStateMismatch { producer: edge.producer, consumer: edge.consumer, }); @@ -281,10 +337,10 @@ impl PostAsapDag { .push(edge.producer); } fn visit( - id: PostAsapNodeId, - children: &HashMap>, - visiting: &mut HashSet, - visited: &mut HashSet, + id: PostASAPNodeId, + children: &HashMap>, + visiting: &mut HashSet, + visited: &mut HashSet, ) -> bool { if visited.contains(&id) { return true; @@ -310,13 +366,13 @@ impl PostAsapDag { &mut HashSet::new(), &mut HashSet::new(), ) { - return Err(PostAsapDagValidationError::Cycle); + return Err(LogicalPostASAPDAGValidationError::Cycle); } let mut reachable = HashSet::new(); fn mark( - id: PostAsapNodeId, - children: &HashMap>, - reachable: &mut HashSet, + id: PostASAPNodeId, + children: &HashMap>, + reachable: &mut HashSet, ) { if !reachable.insert(id) { return; @@ -327,66 +383,236 @@ impl PostAsapDag { } mark(self.root, &children, &mut reachable); if let Some(id) = nodes.keys().find(|id| !reachable.contains(id)) { - return Err(PostAsapDagValidationError::UnreachableNode(*id)); + return Err(LogicalPostASAPDAGValidationError::UnreachableNode(*id)); } Ok(()) } } +/// Lifecycle-assigned timing over one indexed shared logical graph. +/// Cloning an assignment shares the graph and its index; it copies no operators. +#[derive(Debug, Clone)] +pub struct LogicalPostASAPDAGAssignment { + index: Rc, + phases: std::collections::BTreeMap, +} +impl LogicalPostASAPDAGAssignment { + pub fn index(&self) -> &Rc { + &self.index + } + pub fn new( + index: Rc, + phases: std::collections::BTreeMap, + ) -> Result { + if phases.len() != index.nodes.len() + || index.nodes.iter().any(|n| !phases.contains_key(&n.id)) + { + return Err(LogicalPostASAPDAGValidationError::IncompletePhaseAssignment); + } + let result = Self { index, phases }; + result.view().validate()?; + Ok(result) + } + pub fn phases(&self) -> &std::collections::BTreeMap { + &self.phases + } + /// The shared index's records with this assignment's timing overlaid. + pub fn view(&self) -> LogicalPostASAPDAGView<'_> { + LogicalPostASAPDAGView { + phases: Some(&self.phases), + ..self.index.view() + } + } + pub fn to_transport(&self) -> LogicalPostASAPDAGTransport { + let view = self.view(); + LogicalPostASAPDAGTransport { + nodes: view + .nodes + .iter() + .map(|node| LogicalPostASAPDAGNode { + output_state: view.output_state(node), + ..node.clone() + }) + .collect(), + edges: view + .edges + .iter() + .map(|edge| LogicalPostASAPDAGEdge { + data_state: view.edge_state(edge), + ..edge.clone() + }) + .collect(), + root: view.root, + } + } +} + +fn assign_phases( + nodes: &mut [LogicalPostASAPDAGNode], + edges: &mut [LogicalPostASAPDAGEdge], + phases: &std::collections::BTreeMap, +) -> Result<(), LogicalPostASAPDAGValidationError> { + if phases.len() != nodes.len() || nodes.iter().any(|n| !phases.contains_key(&n.id)) { + return Err(LogicalPostASAPDAGValidationError::IncompletePhaseAssignment); + } + for node in nodes.iter_mut() { + node.output_state.timing = phases[&node.id]; + } + let states: HashMap<_, _> = nodes.iter().map(|n| (n.id, n.output_state)).collect(); + for edge in edges { + edge.data_state = *states.get(&edge.producer).ok_or( + LogicalPostASAPDAGValidationError::MissingEdgeEndpoint(edge.producer), + )?; + } + Ok(()) +} + /// Compiler-local identity assignment. It deliberately retains `Rc` handles /// and is not serialized; deployed artifacts persist the post-ASAP node ID /// together with their physical materialization/query IDs. #[derive(Debug, Clone)] -pub struct PostAsapNodeIdentityMap { - nodes_by_id: Vec>, +pub struct PostASAPNodeIdentityMap { + nodes_by_id: Vec>, } -impl PostAsapNodeIdentityMap { - pub fn node_id(&self, node: &Rc) -> Option { +impl PostASAPNodeIdentityMap { + pub fn node_id(&self, node: &Rc) -> Option { self.nodes_by_id .iter() .position(|candidate| Rc::ptr_eq(candidate, node)) - .map(|id| PostAsapNodeId(id as u32)) + .map(|id| PostASAPNodeId(id as u32)) } - pub fn summary_node(&self, id: PostAsapNodeId) -> Option<&Rc> { + pub fn summary_node(&self, id: PostASAPNodeId) -> Option<&Rc> { self.nodes_by_id.get(id.0 as usize) } } +pub fn export_post_asap_dag( + root: &Rc, +) -> Result { + let dag = index_post_asap_dag(root)?.to_transport(); + dag.validate() + .expect("compiler emits a valid post-ASAP DAG"); + Ok(dag) +} + +/// Indexed projection of the authoritative shared graph: node identities, +/// plus node and edge records projected once for compilation and transport. +/// Lifecycle assignments overlay timing on these records without copying them. #[derive(Debug, Clone)] -pub struct PostAsapDagCompilation { - pub dag: PostAsapDag, - pub node_ids: PostAsapNodeIdentityMap, +pub struct LogicalPostASAPDAGIndex { + pub root_id: PostASAPNodeId, + pub node_ids: PostASAPNodeIdentityMap, + nodes: Vec, + edges: Vec, +} + +impl LogicalPostASAPDAGIndex { + /// Node records in ID order, with the logical graph's own timing. + pub fn node_views(&self) -> &[LogicalPostASAPDAGNode] { + &self.nodes + } + pub fn edges(&self) -> &[LogicalPostASAPDAGEdge] { + &self.edges + } + pub fn view(&self) -> LogicalPostASAPDAGView<'_> { + LogicalPostASAPDAGView { + nodes: &self.nodes, + edges: &self.edges, + root: self.root_id, + phases: None, + } + } + pub fn to_transport(&self) -> LogicalPostASAPDAGTransport { + LogicalPostASAPDAGTransport { + nodes: self.nodes.clone(), + edges: self.edges.clone(), + root: self.root_id, + } + } } -pub fn compile_post_asap_dag( - root: &Rc, -) -> Result { - Ok(compile_post_asap_dag_with_node_ids(root)?.dag) +fn project_node( + id: PostASAPNodeId, + node: &PostASAPNode, + state: ExecutionDataState, +) -> LogicalPostASAPDAGNode { + let payload = match &node.expr { + SummaryExpr::KeepPreAsap(expression) => PostASAPOperatorPayload::Fallback { + expression: (**expression).clone(), + }, + SummaryExpr::BinaryOp { operator, .. } => PostASAPOperatorPayload::Binary { + operator: operator.clone(), + }, + + SummaryExpr::ValueOperation { operation, .. } => PostASAPOperatorPayload::Value { + operation: operation.clone(), + }, + SummaryExpr::RelationalJoin { + kind, + pred, + pruning, + .. + } => PostASAPOperatorPayload::RelationalJoin { + join_kind: kind.clone(), + pred: pred.clone(), + pruning: pruning.clone(), + }, + SummaryExpr::SummaryAgg { + family, + input, + reduction, + grouping, + .. + } => PostASAPOperatorPayload::SummaryAgg { + family: family.clone(), + input: input.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + }, + SummaryExpr::SummaryJoin { key, family, .. } => PostASAPOperatorPayload::SummaryJoin { + key: key.clone(), + family: family.clone(), + }, + SummaryExpr::SummarySubtract { .. } => PostASAPOperatorPayload::SummarySubtract, + SummaryExpr::SummaryDelete { key, .. } => { + PostASAPOperatorPayload::SummaryDelete { key: key.clone() } + } + SummaryExpr::SummaryEstimate { query, .. } => PostASAPOperatorPayload::SummaryEstimate { + query: query.clone(), + }, + SummaryExpr::SummaryMerge { .. } => PostASAPOperatorPayload::SummaryMerge, + }; + LogicalPostASAPDAGNode { + id, + payload, + output_state: state, + output_schema: node.schema.clone(), + guarantee: node.guarantee.clone(), + } } -pub fn compile_post_asap_dag_with_node_ids( - root: &Rc, -) -> Result { +/// Assign stable postorder IDs once without copying the logical operators. +pub fn index_post_asap_dag( + root: &super::LogicalPostASAPDAG, +) -> Result { let assignment = validate_execution_data_states(root)?; - let mut nodes = Vec::new(); let mut edges = Vec::new(); let mut ids = HashMap::new(); - let mut nodes_by_id = Vec::new(); + let mut nodes = Vec::new(); fn visit( - node: &Rc, + node: &Rc, assignment: &super::ExecutionDataStateAssignment, - ids: &mut HashMap<*const SummaryNode, PostAsapNodeId>, - nodes: &mut Vec, - edges: &mut Vec, - nodes_by_id: &mut Vec>, - ) -> PostAsapNodeId { + ids: &mut HashMap<*const PostASAPNode, PostASAPNodeId>, + nodes: &mut Vec>, + edges: &mut Vec, + ) -> PostASAPNodeId { if let Some(id) = ids.get(&Rc::as_ptr(node)) { return *id; } - let children: Vec<(&Rc, EdgeRole)> = match &node.expr { + let children: Vec<(&Rc, EdgeRole)> = match &node.expr { SummaryExpr::KeepPreAsap(_) => vec![], SummaryExpr::BinaryOp { lhs, rhs, .. } => { vec![(lhs, EdgeRole::Left), (rhs, EdgeRole::Right)] @@ -414,73 +640,22 @@ pub fn compile_post_asap_dag_with_node_ids( }; let child_ids: Vec<_> = children .iter() - .map(|(c, r)| (visit(c, assignment, ids, nodes, edges, nodes_by_id), *c, *r)) + .map(|(c, r)| (visit(c, assignment, ids, nodes, edges), *c, *r)) .collect(); - let id = PostAsapNodeId(nodes.len() as u32); - let state = assignment - .data_state_of(node) - .expect("validated node has state"); - let payload = match &node.expr { - SummaryExpr::KeepPreAsap(expression) => PostAsapOperatorPayload::Fallback { - expression: (**expression).clone(), - }, - SummaryExpr::BinaryOp { operator, .. } => PostAsapOperatorPayload::Binary { - operator: operator.clone(), - }, - - SummaryExpr::ValueOperation { operation, .. } => PostAsapOperatorPayload::Value { - operation: operation.clone(), - }, - SummaryExpr::RelationalJoin { - kind, - pred, - pruning, - .. - } => PostAsapOperatorPayload::RelationalJoin { - join_kind: kind.clone(), - pred: pred.clone(), - pruning: pruning.clone(), - }, - SummaryExpr::SummaryAgg { - family, - input, - reduction, - grouping, - .. - } => PostAsapOperatorPayload::SummaryAgg { - family: family.clone(), - input: input.clone(), - reduction: reduction.clone(), - grouping: grouping.clone(), - }, - SummaryExpr::SummaryJoin { key, family, .. } => PostAsapOperatorPayload::SummaryJoin { - key: key.clone(), - family: family.clone(), - }, - SummaryExpr::SummarySubtract { .. } => PostAsapOperatorPayload::SummarySubtract, - SummaryExpr::SummaryDelete { key, .. } => { - PostAsapOperatorPayload::SummaryDelete { key: key.clone() } - } - SummaryExpr::SummaryEstimate { query, .. } => { - PostAsapOperatorPayload::SummaryEstimate { - query: query.clone(), - } - } - SummaryExpr::SummaryMerge { .. } => PostAsapOperatorPayload::SummaryMerge, - }; - nodes.push(PostAsapDagNode { - id, - payload, - output_state: state, - output_schema: node.schema.clone(), - guarantee: node.guarantee.clone(), - }); - nodes_by_id.push(Rc::clone(node)); + let id = PostASAPNodeId(nodes.len() as u32); + nodes.push(Rc::clone(node)); ids.insert(Rc::as_ptr(node), id); for (producer, child, role) in child_ids { - let maintenance_dependency = nodes[producer.0 as usize].output_state.timing + let maintenance_dependency = assignment + .data_state_of(&nodes[producer.0 as usize]) + .expect("validated producer") + .timing == ExecutionTiming::IngestionTime - && nodes[id.0 as usize].output_state.timing == ExecutionTiming::IngestionTime; + && assignment + .data_state_of(node) + .expect("validated consumer") + .timing + == ExecutionTiming::IngestionTime; let grouping = match (&child.expr, &node.expr) { ( SummaryExpr::SummaryAgg { @@ -522,7 +697,7 @@ pub fn compile_post_asap_dag_with_node_ids( } _ => GroupingEdgeCompatibility::NotApplicable, }; - edges.push(PostAsapDagEdge { + edges.push(LogicalPostASAPDAGEdge { producer, consumer: id, role, @@ -545,20 +720,25 @@ pub fn compile_post_asap_dag_with_node_ids( id } - let root = visit( - root, - &assignment, - &mut ids, - &mut nodes, - &mut edges, - &mut nodes_by_id, - ); - let dag = PostAsapDag { nodes, edges, root }; - dag.validate() - .expect("compiler emits a valid post-ASAP DAG"); - Ok(PostAsapDagCompilation { - dag, - node_ids: PostAsapNodeIdentityMap { nodes_by_id }, + let root = visit(root, &assignment, &mut ids, &mut nodes, &mut edges); + let projected = nodes + .iter() + .enumerate() + .map(|(id, node)| { + project_node( + PostASAPNodeId(id as u32), + node, + assignment + .data_state_of(node) + .expect("indexed node has a state"), + ) + }) + .collect(); + Ok(LogicalPostASAPDAGIndex { + root_id: root, + node_ids: PostASAPNodeIdentityMap { nodes_by_id: nodes }, + nodes: projected, + edges, }) } @@ -570,7 +750,7 @@ mod tests { SummaryUpdate, ValueOperation, }; use crate::pre_asap::schema::{Column, Schema}; - use crate::pre_asap::{ColumnRef, DataType, QueryExpr, Reduction, Source}; + use crate::pre_asap::{ColumnRef, DataType, PreASAPNode, Reduction, Source}; use std::collections::BTreeMap; #[test] @@ -578,12 +758,12 @@ mod tests { use crate::post_asap::DataPrimitive; use crate::pre_asap::{ArithmeticOpKind, BinaryOpKind, JoinKind, Predicate, ScalarValue}; let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let predicate = Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))); + let predicate = Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(true)))); let payloads = vec![ - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::Literal(ScalarValue::Int64(1)), + PostASAPOperatorPayload::Fallback { + expression: PreASAPNode::Literal(ScalarValue::Int64(1)), }, - PostAsapOperatorPayload::Binary { + PostASAPOperatorPayload::Binary { operator: BinaryOperator { checked_relative_division: false, checked_finite_division: false, @@ -591,56 +771,56 @@ mod tests { vector_match: None, }, }, - PostAsapOperatorPayload::Value { + PostASAPOperatorPayload::Value { operation: ValueOperation::Limit { n: 1, offset: 0, partition_by: Default::default(), }, }, - PostAsapOperatorPayload::RelationalJoin { + PostASAPOperatorPayload::RelationalJoin { join_kind: JoinKind::Semi, pred: predicate, pruning: None, }, - PostAsapOperatorPayload::SummaryAgg { + PostASAPOperatorPayload::SummaryAgg { family: family.clone(), input: SummaryUpdate::column(ColumnRef::SampleValue), reduction: Reduction::by(vec![]), grouping: GroupingStrategy::default(), }, - PostAsapOperatorPayload::SummaryJoin { + PostASAPOperatorPayload::SummaryJoin { key: ColumnRef::SampleValue, family: family.clone(), }, - PostAsapOperatorPayload::SummarySubtract, - PostAsapOperatorPayload::SummaryDelete { + PostASAPOperatorPayload::SummarySubtract, + PostASAPOperatorPayload::SummaryDelete { key: ColumnRef::SampleValue, }, - PostAsapOperatorPayload::SummaryEstimate { + PostASAPOperatorPayload::SummaryEstimate { query: SketchQuery::Cardinality, }, - PostAsapOperatorPayload::SummaryMerge, + PostASAPOperatorPayload::SummaryMerge, ]; for payload in payloads { // This checks physical identity and placement, not kernel availability. let primitive = match &payload { - PostAsapOperatorPayload::Fallback { .. } - | PostAsapOperatorPayload::Binary { .. } - | PostAsapOperatorPayload::Value { .. } - | PostAsapOperatorPayload::RelationalJoin { .. } - | PostAsapOperatorPayload::SummaryEstimate { .. } => DataPrimitive::Raw, - PostAsapOperatorPayload::SummaryAgg { .. } - | PostAsapOperatorPayload::SummaryJoin { .. } - | PostAsapOperatorPayload::SummarySubtract - | PostAsapOperatorPayload::SummaryDelete { .. } - | PostAsapOperatorPayload::SummaryMerge => DataPrimitive::SummaryState, + PostASAPOperatorPayload::Fallback { .. } + | PostASAPOperatorPayload::Binary { .. } + | PostASAPOperatorPayload::Value { .. } + | PostASAPOperatorPayload::RelationalJoin { .. } + | PostASAPOperatorPayload::SummaryEstimate { .. } => DataPrimitive::Raw, + PostASAPOperatorPayload::SummaryAgg { .. } + | PostASAPOperatorPayload::SummaryJoin { .. } + | PostASAPOperatorPayload::SummarySubtract + | PostASAPOperatorPayload::SummaryDelete { .. } + | PostASAPOperatorPayload::SummaryMerge => DataPrimitive::SummaryState, }; - let dag = PostAsapDag { - root: PostAsapNodeId(0), + let dag = LogicalPostASAPDAGTransport { + root: PostASAPNodeId(0), edges: vec![], - nodes: vec![PostAsapDagNode { - id: PostAsapNodeId(0), + nodes: vec![LogicalPostASAPDAGNode { + id: PostASAPNodeId(0), payload: payload.clone(), output_state: ExecutionDataState { timing: ExecutionTiming::QueryTime, @@ -665,7 +845,10 @@ mod tests { assert_eq!(placed.nodes[0].output_state.timing, phase); let wire = serde_json::to_value(&placed).unwrap(); assert!(wire["nodes"][0]["payload"].get("timing").is_none()); - assert_eq!(serde_json::from_value::(wire).unwrap(), placed); + assert_eq!( + serde_json::from_value::(wire).unwrap(), + placed + ); } assert!(dag.with_execution_phases(&BTreeMap::new()).is_err()); } @@ -680,22 +863,22 @@ mod tests { }; let nodes = [0, 1] .into_iter() - .map(|id| PostAsapDagNode { - id: PostAsapNodeId(id), - payload: PostAsapOperatorPayload::Fallback { - expression: QueryExpr::Literal(ScalarValue::Int64(1)), + .map(|id| LogicalPostASAPDAGNode { + id: PostASAPNodeId(id), + payload: PostASAPOperatorPayload::Fallback { + expression: PreASAPNode::Literal(ScalarValue::Int64(1)), }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: schema.clone(), guarantee: None, }) .collect(); - let dag = PostAsapDag { + let dag = LogicalPostASAPDAGTransport { nodes, - root: PostAsapNodeId(1), - edges: vec![PostAsapDagEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), + root: PostASAPNodeId(1), + edges: vec![LogicalPostASAPDAGEdge { + producer: PostASAPNodeId(0), + consumer: PostASAPNodeId(1), role: EdgeRole::Input, intermediate_schema: schema, data_state: ExecutionDataState::QUERY_ROWS, @@ -705,8 +888,8 @@ mod tests { }; let placed = dag .with_execution_phases(&BTreeMap::from([ - (PostAsapNodeId(0), ExecutionTiming::IngestionTime), - (PostAsapNodeId(1), ExecutionTiming::QueryTime), + (PostASAPNodeId(0), ExecutionTiming::IngestionTime), + (PostASAPNodeId(1), ExecutionTiming::QueryTime), ])) .unwrap(); assert_eq!( @@ -716,21 +899,21 @@ mod tests { assert_eq!(dag.edges[0].data_state.timing, ExecutionTiming::QueryTime); assert!(matches!( dag.with_execution_phases(&BTreeMap::from([ - (PostAsapNodeId(0), ExecutionTiming::QueryTime), - (PostAsapNodeId(1), ExecutionTiming::IngestionTime), + (PostASAPNodeId(0), ExecutionTiming::QueryTime), + (PostASAPNodeId(1), ExecutionTiming::IngestionTime), ])), - Err(PostAsapDagValidationError::QueryDependencyInIngestion { .. }) + Err(LogicalPostASAPDAGValidationError::QueryDependencyInIngestion { .. }) )); } #[test] fn exports_summary_over_summary_as_typed_precompute_edges() { - let scan = Rc::new(QueryExpr::Scan { + let scan = Rc::new(PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), }); - let raw = Rc::new(SummaryNode { + let raw = Rc::new(PostASAPNode { expr: SummaryExpr::KeepPreAsap(scan), schema: SummarySchema { fields: vec![SummaryField { @@ -742,9 +925,9 @@ mod tests { }, guarantee: None, }); - let make_agg = |child: Rc, kind, params| { + let make_agg = |child: Rc, kind, params| { let family = SummaryFamilyType::ExactAggregate(kind, params); - Rc::new(SummaryNode { + Rc::new(PostASAPNode { expr: SummaryExpr::SummaryAgg { child, family: family.clone(), @@ -765,7 +948,7 @@ mod tests { }; let inner = make_agg(raw, ExactKind::Sum, ExactParams::Sum); let outer = make_agg(Rc::clone(&inner), ExactKind::Sum, ExactParams::Sum); - let root = Rc::new(SummaryNode { + let root = Rc::new(PostASAPNode { expr: SummaryExpr::ValueOperation { child: outer, operation: ValueOperation::FinalizeExactAccumulator, @@ -782,14 +965,14 @@ mod tests { guarantee: None, }); - let compiled = compile_post_asap_dag_with_node_ids(&root).unwrap(); - assert_eq!(compiled.node_ids.node_id(&root), Some(PostAsapNodeId(3))); + let compiled = index_post_asap_dag(&root).unwrap(); + assert_eq!(compiled.node_ids.node_id(&root), Some(PostASAPNodeId(3))); assert!(Rc::ptr_eq( - compiled.node_ids.summary_node(PostAsapNodeId(1)).unwrap(), + compiled.node_ids.summary_node(PostASAPNodeId(1)).unwrap(), &inner )); - let dag = compiled.dag; - assert_eq!(dag.root, PostAsapNodeId(3)); + let dag = compiled.to_transport(); + assert_eq!(dag.root, PostASAPNodeId(3)); assert_eq!( dag.nodes[1].output_state, ExecutionDataState::INGESTION_SUMMARY @@ -801,7 +984,7 @@ mod tests { let dependency = dag .edges .iter() - .find(|e| e.producer == PostAsapNodeId(1) && e.consumer == PostAsapNodeId(2)) + .find(|e| e.producer == PostASAPNodeId(1) && e.consumer == PostASAPNodeId(2)) .unwrap(); assert_eq!(dependency.data_state, ExecutionDataState::INGESTION_SUMMARY); assert_eq!(dependency.grouping, GroupingEdgeCompatibility::Identical); @@ -814,14 +997,14 @@ mod tests { SummaryFamilyType::ExactAggregate(ExactKind::Sum, _) )); let encoded = serde_json::to_string(&dag).expect("serialize post-ASAP DAG"); - let decoded: PostAsapDag = + let decoded: LogicalPostASAPDAGTransport = serde_json::from_str(&encoded).expect("deserialize post-ASAP DAG"); assert_eq!(decoded, dag); - let document = PostAsapDagDocument::new(decoded); + let document = LogicalPostASAPDAGDocument::new(decoded); document.validate().unwrap(); let mut invalid = serde_json::to_value(&document).unwrap(); invalid["dag"]["nodes"][0]["operator"] = serde_json::json!("Binary"); - assert!(serde_json::from_value::(invalid).is_err()); + assert!(serde_json::from_value::(invalid).is_err()); assert!(document.dag.nodes.iter().all(|node| { let wire = serde_json::to_value(node).unwrap(); wire.get("operator").is_none() && wire["payload"]["kind"].is_string() @@ -830,26 +1013,46 @@ mod tests { old_version.schema_version = 1; assert_eq!( old_version.validate(), - Err(PostAsapDagValidationError::UnsupportedVersion(1)) + Err(LogicalPostASAPDAGValidationError::UnsupportedVersion(1)) ); let mut unknown = serde_json::to_value(&document).unwrap(); unknown["unexpected"] = serde_json::json!(true); - assert!(serde_json::from_value::(unknown).is_err()); + assert!(serde_json::from_value::(unknown).is_err()); assert!(matches!( dag.nodes[2].payload, - PostAsapOperatorPayload::SummaryAgg { + PostASAPOperatorPayload::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), reduction: Reduction::Reduce(_), .. } )); + + // An assignment's view overlays its timing on the shared records; its + // transport equals assigning the same phases to the exported document. + let index = Rc::new(compiled); + let phases: BTreeMap<_, _> = index + .node_views() + .iter() + .map(|n| (n.id, ExecutionTiming::QueryTime)) + .collect(); + let assignment = LogicalPostASAPDAGAssignment::new(index.clone(), phases.clone()).unwrap(); + let summary = &index.node_views()[1]; + assert_eq!(index.view().timing(summary), ExecutionTiming::IngestionTime); + assert_eq!( + assignment.view().timing(summary), + ExecutionTiming::QueryTime + ); + assert_eq!( + assignment.to_transport(), + index.to_transport().with_execution_phases(&phases).unwrap() + ); } #[test] fn post_asap_node_ids_serialize_in_deterministic_binding_order() { let mut bindings = BTreeMap::new(); - bindings.insert(PostAsapNodeId(10), "materialization-10"); - bindings.insert(PostAsapNodeId(2), "query-2"); + bindings.insert(PostASAPNodeId(10), "materialization-10"); + bindings.insert(PostASAPNodeId(2), "query-2"); assert_eq!( serde_json::to_string(&bindings).unwrap(), r#"{"2":"query-2","10":"materialization-10"}"# diff --git a/crates/types/src/post_asap/summary_maintenance.rs b/crates/types/src/post_asap/summary_maintenance.rs index d1e50d7e..75d55514 100644 --- a/crates/types/src/post_asap/summary_maintenance.rs +++ b/crates/types/src/post_asap/summary_maintenance.rs @@ -1,6 +1,6 @@ //! Planner-level construction mode for a materialized summary. //! -//! A [`super::SummaryNode`] is a logical summary expression and deliberately +//! A [`super::PostASAPNode`] is a logical summary expression and deliberately //! does not carry this choice: the same candidate may be built directly for //! one workload or maintained incrementally for another. Planner search //! attaches the selected mode to its lifecycle guarantee; downstream physical diff --git a/crates/types/src/pre_asap/agg_intent.rs b/crates/types/src/pre_asap/agg_intent.rs index 2203f4ec..17d3d14f 100644 --- a/crates/types/src/pre_asap/agg_intent.rs +++ b/crates/types/src/pre_asap/agg_intent.rs @@ -10,7 +10,7 @@ //! heavy-hitter sketch when approximate — is a post-ASAP cost-aware decision, //! not encoded here. The semantic distinction that *is* made at lowering is //! intent vs operator: a heavy-hitter aggregate becomes `TopK`, whereas a -//! generic `ORDER BY value LIMIT k` stays as the `QueryExpr::Sort + Limit` +//! generic `ORDER BY value LIMIT k` stays as the `PreASAPNode::Sort + Limit` //! operator pair. use serde::{Deserialize, Serialize}; @@ -21,14 +21,14 @@ use crate::types::AccuracyTarget; /// "What to compute" — the vocabulary the planner pivots on. /// -/// Grouping for `TopK` rides on the enclosing `QueryExpr::Aggregate.by` +/// Grouping for `TopK` rides on the enclosing `PreASAPNode::Aggregate.by` /// (positional `ColumnId`s), like every other aggregate; the intent itself /// carries only `k` + the accuracy target. /// /// The single-column reducers (`Sum` / `Min` / `Max` / `Avg` / `StdDev` / /// `Variance` / `Quantile`) carry `col: Option` — the input /// column they reduce, generic over the column-reference state the same way -/// [`QueryExpr`](super::query_expr::QueryExpr) is: positional `ColumnId` once +/// [`PreASAPNode`](super::query_expr::PreASAPNode) is: positional `ColumnId` once /// bound (the default, and every existing use of the bare `AggIntent` name), /// or an unresolved name-based `ColumnRef` for a front end constructing this /// intent directly, before the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has run. @@ -128,7 +128,7 @@ pub enum AggIntent { // ── Time-series streaming derivatives ──────────────────────────────── // Counter-reset adjustment; not equivalent to Sum/Count over a window. - // The temporal range lives on the enclosing `QueryExpr::TimeRange` node, + // The temporal range lives on the enclosing `PreASAPNode::TimeRange` node, // not in the intent — this keeps the intent vocabulary range-agnostic. Rate, /// PromQL `irate(v[w])` — reset-aware rate from the final two samples. @@ -234,7 +234,7 @@ pub enum AggIntent { /// A time / calendar accessor (issue #46) — `timestamp`, `minute`, `hour`, /// `day_of_week`, … over each sample's timestamp (or, for the no-arg forms, /// over the evaluation time). Label-preserving per-series value transform. - /// (`time()` is the evaluation time itself — a `QueryExpr::EvalTimestamp` leaf, + /// (`time()` is the evaluation time itself — a `PreASAPNode::EvalTimestamp` leaf, /// not this.) TimeFn(TimeFunc), @@ -379,7 +379,7 @@ pub enum MathFunc { // `requires` / `is_per_series` / `output_column` never read `col`'s value — // only its presence via a `{ .. }` pattern — so, unlike -// `QueryExpr::output_schema` (which genuinely cannot compile for an +// `PreASAPNode::output_schema` (which genuinely cannot compile for an // unresolved tree — see its own doc), nothing stops these from being generic // over every `C`. And a front end constructing `AggIntent` // directly (issue #179) does need `is_per_series` pre-binding — it decides @@ -529,7 +529,7 @@ impl AggIntent { impl AggIntent { /// Output column name + type produced by this intent over `input`. - /// Used by `QueryExpr::Aggregate`'s schema-derivation rule. The PromQL + /// Used by `PreASAPNode::Aggregate`'s schema-derivation rule. The PromQL /// convention names the column after the intent kind so consumers can /// locate it without an alias lookup. pub fn output_column(&self, input: &Column) -> Column { diff --git a/crates/types/src/pre_asap/canonicalize.rs b/crates/types/src/pre_asap/canonicalize.rs index aaf2c06d..18b36995 100644 --- a/crates/types/src/pre_asap/canonicalize.rs +++ b/crates/types/src/pre_asap/canonicalize.rs @@ -1,4 +1,4 @@ -//! Shared post-lowering canonicalization of the resolved [`QueryExpr`]. +//! Shared post-lowering canonicalization of the resolved [`PreASAPNode`]. //! //! Both language front ends funnel through [`resolve_root`](super::resolve::resolve_root), //! which runs this pass over the resolved tree. Its job is to erase @@ -26,17 +26,17 @@ use std::rc::Rc; use super::agg_intent::{topk, AggIntent}; use super::expr_ir::{CompareOpKind, ScalarValue}; -use super::query_expr::{Predicate, QueryExpr, Reduction, SortKey, WindowFuncKind}; +use super::query_expr::{PreASAPNode, Predicate, Reduction, SortKey, WindowFuncKind}; use crate::types::AccuracyTarget; /// Rewrite `expr` into its canonical form (bottom-up). Idempotent: a tree that /// is already canonical is returned unchanged. -pub fn canonicalize(mut expr: QueryExpr) -> QueryExpr { +pub fn canonicalize(mut expr: PreASAPNode) -> PreASAPNode { canon(&mut expr); expr } -fn canon(expr: &mut QueryExpr) { +fn canon(expr: &mut PreASAPNode) { // A `Concat` asserting a caller-proven `discriminator_unique_key` (issue // #228) had that key's `ColumnId`s resolved, in `resolve.rs`, against // exactly the first branch's output schema *as it stood before this @@ -50,7 +50,7 @@ fn canon(expr: &mut QueryExpr) { // was actually resolved against, right here, before recursing into the // children — this is the exact tree state `resolve.rs` saw. let discriminator_branch_schema_before = match expr { - QueryExpr::Concat { + PreASAPNode::Concat { children, discriminator_unique_key: Some(_), } => children.first().and_then(|c| c.output_schema().ok()), @@ -71,7 +71,7 @@ fn canon(expr: &mut QueryExpr) { // `ConcatDiscriminatorKey`'s soundness doc — so this errs conservatively: // any difference at all (not just a column-count/type change) drops the // key, including the schema becoming undecidable in either direction. - if let QueryExpr::Concat { + if let PreASAPNode::Concat { children, discriminator_unique_key: key @ Some(_), } = expr @@ -94,13 +94,13 @@ fn canon(expr: &mut QueryExpr) { } } -/// A `&mut QueryExpr` out of a child `Rc` — clone-on-write via +/// A `&mut PreASAPNode` out of a child `Rc` — clone-on-write via /// [`Rc::make_mut`]: free (no clone) while `r` is uniquely owned, which is /// the overwhelmingly common case (a tree `canonicalize` was just handed by /// value); falls back to cloning just *this* node (its own fields — the /// grandchildren stay shared `Rc`s, not deep-copied) only when some other /// owner still holds the same `Rc`, e.g. a caller that kept its own clone -/// around (`once.clone()` in `is_idempotent` below — `QueryExpr::clone()` is +/// around (`once.clone()` in `is_idempotent` below — `PreASAPNode::clone()` is /// now a cheap `Rc`-bump, not a deep copy, so that clone shares structure /// with `once` until a rewrite here needs to touch it). `Rc::get_mut` would /// panic on exactly that case; `make_mut` degrades to a shallow copy instead @@ -109,18 +109,18 @@ fn canon(expr: &mut QueryExpr) { /// subtree from a *different* query, this is also the mechanism that keeps /// canonicalizing one query from silently corrupting another's view of the /// same shared node. -fn rc_mut(r: &mut Rc) -> &mut QueryExpr { +fn rc_mut(r: &mut Rc) -> &mut PreASAPNode { Rc::make_mut(r) } -/// Mutable references to the direct **operator** `QueryExpr` children of a +/// Mutable references to the direct **operator** `PreASAPNode` children of a /// node — `canon`'s own top-down/bottom-up walk only ever visits the /// relational skeleton, never descending into a scalar position (`Filter.pred`, /// `ProjectItem.expr`, …): none of the three rewrite rules rewrite anything /// inside a scalar subtree, so there's nothing to gain by recursing into one, /// and every scalar variant (issue #205) hits the catch-all below. -fn children_mut(expr: &mut QueryExpr) -> Vec<&mut QueryExpr> { - use QueryExpr::*; +fn children_mut(expr: &mut PreASAPNode) -> Vec<&mut PreASAPNode> { + use PreASAPNode::*; match expr { // `PromqlScalarBridge`'s child is a scalar-sub-language node (issue // #220), not the relational skeleton — same "no children to recurse @@ -165,9 +165,9 @@ fn children_mut(expr: &mut QueryExpr) -> Vec<&mut QueryExpr> { /// `Limit { Sort { [Project] Aggregate([Count | Sum]) } }` and rewrite it to /// the canonical heavy-hitter `Aggregate([TopK])` over the explicit inner /// aggregate. Returns `None` when the shape does not match. -fn try_promote_additive_top_ranking(expr: &QueryExpr) -> Option { +fn try_promote_additive_top_ranking(expr: &PreASAPNode) -> Option { // Limit k, no offset (an OFFSET means "not the top k"). - let QueryExpr::Limit { + let PreASAPNode::Limit { n: k, offset: 0, child, @@ -176,7 +176,7 @@ fn try_promote_additive_top_ranking(expr: &QueryExpr) -> Option { return None; }; // A single ordering key on a column. - let QueryExpr::Sort { + let PreASAPNode::Sort { keys, partition_by, child: sort_child, @@ -185,7 +185,7 @@ fn try_promote_additive_top_ranking(expr: &QueryExpr) -> Option { return None; }; let [SortKey { - expr: QueryExpr::Column(sort_col), + expr: PreASAPNode::Column(sort_col), ascending, .. }] = keys.as_slice() @@ -197,8 +197,8 @@ fn try_promote_additive_top_ranking(expr: &QueryExpr) -> Option { // projection (a bare-column SELECT list). Map the sort key through the // projection to the aggregate's own output column. let (agg_expr, ranked_col) = match sort_child.as_ref() { - QueryExpr::Project { cols, child, .. } => { - let QueryExpr::Column(underlying) = &cols.get(*sort_col)?.expr else { + PreASAPNode::Project { cols, child, .. } => { + let PreASAPNode::Column(underlying) = &cols.get(*sort_col)?.expr else { return None; }; (child.as_ref(), *underlying) @@ -210,7 +210,7 @@ fn try_promote_additive_top_ranking(expr: &QueryExpr) -> Option { // index `by.len()` (after the group keys). A `PerEntity` reduction has no // `by` to rank a measure against — this shape can't be heavy-hitter // promoted, so it's a non-match rather than an error. - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child: aggregate_child, @@ -243,7 +243,7 @@ fn try_promote_additive_top_ranking(expr: &QueryExpr) -> Option { // post-ASAP IR has no candidate-sidecar + exact-rerank node, so keep that // shape as Sort + Limit instead of treating a sketch estimate as final. if matches!(ranked_agg, AggIntent::Sum { .. }) - && matches!(aggregate_child.as_ref(), QueryExpr::Aggregate { .. }) + && matches!(aggregate_child.as_ref(), PreASAPNode::Aggregate { .. }) { return None; } @@ -257,7 +257,7 @@ fn try_promote_additive_top_ranking(expr: &QueryExpr) -> Option { // Outer heavy-hitter `TopK`, grouped by the ranking's partition (empty for a // global `ORDER BY … LIMIT k`; the `by` labels for a partitioned `topk by`), // over the unchanged inner additive aggregate. - Some(QueryExpr::Aggregate { + Some(PreASAPNode::Aggregate { reduction: Reduction::by(partition_by.to_vec()), measures: vec![AggIntent::TopK { k: *k, accuracy }], output_names: Vec::new(), @@ -272,20 +272,20 @@ fn try_promote_additive_top_ranking(expr: &QueryExpr) -> Option { /// #24). The count-ranked case is then promoted to a heavy-hitter `TopK` by /// [`try_promote_additive_top_ranking`], so a SQL `ROW_NUMBER` top-k and the PromQL /// `topk by (…)` it mirrors converge on the same canonical shape. -fn try_rewrite_rownumber_topk(expr: &QueryExpr) -> Option { +fn try_rewrite_rownumber_topk(expr: &PreASAPNode) -> Option { // Filter { pred: `Column(rn) <= k` }. - let QueryExpr::Filter { pred, child } = expr else { + let PreASAPNode::Filter { pred, child } = expr else { return None; }; let Predicate(pred_expr) = pred; - let QueryExpr::Compare { left, op, right } = pred_expr.as_ref() else { + let PreASAPNode::Compare { left, op, right } = pred_expr.as_ref() else { return None; }; // `rn <= k` (top-k). `rn < k` would be off-by-one; require `<=`. if *op != CompareOpKind::Le { return None; } - let (QueryExpr::Column(rn_col), QueryExpr::Literal(ScalarValue::Int64(k))) = + let (PreASAPNode::Column(rn_col), PreASAPNode::Literal(ScalarValue::Int64(k))) = (left.as_ref(), right.as_ref()) else { return None; @@ -297,8 +297,8 @@ fn try_rewrite_rownumber_topk(expr: &QueryExpr) -> Option { // Optionally strip a passthrough projection (the derived table's SELECT that // re-exposes the aggregate columns + rn), mapping the rn column through it. let (wf_expr, rn_in_wf) = match child.as_ref() { - QueryExpr::Project { cols, child, .. } => { - let QueryExpr::Column(underlying) = &cols.get(*rn_col)?.expr else { + PreASAPNode::Project { cols, child, .. } => { + let PreASAPNode::Column(underlying) = &cols.get(*rn_col)?.expr else { return None; }; (child.as_ref(), *underlying) @@ -308,7 +308,7 @@ fn try_rewrite_rownumber_topk(expr: &QueryExpr) -> Option { // The filtered column must be a `ROW_NUMBER()` window output — the single // column the SQLWindowFunc appends after its input, i.e. the last one. - let QueryExpr::SQLWindowFunc { + let PreASAPNode::SQLWindowFunc { func: WindowFuncKind::RowNumber, partition_by, order_by, @@ -328,10 +328,10 @@ fn try_rewrite_rownumber_topk(expr: &QueryExpr) -> Option { // Generic partitioned top-k. The window's ORDER BY keys are relative to its // input (`inner`), so they transfer directly to a `Sort` over `inner`. - Some(QueryExpr::Limit { + Some(PreASAPNode::Limit { n: *k as usize, offset: 0, - child: Rc::new(QueryExpr::Sort { + child: Rc::new(PreASAPNode::Sort { keys: order_by.clone(), partition_by: partition_by.clone(), child: Rc::new(inner.as_ref().clone()), @@ -349,8 +349,8 @@ mod tests { use crate::pre_asap::schema::{Column, DataType, Schema}; use crate::types::AccuracyTarget; - fn scan() -> QueryExpr { - QueryExpr::Scan { + fn scan() -> PreASAPNode { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -366,8 +366,8 @@ mod tests { } /// `Aggregate{ by: [1], [Count] }` over the scan — output cols `[service, count]`. - fn count_by_service() -> QueryExpr { - QueryExpr::Aggregate { + fn count_by_service() -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::by(vec![1]), measures: vec![AggIntent::Count { accuracy: AccuracyTarget::Exact, @@ -380,33 +380,33 @@ mod tests { fn desc(col: usize) -> Vec { vec![SortKey { - expr: QueryExpr::Column(col), + expr: PreASAPNode::Column(col), ascending: false, nulls_first: false, }] } - fn limit(n: usize, offset: usize, child: QueryExpr) -> QueryExpr { - QueryExpr::Limit { + fn limit(n: usize, offset: usize, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Limit { n, offset, child: Rc::new(child), } } - fn sort(keys: Vec, child: QueryExpr) -> QueryExpr { - QueryExpr::Sort { + fn sort(keys: Vec, child: PreASAPNode) -> PreASAPNode { + PreASAPNode::Sort { keys, partition_by: GroupKeys::by(vec![]), child: Rc::new(child), } } - fn is_topk_over_count(qe: &QueryExpr) -> bool { + fn is_topk_over_count(qe: &PreASAPNode) -> bool { matches!(qe, - QueryExpr::Aggregate { measures, child, .. } + PreASAPNode::Aggregate { measures, child, .. } if matches!(measures.as_slice(), [AggIntent::TopK { k: 5, .. }]) - && matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + && matches!(child.as_ref(), PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Count { .. }]))) } @@ -420,15 +420,15 @@ mod tests { #[test] fn promotes_through_a_passthrough_projection() { // …with a `SELECT service, count` projection between the Sort and the Agg. - let proj = QueryExpr::Project { + let proj = PreASAPNode::Project { cols: vec![ ProjectItem { alias: None, - expr: QueryExpr::Column(0), + expr: PreASAPNode::Column(0), }, ProjectItem { alias: Some("c".into()), - expr: QueryExpr::Column(1), + expr: PreASAPNode::Column(1), }, ], qualifier: None, @@ -460,12 +460,12 @@ mod tests { fn concat_discriminator_key_survives_canonicalize_when_first_branch_is_unaffected() { // A plain `Aggregate` first branch matches neither rewrite trigger, // so its schema is identical before and after canonicalize. - let q = QueryExpr::concat_with_discriminator( + let q = PreASAPNode::concat_with_discriminator( vec![count_by_service(), count_by_service()], /* discriminator */ 0, /* inner_key */ vec![1], ); - let QueryExpr::Concat { + let PreASAPNode::Concat { discriminator_unique_key, .. } = canonicalize(q) @@ -489,12 +489,12 @@ mod tests { // at index 0, `inner_key` = `count` at index 1) must not silently // survive pointing at the new 1-column schema. let promotable_branch = limit(5, 0, sort(desc(1), count_by_service())); - let q = QueryExpr::concat_with_discriminator( + let q = PreASAPNode::concat_with_discriminator( vec![promotable_branch, count_by_service()], /* discriminator */ 0, /* inner_key */ vec![1], ); - let QueryExpr::Concat { + let PreASAPNode::Concat { children, discriminator_unique_key, } = canonicalize(q) @@ -517,7 +517,7 @@ mod tests { // rejects it (needs descending), so it stays a generic Sort+Limit — the // same call PromQL `bottomk` makes (issue #38). let asc = vec![SortKey { - expr: QueryExpr::Column(1), + expr: PreASAPNode::Column(1), ascending: true, nulls_first: false, }]; @@ -541,7 +541,7 @@ mod tests { #[test] fn promotes_sum_ranked_limit_sort_as_weighted_heavy_hitter() { - let sum = QueryExpr::Aggregate { + let sum = PreASAPNode::Aggregate { reduction: Reduction::by(vec![1]), measures: vec![AggIntent::Sum { col: None }], output_names: vec![], @@ -550,7 +550,7 @@ mod tests { }; let q = limit(5, 0, sort(desc(1), sum)); let out = canonicalize(q); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { measures, child, .. } = out else { @@ -561,7 +561,7 @@ mod tests { [AggIntent::TopK { k: 5, .. }] )); assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + matches!(child.as_ref(), PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Sum { .. }])) ); } @@ -569,14 +569,14 @@ mod tests { #[test] fn keeps_sum_over_counter_reduction_as_exact_value_ranking() { for counter in [AggIntent::Rate, AggIntent::Increase] { - let derived = QueryExpr::Aggregate { + let derived = PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![counter], output_names: vec![], having: None, child: Rc::new(scan()), }; - let sum = QueryExpr::Aggregate { + let sum = PreASAPNode::Aggregate { reduction: Reduction::by(vec![1]), measures: vec![AggIntent::Sum { col: None }], output_names: vec![], @@ -584,19 +584,19 @@ mod tests { child: Rc::new(derived), }; let out = canonicalize(limit(5, 0, sort(desc(1), sum))); - assert!(matches!(out, QueryExpr::Limit { child, .. } - if matches!(child.as_ref(), QueryExpr::Sort { child, .. } - if matches!(child.as_ref(), QueryExpr::Aggregate { measures, child, .. } + assert!(matches!(out, PreASAPNode::Limit { child, .. } + if matches!(child.as_ref(), PreASAPNode::Sort { child, .. } + if matches!(child.as_ref(), PreASAPNode::Aggregate { measures, child, .. } if matches!(measures.as_slice(), [AggIntent::Sum { .. }]) - && matches!(child.as_ref(), QueryExpr::Aggregate { .. }))))); + && matches!(child.as_ref(), PreASAPNode::Aggregate { .. }))))); } } // ── ROW_NUMBER() partitioned top-k (issue #24) ────────────────────────── /// A scan with `[ts, service, region, value]`. - fn scan4() -> QueryExpr { - QueryExpr::Scan { + fn scan4() -> PreASAPNode { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -614,8 +614,8 @@ mod tests { /// `Aggregate{ by: [1,2] (service, region), [agg] }` — output `[service, /// region, ]` (3 cols), so a ROW_NUMBER over it appends `rn` at index 3. - fn grouped(agg: AggIntent) -> QueryExpr { - QueryExpr::Aggregate { + fn grouped(agg: AggIntent) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::by(vec![1, 2]), measures: vec![agg], output_names: vec![], @@ -636,13 +636,13 @@ mod tests { /// `Filter{ rn(3) <= 5 } { SQLWindowFunc{ RowNumber, PARTITION BY region(2), /// ORDER BY col(2) DESC } { agg } }`. - fn rownumber_topk(agg: QueryExpr) -> QueryExpr { - let wf = QueryExpr::SQLWindowFunc { + fn rownumber_topk(agg: PreASAPNode) -> PreASAPNode { + let wf = PreASAPNode::SQLWindowFunc { func: WindowFuncKind::RowNumber, args: vec![], partition_by: GroupKeys::by(vec![2]), // region order_by: vec![SortKey { - expr: QueryExpr::Column(2), // the aggregate output column + expr: PreASAPNode::Column(2), // the aggregate output column ascending: false, nulls_first: true, }], @@ -650,11 +650,11 @@ mod tests { output_name: "rn".into(), child: Rc::new(agg), }; - QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), // rn = the appended window column + PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(3)), // rn = the appended window column op: CompareOpKind::Le, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(5))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Int64(5))), })), child: Rc::new(wf), } @@ -668,7 +668,7 @@ mod tests { accuracy: AccuracyTarget::Exact, })); let out = canonicalize(q); - let QueryExpr::Aggregate { + let PreASAPNode::Aggregate { reduction, measures, child, @@ -686,7 +686,7 @@ mod tests { )); assert_eq!(**by, vec![2], "outer TopK partitioned by region"); assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + matches!(child.as_ref(), PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Count { .. }])) ); } @@ -697,11 +697,11 @@ mod tests { // top-k: Limit{5}{ Sort{ partition_by: [region] } }. let q = rownumber_topk(grouped(AggIntent::Avg { col: None })); let out = canonicalize(q); - let QueryExpr::Limit { n, child, .. } = &out else { + let PreASAPNode::Limit { n, child, .. } = &out else { panic!("expected a Limit, got {out:?}"); }; assert_eq!(*n, 5); - let QueryExpr::Sort { + let PreASAPNode::Sort { partition_by, child, .. @@ -711,7 +711,7 @@ mod tests { }; assert_eq!(**partition_by, vec![2], "partitioned by region"); assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + matches!(child.as_ref(), PreASAPNode::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Avg { .. }])) ); } @@ -720,12 +720,12 @@ mod tests { fn filter_on_a_non_rownumber_column_is_left_alone() { // `WHERE service_len <= 5` (col 0, not the rn window column) must not be // mistaken for a top-k. - let wf = QueryExpr::SQLWindowFunc { + let wf = PreASAPNode::SQLWindowFunc { func: WindowFuncKind::RowNumber, args: vec![], partition_by: GroupKeys::by(vec![2]), order_by: vec![SortKey { - expr: QueryExpr::Column(2), + expr: PreASAPNode::Column(2), ascending: false, nulls_first: true, }], @@ -735,16 +735,16 @@ mod tests { accuracy: AccuracyTarget::Exact, })), }; - let q = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), // NOT the rn column (index 3) + let q = PreASAPNode::Filter { + pred: Predicate(Rc::new(PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(0)), // NOT the rn column (index 3) op: CompareOpKind::Le, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(5))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Int64(5))), })), child: Rc::new(wf), }; assert!( - matches!(canonicalize(q), QueryExpr::Filter { .. }), + matches!(canonicalize(q), PreASAPNode::Filter { .. }), "left as a Filter" ); } diff --git a/crates/types/src/pre_asap/column_resolution.rs b/crates/types/src/pre_asap/column_resolution.rs index f1db980a..f32c9a39 100644 --- a/crates/types/src/pre_asap/column_resolution.rs +++ b/crates/types/src/pre_asap/column_resolution.rs @@ -14,8 +14,8 @@ use thiserror::Error; use super::agg_intent::AggIntent; use super::expr_ir::ColumnRef; use super::query_expr::{ - aggregate_output_schema, GroupKeys, QueryExpr, QueryExprError, Reduction, ResolvedQueryExpr, - UnresolvedQueryExpr, + aggregate_output_schema, GroupKeys, PreASAPNode, PreASAPNodeError, Reduction, + ResolvedPreASAPNode, UnresolvedPreASAPNode, }; use super::schema::{ColumnId, DataType, Schema}; @@ -112,64 +112,64 @@ pub fn resolve_group_keys_promql( .collect() } -/// Resolve a name-based scalar [`UnresolvedQueryExpr`] (one of `QueryExpr`'s scalar -/// variants, issue #205) into a positional [`ResolvedQueryExpr`] by resolving every +/// Resolve a name-based scalar [`UnresolvedPreASAPNode`] (one of `PreASAPNode`'s scalar +/// variants, issue #205) into a positional [`ResolvedPreASAPNode`] by resolving every /// column reference against `schema`. Structural otherwise. `expr` must be /// one of the scalar variants — an operator variant here is a construction /// bug, not a shape this needs to handle silently. pub fn resolve_expr( - expr: &UnresolvedQueryExpr, + expr: &UnresolvedPreASAPNode, schema: &Schema, -) -> Result { - let rc = |e: &UnresolvedQueryExpr| -> Result, ResolveError> { +) -> Result { + let rc = |e: &UnresolvedPreASAPNode| -> Result, ResolveError> { Ok(Rc::new(resolve_expr(e, schema)?)) }; - let each = |es: &[UnresolvedQueryExpr]| -> Result, ResolveError> { + let each = |es: &[UnresolvedPreASAPNode]| -> Result, ResolveError> { es.iter().map(|e| resolve_expr(e, schema)).collect() }; Ok(match expr { - QueryExpr::Column(c) => QueryExpr::Column(resolve_column_ref(c, schema)?), - QueryExpr::Literal(s) => QueryExpr::Literal(s.clone()), - QueryExpr::EvalTimestamp => QueryExpr::EvalTimestamp, - QueryExpr::CurrentTimestamp => QueryExpr::CurrentTimestamp, - QueryExpr::Compare { left, op, right } => QueryExpr::Compare { + PreASAPNode::Column(c) => PreASAPNode::Column(resolve_column_ref(c, schema)?), + PreASAPNode::Literal(s) => PreASAPNode::Literal(s.clone()), + PreASAPNode::EvalTimestamp => PreASAPNode::EvalTimestamp, + PreASAPNode::CurrentTimestamp => PreASAPNode::CurrentTimestamp, + PreASAPNode::Compare { left, op, right } => PreASAPNode::Compare { left: rc(left)?, op: op.clone(), right: rc(right)?, }, - QueryExpr::BoolAnd(v) => QueryExpr::BoolAnd(each(v)?), - QueryExpr::BoolOr(v) => QueryExpr::BoolOr(each(v)?), - QueryExpr::Not(e) => QueryExpr::Not(rc(e)?), - QueryExpr::IsNull(e) => QueryExpr::IsNull(rc(e)?), - QueryExpr::IsNotNull(e) => QueryExpr::IsNotNull(rc(e)?), - QueryExpr::Cast { expr, to, try_cast } => QueryExpr::Cast { + PreASAPNode::BoolAnd(v) => PreASAPNode::BoolAnd(each(v)?), + PreASAPNode::BoolOr(v) => PreASAPNode::BoolOr(each(v)?), + PreASAPNode::Not(e) => PreASAPNode::Not(rc(e)?), + PreASAPNode::IsNull(e) => PreASAPNode::IsNull(rc(e)?), + PreASAPNode::IsNotNull(e) => PreASAPNode::IsNotNull(rc(e)?), + PreASAPNode::Cast { expr, to, try_cast } => PreASAPNode::Cast { expr: rc(expr)?, to: to.clone(), try_cast: *try_cast, }, - QueryExpr::InList { + PreASAPNode::InList { expr, list, negated, - } => QueryExpr::InList { + } => PreASAPNode::InList { expr: rc(expr)?, list: each(list)?, negated: *negated, }, - QueryExpr::FunctionCall { name, args } => QueryExpr::FunctionCall { + PreASAPNode::FunctionCall { name, args } => PreASAPNode::FunctionCall { name: name.clone(), args: each(args)?, }, - QueryExpr::Arithmetic { op, left, right } => QueryExpr::Arithmetic { + PreASAPNode::Arithmetic { op, left, right } => PreASAPNode::Arithmetic { op: op.clone(), left: rc(left)?, right: rc(right)?, }, - QueryExpr::Case { + PreASAPNode::Case { operand, branches, else_expr, - } => QueryExpr::Case { + } => PreASAPNode::Case { operand: operand.as_deref().map(&rc).transpose()?, branches: branches .iter() @@ -177,12 +177,12 @@ pub fn resolve_expr( .collect::, ResolveError>>()?, else_expr: else_expr.as_deref().map(&rc).transpose()?, }, - other => unreachable!("resolve_expr called on a non-scalar QueryExpr variant: {other:?}"), + other => unreachable!("resolve_expr called on a non-scalar PreASAPNode variant: {other:?}"), }) } /// Output schema produced by an `Aggregate { by, measures }` over `input`. -/// Mirrors `QueryExpr::output_schema_in`'s `Aggregate` arm; out-of-range `by` +/// Mirrors `PreASAPNode::output_schema_in`'s `Aggregate` arm; out-of-range `by` /// ids are silently dropped (callers needing the strict check resolve `by` /// via [`resolve_column_refs`], which surfaces `NotFound`). pub fn output_schema_for_aggregate( @@ -190,9 +190,9 @@ pub fn output_schema_for_aggregate( by: &GroupKeys, measures: &[AggIntent], output_names: &[String], -) -> Result { +) -> Result { // Delegate to the single canonical derivation so HAVING resolution can never - // drift from `QueryExpr::output_schema_in` (issue #41). HAVING is SQL-only + // drift from `PreASAPNode::output_schema_in` (issue #41). HAVING is SQL-only // and cross-series (SQL has no `without`), but detect the child-independent // per-entity case anyway (a lone `rate`/`increase`/`*_over_time` intent) so // the two agree on every shared input — the range-window child marker the @@ -347,7 +347,7 @@ mod tests { #[test] fn having_schema_agrees_with_canonical_for_a_per_series_reduction() { // Issue #41: `output_schema_for_aggregate` (HAVING resolution) and the - // canonical `QueryExpr::output_schema_in` must produce identical schemas + // canonical `PreASAPNode::output_schema_in` must produce identical schemas // for the same aggregate. Before the dedup this diverged on a per-series // reduction — the HAVING mirror lacked the per-series branch and would // collapse `[ts, value]` to a single `rate` column. @@ -362,19 +362,19 @@ mod tests { 0, vec![], ); - let scan = QueryExpr::Scan { + let scan = PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: leaf_schema.clone(), }; // Aggregate{ reduction: PerEntity, [Rate], child: TimeRange{ Scan } } — // a per-series reduction (label-preserving). - let agg = QueryExpr::Aggregate { + let agg = PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Rate], output_names: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: Rc::new(PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(scan), }), diff --git a/crates/types/src/pre_asap/cse.rs b/crates/types/src/pre_asap/cse.rs index cb6aa69a..9d60cf35 100644 --- a/crates/types/src/pre_asap/cse.rs +++ b/crates/types/src/pre_asap/cse.rs @@ -1,5 +1,5 @@ //! Pre-ASAP structural common-subexpression elimination: bottom-up -//! hash-consing over an already-`resolve_root`'d [`QueryExpr`] tree (issue +//! hash-consing over an already-`resolve_root`'d [`PreASAPNode`] tree (issue //! #212, #222, #223). //! //! CSE only runs on an already-bound, already-canonicalized tree — @@ -27,7 +27,7 @@ //! subexpression reachable only through a wrapper position (`Predicate`, //! `ProjectItem.expr`, `Aggregate.having`, `SQLWindowFunc.args`, …) stays //! embedded as opaque data on its owning operator node, compared by -//! `QueryExpr`'s derived `PartialEq` along with the rest of that node's +//! `PreASAPNode`'s derived `PartialEq` along with the rest of that node's //! fields, rather than separately hash-consed — the same scope //! `canonicalize.rs` settled on ("none of the rewrite rules touch a scalar //! subtree, so there's nothing to gain by recursing into one"). Widening this @@ -94,7 +94,7 @@ //! `share_common_subtrees`-actual sharing — see `dag_export`'s module doc). //! Stage 4 (issue #237) is implemented in //! `asap_aware_mapping::cost_model::CostModel::cse_share_decision`, called -//! from `asap_aware_mapping::replacement::CandidatePostASAPDAGs::cost_sorted` (via that +//! from `asap_aware_mapping::replacement::CandidateLogicalPostASAPDAGs::cost_sorted` (via that //! module's own `cse_preference`) — a real, Volcano/Cascades-style cost //! comparison over what this module detects, not a fixed rule. See //! `docs/design_docs/cost-model.md`. This module's own @@ -107,10 +107,10 @@ use std::collections::HashMap; use std::hash::{Hash, Hasher}; use std::rc::Rc; -use super::query_expr::QueryExpr; +use super::query_expr::PreASAPNode; /// Bottom-up hash-consing table: structurally-equal, sharing-legal -/// [`QueryExpr`] nodes collapse onto one `Rc`. +/// [`PreASAPNode`] nodes collapse onto one `Rc`. /// /// `buckets` is keyed by [`structural_hash`] — a coarse candidate filter /// only (see the module-level "Correctness" section). Every entry within one @@ -118,7 +118,7 @@ use super::query_expr::QueryExpr; /// actually decides a match; a hash collision between structurally different /// nodes just means a (harmless) linear scan of a few extra candidates. struct InternTable { - buckets: HashMap>>, + buckets: HashMap>>, /// Memoizes [`structural_hash`] per already-hashed `Rc` pointer, shared /// across every [`intern`](Self::intern) call for the table's whole /// lifetime — see [`structural_hash`]'s own doc on why this matters: @@ -140,7 +140,7 @@ impl InternTable { /// [`structural_hash`], confirm with `PartialEq`, and — only when /// sharing is legal (see "Legality" above) — return the existing `Rc` /// instead of allocating a new one. - fn intern(&mut self, node: QueryExpr) -> Rc { + fn intern(&mut self, node: PreASAPNode) -> Rc { let hash = structural_hash(&node, &mut self.hash_cache); // A node with no provable unique key is never *returned* as a match // for something else — it may still go on to occupy a fresh slot in @@ -162,7 +162,7 @@ impl InternTable { } /// [`structural_hash`]'s memoization cache: maps an already-hashed node's -/// `Rc` pointer to its computed hash. Not tied to any one `QueryExpr` — a +/// `Rc` pointer to its computed hash. Not tied to any one `PreASAPNode` — a /// fresh, empty cache is correct to start with anywhere; what matters is /// letting it *persist* across every node in one bottom-up pass (as /// [`InternTable`] does via its own `hash_cache` field), rather than @@ -174,12 +174,12 @@ impl InternTable { /// parallel reimplementation — the same "one real hash, reused everywhere /// it's needed" rationale [`structural_hash`]'s own doc gives for /// [`dag_export`](crate::dag_export)'s `pub(crate)` reuse. -pub type HashCache = HashMap<*const QueryExpr, u64>; +pub type HashCache = HashMap<*const PreASAPNode, u64>; /// Coarse structural hash used only to bucket [`InternTable::intern`]'s /// candidate search — never the actual sharing decision (`PartialEq` is). /// -/// `QueryExpr` carries `f64`s (`Literal(ScalarValue::Float64)`, `AggIntent::Quantile.q`, …), so it +/// `PreASAPNode` carries `f64`s (`Literal(ScalarValue::Float64)`, `AggIntent::Quantile.q`, …), so it /// cannot derive `std::hash::Hash`. Serializing to a canonical JSON string /// and hashing that sidesteps the `f64` problem — but only for `node`'s own /// tag and non-child fields, *not* its children's full values: each @@ -218,13 +218,13 @@ pub type HashCache = HashMap<*const QueryExpr, u64>; /// candidate-narrowing filter this module's own [`InternTable::intern`] /// already uses, so it doesn't have to reinvent (and risk drifting from) it. /// -/// Exhaustive over every `QueryExpr` variant, matching [`rebuild_children`] +/// Exhaustive over every `PreASAPNode` variant, matching [`rebuild_children`] /// in which fields count as an operator child (must stay in sync — a new /// variant fails to compile in both places until both are extended). -pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { - use QueryExpr::*; +pub fn structural_hash(node: &PreASAPNode, cache: &mut HashCache) -> u64 { + use PreASAPNode::*; - fn child_hash(child: &Rc, cache: &mut HashCache) -> u64 { + fn child_hash(child: &Rc, cache: &mut HashCache) -> u64 { let ptr = Rc::as_ptr(child); if let Some(&h) = cache.get(&ptr) { return h; @@ -236,7 +236,7 @@ pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { /// Hash `own_fields` (this node's own tag and non-child scalar /// fields — anything JSON-serializable and small, i.e. never a - /// `QueryExpr` subtree) via the same canonical-JSON-string trick the + /// `PreASAPNode` subtree) via the same canonical-JSON-string trick the /// whole-subtree version used, just applied to `O(1)` fields instead /// of `O(subtree size)`. fn hash_own_fields(hasher: &mut impl Hasher, own_fields: &impl serde::Serialize) { @@ -456,26 +456,29 @@ pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { /// counted as part of its owning operator node, the same node /// `rebuild_children` treats as a single opaque leaf for interning /// purposes. -pub fn dag_node_count(root: &QueryExpr) -> usize { - let mut seen: std::collections::HashSet<*const QueryExpr> = std::collections::HashSet::new(); +pub fn dag_node_count(root: &PreASAPNode) -> usize { + let mut seen: std::collections::HashSet<*const PreASAPNode> = std::collections::HashSet::new(); count_unique(root, &mut seen) } /// One node's own contribution (`1`) plus each *not-yet-seen* operator -/// child's contribution — exhaustive over every `QueryExpr` variant, +/// child's contribution — exhaustive over every `PreASAPNode` variant, /// enumerating the same fields [`rebuild_children`] does (kept as a /// separate, read-only traversal rather than threaded through /// `rebuild_children` itself, since that function consumes and rebuilds /// its input while this one only ever reads it). -fn count_unique(node: &QueryExpr, seen: &mut std::collections::HashSet<*const QueryExpr>) -> usize { - use QueryExpr::*; +fn count_unique( + node: &PreASAPNode, + seen: &mut std::collections::HashSet<*const PreASAPNode>, +) -> usize { + use PreASAPNode::*; /// Visit one `Rc`-held child: counts (and recurses into) it only the /// first time its pointer is seen, `0` on every later occurrence — /// this is the actual dedup step. fn visit( - child: &Rc, - seen: &mut std::collections::HashSet<*const QueryExpr>, + child: &Rc, + seen: &mut std::collections::HashSet<*const PreASAPNode>, ) -> usize { if seen.insert(Rc::as_ptr(child)) { count_unique(child, seen) @@ -503,8 +506,8 @@ fn count_unique(node: &QueryExpr, seen: &mut std::collections::HashSet<*const Qu | TimeRange { child, .. } | TimeShift { child, .. } | SQLWindowFunc { child, .. } => visit(child, seen), - // `Concat`'s branches are stored by value (`Vec`, not - // `Rc` — see `rebuild_children`'s `intern_owned` use for + // `Concat`'s branches are stored by value (`Vec`, not + // `Rc` — see `rebuild_children`'s `intern_owned` use for // this variant), so a branch has no `Rc` identity of its own to // dedup on at this position; still recurse into each in case an // `Rc`-shared descendant appears further down. @@ -541,7 +544,7 @@ fn count_unique(node: &QueryExpr, seen: &mut std::collections::HashSet<*const Qu /// went through a previous `share_common_subtrees` pass; a structural /// duplicate collapses right back onto `child` itself via `PartialEq`, an /// already-optimal no-op. -fn intern_child(table: &mut InternTable, child: Rc) -> Rc { +fn intern_child(table: &mut InternTable, child: Rc) -> Rc { match Rc::try_unwrap(child) { Ok(owned) => intern_bottom_up(table, owned), Err(shared) => intern_bottom_up(table, (*shared).clone()), @@ -549,32 +552,32 @@ fn intern_child(table: &mut InternTable, child: Rc) -> Rc } /// Like [`intern_child`], for a `Concat` branch — stored by value -/// (`Vec`, not `Rc`), so this position itself can never +/// (`Vec`, not `Rc`), so this position itself can never /// alias another parent. Interning it anyway still lets any `Rc`-typed /// descendant of the branch participate in sharing, and registers the /// branch's own hash/value in the table for a *different* `Concat` elsewhere /// with a structurally identical branch (which — being in its own `Vec` /// slot too — still can't literally share the `Rc`, but this keeps the /// interning behavior uniform and the table's bucket contents consistent). -fn intern_owned(table: &mut InternTable, expr: QueryExpr) -> QueryExpr { +fn intern_owned(table: &mut InternTable, expr: PreASAPNode) -> PreASAPNode { let rc = intern_bottom_up(table, expr); Rc::try_unwrap(rc).unwrap_or_else(|shared| (*shared).clone()) } /// Bottom-up: rebuild `expr`'s children (recursively interning each), then /// intern the rebuilt node itself. -fn intern_bottom_up(table: &mut InternTable, expr: QueryExpr) -> Rc { +fn intern_bottom_up(table: &mut InternTable, expr: PreASAPNode) -> Rc { let rebuilt = rebuild_children(table, expr); table.intern(rebuilt) } /// Rebuild `expr` with each **operator** child (see the module doc on scope) -/// replaced by its interned `Rc`. Exhaustive over every `QueryExpr` variant, +/// replaced by its interned `Rc`. Exhaustive over every `PreASAPNode` variant, /// matching `canonicalize.rs`'s `children_mut` exactly in which fields count /// as an operator child — new variants fail to compile here until this match /// is extended. -fn rebuild_children(table: &mut InternTable, expr: QueryExpr) -> QueryExpr { - use QueryExpr::*; +fn rebuild_children(table: &mut InternTable, expr: PreASAPNode) -> PreASAPNode { + use PreASAPNode::*; match expr { Scan { .. } | EvalTimestamp | CurrentTimestamp => expr, PromqlVectorFromScalar(c) => PromqlVectorFromScalar(intern_child(table, c)), @@ -750,7 +753,7 @@ fn rebuild_children(table: &mut InternTable, expr: QueryExpr) -> QueryExpr { /// `Id` is caller-chosen — a `QueryWorkload` entry's own key, an index, a /// query name, whatever identifies one root through the pipeline; this /// module has no opinion on its shape. -pub fn share_common_subtrees(roots: Vec<(Id, QueryExpr)>) -> Vec<(Id, Rc)> { +pub fn share_common_subtrees(roots: Vec<(Id, PreASAPNode)>) -> Vec<(Id, Rc)> { let mut table = InternTable::new(); roots .into_iter() @@ -767,8 +770,8 @@ mod tests { use crate::types::AccuracyTarget; /// `[ts, service, value, latency]`. - fn scan() -> QueryExpr { - QueryExpr::Scan { + fn scan() -> PreASAPNode { + PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -784,8 +787,8 @@ mod tests { } } - fn quantile_agg(by: Vec, col: Option, q: f64) -> QueryExpr { - QueryExpr::Aggregate { + fn quantile_agg(by: Vec, col: Option, q: f64) -> PreASAPNode { + PreASAPNode::Aggregate { reduction: Reduction::by(by), measures: vec![AggIntent::Quantile { col, @@ -871,7 +874,7 @@ mod tests { // workload of size 1 still interns bottom-up within this one tree — // no separate single-query mechanism needed. let agg = quantile_agg(vec![1], Some(2), 0.5); - let root = QueryExpr::BinaryOp { + let root = PreASAPNode::BinaryOp { op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), lhs: Rc::new(agg.clone()), rhs: Rc::new(agg), @@ -881,7 +884,7 @@ mod tests { let [(_, root)] = shared.as_slice() else { panic!("expected 1 root"); }; - let QueryExpr::BinaryOp { lhs, rhs, .. } = root.as_ref() else { + let PreASAPNode::BinaryOp { lhs, rhs, .. } = root.as_ref() else { panic!("expected BinaryOp root, got {root:?}"); }; assert!( @@ -920,13 +923,13 @@ mod tests { // whole point of memoization is not changing the answer, only the // work needed to reach it. let agg = quantile_agg(vec![1], Some(2), 0.5); - let shared_root = QueryExpr::BinaryOp { + let shared_root = PreASAPNode::BinaryOp { op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), lhs: Rc::new(agg.clone()), rhs: Rc::new(agg.clone()), vector_match: None, }; - let unshared_root = QueryExpr::BinaryOp { + let unshared_root = PreASAPNode::BinaryOp { op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), lhs: Rc::new(agg.clone()), rhs: Rc::new(agg), // a second, independently-allocated Rc with an equal value @@ -949,7 +952,7 @@ mod tests { // recursive walk) for the second occurrence. let agg = quantile_agg(vec![1], Some(2), 0.5); let shared = Rc::new(agg); - let root = QueryExpr::BinaryOp { + let root = PreASAPNode::BinaryOp { op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), lhs: Rc::clone(&shared), rhs: Rc::clone(&shared), @@ -986,7 +989,7 @@ mod tests { // not 5 (which a tree-walk / naive serialization, counting the // shared branch's 2 nodes twice, would report). let agg = quantile_agg(vec![1], Some(2), 0.5); - let root = QueryExpr::BinaryOp { + let root = PreASAPNode::BinaryOp { op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), lhs: Rc::new(agg.clone()), rhs: Rc::new(agg), @@ -1033,7 +1036,7 @@ mod tests { // `Dedup { cols }` adds `cols` as a unique key — so two identical // `Dedup` subtrees over a keyed column *do* merge, exercising the // legality gate on a non-`Aggregate` node. - let dedup = |cols: Vec| QueryExpr::Dedup { + let dedup = |cols: Vec| PreASAPNode::Dedup { cols, child: Rc::new(scan()), }; @@ -1055,7 +1058,7 @@ mod tests { // grouping stays open (no unique key) even though `by` is // non-empty-shaped structurally, so two identical `without` groups // do not merge under the same gate that blocks the ungrouped case. - let without_agg = || QueryExpr::Aggregate { + let without_agg = || PreASAPNode::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(vec![0])), measures: vec![AggIntent::Count { accuracy: AccuracyTarget::Exact, diff --git a/crates/types/src/pre_asap/expr_ir.rs b/crates/types/src/pre_asap/expr_ir.rs index 2aed64aa..6892f302 100644 --- a/crates/types/src/pre_asap/expr_ir.rs +++ b/crates/types/src/pre_asap/expr_ir.rs @@ -1,12 +1,12 @@ //! Column-reference and scalar-operator vocabulary shared by the whole -//! canonical [`QueryExpr`](super::query_expr::QueryExpr) tree. +//! canonical [`PreASAPNode`](super::query_expr::PreASAPNode) tree. //! //! Issue #205: the scalar expression shapes (`Column`/`Literal`/`Compare`/…) //! used to live in a separate, self-recursive `Expr` tree here, reachable -//! from `QueryExpr` only through wrapper fields (`Predicate`, `ProjectItem`, -//! `SortKey`). They're variants of `QueryExpr` itself now — one recursive +//! from `PreASAPNode` only through wrapper fields (`Predicate`, `ProjectItem`, +//! `SortKey`). They're variants of `PreASAPNode` itself now — one recursive //! tree, not two type families joined by wrappers — generic over the same -//! column-reference state `C` the rest of `QueryExpr` already carries +//! column-reference state `C` the rest of `PreASAPNode` already carries //! (issue #179): [`ColumnRef`] (name-based, front-end-emitted) or //! [`ColumnId`](super::schema::ColumnId) (positional, once bound). //! @@ -19,7 +19,7 @@ use serde::{Deserialize, Serialize}; /// A name-based column reference — the front-end-emitted, unresolved state of -/// [`QueryExpr::Column`](super::query_expr::QueryExpr::Column) (`C = +/// [`PreASAPNode::Column`](super::query_expr::PreASAPNode::Column) (`C = /// ColumnRef`); the [`SchemaResolver`](super::schema_resolver::SchemaResolver) resolves it to a /// positional [`ColumnId`](super::schema::ColumnId). Includes the two /// PromQL-conventional synthetic columns. diff --git a/crates/types/src/pre_asap/mod.rs b/crates/types/src/pre_asap/mod.rs index 0f656022..2558415a 100644 --- a/crates/types/src/pre_asap/mod.rs +++ b/crates/types/src/pre_asap/mod.rs @@ -1,24 +1,24 @@ //! The canonical pre-ASAP intent algebra IR. //! //! - [`query_expr`] — the canonical, language- and deployment-independent -//! intent algebra: one recursive [`QueryExpr`] tree (relational operators +//! intent algebra: one recursive [`PreASAPNode`] tree (relational operators //! *and* scalar expression shapes both, since issue #205) + [`AggIntent`], //! generic over the column-reference state (positional [`ColumnId`] once //! bound, name-based [`ColumnRef`] before). //! - [`agg_intent`] — the aggregation-intent vocabulary. //! - [`expr_ir`] — the [`ColumnRef`] column-reference type and the scalar //! operator/literal vocabulary ([`ScalarValue`], [`CompareOpKind`], [`ArithmeticOpKind`]) -//! [`QueryExpr`]'s scalar variants are built from. +//! [`PreASAPNode`]'s scalar variants are built from. //! - [`schema`] — the per-edge [`Schema`] every node carries. //! - [`schema_resolver`] / [`column_resolution`] — name resolution: turn a `ColumnRef` //! into a positional `ColumnId` against an in-scope [`Schema`]. -//! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedQueryExpr`] tree to -//! canonical [`ResolvedQueryExpr`] (issue #179): both front ends -//! (`asap-frontend-promql`, `asap-frontend-sql`) construct `UnresolvedQueryExpr` +//! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedPreASAPNode`] tree to +//! canonical [`ResolvedPreASAPNode`] (issue #179): both front ends +//! (`asap-frontend-promql`, `asap-frontend-sql`) construct `UnresolvedPreASAPNode` //! directly during their own `interpret` step and call //! [`resolve_root`] on the result — there is no separate per-language //! relational tree or converter anymore. -//! - [`canonicalize`] — post-lowering structural normalization of [`QueryExpr`] +//! - [`canonicalize`] — post-lowering structural normalization of [`PreASAPNode`] //! (issue #34), run by [`resolve_root`]. //! - [`cse`] — workload-level structural common-subexpression elimination //! over an already-`resolve_root`'d tree (issue #212, #222, #223), run @@ -28,7 +28,7 @@ //! Formerly the separate `asap-l2` crate; folded in here since //! `schema_resolver`/`column_resolution`/`canonicalize`/`resolve` have no //! front-end-specific logic — they operate directly on this crate's own -//! `QueryExpr`. +//! `PreASAPNode`. pub mod agg_intent; pub mod canonicalize; @@ -54,11 +54,19 @@ pub use cse::share_common_subtrees; pub use expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; pub use query_expr::{ aggregate_output_schema, AtModifier, BinaryOpKind, ColState, DataModel, GroupKeys, GroupSide, - InfoMatcher, JoinKind, Predicate, ProjectItem, PromQLVectorSetOpKind, QueryExpr, - QueryExprError, Reduction, RelationalSetOpKind, ResolvedQueryExpr, SampleKind, SortKey, Source, - TimeShift, UnresolvedQueryExpr, VectorGrouping, VectorMatch, VectorMatchKind, WindowFrame, - WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, + InfoMatcher, JoinKind, PreASAPNode, PreASAPNodeError, Predicate, ProjectItem, + PromQLVectorSetOpKind, Reduction, RelationalSetOpKind, ResolvedPreASAPNode, SampleKind, + SortKey, Source, TimeShift, UnresolvedPreASAPNode, VectorGrouping, VectorMatch, + VectorMatchKind, WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, + WindowFuncKind, }; pub use resolve::{resolve_root, ResolveTreeError}; pub use schema::{Column, ColumnId, DataType, Schema}; pub use schema_resolver::{SchemaCatalog, SchemaResolver, UsageDerivedCatalog}; + +/// Shared root of a frontend-lowered query graph. +pub type PreASAPDAG = std::rc::Rc>; + +/// Frontend candidates keyed by workload entry. Current frontends produce one +/// lowering per entry; distinct entries are independent workload queries. +pub type CandidatePreASAPDAGs = Vec<(Id, PreASAPDAG)>; diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs index c6d00f81..ba615e21 100644 --- a/crates/types/src/pre_asap/query_expr.rs +++ b/crates/types/src/pre_asap/query_expr.rs @@ -1,7 +1,7 @@ //! The canonical pre-ASAP intent algebra IR. //! //! Language- and deployment-independent. `Rc`-owned tree — a child field is -//! `Rc>` rather than `Box>` so a structurally +//! `Rc>` rather than `Box>` so a structurally //! identical sub-expression can be shared (the same `Rc`) across more than //! one parent, within one query or across a `QueryWorkload` batch, instead of //! being duplicated. Nothing in this module produces that sharing on its @@ -23,12 +23,12 @@ use super::agg_intent::AggIntent; use super::expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; use super::schema::{Column, ColumnId, DataType, Schema}; -/// The column-reference resolution state a [`QueryExpr`] tree carries — -/// [`ColumnId`] (the default, and what the bare `QueryExpr` name has always +/// The column-reference resolution state a [`PreASAPNode`] tree carries — +/// [`ColumnId`] (the default, and what the bare `PreASAPNode` name has always /// meant) once the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has resolved every /// reference positionally, or the front-end-emitted, name-based [`ColumnRef`] /// before binding. The only place the two states differ in *shape* rather -/// than just in which type fills `C` is [`QueryExpr::Scan`]'s `schema` field: +/// than just in which type fills `C` is [`PreASAPNode::Scan`]'s `schema` field: /// a bound tree's binding schema is always known (the SchemaResolver is total, so /// [`ScanSchema`](Self::ScanSchema) `= Schema`); an unresolved front-end /// `Scan` knows its schema only when the front end already has it without @@ -37,7 +37,7 @@ use super::schema::{Column, ColumnId, DataType, Schema}; pub trait ColState: Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de> { - /// What [`QueryExpr::Scan`]'s `schema` field holds for a tree in this state. + /// What [`PreASAPNode::Scan`]'s `schema` field holds for a tree in this state. type ScanSchema: Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de>; } @@ -51,14 +51,14 @@ impl ColState for ColumnRef { /// Errors from schema derivation over a canonical tree. #[derive(Debug, Error)] -pub enum QueryExprError { +pub enum PreASAPNodeError { #[error("invalid scalar function signature: {0}")] InvalidScalarSignature(String), #[error("by-column id {0} out of range (input has {1} columns)")] InvalidGroupByColumn(ColumnId, usize), #[error("Concat requires at least one child")] EmptyConcat, - /// [`QueryExpr::output_schema`] called on (or reached, while recursing, a + /// [`PreASAPNode::output_schema`] called on (or reached, while recursing, a /// child that is) one of the scalar variants (issue #205) — those have no /// independent row schema of their own; a scalar expression's *type* only /// makes sense against the schema it's embedded in (see `infer_expr_type`, @@ -251,10 +251,10 @@ impl Source { /// operator has exactly one representation (and one `Display`) across the IR. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub enum BinaryOpKind { - /// Arithmetic — `Add/Sub/Mul/Div/Mod` (shared with `QueryExpr::Arithmetic`). + /// Arithmetic — `Add/Sub/Mul/Div/Mod` (shared with `PreASAPNode::Arithmetic`). Arithmetic(ArithmeticOpKind), /// Comparison — `Eq/Ne/Lt/Le/Gt/Ge` + `Like/ILike/Regex` family (shared - /// with `QueryExpr::Compare`). PromQL keeps the matched series whose + /// with `PreASAPNode::Compare`). PromQL keeps the matched series whose /// comparison holds. Compare(CompareOpKind), /// PromQL comparison with the `bool` modifier: every matched series @@ -402,7 +402,7 @@ pub enum WindowFrameBound { } /// A symbolic label matcher on the **info metric** side of an -/// [`QueryExpr::PromqlInfoEnrich`] (issue #84). Unlike a `Scan` predicate it is not +/// [`PreASAPNode::PromqlInfoEnrich`] (issue #84). Unlike a `Scan` predicate it is not /// resolved positionally — it references the info metric's labels (`__name__` /// picks the metric, the rest constrain data labels), which aren't in the input /// vector's schema; the post-ASAP realization pass applies it against the info metric. @@ -415,7 +415,7 @@ pub struct InfoMatcher { } /// Series-sampling selection mode (PromQL `limitk` / `limit_ratio`, issue #86). -/// A [`QueryExpr::PromqlSeriesSample`] keeps a *subset of whole series*, unchanged — it does +/// A [`PreASAPNode::PromqlSeriesSample`] keeps a *subset of whole series*, unchanged — it does /// not rank or reduce, so it is distinct from `TopK` and from `Sort → Limit`. #[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] #[serde(rename_all = "snake_case")] @@ -431,7 +431,7 @@ pub enum SampleKind { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] pub struct SortKey { - pub expr: QueryExpr, + pub expr: PreASAPNode, pub ascending: bool, pub nulls_first: bool, } @@ -459,7 +459,7 @@ pub enum AtModifier { /// PromQL per-selector **time-shift** modifiers — `offset` and `@` (issue #40). /// Neither changes a selector's *schema*; both move *when* it is evaluated, so -/// the shift is a pass-through wrapper ([`QueryExpr::TimeShift`]) over the +/// the shift is a pass-through wrapper ([`PreASAPNode::TimeShift`]) over the /// selector rather than a new leaf shape. The runtime resolves the anchor and /// applies the offset. #[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)] @@ -500,19 +500,19 @@ pub enum GroupSide { /// A row-level filter predicate (WHERE clause / PromQL label matcher). /// Boxed: `Predicate` sits directly (not behind a `Vec`) in -/// `Filter.pred`/`Join.pred`/`Aggregate.having`, and `QueryExpr` is +/// `Filter.pred`/`Join.pred`/`Aggregate.having`, and `PreASAPNode` is /// self-recursive without further indirection once the scalar variants are /// part of it — the box is what makes the recursive type's size finite there. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct Predicate(pub Rc>); +pub struct Predicate(pub Rc>); /// One item in a SELECT projection list. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] pub struct ProjectItem { pub alias: Option, - pub expr: QueryExpr, + pub expr: PreASAPNode, } // ── Intent algebra IR ──────────────────────────────────────────────────────── @@ -568,8 +568,8 @@ impl Reduction { } } -/// A caller-proven compound unique key for a [`QueryExpr::Concat`] (issue -/// #228) — built only via [`QueryExpr::concat_with_discriminator`] / +/// A caller-proven compound unique key for a [`PreASAPNode::Concat`] (issue +/// #228) — built only via [`PreASAPNode::concat_with_discriminator`] / /// [`ConcatDiscriminatorKey::new`], never by naming `discriminator` directly /// in a struct literal (both fields are private): from *other Rust code*, /// the only way to end up with one of these is to hand over a specific @@ -579,7 +579,7 @@ impl Reduction { /// derived `Deserialize` impl below builds a `ConcatDiscriminatorKey` /// directly from field values, bypassing `new()`. Deserialization is therefore /// equivalent to a caller supplying the assertion directly; it does not prove -/// either fact below. An external boundary accepting `QueryExpr` data must +/// either fact below. An external boundary accepting `PreASAPNode` data must /// reject this field or validate both obligations before treating it as /// uniqueness evidence. /// @@ -598,7 +598,7 @@ impl Reduction { /// hold: `inner_key` uniquely identifies rows **within every branch**, and /// `discriminator`'s value is **guaranteed to differ between branches** — a /// literal the producer just tagged the branch with (PromQL φ riding along via -/// [`QueryExpr::PromqlRelabel`], a Postgres-style synthetic `GROUPING()` id +/// [`PreASAPNode::PromqlRelabel`], a Postgres-style synthetic `GROUPING()` id /// for `ROLLUP`/`CUBE`, …), never something inferred structurally from the /// branches' own data — then `discriminator` alone partitions rows into /// disjoint sets independent of what the branches actually contain, so @@ -611,7 +611,7 @@ impl Reduction { /// repeats across branches, in which case the resulting `unique_keys` claim /// is simply wrong — `output_schema` trusts it without checking. The /// obligation is on the constructor call site, exactly as it is on -/// [`QueryExpr::Dedup`]'s `cols` or any other unverified `unique_keys` +/// [`PreASAPNode::Dedup`]'s `cols` or any other unverified `unique_keys` /// producer in this module. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(deny_unknown_fields)] @@ -643,7 +643,7 @@ impl ConcatDiscriminatorKey { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub enum QueryExpr { +pub enum PreASAPNode { /// Outermost leaf. `schema` is the **binding schema** — the resolved column /// set every positional `ColumnId` in the tree indexes into, *not* a full /// description of the runtime row — once bound (`schema: Schema`, always @@ -682,7 +682,7 @@ pub enum QueryExpr { /// `canonicalize`, and `resolve` now key off to tell "this operand has /// its own row schema" from "this is a nested scalar leaf with none," /// in place of the old `PromqlScalar` vs. `Literal` variant tag. - PromqlScalarBridge(Rc>), + PromqlScalarBridge(Rc>), /// The query **evaluation timestamp** as Unix seconds, exposed by PromQL /// `time()`. This is not inherently the current wall-clock time: its value @@ -699,13 +699,13 @@ pub enum QueryExpr { /// scalar-typed child to a single label-less series carrying the scalar's /// value at every step. Lets a scalar participate where a vector is required /// (`up or vector(0)` dead-man's-switch). Issue #48. - PromqlVectorFromScalar(Rc>), + PromqlVectorFromScalar(Rc>), /// PromQL `scalar(v)` — the instant-vector→scalar bridge. Collapses a /// single-element vector to its value (NaN at runtime if the input is not /// exactly one series). Lets a vector feed a scalar position (`vector` / /// aggregation `k` args, thresholds). Issue #48. - PromqlScalarFromVector(Rc>), + PromqlScalarFromVector(Rc>), /// ρ — a per-series **label rewrite** (PromQL `label_replace` / /// `label_join`). Every input row passes through unchanged except for the @@ -717,8 +717,8 @@ pub enum QueryExpr { PromqlRelabel { /// The label written by this rewrite (PromQL `dst_label`). dst: String, - value: Rc>, - child: Rc>, + value: Rc>, + child: Rc>, }, /// PromQL `info(v, [selector])` — left-join **label enrichment** (#84). Each @@ -734,7 +734,7 @@ pub enum QueryExpr { PromqlInfoEnrich { #[serde(default)] selector: Vec, - child: Rc>, + child: Rc>, }, /// Series-sampling **selection** — PromQL `limitk` / `limit_ratio` (#86). @@ -745,13 +745,13 @@ pub enum QueryExpr { #[serde(default)] by: GroupKeys, kind: SampleKind, - child: Rc>, + child: Rc>, }, /// σ — row-level filter. Output schema = child schema. Filter { pred: Predicate, - child: Rc>, + child: Rc>, }, /// π — column projection. Project { @@ -760,7 +760,7 @@ pub enum QueryExpr { /// table / inline view). `None` for an ordinary SELECT list. #[serde(default)] qualifier: Option, - child: Rc>, + child: Rc>, }, /// γ + α — GROUP BY (positional) + aggregate intents. @@ -777,18 +777,18 @@ pub enum QueryExpr { output_names: Vec, #[serde(default)] having: Option>, - child: Rc>, + child: Rc>, }, /// δ — SQL `DISTINCT` / row deduplication. Positional like every other /// column reference here; empty = dedup on all columns (`SELECT DISTINCT *`). Dedup { cols: Vec, - child: Rc>, + child: Rc>, }, /// ⊕ — exact, n-ary `UNION ALL` of independent branches. Rows are /// concatenated, never deduplicated; SQL's `UNION`/`INTERSECT`/`EXCEPT` are - /// [`QueryExpr::SetOp`], not this. + /// [`PreASAPNode::SetOp`], not this. /// /// Used for the branches of one query that a single `Aggregate` cannot /// express — PromQL `histogram_quantiles` (one branch per φ, issue #109) and @@ -804,19 +804,19 @@ pub enum QueryExpr { /// A row may appear in several branches, so no branch's unique key survives /// the union — `unique_keys` is dropped, as in `SetOp`. **Unless** the /// constructor asserted `discriminator_unique_key` (issue #228, - /// [`QueryExpr::concat_with_discriminator`]): a caller-proven claim that + /// [`PreASAPNode::concat_with_discriminator`]): a caller-proven claim that /// one column's value is guaranteed distinct per branch, which makes /// `(discriminator, inner_key)` a sound compound unique key regardless of /// whether `inner_key` alone repeats across branches. `None` — every /// ordinary construction path, including the plain struct literal and - /// [`QueryExpr::concat`] — reproduces the old, unconditional-drop + /// [`PreASAPNode::concat`] — reproduces the old, unconditional-drop /// behavior exactly; see [`ConcatDiscriminatorKey`]'s doc for the /// soundness argument and the obligation this puts on whoever asserts it. /// - /// Empty children is an error ([`QueryExprError::EmptyConcat`]), not an + /// Empty children is an error ([`PreASAPNodeError::EmptyConcat`]), not an /// empty relation: there would be no schema to derive. Concat { - children: Vec>, + children: Vec>, /// See the field-level doc above and [`ConcatDiscriminatorKey`]. #[serde(default)] discriminator_unique_key: Option>, @@ -826,14 +826,14 @@ pub enum QueryExpr { Join { kind: JoinKind, pred: Predicate, - left: Rc>, - right: Rc>, + left: Rc>, + right: Rc>, }, SetOp { kind: RelationalSetOpKind, all: bool, - left: Rc>, - right: Rc>, + left: Rc>, + right: Rc>, }, /// Generic order-by for non-heavy-hitter cases. @@ -850,12 +850,12 @@ pub enum QueryExpr { keys: Vec>, #[serde(default)] partition_by: GroupKeys, - child: Rc>, + child: Rc>, }, Limit { n: usize, offset: usize, - child: Rc>, + child: Rc>, }, /// PromQL sub-query (`[range:resolution]`). Logical pass-through. @@ -863,7 +863,7 @@ pub enum QueryExpr { range: Duration, #[serde(default)] resolution: Option, - child: Rc>, + child: Rc>, }, /// Temporal range selection — "look back `range` of history for this @@ -875,7 +875,7 @@ pub enum QueryExpr { /// plain `Scan` or another `Aggregate` is a *cross-series* reduction. TimeRange { range: Duration, - child: Rc>, + child: Rc>, }, /// PromQL `offset` / `@` **time shift** on a selector (issue #40). A @@ -889,7 +889,7 @@ pub enum QueryExpr { /// selector when neither modifier is present). TimeShift { shift: TimeShift, - child: Rc>, + child: Rc>, }, /// SQL analytic window function: `func(args) OVER (PARTITION BY … ORDER BY … @@ -899,7 +899,7 @@ pub enum QueryExpr { func: WindowFuncKind, /// Operand expressions (`LAG(value)` → `[Column(value_id)]`); empty for /// the rank-only functions (`ROW_NUMBER`/`RANK`/`DENSE_RANK`). - args: Vec>, + args: Vec>, partition_by: GroupKeys, order_by: Vec>, /// `None` is accepted only for backward compatibility with serialized @@ -911,14 +911,14 @@ pub enum QueryExpr { /// The output column's name — DataFusion's window-expr field name, so a /// `Project` above resolves it (cf. `Aggregate.output_names`). output_name: String, - child: Rc>, + child: Rc>, }, /// Arithmetic / comparison / boolean composition (PromQL binary ops). BinaryOp { op: BinaryOpKind, - lhs: Rc>, - rhs: Rc>, + lhs: Rc>, + rhs: Rc>, #[serde(default)] vector_match: Option, }, @@ -935,7 +935,7 @@ pub enum QueryExpr { // `Expr` variant set did — nothing stops constructing, say, a `Scan` // where a `Compare`'s `left` operand belongs. `output_schema` and every // scalar-position consumer (`resolve`, `canonicalize`, `infer_expr_type`) - // reject a non-scalar variant found there instead (a `QueryExprError` or + // reject a non-scalar variant found there instead (a `PreASAPNodeError` or // an `unreachable!`, depending on the call site) — the accepted // replacement, since the alternative (a marker-trait/sub-enum bound // restricting which variants are constructible in a scalar position) adds @@ -949,60 +949,60 @@ pub enum QueryExpr { Literal(ScalarValue), /// `left op right` — binary comparison. Compare { - left: Rc>, + left: Rc>, op: CompareOpKind, - right: Rc>, + right: Rc>, }, /// Flat conjunction (logical AND). An empty list is vacuously true. - BoolAnd(Vec>), + BoolAnd(Vec>), /// Flat disjunction (logical OR). An empty list is vacuously false. - BoolOr(Vec>), + BoolOr(Vec>), /// Logical NOT. - Not(Rc>), + Not(Rc>), /// `expr IS NULL`. - IsNull(Rc>), + IsNull(Rc>), /// `expr IS NOT NULL`. - IsNotNull(Rc>), + IsNotNull(Rc>), /// `CAST(expr AS to)`; `try_cast` for SQL `TRY_CAST` (NULL on failure). Cast { - expr: Rc>, + expr: Rc>, to: DataType, try_cast: bool, }, /// `expr [NOT] IN (v1, v2, …)`. InList { - expr: Rc>, - list: Vec>, + expr: Rc>, + list: Vec>, negated: bool, }, /// Scalar function call, e.g. `LOWER(col)`, `ABS(x)`. FunctionCall { name: String, - args: Vec>, + args: Vec>, }, /// Binary arithmetic: `left op right`. Arithmetic { op: ArithmeticOpKind, - left: Rc>, - right: Rc>, + left: Rc>, + right: Rc>, }, /// SQL `CASE` (both searched and simple forms). `operand` present for the /// simple form (`CASE expr WHEN …`), absent for searched. Case { - operand: Option>>, - branches: Vec<(QueryExpr, QueryExpr)>, - else_expr: Option>>, + operand: Option>>, + branches: Vec<(PreASAPNode, PreASAPNode)>, + else_expr: Option>>, }, } -impl QueryExpr { +impl PreASAPNode { /// Construct the [`PromqlScalarBridge`](Self::PromqlScalarBridge) leaf /// for a bare PromQL numeric literal / folded constant scalar (issue /// #220) — `Literal(ScalarValue::Float64(v))` at an operator-tree /// position. The one constructor every front end / test that used to - /// write `QueryExpr::PromqlScalar(v)` should use instead. + /// write `PreASAPNode::PromqlScalar(v)` should use instead. pub fn promql_scalar(v: f64) -> Self { - QueryExpr::PromqlScalarBridge(Rc::new(QueryExpr::Literal(ScalarValue::Float64(v)))) + PreASAPNode::PromqlScalarBridge(Rc::new(PreASAPNode::Literal(ScalarValue::Float64(v)))) } /// Build an ordinary [`Concat`](Self::Concat) — the ordinary/default @@ -1012,8 +1012,8 @@ impl QueryExpr { /// [`concat_with_discriminator`](Self::concat_with_discriminator) instead /// when the caller can prove branch disjointness via a discriminator /// column. - pub fn concat(children: Vec>) -> Self { - QueryExpr::Concat { + pub fn concat(children: Vec>) -> Self { + PreASAPNode::Concat { children, discriminator_unique_key: None, } @@ -1027,11 +1027,11 @@ impl QueryExpr { /// checks that `inner_key` is unique within every branch or that /// `discriminator`'s value is distinct between branches. pub fn concat_with_discriminator( - children: Vec>, + children: Vec>, discriminator: C, inner_key: Vec, ) -> Self { - QueryExpr::Concat { + PreASAPNode::Concat { children, discriminator_unique_key: Some(ConcatDiscriminatorKey::new(discriminator, inner_key)), } @@ -1044,8 +1044,8 @@ impl QueryExpr { /// something else (not constructed today, but not precluded by the type). pub fn as_promql_scalar(&self) -> Option { match self { - QueryExpr::PromqlScalarBridge(inner) => match inner.as_ref() { - QueryExpr::Literal(ScalarValue::Float64(v)) => Some(*v), + PreASAPNode::PromqlScalarBridge(inner) => match inner.as_ref() { + PreASAPNode::Literal(ScalarValue::Float64(v)) => Some(*v), _ => None, }, _ => None, @@ -1054,18 +1054,18 @@ impl QueryExpr { /// If this expression is a `BoolAnd`, return its elements; otherwise a /// single-element slice containing `self`. - pub fn conjuncts(&self) -> &[QueryExpr] { + pub fn conjuncts(&self) -> &[PreASAPNode] { match self { - QueryExpr::BoolAnd(v) => v.as_slice(), + PreASAPNode::BoolAnd(v) => v.as_slice(), _ => std::slice::from_ref(self), } } /// If this expression is a `BoolOr`, return its elements; otherwise a /// single-element slice containing `self`. - pub fn disjuncts(&self) -> &[QueryExpr] { + pub fn disjuncts(&self) -> &[PreASAPNode] { match self { - QueryExpr::BoolOr(v) => v.as_slice(), + PreASAPNode::BoolOr(v) => v.as_slice(), _ => std::slice::from_ref(self), } } @@ -1075,37 +1075,38 @@ impl QueryExpr { /// leaf schemas, and available to post-ASAP binding for column-lineage / /// selectivity. /// `self` must be one of the scalar variants (see the module doc on - /// [`QueryExpr`]'s scalar shapes) — every caller already only reaches + /// [`PreASAPNode`]'s scalar shapes) — every caller already only reaches /// this through a scalar-typed position (`Predicate`, `ProjectItem.expr`, /// …), so an operator variant here indicates a construction bug, not a /// shape this needs to handle silently. pub fn columns_referenced(&self) -> Vec<&C> { match self { - QueryExpr::Column(c) => vec![c], - QueryExpr::Literal(_) => vec![], - QueryExpr::EvalTimestamp => vec![], - QueryExpr::CurrentTimestamp => vec![], - QueryExpr::Compare { left, right, .. } | QueryExpr::Arithmetic { left, right, .. } => { + PreASAPNode::Column(c) => vec![c], + PreASAPNode::Literal(_) => vec![], + PreASAPNode::EvalTimestamp => vec![], + PreASAPNode::CurrentTimestamp => vec![], + PreASAPNode::Compare { left, right, .. } + | PreASAPNode::Arithmetic { left, right, .. } => { let mut v = left.columns_referenced(); v.extend(right.columns_referenced()); v } - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + PreASAPNode::BoolAnd(parts) | PreASAPNode::BoolOr(parts) => { parts.iter().flat_map(|e| e.columns_referenced()).collect() } - QueryExpr::Not(e) | QueryExpr::IsNull(e) | QueryExpr::IsNotNull(e) => { + PreASAPNode::Not(e) | PreASAPNode::IsNull(e) | PreASAPNode::IsNotNull(e) => { e.columns_referenced() } - QueryExpr::Cast { expr, .. } => expr.columns_referenced(), - QueryExpr::InList { expr, list, .. } => { + PreASAPNode::Cast { expr, .. } => expr.columns_referenced(), + PreASAPNode::InList { expr, list, .. } => { let mut v = expr.columns_referenced(); v.extend(list.iter().flat_map(|e| e.columns_referenced())); v } - QueryExpr::FunctionCall { args, .. } => { + PreASAPNode::FunctionCall { args, .. } => { args.iter().flat_map(|e| e.columns_referenced()).collect() } - QueryExpr::Case { + PreASAPNode::Case { operand, branches, else_expr, @@ -1124,43 +1125,43 @@ impl QueryExpr { v } other => unreachable!( - "columns_referenced called on a non-scalar QueryExpr variant: {other:?}" + "columns_referenced called on a non-scalar PreASAPNode variant: {other:?}" ), } } } -/// The canonical, positional, resolved tree — what the bare `QueryExpr` name +/// The canonical, positional, resolved tree — what the bare `PreASAPNode` name /// has always meant (the default `C = ColumnId`). Every existing consumer -/// keeps using `QueryExpr` unparameterized; this alias exists only to name +/// keeps using `PreASAPNode` unparameterized; this alias exists only to name /// the resolved state explicitly at a use site that also wants to name -/// [`UnresolvedQueryExpr`] nearby. -pub type ResolvedQueryExpr = QueryExpr; +/// [`UnresolvedPreASAPNode`] nearby. +pub type ResolvedPreASAPNode = PreASAPNode; /// The front-end-emitted, name-based, unresolved tree — -/// `QueryExpr`: front ends construct this directly during their +/// `PreASAPNode`: front ends construct this directly during their /// own `interpret` step (issue #179), and the [`SchemaResolver`](super::schema_resolver) -/// resolves it into [`ResolvedQueryExpr`]. -pub type UnresolvedQueryExpr = QueryExpr; +/// resolves it into [`ResolvedPreASAPNode`]. +pub type UnresolvedPreASAPNode = PreASAPNode; // `output_schema` needs a fully bound tree — it reads `Scan.schema` as a plain // `Schema` and resolves every scalar `Expr::Column` positionally — so it lives -// only on the resolved instantiation, not `impl QueryExpr`. +// only on the resolved instantiation, not `impl PreASAPNode`. // Same reasoning as `AggIntent`'s `output_column`/`requires`/`is_per_series` // (#205): a schema-shaped property that is only meaningful post-binding. -impl QueryExpr { +impl PreASAPNode { /// Infer a scalar expression against its input relation using the same /// canonical rules as projection schema derivation. - pub fn scalar_type(&self, input: &Schema) -> Result<(DataType, bool), QueryExprError> { + pub fn scalar_type(&self, input: &Schema) -> Result<(DataType, bool), PreASAPNodeError> { infer_expr_type(self, input) } /// Output schema of the root of a canonical tree. - pub fn output_schema(&self) -> Result { + pub fn output_schema(&self) -> Result { match self { - QueryExpr::Scan { schema, .. } => Ok(schema.clone()), + PreASAPNode::Scan { schema, .. } => Ok(schema.clone()), - QueryExpr::Aggregate { + PreASAPNode::Aggregate { reduction, measures, output_names, @@ -1171,20 +1172,20 @@ impl QueryExpr { aggregate_output_schema(&in_schema, reduction, measures, output_names) } - QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } + PreASAPNode::Filter { child, .. } + | PreASAPNode::Sort { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } // Series sampling keeps a subset of whole series unchanged, so the // output schema (and row-uniqueness) is exactly the child's (#86). - | QueryExpr::PromqlSeriesSample { child, .. } + | PreASAPNode::PromqlSeriesSample { child, .. } // Info enrichment adds runtime info labels — the statically-known // schema is the child's (open), so it passes through (#84). - | QueryExpr::PromqlInfoEnrich { child, .. } - | QueryExpr::TimeRange { child, .. } + | PreASAPNode::PromqlInfoEnrich { child, .. } + | PreASAPNode::TimeRange { child, .. } // A time shift (`offset`/`@`) moves *when* the child is evaluated, // never its columns — schema passes through (#40). - | QueryExpr::TimeShift { child, .. } => child.output_schema(), + | PreASAPNode::TimeShift { child, .. } => child.output_schema(), // ρ — relabel preserves every input column and writes one label // `dst` (Utf8): overwritten in place if it already exists, else @@ -1192,7 +1193,7 @@ impl QueryExpr { // unset). The schema stays open (other labels remain runtime-only). // A rewrite can collapse two label sets into one, so row-uniqueness // is no longer provable — drop unique_keys. - QueryExpr::PromqlRelabel { dst, child, .. } => { + PreASAPNode::PromqlRelabel { dst, child, .. } => { let mut out = child.output_schema()?; if let Some(existing) = out.columns.iter_mut().find(|c| c.name == *dst) { existing.dtype = DataType::Utf8; @@ -1211,7 +1212,7 @@ impl QueryExpr { // as a bare `Column` item (possibly reordered or aliased). Derived // expressions cannot carry key identity. `time_index` is re-found // by name. - QueryExpr::Project { cols, qualifier, child } => { + PreASAPNode::Project { cols, qualifier, child } => { let in_schema = child.output_schema()?; let columns: Vec = cols .iter() @@ -1231,7 +1232,7 @@ impl QueryExpr { None => c, }) }) - .collect::, QueryExprError>>()?; + .collect::, PreASAPNodeError>>()?; let time_index = columns.iter().position(|c| c.name == "ts"); let unique_keys = in_schema .unique_keys @@ -1240,7 +1241,7 @@ impl QueryExpr { key.iter() .map(|input_col| { cols.iter().position(|item| { - matches!(&item.expr, QueryExpr::Column(col) if col == input_col) + matches!(&item.expr, PreASAPNode::Column(col) if col == input_col) }) }) .collect::>>() @@ -1255,7 +1256,7 @@ impl QueryExpr { }) } - QueryExpr::Dedup { cols, child } => { + PreASAPNode::Dedup { cols, child } => { let mut out = child.output_schema()?; // Deduplicating on `cols` makes them a unique key of the result. if !cols.is_empty() { @@ -1273,13 +1274,13 @@ impl QueryExpr { // assertion is trusted verbatim here, never checked: see // `ConcatDiscriminatorKey`'s doc for the soundness argument and // whose obligation it is. - QueryExpr::Concat { + PreASAPNode::Concat { children, discriminator_unique_key, } => { let mut s = children .first() - .ok_or(QueryExprError::EmptyConcat) + .ok_or(PreASAPNodeError::EmptyConcat) .and_then(|c| c.output_schema())?; s.unique_keys.clear(); if let Some(key) = discriminator_unique_key { @@ -1292,7 +1293,7 @@ impl QueryExpr { // Set operations are union-compatible: both sides share the left's // column shape, so the output schema is the left's. (Row identity // is not preserved across a UNION, so unique_keys are dropped.) - QueryExpr::SetOp { left, .. } => { + PreASAPNode::SetOp { left, .. } => { let mut s = left.output_schema()?; s.unique_keys.clear(); Ok(s) @@ -1300,7 +1301,7 @@ impl QueryExpr { // ⋈ — output is the concatenation of both inputs' columns. Outer // joins make the non-preserved side nullable. Post-join row // identity isn't provable in general, so unique_keys reset. - QueryExpr::Join { + PreASAPNode::Join { kind, left, right, .. } => { let l = left.output_schema()?; @@ -1342,7 +1343,7 @@ impl QueryExpr { }) } // ψ-analytic — child schema + one appended window-output column. - QueryExpr::SQLWindowFunc { + PreASAPNode::SQLWindowFunc { func, args, output_name, @@ -1353,7 +1354,7 @@ impl QueryExpr { // First operand's (dtype, nullable) from the child schema, owned // so the borrow ends before we append. let arg = args.first().and_then(|a| match a { - QueryExpr::Column(id) => out.columns.get(*id), + PreASAPNode::Column(id) => out.columns.get(*id), _ => None, }); let arg_dtype = || arg.map_or(DataType::Float64, |c| c.dtype.clone()); @@ -1387,14 +1388,14 @@ impl QueryExpr { // `PromqlScalarBridge` constructed today wraps a plain // `Literal(Float64)` (issue #220), so the schema doesn't need to // inspect the inner node. - QueryExpr::PromqlScalarBridge(_) | QueryExpr::EvalTimestamp => Ok(Schema { + PreASAPNode::PromqlScalarBridge(_) | PreASAPNode::EvalTimestamp => Ok(Schema { columns: vec![Column::new("value", DataType::Float64, false)], time_index: None, unique_keys: Vec::new(), closed: true, }), - QueryExpr::CurrentTimestamp => Ok(Schema { + PreASAPNode::CurrentTimestamp => Ok(Schema { columns: vec![Column::new("value", DataType::Timestamp, false)], time_index: None, unique_keys: Vec::new(), @@ -1404,7 +1405,7 @@ impl QueryExpr { // `vector(s)` yields a label-less instant vector: the (ts, value) // floor and nothing else. `closed` — its full label set (empty) is // known statically (#48). - QueryExpr::PromqlVectorFromScalar(_) => Ok(Schema { + PreASAPNode::PromqlVectorFromScalar(_) => Ok(Schema { columns: vec![ Column::new("ts", DataType::Timestamp, false), Column::new("value", DataType::Float64, false), @@ -1416,7 +1417,7 @@ impl QueryExpr { // `scalar(v)` collapses to a single `value`, no time index — the same // scalar shape as a constant or `time()` (#48). - QueryExpr::PromqlScalarFromVector(_) => Ok(Schema { + PreASAPNode::PromqlScalarFromVector(_) => Ok(Schema { columns: vec![Column::new("value", DataType::Float64, false)], time_index: None, unique_keys: Vec::new(), @@ -1427,11 +1428,11 @@ impl QueryExpr { // `) is the vector side's — a scalar operand (a constant or // `time()`) contributes only its value, no labels. Prefer the // non-scalar side. - QueryExpr::BinaryOp { lhs, rhs, op, vector_match } => { - fn scalar(expression: &QueryExpr) -> bool { + PreASAPNode::BinaryOp { lhs, rhs, op, vector_match } => { + fn scalar(expression: &PreASAPNode) -> bool { match expression { - QueryExpr::PromqlScalarBridge(_) | QueryExpr::EvalTimestamp | QueryExpr::PromqlScalarFromVector(_) => true, - QueryExpr::BinaryOp { lhs, rhs, .. } => scalar(lhs) && scalar(rhs), + PreASAPNode::PromqlScalarBridge(_) | PreASAPNode::EvalTimestamp | PreASAPNode::PromqlScalarFromVector(_) => true, + PreASAPNode::BinaryOp { lhs, rhs, .. } => scalar(lhs) && scalar(rhs), _ => false, } } @@ -1457,20 +1458,20 @@ impl QueryExpr { Ok(output) }, - // The scalar variants (issue #205) — see `QueryExprError::ScalarHasNoRowSchema`. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => Err(QueryExprError::ScalarHasNoRowSchema), + // The scalar variants (issue #205) — see `PreASAPNodeError::ScalarHasNoRowSchema`. + PreASAPNode::Column(_) + | PreASAPNode::Literal(_) + | PreASAPNode::Compare { .. } + | PreASAPNode::BoolAnd(_) + | PreASAPNode::BoolOr(_) + | PreASAPNode::Not(_) + | PreASAPNode::IsNull(_) + | PreASAPNode::IsNotNull(_) + | PreASAPNode::Cast { .. } + | PreASAPNode::InList { .. } + | PreASAPNode::FunctionCall { .. } + | PreASAPNode::Arithmetic { .. } + | PreASAPNode::Case { .. } => Err(PreASAPNodeError::ScalarHasNoRowSchema), } } } @@ -1480,18 +1481,21 @@ impl QueryExpr { /// one value per series, so every label column of `input` is preserved and only /// the sample value is replaced — kept named `value` so the PromQL sample-value /// convention (and any outer `SampleValue` reference) still resolves it by name. -fn per_series_reduction_schema(input: &Schema, agg: &AggIntent) -> Result { +fn per_series_reduction_schema( + input: &Schema, + agg: &AggIntent, +) -> Result { let vi = if let Some(index) = agg.input_cols().first() { *index } else { super::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, input) - .map_err(|error| QueryExprError::InvalidSampleColumn(error.to_string()))? + .map_err(|error| PreASAPNodeError::InvalidSampleColumn(error.to_string()))? }; if !matches!( input.columns.get(vi).map(|column| &column.dtype), Some(DataType::Float64 | DataType::Int64) ) { - return Err(QueryExprError::InvalidSampleColumn(format!( + return Err(PreASAPNodeError::InvalidSampleColumn(format!( "column {vi} is not numeric" ))); } @@ -1518,7 +1522,7 @@ fn per_series_reduction_schema(input: &Schema, agg: &AggIntent) -> Result Result { +) -> Result { let by = match reduction { Reduction::PerEntity => { debug_assert_eq!( @@ -1559,7 +1563,7 @@ pub fn aggregate_output_schema( let c = in_schema .columns .get(id) - .ok_or(QueryExprError::InvalidGroupByColumn( + .ok_or(PreASAPNodeError::InvalidGroupByColumn( id, in_schema.columns.len(), ))?; @@ -1610,7 +1614,7 @@ pub fn aggregate_output_schema( } if let Some((arg, _)) = intent .arg_selector_columns(in_schema) - .map_err(QueryExprError::InvalidScalarSignature)? + .map_err(PreASAPNodeError::InvalidScalarSignature)? { out.dtype = in_schema.columns[arg].dtype.clone(); out.nullable = in_schema.columns[arg].nullable; @@ -1652,10 +1656,10 @@ fn without_output_schema( excluded: &[ColumnId], measures: &[AggIntent], output_names: &[String], -) -> Result { +) -> Result { for &id in excluded { if id >= in_schema.columns.len() { - return Err(QueryExprError::InvalidGroupByColumn( + return Err(PreASAPNodeError::InvalidGroupByColumn( id, in_schema.columns.len(), )); @@ -1688,7 +1692,7 @@ fn without_output_schema( let mut out = intent.output_column(in_col); if let Some((arg, _)) = intent .arg_selector_columns(in_schema) - .map_err(QueryExprError::InvalidScalarSignature)? + .map_err(PreASAPNodeError::InvalidScalarSignature)? { out.dtype = in_schema.columns[arg].dtype.clone(); out.nullable = in_schema.columns[arg].nullable; @@ -1708,24 +1712,24 @@ fn without_output_schema( }) } -/// Infer the `(DataType, nullable)` a scalar [`QueryExpr`] produces against an +/// Infer the `(DataType, nullable)` a scalar [`PreASAPNode`] produces against an /// input [`Schema`]. Used by `Project` schema derivation. Approximate here: /// unknown columns and bare `FunctionCall`s fall back to a permissive default /// (post-ASAP binding refines with a real function/type registry). `expr` /// must be one of the scalar variants (issue #205) — an operator variant here /// is a construction bug, not a shape this needs to handle silently. fn infer_expr_type( - expr: &QueryExpr, + expr: &PreASAPNode, schema: &Schema, -) -> Result<(DataType, bool), QueryExprError> { +) -> Result<(DataType, bool), PreASAPNodeError> { Ok(match expr { - QueryExpr::CurrentTimestamp => (DataType::Timestamp, false), - QueryExpr::Column(id) => schema + PreASAPNode::CurrentTimestamp => (DataType::Timestamp, false), + PreASAPNode::Column(id) => schema .columns .get(*id) .map(|c| (c.dtype.clone(), c.nullable)) .unwrap_or((DataType::Float64, true)), - QueryExpr::Literal(s) => match s { + PreASAPNode::Literal(s) => match s { ScalarValue::Int64(_) => (DataType::Int64, false), ScalarValue::Float64(_) => (DataType::Float64, false), ScalarValue::Utf8(_) => (DataType::Utf8, false), @@ -1734,14 +1738,14 @@ fn infer_expr_type( ScalarValue::Interval { .. } => (DataType::Interval, false), }, // Boolean-valued expressions (SQL three-valued logic → nullable). - QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::InList { .. } => (DataType::Bool, true), - QueryExpr::Arithmetic { op, left, right } => { + PreASAPNode::Compare { .. } + | PreASAPNode::BoolAnd(_) + | PreASAPNode::BoolOr(_) + | PreASAPNode::Not(_) + | PreASAPNode::IsNull(_) + | PreASAPNode::IsNotNull(_) + | PreASAPNode::InList { .. } => (DataType::Bool, true), + PreASAPNode::Arithmetic { op, left, right } => { let (lt, ln) = infer_expr_type(left, schema)?; let (rt, rn) = infer_expr_type(right, schema)?; // Temporal subtraction yields a fixed duration with a unit, not a @@ -1751,7 +1755,7 @@ fn infer_expr_type( && matches!(lt, DataType::Date | DataType::Timestamp) && matches!(rt, DataType::Date | DataType::Timestamp) { - return Err(QueryExprError::InvalidScalarSignature( + return Err(PreASAPNodeError::InvalidScalarSignature( "temporal subtraction produces an unsupported duration type".into(), )); } @@ -1777,17 +1781,17 @@ fn infer_expr_type( }; (dtype, ln || rn) } - QueryExpr::Cast { to, try_cast, expr } => { + PreASAPNode::Cast { to, try_cast, expr } => { let (_, nullable) = infer_expr_type(expr, schema)?; (to.clone(), *try_cast || nullable) } - QueryExpr::FunctionCall { name, args } => { + PreASAPNode::FunctionCall { name, args } => { if name == "asap_element_access" { super::scalar_signature::element_access_type(args, schema) - .map_err(QueryExprError::InvalidScalarSignature)? + .map_err(PreASAPNodeError::InvalidScalarSignature)? } else if name == "asap_struct_field" { super::scalar_signature::struct_field_type(args, schema) - .map_err(QueryExprError::InvalidScalarSignature)? + .map_err(PreASAPNodeError::InvalidScalarSignature)? } else if let Some(function) = super::scalar_signature::MapScalarFunction::from_name(name) { @@ -1797,13 +1801,13 @@ fn infer_expr_type( .collect::, _>>()?; function .output_type(&arguments) - .map_err(QueryExprError::InvalidScalarSignature)? + .map_err(PreASAPNodeError::InvalidScalarSignature)? } else { // Legacy unknown functions retain their existing policy. (DataType::Float64, true) } } - QueryExpr::Case { + PreASAPNode::Case { branches, else_expr, .. @@ -1817,16 +1821,16 @@ fn infer_expr_type( } } other => { - unreachable!("infer_expr_type called on a non-scalar QueryExpr variant: {other:?}") + unreachable!("infer_expr_type called on a non-scalar PreASAPNode variant: {other:?}") } }) } /// Default output-column name for a projection item with no explicit alias: /// a bare column keeps its (schema) name; anything else gets `col_{i}`. -fn default_proj_name(expr: &QueryExpr, idx: usize, schema: &Schema) -> String { +fn default_proj_name(expr: &PreASAPNode, idx: usize, schema: &Schema) -> String { match expr { - QueryExpr::Column(id) => schema + PreASAPNode::Column(id) => schema .columns .get(*id) .map(|c| c.name.clone()) @@ -1854,15 +1858,15 @@ mod tests { col("d", DataType::Date, false), ]); let thirty_days = || { - Rc::new(QueryExpr::Literal(ScalarValue::Interval { + Rc::new(PreASAPNode::Literal(ScalarValue::Interval { months: 0, days: 30, nanos: 0, })) }; - let shift = |column, op| QueryExpr::Arithmetic { + let shift = |column, op| PreASAPNode::Arithmetic { op, - left: Rc::new(QueryExpr::Column(column)), + left: Rc::new(PreASAPNode::Column(column)), right: thirty_days(), }; @@ -1881,7 +1885,7 @@ mod tests { DataType::Date ); assert_eq!( - QueryExpr::Arithmetic { + PreASAPNode::Arithmetic { op: ArithmeticOpKind::Add, left: thirty_days(), right: thirty_days(), @@ -1897,8 +1901,8 @@ mod tests { columns: Vec, time_index: Option, uk: Vec>, - ) -> QueryExpr { - QueryExpr::Scan { + ) -> PreASAPNode { + PreASAPNode::Scan { source: Source::Table { table_ref: "t".into(), }, @@ -1923,22 +1927,22 @@ mod tests { None, vec![vec![0, 1]], )); - let projected = QueryExpr::Project { + let projected = PreASAPNode::Project { cols: vec![ ProjectItem { alias: Some("r".into()), - expr: QueryExpr::Column(1), + expr: PreASAPNode::Column(1), }, ProjectItem { alias: Some("t".into()), - expr: QueryExpr::Column(0), + expr: PreASAPNode::Column(0), }, ProjectItem { alias: None, - expr: QueryExpr::Arithmetic { + expr: PreASAPNode::Arithmetic { op: ArithmeticOpKind::Add, - left: Rc::new(QueryExpr::Column(2)), - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + left: Rc::new(PreASAPNode::Column(2)), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Int64(1))), }, }, ], @@ -1962,10 +1966,10 @@ mod tests { None, vec![vec![0, 1]], )); - let projected = QueryExpr::Project { + let projected = PreASAPNode::Project { cols: vec![ProjectItem { alias: None, - expr: QueryExpr::Column(0), + expr: PreASAPNode::Column(0), }], qualifier: None, child: input, @@ -1976,7 +1980,7 @@ mod tests { #[test] fn legacy_window_json_without_frame_deserializes_as_unspecified() { - let window = QueryExpr::SQLWindowFunc { + let window = PreASAPNode::SQLWindowFunc { func: WindowFuncKind::RowNumber, args: vec![], partition_by: GroupKeys::by(vec![]), @@ -1997,10 +2001,10 @@ mod tests { .unwrap() .remove("frame"); - let decoded: QueryExpr = serde_json::from_value(json).unwrap(); + let decoded: PreASAPNode = serde_json::from_value(json).unwrap(); assert!(matches!( decoded, - QueryExpr::SQLWindowFunc { frame: None, .. } + PreASAPNode::SQLWindowFunc { frame: None, .. } )); } @@ -2010,7 +2014,7 @@ mod tests { /// not have — `unique_keys` feeds CSE's producer-sharing legality check. #[test] fn merge_drops_the_branches_unique_keys() { - let branch = || QueryExpr::Dedup { + let branch = || PreASAPNode::Dedup { cols: vec![0], child: Rc::new(scan( vec![ @@ -2027,7 +2031,7 @@ mod tests { "a Dedup branch does have a unique key on its own" ); - let merged = QueryExpr::concat(vec![branch(), branch()]); + let merged = PreASAPNode::concat(vec![branch(), branch()]); let schema = merged.output_schema().unwrap(); assert!( schema.unique_keys.is_empty(), @@ -2040,12 +2044,12 @@ mod tests { /// Same rule as `SetOp`, which already dropped them. #[test] fn merge_and_setop_agree_on_unique_keys() { - let branch = || QueryExpr::Dedup { + let branch = || PreASAPNode::Dedup { cols: vec![0], child: Rc::new(scan(vec![col("k", DataType::Utf8, false)], None, vec![])), }; - let merged = QueryExpr::concat(vec![branch(), branch()]); - let setop = QueryExpr::SetOp { + let merged = PreASAPNode::concat(vec![branch(), branch()]); + let setop = PreASAPNode::SetOp { kind: RelationalSetOpKind::Union, all: true, left: Rc::new(branch()), @@ -2060,8 +2064,8 @@ mod tests { #[test] fn an_empty_merge_has_no_schema() { assert!(matches!( - QueryExpr::concat(vec![]).output_schema(), - Err(QueryExprError::EmptyConcat) + PreASAPNode::concat(vec![]).output_schema(), + Err(PreASAPNodeError::EmptyConcat) )); } @@ -2079,7 +2083,7 @@ mod tests { // branch (PromQL φ, a synthetic `GROUPING()` id, ...) — this // schema-level test only checks the shape `output_schema` derives // from asserting one, not how a real caller proves distinctness. - let branch = || QueryExpr::Dedup { + let branch = || PreASAPNode::Dedup { cols: vec![0], child: Rc::new(scan( vec![ @@ -2090,7 +2094,7 @@ mod tests { vec![], )), }; - let merged = QueryExpr::concat_with_discriminator( + let merged = PreASAPNode::concat_with_discriminator( vec![branch(), branch()], /* discriminator */ 1, /* inner_key */ vec![0], @@ -2120,11 +2124,11 @@ mod tests { /// unchanged. #[test] fn ordinary_concat_struct_literal_still_drops_unique_keys_by_default() { - let branch = || QueryExpr::Dedup { + let branch = || PreASAPNode::Dedup { cols: vec![0], child: Rc::new(scan(vec![col("k", DataType::Utf8, false)], None, vec![])), }; - let merged = QueryExpr::Concat { + let merged = PreASAPNode::Concat { children: vec![branch(), branch()], discriminator_unique_key: None, }; @@ -2141,13 +2145,13 @@ mod tests { /// as an explicit, named argument. #[test] fn no_way_to_fabricate_a_unique_key_without_naming_a_discriminator() { - let branch = || QueryExpr::Dedup { + let branch = || PreASAPNode::Dedup { cols: vec![0], child: Rc::new(scan(vec![col("k", DataType::Utf8, false)], None, vec![])), }; // The ordinary builder. assert_eq!( - QueryExpr::concat(vec![branch(), branch()]) + PreASAPNode::concat(vec![branch(), branch()]) .output_schema() .unwrap() .unique_keys, @@ -2155,7 +2159,7 @@ mod tests { ); // The bare struct literal, explicitly opting out. assert_eq!( - QueryExpr::Concat { + PreASAPNode::Concat { children: vec![branch(), branch()], discriminator_unique_key: None, } @@ -2177,30 +2181,30 @@ mod tests { Some(0), vec![vec![0, 1]], ); - let q = QueryExpr::Project { + let q = PreASAPNode::Project { qualifier: None, cols: vec![ // bare column passthrough keeps its (schema) name + type: host=col 1 ProjectItem { alias: None, - expr: QueryExpr::Column(1), + expr: PreASAPNode::Column(1), }, // arithmetic over value (col 2) → Float64 ProjectItem { alias: Some("dbl".into()), - expr: QueryExpr::Arithmetic { + expr: PreASAPNode::Arithmetic { op: ArithmeticOpKind::Add, - left: Rc::new(QueryExpr::Column(2)), - right: Rc::new(QueryExpr::Column(2)), + left: Rc::new(PreASAPNode::Column(2)), + right: Rc::new(PreASAPNode::Column(2)), }, }, // comparison → Bool (nullable under 3-valued logic) ProjectItem { alias: Some("flag".into()), - expr: QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(2)), + expr: PreASAPNode::Compare { + left: Rc::new(PreASAPNode::Column(2)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(0.0))), + right: Rc::new(PreASAPNode::Literal(ScalarValue::Float64(0.0))), }, }, ], @@ -2252,7 +2256,7 @@ mod tests { // kept labels are the input labels minus the excluded `instance` (and ts // / value), followed by the `sum` column, and the schema stays OPEN // (issue #39). `job` survives; `instance` is dropped. - let scan_node = QueryExpr::Scan { + let scan_node = PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -2266,7 +2270,7 @@ mod tests { vec![], ), }; - let agg = QueryExpr::Aggregate { + let agg = PreASAPNode::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(vec![2])), // exclude `instance` measures: vec![AggIntent::Sum { col: None }], output_names: vec![], @@ -2285,7 +2289,7 @@ mod tests { #[test] fn without_aggregate_drops_a_renamed_sample_value() { // `sum without (inst) (sum by (inst, job) (m))` over `[inst, job, sum]`. - let inner = QueryExpr::Scan { + let inner = PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::new(vec![ @@ -2294,7 +2298,7 @@ mod tests { col("sum", DataType::Float64, false), ]), }; - let agg = QueryExpr::Aggregate { + let agg = PreASAPNode::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(vec![0])), measures: vec![AggIntent::Sum { col: None }], output_names: vec![], @@ -2319,7 +2323,7 @@ mod tests { Some(0), vec![], ); - let shifted = QueryExpr::TimeShift { + let shifted = PreASAPNode::TimeShift { shift: TimeShift { offset_ms: 3_600_000, at: Some(AtModifier::Timestamp(1_609_746_000_000)), @@ -2389,12 +2393,12 @@ mod tests { Some(0), vec![], ); - let rate = QueryExpr::Aggregate { + let rate = PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Rate], output_names: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: Rc::new(PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(scan_node), }), @@ -2427,12 +2431,12 @@ mod tests { Some(0), vec![], ); - let avg_over_time = QueryExpr::Aggregate { + let avg_over_time = PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Avg { col: None }], output_names: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: Rc::new(PreASAPNode::TimeRange { range: Duration::from_secs(300), child: Rc::new(scan_node), }), @@ -2457,7 +2461,7 @@ mod tests { // A schemaless (PromQL-style) leaf is *open*; it stays open through a // per-series reduction (`rate`), then is **frozen to closed** by a // cross-series aggregate (which enumerates exactly its output columns). - let open_leaf = QueryExpr::Scan { + let open_leaf = PreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], // `with_time_index` defaults to `closed: false` (open). @@ -2476,7 +2480,7 @@ mod tests { "schemaless leaf is open" ); - let rate = QueryExpr::Aggregate { + let rate = PreASAPNode::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Rate], output_names: vec![], @@ -2488,7 +2492,7 @@ mod tests { "per-series rate is label-preserving → stays open" ); - let sum_by_job = QueryExpr::Aggregate { + let sum_by_job = PreASAPNode::Aggregate { reduction: Reduction::by(vec![2]), // `job` measures: vec![AggIntent::Sum { col: None }], output_names: vec![], @@ -2511,17 +2515,17 @@ mod tests { Some(0), vec![], ); - let q = QueryExpr::Project { + let q = PreASAPNode::Project { qualifier: None, cols: vec![ // value=col 1, ts=col 0 ProjectItem { alias: None, - expr: QueryExpr::Column(1), + expr: PreASAPNode::Column(1), }, ProjectItem { alias: None, - expr: QueryExpr::Column(0), + expr: PreASAPNode::Column(0), }, ], child: Rc::new(child), @@ -2532,12 +2536,12 @@ mod tests { assert_eq!(s.time_index, Some(1)); } - fn join(kind: JoinKind) -> QueryExpr { + fn join(kind: JoinKind) -> PreASAPNode { let left = scan(vec![col("a", DataType::Int64, false)], None, vec![vec![0]]); let right = scan(vec![col("b", DataType::Utf8, false)], None, vec![]); - QueryExpr::Join { + PreASAPNode::Join { kind, - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + pred: Predicate(Rc::new(PreASAPNode::Literal(ScalarValue::Boolean(true)))), left: Rc::new(left), right: Rc::new(right), } @@ -2585,7 +2589,7 @@ mod tests { None, vec![vec![0]], ); - let q = QueryExpr::SetOp { + let q = PreASAPNode::SetOp { kind: RelationalSetOpKind::Union, all: false, left: Rc::new(left), @@ -2602,17 +2606,19 @@ mod tests { // ── PromqlScalarBridge / Literal dedup (issue #220) ───────────────────── - /// `QueryExpr::promql_scalar(v)` — what every front end now constructs in + /// `PreASAPNode::promql_scalar(v)` — what every front end now constructs in /// place of the old `PromqlScalar(v)` leaf — wraps exactly /// `Literal(ScalarValue::Float64(v))`: the same value a SQL-emitted typed /// float literal in a scalar-sub-language position would carry, just at a /// different tree position. `as_promql_scalar` is the round-trip inverse. #[test] fn promql_scalar_bridges_a_literal_float_at_an_operator_position() { - let bridge = QueryExpr::::promql_scalar(2.5); + let bridge = PreASAPNode::::promql_scalar(2.5); assert_eq!( bridge, - QueryExpr::PromqlScalarBridge(Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.5)))) + PreASAPNode::PromqlScalarBridge(Rc::new(PreASAPNode::Literal(ScalarValue::Float64( + 2.5 + )))) ); assert_eq!(bridge.as_promql_scalar(), Some(2.5)); @@ -2620,7 +2626,7 @@ mod tests { // its native (unwrapped, no row schema) scalar-sub-language position — // no longer a different variant, just not bridged to this tree // position. - let sql_literal = QueryExpr::::Literal(ScalarValue::Float64(2.5)); + let sql_literal = PreASAPNode::::Literal(ScalarValue::Float64(2.5)); assert_eq!(bridge.as_promql_scalar(), Some(2.5)); assert_ne!( bridge, sql_literal, @@ -2641,7 +2647,7 @@ mod tests { /// duplicate variants was used, only by whether the wrapper is present. #[test] fn row_schema_rides_on_the_bridge_wrapper_not_the_literal_variant() { - let bridged = QueryExpr::::promql_scalar(42.0); + let bridged = PreASAPNode::::promql_scalar(42.0); let schema = bridged.output_schema().expect("bridge has a row schema"); assert_eq!(schema.columns.len(), 1); assert_eq!(schema.columns[0].name, "value"); @@ -2652,10 +2658,10 @@ mod tests { // `Compare`/`Arithmetic` operand would occupy) has no row schema of // its own — it's a construction bug to call `output_schema` on it // directly, caught as `ScalarHasNoRowSchema` rather than panicking. - let bare = QueryExpr::::Literal(ScalarValue::Float64(42.0)); + let bare = PreASAPNode::::Literal(ScalarValue::Float64(42.0)); assert!(matches!( bare.output_schema(), - Err(QueryExprError::ScalarHasNoRowSchema) + Err(PreASAPNodeError::ScalarHasNoRowSchema) )); } @@ -2679,14 +2685,14 @@ mod tests { labels: vec!["host".into()], grouping: None, }; - let op = QueryExpr::BinaryOp { + let op = PreASAPNode::BinaryOp { op: BinaryOpKind::Compare(CompareOpKind::Gt), lhs: Rc::new(vector.clone()), - rhs: Rc::new(QueryExpr::promql_scalar(1.0)), + rhs: Rc::new(PreASAPNode::promql_scalar(1.0)), vector_match: Some(vm.clone()), }; assert_eq!(op.output_schema().unwrap(), vector.output_schema().unwrap()); - let QueryExpr::BinaryOp { vector_match, .. } = &op else { + let PreASAPNode::BinaryOp { vector_match, .. } = &op else { unreachable!() }; assert_eq!(vector_match.as_ref(), Some(&vm)); diff --git a/crates/types/src/pre_asap/resolve.rs b/crates/types/src/pre_asap/resolve.rs index 2ad60eb6..f878afa9 100644 --- a/crates/types/src/pre_asap/resolve.rs +++ b/crates/types/src/pre_asap/resolve.rs @@ -1,21 +1,21 @@ -//! Resolve a front-end-emitted, unresolved [`UnresolvedQueryExpr`] (`QueryExpr`) -//! into the canonical, positional [`ResolvedQueryExpr`] (`QueryExpr`). +//! Resolve a front-end-emitted, unresolved [`UnresolvedPreASAPNode`] (`PreASAPNode`) +//! into the canonical, positional [`ResolvedPreASAPNode`] (`PreASAPNode`). //! //! Both front ends (`asap-frontend-promql`, `asap-frontend-sql`) construct -//! canonical `QueryExpr` shapes directly during their own `interpret` step +//! canonical `PreASAPNode` shapes directly during their own `interpret` step //! (issue #179) — heavy-hitter `topk` recognition, the window-over-aggregate //! fold, the `PerEntity`/`Reduce` reduction choice, and every other //! *structural* decision happen right there, since a front end already knows //! the answer at parse time. What's left for [`resolve_root`] is exactly the //! "mechanical, schema-dependent substitution" #179 describes: a single -//! generic, shape-preserving walk — every [`UnresolvedQueryExpr`] variant maps to the -//! identical [`ResolvedQueryExpr`] variant — that resolves every [`ColumnRef`] to +//! generic, shape-preserving walk — every [`UnresolvedPreASAPNode`] variant maps to the +//! identical [`ResolvedPreASAPNode`] variant — that resolves every [`ColumnRef`] to //! the [`SchemaResolver`](super::schema_resolver::SchemaResolver)-computed positional [`ColumnId`]. //! //! ## Why positional `ColumnId`, not just carrying names all the way through (issue #216) //! //! A mature query engine can legitimately choose either design — DataFusion's -//! own logical plan (what `asap-frontend-sql` walks to build its `QueryExpr`) +//! own logical plan (what `asap-frontend-sql` walks to build its `PreASAPNode`) //! and Calcite both keep names, with an optional table qualifier, all the way //! through logical optimization, only going positional once they lower to a //! physical plan. Resolving once, immediately after each front end's own @@ -58,13 +58,13 @@ use super::column_resolution::{ }; use super::expr_ir::ColumnRef; use super::query_expr::{ - aggregate_output_schema, ConcatDiscriminatorKey, GroupKeys, Predicate, ProjectItem, - QueryExprError, Reduction, ResolvedQueryExpr, SortKey, UnresolvedQueryExpr, + aggregate_output_schema, ConcatDiscriminatorKey, GroupKeys, PreASAPNodeError, Predicate, + ProjectItem, Reduction, ResolvedPreASAPNode, SortKey, UnresolvedPreASAPNode, }; use super::schema::{ColumnId, Schema}; use super::schema_resolver::SchemaResolver; -/// Errors from resolving a canonical, unresolved [`UnresolvedQueryExpr`] tree. +/// Errors from resolving a canonical, unresolved [`UnresolvedPreASAPNode`] tree. #[derive(Debug, Error)] pub enum ResolveTreeError { /// A column reference did not resolve against its in-scope schema. @@ -73,23 +73,23 @@ pub enum ResolveTreeError { /// Deriving the schema of an already-resolved child failed (needed to /// resolve positional column references against it). #[error("schema derivation failed: {0}")] - Schema(#[from] QueryExprError), + Schema(#[from] PreASAPNodeError), } -/// Resolve a whole [`UnresolvedQueryExpr`] tree rooted at `tree` into canonical -/// [`ResolvedQueryExpr`]: binds every `ColumnRef` to a `ColumnId` via the +/// Resolve a whole [`UnresolvedPreASAPNode`] tree rooted at `tree` into canonical +/// [`ResolvedPreASAPNode`]: binds every `ColumnRef` to a `ColumnId` via the /// [`SchemaResolver`], then [`canonicalize`](super::canonicalize::canonicalize)s the /// result. -pub fn resolve_root(tree: &UnresolvedQueryExpr) -> Result { +pub fn resolve_root(tree: &UnresolvedPreASAPNode) -> Result { resolve_root_with_inherited(tree, &[]) } /// [`resolve_root`] with label names inherited from an enclosing scope seeded /// into the leaf schema, used when re-binding a `BinaryOp` side (issue #52). fn resolve_root_with_inherited( - tree: &UnresolvedQueryExpr, + tree: &UnresolvedPreASAPNode, inherited: &[String], -) -> Result { +) -> Result { let fallback = SchemaResolver::new().resolve_schema_with_inherited(tree, inherited); let l3 = resolve(tree, &fallback)?; Ok(super::canonicalize::canonicalize(l3)) @@ -100,10 +100,10 @@ fn resolve_root_with_inherited( /// derived output schema — so a `JOIN`'s concatenated schema and a cross- /// series aggregate's frozen-closed output bind to the right positions. fn resolve( - tree: &UnresolvedQueryExpr, + tree: &UnresolvedPreASAPNode, fallback: &Schema, -) -> Result { - use super::query_expr::QueryExpr as QE; +) -> Result { + use super::query_expr::PreASAPNode as QE; Ok(match tree { QE::Scan { source, @@ -263,7 +263,7 @@ fn resolve( .map(|key| -> Result<_, ResolveTreeError> { let schema = children .first() - .ok_or(QueryExprError::EmptyConcat)? + .ok_or(PreASAPNodeError::EmptyConcat)? .output_schema()?; Ok(ConcatDiscriminatorKey::new( resolve_column_ref(key.discriminator(), &schema)?, @@ -438,7 +438,7 @@ fn resolve( | QE::FunctionCall { .. } | QE::Arithmetic { .. } | QE::Case { .. }) => { - unreachable!("resolve reached a scalar QueryExpr variant directly: {other:?}") + unreachable!("resolve reached a scalar PreASAPNode variant directly: {other:?}") } }) } @@ -616,7 +616,7 @@ mod tests { use super::*; use crate::pre_asap::expr_ir::CompareOpKind; use crate::pre_asap::query_expr::{ - BinaryOpKind, QueryExpr, Source, VectorMatch, VectorMatchKind, + BinaryOpKind, PreASAPNode, Source, VectorMatch, VectorMatchKind, }; // Both sides resolve with qualifiers; an unknown right input is an error. @@ -699,21 +699,21 @@ mod tests { labels: vec!["job".into()], grouping: None, }; - let unresolved: UnresolvedQueryExpr = QueryExpr::BinaryOp { + let unresolved: UnresolvedPreASAPNode = PreASAPNode::BinaryOp { op: BinaryOpKind::Compare(CompareOpKind::Gt), - lhs: Rc::new(UnresolvedQueryExpr::Scan { + lhs: Rc::new(UnresolvedPreASAPNode::Scan { source: Source::TimeSeries { metric: "up".into(), }, predicates: vec![], schema: None, }), - rhs: Rc::new(UnresolvedQueryExpr::promql_scalar(1.0)), + rhs: Rc::new(UnresolvedPreASAPNode::promql_scalar(1.0)), vector_match: Some(vm.clone()), }; let resolved = resolve_root(&unresolved).expect("resolves"); - let QueryExpr::BinaryOp { + let PreASAPNode::BinaryOp { lhs, rhs, vector_match, @@ -722,7 +722,7 @@ mod tests { else { panic!("expected a resolved BinaryOp, got {resolved:?}"); }; - assert!(matches!(lhs.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(lhs.as_ref(), PreASAPNode::Scan { .. })); assert_eq!(rhs.as_promql_scalar(), Some(1.0)); assert_eq!(vector_match.as_ref(), Some(&vm)); @@ -744,19 +744,19 @@ mod tests { /// the branch's own (usage-derived) schema. #[test] fn resolve_root_seeds_and_resolves_an_otherwise_unreferenced_discriminator_column() { - let branch = || UnresolvedQueryExpr::Scan { + let branch = || UnresolvedPreASAPNode::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: None, }; - let unresolved = UnresolvedQueryExpr::concat_with_discriminator( + let unresolved = UnresolvedPreASAPNode::concat_with_discriminator( vec![branch(), branch()], ColumnRef::Named("phi".into()), vec![ColumnRef::Named("host".into())], ); let resolved = resolve_root(&unresolved).expect("resolves"); - let QueryExpr::Concat { + let PreASAPNode::Concat { children, discriminator_unique_key, } = &resolved diff --git a/crates/types/src/pre_asap/scalar_signature.rs b/crates/types/src/pre_asap/scalar_signature.rs index 7d61f624..23c14f48 100644 --- a/crates/types/src/pre_asap/scalar_signature.rs +++ b/crates/types/src/pre_asap/scalar_signature.rs @@ -200,16 +200,16 @@ mod tests { #[cfg(test)] mod projection_tests { use super::*; - use crate::pre_asap::{Column, ProjectItem, QueryExpr, ScalarValue, Schema, Source}; + use crate::pre_asap::{Column, PreASAPNode, ProjectItem, ScalarValue, Schema, Source}; use std::rc::Rc; - fn project(expr: QueryExpr) -> QueryExpr { - QueryExpr::Project { + fn project(expr: PreASAPNode) -> PreASAPNode { + PreASAPNode::Project { cols: vec![ProjectItem { alias: Some("result".into()), expr, }], qualifier: None, - child: Rc::new(QueryExpr::Scan { + child: Rc::new(PreASAPNode::Scan { source: Source::Table { table_ref: "t".into(), }, @@ -223,9 +223,9 @@ mod projection_tests { } #[test] fn canonical_projection_uses_map_signature_and_rejects_invalid_arity() { - let map = QueryExpr::FunctionCall { + let map = PreASAPNode::FunctionCall { name: "map".into(), - args: vec![QueryExpr::Column(0), QueryExpr::Column(1)], + args: vec![PreASAPNode::Column(0), PreASAPNode::Column(1)], }; let schema = project(map.clone()).output_schema().unwrap(); assert_eq!( @@ -237,17 +237,20 @@ mod projection_tests { } ); assert!(!schema.columns[0].nullable); - let lookup = QueryExpr::FunctionCall { + let lookup = PreASAPNode::FunctionCall { name: "asap_map_access".into(), - args: vec![map, QueryExpr::Literal(ScalarValue::Utf8("missing".into()))], + args: vec![ + map, + PreASAPNode::Literal(ScalarValue::Utf8("missing".into())), + ], }; assert_eq!( project(lookup).output_schema().unwrap().columns[0], Column::new("result", DataType::Int64, true) ); - assert!(project(QueryExpr::FunctionCall { + assert!(project(PreASAPNode::FunctionCall { name: "map".into(), - args: vec![QueryExpr::Column(0)] + args: vec![PreASAPNode::Column(0)] }) .output_schema() .is_err()); @@ -260,10 +263,10 @@ mod projection_tests { /// Dynamic/negative/defaulted selectors and nullable containers are intentionally /// unsupported here; this is not a claim of complete native tupleElement support. pub fn struct_field_type( - args: &[super::QueryExpr], + args: &[super::PreASAPNode], schema: &super::Schema, ) -> Result<(DataType, bool), String> { - use super::{QueryExpr, ScalarValue}; + use super::{PreASAPNode, ScalarValue}; let [input, selector] = args else { return Err("struct field access requires a struct and constant selector".into()); }; @@ -277,11 +280,13 @@ pub fn struct_field_type( return Err("struct field access requires a Struct input".into()); }; let field = match selector { - QueryExpr::Literal(ScalarValue::Int64(index)) if *index > 0 => usize::try_from(*index - 1) - .ok() - .and_then(|index| fields.get(index)) - .ok_or("struct field ordinal is out of bounds")?, - QueryExpr::Literal(ScalarValue::Utf8(name)) => { + PreASAPNode::Literal(ScalarValue::Int64(index)) if *index > 0 => { + usize::try_from(*index - 1) + .ok() + .and_then(|index| fields.get(index)) + .ok_or("struct field ordinal is out of bounds")? + } + PreASAPNode::Literal(ScalarValue::Utf8(name)) => { let mut matches = fields.iter().filter(|field| field.name == *name); let field = matches.next().ok_or("struct field name does not exist")?; if matches.next().is_some() { @@ -301,7 +306,7 @@ pub fn struct_field_type( #[cfg(test)] mod struct_field_tests { use super::*; - use crate::pre_asap::{Column, QueryExpr, ScalarValue, Schema}; + use crate::pre_asap::{Column, PreASAPNode, ScalarValue, Schema}; fn schema() -> Schema { Schema::new(vec![Column::new( "record", @@ -320,23 +325,23 @@ mod struct_field_tests { false, )]) } - fn access(selector: QueryExpr) -> QueryExpr { - QueryExpr::FunctionCall { + fn access(selector: PreASAPNode) -> PreASAPNode { + PreASAPNode::FunctionCall { name: "asap_struct_field".into(), - args: vec![QueryExpr::Column(0), selector], + args: vec![PreASAPNode::Column(0), selector], } } #[test] fn field_access_reuses_nested_field_type_and_nullability() { let schema = schema(); assert_eq!( - access(QueryExpr::Literal(ScalarValue::Int64(1))) + access(PreASAPNode::Literal(ScalarValue::Int64(1))) .scalar_type(&schema) .unwrap(), (DataType::Int64, false) ); - let named = access(QueryExpr::Literal(ScalarValue::Utf8("values".into()))); - let ordinal = access(QueryExpr::Literal(ScalarValue::Int64(2))); + let named = access(PreASAPNode::Literal(ScalarValue::Utf8("values".into()))); + let ordinal = access(PreASAPNode::Literal(ScalarValue::Int64(2))); assert_eq!( named.scalar_type(&schema).unwrap(), ordinal.scalar_type(&schema).unwrap() @@ -350,18 +355,18 @@ mod struct_field_tests { true ) ); - let roundtrip: QueryExpr = + let roundtrip: PreASAPNode = serde_json::from_str(&serde_json::to_string(&named).unwrap()).unwrap(); assert_eq!(roundtrip, named); } #[test] fn unsupported_field_access_is_an_error_not_placeholder_typing() { for selector in [ - QueryExpr::Column(0), - QueryExpr::Literal(ScalarValue::Int64(0)), - QueryExpr::Literal(ScalarValue::Int64(-1)), - QueryExpr::Literal(ScalarValue::Int64(3)), - QueryExpr::Literal(ScalarValue::Utf8("missing".into())), + PreASAPNode::Column(0), + PreASAPNode::Literal(ScalarValue::Int64(0)), + PreASAPNode::Literal(ScalarValue::Int64(-1)), + PreASAPNode::Literal(ScalarValue::Int64(3)), + PreASAPNode::Literal(ScalarValue::Utf8("missing".into())), ] { assert!(access(selector).scalar_type(&schema()).is_err()); } @@ -369,12 +374,12 @@ mod struct_field_tests { if let DataType::Struct { fields } = &mut ambiguous.columns[0].dtype { fields.push(Column::new("ts", DataType::Utf8, false)); } - assert!(access(QueryExpr::Literal(ScalarValue::Utf8("ts".into()))) + assert!(access(PreASAPNode::Literal(ScalarValue::Utf8("ts".into()))) .scalar_type(&ambiguous) .is_err()); let mut nullable = schema(); nullable.columns[0].nullable = true; - assert!(access(QueryExpr::Literal(ScalarValue::Int64(1))) + assert!(access(PreASAPNode::Literal(ScalarValue::Int64(1))) .scalar_type(&nullable) .is_err()); } @@ -387,10 +392,10 @@ mod struct_field_tests { /// behavior depends on whether the input array is constant. Nullable containers /// are unsupported; nullable indices produce nullable results. pub fn element_access_type( - args: &[super::QueryExpr], + args: &[super::PreASAPNode], schema: &super::Schema, ) -> Result<(DataType, bool), String> { - use super::{QueryExpr, ScalarValue}; + use super::{PreASAPNode, ScalarValue}; let [input, index] = args else { return Err("element access requires a collection and index".into()); }; @@ -405,7 +410,7 @@ pub fn element_access_type( if !matches!(key.0, DataType::Int64 | DataType::Null) { return Err("List index must have integer type".into()); } - if matches!(index, QueryExpr::Literal(ScalarValue::Int64(0))) { + if matches!(index, PreASAPNode::Literal(ScalarValue::Int64(0))) { return Err( "literal zero List index is unsupported without constant-array proof".into(), ); @@ -422,11 +427,11 @@ pub fn element_access_type( #[cfg(test)] mod element_access_tests { use super::*; - use crate::pre_asap::{Column, QueryExpr, ScalarValue, Schema}; - fn access(index: QueryExpr) -> QueryExpr { - QueryExpr::FunctionCall { + use crate::pre_asap::{Column, PreASAPNode, ScalarValue, Schema}; + fn access(index: PreASAPNode) -> PreASAPNode { + PreASAPNode::FunctionCall { name: "asap_element_access".into(), - args: vec![QueryExpr::Column(0), index], + args: vec![PreASAPNode::Column(0), index], } } #[test] @@ -449,34 +454,34 @@ mod element_access_tests { ]); for index in [1, -1, 100] { assert_eq!( - access(QueryExpr::Literal(ScalarValue::Int64(index))) + access(PreASAPNode::Literal(ScalarValue::Int64(index))) .scalar_type(&schema) .unwrap(), (element.clone(), false) ); } assert_eq!( - access(QueryExpr::Column(1)).scalar_type(&schema).unwrap(), + access(PreASAPNode::Column(1)).scalar_type(&schema).unwrap(), (element.clone(), true) ); - assert!(access(QueryExpr::Literal(ScalarValue::Int64(0))) + assert!(access(PreASAPNode::Literal(ScalarValue::Int64(0))) .scalar_type(&schema) .is_err()); - assert!(access(QueryExpr::Literal(ScalarValue::Float64(1.0))) + assert!(access(PreASAPNode::Literal(ScalarValue::Float64(1.0))) .scalar_type(&schema) .is_err()); - let nested = QueryExpr::FunctionCall { + let nested = PreASAPNode::FunctionCall { name: "asap_struct_field".into(), args: vec![ - access(QueryExpr::Literal(ScalarValue::Int64(1))), - QueryExpr::Literal(ScalarValue::Int64(2)), + access(PreASAPNode::Literal(ScalarValue::Int64(1))), + PreASAPNode::Literal(ScalarValue::Int64(2)), ], }; assert_eq!( nested.scalar_type(&schema).unwrap(), (DataType::Float64, true) ); - let roundtrip: QueryExpr = + let roundtrip: PreASAPNode = serde_json::from_value(serde_json::to_value(&nested).unwrap()).unwrap(); assert_eq!(roundtrip, nested); } @@ -491,10 +496,10 @@ mod element_access_tests { }, false, )]); - let key = QueryExpr::Literal(ScalarValue::Utf8("k".into())); - let legacy = QueryExpr::FunctionCall { + let key = PreASAPNode::Literal(ScalarValue::Utf8("k".into())); + let legacy = PreASAPNode::FunctionCall { name: "asap_map_access".into(), - args: vec![QueryExpr::Column(0), key.clone()], + args: vec![PreASAPNode::Column(0), key.clone()], }; assert_eq!( access(key).scalar_type(&schema).unwrap(), diff --git a/crates/types/src/pre_asap/schema.rs b/crates/types/src/pre_asap/schema.rs index 9d5b2e22..66003f49 100644 --- a/crates/types/src/pre_asap/schema.rs +++ b/crates/types/src/pre_asap/schema.rs @@ -167,13 +167,15 @@ pub const PROMQL_SERIES_IDENTITY: &str = "$promql_series_identity"; /// Operators that rewrite or implicitly match dynamic label sets require their /// own realization; they must not accidentally treat the opaque identity as a /// user label or silently discard it. -pub fn with_promql_series_identity(root: &super::QueryExpr) -> Result { - use super::{QueryExpr, Source}; +pub fn with_promql_series_identity( + root: &super::PreASAPNode, +) -> Result { + use super::{PreASAPNode, Source}; use std::rc::Rc; let mut root = root.clone(); - fn visit(node: &mut QueryExpr) -> Result<(), String> { + fn visit(node: &mut PreASAPNode) -> Result<(), String> { match node { - QueryExpr::Scan { + PreASAPNode::Scan { source: Source::TimeSeries { .. }, schema, .. @@ -194,29 +196,29 @@ pub fn with_promql_series_identity(root: &super::QueryExpr) -> Result visit(Rc::make_mut(child)), + PreASAPNode::TimeRange { child, .. } + | PreASAPNode::Limit { child, .. } + | PreASAPNode::TimeShift { child, .. } + | PreASAPNode::PromqlSubquery { child, .. } + | PreASAPNode::PromqlScalarFromVector(child) + | PreASAPNode::PromqlRelabel { child, .. } => visit(Rc::make_mut(child)), // Constants read no series. - QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::Literal(super::ScalarValue::Float64(_)) => Ok(()), - QueryExpr::PromqlVectorFromScalar(child) => visit(Rc::make_mut(child)), - QueryExpr::BinaryOp { lhs, rhs, .. } => { + PreASAPNode::PromqlScalarBridge(_) + | PreASAPNode::EvalTimestamp + | PreASAPNode::Literal(super::ScalarValue::Float64(_)) => Ok(()), + PreASAPNode::PromqlVectorFromScalar(child) => visit(Rc::make_mut(child)), + PreASAPNode::BinaryOp { lhs, rhs, .. } => { visit(Rc::make_mut(lhs))?; visit(Rc::make_mut(rhs)) } - QueryExpr::Concat { children, .. } => { + PreASAPNode::Concat { children, .. } => { for child in children { visit(child)?; } Ok(()) } - QueryExpr::Aggregate { child, .. } => visit(Rc::make_mut(child)), - QueryExpr::Sort { + PreASAPNode::Aggregate { child, .. } => visit(Rc::make_mut(child)), + PreASAPNode::Sort { child, partition_by, .. @@ -396,8 +398,8 @@ mod tests { // Direct scalar literals remain valid vector inputs when series typing runs. #[test] fn series_identity_accepts_direct_vector_literal() { - let root = super::super::QueryExpr::PromqlVectorFromScalar(std::rc::Rc::new( - super::super::QueryExpr::Literal(super::super::ScalarValue::Float64(1.0)), + let root = super::super::PreASAPNode::PromqlVectorFromScalar(std::rc::Rc::new( + super::super::PreASAPNode::Literal(super::super::ScalarValue::Float64(1.0)), )); assert!(with_promql_series_identity(&root).is_ok()); } diff --git a/crates/types/src/pre_asap/schema_resolver.rs b/crates/types/src/pre_asap/schema_resolver.rs index 8244d9be..ef6786a5 100644 --- a/crates/types/src/pre_asap/schema_resolver.rs +++ b/crates/types/src/pre_asap/schema_resolver.rs @@ -12,7 +12,7 @@ //! lands, only the catalog impl swaps. use super::expr_ir::ColumnRef; -use super::query_expr::UnresolvedQueryExpr; +use super::query_expr::UnresolvedPreASAPNode; use super::schema::{Column, DataType, Schema}; /// The DB / source-schema metadata source — resolves a source (metric / @@ -71,7 +71,7 @@ impl SchemaResolver { /// Contains the time axis, the synthetic `value` column, and one column /// per distinct name referenced anywhere in the tree — so positional /// `ColumnId` resolution downstream is total. - pub fn resolve_schema(&self, tree: &UnresolvedQueryExpr) -> Schema { + pub fn resolve_schema(&self, tree: &UnresolvedPreASAPNode) -> Schema { self.resolve_schema_with_inherited(tree, &[]) } @@ -83,7 +83,7 @@ impl SchemaResolver { /// side's own matchers (issue #52). pub fn resolve_schema_with_inherited( &self, - tree: &UnresolvedQueryExpr, + tree: &UnresolvedPreASAPNode, inherited: &[String], ) -> Schema { let mut columns: Vec = leftmost_scan_name(tree) @@ -136,12 +136,12 @@ fn push_ref_name(c: &ColumnRef, out: &mut Vec) { } } -/// The leftmost `Scan`'s source name in a canonical (`UnresolvedQueryExpr`) tree — +/// The leftmost `Scan`'s source name in a canonical (`UnresolvedPreASAPNode`) tree — /// the [`collect_referenced_columns`] counterpart to what a dedicated /// `Source` leaf type would carry as a method; the canonical tree's `Scan` /// leaf needs this walk written out instead. -fn leftmost_scan_name(tree: &UnresolvedQueryExpr) -> Option<&str> { - use UnresolvedQueryExpr as QE; +fn leftmost_scan_name(tree: &UnresolvedPreASAPNode) -> Option<&str> { + use UnresolvedPreASAPNode as QE; match tree { QE::Scan { source, .. } => Some(match source { super::query_expr::Source::TimeSeries { metric } => metric.as_str(), @@ -193,16 +193,16 @@ fn leftmost_scan_name(tree: &UnresolvedQueryExpr) -> Option<&str> { /// Collect every distinct column name referenced anywhere in `tree` that /// resolves positionally — every place a front end constructing -/// [`QueryExpr`](super::query_expr::QueryExpr) directly (issue +/// [`PreASAPNode`](super::query_expr::PreASAPNode) directly (issue /// #179) puts a name-based reference: `Scan.predicates`, `Aggregate`'s /// `reduction`/`having`/per-measure `col`, `Dedup.cols`, `PromqlSeriesSample.by`, /// `Filter.pred`, `Project.cols`, `Sort.keys`/`partition_by`, /// `SQLWindowFunc.args`/`partition_by`/`order_by`, `Join.pred`, `PromqlRelabel.value`. /// The SchemaResolver seeds these into the usage-derived leaf so positional /// resolution downstream is total. -pub(crate) fn collect_referenced_columns(tree: &UnresolvedQueryExpr) -> Vec { - use UnresolvedQueryExpr as QE; - fn named(expr: &UnresolvedQueryExpr, out: &mut Vec) { +pub(crate) fn collect_referenced_columns(tree: &UnresolvedPreASAPNode) -> Vec { + use UnresolvedPreASAPNode as QE; + fn named(expr: &UnresolvedPreASAPNode, out: &mut Vec) { for c in expr.columns_referenced() { push_ref_name(c, out); } @@ -217,7 +217,7 @@ pub(crate) fn collect_referenced_columns(tree: &UnresolvedQueryExpr) -> Vec) { + fn walk(node: &UnresolvedPreASAPNode, out: &mut Vec) { match node { QE::Scan { predicates, .. } => { for super::query_expr::Predicate(p) in predicates { @@ -354,7 +354,7 @@ pub(crate) fn collect_referenced_columns(tree: &UnresolvedQueryExpr) -> Vec { - unreachable!("walk reached a scalar QueryExpr variant directly: {node:?}") + unreachable!("walk reached a scalar PreASAPNode variant directly: {node:?}") } } } @@ -372,8 +372,8 @@ mod tests { use super::super::query_expr::{GroupKeys, Source}; use super::*; - fn src(name: &str) -> UnresolvedQueryExpr { - UnresolvedQueryExpr::Scan { + fn src(name: &str) -> UnresolvedPreASAPNode { + UnresolvedPreASAPNode::Scan { source: Source::TimeSeries { metric: name.into(), }, @@ -386,7 +386,7 @@ mod tests { #[test] fn pearson_corr_inputs_seed_usage_derived_schema() { use crate::pre_asap::{AggIntent, Reduction}; - let tree = UnresolvedQueryExpr::Aggregate { + let tree = UnresolvedPreASAPNode::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::PearsonCorr { left: ColumnRef::Named("x".into()), @@ -415,9 +415,9 @@ mod tests { fn sort_partition_keys_land_in_schema() { // Per-group ranking keys (`topk by (host)` → `Sort.partition_by`) must be // seeded into the usage-derived leaf so they resolve positionally. - let tree = UnresolvedQueryExpr::Sort { + let tree = UnresolvedPreASAPNode::Sort { keys: vec![super::super::query_expr::SortKey { - expr: UnresolvedQueryExpr::Column(ColumnRef::SampleValue), + expr: UnresolvedPreASAPNode::Column(ColumnRef::SampleValue), ascending: false, nulls_first: false, }], @@ -435,7 +435,7 @@ mod tests { /// column the caller correctly named. #[test] fn concat_discriminator_key_is_seeded_into_the_resolver_schema() { - let tree = UnresolvedQueryExpr::concat_with_discriminator( + let tree = UnresolvedPreASAPNode::concat_with_discriminator( vec![src("m")], ColumnRef::Named("phi".into()), vec![ColumnRef::Named("host".into())], diff --git a/crates/types/tests/planner_vocabulary.rs b/crates/types/tests/planner_vocabulary.rs index 567e2c28..210fcd40 100644 --- a/crates/types/tests/planner_vocabulary.rs +++ b/crates/types/tests/planner_vocabulary.rs @@ -1,7 +1,7 @@ use asap_types::post_asap::{ validate_pane_coverage, PaneLayout, WindowEdgeCompatibility, WindowEdgeCoverage, }; -use asap_types::pre_asap::{SchemaResolver, Source, UnresolvedQueryExpr}; +use asap_types::pre_asap::{SchemaResolver, Source, UnresolvedPreASAPNode}; use asap_types::resources::{PhysicalHandoffBytes, PhysicalHandoffKind}; // Renamed pane APIs still read and emit the deployed wire contract. @@ -33,7 +33,7 @@ fn window_edge_names_preserve_wire_values() { // External consumers can use the new resolver and resource names without changing behavior. #[test] fn renamed_schema_and_handoff_apis_are_public() { - let tree = UnresolvedQueryExpr::Scan { + let tree = UnresolvedPreASAPNode::Scan { source: Source::TimeSeries { metric: "requests".into(), }, diff --git a/docs/design_docs/architecture/README.md b/docs/design_docs/architecture/README.md index 9c32f623..12241957 100644 --- a/docs/design_docs/architecture/README.md +++ b/docs/design_docs/architecture/README.md @@ -7,7 +7,7 @@ as ASAPQuery-backend bind the candidates to physical alternatives, make the deployment-level decision, and run the selected contract. For the integration workflow, start with [ASAPPlanner input, output, and -workflows](input-output-workflow.md). It defines inputs, `CandidatePostASAPDAGs`, selection +workflows](input-output-workflow.md). It defines inputs, `CandidateLogicalPostASAPDAGs`, selection and summary-maintenance lifecycle workflows, and future replanning support. ## Planner component flow @@ -17,16 +17,16 @@ flowchart TD W["PlanningWorkload: query demand + optional data facts"] F["Frontend dependencies: SQL catalog or PromQL time"] E["Strategy, accuracy model, and applicable evidence"] - PRE["Frontend lowering → canonical Pre-ASAP QueryExpr roots"] + PRE["Frontend lowering → canonical Pre-ASAP PreASAPNode roots"] SEARCH["Whole-workload candidate search: sharing, legality, accuracy"] - SPACE["CandidatePostASAPDAGs: compact logical candidate DAG space"] + SPACE["CandidateLogicalPostASAPDAGs: compact logical candidate DAG space"] RANK["Optional cost_sorted: ranked inspection view"] SELECT["Optional global_selection + assemble_selected_dag"] DAG["Selected logical Post-ASAP DAG"] LINPUT["Optional lifecycle inputs: horizon, rates, capabilities, costs"] LIFE["global_selection_with_summary_maintenance_lifecycles"] LMAT["assemble_selected_dag_with_summary_maintenance_lifecycles"] - LPLAN["SummaryMaintenanceLifecyclePlan: DAG root + lifecycle decisions"] + LPLAN["LifecyclePostASAPDAG: DAG root + lifecycle decisions"] BACKEND["Downstream: bind physical alternatives, decide deployment, compile and execute"] W --> PRE F --> PRE @@ -39,17 +39,17 @@ flowchart TD LINPUT --> LIFE --> LMAT --> LPLAN --> BACKEND ``` -`CandidatePostASAPDAGs` is the output of logical candidate search. Each target's candidate set holds +`CandidateLogicalPostASAPDAGs` is the output of logical candidate search. Each target's candidate set holds alternatives and rejection reasons, but no selected maintenance lifecycle. Choose among the three branches: inspect candidates (optionally ranked), select and assemble logical DAGs, or select and assemble with summary-maintenance lifecycle decisions. Use the last branch when Planner owns the maintenance decision; otherwise the backend owns it. Its first call returns a `GlobalSelection`; the second returns a -`SummaryMaintenanceLifecyclePlan` with an assembled DAG root and lifecycle +`LifecyclePostASAPDAG` with an assembled DAG root and lifecycle decisions. No branch by itself deploys or executes a physical plan. Known-invalid evidence rejects a logical candidate. Missing accuracy evidence -leaves a constructible candidate visible in `CandidatePostASAPDAGs` but uncertified; default +leaves a constructible candidate visible in `CandidateLogicalPostASAPDAGs` but uncertified; default selection does not commit it without the required guarantee. Cost evidence can rank eligible candidates, but it cannot establish a missing guarantee or turn an unsupported physical alternative into a deployable plan. @@ -72,7 +72,7 @@ requirements, the planning horizon, available materialized state, downstream capabilities, and complete cost evidence. Missing or stale evidence must remain explicit rather than being treated as zero. -The primary output is `CandidatePostASAPDAGs`; `cost_sorted` derives an optional ranked +The primary output is `CandidateLogicalPostASAPDAGs`; `cost_sorted` derives an optional ranked view with index-aligned costs. Downstream may inspect compatible choices across targets rather than assuming the first candidate is a feasible physical workload plan. Candidates carry logical summary algorithms, @@ -87,7 +87,7 @@ serving, and operational feedback. Their physical planning can reorder candidates because it has evidence that the reusable Planner does not, but it must not silently change Planner-owned semantics. -`CandidatePostASAPDAGs::global_selection` optionally coordinates structural choices across +`CandidateLogicalPostASAPDAGs::global_selection` optionally coordinates structural choices across targets; `GlobalSelection::assemble_selected_dag` constructs a selected semantic DAG. Those plain APIs do not establish physical feasibility or a maintenance-versus-recompute decision. The lifecycle-aware selection call uses diff --git a/docs/design_docs/architecture/asap-aware-mapping.md b/docs/design_docs/architecture/asap-aware-mapping.md index bf88f9b4..9ea683b2 100644 --- a/docs/design_docs/architecture/asap-aware-mapping.md +++ b/docs/design_docs/architecture/asap-aware-mapping.md @@ -7,7 +7,7 @@ ASAP-aware mapping decides **whether and how a query intent can be answered usin Given a logical query plan, the mapping layer explores alternative plans that may use sketches, exact summaries, shared computation, roll-ups, semantic rewrites, or combinations of these techniques. Candidate search takes canonical **Pre-ASAP query roots** and produces -`CandidatePostASAPDAGs`, a compact set of **candidate Post-ASAP DAGs**. Ranking, selection, +`CandidateLogicalPostASAPDAGs`, a compact set of **candidate Post-ASAP DAGs**. Ranking, selection, and summary-maintenance lifecycle decisions are subsequent operations over it; see [input, output, and workflows](input-output-workflow.md). @@ -71,7 +71,7 @@ Check compatibility between replacement sub-DAGs Apply applicable accuracy and semantic checks | v -CandidatePostASAPDAGs: compact candidate Post-ASAP DAGs +CandidateLogicalPostASAPDAGs: compact candidate Post-ASAP DAGs | +--> inspect / rank +--> select and assemble logical DAGs diff --git a/docs/design_docs/architecture/asap-aware-plan-search.md b/docs/design_docs/architecture/asap-aware-plan-search.md index 6f8a5bd7..f03e3501 100644 --- a/docs/design_docs/architecture/asap-aware-plan-search.md +++ b/docs/design_docs/architecture/asap-aware-plan-search.md @@ -80,7 +80,7 @@ sets. This avoids copying every full plan when most structure is shared. ## Current implementation boundary -`CandidatePostASAPDAGs` stores per-target candidates rather than eagerly enumerating their +`CandidateLogicalPostASAPDAGs` stores per-target candidates rather than eagerly enumerating their Cartesian product. `cost_sorted` returns a `RankedTargetSubDAGCandidates` view for each target. `global_selection` coordinates supported sharing and composition choices; `assemble_selected_dag(root)` assembles one selected DAG diff --git a/docs/design_docs/architecture/evidence-dependent-candidates.md b/docs/design_docs/architecture/evidence-dependent-candidates.md index 1a53c0b7..5a248cb0 100644 --- a/docs/design_docs/architecture/evidence-dependent-candidates.md +++ b/docs/design_docs/architecture/evidence-dependent-candidates.md @@ -2,7 +2,7 @@ Audience: ASAPPlanner library integrators, especially ASAPQuery-backend. -`CandidatePostASAPDAGs` is a space of constructible logical alternatives, not a list of +`CandidateLogicalPostASAPDAGs` is a space of constructible logical alternatives, not a list of certified deployment choices. Missing external evidence must not erase a candidate whose semantics and Post-ASAP shape are already known. It also must not turn an unknown guarantee into a satisfied accuracy requirement. @@ -21,7 +21,7 @@ The exact `KeepPreAsap` path has an exact guarantee. |---|---|---| | Known guarantee | `ResultGuarantee` with evaluable bound and failure probability | Planner checks whether the guarantee satisfies the query's accuracy target. If this candidate is selected, the backend checks whether its implementation can realize the selected summary; it does not re-decide the accuracy target. | | Missing accuracy/domain evidence | Symbolic `BoundExpr::Unknown` or `ProbabilityExpr::Unknown`, or `guarantee: None` on a constructible summary | Inspect `ReplacementSubDAG::has_missing_accuracy_evidence()`, obtain applicable evidence or apply explicit policy; do not claim certification. | -| Missing cost | `CostModel::candidate_cost()` returns `None` for a `ReplacementSubDAG` (including a non-finite or negative legacy estimate) | Keep that logical summary/rewrite candidate in `CandidatePostASAPDAGs` for inspection; provide a comparable cost before selecting it by cost. This does not make it a deployable physical plan. | +| Missing cost | `CostModel::candidate_cost()` returns `None` for a `ReplacementSubDAG` (including a non-finite or negative legacy estimate) | Keep that logical summary/rewrite candidate in `CandidateLogicalPostASAPDAGs` for inspection; provide a comparable cost before selecting it by cost. This does not make it a deployable physical plan. | | Unknown runtime support | `ReplacementSubDAG::runtime_support_evidence(model)` returns `None` | Candidate remains visible; bind a concrete implementation and confirm support before deployment. | | Known invalid evidence or impossible semantics | No candidate; where supported, a `RejectedCandidate` records the error | Do not deploy. | @@ -58,10 +58,10 @@ not emit a `RejectedCandidate` for that case. | HLL confidence | Symbolic failure probability | Reject a fully known unmet root target. | | Relative-value composition | Symbolic bound when input sign is unknown | Reject known signed input for this rule. | | Exact sum/average/extremum | Symbolic row-count probability term | Reject unsupported metric combinations. | -| Cost/rate/physical evidence | `None` cost or missing workload rate; candidate remains in `CandidatePostASAPDAGs` | Physical/lifecycle evaluation reports unavailable or rejected evidence. | -| Mixed exact/summary operator | Unknown runtime support; candidate remains in `CandidatePostASAPDAGs` | `Some(false)` prevents construction. | +| Cost/rate/physical evidence | `None` cost or missing workload rate; candidate remains in `CandidateLogicalPostASAPDAGs` | Physical/lifecycle evaluation reports unavailable or rejected evidence. | +| Mixed exact/summary operator | Unknown runtime support; candidate remains in `CandidateLogicalPostASAPDAGs` | `Some(false)` prevents construction. | -Lifecycle deployment choices are a separate output from `CandidatePostASAPDAGs`; their +Lifecycle assignments (`CandidateLifecyclePostASAPDAGs`) are a separate output from `CandidateLogicalPostASAPDAGs`; their capability/cost rejections do not erase the logical summary candidate. The backend must still check ordinary summary family, window, and state-operation capabilities before deployment. @@ -87,7 +87,7 @@ capabilities before deployment. The default `global_selection()` skips summaries that `has_missing_accuracy_evidence()` identifies as uncertified. Its `GlobalSelection::assemble_selected_dag()` result is a selected logical plan, -not an instruction to deploy every candidate in `CandidatePostASAPDAGs`. If no alternative +not an instruction to deploy every candidate in `CandidateLogicalPostASAPDAGs`. If no alternative is chosen at a site, DAG assembly retains the exact `KeepPreAsap` path. The backend can inspect alternatives, apply its own evidence and policy, then choose a physically supported one; it must not equate candidate presence with approval. @@ -108,7 +108,7 @@ backend. | PromQL input | Before this PR | After this PR | |---|---|---| -| `count by(job)(up)` with an ε/δ target | Hydra's shared CMS/CountSketch alternatives are absent: missing shared-grid bounds make the strategy decline the target. | Both Hydra alternatives remain in `CandidatePostASAPDAGs` with symbolic unknown bound/probability terms. `has_missing_accuracy_evidence()` is true; default `global_selection()` does not choose either as a certified answer. | +| `count by(job)(up)` with an ε/δ target | Hydra's shared CMS/CountSketch alternatives are absent: missing shared-grid bounds make the strategy decline the target. | Both Hydra alternatives remain in `CandidateLogicalPostASAPDAGs` with symbolic unknown bound/probability terms. `has_missing_accuracy_evidence()` is true; default `global_selection()` does not choose either as a certified answer. | | `entropy_over_time(m[5m])` with an ε target | The uncalibrated frequency readout has no `SummaryEstimate` candidate. | Its `SummaryEstimate` remains inspectable with `guarantee: None`. Default selection still skips it, so candidate visibility is not an accuracy certificate. | | `quantile_over_time(0.9,data[5m]) / quantile_over_time(0.5,data[5m])` with an ε target | The uncertified direct DDSketch ratio is **already** visible because of #449. | Still visible with `guarantee: None`, and still skipped by default selection. This is a regression/control example, not a new candidate introduced by this PR. | @@ -139,7 +139,7 @@ The same loop applies to grouped Count with Hydra and to Count-ranked TopK: their symbolic guarantees keep them visible until shared-grid or interval evidence is available. Re-run planning with a provider when a certified Planner selection is needed; supplying evidence to the backend alone does -not retroactively change the guarantees stored in the existing `CandidatePostASAPDAGs`. +not retroactively change the guarantees stored in the existing `CandidateLogicalPostASAPDAGs`. Backend integration work is tracked in [ASAPQuery-backend#752](https://github.com/ProjectASAP/ASAPQuery-backend/issues/752). diff --git a/docs/design_docs/architecture/input-output-workflow.md b/docs/design_docs/architecture/input-output-workflow.md index f57b31f5..a2807f31 100644 --- a/docs/design_docs/architecture/input-output-workflow.md +++ b/docs/design_docs/architecture/input-output-workflow.md @@ -5,10 +5,12 @@ This document is for library integrators such as ASAPQuery-backend, not users submitting queries through a backend. -ASAPPlanner is a **logical planning library**. Its input is a planning workload -plus the models, evidence, and deployment capabilities needed by the requested -planning workflow. Its canonical output is a `CandidatePostASAPDAGs` containing the legal -Post-ASAP alternatives for the workload. +ASAPPlanner provides frontend lowering, logical candidate search, summary +maintenance lifecycles, and physical compilation. Its input is a planning +workload plus the models, evidence, and deployment capabilities required by +the chosen workflow. The graph stages are `PreASAPDAG`, `LogicalPostASAPDAG`, and +`PhysicalPostASAPDAG`. Logical planning outputs `CandidateLogicalPostASAPDAGs`; +`CandidatePhysicalPostASAPDAGs` forms the execution boundary with a deployment. ### Input fields at a glance @@ -28,28 +30,42 @@ fields and [frontend dependencies](#frontend-specific-dependencies). | Output | Fields or contents | Meaning | |---|---|---| -| `CandidatePostASAPDAGs` | The legal candidate Post-ASAP DAGs for the workload, represented compactly as canonical roots, one candidate set per target sub-DAG, and cross-target composition information | The ASAPPlanner output | +| `CandidateLogicalPostASAPDAGs` | All legal logical candidates for the workload, represented compactly as canonical roots, one candidate set per target sub-DAG, and cross-target composition information | The candidate space: nothing is selected yet | +| `PlanOutput` | One selected LogicalPostASAPDAG root (`Rc`) per workload entry, optionally with summary-maintenance lifecycle decisions | One optimization pass's selection, returned by `e2e_plan` and `optimize` | + +`PlanOutput` is selected from the candidate space, not a second output beside +it. [Output layers](#output-layers) places both in the `PreASAPDAG` → +`LogicalPostASAPDAG` → `PhysicalPostASAPDAG` pipeline and maps these names to current APIs. [Ranking](#ranked-view), [selection and DAG assembly](#selection-and-dag-assembly), and -[summary-maintenance lifecycle](#summary-maintenance-lifecycle-aware-helper) APIs operate on this `CandidatePostASAPDAGs`. -These are alternative uses of the candidate space, not mandatory sequential -stages. `CandidatePostASAPDAGs` itself has no selected summary-maintenance lifecycle, and +[summary-maintenance lifecycle](#summary-maintenance-lifecycle-aware-helper) APIs operate on this candidate collection. +These are different uses of the candidate space, not mandatory sequential +stages. `CandidateLogicalPostASAPDAGs` itself has no selected summary-maintenance lifecycle, and its candidates do not choose precompute versus query-time placement: a chosen lifecycle assignment sets each node's execution timing. -The candidate DAGs are logical planning artifacts. ASAPPlanner does **not** -produce a deployed executable plan; downstream systems bind physical operators, -choose placement and storage, deploy state, and execute queries. +These candidates are logical `LogicalPostASAPDAG`s. Planner also compiles +physical candidates. A deployment supplies prices for lifecycle assignments +and physical candidates, with its accuracy requirements and capabilities; +Planner selection returns the optimal plan. The deployment then binds the +compiled typed inputs, owns storage, installs state, and executes the +precompute and query DAGs. Planner does not install a deployment plan. ```text PlanningWorkload + frontend dependencies + planning models/evidence | v - ASAPPlanner + Frontends -> CandidatePreASAPDAGs + | + v + CandidateLogicalPostASAPDAGs | v - CandidatePostASAPDAGs: candidate Post-ASAP DAGs + lifecycle assignments -> CandidateLifecyclePostASAPDAGs + | + v + compile and cut by timing -> CandidatePhysicalPostASAPDAGs ``` --- @@ -66,44 +82,45 @@ flowchart TD D["data_workload: continuous arrival; declared ingestion interval 15 s"] T["Frontend argument: now_ms"] F["PromQL lowering"] - R["One canonical QueryExpr root"] + R["One canonical PreASAPDAG root"] S["Candidate search"] - P["CandidatePostASAPDAGs: logical choices for this root"] + P["CandidateLogicalPostASAPDAGs for this root"] I["cost_sorted: inspect choices"] G["global_selection + assemble_selected_dag(root)"] - L["One selected Post-ASAP DAG; exact KeepPreAsap if no optimization is selected"] + L["One selected LogicalPostASAPDAG; exact KeepPreAsap if no optimization is selected"] X["Extra lifecycle inputs: horizon; update rate; capabilities; comparable summary/raw costs"] H["Summary-maintenance-lifecycle-aware selection"] HM["Assemble one selected DAG and decide summary maintenance"] - O["SummaryMaintenanceLifecyclePlan: assembled DAG root + maintenance/recompute decision"] - B["Backend: bind physical operators, deploy, and execute"] + O["LifecyclePostASAPDAG: assembled DAG root annotated with its lifecycle assignment"] + PC["Planner: lifecycle timing, compile PhysicalPostASAPDAG, cut PhysicalCandidate"] + B["Backend: bind typed inputs, deploy, and execute"] Q --> F D --> F T --> F F --> R --> S --> P P --> I - P --> G --> L --> B + P --> G --> L --> PC --> B P --> H - X --> H --> HM --> O --> B + X --> H --> HM --> O --> PC ``` “Predictable” says the query is known in advance; it is independent of its -one-minute recurrence. The `CandidatePostASAPDAGs` may contain an exact count-summary +one-minute recurrence. The candidate collection may contain an exact count-summary realization, but it is not a deployed query. Without the extra lifecycle inputs, the caller can still inspect candidates or obtain a logical DAG; it cannot conclude that maintaining a summary is cheaper than recomputing raw results. For contrast, a one-time SQL query needs a catalog but need not supply data -arrival evidence merely to inspect logical alternatives: +arrival evidence merely to inspect logical candidates: ```mermaid flowchart LR Q["query_batch: SELECT COUNT(*) FROM metrics; invocations 1; AdHoc"] C["SqlCatalog: resolves metrics and its columns"] F["SQL lowering"] - R["One QueryExpr root"] - P["Candidate search → CandidatePostASAPDAGs"] + R["One PreASAPDAG root"] + P["Candidate search → CandidateLogicalPostASAPDAGs"] Q --> F C --> F F --> R --> P @@ -111,7 +128,7 @@ flowchart LR In this SQL example, `data_workload` can be `None` if the chosen lowering and search rules do not consume it. The lifecycle helper is not needed merely to -inspect the `CandidatePostASAPDAGs`. +inspect `CandidateLogicalPostASAPDAGs`. --- @@ -214,7 +231,7 @@ fields expand as follows: | `TimeSelection` | `lookback` | Optional event-time duration selected before the upper bound. | | `TimeSelection` | `as_of` | Optional fixed upper-bound timestamp; `None` means planning/evaluation time. | -Frontend lowering produces one Pre-ASAP `QueryExpr` root for each normalized +Frontend lowering produces one Pre-ASAP `PreASAPNode` root for each normalized query entry. The caller must retain each root's association with its workload entry for later recurrence and lifecycle planning. @@ -303,7 +320,7 @@ a blanket reason to discard unrelated candidates. For the direct DDSketch ratio above, search retains a candidate without a proven root guarantee when domain evidence is missing; automatic `global_selection` does not choose it. See the [candidate-search reference](../../develop_docs/library-api.md#generate-and-rank-candidates) -for this backend-selection path. +for how such a candidate stays available to selection that uses deployment-supplied evidence. Additional inputs for a Planner-owned maintenance decision are listed with the [summary-maintenance-lifecycle-aware helper](#summary-maintenance-lifecycle-aware-helper). @@ -312,9 +329,11 @@ Additional inputs for a Planner-owned maintenance decision are listed with the ## Output -### `CandidatePostASAPDAGs` +### `CandidateLogicalPostASAPDAGs` -`CandidatePostASAPDAGs` is Planner's canonical output. It contains: +The logical layer outputs `CandidateLogicalPostASAPDAGs`. The current +Rust API represents this collection compactly as `CandidateLogicalPostASAPDAGs`; this is an +implementation name, not an additional design stage. It contains: * canonical workload roots; * one `TargetSubDAGCandidates` entry for each discovered target sub-DAG; @@ -322,48 +341,101 @@ Additional inputs for a Planner-owned maintenance decision are listed with the * rejected candidates and reasons; and * information needed to select compatible candidates across targets. -A `CandidatePostASAPDAGs` represents a **space of logical DAG choices**, not a single plan. -It is exposed to integrators because the backend may choose among candidates -using implementation support, measured costs, and available resources that -candidate search does not have. A summary that is cheap on one backend may be +This collection represents a **space of logical DAG choices**, not a single plan. +It is exposed to integrators because a deployment supplies prices from +implementation support, measured costs, and available resources that candidate +search does not have, and selection must consider every candidate under them. A summary that is cheap on one backend may be expensive or unsupported on another. Returning only one plan during search would discard those choices too early. Callers with suitable models can instead use the [selection workflows](#workflows) -below. A future higher-level API could hide `CandidatePostASAPDAGs` behind those decisions; -the current interface lets an integrator own them. DAG assembly connects choices -after selection and does not replace this candidate interface. - -Here, a **root** is the top-level `Rc` for a workload query. A +below, or let an optimization pass run them through +[`e2e_plan` or `optimize`](updated_interface_with_pluggable_optimization.md). +Either way, selection keeps one candidate per target and drops the rest. DAG +assembly connects choices after selection and does not replace this candidate +interface. + +Here, a **root** is the top-level node of a workload query's `PreASAPDAG` +(currently `Rc`). A **target** is any discovered sub-DAG that may be replaced, including roots. For `count(up) + 1`, the addition is a root and `count(up)` can be an inner -target. `TargetSubDAGCandidates` holds the alternatives for one such target. +target. `TargetSubDAGCandidates` holds the candidates for one such target. It represents that space compactly instead of eagerly copying every complete -DAG. `CandidatePostASAPDAGs` stores the workload's canonical roots once, creates one +DAG. The current representation stores the workload's canonical roots once, creates one `TargetSubDAGCandidates` entry for each distinct target sub-DAG, and stores -that target's replacement alternatives once inside the entry. Candidate +that target's replacement candidates once inside the entry. Candidate children refer back to canonical -targets, so common subexpressions and shared alternatives are not duplicated +targets, so common subexpressions and shared candidates are not duplicated across roots. -For example, if one target has three alternatives and its child has two, -eager enumeration could create six complete DAGs. `CandidatePostASAPDAGs` stores the three -parent alternatives, the two child alternatives, and their relationship. -Whole-plan selection chooses compatible alternatives across those targets; -`assemble_selected_dag(root)` then recursively substitutes the selected alternatives to -construct a complete Post-ASAP DAG. Sharing each target's candidate set avoids the +For example, if one target has three candidates and its child has two, +eager enumeration could create six complete DAGs. The representation stores the three +parent candidates, the two child candidates, and their relationship. +Whole-plan selection chooses compatible candidates across those targets; +`assemble_selected_dag(root)` then recursively substitutes the selected candidates to +construct a complete LogicalPostASAPDAG. Sharing each target's candidate set avoids the Cartesian-product expansion of complete DAGs and preserves shared nodes. +### Output layers + +Planning proceeds through these layers, from what to compute to how to run it. +The [Planner and deployment layering](../proposals/planner-layering.md) +proposal describes the full contract, including deployment costs and mixed +placement. + +The Rust graph APIs are `PreASAPDAG = Rc`, +`LogicalPostASAPDAG = Rc`, and `PhysicalPostASAPDAG`. Named candidate +collections carry alternatives between layers. The +[API migration guide](../../develop_docs/dag-api-migration.md) describes shared +timing assignments, physical candidate generation and explicit transport exports. + +| Layer | Form | Decides | +|---|---|---| +| 0. Frontends | `CandidatePreASAPDAGs` | Parse and lower PromQL, SQL, or MetricsQL; reject unsupported semantics such as PromQL `fill`. | +| 1. Logical Post-ASAP | `CandidateLogicalPostASAPDAGs` | What to compute: summary families, rewrites, and exact candidates. No placement. `PlanOutput` is an optional library-selected logical result; selection with deployment prices uses the candidate collection itself. | +| 2. Summary lifecycle planning | Lifecycle choices per unique summary state and maintained population → `CandidateLifecyclePostASAPDAGs` | `Ephemeral`, `Prepared`, `Shared`, or `ContinuouslyMaintained`. Each admissible assignment produces a candidate with node timing, window framework, and retention; this layer is the only source of timing. | +| 3. Physical compilation | `CandidatePhysicalPostASAPDAGs` | Compile operators and kernels once, then cut by timing into precompute and query DAGs with typed `InputContracts`. | +| 4. Selection | Planner library function | Takes the deployment's prices (including summary store cost), accuracy requirements and capabilities, and returns one optimal physical plan. | +| 5. Deployment execution | Deployment-owned | Binds, installs state, and executes the selected plan. | + +Each layer preserves all legal candidates under the supplied constraints; +none selects a winner in this pipeline. Frontend output may be a singleton +set. Candidate sets may remain compact or be enumerated lazily, and rejected +candidates retain reasons. The selection workflows below are opt-in library +helpers, not mandatory stages that discard candidates before deployment. + +Selection with a deployment's prices, such as ASAPQuery-backend's, must see the +whole candidate collection, not `PlanOutput`; otherwise candidates such as those +added in #472 never reach pricing. + +`LogicalPostASAPDAGTransport` is an explicit export/import format. Physical compilation +can read the shared logical graph directly through its indexed view. `PlanOutput` carries logical graphs in the +current `PostASAPNode` representation. Its `DagWithLifecycle` variant is the +library path doing layer 2 as well, under the +caller's cost model; with deployment-supplied prices, selection makes that +lifecycle choice over the timed candidates. +Both forms are described in +[Post-ASAP IR](../concepts/post-asap-ir.md#tree-and-exported-dag-forms). + +Placement does not belong to logical candidate generation. The grouped `Rate`→`Sum` +"maintenance versus query placement" candidate pair from #472 is a placement +choice; #485 replaces it with one candidate placed by lifecycle choice. + +Cross-query sharing is expressed by common-subexpression elimination in the +logical layer and by explicit node identity in a multi-root `LogicalPostASAPDAGTransport` at +physical compilation. The per-query `Vec` in `PlanOutput` is not the sharing +contract. + ## Workflows All paths start by lowering the workload and searching for candidates: ```text PlanningWorkload + frontend dependencies + planning models/evidence - -> frontend lowering: one QueryExpr root per normalized query entry + -> frontend lowering: one PreASAPDAG root per normalized query entry -> search_workload_with_targets - -> CandidatePostASAPDAGs + -> CandidateLogicalPostASAPDAGs ``` The integration associates each lowered root with a caller-owned `Id` and its @@ -374,23 +446,23 @@ Then choose the operation matching the caller's responsibility: | Purpose | Operation | Result | |---|---|---| -| Inspect candidates or let the backend choose | [Ranked view](#ranked-view), if ranking is useful | Per-target candidate lists and costs | -| Ask Planner to choose logical computations; backend owns summary maintenance | [Selection and DAG assembly](#selection-and-dag-assembly) | One selected Post-ASAP DAG root per query | +| Inspect candidates, e.g. to supply prices for them | [Ranked view](#ranked-view), if ranking is useful | Per-target candidate lists and costs | +| Ask Planner to choose logical computations; summary maintenance decided separately | [Selection and DAG assembly](#selection-and-dag-assembly) | One selected LogicalPostASAPDAG root per query | | Ask Planner to also decide summary maintenance versus raw recomputation | [Summary-maintenance-lifecycle-aware helper](#summary-maintenance-lifecycle-aware-helper) | One plan containing a DAG root and maintenance decisions per query | ### Ranked view -`CandidatePostASAPDAGs::cost_sorted` returns one `RankedTargetSubDAGCandidates` for each +`CandidateLogicalPostASAPDAGs::cost_sorted` returns one `RankedTargetSubDAGCandidates` for each `TargetSubDAGCandidates` entry. Conceptually, it is the same target's -alternatives in cost-model preference order where the model defines one -(otherwise discovery order), with one displayed cost per alternative. It is +candidates in cost-model preference order where the model defines one +(otherwise discovery order), with one displayed cost per candidate. It is a **view of one decision point**, not a complete DAG or a selected plan. The return type is `Vec>`; each element has this shape: ```rust struct RankedTargetSubDAGCandidates<'a> { - target: &'a Rc, + target: &'a Rc, consumer_count: usize, candidates: Vec<&'a ReplacementSubDAG>, costs: Vec, // costs[i] describes candidates[i] @@ -405,32 +477,32 @@ physical deployability. ### Selection and DAG assembly -The input is `CandidatePostASAPDAGs` and a cost model. Call -`CandidatePostASAPDAGs::global_selection(&cost_model)` once for the workload, then +The input is the candidate collection and a cost model. In the current API, call +`CandidateLogicalPostASAPDAGs::global_selection(&cost_model)` once for the workload, then `GlobalSelection::assemble_selected_dag(root)` for each wanted query root. These are two public APIs, not one combined call: N roots require one selection and N assembly calls. Each successful assembly returns one DAG root; the caller collects them for the workload, with shared nodes where applicable. -Selection chooses compatible alternatives across the workload. For example, +Selection chooses compatible candidates across the workload. For example, if two queries can share a summary, their choices must agree on the shared -computation. `assemble_selected_dag(root)` then connects the chosen alternatives +computation. `assemble_selected_dag(root)` then connects the chosen candidates into each query's DAG in memory, preserving shared nodes. Candidate search has already built candidate sub-DAGs; assembly connects the selected choices into the result for one query root. | Input → decisions → output (click a step for details) | |:---:| -| **Input:** [CandidatePostASAPDAGs](asap-aware-plan-search.md) + cost model | +| **Input:** [CandidateLogicalPostASAPDAGs](asap-aware-plan-search.md) + cost model | | ↓ | -| **Select:** [global_selection](../../develop_docs/library-api.md#what-does-global-selection-mean) chooses compatible alternatives | +| **Select:** [global_selection](../../develop_docs/library-api.md#what-does-global-selection-mean) chooses compatible candidates | | ↓ | | **Assemble:** [assemble_selected_dag(root)](../../develop_docs/library-api.md#api-definition-and-example) connects those choices for each query root | | ↓ | -| **Output:** one selected logical [Post-ASAP DAG](../concepts/post-asap-ir.md) per query root | +| **Output:** one selected logical [LogicalPostASAPDAG](../concepts/post-asap-ir.md) per query root | Each output DAG specifies the chosen operators, parameters, and accuracy -guarantees. Its root is represented by `Rc`; the +guarantees. Its root is represented by `Rc`; the [API reference](../../develop_docs/library-api.md#api-definition-and-example) describes the function signatures and return handling. @@ -443,7 +515,7 @@ summary-maintenance lifecycle costs. Use it when ASAPPlanner owns the decision to maintain summaries versus recompute raw data. It is not needed for candidate inspection or when the downstream backend owns that decision. -Starting from an existing `CandidatePostASAPDAGs`, call these two public helpers in order; +Starting from the candidate collection (currently `CandidateLogicalPostASAPDAGs`), call these two public helpers in order; there is no need to run the ordinary selection/assembly workflow first: 1. `global_selection_with_summary_maintenance_lifecycles` uses the workload @@ -451,9 +523,9 @@ there is no need to run the ordinary selection/assembly workflow first: candidates across target sub-DAGs. It returns `GlobalSelection`, not a DAG or a deployment plan. 2. For each wanted query root, `assemble_selected_dag_with_summary_maintenance_lifecycles` - takes that selection and root, constructs a Post-ASAP DAG, compares the + takes that selection and root, constructs a LogicalPostASAPDAG, compares the selected summary's maintenance cost with raw recomputation, and returns - `Result, SummaryMaintenanceLifecycleAssemblyError>`. + `Result, SummaryMaintenanceLifecycleAssemblyError>`. When a summary does not beat a known raw cost, or a required comparable cost is unavailable, the result retains the exact `KeepPreAsap` root and no summary deployments. @@ -472,13 +544,13 @@ Across the two calls, the caller supplies these parameters: | Helper parameter | Source | Required | |---|---|---:| -| `CandidatePostASAPDAGs` | Canonical ASAPPlanner output; passed to selection | Yes | -| `GlobalSelection` and one root | Selection result and a root in that `CandidatePostASAPDAGs`; passed to DAG assembly | Yes for each assembled root | +| Candidate collection (`CandidateLogicalPostASAPDAGs` in the current API) | Logical candidates passed to selection | Yes | +| `GlobalSelection` and one root | Selection result and a root in that candidate collection; passed to DAG assembly | Yes for each assembled root | | Workload binding | `QueryWorkload` plus the workload-entry indices associated with each root | Yes | | Planning time (`now_ms`) | Caller clock in Unix milliseconds | Yes | | Planning horizon | Caller policy | Conditional: required for finite totals over recurring demand | | Data arrival and update rate | `DataWorkload` evidence | Conditional: required to cost continuous maintenance | -| Lifecycle capabilities | Deployment/runtime provider | Yes for checking deployable lifecycle alternatives | +| Lifecycle capabilities | Deployment/runtime provider | Yes for checking deployable lifecycle choices | | Summary and raw cost information | Cost model and physical-evidence provider | Yes for a cost-based maintenance-versus-recompute decision | Recurrence and time selection are already fields of the bound `QueryWorkload`; @@ -486,10 +558,10 @@ they are not duplicated as separate top-level inputs. Similarly, data arrival and update rate are read from the optional `DataWorkload`. Missing required facts remain unknown rather than being treated as zero. -The per-query output, `SummaryMaintenanceLifecyclePlan`, **contains** the -Post-ASAP DAG rather than being a parallel representation. It records: +The per-query output, `LifecyclePostASAPDAG`, **contains** the +LogicalPostASAPDAG rather than being a parallel representation. It records: -* the assembled Post-ASAP DAG root (`Rc`); +* the assembled LogicalPostASAPDAG root (`Rc`); * lifecycle choices for summary state; * planning horizon and expected reads/updates; * selected window implementation and guarantees; diff --git a/docs/design_docs/architecture/metricsql-frontend.md b/docs/design_docs/architecture/metricsql-frontend.md index ddc46323..f88f6c61 100644 --- a/docs/design_docs/architecture/metricsql-frontend.md +++ b/docs/design_docs/architecture/metricsql-frontend.md @@ -10,7 +10,7 @@ rollup expressions, step-relative durations, MetricsQL binary operators, aggregate limits, or-delimited matchers, and `keep_metric_names`. The frontend walks that AST directly and emits the existing canonical -`QueryExpr`. It does not add MetricsQL fields to `QueryExpr`, SDS descriptors, +`PreASAPNode`. It does not add MetricsQL fields to `PreASAPNode`, SDS descriptors, or the physical summary DAG. ```text @@ -18,7 +18,7 @@ MetricsQL source | MetricsqlExpr (extension semantics retained) | -canonical QueryExpr +canonical PreASAPNode | existing ASAP-aware mapping and physical Summary DAG ``` @@ -33,8 +33,8 @@ existing ASAP-aware mapping and physical Summary DAG | Common rollups: rate/increase/derivatives and statistical `*_over_time` | Existing per-entity canonical intents over the lowered range. | | PromQL arithmetic, comparison, and set binary operators without modifiers | Existing canonical `BinaryOp`. | | `default_rollup(selector[range])` | Lower to `Aggregate(LastOverTime)` over the explicit `TimeRange`. | -| `default_rollup(selector)` | Reject for exact fallback because the implicit lookbehind window depends on the runtime evaluation step, which is not a property of canonical `QueryExpr`. | -| `expr keep_metric_names` | Parsed natively, then rejected for exact fallback because canonical `QueryExpr` does not carry metric-name lineage. | +| `default_rollup(selector)` | Reject for exact fallback because the implicit lookbehind window depends on the runtime evaluation step, which is not a property of canonical `PreASAPNode`. | +| `expr keep_metric_names` | Parsed natively, then rejected for exact fallback because canonical `PreASAPNode` does not carry metric-name lineage. | | `if`, `ifnot`, `default`, aggregate `limit`, or-delimited matchers, binary match modifiers | Parsed natively and rejected until the canonical executor has the exact semantics. | | `WITH` | Expanded by the native parser; the expanded expression lowers when every resulting node is supported. | diff --git a/docs/design_docs/architecture/physical-plan-integration.md b/docs/design_docs/architecture/physical-plan-integration.md index b62a6772..82bf9421 100644 --- a/docs/design_docs/architecture/physical-plan-integration.md +++ b/docs/design_docs/architecture/physical-plan-integration.md @@ -10,7 +10,7 @@ the cost model from being coupled directly to either logical IR. The integration pipeline is: ```text -pre-ASAP QueryExpr ─┐ +pre-ASAP PreASAPNode ─┐ ├─ physical lowering ─> PhysicalOperator DAG post-ASAP SummaryExpr┘ │ v @@ -30,7 +30,7 @@ Each representation is authoritative for a different concern: | Representation | Authoritative concern | |---|---| -| `QueryExpr` | Original exact query semantics: sources, predicates, relational and PromQL operations, and output shape. | +| `PreASAPNode` | Original exact query semantics: sources, predicates, relational and PromQL operations, and output shape. | | `SummaryExpr` | Logical summary semantics: selected family, grouping strategy, summary composition, and summary readout. | | `PhysicalOperator` DAG | Selected executable algorithms, their configuration, physical identity, edges, and execution multiplicity. | | `OperatorStatistics` | Workload-dependent evidence required by each selected physical operator's resource formula. | @@ -66,7 +66,7 @@ Examples include: - shared logical sub-DAGs become shared physical nodes only when they refer to the same physical identity and compatible evidence. -For this reason, aligning `OperatorStatistics` directly with `QueryExpr` would +For this reason, aligning `OperatorStatistics` directly with `PreASAPNode` would lose post-ASAP summary implementations, while aligning it directly with `SummaryExpr` would lose raw query operators and physical algorithm choices. @@ -91,7 +91,7 @@ its modeled descendants is invalid because it undercounts the candidate. ### Pre-ASAP lowering -`KeepPreAsap` recursively lowers its contained `QueryExpr`. Typical physical +`KeepPreAsap` recursively lowers its contained `PreASAPNode`. Typical physical operators include scans, filters, projections, hash aggregates, joins, ordering, bounded Top-K, limits, and PromQL-specific operators. The selected physical algorithm, rather than the logical spelling, determines the formula. @@ -108,7 +108,7 @@ Every `SummaryExpr` operation also needs explicit physical realization: | `SummarySubtract` | subtract operator supported by the selected state representation | | `SummaryDelete` | physical deletion/update operator supported by the selected representation | | `SummaryEstimate` | family- and query-specific readout operator | -| `KeepPreAsap` | recursive lowering of the contained `QueryExpr` | +| `KeepPreAsap` | recursive lowering of the contained `PreASAPNode` | | `BinaryOp` | binary evaluation preserving operand order, execution timing and any typed finite/relative-division guard | | `ValueOperation` | concrete realization of the value operation with its required execution timing and data state | | `RelationalJoin` | concrete row-join algorithm preserving join kind and predicate | @@ -120,7 +120,7 @@ statistics contract, validation rule, and resource formula for an operation, a candidate containing it is unavailable. The streaming integration can consume a complete binding through -`StreamingNodeEvidence`. That binding is keyed to exact `SummaryNode` +`StreamingNodeEvidence`. That binding is keyed to exact `PostASAPNode` identities and uses structured evidence for aggregate state, join, merge, subtract, delete, readout, and retained pre-ASAP work. It is a physical evidence boundary, not automatic physical lowering: a deployment must still @@ -128,7 +128,7 @@ select each concrete implementation and provide all edges, resource facts, multiplicities, source ownership, and stable physical identities. The planner fails closed when any reachable `SummaryExpr` node lacks that binding. -The raw/query portion of a streaming comparison remains a `PhysicalDag` using +The raw/query portion of a streaming comparison remains a `PhysicalExecution` using the canonical `PhysicalOperator` and `OperatorStatistics` pairing. Summary evidence is kept separate only where lifecycle-driven update, retention, and expiration multiplicities require facts beyond the query-DAG diff --git a/docs/design_docs/architecture/planner-runtime-contract.md b/docs/design_docs/architecture/planner-runtime-contract.md index 41a3e719..adcd9496 100644 --- a/docs/design_docs/architecture/planner-runtime-contract.md +++ b/docs/design_docs/architecture/planner-runtime-contract.md @@ -2,16 +2,16 @@ ## Purpose -ASAPPlanner produces `CandidatePostASAPDAGs`, a compact logical candidate space. Integrators -may select candidates downstream or ask Planner's helpers to select and assemble -DAGs. Summary-maintenance lifecycle decisions belong to Planner only when the +ASAPPlanner produces `CandidateLogicalPostASAPDAGs`, a compact logical candidate space. +Selection is a Planner function that uses the deployment's cost model; Planner's +helpers select and assemble DAGs. Summary-maintenance lifecycle decisions belong to Planner only when the integration uses its lifecycle-aware workflow; physical deployment and execution remain downstream. The [input/output/workflow design](input-output-workflow.md) defines this boundary. A downstream provider can report implementation alternatives and their cost and accuracy evidence for a Planner-owned comparison. The resulting -`SummaryMaintenanceLifecyclePlan` contains a Post-ASAP DAG root and maintenance +`LifecyclePostASAPDAG` contains a Post-ASAP DAG root and maintenance decisions; it is not an executable plan. Repeated provider calls do not constitute an implemented end-to-end replanning or deployment-transition protocol. @@ -38,8 +38,8 @@ exponential histogram—because those alternatives have different accuracy, CPU, memory, and I/O behavior. ASAPQuery-backend then implements the selected framework. For example, after -Planner chooses a sliding-window realization, the backend chooses the concrete -pane representation, runtime operator implementation, placement, sharding, +Planner chooses a sliding-window realization, the backend implements it with its +own pane representation, runtime operator implementation, machine placement, sharding, watermark behavior, and materialization identifiers. ASAPCollector maintains the compiled panes and summary state. @@ -51,7 +51,7 @@ backend still owns how the selected algorithms are physically realized. ## Summary-algorithm analogy The same contract applies when ASAPPlanner selects a summary algorithm. Planner -can choose KLL rather than DDSketch, while downstream chooses the concrete KLL +can choose KLL rather than DDSketch, while downstream supplies the concrete KLL implementation and runtime configuration that satisfies the selected parameter and accuracy contract. Empirical KLL error, update work, state size, and readout work observed on a particular workload can be fed back as evidence for later @@ -76,7 +76,7 @@ and rollback are not an end-to-end Planner protocol. source coverage, input/output edges, operation counts, update and bootstrap fanout, retained state, CPU, memory, I/O, and accuracy facts. 4. ASAPPlanner keeps constructible candidates with missing evidence visible - in `CandidatePostASAPDAGs` but does not certify unknown accuracy. The + in `CandidateLogicalPostASAPDAGs` but does not certify unknown accuracy. The summary-maintenance-lifecycle-aware workflow compares supported alternatives over the same workload horizon. Missing or incomparable costs do not establish that maintaining a summary beats raw recomputation; structural scores and @@ -136,7 +136,7 @@ in one cost formula. execution. - A selected realization framework is a contract, not executor code. - Physical capabilities and evidence constrain deployment choices, not every - logical candidate's presence in `CandidatePostASAPDAGs`. Known unsupported capabilities + logical candidate's presence in `CandidateLogicalPostASAPDAGs`. Known unsupported capabilities and unknown algorithms cannot become deployable alternatives. - Complete physical alternatives need identity and comparable evidence for cost-based deployment decisions. Stale evidence cannot certify or cost a diff --git a/docs/design_docs/architecture/updated_interface_with_pluggable_optimization.md b/docs/design_docs/architecture/updated_interface_with_pluggable_optimization.md index 6ddf4984..61a1ce30 100644 --- a/docs/design_docs/architecture/updated_interface_with_pluggable_optimization.md +++ b/docs/design_docs/architecture/updated_interface_with_pluggable_optimization.md @@ -7,14 +7,14 @@ What that buys: -* One call in place of six across three stages. `PlanSpace` and +* One call in place of six across three stages. `CandidateLogicalPostASAPDAGs` and `GlobalSelection` no longer appear in user code. * The root-to-entry bindings a caller used to build by hand are derived, and their ordering contract is checked rather than assumed. * A new optimization algorithm can be freely implemented as a trait implementation, rather than a rule disguised to fit a two-phase pipeline it does not share. -Unchanged: `PlanSpace`, `cost_sorted`, `global_selection`, and the interface +Unchanged: `CandidateLogicalPostASAPDAGs`, `cost_sorted`, `global_selection`, and the interface [input, output, and workflows](input-output-workflow.md) describes. ```text @@ -89,17 +89,30 @@ pub enum PlanOutput { pub struct QueryPlan { pub entry_index: usize, // index into QueryWorkload::entries() - pub dag: Rc, + pub dag: Rc, } pub struct QueryLifecyclePlan { pub entry_index: usize, - pub plan: SummaryMaintenanceLifecyclePlan, // its `root` is the DAG + pub plan: LifecyclePostASAPDAG, // its `root` is the DAG } ``` The variant follows from whether `lifecycle` was supplied in the input. +`PlanOutput` is an opt-in selection helper's result. The candidate-preserving +Planner pipeline outputs all legal candidates at each layer and does not pass +through this single-selection result. `PlanOutput` is not a replacement for +`CandidateLogicalPostASAPDAGs`: candidates the pass dropped are +not in it, so selection with deployment-supplied prices uses the full +candidate collection. Each selected +logical graph is a `LogicalPostASAPDAG`, currently represented by `Rc`. +The lifecycle layer alone supplies timing; `DagWithLifecycle` includes that +choice, while `Dag` does not. Physical compilation then produces a +`PhysicalPostASAPDAG` (currently `PhysicalPostASAPDAG`) and cuts a `PhysicalCandidate`. +These are design names, not renamed Rust APIs. See +[output layers](input-output-workflow.md#output-layers). + --- ## 3. The pluggable optimization pass diff --git a/docs/design_docs/concepts/glossary.md b/docs/design_docs/concepts/glossary.md index 6874ffce..ca158072 100644 --- a/docs/design_docs/concepts/glossary.md +++ b/docs/design_docs/concepts/glossary.md @@ -22,7 +22,7 @@ A logical plan representation that can contain ASAP primitives, alongside relati ## Candidate -One legal Post-ASAP alternative for answering an intent. Planner preserves alternatives for downstream selection rather than committing to one. +One legal Post-ASAP alternative for answering an intent. Planner preserves alternatives until its selection, which uses the deployment's cost model, rather than committing to one early. ## Summary @@ -30,4 +30,4 @@ Maintained state, exact or approximate, that can answer some query intent more e ## Physical plan -A concrete execution and deployment choice: runtime topology of data lifecycle stages, placement of computation, transmission, storage. Physical plans are owned by downstream systems. +A precompute DAG and a query DAG of physical operators, with the lifecycle, window framework and retention of every stored output. Planner compiles and selects it; the deployment binds its inputs, places, stores and executes it. diff --git a/docs/design_docs/concepts/planner-pipeline.md b/docs/design_docs/concepts/planner-pipeline.md index 5e013e48..1cccd39b 100644 --- a/docs/design_docs/concepts/planner-pipeline.md +++ b/docs/design_docs/concepts/planner-pipeline.md @@ -1,6 +1,6 @@ # Planner pipeline -ASAPPlanner accepts a planning workload and produces `CandidatePostASAPDAGs`, a compact +ASAPPlanner accepts a planning workload and produces `CandidateLogicalPostASAPDAGs`, a compact representation of candidate Post-ASAP DAGs. Ranking and selection are operations over that output, not mandatory stages of candidate search. @@ -10,7 +10,7 @@ over that output, not mandatory stages of candidate search. Pre-ASAP IR: exact, language-independent query intent | enumerate legal summary-aware alternatives v - CandidatePostASAPDAGs: candidate Post-ASAP DAGs + CandidateLogicalPostASAPDAGs: candidate Post-ASAP DAGs | +--> inspect candidates, optionally using cost_sorted +--> select and assemble logical DAGs diff --git a/docs/design_docs/concepts/post-asap-ir.md b/docs/design_docs/concepts/post-asap-ir.md index 67b5535a..4714d9c0 100644 --- a/docs/design_docs/concepts/post-asap-ir.md +++ b/docs/design_docs/concepts/post-asap-ir.md @@ -48,7 +48,7 @@ summary family supports incremental maintenance. values; the right input supplies keys. Grouped Sort followed by grouped Limit ranks and selects the joined rows. Completeness evidence belongs to pruning, not ranking. -A `SummaryNode` carries its expression, schema and optional result guarantee. +A `PostASAPNode` carries its expression, schema and optional result guarantee. State and query values have different contracts. Exact operations over approximate readouts still require composed accuracy guarantees. See the [accuracy implementation companion](../../develop_docs/end-to-end-accuracy-guarantees.md) @@ -59,11 +59,11 @@ for the corresponding correctness and realization requirements. The Pre-ASAP DAG and the Post-ASAP DAG are both logical: they describe what is computed, not which physical operators execute it. The Post-ASAP DAG has two -forms of the same content. Planning builds and shares `SummaryNode` trees. -`compile_post_asap_dag` converts a selected tree into a -[`PostAsapDag`](../../../crates/types/src/post_asap/post_asap_dag.rs) with -stable node IDs and typed edges; `PostAsapDagDocument` is its versioned wire -envelope. Physical compilation consumes `PostAsapDag` and produces a separate +forms of the same content. Planning builds and shares `PostASAPNode` trees. +`export_post_asap_dag` converts a selected tree into a +[`LogicalPostASAPDAGTransport`](../../../crates/types/src/post_asap/post_asap_dag.rs) with +stable node IDs and typed edges; `LogicalPostASAPDAGDocument` is its versioned wire +envelope. Physical compilation consumes `LogicalPostASAPDAGTransport` and produces a separate physical DAG. ## Execution phase @@ -74,8 +74,8 @@ one of these phases. Backend capability restrictions are implementation gaps, not definitions of the operator. Every post-ASAP operator payload supports both phase assignments. Phase is -stored on the `PostAsapDag` node, independently of its operator payload. -`PostAsapDag::with_execution_phases` assigns a phase to every node and updates +stored on the `LogicalPostASAPDAGTransport` node, independently of its operator payload. +`LogicalPostASAPDAGTransport::with_execution_phases` assigns a phase to every node and updates its edges. Ingestion work cannot depend on a future query result. Default semantic realization still proposes an initial layout; it does not restrict which phase an operator may use. Deployments must separately check that @@ -109,7 +109,7 @@ failure probabilities are combined, and the score guarantee remains in the membership guarantee's child provenance. An exact request does not accept this approximate output path merely because its selected identities are certified. -Deployment chooses ingestion time or query time for these operators. The +The selected lifecycle assignment decides ingestion time or query time for these operators. The semantic constructor proposes a layout; `with_execution_phases` assigns the placement. Either deployment must give each evaluation a complete rate window and an isolated summary state, or maintain an equivalent replacement diff --git a/docs/design_docs/decisions/concat-unique-keys.md b/docs/design_docs/decisions/concat-unique-keys.md index 778f6061..7fc05f76 100644 --- a/docs/design_docs/decisions/concat-unique-keys.md +++ b/docs/design_docs/decisions/concat-unique-keys.md @@ -14,7 +14,7 @@ ## Context -[`QueryExpr::Concat`](../../../crates/types/src/pre_asap/query_expr.rs) (the +[`PreASAPNode::Concat`](../../../crates/types/src/pre_asap/query_expr.rs) (the n-ary exact `UNION ALL` node, renamed from `Merge` in #226) always drops `unique_keys` on its output — `merge_drops_the_branches_unique_keys` and `merge_and_setop_agree_on_unique_keys` pin this down. Issue #228 asks whether @@ -61,7 +61,7 @@ Both current `Concat`-constructing call sites, and every consumer of site; wiring one in would mean *first* deciding to stop rejecting `GROUPING()` and surfacing `__grouping_id` as real IR — a separate, larger change outside #228's scope, not a small addition to this lowering. -- **Every other `QueryExpr::Concat { … }` construction site** in the repo is +- **Every other `PreASAPNode::Concat { … }` construction site** in the repo is a test/tooling AST match (`promql_lowering.rs`, `promql_conformance.rs`, `sql_lowering.rs`, `dag_export.rs`, `variant_coverage.rs`, netflow/synthetic test fixtures) — none of them builds a fresh `Concat` with a `Dedup` on top @@ -98,7 +98,7 @@ for why it's fine to ship unused. ### What shipped -`QueryExpr::Concat` gained an opt-in field, `discriminator_unique_key: Option>` +`PreASAPNode::Concat` gained an opt-in field, `discriminator_unique_key: Option>` (`crates/types/src/pre_asap/query_expr.rs`), plus: - `ConcatDiscriminatorKey` — a small struct with **private** `discriminator: C` / @@ -112,11 +112,11 @@ for why it's fine to ship unused. arbitrary field values in untrusted JSON. Not a reachable concern today (see "Safety argument" below for why), but stated precisely rather than overclaimed. -- `QueryExpr::concat(children)` — the ordinary constructor (`discriminator_unique_key: None`), - meant to replace the bare `QueryExpr::Concat { children }` struct literal +- `PreASAPNode::concat(children)` — the ordinary constructor (`discriminator_unique_key: None`), + meant to replace the bare `PreASAPNode::Concat { children }` struct literal everywhere in the tree so a future field addition doesn't force every call site to re-litigate this choice. -- `QueryExpr::concat_with_discriminator(children, discriminator, inner_key)` — +- `PreASAPNode::concat_with_discriminator(children, discriminator, inner_key)` — the override constructor. - `output_schema()`'s `Concat` arm: unchanged default (`unique_keys` cleared unconditionally) when `discriminator_unique_key` is `None`; when `Some`, @@ -155,7 +155,7 @@ there) and are not attempted here. Three independent things hold `discriminator_unique_key: None` as the observable behavior for every caller that doesn't ask for the override: -1. **Every real construction path defaults to `None`.** `QueryExpr::concat` +1. **Every real construction path defaults to `None`.** `PreASAPNode::concat` hardcodes it; every call site in the tree (including both real lowering call sites) uses `concat`, not `concat_with_discriminator`, so nothing in the current tree can produce `Some` at all. @@ -189,7 +189,7 @@ file for its own `by`/`without` invariant. Concretely, this means: unique within every branch. The type system enforces "you named the columns," not "the compound key is valid." Both obligations are documented on `ConcatDiscriminatorKey` itself and have the same shape of - unverified claim `QueryExpr::Dedup.cols` already carries elsewhere in this + unverified claim `PreASAPNode::Dedup.cols` already carries elsewhere in this module (nothing checks a `Dedup`'s `cols` are actually a real key of its child either). - The `no_way_to_fabricate_a_unique_key_without_naming_a_discriminator` test @@ -202,7 +202,7 @@ file for its own `by`/`without` invariant. Concretely, this means: guarantee against *other Rust code*, not against arbitrary data.** Derived deserialization builds the assertion directly from input values, bypassing `new()`. Deserialization must therefore be treated exactly like a direct -caller assertion: an external `QueryExpr` boundary must reject this field or +caller assertion: an external `PreASAPNode` boundary must reject this field or establish both obligations above before using it as uniqueness evidence. Rejecting unknown fields protects the assertion object from schema drift, but cannot prove facts about the underlying rows. diff --git a/docs/design_docs/decisions/cse-cost-model.md b/docs/design_docs/decisions/cse-cost-model.md index 6cad1730..09cdf3d1 100644 --- a/docs/design_docs/decisions/cse-cost-model.md +++ b/docs/design_docs/decisions/cse-cost-model.md @@ -49,10 +49,10 @@ carve-out — a cheap-to-recompute candidate naturally loses the comparison on its own. This decision does not need search infrastructure of its own. Issue #252's -MEMO-based search engine (`CandidatePostASAPDAGs`/`TargetSubDAGCandidates` in `replacement.rs`) already +MEMO-based search engine (`CandidateLogicalPostASAPDAGs`/`TargetSubDAGCandidates` in `replacement.rs`) already enumerates and ranks the larger, workload-wide candidate space. The choice between sharing and recomputing one already-detected CSE candidate is binary, -so `CandidatePostASAPDAGs::cost_sorted` reuses one direct +so `CandidateLogicalPostASAPDAGs::cost_sorted` reuses one direct `CostModel::cse_share_decision` comparison per group. This preserves the policy described here—compare costs rather than applying a fixed rule—inside the larger search engine. `search_workload_with`'s @@ -72,18 +72,18 @@ gate) and the cost-aware decision is applied downstream, in ## Where it hooks in -[`CandidatePostASAPDAGs::cost_sorted`](../../../crates/asap-aware-mapping/src/replacement.rs) +[`CandidateLogicalPostASAPDAGs::cost_sorted`](../../../crates/asap-aware-mapping/src/replacement.rs) is where this hooks in today. `search_workload_with` computes each shared subtree's true `consumer_count` across the whole workload up front (the same role `implement_workload_with`'s pre-pass used to play, before that function was retired along with `bind.rs` — this crate no longer commits to one physically-materialized answer at all; picking and building one final -`SummaryNode` per shared subtree is a downstream deployment's job, not this +`PostASAPNode` per shared subtree is a downstream deployment's job, not this crate's). For a `TargetSubDAGCandidates` whose candidates are a [`SharedSubtreeStrategy`](../../../crates/asap-aware-mapping/src/replacement.rs) share-vs-recompute pair, `cost_sorted`'s ranking step (`rank_group`/ `cse_preference`) asks `CostModel::cse_share_decision` once per group — using -one representative bound `SummaryNode` built just for that comparison, not +one representative bound `PostASAPNode` built just for that comparison, not cached anywhere — and sorts the pair so the preferred candidate (`Share` or `RecomputeIndependently`) comes first. Both candidates are still returned; ranking never drops one: a `CostModel` orders and parameterizes candidates; it @@ -110,7 +110,7 @@ either or both, same as `size_params` already lets a deployment override ## Scope -This decision, and `cse_share_decision`'s wiring into `CandidatePostASAPDAGs::cost_sorted` +This decision, and `cse_share_decision`'s wiring into `CandidateLogicalPostASAPDAGs::cost_sorted` (originally into `implement_workload_with`, before `bind.rs` was retired — see above), close out #223's stage 4 and #212's original "add CSE" tracking issue. Stage 3 (`dag_export::structural_hash` unification) landed separately diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 14af31a1..17aec636 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -28,17 +28,17 @@ implementation library. Deployment systems such as ASAPQuery and asap-fusion own deployment compilation and operation. The lifecycle is a planning contract associated with the logical DAG, not a separate computation IR. -The Logical Post-ASAP DAG is preceded by the Pre-ASAP DAG (`QueryExpr`), the +The Logical Post-ASAP DAG is preceded by the Pre-ASAP DAG (`PreASAPNode`), the language-independent query semantics before summary selection. Both are -logical. Planning builds Post-ASAP `SummaryNode` trees; `compile_post_asap_dag` -exports the selected tree as a `PostAsapDag`, which is the Physical Plan +logical. Planning builds Post-ASAP `PostASAPNode` trees; `export_post_asap_dag` +exports the selected tree as a `LogicalPostASAPDAGTransport`, which is the Physical Plan Compiler's input. Its per-node execution phase (ingestion or query time) is decided by the selected summary maintenance lifecycle, as the layer contract below states. ### Layer contract -1. **Logical Post-ASAP** (`CandidatePostASAPDAGs`) decides what to compute: summary +1. **Logical Post-ASAP** (`CandidateLogicalPostASAPDAGs`) decides what to compute: summary families, readouts and sharing. It does not decide placement; timing that a realization strategy writes while building a candidate is provisional. 2. **Summary maintenance lifecycle** (Planner) lists the lifecycle choices for @@ -46,7 +46,7 @@ below states. maintained population that does not feed a summary state. A chosen assignment determines every node's `ExecutionTiming`, plus window framework and retention. - `SummaryMaintenanceLifecyclePlan::execution_timed_dag` applies it: a retained + `LifecyclePostASAPDAG::export_timed_dag` applies it: a retained (non-`Ephemeral`) state and all of its inputs run at ingestion time; readouts, other consumers, and `Ephemeral` states not consumed by retained state run at query time. A population that feeds a summary state is one of @@ -54,28 +54,31 @@ below states. 3. **Physical compile** (Planner) reads timing: ingestion-time nodes form the precompute DAG and the rest form the query DAG, joined by typed outputs. It does not see raw ingestion, panes, storage or stored-state readout. -4. **Backend** chooses the lifecycle assignment with its own `CostModel`: - precompute CPU (`maintenance_cost_per_update`), sketch/summary store cost +4. **Selection** (Planner) chooses the lifecycle assignment from prices the + backend supplies through its own `CostModel`: precompute CPU + (`maintenance_cost_per_update`), sketch/summary store cost (`retention_cost_rate`), query reads (`summary_read_cost`) and per-query builds (`build_cost`, for `Ephemeral`), counting shared state once. `Ephemeral` requires the deployment to supply the state's raw input as a query-time source. -### Candidate generation and deployment selection +### Candidate generation and selection with deployment prices -Planner exposes the supported, semantically legal **physical plan candidates**. -It does not discard a computation family or materialization placement merely -because a deployment-independent cost estimate prefers another candidate. -Logical candidates are an internal search stage, not the deployment handoff. +Planner generates every supported, semantically legal **physical plan +candidate** and selects among them with the deployment's prices. Generation does +not discard a computation family or materialization placement merely because a +deployment-independent cost estimate prefers another candidate. The deployment +receives the one selected plan, not the candidates. ```text Query semantics + accuracy and lifecycle requirements ↓ Planner Supported Physical DAG candidates + typed inputs/outputs + requirements - ↓ backend -Binding feasibility + runtime statistics + resource limits + ERP - ↓ backend deployment compiler + ↓ Planner selection, with backend-supplied binding + feasibility, runtime statistics, resource limits and ERP Selected PrecomputePlan + QueryPlan + StoredOutputReferences + ↓ backend deployment compiler +Bound and installed plan ``` Planner owns operators, dependencies, sharing, and each candidate's @@ -92,7 +95,7 @@ separate unsupported compilation, deployment infeasibility, missing evidence, and a feasible candidate that loses on cost. Absence is not a cost comparison. For `sum by(job)(rate(m[1m]))`, Rate remains per series before grouped Sum. -`CandidatePostASAPDAGs` offers one such candidate, with a per-series Rate state and a grouped +`CandidateLogicalPostASAPDAGs` offers one such candidate, with a per-series Rate state and a grouped Sum state. Its lifecycle assignment places it: a retained Sum state finalizes Rate and builds Sum within a bounded precompute run; an `Ephemeral` Sum over a retained Rate state leaves the Rate readout and Sum in the query DAG. Storing a @@ -151,7 +154,7 @@ readiness; those require runtime checks. Physical location, encoding, scheduling and retention are separate execution/deployment contracts. Persisted semantic identity, its wire format and any tenant or dataset binding -belong to the deployment. Planner provides the typed `PostAsapDag` that a +belong to the deployment. Planner provides the typed `LogicalPostASAPDAGTransport` that a deployment canonicalizes; it does not define a stored-definition format. ### Running example @@ -259,22 +262,22 @@ two readouts. It does not determine when KLL states are built or retained. **Summary Maintenance Candidate Generation** enumerates legal lifecycle choices using workload demand, window/freshness requirements and supported physical -implementations. Backend selection uses runtime feasibility and cost after -physical compilation. The following example follows one candidate. +implementations. Selection uses the backend's runtime feasibility and cost +after physical compilation. The following example follows one candidate. Candidate generation and selection are separate steps. For every unique retained state, enumeration reports each lifecycle (ephemeral, prepared, shared, continuously maintained) as legal, with a Planner cost or explicitly unknown cost, or as rejected with a reason. Planner does not remove a legal alternative -because its own estimate prefers another. A deployment prices the legal -alternatives over the whole workload, counting shared state once, and binds one -lifecycle per state. Binding checks that the choice is legal and that states on +because its own estimate prefers another. A deployment supplies prices for the +legal alternatives; selection compares them over the whole workload, counting +shared state once, and binds one lifecycle per state. Binding checks that the choice is legal and that states on one maintenance path share an evaluation schedule. An alternative whose cost is -unknown can be bound only when the deployment's cost model is authoritative for +unknown can be bound only when the deployment-supplied cost model is authoritative for complete-candidate cost; unknown cost is never treated as zero. It then yields the same lifecycle guarantee and window framework the physical compiler consumes when -Planner selects. Planner's own cheapest-alternative selection remains available -for callers without deployment pricing. The window framework is decided for the +Planner selects with its default costs. Planner's default-cost selection remains +available for callers that supply no prices. The window framework is decided for the complete combination, not for one alternative in isolation. A maintained population (for example, the current series of `topk by(job)(1, m)`) @@ -319,7 +322,7 @@ The **Physical Plan Compiler** consumes both computation semantics and maintenan requirements: ```text -Logical Post-ASAP DAG (PostAsapDag) +Logical Post-ASAP DAG (LogicalPostASAPDAGTransport) + Summary Maintenance Lifecycle + physical capabilities ↓ @@ -433,16 +436,15 @@ frontiers and cost evidence, including updates, retention, recurrence and sharin The lifecycle layer decides timing; physical compilation reads it. Lowering a node does not depend on the frontier, so each query DAG is lowered once and different lifecycle assignments are different cuts of that lowering. -`compile(dag, inputs, roots)` yields the complete `CompiledPhysicalDag`. -`frontier_from_timing(&timed_dag)` reads an assignment's timed DAG (from -`execution_timed_dag`) and returns its frontier: ingestion-time nodes read by +`compile(dag, inputs, roots)` yields the complete `PhysicalPostASAPDAG`. +`frontier_from_timing` reads an assignment's timing (a lifecycle assignment's +view, or a DAG from `export_timed_dag`) and returns its frontier: ingestion-time nodes read by query-time nodes, or an ingestion-time root; a query-time node feeding an ingestion-time node is rejected. `cut_candidate(&compiled, &frontier)` then partitions the lowered operators: the frontier's ancestors form the precompute DAG and the rest form the query DAG. Helper operators are numbered by their Planner node (`u64::MAX - (node_id << 16) - index`), so a cut is byte-identical to -`compile_candidate` for that frontier. One exception: an ingestion-time -`Binary` lowers differently, so its timing must match at compile time. +`compile_candidate` for that frontier. `compile_candidate(s)` and `enumerate_frontiers` wrap the same path. Temporal pane candidates remain a separate lowering. @@ -572,7 +574,6 @@ operator/runtime fixtures: | `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled precompute/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | | `summary_maintenance_lifecycle_e2e::chosen_lifecycle_timing_decides_precompute_contents` | PromQL workload → enumerated lifecycles → explicit choice → timed DAG → compiled candidate; ContinuouslyMaintained stores the state in precompute, Ephemeral leaves precompute empty and reads the raw source at query time; both return the same p99 | | `summary_maintenance_lifecycle_e2e::lifecycle_timing_cuts_one_compilation` | KLL quantile and grouped Rate→Sum: one compilation cut by the ContinuouslyMaintained and Ephemeral timed DAGs equals `compile_candidate` for each; the frontier is the retained state or empty | - | `summary_maintenance_lifecycle_e2e::chosen_population_lifecycle_decides_precompute_contents` | PromQL `topk by(job)` over a maintained population → explicit choice → timed DAG → compiled candidate; ContinuouslyMaintained stores the population in precompute, Ephemeral rebuilds it from raw samples at query time; both rank alike | | `summary_maintenance_lifecycle_e2e::planner_lifecycle_selection_reproduces_strategy_timing` | For PromQL summary fixtures, the timed DAG from Planner's retained selection equals the DAG realization strategies produce | | `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute precompute DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | diff --git a/docs/design_docs/proposals/README.md b/docs/design_docs/proposals/README.md index 2cebb3b2..731b0753 100644 --- a/docs/design_docs/proposals/README.md +++ b/docs/design_docs/proposals/README.md @@ -10,3 +10,5 @@ extensions. A design document is not a promise of downstream runtime support. - [ASAP-aware mapping proposals](asap-aware-mapping/README.md) - [Operator sharing](operator-sharing.md) - [Decoupling operators from scalar expressions](decoupling_op_and_expr.md) +- [Planner and deployment layering](planner-layering.md) +- [DAG API alignment](dag-api-alignment.md) diff --git a/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md b/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md index 6633247e..beea37b8 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md +++ b/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md @@ -39,7 +39,7 @@ an unavailable estimate, never an assumed zero or a structural-cost fallback. The implementation keeps five layers distinct: - physical query lowering (`query_physical_lowering`), which recursively maps - supported resolved `QueryExpr` operators to the physical representation; + supported resolved `PreASAPNode` operators to the physical representation; - physical evidence (`physical_operator_statistics`), which pairs every physical operator with the statistics required by its formula; - analytical estimation (`analytical_cost`), which composes any evidenced DAG @@ -279,7 +279,7 @@ resource formula. Neither logical IR is the statistics schema: ```text -pre-ASAP QueryExpr ─┐ +pre-ASAP PreASAPNode ─┐ ├─ physical lowering ─> PhysicalDagNode/PhysicalOperator post-ASAP SummaryExpr┘ │ v @@ -590,7 +590,7 @@ deduplicated by physical identity. ### Query-DAG lowering and statistics contract -`lower_query_physical_dag` recursively lowers a resolved `Rc` and +`lower_query_physical_dag` recursively lowers a resolved `Rc` and returns an `EvidenceBackedPhysicalDag` containing both its nodes and root ID. It consumes the existing query and physical-operator enums; it does not introduce a parallel logical operator vocabulary. For every occurrence, the lowerer sends a @@ -634,7 +634,7 @@ The lowering validates every physical edge before costing: The supported mappings are: -| Existing `QueryExpr` shape | Physical DAG | +| Existing `PreASAPNode` shape | Physical DAG | |---|---| | Scan without predicates | Scan | | Scan with pushed predicates | Scan → Filter | @@ -668,7 +668,7 @@ also uses the bound left and right output schemas to prove that every equality compares one column from each side; same-side or out-of-range `ColumnId`s fail closed. -An `Rc` address is not physical identity. Every logical occurrence +An `Rc` address is not physical identity. Every logical occurrence is independent unless the provider returns the same non-empty `physical_id`. Repeated IDs deduplicate only when operator, children, coverage, statistics, and buffer evidence are identical; conflicting reuse fails closed. This @@ -803,8 +803,8 @@ horizon. This deliberately avoids reconstructing raw work with a special-case `input_rows × cpu_per_row` formula that would omit joins, windows, sorts, or other operators. -Flat single-summary evidence is bound to the exact `SummaryNode` and raw -`QueryExpr` identities for which it was produced. It cannot be reused for a +Flat single-summary evidence is bound to the exact `PostASAPNode` and raw +`PreASAPNode` identities for which it was produced. It cannot be reused for a structurally similar node or for multiple summary states. A complete multi-summary `SummaryExpr` DAG requires per-node physical evidence and physical-identity deduplication. @@ -1055,9 +1055,9 @@ may be fitted from measurements or encode a deployment resource policy. The calibration version is exported so results from different policies are not treated as directly comparable. -Changing calibration may change the selected plan. A memory-constrained -deployment and an I/O-constrained deployment need not choose the same legal -candidate. +Changing calibration may change the selected plan. Selection under a +memory-constrained deployment's costs and under an I/O-constrained deployment's +costs need not return the same legal candidate. ## Candidate selection and provenance @@ -1072,7 +1072,7 @@ The intended end-to-end selection pipeline is: The query lowerer and physical estimator cover the supported raw-query shapes listed above. `PhysicalPlanCostModel` executes this pipeline for every -candidate supplied to `CandidatePostASAPDAGs::global_selection`. Logical rewrites are +candidate supplied to `CandidateLogicalPostASAPDAGs::global_selection`. Logical rewrites are lowered recursively. Summary candidates participate only after the deployment has bound their complete `SummaryExpr` DAG; there is no optimistic generic summary fallback. The streaming adapter connects raw recomputation and diff --git a/docs/design_docs/proposals/asap-aware-mapping/ddsketch-quantile-ratios.md b/docs/design_docs/proposals/asap-aware-mapping/ddsketch-quantile-ratios.md index 620498c9..7bce25c1 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/ddsketch-quantile-ratios.md +++ b/docs/design_docs/proposals/asap-aware-mapping/ddsketch-quantile-ratios.md @@ -8,7 +8,7 @@ The formula alone does not establish its preconditions. `AccuracyEvidenceProvide Certification requires each domain to lie wholly within the pinned mapping's positive or negative indexable range (or be zero alone). Same-sign interpolation preserves the relative bound. Mixed-sign interpolation can cancel: for example, `[-1, 1.001]` can have estimated median zero despite a nonzero exact median. Nonzero values below the mapping minimum are counted as zero and also do not carry the relative bound. -The denominator range must exclude zero. True and perturbed quotient ranges must stay finite and outside Float64's subnormal range, with an exact zero numerator allowed. Invalid quantile parameters and missing/invalid proofs do not receive a ratio certificate. Missing proof does not suppress candidate generation: the candidate has no root guarantee and remains visible even in `search_workload_with_targets`. Downstream selection must not treat it as certified. Invalid supplied domains are rejected. These checks are conservative: an actual window may be safe even when its declared bounds cannot prove it. +The denominator range must exclude zero. True and perturbed quotient ranges must stay finite and outside Float64's subnormal range, with an exact zero numerator allowed. Invalid quantile parameters and missing/invalid proofs do not receive a ratio certificate. Missing proof does not suppress candidate generation: the candidate has no root guarantee and remains visible even in `search_workload_with_targets`. Selection must not treat it as certified. Invalid supplied domains are rejected. These checks are conservative: an actual window may be safe even when its declared bounds cannot prove it. The final guarantee records both input ranges and their contract identifiers. The integration layer must only provide contracts it enforces for the plan's lifetime. This change adds no runtime guard, fallback executor, or automatic proof inference, and does not change standalone DDSketch readout certification outside this ratio rule. @@ -22,6 +22,6 @@ rule or remain exact. Callers that require a certified end-to-end accuracy target must use evidence or select another candidate. Planner's automatic whole-plan selection skips -uncertified ratios while retaining them in `CandidatePostASAPDAGs`; a backend can inspect +uncertified ratios while retaining them in `CandidateLogicalPostASAPDAGs`; a backend can inspect the candidate and make its own evidence-based selection. Runtime or statically enforced domain contracts remain future work driven by observed v1 correctness needs. diff --git a/docs/design_docs/proposals/asap-aware-mapping/workload-demand-and-summary-lifecycle.md b/docs/design_docs/proposals/asap-aware-mapping/workload-demand-and-summary-lifecycle.md index 64c3518e..a7596917 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/workload-demand-and-summary-lifecycle.md +++ b/docs/design_docs/proposals/asap-aware-mapping/workload-demand-and-summary-lifecycle.md @@ -66,8 +66,8 @@ For the broader lifecycle design, four categories of information matter distribution; 4. existing summaries and the lifecycle actions available to the deployment. -Candidate search outputs `CandidatePostASAPDAGs`. The implemented lifecycle-aware workflow -then returns a `SummaryMaintenanceLifecyclePlan` per query root, containing the +Candidate search outputs `CandidateLogicalPostASAPDAGs`. The implemented lifecycle-aware workflow +then returns a `LifecyclePostASAPDAG` per query root, containing the Post-ASAP DAG and maintenance decisions. It can choose exact raw recomputation when summary maintenance does not beat raw cost or comparable costs are missing. A state deployment states whether a summary is ephemeral, prepared, shared for a @@ -142,7 +142,7 @@ normalize query and data workloads report, or data at rest from continuous ingestion as an explicit mode. - **What is new, and why will it succeed?** Orthogonal workload axes and an explicit state lifecycle let the existing recurrence formulas compare the - same summary under different deployment choices without changing query + same summary under different lifecycle choices without changing query semantics. - **Who cares?** Users need predictable latency and cost; operators need to know what state will exist and for how long; planner developers need demand @@ -417,7 +417,7 @@ cost. It is not a query correctness requirement. ### Separate operator state, schedule, and output -The physical design must not use `SummaryAgg` as shorthand for incremental +Summary lifecycle planning must not use `SummaryAgg` as shorthand for incremental maintenance. ```rust diff --git a/docs/design_docs/proposals/asapquery-rule-coverage.md b/docs/design_docs/proposals/asapquery-rule-coverage.md index dd6b9a92..3cc3e439 100644 --- a/docs/design_docs/proposals/asapquery-rule-coverage.md +++ b/docs/design_docs/proposals/asapquery-rule-coverage.md @@ -5,7 +5,7 @@ This document compares ASAPPlanner's rule system with the Rust planner in ASAPQuery at upstream commit `2586400b3b0436a5414c901ebce07065d20b5223`. The comparison is by semantic capability, not source-file or function-name -parity: ASAPPlanner operates on a front-end-independent `QueryExpr` DAG, while +parity: ASAPPlanner operates on a front-end-independent `PreASAPNode` DAG, while ASAPQuery recognizes a smaller set of PromQL expression shapes. The audit covers ASAPQuery's `planner/patterns.rs`, `planner/window.rs`, diff --git a/docs/design_docs/proposals/dag-api-alignment.md b/docs/design_docs/proposals/dag-api-alignment.md new file mode 100644 index 00000000..3f526ef3 --- /dev/null +++ b/docs/design_docs/proposals/dag-api-alignment.md @@ -0,0 +1,146 @@ +# DAG API alignment + +Status: collection API unification implemented and validated on #480. +Audience: Planner developers and API integrators. + +## Objective + +Align the Rust API with the DAG names in +[Planner and deployment layering](planner-layering.md#dags-and-what-each-encodes), +without duplicating graph implementations or eagerly copying every candidate. + +The public pipeline is: + +```text +CandidatePreASAPDAGs + -> CandidateLogicalPostASAPDAGs + -> CandidateLifecyclePostASAPDAGs + -> CandidatePhysicalPostASAPDAGs + -> selection with deployment-supplied prices + -> deployment execution of the selected plan +``` + +Each generation stage retains supported legal alternatives. A workload entry +and an alternative for that entry are distinct identities. Timing comes from +lifecycle assignments; generation does not implicitly select a winner. + +## Baseline + +The implementation builds on the physical compilation and lifecycle APIs of the +stack under #508; main does not contain them yet. + +## Changes + +1. **Name the existing representations.** Rename `QueryExpr` to `PreASAPNode` + and `SummaryNode` to `PostASAPNode`. Define `PreASAPDAG` and `LogicalPostASAPDAG` as + root-reference aliases, retaining shared subgraphs and the pre-ASAP column + binding parameter. Rename `CompiledPhysicalDag` to `PhysicalPostASAPDAG`. Update + callers, exports, examples and tests together; avoid compatibility aliases + that leave two public names for the same role. +2. **Name frontend candidate outputs.** Introduce `CandidatePreASAPDAGs` using + the existing collection machinery. Preserve workload entry identity and + distinguish alternative lowering results from independent workload roots. + Deterministic lowering produces one alternative per entry. Do not add an + independent frontend candidate search implementation. +3. **Use one authoritative logical graph.** Keep the shared `PostASAPNode` graph + as `LogicalPostASAPDAG`. Move physical compilation onto this graph and reuse the + existing node-identity mapping and validation. Keep a flat node/edge form + only as an explicitly named transport document when serialization requires + it; derive it from the authoritative graph. Do not maintain a second rewrite + or computation implementation in the transport format. +4. **Attach lifecycle assignments without duplicating logical graphs.** Keep + the existing compact `CandidateLogicalPostASAPDAGs` search representation. Reuse + lifecycle enumeration and validation to expose candidate DAG references + with timing, window and retention assignments. Enumerate combinations lazily + or under an explicit expansion budget; budget exhaustion must not appear as + a complete candidate collection. Do not create a second timed graph IR. +5. **Share physical compilation across timing cuts.** `CandidatePhysicalPostASAPDAGs` + owns shared `PhysicalPostASAPDAG` realizations and lightweight candidate entries + identifying their timing cuts, contracts and lifecycle metadata. Reuse the + existing cut and validation implementation. Materialize precompute/query + execution graphs on demand, including after selection. Preserve + separate compilations when timing changes operator lowering. Keep failures + attributable to their candidates. +6. **Update the public boundary and documentation.** Route collection APIs + through the existing lowering, search, lifecycle and compilation algorithms. + Preserve explicit opt-in selection helpers. Update the DAG naming table to + describe implemented representations, and document breaking API migration. + +## Acceptance and validation + +- Public names correspond to individual nodes, DAG roots and candidate + collections without ambiguous aliases or redundant computation algorithms. +- Frontend entry identity survives lowering; independent queries are not + presented as mutually exclusive alternatives. +- Shared logical subgraphs retain identity through enumeration and compilation. +- Lifecycle alternatives reuse the logical graph and assign all required node + timing, window and retention information. Illegal timing edges are rejected. +- Equivalent timing cuts share one compilation; a realization is never reused + for timing that lowers differently. +- Candidate generation preserves valid alternatives and visible rejection + reasons; candidate inspection never silently invokes winner selection. +- Selected physical execution graphs preserve their typed producer/reader + contracts, results and serialization validation. +- Add focused regression tests for graph identity, timing and candidate sharing; + run formatting, workspace clippy and the workspace test suite. Review docs, + migration examples and any affected downstream call sites. + +Implement in reviewable commits: naming and frontend collection; logical graph +and timing consolidation; physical candidate sharing; documentation and final +integration validation. Record actual completion and any remaining limitations +rather than treating a rename as completion of the structural work. + +## Implementation record + +- Named the existing node types and shared root aliases; added frontend + `lower_pre_asap_dag_candidates` and the ID-preserving `CandidatePreASAPDAGs` collection. +- Added `LogicalPostASAPDAGIndex`, retaining shared node references and projecting + node and edge records once. Physical compilation takes a borrowed + `LogicalPostASAPDAGView` of a transport document, an index, or a lifecycle + assignment; an assignment's view overlays its timing on the index's records + instead of copying them. Transport import and direct compilation share + validation and operator lowering; no second rewrite implementation was introduced. +- Added the lazy, budget-checked timed collection `CandidateLifecyclePostASAPDAGs`. + Unpriced legal choices retain unknown cost, and rejected combinations retain + their choices and errors. `execution_assignment` attaches timing to the shared + index. `SummaryMaintenanceLifecyclePlan` is merged into `LifecyclePostASAPDAG`: + one DAG root with its per-state lifecycle, window framework and retention, from + which the timing is derived; unresolved window evidence remains explicit and must be supplied before + deployment installation. +- Added `compile_physical_dag_candidates` and `CandidatePhysicalPostASAPDAGs`. Compatible + assignments share `Arc`; timing that changes lowering produces a + separate compilation. Cuts materialize on demand through the existing implementation. +- Kept explicit eager cut and winner-selection helpers for callers requesting + them. These helpers are not the candidate-preserving generation pipeline. +- Added regression coverage for workload entry IDs, shared logical identity, + assignment budgets and unknown costs, physical compilation sharing, agreement + with independent compilation of each assignment's transport, and rejected timing. + +The Rust API migration is documented in +[dag-api-migration.md](../../develop_docs/dag-api-migration.md). Downstream +consumers must adopt the breaking names when they update their Planner dependency. + +## Collection boundary completion + +The logical collection `CandidateLogicalPostASAPDAGs` and the timed collection +`CandidateLifecyclePostASAPDAGs<'a, Id>` are separate plain types. The timed +collection encapsulates the existing lifecycle enumerator: `with_timing_for_root` +handles logical realization, index sharing, assignment budgets and lazy +generation, and single already-assembled graphs use `from_post_asap_dag`. It +also answers `lifecycle_guarantee`, so a deployment can supply a price for an +alternative before selection binds it. A state without lifecycle alternatives yields a diagnostic +entry rather than disappearing. + +`compile_physical_dag_candidates` consumes the timed collection's entries and +returns `CandidatePhysicalPostASAPDAGs`, which owns shared graphs and cut +descriptors. The metadata type `M` and the timing error type `E` belong to the +caller, so the physical crate does not depend on lifecycle planning; a timing +failure stays typed as `PhysicalCandidateError::Timing`, separate from +`PhysicalCandidateError::Compile`. Both transitions retain IDs, lifecycle +metadata and errors. Reuse also checks contracts and requested roots, +preventing one candidate's input boundary from contaminating another. + +Existing explicit lifecycle selection and cut APIs delegate to the same +implementation. No second lifecycle enumeration or operator compiler was added. +Binding runtime sources is an execution step: `PhysicalPostASAPDAG::instantiate` +returns a `PhysicalExecution` handle for one run, not another DAG. diff --git a/docs/design_docs/proposals/decoupling_op_and_expr.md b/docs/design_docs/proposals/decoupling_op_and_expr.md index 2e9a9342..e708b059 100644 --- a/docs/design_docs/proposals/decoupling_op_and_expr.md +++ b/docs/design_docs/proposals/decoupling_op_and_expr.md @@ -1,17 +1,17 @@ # Decoupling Operators From Scalar Expressions > Status: proposed, not implemented. Companion to [Operator sharing](operator-sharing.md) -> (same PR): this document splits `QueryExpr`; that one builds the shared operator +> (same PR): this document splits `PreASAPNode`; that one builds the shared operator > language on the result. Code is referenced by file and function; counts are > approximate, measured on `main` at `8acb472`. -**The idea.** `QueryExpr` holds two different kinds of node in one enum. This proposal +**The idea.** `PreASAPNode` holds two different kinds of node in one enum. This proposal splits it into `NonASAPOp` (operators) and `ScalarExpr` (scalar expressions). ``` Today Proposed -Filter { pred: Rc, Filter { pred: Predicate(Rc), - child: Rc } child: Rc } +Filter { pred: Rc, Filter { pred: Predicate(Rc), + child: Rc } child: Rc } ``` ## 1. Problem @@ -31,10 +31,10 @@ Project ← operator: outputs a table - A **scalar expression** has no table of its own. `Column(4)` means "column 4 of the input of the operator I sit in"; outside that operator it means nothing. -Today both are `QueryExpr` variants, told apart only by field position. The code already +Today both are `PreASAPNode` variants, told apart only by field position. The code already separates them, but only by convention: -- `QueryExpr::output_schema` returns `ScalarHasNoRowSchema` for all 13 scalar variants +- `PreASAPNode::output_schema` returns `ScalarHasNoRowSchema` for all 13 scalar variants (`query_expr.rs`), so `Filter { child: Literal(2) }` compiles and fails at run time. - `pre_asap/cse.rs` never descends into a scalar, and repeats a "scalar: nothing to do" arm in each of its three traversals; `canonicalize` likewise never rewrites one. @@ -66,7 +66,7 @@ NonASAPOp ``` - **Naming**: `NonASAPOp` is named for [Operator sharing](operator-sharing.md), where it - becomes the non-ASAP category of `Operator`. `QueryExpr` goes away. + becomes the non-ASAP category of `Operator`. `PreASAPNode` goes away. - **Scalar fields**: `Filter.pred`, `Join.pred`, `Aggregate.having` (`Predicate`); `Project.cols` (`ProjectItem`); `Sort` / `SQLWindowFunc` sort keys (`SortKey`); `SQLWindowFunc.args`; `PromqlRelabel.value`. @@ -86,7 +86,7 @@ NonASAPOp | `resolve`, `column_resolution.rs` | already separate: `resolve` walks operators and calls `resolve_expr` for scalars. Each scalar resolves against one schema its operator picks (usually the child's output; the `Aggregate`'s output for `HAVING`, left + right for a `Join` predicate, the `Scan`'s own schema for `Scan` predicates). The split only changes their signatures: `resolve` takes `NonASAPOp`, `resolve_expr` takes `ScalarExpr`. Leaf schemas are still inferred from scalar column references across the whole tree | | `canonicalize`, `pre_asap/cse.rs` | the "scalar: nothing to do" arms go; scalars are hashed as plain data | | `scalar_signature.rs`, `infer_expr_type` | take `ScalarExpr` | -| `QueryExpr::output_schema` | becomes `NonASAPOp::output_schema`; the scalar arms and `ScalarHasNoRowSchema` go | +| `PreASAPNode::output_schema` | becomes `NonASAPOp::output_schema`; the scalar arms and `ScalarHasNoRowSchema` go | ## 4. Implementation and tests @@ -94,14 +94,14 @@ This is stage 1 of the joint plan ([Operator sharing §8](operator-sharing.md#8- children stay `Rc`; operator sharing widens them to `Rc` in its stage 2. -**No wire change.** The `fallback` payload serializes a `QueryExpr`, externally tagged. +**No wire change.** The `fallback` payload serializes a `PreASAPNode`, externally tagged. Variant names are kept, so a tree serializes the same; `ScalarBridge` keeps the name `PromqlScalarBridge` with `#[serde(rename)]`. - Existing tests pass unchanged apart from construction syntax. - Tests that place a scalar in operator position no longer compile and are rewritten or deleted: the `CurrentTimestamp` unit test, the `ScalarHasNoRowSchema` tests, and the - `post_asap_dag.rs` tests using `QueryExpr::Literal` as a `fallback` expression. + `post_asap_dag.rs` tests using `PreASAPNode::Literal` as a `fallback` expression. ## 5. Limits diff --git a/docs/design_docs/proposals/operator-sharing.md b/docs/design_docs/proposals/operator-sharing.md index 99f5d5e2..55145f41 100644 --- a/docs/design_docs/proposals/operator-sharing.md +++ b/docs/design_docs/proposals/operator-sharing.md @@ -2,7 +2,7 @@ > - Status: proposed, not implemented. > - Problem statement: [#468](https://github.com/ProjectASAP/ASAPPlanner/issues/468). -> - Builds on [Decoupling operators from scalar expressions](decoupling_op_and_expr.md) (same PR), which splits `QueryExpr` into `NonASAPOp` and `ScalarExpr`. +> - Builds on [Decoupling operators from scalar expressions](decoupling_op_and_expr.md) (same PR), which splits `PreASAPNode` into `NonASAPOp` and `ScalarExpr`. **The idea.** Today a post-ASAP plan is glued together from two sets of operator types. This proposal keeps one operator language and makes summary operators extra node kinds in it: any relational operator can sit above a summary, and a summary can read any relational subtree. @@ -41,7 +41,7 @@ We define the `Operator` type structure based on the breadth of its attributes. ```rust pub enum Operator { - NonASAP(NonASAPOp), // today's relational and timeseries operators in `QueryExpr` (§1.2) + NonASAP(NonASAPOp), // today's relational and timeseries operators in `PreASAPNode` (§1.2) ASAP(ASAPOp), // summary operators (§1.3) } @@ -106,7 +106,7 @@ Operator ### 1.2 `NonASAPOp` `NonASAPOp` is the non-ASAP category of `Operator`. -It comes from splitting `QueryExpr` into "operator" and "scalar expression" parts ([decoupling doc](decoupling_op_and_expr.md#2-types)). +It comes from splitting `PreASAPNode` into "operator" and "scalar expression" parts ([decoupling doc](decoupling_op_and_expr.md#2-types)). ```rust pub enum NonASAPOp { @@ -163,7 +163,7 @@ Following table shows how some legacy types get expressed in the new framework. | `ValueOperation::{Project, Filter, Sort, Limit}` | `NonASAPOp::{Project, Filter, Sort, Limit}` | | `SummaryExpr::{BinaryOp, RelationalJoin}` | `NonASAPOp::{BinaryOp, Join}` | | `ValueOperation::Exact(Aggregate)`, `ExactOperation` | `NonASAPOp::Aggregate` | -| `SummaryNode` | `Operator` itself: `schema` is computed, `timing` / `guarantee` are slots on every variant (§2) | +| `PostASAPNode` | `Operator` itself: `schema` is computed, `timing` / `guarantee` are slots on every variant (§2) | ### 1.4 Child field @@ -178,7 +178,7 @@ Filter { pred: Predicate(Rc), Filter { pred: Predicate(Rc` today: branches are stored by value and have no `Rc` identity, so the planner (§4), which identifies targets by pointer, +`Concat.children` is `Vec` today: branches are stored by value and have no `Rc` identity, so the planner (§4), which identifies targets by pointer, cannot replace a branch — e.g. the branches of SQL `ROLLUP` or PromQL `histogram_quantiles`. It becomes `Vec>` (§8 stage 0). ## 2. Schema, Guarantee, and Timing @@ -187,8 +187,8 @@ This section discusses three key per-node attributes, `schema`, `guarantee`, and | Field | Meaning | Today | After | |---|---|---|---| -| `schema` | output columns and their types | pre-ASAP: computed by `QueryExpr::output_schema()`
post-ASAP: a `SummarySchema` stored on every `SummaryNode` | can be obtained by `output_schema()` | -| `guarantee` | accuracy bound | pre-ASAP: none
post-ASAP: stored on every `SummaryNode` | can be obtained by `guarantee()`
binding stores only each operator's own error
complete error bound need to be derived by `derive_guarantees()` | +| `schema` | output columns and their types | pre-ASAP: computed by `PreASAPNode::output_schema()`
post-ASAP: a `SummarySchema` stored on every `PostASAPNode` | can be obtained by `output_schema()` | +| `guarantee` | accuracy bound | pre-ASAP: none
post-ASAP: stored on every `PostASAPNode` | can be obtained by `guarantee()`
binding stores only each operator's own error
complete error bound need to be derived by `derive_guarantees()` | | `timing` | execution time | pre-ASAP: none
post-ASAP, stored: a field on `BinaryOp` / `ValueOperation` / `SummaryMerge`
post-ASAP, not stored: `KeepPreAsap` from the consuming edge, `SummaryAgg` from the child. | can be obtained by `timing()`
set by binding (`SummaryAgg`) or the planner (`FinalizeExactAccumulator`)
timing of the rest of operators need to be derived by `derive_timings()` | ### 2.1 Schema: fused into one type @@ -224,7 +224,7 @@ impl Field { | Node | Today | After | |---|---|---| -| `NonASAPOp` | post-ASAP `KeepPreAsap`: `QueryExpr` schema lifted to `SummarySchema` and stored
post-ASAP `ValueOperation` / `BinaryOp` / `RelationalJoin` copies: stored at construction | using the same logic as `QueryExpr::output_schema()` | +| `NonASAPOp` | post-ASAP `KeepPreAsap`: `PreASAPNode` schema lifted to `SummarySchema` and stored
post-ASAP `ValueOperation` / `BinaryOp` / `RelationalJoin` copies: stored at construction | using the same logic as `PreASAPNode::output_schema()` | | `SummaryAgg` | the replaced `Aggregate`'s output with the measure column retyped to `family` | grouping columns + one `ASAPType(family)` column | | `SummaryEstimate` | the replaced operator's output schema | the child's grouping columns + the value columns of the `SketchQuery` | | `FinalizeExactAccumulator` | the logical operator's output, lifted | the child's schema, `ASAPType(ExactAggregate ..)` columns changed into `DataType(..)` | @@ -251,7 +251,7 @@ Per node kind: |---|---|---| | `SummaryEstimate` | stored at binding: the sketch's own error composed with the child's (`compose_guarantee`) | **derived**: `local_guarantee` composed with the child's. `local_guarantee` is set at binding: the sketch's error over an exact input, `None` when the model has no error model for the family | | `SummaryAgg` | stored: ExactAggregate family composed with the child's; sketch families `None` | **derived**: ExactAggregate family: exact, composed with the child's under `exact_rule`, except `ExactKind::Count`, exact whatever the child (as today); sketch families `Set(None)`, state has no guarantee | -| `NonASAPOp` | pre-ASAP `QueryExpr`: none
post-ASAP `KeepPreAsap`: exact
post-ASAP `ValueOperation` / `BinaryOp` / `RelationalJoin` copies: composed at construction | **derived**: composed from the children; exact if no `ASAP` descendant | +| `NonASAPOp` | pre-ASAP `PreASAPNode`: none
post-ASAP `KeepPreAsap`: exact
post-ASAP `ValueOperation` / `BinaryOp` / `RelationalJoin` copies: composed at construction | **derived**: composed from the children; exact if no `ASAP` descendant | | `FinalizeExactAccumulator` | copies the child's | **derived**: the child's | | `MaintainPopulation` / `ReadPopulation` | stored: exact | **derived**: exact | | unused variants | `None`: state has no guarantee of its own | unimplemented (§1.3) | @@ -283,7 +283,7 @@ Per node kind: | Node | Today | After | |---|---|---| -| `NonASAPOp` | pre-ASAP `QueryExpr`: none
post-ASAP `KeepPreAsap`: from the consuming edge
post-ASAP `ValueOperation` / `BinaryOp` copies: a stored field | **derived** from the consuming edge (§5) | +| `NonASAPOp` | pre-ASAP `PreASAPNode`: none
post-ASAP `KeepPreAsap`: from the consuming edge
post-ASAP `ValueOperation` / `BinaryOp` copies: a stored field | **derived** from the consuming edge (§5) | | `SummaryAgg` | from the child; ingestion time under `KeepPreAsap` | **set** by binding, as today's fallback: `IngestionTime`, or `QueryTime` over a query-time child | | `FinalizeExactAccumulator` | a stored field, set by the planner | **set** by the planner: the same position allows either time | | `SummaryEstimate` | query time, fixed by the kind | **derived** from the kind: query time | @@ -298,7 +298,7 @@ DAG and runs only after assembly. Today: ``` -search / binding each SummaryNode's guarantee is composed when the node is built; +search / binding each PostASAPNode's guarantee is composed when the node is built; BinaryOp / ValueOperation store their timing selection reads each candidate's stored guarantee against its target assembly assemble_residual builds kept nodes and composes their guarantee; @@ -348,7 +348,7 @@ input (frontends after `resolve`, deserialized plans, test IR) with simpler type ```rust pub enum Replacement { - Subtree(Rc), // formerly Summary(Rc) and Rewrite(Rc) + Subtree(Rc), // formerly Summary(Rc) and Rewrite(Rc) ExactComposition { .. }, // its plan becomes Rc } ``` @@ -361,7 +361,7 @@ returns the child's original subtree, and assembly keeps it as is. **Accuracy check during search**: binding sets no `guarantee` slot (§2.2), so the candidate filter in `search_workload_with_targets` and `prepare_compositions` run `derive_guarantees` on the candidate alone, with a fresh memo, then check its accuracy target. The derived -copy is only read, then dropped: CandidatePostASAPDAGs keeps the original candidate, whose nodes are +copy is only read, then dropped: CandidateLogicalPostASAPDAGs keeps the original candidate, whose nodes are shared with other queries. **Assembly** — one rule replaces `assemble_residual`: @@ -404,7 +404,7 @@ Deleted: `assemble_residual`, `keep_pre_asap` / `keep_pre_asap_rc`, and the | #468 problem | Resolution | |---|---| -| 1. A `Project` is a `QueryExpr` inside `KeepPreAsap` and a `ValueOperation` outside | one set of types | +| 1. A `Project` is a `PreASAPNode` inside `KeepPreAsap` and a `ValueOperation` outside | one set of types | | 2. Nothing outside `KeepPreAsap` can reference the `Scan` inside, so an exact aggregate and a sketch cannot share a scan | `Aggregate` and `SummaryAgg` can point to the same `Scan`. This holds when both run at the same time; otherwise the scan is copied (above). With today's defaults the sketch runs at ingestion time and the exact `Aggregate` at query time, so they share only after the `QueryTime` default (separate PR, §2.3). Splitting a multi-measure `Aggregate` into exact + sketch is a binding rule, out of scope (§9) | | 3. `SetOp` and similar have no post-ASAP copy, so no summary below them | `SetOp` takes `None => t`; both children are assembled | @@ -436,22 +436,22 @@ The `KeepPreAsap` / `BinaryOp` / `ValueOperation` / `RelationalJoin` arms of tod ## 6. Export: fragments in the post-ASAP DAG -The four original-operator payloads (`fallback{expression: QueryExpr}`, `binary`, +The four original-operator payloads (`fallback{expression: PreASAPNode}`, `binary`, `value`, `relational_join`) become one: ```rust -PostAsapOperatorPayload::Relational { +PostASAPOperatorPayload::Relational { /// No ASAP node inside. Leaves are Scans, or Scan { source: Source::DagInput { role } } /// for an incoming edge whose schema is the edge's intermediate_schema. expression: Operator, } ``` -`compile_post_asap_dag` takes each **largest connected subtree without `ASAP`** as one +`export_post_asap_dag` takes each **largest connected subtree without `ASAP`** as one fragment, cutting an edge with a `DagInput` leaf wherever it meets an `ASAP` node. `ASAP` nodes map one-to-one onto the existing summary payloads; `FinalizeExactAccumulator` / `MaintainPopulation` / `ReadPopulation` stay -`value{operation}`. A backend lowers every fragment with its existing `QueryExpr` +`value{operation}`. A backend lowers every fragment with its existing `PreASAPNode` lowering plus a `DagInput` arm (an incoming edge as a materialized table); the `binary` / `value::Project` / `relational_join` lowerings go. @@ -460,7 +460,7 @@ lowering plus a `DagInput` arm (an incoming edge as a materialized table); the stage 4), together with the downstream readers. - **Timing and guarantee** are read from the node slots; an `Unset` slot is rejected. An edge's `data_state` is its producer's timing plus the primitive of its kind (§5). - `compile_post_asap_dag` no longer re-runs data-state validation. + `export_post_asap_dag` no longer re-runs data-state validation. - **`SummaryMerge`** stays a wire payload, although its planner-side variant is unimplemented (§1.3, §10). - **Phases** become per fragment. Switching phase inside a fragment would need a @@ -493,9 +493,9 @@ lowering plus a `DagInput` arm (an incoming edge as a materialized table); the | 1 Split | [decoupling doc](decoupling_op_and_expr.md): `NonASAPOp` + `ScalarExpr`; children stay `Rc` | scalar code ([decoupling doc §3](decoupling_op_and_expr.md#3-changes)) | | 2 Two levels | §1.1, §1.4: `Operator`, an empty `ASAPOp`, `contains_asap()`, `expect_non_asap()`; child slots become `Rc>`; every variant gets `timing` / `guarantee` slots, and nodes are built through constructors that leave both `Unset` | every crate; the same mechanical change everywhere | | 3 One schema | §2.1: `Column` → `Field` and `Schema.columns` → `fields` (serde keeps the name `columns` until stage 4); `FieldType`, `ASAPType`, `PlainField`, `Schema` everywhere except the `post_asap_dag.rs` wire types, which keep `SummarySchema` until stage 4. **No wire change** | `asap-types` + schema construction in every crate | -| 4 New types | fill `ASAPOp`; `ASAP` arms of `output_schema`; `derive_guarantees` and `derive_timings` (§2.2, §5); the entry check (§3); `flatten(&SummaryNode) -> Rc` so export runs on the new types, copying each node's guarantee and today's derived timing into the slots, so the export is unchanged; wire types become `Schema`, and `Schema.fields` serializes as `fields`. Wire → 6. **The only wire-breaking stage**; merged together with ASAPQuery-backend and ASAPCollector | `asap-types`, `devtools`, viewer | +| 4 New types | fill `ASAPOp`; `ASAP` arms of `output_schema`; `derive_guarantees` and `derive_timings` (§2.2, §5); the entry check (§3); `flatten(&PostASAPNode) -> Rc` so export runs on the new types, copying each node's guarantee and today's derived timing into the slots, so the export is unchanged; wire types become `Schema`, and `Schema.fields` serializes as `fields`. Wire → 6. **The only wire-breaking stage**; merged together with ASAPQuery-backend and ASAPCollector | `asap-types`, `devtools`, viewer | | 5 Planner | §4: candidates and assembly on `Rc`; §7 moves to the new types; delete `flatten` | `asap-aware-mapping` | -| 6 Cleanup | delete `SummaryExpr`, `SummaryNode`, extra `ValueOperation` variants, `ExactOperation`, `post_asap/cse.rs`; update `post-asap-ir.md`, `physical-plan-integration.md`, developer and viewer docs | docs | +| 6 Cleanup | delete `SummaryExpr`, `PostASAPNode`, extra `ValueOperation` variants, `ExactOperation`, `post_asap/cse.rs`; update `post-asap-ir.md`, `physical-plan-integration.md`, developer and viewer docs | docs | Wrapping pre-ASAP operators in `ValueOperation` first is not planned: stage 4 gives the same early flat export, on the final types. diff --git a/docs/design_docs/proposals/planner-layering.md b/docs/design_docs/proposals/planner-layering.md new file mode 100644 index 00000000..f6355166 --- /dev/null +++ b/docs/design_docs/proposals/planner-layering.md @@ -0,0 +1,169 @@ +# Planner and deployment layering + +Status: proposal. Audience: designers of ASAPPlanner and of deployments such +as ASAPQuery-backend. + +## Goal + +ASAPPlanner takes a query workload, a data workload and the deployment's +inputs, and returns one optimal physical plan. It decides what is computed, how +it is computed, and which plan is best. The deployment only supplies inputs and +executes the plan: it supplies its own cost model but never ranks or selects, +never re-derives the computation, and keeps no operators of its own. + +## Layers + +```text + Query workload (PromQL / SQL / MetricsQL, query repeating pattern, etc.) + + data workload + + deployment inputs: cost model, accuracy requirements, capabilities + │ +┌───────────────────────┴───────── ASAPPlanner ────────────────────┐ +│ 0. Frontends │ +│ Parse + lower -> CandidatePreASAPDAGs │ +│ Reject unsupported constructs, such as PromQL fill. │ +│ │ │ +│ Logical planning (what) │ +│ 1. Logical optimization │ +│ Summary families, rewrites, exact candidates │ +│ -> CandidateLogicalPostASAPDAGs │ +│ │ │ +│ Physical planning (how) │ +│ 2. Summary lifecycle planning │ +│ Per summary state: Ephemeral | Prepared | Shared | │ +│ ContinuouslyMaintained -> node timing, window framework, │ +│ retention -> CandidateLifecyclePostASAPDAGs │ +│ │ │ +│ 3. Compilation │ +│ Lower each node to operators; cut by timing │ +│ -> CandidatePhysicalPostASAPDAGs │ +│ │ │ +│ 4. Selection │ +│ Cost every candidate with the deployment's cost model; │ +│ choose the cheapest admissible one for the workload. │ +└──────────────────────┬───────────────────────────────────────────┘ + one optimal PhysicalPostASAPDAG (the boundary) +┌──────────────────────┴──────── Deployment ───────────────────────┐ +│ 5. Execution: ingest, panes, storage, readout, run │ +└──────────────────────────────────────────────────────────────────┘ +``` + +Layers 0 to 3 each output a candidate set, `Candidates`, holding every +legal candidate of their stage (for example KLL and DDSketch for one quantile, +or each lifecycle assignment), and selection is the only step that chooses. Candidate sets are internal to ASAPPlanner: they may +be shared or enumerated lazily, and the deployment never sees them. Unsupported +or infeasible candidates are rejected with reasons, not silently dropped. + +## DAGs and what each encodes + +Each stage adds decisions to the DAG it receives. The table shows which +decisions each DAG carries. + +| | `PreASAPDAG` | `LogicalPostASAPDAG` | `LifecyclePostASAPDAG` | `PhysicalPostASAPDAG` | +|---|---|---|---|---| +| Produced by | 0. Frontends | 1. Logical optimization | 2. Summary lifecycle planning | 3. Compilation; 4. selects one | +| Node | Query operation | Logical operation, including summary operations | Same, plus annotations | Physical operator | +| Logical optimization (summary family, rewrites) | No | Yes | Yes | Yes | +| Materialization decided (which summary states persist) | No | No | Yes | Yes | +| Data lifecycle (how each state is maintained: `Ephemeral`, `Prepared`, `Shared`, `ContinuouslyMaintained`) | No | No | Yes | Yes | +| Retention (how long each state lives) and window framework | No | No | Yes | Yes | +| Execution time (ingestion or query) | No | No | Yes, per node | Yes, as the precompute / query split | +| Physical optimization (operator choice, e.g. TopK as sort + limit) | No | No | No | Yes | +| Seen by the deployment | No | No | No | Only the selected one | + +Name mapping to code: + +| Design name | Current main | Target API (open PRs #508, #480) | +|---|---|---| +| `PreASAPDAG` | `Rc` | `PreASAPDAG` | +| `LogicalPostASAPDAG` | `Rc` tree; exported as `PostAsapDag` | `LogicalPostASAPDAG` | +| `LifecyclePostASAPDAG` | `SummaryMaintenanceLifecyclePlan`, one selected assignment beside the DAG | `LifecyclePostASAPDAG`; `SummaryMaintenanceLifecyclePlan` is merged into it | +| `PhysicalPostASAPDAG` | None | `PhysicalPostASAPDAG` | +| `CandidatePreASAPDAGs` | None; one `QueryExpr` root per entry | `CandidatePreASAPDAGs` | +| `CandidateLogicalPostASAPDAGs` | `PlanSpace` | `CandidateLogicalPostASAPDAGs` | +| `CandidateLifecyclePostASAPDAGs` | None | `CandidateLifecyclePostASAPDAGs` | +| `CandidatePhysicalPostASAPDAGs` | None | `CandidatePhysicalPostASAPDAGs` | + +**Post-ASAP DAG to lifecycle DAG.** A summary state's *lifecycle* +(`Ephemeral`, `Prepared`, `Shared`, `ContinuouslyMaintained`) fixes several +separate aspects together: whether the state is materialized (kept across +executions, like a materialized view), when it is computed, how it is +maintained, how long it is retained, and its window framework. The table above +lists these aspects separately; the lifecycle is the one choice that sets them. +Choosing lifecycles is a workload-level decision, like a database's +materialized-view selection: it spans queries (a shared state is kept once) and +depends on workload demand (read and update rates, horizon). A lifecycle +assignment annotates every node with execution timing, and every stored state +with window framework and retention: + +* a retained state (`ContinuouslyMaintained`, `Shared`, `Prepared`) and every + node feeding it run at ingestion time; +* readouts, other consumers and `Ephemeral` states run at query time; +* an `Ephemeral` state that feeds a retained state runs at ingestion time, + because query-time work may not feed ingestion-time work. + +The result is still a Post-ASAP DAG, much as physical properties annotate +logical expressions in a database optimizer. This is the only source of +timing; logical optimization proposes computations, never timing or placement. + +**Lifecycle DAG to physical DAG.** Compilation lowers each node to physical +operators and cuts the graph at the timing frontier (ingestion-time nodes read +by query-time nodes, plus an ingestion-time root) into a precompute and a query +DAG. Materialization is decided in layer 2 and realized here: the precompute +DAG's outputs at the cut are the materialized states, and the query DAG reads +them through typed input slots. A physical DAG corresponds to its +Post-ASAP DAG node by node; a node may expand into several operators, whose +helper operators are numbered from their source node. The one exception, a +`Fallback` node wrapping a whole Pre-ASAP expression, is removed by the +operator-flattening proposal ([operator sharing](operator-sharing.md), #469, +#481). Operator materialization (sort, aggregation, summary build) and +computing a shared subexpression once are compilation and runtime details. + +Binding runtime sources is an execution step of a `PhysicalPostASAPDAG`, not +another DAG: the deployment supplies a source for each typed input slot, the +slots are checked against their contracts, and the graph runs. + +## Responsibilities + +| Layer | Owns | Does not own | +|---|---|---| +| 0. Frontends | Language semantics and lowering. A construct that cannot be represented faithfully is rejected, never ignored (for example PromQL `fill`). | Summaries, placement | +| 1. Logical optimization | All legal logical candidates: summary families, exact rewrites, compositions, series-identity typing. | Placement, timing | +| 2. Summary lifecycle planning | For each unique summary state and maintained population, the admissible lifecycle assignments and their timing, window framework and retention. | Cost values; operator implementation | +| 3. Compilation | All computation: value operations, aggregation, PromQL functions and subqueries, vector matching, comparisons and set operators, `histogram_quantile`, summary build, merge and estimate, sort, limit, joins. | Raw ingestion, pane construction, storage formats, decoding persisted state, scheduling | +| 4. Selection | Costing every candidate with the deployment's cost model and returning the cheapest admissible `PhysicalPostASAPDAG` for the whole workload that meets the accuracy requirements. A state shared by several queries is costed once with all consumers' demand (only when compilation installs one shared output: same window layout, evaluation interval and phase). Unknown cost stays unknown and such a candidate is not selected. | The cost values | +| 5. Deployment | Inputs: the cost model (build, per-update maintenance, read, store price per byte-second, retirement, query-time raw processing; optionally whole-plan quotes), accuracy requirements and capabilities (for example whether query-time raw data is available). Execution: ingestion and routing, panes and completeness, lateness and revisions, storage and codecs over Planner kernel states, reading stored state into typed inputs, query-time raw sources, the exact-engine fallback. Sampled or delta edge frames are rejected. | Any computation algorithm | + +## The boundary + +The deployment passes its inputs to ASAPPlanner and receives one optimal +`PhysicalPostASAPDAG`, which contains: + +* a **precompute DAG**, whose inputs are raw-sample contracts (rows carrying + series labels, timestamp and value; the label set is the complete series + identity) and whose outputs are typed summary states; +* a **query DAG**, whose inputs are stored-state contracts, query-time + raw-series contracts, or both; +* the lifecycle, window framework and retention of every stored output. + +The deployment binds each input contract, stores each precompute output under +its own storage identity, and returns the query DAG's result. Semantic identity +of stored outputs is defined by the logical DAG they compute. Storage identity +and encoding belong to the deployment. + +## Example + +This traces `sum by (job) (rate(m[1m]))`, evaluated every 10 s, through the +four DAGs, and shows that only the deployment's store price changes the plan. + +* `PreASAPDAG`: `sum by (job)` over `rate` over the range selector `m[1m]`. +* `LogicalPostASAPDAG`: a per-series Rate state feeding a grouped Sum state. +* `LifecyclePostASAPDAG`: one per lifecycle assignment, for example + (a) both retained, (b) Rate retained and Sum `Ephemeral`, (c) both + `Ephemeral`. +* `PhysicalPostASAPDAG`: one compilation, cut three ways. (a) Precompute builds + Rate and Sum per pane; the query only reads Sum. (b) Precompute keeps Rate; + the query builds Sum. (c) No precompute; the query reads raw series at `t_q`. + +Selection returns (a) when storage is cheap, (b) when it is expensive, and (c) +when it is more expensive still. The deployment only changed its store price. diff --git a/docs/develop_docs/README.md b/docs/develop_docs/README.md index 817722ad..1ecc8cb1 100644 --- a/docs/develop_docs/README.md +++ b/docs/develop_docs/README.md @@ -16,5 +16,6 @@ formats, evidence, and verification workflows. - [Physical handoff cost references](physical-handoff-costs.md), [storage operations](storage-operation-costs.md) - [Replacement explanations](replacement-explanations.md) - [Physical compile coverage for deployment computation](physical-compile-coverage.md) +- [DAG API migration](dag-api-migration.md) - [Planner vocabulary migration (#427)](planner-vocabulary-migration.md) diff --git a/docs/develop_docs/asap-aware-mapping-architecture.md b/docs/develop_docs/asap-aware-mapping-architecture.md index 06f47bfc..37776e43 100644 --- a/docs/develop_docs/asap-aware-mapping-architecture.md +++ b/docs/develop_docs/asap-aware-mapping-architecture.md @@ -55,7 +55,7 @@ The diagram below follows a workload of one or more query roots through target d Terminology used in the diagram: - A **workload** is the set of named queries planned together. A **query root** - is the top-level `QueryExpr` (the logical query-expression type) for one of + is the top-level `PreASAPNode` (the logical query-expression type) for one of those queries. **Pre-ASAP** means this logical input form, before the planner realizes an operation as a concrete ASAP realization; **post-ASAP** means the resulting realization form. @@ -72,7 +72,7 @@ Terminology used in the diagram: is a compact data structure that trades exactness for bounded error. A query's **accuracy target** states the allowed error and failure probability. A candidate's **rationale** is its human-readable explanation. -- `CandidatePostASAPDAGs` is a compact candidate space with one +- `CandidateLogicalPostASAPDAGs` is a compact candidate space with one `TargetSubDAGCandidates` per target instead of one full plan per combination of choices. A `node_hash` is a structural fingerprint used to narrow explanation lookup; exact structural equality is still checked afterward. @@ -86,9 +86,9 @@ flowchart TB classDef report fill:#f2eafe,stroke:#7950b3,color:#34204f subgraph DISCOVERY[1. Discover every replaceable site] - WL["Input workload
one or more named pre-ASAP QueryExpr roots"]:::input + WL["Input workload
one or more named pre-ASAP PreASAPNode roots"]:::input SEARCH["search_workload_with
run CSE once, then visit every node in every root DAG"]:::generate - TARGET["TargetSubDAG
one candidate site plus the number of workload locations
that reference the same Rc<QueryExpr>"]:::generate + TARGET["TargetSubDAG
one candidate site plus the number of workload locations
that reference the same Rc<PreASAPNode>"]:::generate WL -->|"roots"| SEARCH -->|"one target per distinct node"| TARGET end @@ -101,12 +101,12 @@ flowchart TB end subgraph SEARCHSPACE[3. Store the workload-wide search space] - SPACE["CandidatePostASAPDAGs
one TargetSubDAGCandidates per target; each candidate set keeps
all candidates, including dependent compositions"]:::store + SPACE["CandidateLogicalPostASAPDAGs
one TargetSubDAGCandidates per target; each candidate set keeps
all candidates, including dependent compositions"]:::store CAND -->|"deduplicate by target and candidate identity"| SPACE end subgraph RANKING[Optional ranked view] - SORT["CandidatePostASAPDAGs::cost_sorted
use the CostModel to order each candidate set
and cost every candidate"]:::choose + SORT["CandidateLogicalPostASAPDAGs::cost_sorted
use the CostModel to order each candidate set
and cost every candidate"]:::choose RANKED["RankedTargetSubDAGCandidates
the same candidates in preferred order,
with costs aligned by index"]:::choose SPACE --> SORT -->|"reorder only; preserve every candidate"| RANKED end @@ -157,7 +157,7 @@ flowchart LR classDef workload fill:#e7f7ef,stroke:#31835e,color:#173f2d classDef common fill:#fff6dd,stroke:#b78922,color:#513d0c - ROOTS["Input
one or more named QueryExpr roots"]:::workload + ROOTS["Input
one or more named PreASAPNode roots"]:::workload ROOTS --> CSE["Canonicalize sharing
merge structurally identical, legally shareable subtrees"]:::workload CSE --> WALK["Discover sites
walk the complete DAG, including nodes below unshared parents"]:::workload WALK --> T["Build TargetSubDAG
retain the subtree's Rc identity and measured consumer_count"]:::workload @@ -200,7 +200,7 @@ cost. The default context-free registry contains five `ReplacementStrategy` implementations: - `SketchAlgorithmStrategy` matches supported aggregate and binary shapes. Its - `replacements(target)` method constructs every legal post-ASAP `SummaryNode`, + `replacements(target)` method constructs every legal post-ASAP `PostASAPNode`, including applicable sketch, exact-accumulator, and pass-through realizations. Candidates are sized and ordered for the target's accuracy requirement; candidates without a sufficient guarantee are rejected before @@ -229,13 +229,13 @@ strategy-specific discovery logic. ### 3.4 Store and rank the complete search space -Workload search deduplicates candidates into a `CandidatePostASAPDAGs`. Each distinct +Workload search deduplicates candidates into a `CandidateLogicalPostASAPDAGs`. Each distinct target has one `TargetSubDAGCandidates` containing retained alternatives and rejection reasons. This compact representation preserves independent choices without enumerating a flat list of `2^N` complete plans for `N` replaceable targets. -`CandidatePostASAPDAGs::cost_sorted` ranks each target's existing candidates with the +`CandidateLogicalPostASAPDAGs::cost_sorted` ranks each target's existing candidates with the supplied `CostModel`. It returns the same candidates in preferred order, with costs aligned by index; ranking does not select or remove a candidate. @@ -253,7 +253,7 @@ execution policy. Constructing all candidates before taking the first costs more than constructing only the preferred candidate, but it keeps the strategy contract consistent and preserves the full choice set for other callers. -`CandidatePostASAPDAGs::global_selection` optionally coordinates cross-target sharing and +`CandidateLogicalPostASAPDAGs::global_selection` optionally coordinates cross-target sharing and composition choices. `GlobalSelection::assemble_selected_dag` constructs the selected semantic DAG. These plain APIs do not establish lifecycle or physical deployment feasibility. Recurrence and lifecycle-aware variants require the corresponding diff --git a/docs/develop_docs/asap-aware-mapping-contracts.md b/docs/develop_docs/asap-aware-mapping-contracts.md index 3ea1d154..3ce93741 100644 --- a/docs/develop_docs/asap-aware-mapping-contracts.md +++ b/docs/develop_docs/asap-aware-mapping-contracts.md @@ -10,18 +10,18 @@ first; use the [extension guide](extend-asap-aware-mapping.md) when changing one ### `TargetSubDAG` -A pre-ASAP `QueryExpr` node that a strategy may replace. +A pre-ASAP `PreASAPNode` node that a strategy may replace. ```rust pub struct TargetSubDAG<'a> { - pub root: &'a Rc, + pub root: &'a Rc, pub consumer_count: usize, } ``` -`root` is the actual `Rc` from the workload. +`root` is the actual `Rc` from the workload. -`consumer_count` counts structural references, not runtime executions. It is the number of places in the workload DAG that point to this exact `Rc` node. +`consumer_count` counts structural references, not runtime executions. It is the number of places in the workload DAG that point to this exact `Rc` node. For example, consider two top-level queries: @@ -58,15 +58,15 @@ There are currently three forms: ```rust pub enum Replacement { - Summary(Rc), - Rewrite(Rc), + Summary(Rc), + Rewrite(Rc), ExactComposition(ExactComposition), } ``` Use `Replacement::Summary` when the alternative is a constructed post-ASAP summary plan. -Use `Replacement::Rewrite` when the alternative is still a logical pre-ASAP `QueryExpr`. +Use `Replacement::Rewrite` when the alternative is still a logical pre-ASAP `PreASAPNode`. Use `Replacement::ExactComposition` when an exact operation refers to a child target whose realization must remain undecided. Selection coordinates the @@ -77,7 +77,7 @@ Examples: ```text Quantile(...) - -> KLL SummaryNode + -> KLL PostASAPNode ``` is a `Summary`; KLL (Karnin–Lang–Liberty) is a quantile-sketch algorithm. @@ -173,7 +173,7 @@ This guide uses the Cascades/Volcano terminology: operation. In this crate, that kind of candidate is represented by `Replacement::Rewrite`. - A **replacement candidate** packages either kind of result as a - `ReplacementSubDAG` for search. `CandidatePostASAPDAGs` stores and ranks these candidates. + `ReplacementSubDAG` for search. `CandidateLogicalPostASAPDAGs` stores and ranks these candidates. - **Physical commitment and placement** happen downstream. An `Realization` therefore does not mean that the planner has committed the workload to that choice. @@ -184,8 +184,9 @@ The concrete flow is: AggIntent -> realizations_for_intent(): enumerate Realization values -> SketchAlgorithmStrategy: construct ReplacementSubDAG candidates - -> CandidatePostASAPDAGs: store and rank candidates - -> downstream deployment: select and place a final choice + -> CandidateLogicalPostASAPDAGs: store candidates + -> Planner selection with the deployment's cost model + -> deployment: place and execute the selected plan ``` `ReplacementStrategy` enumerates supported legal candidates. The caller can @@ -273,7 +274,7 @@ bounds, but does not execute workloads or own deployment measurements. Most hook fn cse_share_decision(&self, candidate: &CseCandidate) -> ShareDecision; ``` -- **`estimate_cost`** — attach a comparable numeric cost to an already-constructed replacement. `CandidatePostASAPDAGs::cost_sorted` calls it for every candidate and keeps the returned values aligned with the ranked candidates. The trait default returns `f64::NAN` deliberately; override it when a custom model's callers need displayable or otherwise consumable numeric costs. `DefaultCostModel` provides real values derived from its CSE cost hooks. +- **`estimate_cost`** — attach a comparable numeric cost to an already-constructed replacement. `CandidateLogicalPostASAPDAGs::cost_sorted` calls it for every candidate and keeps the returned values aligned with the ranked candidates. The trait default returns `f64::NAN` deliberately; override it when a custom model's callers need displayable or otherwise consumable numeric costs. `DefaultCostModel` provides real values derived from its CSE cost hooks. ```rust fn estimate_cost( @@ -287,9 +288,9 @@ A custom cost model does not necessarily need to override every hook. The curren --- -### `CandidatePostASAPDAGs` / `TargetSubDAGCandidates` / `RankedTargetSubDAGCandidates` — the whole-workload view +### `CandidateLogicalPostASAPDAGs` / `TargetSubDAGCandidates` / `RankedTargetSubDAGCandidates` — the whole-workload view -`ReplacementStrategy` answers "what are the candidates for this one target?" `CandidatePostASAPDAGs` answers the same question for every target in a whole workload at once, without enumerating `2^N` fully-copied plans for `N` independently-choosable sites. +`ReplacementStrategy` answers "what are the candidates for this one target?" `CandidateLogicalPostASAPDAGs` answers the same question for every target in a whole workload at once, without enumerating `2^N` fully-copied plans for `N` independently-choosable sites. ```rust // replacement.rs @@ -297,14 +298,14 @@ A custom cost model does not necessarily need to override every hook. The curren // One TargetSubDAGCandidates per distinct TargetSubDAG in the whole workload — // never a flat list of fully assembled plans. pub struct TargetSubDAGCandidates { - pub target: Rc, + pub target: Rc, pub consumer_count: usize, pub candidates: Vec, // accepted alternatives, unranked pub rejected: Vec, // failed accuracy checks } pub struct RankedTargetSubDAGCandidates<'a> { - pub target: &'a Rc, + pub target: &'a Rc, pub consumer_count: usize, pub candidates: Vec<&'a ReplacementSubDAG>, // same candidates, ranked pub costs: Vec, // costs[i] <-> candidates[i] @@ -313,7 +314,7 @@ pub struct RankedTargetSubDAGCandidates<'a> { `search_workload(roots)` runs the shared-subtree pass once, discovers every target across every root's whole DAG (not just root-level sharing — a `SharedSubtreeStrategy` candidate three levels under an unshared `Filter` is exactly as real a site as a shared whole root), and asks every registered strategy to a fixpoint. Two logically different candidates at two different targets are never copied into two separate plans — they're two entries in two different `TargetSubDAGCandidates`s, sharing every other node in the workload by construction. -`CandidatePostASAPDAGs::cost_sorted(cost_model)` is the one ranking step: for each candidate set, it dispatches by candidate shape — a same-shape `Rewrite` pair (a `SharedSubtreeStrategy` share/recompute choice) goes through `CostModel::cse_share_decision`; a same-shape run of `Summary` candidates realizing sketches (a `SketchAlgorithmStrategy` choice) goes through `CostModel::rank_candidates`; and a mixed candidate set is ordered by each candidate's `CostModel::estimate_cost`. Every candidate gets a numeric cost aligned index-for-index in `costs`. Count in, count out—nothing is dropped to produce a ranking. Legality checks +`CandidateLogicalPostASAPDAGs::cost_sorted(cost_model)` is the one ranking step: for each candidate set, it dispatches by candidate shape — a same-shape `Rewrite` pair (a `SharedSubtreeStrategy` share/recompute choice) goes through `CostModel::cse_share_decision`; a same-shape run of `Summary` candidates realizing sketches (a `SketchAlgorithmStrategy` choice) goes through `CostModel::rank_candidates`; and a mixed candidate set is ordered by each candidate's `CostModel::estimate_cost`. Every candidate gets a numeric cost aligned index-for-index in `costs`. Count in, count out—nothing is dropped to produce a ranking. Legality checks may already have removed proposals before this boundary. In particular, `search_workload_with_targets` checks explicit per-root targets, while retaining direct DDSketch ratios with missing domain evidence and no root guarantee for @@ -370,11 +371,11 @@ The crate provides no default `Matcher` implementation because the answer depend ## 2. Replacement explanations (`explanation.rs`) -`explanation::explain_replacements`/`explain_replacements_with` answer a different question than everything above: not "what could this target become" (`ReplacementStrategy::replacements`) but "why does the replacement already discovered for this target exist, and where." It is a **reporting view over `CandidatePostASAPDAGs`**, not a second search or a second rule engine — this crate's *explanation of a replacement*, not an applicability classifier deciding admissibility from scratch. +`explanation::explain_replacements`/`explain_replacements_with` answer a different question than everything above: not "what could this target become" (`ReplacementStrategy::replacements`) but "why does the replacement already discovered for this target exist, and where." It is a **reporting view over `CandidateLogicalPostASAPDAGs`**, not a second search or a second rule engine — this crate's *explanation of a replacement*, not an applicability classifier deciding admissibility from scratch. ### The rule -> A `TargetSubDAG` is worth explaining exactly when its `CandidatePostASAPDAGs` candidate list contains something beyond the trivial, no-op realization. +> A `TargetSubDAG` is worth explaining exactly when its `CandidateLogicalPostASAPDAGs` candidate list contains something beyond the trivial, no-op realization. Concretely, `explanation.rs` reports three candidate kinds from each `TargetSubDAGCandidates`: @@ -390,10 +391,10 @@ Each `ReplacementExplanation::reason` is copied verbatim from the matching candi ### Why there is no `ExplanationRule` trait -Explanations are derived from candidates already present in `CandidatePostASAPDAGs`. A new candidate kind therefore requires an `impl ReplacementStrategy` wired into `default_strategies`/`default_strategies_with`; a second explanation-specific trait would duplicate registration and could drift from the actual search space. Custom callers supply strategies through `explain_replacements_with`, using the same extension point exposed by `search_workload_with`. +Explanations are derived from candidates already present in `CandidateLogicalPostASAPDAGs`. A new candidate kind therefore requires an `impl ReplacementStrategy` wired into `default_strategies`/`default_strategies_with`; a second explanation-specific trait would duplicate registration and could drift from the actual search space. Custom callers supply strategies through `explain_replacements_with`, using the same extension point exposed by `search_workload_with`. ### How it derives `location` text -`CandidatePostASAPDAGs`/`TargetSubDAGCandidates` track `Rc` pointer identity, not human-readable breadcrumbs. `ReplacementExplanation::location` provides prose such as `root "dash_a" > lhs` so reporting consumers can identify the relevant part of the query without interpreting pointer identity. Location derivation does not make replacement or costing decisions. +`CandidateLogicalPostASAPDAGs`/`TargetSubDAGCandidates` track `Rc` pointer identity, not human-readable breadcrumbs. `ReplacementExplanation::location` provides prose such as `root "dash_a" > lhs` so reporting consumers can identify the relevant part of the query without interpreting pointer identity. Location derivation does not make replacement or costing decisions. --- diff --git a/docs/develop_docs/dag-api-migration.md b/docs/develop_docs/dag-api-migration.md new file mode 100644 index 00000000..5e392ed3 --- /dev/null +++ b/docs/develop_docs/dag-api-migration.md @@ -0,0 +1,112 @@ +# DAG API migration + +Audience: Rust integrators. This is a breaking API change on the #508 integration. +There are no compatibility aliases for the former graph names. + +| Previous API | Replacement | +|---|---| +| `QueryExpr`, `ResolvedQueryExpr`, `UnresolvedQueryExpr`, `QueryExprError` | `PreASAPNode`, `ResolvedPreASAPNode`, `UnresolvedPreASAPNode`, `PreASAPNodeError` | +| `Rc` | `PreASAPDAG` (same shared root representation) | +| `SummaryNode`, `Rc` | `PostASAPNode`, `LogicalPostASAPDAG` | +| `PlanSpace` | `CandidateLogicalPostASAPDAGs` | +| `PostAsapDag`, `PostAsapDagDocument` | `LogicalPostASAPDAGTransport`, `LogicalPostASAPDAGDocument` for explicit transport | +| `PostAsapNodeId`, `PostAsapOperatorPayload`, `PostAsapDagNode`, `PostAsapDagEdge`, `PostAsapDagValidationError`, `PostAsapNodeIdentityMap`, `PostAsapSubstitution` | `PostASAPNodeId`, `PostASAPOperatorPayload`, `LogicalPostASAPDAGNode`, `LogicalPostASAPDAGEdge`, `LogicalPostASAPDAGValidationError`, `PostASAPNodeIdentityMap`, `PostASAPSubstitution` | +| `InvalidPostAsapDag` error variants | `InvalidLogicalPostASAPDAG` | +| `compile_post_asap_dag` | `export_post_asap_dag` | +| `compile_post_asap_dag_with_node_ids`, `PostAsapDagCompilation` | `index_post_asap_dag` returning `LogicalPostASAPDAGIndex` (`node_ids`, `view()`, `to_transport()`) | +| `execution_timed_dag` | `export_timed_dag` for transport; use `execution_assignment` for shared compilation | +| `CompiledPhysicalDag` | `PhysicalPostASAPDAG` | +| Runtime-bound `PhysicalDag` | `PhysicalExecution`, the execution handle returned by `PhysicalPostASAPDAG::instantiate` | +| `compile(&dag, …)`, `frontier_from_timing(&dag)` over a transport | `compile(dag.as_view(), …)`, `frontier_from_timing(dag.as_view())`; an index or assignment passes `view()` | +| `enumerate_summary_maintenance_lifecycles`, `SummaryMaintenanceLifecycleCandidates` | `CandidateLifecyclePostASAPDAGs` (see below) | +| `SummaryMaintenanceLifecyclePlan`, `SummaryMaintenanceLifecyclePlanError` | `LifecyclePostASAPDAG` (same fields), `LifecyclePostASAPDAGError` | +| Earlier #508/#480 names: `PostASAPDAG`, `CandidatePostASAPDAGs`, `PhysicalDAG`, `CandidatePhysicalDAGs`; `PostASAPDAG*` compounds (`Node`, `Edge`, `Transport`, `Document`, `ValidationError`, `View`, `Assignment`, `Index`); `InvalidPostASAPDAG` | `LogicalPostASAPDAG`, `CandidateLogicalPostASAPDAGs`, `PhysicalPostASAPDAG`, `CandidatePhysicalPostASAPDAGs`; `LogicalPostASAPDAG*` compounds; `InvalidLogicalPostASAPDAG` | + +## Candidate generation + +`asap_planner::lower_pre_asap_dag_candidates(&input).await` returns +`CandidatePreASAPDAGs`, keyed by normalized workload entry index. Batch +and repeating entries retain their identities. Current frontends lower each +entry deterministically. This function does not invoke an optimization pass. +`search_workload` and the target-aware search APIs consume the same root/ID +collection and produce the compact `CandidateLogicalPostASAPDAGs`. + +The normal stage transition is: + +```rust,ignore +let timed = logical.with_timing_for_root( + &workload_entry_id, + timing_context, + logical_expansion_limit, + assignment_expansion_limit, +)?; +let physical = compile_physical_dag_candidates( + timed.iter(), + timed.rejected_assemblies().to_vec(), + |metadata, assignment| { + // Supply typed contracts and requested root IDs for this realization. + resolve_contracts(metadata, assignment) + }, +); +``` + +`timed` has type `CandidateLifecyclePostASAPDAGs<'a, Id>`; `physical` has type +`CandidatePhysicalPostASAPDAGs, Rc>`. +`CandidateTimingContext` binds the root's `WorkloadDemand`, planning clock, +horizon, lifecycle capabilities and cost model. No winner-selection helper runs +during these transitions. + +The timed collection enumerates assignments lazily and owns the shared graph +indices. Both expansion budgets are checked before iteration, including the +total assignment count across logical alternatives; exceeding either returns +`CandidateTimingError::ExpansionLimit`. Rejected logical assemblies remain +accessible through `rejected_assemblies()`. Iterator entries retain workload ID, +logical/assignment indices, choices, lifecycle plan and timing or rejection. A +state with no lifecycle alternative appears as one entry with a `NoAlternatives` +error. Unknown cost stays unknown; absent window evidence must still be resolved +before installation. + +Callers with an already assembled logical graph enter the same collection via +`CandidateLifecyclePostASAPDAGs::from_post_asap_dag(id, root, context, limit)`. +`lifecycle_alternatives(logical_index)` supports inspection, +`lifecycle_guarantee(logical_index, state, lifecycle)` answers the guarantee of one +of a state's alternatives, so the deployment can supply its price before it +is bound, and `select_lifecycles(logical_index, choices)` remains an explicit +opt-in selection operation. + +`compile_physical_dag_candidates` accepts any iterator of +`(metadata, Result)`, so the physical crate does not +depend on the mapping crate. Its contract resolver can supply different inputs +and roots for different realizations. The physical collection keeps every +timing or compilation failure with its original metadata: +`PhysicalCandidateError::Timing(E)` keeps the caller's typed timing error, and +`PhysicalCandidateError::Compile` holds a compilation error. It exposes shared +`PhysicalPostASAPDAG`s through `iter()`, frontiers through `frontier(index)`, and +execution cuts through `materialize(index)`. Cut descriptors and compilation +reuse are internal to `CandidatePhysicalPostASAPDAGs`. + +Whole-workload selection must still account for shared state and compatible +assignments across roots. Independent per-root minima do not prove a workload +minimum. Existing explicit selection helpers remain available. + +## Representation and validation + +`LogicalPostASAPDAG` is the authoritative shared logical graph. Its index keeps the +shared node references and projects node and edge records once for +compilation; the compiler uses the same validator and operator compiler for +those records and for imported transport documents. Export a flat graph +only when a transport consumer needs it. Serialization contracts remain checked. + +Timing is a total assignment over indexed node IDs. Missing assignments and +query-time producers feeding ingestion-time consumers are rejected. Assignments +share the index and graph; an assignment's `view()` overlays its timing on the +index's records, so read node and edge states through the view's `timing`, +`output_state` and `edge_state`. Lifecycle choices do not clone logical operators. + +Physical candidate generation shares each compiled graph across assignments +with the same index, Binary timings, input contracts and requested roots. +Ingestion-time Binary changes lowering, so incompatible assignments get +separate compilations. Candidate cuts are +materialized on demand using the existing cut implementation. The convenience +`compile_candidate` and `compile_candidates` APIs still eagerly materialize +explicit requested cuts; they are not the shared candidate-generation path. diff --git a/docs/develop_docs/end-to-end-accuracy-guarantees.md b/docs/develop_docs/end-to-end-accuracy-guarantees.md index 072c10c5..145c7355 100644 --- a/docs/develop_docs/end-to-end-accuracy-guarantees.md +++ b/docs/develop_docs/end-to-end-accuracy-guarantees.md @@ -77,7 +77,7 @@ trait AccuracyModel { approximate layers. Every candidate must then be resized, propagated, and checked before being treated as satisfying the target. This is distinct from candidate visibility: direct DDSketch ratios lacking domain evidence remain in -`CandidatePostASAPDAGs` with `guarantee: None` and can appear in `cost_sorted`, but automatic +`CandidateLogicalPostASAPDAGs` with `guarantee: None` and can appear in `cost_sorted`, but automatic `global_selection` skips them. Presence and cost are not accuracy certification. See the [workflow design](../design_docs/architecture/input-output-workflow.md#planning-evidence-inputs) for this boundary. diff --git a/docs/develop_docs/extend-asap-aware-mapping.md b/docs/develop_docs/extend-asap-aware-mapping.md index fddf6091..e194d4dd 100644 --- a/docs/develop_docs/extend-asap-aware-mapping.md +++ b/docs/develop_docs/extend-asap-aware-mapping.md @@ -257,7 +257,7 @@ flowchart LR A["Input TargetSubDAG
root is a supported Aggregate"] --> B["SketchAlgorithmStrategy::matches
check whether the target shape can produce summaries"] B -->|"true"| C["SketchAlgorithmStrategy::replacements
use CostModel preferences and sizing while preserving
every semantically valid realization"] B -->|"false"| NONE["Empty candidate list"] - C --> F["Output Vec<ReplacementSubDAG>
each entry contains a constructed SummaryNode and rationale;
all candidates retained in preferred order"] + C --> F["Output Vec<ReplacementSubDAG>
each entry contains a constructed PostASAPNode and rationale;
all candidates retained in preferred order"] ``` For an approximate quantile, both KLL and DDSketch remain candidates when @@ -277,7 +277,7 @@ let candidates = strategy.replacements(&target); for candidate in candidates { match candidate.replacement { Replacement::Summary(summary) => { - // Inspect or execute this constructed SummaryNode. + // Inspect or execute this constructed PostASAPNode. } Replacement::Rewrite(_) => unreachable!( "SketchAlgorithmStrategy produces summary candidates" @@ -310,7 +310,7 @@ and returns two alternatives: 2. Build independently for each consumer. ``` -The shared candidate reuses the same `Rc`: +The shared candidate reuses the same `Rc`: ```rust Replacement::Rewrite(Rc::clone(target.root)) @@ -327,7 +327,7 @@ Replacement::Rewrite( This strategy does **not** decide whether sharing is cheaper. That preference belongs to the cost model. -`CandidatePostASAPDAGs::cost_sorted` calls `CostModel::cse_share_decision` when it ranks a +`CandidateLogicalPostASAPDAGs::cost_sorted` calls `CostModel::cse_share_decision` when it ranks a share-versus-recompute candidate pair. The strategy still returns both alternatives because enumeration and ranking are separate steps: @@ -711,7 +711,7 @@ fn estimate_cost( ) -> f64; ``` -The default returns `f64::NAN`, making the absence of a numeric model explicit. Override this hook when passing the model to `CandidatePostASAPDAGs::cost_sorted` if downstream code displays or otherwise consumes the `costs` values. Prefer to derive the result from the same inputs used by `rank_candidates` and the CSE cost hooks so numeric costs do not disagree with relative ordering. +The default returns `f64::NAN`, making the absence of a numeric model explicit. Override this hook when passing the model to `CandidateLogicalPostASAPDAGs::cost_sorted` if downstream code displays or otherwise consumes the `costs` values. Prefer to derive the result from the same inputs used by `rank_candidates` and the CSE cost hooks so numeric costs do not disagree with relative ordering. --- @@ -778,7 +778,7 @@ Therefore, when adding a new built-in sketch algorithm, the intended flow is: flowchart LR MAP["1. Declare legality
add the algorithm to summary_candidates
for each AggIntent it can answer"] MAP --> MODEL["2. Define costing
rank it, derive its SketchParams,
and provide a comparable numeric cost"] - MODEL --> BUILD["3. Define realization behavior
ensure the public strategy output contains a valid SummaryNode
with the correct maintained state and readout"] + MODEL --> BUILD["3. Define realization behavior
ensure the public strategy output contains a valid PostASAPNode
with the correct maintained state and readout"] BUILD --> ACC["4. Certify accuracy
derive from committed parameters;
propagate and check the final target"] ACC --> ENUM["5. Verify integration
SketchAlgorithmStrategy includes it automatically;
tests confirm enumeration, ordering, sizing, and cost"] ``` @@ -895,7 +895,7 @@ silently disagree. ### Mistake: reimplementing summary construction inside a strategy -If the candidate should produce a normal `SummaryNode`, use the existing +If the candidate should produce a normal `PostASAPNode`, use the existing summary-construction path. A strategy should steer or wrap that path when necessary, not recreate schema derivation, column resolution, readout construction, or parameter sizing. @@ -922,7 +922,7 @@ Workload-wide target discovery, deduplication, and consumer counting are separat For CSE-style decisions, pointer identity can encode actual sharing. -Two `Rc` values can be structurally equal but deliberately represent independent computation. +Two `Rc` values can be structurally equal but deliberately represent independent computation. Use the distinction intentionally. @@ -983,9 +983,9 @@ Use this table to find the right place for a change. | Produce a normal (ranked-first) post-ASAP summary for one target | `SketchAlgorithmStrategy::replacements(...).into_iter().next()` | | Search a whole workload for supported legal candidates | `search_workload`/`search_workload_with` | | Enforce per-root result accuracy requirements | `search_workload_with_targets` | -| Coordinate compatible choices across groups | `CandidatePostASAPDAGs::global_selection` | +| Coordinate compatible choices across groups | `CandidateLogicalPostASAPDAGs::global_selection` | | Assemble the selected logical DAG | `GlobalSelection::assemble_selected_dag` | -| Get every candidate ranked best-first, across a whole workload | `CandidatePostASAPDAGs::cost_sorted` | +| Get every candidate ranked best-first, across a whole workload | `CandidateLogicalPostASAPDAGs::cost_sorted` | | Get a real numeric cost per candidate, not just a relative rank | `CostModel::estimate_cost` | | Enumerate valid sketch algorithms | `summary_candidates` | | Build a target with no workload context | `TargetSubDAG::new` | diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index 1bda9f5a..9cc4602e 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -4,7 +4,7 @@ Audience: developers embedding ASAPPlanner or adding strategies/models. This is a compact reference for the public workflow APIs, not an exhaustive symbol reference. The [CLI guide](../user_guide_docs/run-a-query.md) covers command-line inspection; the [design overview](../design_docs/architecture/README.md) defines ownership. -ASAPPlanner's primary output is `CandidatePostASAPDAGs`; ranking is a view over its candidates. +ASAPPlanner's primary output is `CandidateLogicalPostASAPDAGs`; ranking is a view over its candidates. Downstream owns physical binding and commitment. Selection/DAG assembly helpers do not deploy a plan, and a serializable DAG is not evidence of runtime readiness. @@ -38,9 +38,9 @@ asap-types = { git = "https://github.com/ProjectASAP/ASAPPlanner", rev = "e7fdb2 | Public function | Required input | Output | | --- | --- | --- | -| `asap_frontend_promql::lower_promql_workload` | PromQL `PlanningWorkload` with a nonzero `data_ingestion_interval` | All-or-nothing `Result, PromqlError>` for normalized batch and repeating entries | -| `asap_frontend_metricsql::lower_metricsql` | Query string, `AccuracyTarget` | `Result` | -| `asap_frontend_sql::lower_sql` | Query string, `SqlCatalog`, accuracy | Async `Result`; default SQL dialect is DataFusionSQL | +| `asap_frontend_promql::lower_promql_workload` | PromQL `PlanningWorkload` with a nonzero `data_ingestion_interval` | All-or-nothing `Result, PromqlError>` for normalized batch and repeating entries | +| `asap_frontend_metricsql::lower_metricsql` | Query string, `AccuracyTarget` | `Result` | +| `asap_frontend_sql::lower_sql` | Query string, `SqlCatalog`, accuracy | Async `Result`; default SQL dialect is DataFusionSQL | | `asap_frontend_sql::lower_sql_dialect` | Same inputs plus `SqlDialect` | Async resolved Pre-ASAP query or error | | `asap_frontend_sql::lower_sql_batch` | `QueryWorkload` and catalog | Per-query results for `query_batch`; does not iterate `repeating_queries` | @@ -58,7 +58,7 @@ PromQL's public signature (types are imported from their respective crates): ```text lower_promql_workload(workload: &PlanningWorkload, now_ms: u64) - -> Result, PromqlError> + -> Result, PromqlError> ``` `DataWorkload.data_ingestion_interval` must contain a nonzero `Evidence`. @@ -125,9 +125,9 @@ For SQL, the corresponding signatures are: ```text async lower_sql(query: &str, catalog: &SqlCatalog, accuracy: AccuracyTarget) - -> Result + -> Result async lower_sql_dialect(query: &str, catalog: &SqlCatalog, - dialect: SqlDialect, accuracy: AccuracyTarget) -> Result + dialect: SqlDialect, accuracy: AccuracyTarget) -> Result ``` | `SqlDialect` value | Current behavior | @@ -144,7 +144,7 @@ example, see [the CLI frontend example](../../crates/devtools/src/bin/show_pre_a ### Target sub-DAG candidates `TargetSubDAGCandidates` collects alternatives for one query subexpression -discovered by search. `CandidatePostASAPDAGs` contains these per-target candidate sets and +discovered by search. `CandidateLogicalPostASAPDAGs` contains these per-target candidate sets and the workload's query roots. A root is a whole query; an inner expression can also be a target. @@ -162,12 +162,12 @@ It keeps the alternatives available; it does not select an entire workload plan. ```text search_workload_with_targets<'s, Id>( - roots: Vec<(Id, Rc, Option)>, + roots: Vec<(Id, Rc, Option)>, strategies: &[Box], accuracy_model: &dyn AccuracyModel, -) -> CandidatePostASAPDAGs +) -> CandidateLogicalPostASAPDAGs -CandidatePostASAPDAGs::cost_sorted(&self, cost_model: &dyn CostModel) +CandidateLogicalPostASAPDAGs::cost_sorted(&self, cost_model: &dyn CostModel) -> Vec> ``` @@ -181,13 +181,13 @@ CandidatePostASAPDAGs::cost_sorted(&self, cost_model: &dyn CostModel) `search_workload_with_targets` normally rejects candidates without a guarantee that satisfies the root target. One exception is a direct DDSketch quantile -ratio: without input-domain evidence, it remains in `CandidatePostASAPDAGs` with -`guarantee: None` so the downstream backend can decide whether to select it. +ratio: without input-domain evidence, it remains in `CandidateLogicalPostASAPDAGs` with +`guarantee: None` so selection with backend-supplied domain evidence can still consider it. Its presence does **not** mean it satisfies the target. `cost_sorted` still shows it, but `global_selection` skips it and DAG assembly uses the exact fallback -unless a certified alternative is available. A backend that wants the -uncertified candidate must explicitly inspect it and check its own domain -evidence and execution requirements before selecting or deploying it. +unless a certified alternative is available. Using the uncertified candidate +requires the backend to supply domain evidence and execution requirements that +certify it; without them it is not selected. ### Example @@ -254,11 +254,11 @@ fn main() -> Result<(), Box> { | API (`asap_aware_mapping`, unless qualified) | Inputs | Output and limits | | --- | --- | --- | -| `search_workload` | `(query_id, Rc)` roots | `CandidatePostASAPDAGs` with built-in strategies/model; no explicit per-root target argument | -| `search_workload_with` | Roots, strategy slice | `CandidatePostASAPDAGs`; callers choose context-free replacement strategies | +| `search_workload` | `(query_id, Rc)` roots | `CandidateLogicalPostASAPDAGs` with built-in strategies/model; no explicit per-root target argument | +| `search_workload_with` | Roots, strategy slice | `CandidateLogicalPostASAPDAGs`; callers choose context-free replacement strategies | | `search_workload_with_targets` | Roots with optional end-to-end targets, strategies, accuracy model | Candidate space with supplied root-target checks; `None` does not supply a root-level requirement; uncertified direct DDSketch ratios remain available for backend selection | -| `CandidatePostASAPDAGs::cost_sorted` | Cost model | `Vec`; retains alternatives and pairs `candidates[i]` with `costs[i]` | -| `CandidatePostASAPDAGs::cost_sorted_with_recurrence` | Cost model, recurrence profiles, optional horizon | Ranked per-target candidate sets or `RecurrenceError`; uses recurrence for applicable share/recompute comparisons | +| `CandidateLogicalPostASAPDAGs::cost_sorted` | Cost model | `Vec`; retains alternatives and pairs `candidates[i]` with `costs[i]` | +| `CandidateLogicalPostASAPDAGs::cost_sorted_with_recurrence` | Cost model, recurrence profiles, optional horizon | Ranked per-target candidate sets or `RecurrenceError`; uses recurrence for applicable share/recompute comparisons | | `SketchAlgorithmStrategy::replacements` through `ReplacementStrategy` | One `TargetSubDAG` | Alternatives at that target; not whole-workload search | `cost_sorted` is a ranking view, not a request to discard all but the first @@ -270,7 +270,7 @@ before physical selection; do not treat their presence as deployment permission. ### Enumerate candidate DAGs per root ```text -CandidatePostASAPDAGs::enumerate_candidate_dags_for_root(&self, id: &Id, expansion_limit: usize) +CandidateLogicalPostASAPDAGs::enumerate_candidate_dags_for_root(&self, id: &Id, expansion_limit: usize) -> Result, RealizationError> ``` @@ -285,8 +285,8 @@ carrying the complete series identity (`$promql_series_identity`). They are finalized, deduplicated, and marked `ReplacementProvenance::RootPhysicalRealization`. Callers do not apply `with_series_identity` themselves. Compile each with `promql_rows::compile_current_series_readout`; other queries keep their previous -inventory. `global_selection` never commits these candidates; the backend -compiles and prices them. CandidatePostASAPDAGs lists no placement variants: node timing +inventory. `global_selection` never commits these candidates; they are +compiled, and the backend supplies their prices for selection. CandidateLogicalPostASAPDAGs lists no placement variants: node timing comes from the summary maintenance lifecycle. ## Choose strategies and models @@ -525,7 +525,7 @@ Use this workflow when Planner owns summary-maintenance lifecycle decisions; otherwise the backend may make them from logical candidates. It includes both selection and DAG assembly, so callers do not first run the ordinary workflow. The first helper returns one `GlobalSelection`; the second is called per root -and returns a plan containing `root: Rc` plus maintenance decisions. +and returns a plan containing `root: Rc` plus maintenance decisions. See the [workflow design](../design_docs/architecture/input-output-workflow.md#summary-maintenance-lifecycle-aware-helper). Two capabilities are distinct: the runtime can orchestrate a lifecycle, and the @@ -536,16 +536,16 @@ hold. Workload legality and known cost evidence can further restrict alternative ```text global_selection_with_summary_maintenance_lifecycles<'a, Id>( - space: &'a CandidatePostASAPDAGs, demand: WorkloadDemand<'_>, + space: &'a CandidateLogicalPostASAPDAGs, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &dyn CostModel, ) -> Result, SummaryMaintenanceLifecycleSelectionError> assemble_selected_dag_with_summary_maintenance_lifecycles( - selection: &GlobalSelection<'_>, target: &Rc, + selection: &GlobalSelection<'_>, target: &Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &dyn CostModel, -) -> Result, SummaryMaintenanceLifecycleAssemblyError> +) -> Result, SummaryMaintenanceLifecycleAssemblyError> ``` | Argument | Values / requirements | @@ -578,20 +578,20 @@ entries, construct demand using all applicable indices. ```rust use asap_aware_mapping::{ global_selection_with_summary_maintenance_lifecycles, - assemble_selected_dag_with_summary_maintenance_lifecycles, CostModel, Horizon, CandidatePostASAPDAGs, - SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecyclePlan, + assemble_selected_dag_with_summary_maintenance_lifecycles, CostModel, Horizon, CandidateLogicalPostASAPDAGs, + SummaryMaintenanceLifecycleCapabilities, LifecyclePostASAPDAG, WorkloadDemand, }; use asap_types::workload::PlanningWorkload; fn plan_batch_root( - space: &CandidatePostASAPDAGs<&str>, + space: &CandidateLogicalPostASAPDAGs<&str>, workload: &PlanningWorkload, entry_index: usize, now_ms: u64, horizon: Option, model: &dyn CostModel, -) -> Result, Box> { +) -> Result, Box> { if space.roots.len() != 1 { return Err("this example requires exactly one root".into()); } @@ -630,11 +630,12 @@ that prepared or retained shared state is supported. | Function | Inputs | Output / promise | | --- | --- | --- | -| `plan_summary_maintenance_lifecycles` | Assembled logical DAG root, `WorkloadDemand`, `now_ms`, optional horizon, runtime capabilities, cost model | `Result` for that fixed root; does not revisit all semantic candidates | -| `global_selection_with_summary_maintenance_lifecycles` | `CandidatePostASAPDAGs`, workload/root-entry associations, time, horizon, capabilities, cost model | Lifecycle-aware compatible selection/error, using eligible cost evidence | +| `plan_summary_maintenance_lifecycles` | Assembled logical DAG root, `WorkloadDemand`, `now_ms`, optional horizon, runtime capabilities, cost model | `Result` for that fixed root; does not revisit all semantic candidates | +| `global_selection_with_summary_maintenance_lifecycles` | `CandidateLogicalPostASAPDAGs`, workload/root-entry associations, time, horizon, capabilities, cost model | Lifecycle-aware compatible selection/error, using eligible cost evidence | | `assemble_selected_dag_with_summary_maintenance_lifecycles` | Selection, target root and lifecycle context | Optional lifecycle plan/error; attaches state deployment decisions | -| `enumerate_summary_maintenance_lifecycles` | Same inputs as `plan_summary_maintenance_lifecycles` | `SummaryMaintenanceLifecycleCandidates`: per unique retained state, every alternative with its cost or rejection; nothing selected. `guarantee(&lifecycle)` gives the mode/schedule that alternative would carry | -| `SummaryMaintenanceLifecycleCandidates::select(choices)` | One `(PostAsapNodeId, SummaryMaintenanceLifecycle)` per state, copied from `deployments()` | The same `SummaryMaintenanceLifecyclePlan` Planner selection would produce for that combination, or `SummaryMaintenanceLifecycleChoiceError` when a choice is unknown, missing, duplicated, rejected, schedule-incompatible, or not completely estimable | +| `CandidateLogicalPostASAPDAGs::with_timing_for_root` | Root ID, `CandidateTimingContext`, logical and assignment expansion limits | `CandidateLifecyclePostASAPDAGs<'a, Id>`; lazy assignments and rejections, with shared logical graphs and lifecycle metadata | +| Timed collection `lifecycle_guarantee(logical_index, state, lifecycle)` | A state and one of its lifecycle alternatives | The guarantee that choosing it would attach, so the deployment can supply its price; a lifecycle the state does not offer is rejected | +| Timed collection `select_lifecycles(logical_index, choices)` | One `(PostASAPNodeId, SummaryMaintenanceLifecycle)` per state, copied from `lifecycle_alternatives(logical_index)` | The same lifecycle plan and validation as explicit Planner selection, with typed rejection on failure | Inspect `deployments`, their selected lifecycle/alternatives/rejections, `selected_raw_recompute`, and optional summary/raw costs. Success of a function @@ -643,19 +644,21 @@ costed. A raw alternative remains a downstream execution obligation. Lifecycle feasibility and costs must affect final deployment comparison. Running lifecycle analysis after structural selection can evaluate the selected root, -but does not make the earlier selection lifecycle-optimal. An application may -consume ranked candidates and perform this comparison downstream instead. - -A deployment that prices lifecycles itself calls -`enumerate_summary_maintenance_lifecycles`, prices the alternatives, and binds -its choice with `select`. A choice is accepted only if Planner could select it: +but does not make the earlier selection lifecycle-optimal. An application can +instead supply its own prices to that comparison. + +A deployment that supplies its own lifecycle prices obtains the timed candidates +with `with_timing_for_root`; selection over those prices binds one choice per +state, through `select_lifecycles`, which validates a given choice. Callers with an assembled root use +`CandidateLifecyclePostASAPDAGs::from_post_asap_dag` to enter the same timed collection. +A choice is accepted only if Planner could select it: an alternative with `MissingCostEvidence` is accepted only when the cost model's complete-candidate hook covers lifecycle costs. Window frameworks and totals come from that hook, as in Planner selection. A lifecycle choice then fixes each physical placement through timing: a continuously maintained state and its inputs run at ingestion time, while an -ephemeral one stays at query time. Compile each query's `PostAsapDag` once and +ephemeral one stays at query time. Compile each query's `LogicalPostASAPDAGTransport` once and cut every chosen assignment from that result: ```rust @@ -663,12 +666,12 @@ use asap_physical_operators::physical_planner::{ compile, cut_candidate, frontier_from_timing, }; -let compiled = compile(&dag, inputs, &roots)?; // each node lowered once +let compiled = compile(dag.as_view(), inputs, &roots)?; // each node lowered once for plan in lifecycle_plans { - let frontier = frontier_from_timing(&plan.execution_timed_dag()?)?; + let frontier = frontier_from_timing(plan.export_timed_dag()?.as_view())?; // Precompute/query DAGs split at `frontier`; no logical lowering. let candidate = cut_candidate(&compiled, &frontier)?; - // Check feasibility and price `candidate`; bind the selected one as is. + // The deployment prices `candidate`; Planner's selection binds one as is. } ``` @@ -688,7 +691,7 @@ state's input. The lifecycle cost hooks (`summary_maintenance_capabilities`, hook therefore also receive `MaintainPopulation` nodes. A model that does not recognize one should return unknown costs, which keep its alternatives unselected; a model that prices every node uniformly now also prices -populations, so population candidates can win lifecycle-aware selection. `SummaryMaintenanceLifecyclePlan::execution_timed_dag` times a +populations, so population candidates can win lifecycle-aware selection. `LifecyclePostASAPDAG::export_timed_dag` times a population as it times a summary state: retained at ingestion, `Ephemeral` at query time from the raw source. @@ -731,11 +734,11 @@ Plain `global_selection()` does not automatically perform lifecycle planning or establish physical deployment feasibility. Use the corresponding evidence-aware workflow for those decisions. Downstream still owns physical commitment. -| Method on `CandidatePostASAPDAGs` / `GlobalSelection` | Behavior | +| Method on `CandidateLogicalPostASAPDAGs` / `GlobalSelection` | Behavior | | --- | --- | -| `CandidatePostASAPDAGs::global_selection(&model)` | Compatible structural selection across targets; no recurrence or lifecycle planning implied | -| `CandidatePostASAPDAGs::global_selection_with_recurrence(...)` | Compatible selection using supplied recurrence profiles/horizon; no lifecycle commitments implied | -| `GlobalSelection::assemble_selected_dag(&target)` | `Result>, RealizationError>`; constructs semantic IR, not stored summary data | +| `CandidateLogicalPostASAPDAGs::global_selection(&model)` | Compatible structural selection across targets; no recurrence or lifecycle planning implied | +| `CandidateLogicalPostASAPDAGs::global_selection_with_recurrence(...)` | Compatible selection using supplied recurrence profiles/horizon; no lifecycle commitments implied | +| `GlobalSelection::assemble_selected_dag(&target)` | `Result>, RealizationError>`; constructs semantic IR, not stored summary data | Use a target associated with the searched space; DAG assembly can return `None` when that target is absent. A downstream integration can use these convenience @@ -746,9 +749,9 @@ for checking complete physical alternatives and deployment constraints. ### API definition and example ```text -CandidatePostASAPDAGs::global_selection(&self, cost_model: &dyn CostModel) -> GlobalSelection<'_> -GlobalSelection::assemble_selected_dag(&self, target: &Rc) - -> Result>, RealizationError> +CandidateLogicalPostASAPDAGs::global_selection(&self, cost_model: &dyn CostModel) -> GlobalSelection<'_> +GlobalSelection::assemble_selected_dag(&self, target: &Rc) + -> Result>, RealizationError> ``` For structural inspection only, this complete example selects a semantic root @@ -793,7 +796,7 @@ fn main() -> Result<(), Box> { let root = Rc::new(lower_promql_workload(&workload, 0)?.remove(0)); let space = search_workload(vec![("q1", root)]); let selection = space.global_selection(&DefaultCostModel); - // Search may canonicalize roots; use the root returned by CandidatePostASAPDAGs. + // Search may canonicalize roots; use the root returned by CandidateLogicalPostASAPDAGs. if let Some(summary) = selection.assemble_selected_dag(&space.roots[0].1)? { let graph = asap_types::dag_export::export_summary(&summary); println!("{graph:#?}"); @@ -808,8 +811,8 @@ fn main() -> Result<(), Box> { | --- | --- | | `asap_types::dag_export::export(&query)` | Pre-ASAP inspection graph | | `asap_types::dag_export::export_summary(&summary)` | Post-ASAP inspection graph | -| `asap_types::post_asap::compile_post_asap_dag(&root)` | Compile a semantic DAG with execution-data-state validation; not a physical plan | -| `PostAsapDagDocument::new(dag)` and `.validate()` | Versioned semantic envelope and explicit validation; constructing it alone does not validate | +| `asap_types::post_asap::export_post_asap_dag(&root)` | Compile a semantic DAG with execution-data-state validation; not a physical plan | +| `LogicalPostASAPDAGDocument::new(dag)` and `.validate()` | Versioned semantic envelope and explicit validation; constructing it alone does not validate | | `asap_aware_mapping::export_summary_maintenance_plan(&plan)` | Graph plus lifecycle deployments, alternatives and available cost/guarantee information | | `explain_replacements` / `explain_replacements_with` | Findings from default/custom-strategy search; not a complete physical feasibility report | diff --git a/docs/develop_docs/metrics-observability-corpora.md b/docs/develop_docs/metrics-observability-corpora.md index 3dd1d2b8..22c87b68 100644 --- a/docs/develop_docs/metrics-observability-corpora.md +++ b/docs/develop_docs/metrics-observability-corpora.md @@ -49,7 +49,7 @@ o11y-bench, and awesome-prometheus-alerts. They are not duplicated here. The test prints totals, parse errors, lowering errors, pre-ASAP successes, post-ASAP candidates, unchanged queries, and post-ASAP errors. `Pre-ASAP` means -that parsing and lowering produced a `QueryExpr`. `Post-ASAP candidate` means +that parsing and lowering produced a `PreASAPNode`. `Post-ASAP candidate` means the isolated `SketchAlgorithmStrategy` produced a non-`KeepPreAsap` summary candidate. `Unchanged` is a successful pre-ASAP query for which that strategy returned only the pre-ASAP fallback. diff --git a/docs/develop_docs/native-promql-inputs.md b/docs/develop_docs/native-promql-inputs.md index aa1a52b6..2eb8bbe5 100644 --- a/docs/develop_docs/native-promql-inputs.md +++ b/docs/develop_docs/native-promql-inputs.md @@ -44,7 +44,7 @@ that its result satisfies a query's accuracy requirements. Tests cover open-label Rate → CMS/CountSketch heaps, hidden-label round trips, reset and zero-rate cases, snapshot replacement/decrease/expiry/staleness, serialized physical recovery, and resource rejection. These are shared-library -tests, not proof of Backend candidate selection or durable deployment execution. +tests, not proof of candidate selection under Backend prices or durable deployment execution. Spatial heap candidates use the same complete series identity. Planner's `current_series_topk_candidates` explores a CountSketch-with-heap realization diff --git a/docs/develop_docs/offline-sketch-evidence.md b/docs/develop_docs/offline-sketch-evidence.md index cd331e57..68372e0a 100644 --- a/docs/develop_docs/offline-sketch-evidence.md +++ b/docs/develop_docs/offline-sketch-evidence.md @@ -92,7 +92,7 @@ for a single independently instantiated state. It deliberately leaves retention, retirement and read costs unknown. In particular, a point-frequency benchmark read does not price a total-count read, even when both use CMS. A deployment must match readout semantics and supply the missing lifecycle and raw-query evidence -before selecting and pricing a complete physical plan. Never combine these +before Planner selects a complete physical plan under the deployment's prices. Never combine these nanosecond costs with CPU operation counts without explicit calibration. `error` contains offline observed statistics and a query descriptor. Its metric diff --git a/docs/develop_docs/physical-compile-coverage.md b/docs/develop_docs/physical-compile-coverage.md index 616008b8..c4ca599d 100644 --- a/docs/develop_docs/physical-compile-coverage.md +++ b/docs/develop_docs/physical-compile-coverage.md @@ -6,12 +6,13 @@ Audience: developers moving computation from ASAPQuery-backend into ## Contract Logical selection decides what to compute. The maintenance lifecycle sets node -timing. `physical_planner::compile` turns a timed `PostAsapDag` into physical +timing. `physical_planner::compile` turns a timed `LogicalPostASAPDAGTransport` into physical operator DAGs. The backend owns ingestion, panes, storage, stored-state -readout, external exact engines, pricing/selection, and execution scheduling. +readout, external exact engines, its cost model, and execution scheduling; +selection is a Planner function. A backend lowering is *covered* when `compile` accepts the corresponding -`PostAsapDag` node and produces operators with the same result. The backend +`LogicalPostASAPDAGTransport` node and produces operators with the same result. The backend should then pass the timed DAG and its input contracts to `compile`. It should not rebuild operator choices from PromQL text or construct operators itself. @@ -29,7 +30,7 @@ Status values: | # | Backend site | Computation | Planner node | Status at #475 | Notes | |---|---|---|---|---|---| -| 1 | `query_time.rs` `Lower::lower`, `compile_logical` | PromQL AST → `QueryTimeOperator` graph for a native query | `Fallback { QueryExpr }` subtrees plus value payloads | Missing | `compile` lowers `Fallback` only as a raw `Scan` source. | +| 1 | `query_time.rs` `Lower::lower`, `compile_logical` | PromQL AST → `QueryTimeOperator` graph for a native query | `Fallback { PreASAPNode }` subtrees plus value payloads | Missing | `compile` lowers `Fallback` only as a raw `Scan` source. | | 2 | `QueryTimeOperator::Aggregate` (sum/min/max/avg/count) | Grouped value aggregation | `Value::Exact(Aggregate)`; `SummaryAgg{ExactAggregate, Reduce}` over finalized values | Supported | Also `promql_values::compile_aggregate`. | | 3 | `QueryTimeOperator::Sort`, `Limit` (topk, sort, sort_desc) | Ordering and per-group limits | `Value::Sort`, `Value::Limit` | Supported | | | 4 | `QueryTimeOperator::Binary`, `QueryPlanNode::Binary` (vector ⊗ scalar) | Arithmetic with a scalar operand | `Binary` whose operand is `Fallback{PromqlScalarBridge(Literal)}` | Missing | Query-time `Binary` accepts only label-map vector schemas. The literal node has no native binding. | @@ -56,7 +57,7 @@ Status values: | 25 | `raw_dag.rs` weight `Constant` | Unit/constant-weight update | `SummaryAgg` | Missing | `compile_node` requires a column weight. | | 26 | `raw_dag.rs` item `Column` / `Tuple` | Keyed update item | `SummaryAgg{item}` | Supported | `keyed_summary_build`. | | 27 | `raw_dag.rs` item `EntityIdentity` | Series-identity item | `SummaryAgg{item}` | Missing | Needs the series-identity column. | -| 28 | `physical_values.rs` `compile`, `combine` | Translate `QueryTimeOperator` to `promql_values::*`; compose fragments | n/a | Supported | Exists only because of row 1. `CompiledPhysicalDag::compose` is Planner API. | +| 28 | `physical_values.rs` `compile`, `combine` | Translate `QueryTimeOperator` to `promql_values::*`; compose fragments | n/a | Supported | Exists only because of row 1. `PhysicalPostASAPDAG::compose` is Planner API. | | 29 | `query_plan.rs` `compile_native_fragment` (Semi join, Exact aggregate, Sort, Limit, Filter) | Relational value ops | `RelationalJoin`, `Value::*` | Supported | Already calls `compile`. | | 30 | `query_time.rs` `selected_query_time_nodes`, `selected_native_expression`, `selected_aggregate_operator` | Recover operator identity from original PromQL text | Payload variants (`ExactKind::Min`/`Max`, `AggIntent`) | Supported | Payloads already carry the identity. These witnesses are needed only while row 1 remains. | | 31 | Scan, ExactSubquery, CandidateExactSubquery, CurrentSeries ingest, ReadMaterialization, ExternalExact | Storage reads and external engines | Input contracts | Backend | | @@ -76,7 +77,7 @@ Totals after this change: 17 Supported, 4 Partial, 8 Missing, 2 Backend. ## Covered by PromQL fallback compilation -`compile` now lowers a `Fallback{QueryExpr}` node from its typed expression, +`compile` now lowers a `Fallback{PreASAPNode}` node from its typed expression, realized with `promql_rows::with_series_identity`. The deployment supplies the raw rows of the `i`th selector returned by `promql_fallback::raw_series` at `promql_fallback::raw_series_input(node, i)`, with that selector's schema. Each diff --git a/docs/develop_docs/pre-asap-ir.md b/docs/develop_docs/pre-asap-ir.md index 94677516..171d4424 100644 --- a/docs/develop_docs/pre-asap-ir.md +++ b/docs/develop_docs/pre-asap-ir.md @@ -13,7 +13,7 @@ Only operations that are semantically relevant to answering the query and select > Notes: **SQL and PromQL use different schema models**. SQL typically uses a closed schema, where tables, columns, and types are predefined, while PromQL uses an open (schemaless) schema, where metrics and labels can evolve without a fixed table schema. Closed schemas provide stronger structure and validation; open schemas provide greater flexibility and makes it easier to evolve or ingest diverse data, but can require more care around naming conventions, label cardinality, and query consistency. -The pre-ASAP IR is defined using the `QueryExpr` enum. We discuss some of important enum types below. +The pre-ASAP IR is defined using the `PreASAPNode` enum. We discuss some of important enum types below. ## Node index @@ -464,7 +464,7 @@ did. up > 1 ``` -**Fields:** a single unnamed child `QueryExpr` — the wrapped scalar sub-expression. +**Fields:** a single unnamed child `PreASAPNode` — the wrapped scalar sub-expression. ### EvalTimestamp @@ -488,7 +488,7 @@ patterns (`up or vector(0)`). vector(1) ``` -**Fields:** a single unnamed child `QueryExpr` — the scalar-typed expression being promoted to a vector. +**Fields:** a single unnamed child `PreASAPNode` — the scalar-typed expression being promoted to a vector. ### PromqlScalarFromVector @@ -499,7 +499,7 @@ its value (NaN at runtime if the input isn't exactly one series). scalar(up) ``` -**Fields:** a single unnamed child `QueryExpr` — the single-series vector being collapsed to a scalar. +**Fields:** a single unnamed child `PreASAPNode` — the single-series vector being collapsed to a scalar. ### PromqlRelabel diff --git a/docs/develop_docs/target-candidate-api-migration.md b/docs/develop_docs/target-candidate-api-migration.md index e2b6ab82..1da232e6 100644 --- a/docs/develop_docs/target-candidate-api-migration.md +++ b/docs/develop_docs/target-candidate-api-migration.md @@ -6,14 +6,14 @@ are unchanged. #453 separately defines the integration API surface. | Previous name | New name | Meaning | |---|---|---| -| `CandidatePostASAPDAGs::groups()` | `CandidatePostASAPDAGs::target_subdag_candidates()` | Iterate candidate sets, one per target, in discovery order | -| `CandidatePostASAPDAGs::group_for(target)` | `CandidatePostASAPDAGs::candidates_for_target(target)` | Look up one target's candidate set | +| `CandidateLogicalPostASAPDAGs::groups()` | `CandidateLogicalPostASAPDAGs::target_subdag_candidates()` | Iterate candidate sets, one per target, in discovery order | +| `CandidateLogicalPostASAPDAGs::group_for(target)` | `CandidateLogicalPostASAPDAGs::candidates_for_target(target)` | Look up one target's candidate set | | `SelectedGroup` | `TargetSubDAGSelection` | Selected choice and usage information for one target; the choice may be absent | | `GlobalSelection::groups()` | `GlobalSelection::target_selections()` | Iterate decisions, not alternative sets | | `MaterializeSummaryMaintenanceLifecycleError` | `SummaryMaintenanceLifecycleAssemblyError` | Failure assembling a DAG or deriving maintenance decisions | | Error variant `Materialize` | `AssembleDag` | Wrap an underlying `RealizationError` from DAG assembly | | Internal `materialize_inner` / `materialize_residual` | `assemble_target` / `assemble_residual` | Assemble selected nodes, not runtime materialized views | -| Internal assembly cache `materialized` | `assembled_nodes` | Preserve shared `Rc` identity | +| Internal assembly cache `materialized` | `assembled_nodes` | Preserve shared `Rc` identity | Update imports and calls together; old public names are not retained as aliases. Downstream Rust integrations using these symbols must migrate. No serialized @@ -24,5 +24,5 @@ The earlier #445 renames (`TargetSubDAGCandidates`, counterpart) are prerequisites, not additional changes here. The workflow remains one selection call per workload followed by one assembly -call per query root. `SummaryMaintenanceLifecyclePlan` contains the assembled +call per query root. `LifecyclePostASAPDAG` contains the assembled Post-ASAP DAG root plus maintenance decisions; it is not an executable plan. diff --git a/docs/user_guide_docs/run-a-query.md b/docs/user_guide_docs/run-a-query.md index 89c47c42..f2ef8bf7 100644 --- a/docs/user_guide_docs/run-a-query.md +++ b/docs/user_guide_docs/run-a-query.md @@ -104,7 +104,7 @@ and provide your own models. The default strategy generates DDSketch quantile-ratio candidates even when no input-domain evidence is available. Such candidates have `guarantee: None`: they do not claim a certified end-to-end accuracy bound. They also remain -visible in a target-aware `CandidatePostASAPDAGs` so the downstream backend can decide +visible in a target-aware `CandidateLogicalPostASAPDAGs` so the downstream backend can decide whether to select them using its own evidence. Planner's automatic `global_selection` skips them; their presence alone does not show that they meet the requested target. diff --git a/tools/dag-viewer/README.md b/tools/dag-viewer/README.md index f79d7bf1..279cf81e 100644 --- a/tools/dag-viewer/README.md +++ b/tools/dag-viewer/README.md @@ -69,10 +69,10 @@ cargo run -p asap-devtools --bin dag_export -- \ Load the JSON with the page's file picker. `--planner-cost-json` is a complete physical-evidence document: an immutable `evidence_version`, calibration, and -target records containing the exact target `QueryExpr` and comparison scope. +target records containing the exact target `PreASAPNode` and comparison scope. Each exact replacement candidate owns its complete logical-node -`PhysicalNodeEvidence`; summary candidates additionally own their bound -`PhysicalDag`. Candidate-local evidence prevents statistics for one physical +`PhysicalNodeEvidence`; summary candidates additionally own their +`EvidenceBackedPhysicalDag`. Candidate-local evidence prevents statistics for one physical alternative from satisfying another. Candidate matching includes the complete exported plan, including accuracy guarantees, and never uses a hash or strategy name; derived floating constants allow only a one-ULP JSON round-trip tolerance. diff --git a/tools/dag-viewer/viewer.js b/tools/dag-viewer/viewer.js index 62b2e84c..716dc916 100644 --- a/tools/dag-viewer/viewer.js +++ b/tools/dag-viewer/viewer.js @@ -189,7 +189,7 @@ function render() { sidepanel.style.display = 'block'; sideResizeHandle.style.display = 'block'; - renderPrePostAsap(); + renderPrePostASAP(); renderLegend(); } @@ -476,7 +476,7 @@ function hideModeHint() { } // ── Pre/Post-ASAP: one query or two workload-union DAGs ────────────────── -function renderPrePostAsap() { +function renderPrePostASAP() { const chosen = getParticipants(); const selected = chosen.map((i) => queries[i]); renderScopeSummary(selected);