diff --git a/Cargo.lock b/Cargo.lock index b0dd1db9a..67e04c17a 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -329,10 +329,20 @@ dependencies = [ "tokio", ] +[[package]] +name = "asap-frontend-common" +version = "0.1.0" +dependencies = [ + "asap-types", + "serde", + "thiserror 2.0.18", +] + [[package]] name = "asap-frontend-metricsql" version = "0.1.0" dependencies = [ + "asap-frontend-common", "asap-types", "metricsql_parser", "thiserror 2.0.18", @@ -343,6 +353,7 @@ name = "asap-frontend-promql" version = "0.1.0" dependencies = [ "asap-aware-mapping", + "asap-frontend-common", "asap-types", "promql-parser", ] @@ -352,6 +363,7 @@ name = "asap-frontend-sql" version = "0.1.0" dependencies = [ "asap-aware-mapping", + "asap-frontend-common", "asap-sql-function-catalog", "asap-types", "datafusion", @@ -369,6 +381,7 @@ dependencies = [ "asap-frontend-promql", "asap-frontend-sql", "asap-physical-operators", + "asap-planner", "asap-types", "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", "futures", @@ -384,6 +397,7 @@ dependencies = [ "asap-frontend-promql", "asap-types", "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369)", + "chrono", "futures", "regex", "rmp-serde", diff --git a/Cargo.toml b/Cargo.toml index a2b019af8..4f3d44f6e 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -7,6 +7,7 @@ members = [ "crates/frontend-promql", "crates/frontend-metricsql", "crates/metricsql-common-parser-support", + "crates/frontend-common", "crates/frontend-sql", "crates/planner", "crates/devtools", diff --git a/crates/asap-aware-mapping/src/accuracy/allocation.rs b/crates/asap-aware-mapping/src/accuracy/allocation.rs index e06761ec6..77849d9c8 100644 --- a/crates/asap-aware-mapping/src/accuracy/allocation.rs +++ b/crates/asap-aware-mapping/src/accuracy/allocation.rs @@ -23,7 +23,7 @@ pub struct AccuracyAllocation { impl AccuracyAllocation { /// The end-to-end budget left for everything below `layers[0]` — what - /// the inner subtree must satisfy as a whole (it re-splits internally). + /// the inner sub-DAG must satisfy as a whole (it re-splits internally). /// `None` for a single-layer allocation. pub fn inner_target(&self, shape: &CompositionShape) -> Option { let inner = &self.layers[1..]; diff --git a/crates/asap-aware-mapping/src/accuracy/composition.rs b/crates/asap-aware-mapping/src/accuracy/composition.rs index f3355b281..86a8c8ddc 100644 --- a/crates/asap-aware-mapping/src/accuracy/composition.rs +++ b/crates/asap-aware-mapping/src/accuracy/composition.rs @@ -448,9 +448,7 @@ fn composed_provenance( } pub(super) fn exact_operation_rule(operation: &ExactOperation) -> Option { - let ExactOperation::Aggregate { measures, .. } = operation else { - return None; - }; + let ExactOperation::Aggregate { measures, .. } = operation; match measures.as_slice() { [intent] => crate::function_rules::function_rules(intent).map(|rules| rules.accuracy), // The remaining functions are exact over exact samples, but have @@ -1035,8 +1033,8 @@ mod tests { reduction: asap_types::pre_asap::Reduction::PerEntity, measures: vec![intent], output_names: vec![], - having: None, filters: vec![], + having: None, }; assert_eq!( DefaultAccuracyModel.exact_operation_rule(&operation(AggIntent::Rate)), diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/cardinality.rs b/crates/asap-aware-mapping/src/accuracy/estimators/cardinality.rs index 8830be744..c53a3b5a1 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/cardinality.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/cardinality.rs @@ -4,7 +4,7 @@ use super::*; pub(super) fn guarantee( algorithm: &SketchAlgorithm, params: &SketchParams, - query: &SketchQuery, + query: &SketchStatistic, ) -> Option { let (SketchParams::Kmv { k } | SketchParams::Theta { k }) = params else { return None; diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs b/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs index 0f2f37c63..9482c539a 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs @@ -4,7 +4,7 @@ use super::*; pub(super) fn guarantee( algorithm: &SketchAlgorithm, params: &SketchParams, - query: &SketchQuery, + query: &SketchStatistic, ) -> Option { let (SketchParams::Cms { width, depth } | SketchParams::CmsWithHeap { width, depth, .. }) = params @@ -47,11 +47,11 @@ mod tests { let params = default_size_params(SketchAlgorithm::Cms, &c, 0.01, 0.001); let g = DefaultAccuracyModel .local_guarantee( - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Cms, params), GroupingStrategy::default(), ), - &SketchQuery::Cardinality, + &SketchStatistic::Cardinality, ) .unwrap(); assert_eq!(g.metric, ErrorMetric::Frequency); @@ -65,7 +65,7 @@ mod tests { } #[test] - fn heap_readout_retains_frequency_metric() { + fn heap_evaluation_retains_frequency_metric() { use asap_types::post_asap::{GroupingStrategy, SketchKind}; let cms_heap = SketchParams::CmsWithHeap { width: 272, @@ -74,11 +74,11 @@ mod tests { }; let topk_frequency = DefaultAccuracyModel .local_guarantee( - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::CmsWithHeap, cms_heap), GroupingStrategy::default(), ), - &SketchQuery::TopK { k: 10 }, + &SketchStatistic::TopK { k: 10 }, ) .expect("heap sketch still provides per-key frequency intervals"); assert_eq!(topk_frequency.metric, ErrorMetric::Frequency); diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/count_sketch.rs b/crates/asap-aware-mapping/src/accuracy/estimators/count_sketch.rs index 73af6ffc9..c95059637 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/count_sketch.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/count_sketch.rs @@ -4,7 +4,7 @@ use super::*; pub(super) fn guarantee( algorithm: &SketchAlgorithm, params: &SketchParams, - query: &SketchQuery, + query: &SketchStatistic, ) -> Option { let (SketchParams::CountSketch { width, depth } | SketchParams::CountSketchWithHeap { width, depth, .. }) = params @@ -62,11 +62,11 @@ mod tests { let count_sketch = default_size_params(SketchAlgorithm::CountSketch, &intent, 0.01, 0.01); let guarantee = DefaultAccuracyModel .local_guarantee( - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::CountSketch, count_sketch), GroupingStrategy::default(), ), - &SketchQuery::PointCount { + &SketchStatistic::PointCount { key: asap_types::pre_asap::expr_ir::ColumnRef::SampleValue, value: None, }, diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/ddsketch.rs b/crates/asap-aware-mapping/src/accuracy/estimators/ddsketch.rs index 233ec255e..c7be9ed3e 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/ddsketch.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/ddsketch.rs @@ -4,7 +4,7 @@ use super::*; pub(super) fn guarantee( algorithm: &SketchAlgorithm, params: &SketchParams, - query: &SketchQuery, + query: &SketchStatistic, ) -> Option { let SketchParams::DDSketch { alpha } = params else { return None; diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs b/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs index 6a17e1594..f4556446c 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs @@ -1,7 +1,7 @@ //! Estimator-specific confidence for classic HLL's linear-counting branch. //! //! This is conditional on independent uniform bucket hashes and an enforced -//! upper bound on distinct items in the complete readout population (including +//! upper bound on distinct items in the complete evaluation population (including //! all merged panes). It is not an RSE-to-normal conversion or an ERP fit. use super::*; @@ -9,7 +9,7 @@ use super::*; pub(super) fn generic_guarantee( algorithm: &SketchAlgorithm, params: &SketchParams, - query: &SketchQuery, + query: &SketchStatistic, ) -> Option { let SketchParams::Hll { precision } = params else { return None; @@ -56,13 +56,13 @@ impl ClassicHllConfidence { value: self.relative_error, }, failure_probability: ProbabilityExpr::Constant { value: delta }, - provenance: vec![GuaranteeSource::SketchReadout { + provenance: vec![GuaranteeSource::SketchEvaluation { algorithm: "Hll".into(), contract: "classic_hll_linear_counting_collision_bound_v1".into(), params: serde_json::json!({"precision": precision, "max_distinct": self.max_distinct, "relative_error": self.relative_error, "hash_assumption": "independent_uniform_buckets", - "population_scope": "complete_readout_including_merged_panes"}), + "population_scope": "complete_evaluation_including_merged_panes"}), query: "Cardinality".into(), }], }) @@ -187,7 +187,7 @@ mod tests { } } } - /// The model's readout formula matches the actual classic estimator after merge. + /// The model's evaluation formula matches the actual classic estimator after merge. #[test] fn native_classic_estimator_and_merged_registers_use_the_same_contract() { use asap_sketchlib::sketches::hll::{Classic, HyperLogLogP16}; @@ -237,11 +237,11 @@ mod tests { let params = default_size_params(SketchAlgorithm::Hll, &c, 0.01, 0.01); let g = DefaultAccuracyModel .local_guarantee( - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Hll, params), GroupingStrategy::default(), ), - &SketchQuery::Cardinality, + &SketchStatistic::Cardinality, ) .unwrap(); assert_eq!(g.metric, ErrorMetric::Cardinality); diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs b/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs index 005d34ad2..4f880894d 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs @@ -4,7 +4,7 @@ use super::*; pub(super) fn guarantee( algorithm: &SketchAlgorithm, params: &SketchParams, - query: &SketchQuery, + query: &SketchStatistic, ) -> Option { let SketchParams::Kll { k } = params else { return None; @@ -49,11 +49,11 @@ mod tests { let params = default_size_params(SketchAlgorithm::Kll, &q, 0.01, 0.01); let g = DefaultAccuracyModel .local_guarantee( - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, params), GroupingStrategy::default(), ), - &SketchQuery::Quantile { q: 0.99 }, + &SketchStatistic::Quantile { q: 0.99 }, ) .unwrap(); assert_eq!(g.metric, ErrorMetric::Rank); @@ -69,7 +69,7 @@ mod tests { assert_eq!(g.approximate_layer_count(), 1); assert!(g.provenance.iter().any(|source| matches!( source, - GuaranteeSource::SketchReadout { contract, .. } + GuaranteeSource::SketchEvaluation { contract, .. } if contract == "apache_datasketches_kll_empirical_99_a9b42755072b" ))); } diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs b/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs index df9ef4d9f..0cd5fcb4f 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs @@ -14,7 +14,7 @@ pub mod univmon; pub(super) fn sketch_guarantee( algorithm: &SketchAlgorithm, params: &SketchParams, - query: &SketchQuery, + query: &SketchStatistic, ) -> Option { match params { SketchParams::Kll { .. } => kll::guarantee(algorithm, params, query), @@ -36,7 +36,7 @@ pub(super) fn sketch_guarantee( fn bounded_guarantee( algorithm: &SketchAlgorithm, params: &SketchParams, - query: &SketchQuery, + query: &SketchStatistic, metric: ErrorMetric, bound: f64, delta: ProbabilityExpr, @@ -46,7 +46,7 @@ fn bounded_guarantee( metric, bound: BoundExpr::Constant { value: bound }, failure_probability: delta, - provenance: vec![GuaranteeSource::SketchReadout { + provenance: vec![GuaranteeSource::SketchEvaluation { algorithm: format!("{algorithm:?}"), contract: contract.into(), params: serde_json::to_value(params).unwrap_or(serde_json::Value::Null), @@ -56,21 +56,19 @@ fn bounded_guarantee( } pub(super) fn local_guarantee( - family: &SummaryFamilyType, - query: &SketchQuery, + family: &FieldDataType, + query: &SketchStatistic, ) -> Option { match family { - SummaryFamilyType::Plain(_) => Some(ResultGuarantee::exact("Plain value")), - SummaryFamilyType::ExactAggregate(kind, _) => { + FieldDataType::Plain(_) => Some(ResultGuarantee::exact("Plain value")), + FieldDataType::ExactAggregate(kind, _) => { Some(ResultGuarantee::exact(format!("ExactAggregate({kind:?})"))) } - SummaryFamilyType::Sketch(kind, _) => { - sketch_guarantee(kind.algorithm(), kind.params(), query) - } + FieldDataType::Sketch(kind, _) => sketch_guarantee(kind.algorithm(), kind.params(), query), // No error model is registered for these families. - SummaryFamilyType::Sample(..) - | SummaryFamilyType::Wavelet(..) - | SummaryFamilyType::StatModel(..) => None, + FieldDataType::Sample(..) | FieldDataType::Wavelet(..) | FieldDataType::StatModel(..) => { + None + } } } pub(crate) fn size_params( @@ -171,9 +169,9 @@ impl<'a> EstimatorAccuracy<'a> { fn hll(&self) -> Option { let EstimatorContract::ClassicHll { - max_distinct_per_readout, + max_distinct_per_evaluation, } = self.contract?; - hll::ClassicHllConfidence::new(max_distinct_per_readout, self.epsilon) + hll::ClassicHllConfidence::new(max_distinct_per_evaluation, self.epsilon) } pub(crate) fn size_params(&self, algorithm: &SketchAlgorithm) -> Option { @@ -197,10 +195,10 @@ impl AccuracyModel for EstimatorAccuracy<'_> { } fn local_guarantee( &self, - family: &SummaryFamilyType, - query: &SketchQuery, + family: &FieldDataType, + query: &SketchStatistic, ) -> Option { - if let (Some(_), SummaryFamilyType::Sketch(kind, grouping), SketchQuery::Cardinality) = + if let (Some(_), FieldDataType::Sketch(kind, grouping), SketchStatistic::Cardinality) = (self.contract, family, query) { if let (SketchAlgorithm::Hll, SketchParams::Hll { precision }) = diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/univmon.rs b/crates/asap-aware-mapping/src/accuracy/estimators/univmon.rs index e832fd1a1..6a89a105e 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/univmon.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/univmon.rs @@ -1,8 +1,8 @@ -//! UnivMon currently certifies only its exact unit-update total readout. +//! UnivMon currently certifies only its exact unit-update total evaluation. use super::*; -pub(super) fn guarantee(query: &SketchQuery) -> Option { - matches!(query, SketchQuery::PointCount { value: None, .. }) +pub(super) fn guarantee(query: &SketchStatistic) -> Option { + matches!(query, SketchStatistic::PointCount { value: None, .. }) .then(|| ResultGuarantee::exact("univmon_unit_update_total")) } diff --git a/crates/asap-aware-mapping/src/accuracy/evidence.rs b/crates/asap-aware-mapping/src/accuracy/evidence.rs index e73a1fc84..f779d5782 100644 --- a/crates/asap-aware-mapping/src/accuracy/evidence.rs +++ b/crates/asap-aware-mapping/src/accuracy/evidence.rs @@ -2,12 +2,12 @@ use super::*; /// A trusted source assertion scoped by `AccuracyEvidenceProvider` to one -/// complete readout. Choosing this variant asserts the estimator and hash +/// complete evaluation. Choosing this variant asserts the estimator and hash /// assumptions; it must not be inferred from sampled population statistics. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum EstimatorContract { /// Classic HLL with independent uniform bucket hashing, including merged panes. - ClassicHll { max_distinct_per_readout: u32 }, + ClassicHll { max_distinct_per_evaluation: u32 }, } /// An enforced domain for every sample of a direct quantile operand, in every @@ -18,7 +18,7 @@ pub enum EstimatorContract { pub struct QuantileInputDomain { pub lower: f64, pub upper: f64, - /// Upper bound on samples per evaluation, matching the pinned readout's + /// Upper bound on samples per evaluation, matching the pinned evaluation's /// exact Float64 rank limit. The population must also be nonempty. pub max_samples: u64, pub contract: String, @@ -87,39 +87,30 @@ pub struct PropagationStats { /// Supplies typed planning-time evidence required by propagation rules. pub trait AccuracyEvidenceProvider { /// Trusted estimator contract for this complete aggregate expression, - /// including source, filters, grouping and all panes in each readout. + /// including source, filters, grouping and all panes in each evaluation. /// An observed cardinality is not an enforced population bound. - fn estimator_contract( - &self, - _expression: &asap_types::pre_asap::QueryExpr, - ) -> Option { + fn estimator_contract(&self, _expression: &OperatorNode) -> Option { None } /// Enforced upper bound on distinct (partition, item) identities across a - /// complete TopK readout. Used to union-bound score errors for adaptively + /// complete TopK evaluation. Used to union-bound score errors for adaptively /// selected candidates. Observed cardinality is not sufficient evidence. - fn topk_max_distinct_items( - &self, - _expression: &asap_types::pre_asap::QueryExpr, - ) -> Option { + fn topk_max_distinct_items(&self, _expression: &OperatorNode) -> Option { None } /// Proof scoped to this complete quantile expression, including its source, /// filters, grouping and window. `None` means unknown, including emptiness. - fn quantile_input_domain( - &self, - _operand: &asap_types::pre_asap::query_expr::QueryExpr, - ) -> Option { + fn quantile_input_domain(&self, _operand: &OperatorNode) -> Option { None } fn propagation_stats( &self, _op: &CompositionOperator, - _family: &SummaryFamilyType, - _query: Option<&SketchQuery>, + _family: &FieldDataType, + _query: Option<&SketchStatistic>, ) -> PropagationStats { PropagationStats::default() } @@ -142,8 +133,8 @@ impl AccuracyEvidenceProvider for WorkloadAccuracyEvidence<'_> { fn propagation_stats( &self, _op: &CompositionOperator, - _family: &SummaryFamilyType, - _query: Option<&SketchQuery>, + _family: &FieldDataType, + _query: Option<&SketchStatistic>, ) -> PropagationStats { PropagationStats { input_row_count: self.data.input_cardinality.value_at(self.now_ms).copied(), @@ -180,7 +171,7 @@ mod tests { }; let fresh = provider.propagation_stats( &CompositionOperator::ExactSum, - &SummaryFamilyType::ExactAggregate( + &FieldDataType::ExactAggregate( asap_types::post_asap::ExactKind::Sum, asap_types::post_asap::ExactParams::Sum, ), @@ -195,7 +186,7 @@ mod tests { } .propagation_stats( &CompositionOperator::ExactSum, - &SummaryFamilyType::ExactAggregate( + &FieldDataType::ExactAggregate( asap_types::post_asap::ExactKind::Sum, asap_types::post_asap::ExactParams::Sum, ), diff --git a/crates/asap-aware-mapping/src/accuracy/mod.rs b/crates/asap-aware-mapping/src/accuracy/mod.rs index 9eef3315d..a2f3354c0 100644 --- a/crates/asap-aware-mapping/src/accuracy/mod.rs +++ b/crates/asap-aware-mapping/src/accuracy/mod.rs @@ -20,18 +20,20 @@ pub use evidence::{ QuantileInputDomain, WorkloadAccuracyEvidence, }; +use asap_types::ir::OperatorNode; use asap_types::post_asap::{ - AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ExactOperation, GuaranteeSource, - ProbabilityExpr, ResultGuarantee, SketchAlgorithm, SketchParams, SketchQuery, - SummaryFamilyType, + AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, FieldDataType, GuaranteeSource, + ProbabilityExpr, ResultGuarantee, SketchAlgorithm, SketchParams, SketchStatistic, }; use asap_types::types::AccuracyTarget; +use crate::exact_composition::ExactOperation; + /// The deployment-extensible accuracy algebra. `asap-aware-mapping` ships /// [`DefaultAccuracyModel`]; a deployment with a proof for a composition the /// default rejects (a registered cross-metric conversion, say) implements /// this trait and passes it to -/// [`crate::replacement::SketchAlgorithmStrategy::new_with_planning_inputs`]. +/// [`crate::replacement::ASAPStrategies::new_with_planning_inputs`]. pub trait AccuracyModel { /// The definition-registered rule for applying `operation` to an /// approximate input. `None` means the function is exact only over exact @@ -48,8 +50,8 @@ pub trait AccuracyModel { /// `Sample`/`Wavelet`/`StatModel`). fn local_guarantee( &self, - family: &SummaryFamilyType, - query: &SketchQuery, + family: &FieldDataType, + query: &SketchStatistic, ) -> Option; /// Compose `inputs`' guarantees (in the parent's child order) with the @@ -78,11 +80,11 @@ pub struct DefaultAccuracyModel; const SATISFACTION_TOLERANCE: f64 = 1e-9; impl DefaultAccuracyModel { - /// Derive the guarantee for the committed estimator parameters and readout. + /// Derive the guarantee for the committed estimator parameters and evaluation. pub fn sketch_guarantee( algorithm: &SketchAlgorithm, params: &SketchParams, - query: &SketchQuery, + query: &SketchStatistic, ) -> Option { estimators::sketch_guarantee(algorithm, params, query) } @@ -94,8 +96,8 @@ impl AccuracyModel for DefaultAccuracyModel { } fn local_guarantee( &self, - family: &SummaryFamilyType, - query: &SketchQuery, + family: &FieldDataType, + query: &SketchStatistic, ) -> Option { estimators::local_guarantee(family, query) } diff --git a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs index 46cbc06f5..9982253e3 100644 --- a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs +++ b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs @@ -3,20 +3,20 @@ //! //! ## The gap this closes //! -//! `asap_types::pre_asap::cse::share_common_subtrees` (pre-ASAP CSE) only -//! ever merges two subtrees that are *exactly* [`PartialEq`]-equal, +//! `asap_types::pre_asap::cse::share_common_subdags` (pre-ASAP CSE) only +//! ever merges two sub-DAGs that are *exactly* [`PartialEq`]-equal, //! including their [`AggIntent`]'s `accuracy: AccuracyTarget` field. Two //! otherwise-identical aggregates that differ *only* in how tight an //! accuracy bound they ask for — `quantile(0.99, x)` at `epsilon=0.01` for //! one consumer, the same `quantile(0.99, x)` at `epsilon=0.05` for //! another — are therefore never the same `Rc`, never collapse into one -//! [`crate::replacement::TargetSubDAGCandidates`], and [`crate::replacement::SharedSubtreeStrategy`] +//! [`crate::replacement::TargetSubDAGCandidates`], and [`crate::replacement::SharedSubDagStrategy`] //! never even gets a `TargetSubDAG` with `consumer_count >= 2` to propose //! sharing for. This crate would build two entirely independent sketches //! for what is conceptually one computation, even though a single sketch //! built to the tighter of the two bounds would answer both. //! -//! This module is **additive**, not a relaxation of `share_common_subtrees` +//! This module is **additive**, not a relaxation of `share_common_subdags` //! itself: `accuracy` still participates in exact structural equality //! everywhere else in this crate (correctness elsewhere — e.g. a downstream //! consumer that pattern-matches on a specific `AccuracyTarget` — depends on @@ -24,14 +24,14 @@ //! enough to share" that sits entirely inside the [`ReplacementStrategy`] //! extension point: one more candidate a [`crate::cost_model::CostModel`] //! may or may not prefer, never a forced rewrite and never a change to what -//! `share_common_subtrees` itself merges. +//! `share_common_subdags` itself merges. //! //! ## What counts as a "near-duplicate", and why //! -//! Two [`QueryExpr::Aggregate`] nodes are accuracy-near-duplicates here iff, +//! Two `NonASAPOp::Aggregate` nodes are accuracy-near-duplicates here iff, //! **in this order**: //! -//! 1. Both are the same bindable shape [`crate::replacement::SketchAlgorithmStrategy`] +//! 1. Both are the same bindable shape [`crate::replacement::ASAPStrategies`] //! itself targets — a single measure, no `HAVING` (`bindable_intent`'s own //! scope) — **and** that one measure is one of the four accuracy-bearing //! [`AggIntent`] variants ([`crate::replacement::accuracy_target`]'s own @@ -39,7 +39,7 @@ //! intent has no `AccuracyTarget` to reconcile in the first place. //! 2. Same `reduction` (grouping), same `output_names`, and the same shared //! `child` (`Rc::ptr_eq`, or value-equal for two independently-built but -//! identical subtrees CSE conservatively declined to alias) — the same +//! identical sub-DAGs CSE conservatively declined to alias) — the same //! "identical everything else" bar [`crate::rollup::RollupStrategy`] and //! [`crate::topk_reuse::TopKLimitReuseStrategy`] already hold their own //! sibling-reuse candidates to. @@ -50,7 +50,7 @@ //! trivially "always tightest"). //! 5. The tighter candidate's own **output** schema carries a provable //! unique key (`Schema::has_unique_key`) — the exact legality gate -//! `share_common_subtrees` itself applies (see `cse.rs`'s "Legality" +//! `share_common_subdags` itself applies (see `cse.rs`'s "Legality" //! section) and [`crate::rollup::RollupStrategy::is_legal_rollup_source`] //! already reuses verbatim for the identical reason: a producer's output //! is only safely reusable across a second, independent consumer when @@ -105,7 +105,7 @@ //! //! Like every [`ReplacementStrategy`], this only ever *proposes* — the //! looser-accuracy consumer's own independently-sized candidate (from -//! [`crate::replacement::SketchAlgorithmStrategy`]) stays in its +//! [`crate::replacement::ASAPStrategies`]) stays in its //! [`crate::replacement::TargetSubDAGCandidates`] right alongside this strategy's //! "read the tighter sibling instead" [`Replacement::Rewrite`] candidate; //! [`crate::cost_model::CostModel`]-driven ranking picks between them; @@ -120,12 +120,12 @@ //! tag), because it needs its own cost treatment in //! [`crate::cost_model::DefaultCostModel::estimate_cost`], not just its own //! label. Every other `Replacement::Rewrite` shape that reaches -//! `estimate_cost` (`SharedSubtreeStrategy`'s `CseRecompute`, `Rollup`'s and +//! `estimate_cost` (`SharedSubDagStrategy`'s `CseRecompute`, `Rollup`'s and //! `TopKLimitReuse`'s `LogicalRewrite`) really does rebuild `target` from a //! different source, so pricing it as "one `cse_recompute_cost` of `target` //! itself, per consumer" is the right shape of cost. This strategy's //! candidate never rebuilds `target` at all — it reads `rc` (the tighter -//! sibling), which — per this module's own safety argument — is a subtree +//! sibling), which — per this module's own safety argument — is a sub-DAG //! this crate is already going to build regardless of whether `target` //! reads from it too. Pricing it with the same "rebuild `target`, once per //! consumer" formula would charge it for work it never does, and — because @@ -137,7 +137,7 @@ //! pin against. `estimate_cost` instead prices this shape as a //! [`crate::cost_model::CostModel::cse_shared_maintenance_cost`] read //! against `rc`'s **own** bound summary — the same order-of-magnitude, -//! per-family cost `SharedSubtreeStrategy`'s own `CseShare` candidate is +//! per-family cost `SharedSubDagStrategy`'s own `CseShare` candidate is //! priced with, reflecting "one more reference into a structure that's //! already being maintained" rather than "build a whole new one." //! @@ -149,11 +149,13 @@ //! same structural child, so adding the edge preserves the reference graph's //! parent-before-child topological ordering. +use asap_types::ir::non_asap::any_measure_filtered; use std::cmp::Ordering; use std::rc::Rc; +use asap_types::ir::operator_properties::Reduction; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{any_measure_filtered, QueryExpr, Reduction}; use asap_types::types::AccuracyTarget; use crate::replacement::{ @@ -169,27 +171,27 @@ type BindableAccuracyAggregate<'a> = ( &'a AggIntent, &'a AccuracyTarget, &'a [String], - &'a Rc, + &'a Rc, ); /// The `(reduction, intent, accuracy, output_names, child)` shape this /// module operates on: the same single-measure, no-`HAVING` bindable shape -/// [`crate::replacement::SketchAlgorithmStrategy`] targets (see that +/// [`crate::replacement::ASAPStrategies`] targets (see that /// module's private `bindable_intent`), further narrowed to a measure whose /// intent actually carries an [`AccuracyTarget`] /// ([`crate::replacement::accuracy_target`]'s own scope: `Count` / /// `Quantile` / `Cardinality` / `TopK`). `None` for anything else, including /// a multi-measure or `HAVING` aggregate, a non-`Aggregate` node, or an /// accuracy-free intent (`Sum`, `Avg`, …). -fn bindable_accuracy_aggregate(node: &QueryExpr) -> Option> { - let QueryExpr::Aggregate { +fn bindable_accuracy_aggregate(node: &OperatorNode) -> Option> { + let Some(NonASAPOp::Aggregate { reduction, measures, output_names, filters, having, child, - } = node + }) = node.non_asap() else { return None; }; @@ -278,7 +280,7 @@ fn strictly_tighter(a: &AccuracyTarget, b: &AccuracyTarget) -> bool { /// this strategy from the same post-CSE `Aggregate` sibling set it already /// builds for `RollupStrategy`. pub struct AccuracyReconciliationStrategy { - siblings: Vec>, + siblings: Vec>, } impl AccuracyReconciliationStrategy { @@ -286,7 +288,7 @@ impl AccuracyReconciliationStrategy { /// each as a candidate tighter-accuracy source (or looser-accuracy /// target) — typically the full set of `Aggregate` nodes a workload-wide /// discovery pass already found. - pub fn new(siblings: &[Rc]) -> Self { + pub fn new(siblings: &[Rc]) -> Self { Self { siblings: siblings.to_vec(), } @@ -301,7 +303,7 @@ impl AccuracyReconciliationStrategy { /// /// Also requires the candidate's own *output* schema to carry a provable /// unique key ([`Schema::has_unique_key`]) — the exact legality gate - /// `pre_asap::cse::share_common_subtrees` already applies to its own + /// `pre_asap::cse::share_common_subdags` already applies to its own /// sharing decisions, and [`crate::rollup::RollupStrategy`] already /// reuses verbatim for the identical reason (see that module's /// `is_legal_rollup_source` doc, point 4): a producer's output is only @@ -311,14 +313,14 @@ impl AccuracyReconciliationStrategy { /// reports no unique key — see `cse.rs`'s "Legality" section) would get /// proposed for reconciliation even though nothing guarantees a second /// read of it lines up row-for-row with the first. - fn tighter_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { + fn tighter_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { let Some((target_reduction, target_intent, target_accuracy, target_names, target_child)) = bindable_accuracy_aggregate(target.root) else { return Vec::new(); }; - let mut sources: Vec<&Rc> = self + let mut sources: Vec<&Rc> = self .siblings .iter() .filter(|candidate| { @@ -335,9 +337,7 @@ impl AccuracyReconciliationStrategy { && (Rc::ptr_eq(child, target_child) || child == target_child) && same_intent_except_accuracy(intent, target_intent) && strictly_tighter(accuracy, target_accuracy) - && candidate - .output_schema() - .is_ok_and(|schema| schema.has_unique_key()) + && candidate.schema.has_unique_key() }) .collect(); sources.sort_by(|a, b| { @@ -369,7 +369,7 @@ impl ReplacementStrategy for AccuracyReconciliationStrategy { .expect("tighter_sources only returns bindable_accuracy_aggregate matches"); ReplacementSubDAG { strategy: self.name(), - replacement: Replacement::Rewrite(Rc::clone(source)), + replacement: Replacement::SubDag(Rc::clone(source)), provenance: ReplacementProvenance::AccuracyReconciliation, rationale: format!( "reuses a near-duplicate sibling aggregate — identical intent and grouping \ @@ -391,34 +391,35 @@ impl ReplacementStrategy for AccuracyReconciliationStrategy { mod tests { use super::*; use crate::cost_model::{CostModel, DefaultCostModel}; + use asap_types::ir::cse::share_common_subdags; + use asap_types::ir::operator_properties::{GroupKeys, Source}; use asap_types::post_asap::SketchAlgorithm; - use asap_types::pre_asap::cse::share_common_subtrees; - use asap_types::pre_asap::query_expr::{GroupKeys, Source}; - use asap_types::pre_asap::schema::{Column, ColumnId, DataType, Schema}; + use asap_types::pre_asap::schema::{ColumnId, DataType, Field, Schema}; /// `[ts(0), value(1), job(2)]`. - /// A unique-keyed scan (`[ts]`) so `share_common_subtrees` is actually + /// A unique-keyed scan (`[ts]`) so `share_common_subdags` is actually /// willing to hoist it — see `Schema::has_unique_key`/`cse.rs`'s own /// "Legality" section: a producer with no provable unique key is always /// inserted fresh, never hoisted, regardless of structural equality. - fn metric_scan() -> Rc { - Rc::new(QueryExpr::Scan { + fn metric_scan() -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - Column::new("job", DataType::Utf8, true), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("job", DataType::Utf8, true), ], 0, vec![vec![0]], ), }) + .unwrap() } - fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], @@ -426,9 +427,10 @@ mod tests { having: None, child: Rc::clone(child), }) + .unwrap() } - fn quantile(q: f64, accuracy: AccuracyTarget, child: &Rc) -> Rc { + fn quantile(q: f64, accuracy: AccuracyTarget, child: &Rc) -> Rc { agg( vec![2], AggIntent::Quantile { @@ -441,10 +443,14 @@ mod tests { } /// A globally-grouped (`by(vec![])`) quantile — `aggregate_output_schema` - /// reports no unique key for an empty `by` (see `query_expr.rs`'s own + /// reports no unique key for an empty `by` (see `aggregate_schema.rs`'s own /// `unique_keys = if by.is_empty() || has_count_values { vec![] } else /// { .. }`). - fn global_quantile(q: f64, accuracy: AccuracyTarget, child: &Rc) -> Rc { + fn ungrouped_quantile( + q: f64, + accuracy: AccuracyTarget, + child: &Rc, + ) -> Rc { agg( vec![], AggIntent::Quantile { @@ -463,9 +469,9 @@ mod tests { q: f64, accuracy: AccuracyTarget, excluded: Vec, - child: &Rc, - ) -> Rc { - Rc::new(QueryExpr::Aggregate { + child: &Rc, + ) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(excluded)), measures: vec![AggIntent::Quantile { col: None, @@ -477,6 +483,7 @@ mod tests { having: None, child: Rc::clone(child), }) + .unwrap() } // ── dominates / strictly_tighter ───────────────────────────────────── @@ -548,7 +555,7 @@ mod tests { assert!(strategy.matches(&TargetSubDAG::new(&loose))); let replacements = strategy.replacements(&TargetSubDAG::new(&loose)); assert_eq!(replacements.len(), 1); - let Replacement::Rewrite(rc) = &replacements[0].replacement else { + let Replacement::SubDag(rc) = &replacements[0].replacement else { panic!("expected a Rewrite candidate"); }; assert!(Rc::ptr_eq(rc, &tight)); @@ -580,7 +587,7 @@ mod tests { assert!( loose_group.candidates.iter().any(|candidate| { candidate.strategy == "AccuracyReconciliationStrategy" - && matches!(candidate.replacement, Replacement::Rewrite(_)) + && matches!(candidate.replacement, Replacement::SubDag(_)) }), "expected an AccuracyReconciliationStrategy candidate for the looser consumer, got: \ {:?}", @@ -658,24 +665,24 @@ mod tests { assert!(!strategy.matches(&TargetSubDAG::new(&exact))); } - // ── exact structural equality / share_common_subtrees is unchanged ──── + // ── exact structural equality / share_common_subdags is unchanged ──── #[test] - fn share_common_subtrees_still_never_merges_differing_accuracy() { + fn share_common_subdags_still_never_merges_differing_accuracy() { // The additive guarantee this issue explicitly must not violate: // pre-ASAP CSE's own exact-equality merge stays exact. Two // aggregates differing only in `accuracy` must come back as two // distinct `Rc`s, not one shared `Rc` — AccuracyReconciliationStrategy // is the *only* place cross-accuracy sharing gets proposed, never - // `share_common_subtrees` itself. + // `share_common_subdags` itself. let scan = metric_scan(); - let a = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); - let b = (*quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan)).clone(); + let a = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); + let b = quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); - let roots = share_common_subtrees(vec![("a", a), ("b", b)]); + let roots = share_common_subdags(vec![("a", a), ("b", b)]); assert!( !Rc::ptr_eq(&roots[0].1, &roots[1].1), - "share_common_subtrees must not merge aggregates with different AccuracyTarget" + "share_common_subdags must not merge aggregates with different AccuracyTarget" ); assert_ne!( roots[0].1, roots[1].1, @@ -684,10 +691,10 @@ mod tests { // The identical scan child, though, is still shared exactly as // before — this module changes nothing about that. - let QueryExpr::Aggregate { child: child_a, .. } = roots[0].1.as_ref() else { + let Some(NonASAPOp::Aggregate { child: child_a, .. }) = roots[0].1.non_asap() else { panic!("expected an Aggregate root"); }; - let QueryExpr::Aggregate { child: child_b, .. } = roots[1].1.as_ref() else { + let Some(NonASAPOp::Aggregate { child: child_b, .. }) = roots[1].1.non_asap() else { panic!("expected an Aggregate root"); }; assert!(Rc::ptr_eq(child_a, child_b)); @@ -696,14 +703,14 @@ mod tests { #[test] fn identical_accuracy_still_merges_via_ordinary_cse() { // Sanity check the fixture itself: truly identical aggregates - // (same accuracy too) still merge via share_common_subtrees's own + // (same accuracy too) still merge via share_common_subdags's own // exact equality — unrelated to this module, but pins the contrast // with the test above. let scan = metric_scan(); - let a = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); - let b = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); + let a = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); + let b = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); - let roots = share_common_subtrees(vec![("a", a), ("b", b)]); + let roots = share_common_subdags(vec![("a", a), ("b", b)]); assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); } @@ -714,14 +721,14 @@ mod tests { // `by(vec![])` (global aggregation) reports no unique key // (`aggregate_output_schema`'s own `unique_keys = if by.is_empty() .. // { vec![] } ..`) — the same legality gate - // `share_common_subtrees`/`RollupStrategy` apply, which this + // `share_common_subdags`/`RollupStrategy` apply, which this // strategy must not bypass (module docs, point 5). let scan = metric_scan(); - let tight = global_quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); - let loose = global_quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); + let tight = ungrouped_quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); + let loose = ungrouped_quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); assert!( - !tight.output_schema().unwrap().has_unique_key(), + !tight.schema.clone().has_unique_key(), "fixture sanity: a globally-grouped aggregate has no provable unique key" ); @@ -741,7 +748,7 @@ mod tests { let loose = without_quantile(0.99, AccuracyTarget::Epsilon(0.05), vec![2], &scan); assert!( - !tight.output_schema().unwrap().has_unique_key(), + !tight.schema.clone().has_unique_key(), "fixture sanity: a without(...) aggregate has no provable unique key" ); @@ -831,7 +838,7 @@ mod tests { "global_selection must commit to some candidate for a single-consumer looser target" ); // With no recompute term at all (it never rebuilds `target`), this - // candidate strictly undercuts every SketchAlgorithmStrategy + // candidate strictly undercuts every ASAPStrategies // candidate (which each pay a recompute term on top of their own // maintenance term) under DefaultCostModel's numbers — the sane // direction: reading an already-necessary sibling should be able to @@ -848,7 +855,7 @@ mod tests { // independently-built but structurally identical loose queries // merge onto one Rc via ordinary CSE), *and* a separate, // single-consumer tight sibling exists over the same input — the - // scenario the issue itself targets: `SharedSubtreeStrategy`'s own + // scenario the issue itself targets: `SharedSubDagStrategy`'s own // CseShare/CseRecompute pair is on the table for the loose target's // own 2 consumers at the same time as this strategy's "read the // tight sibling instead" candidate. diff --git a/crates/asap-aware-mapping/src/analytical_cost.rs b/crates/asap-aware-mapping/src/analytical_cost.rs index ed2a316db..6cd7dce67 100644 --- a/crates/asap-aware-mapping/src/analytical_cost.rs +++ b/crates/asap-aware-mapping/src/analytical_cost.rs @@ -2969,7 +2969,7 @@ mod tests { } fn comparison_scope() -> ComparisonScope { - use asap_types::pre_asap::query_expr::Source; + use asap_types::ir::operator_properties::Source; use asap_types::workload::{ DurationMs, QueryRecurrence, QueryTimeScope, RepeatedDemand, RepetitionInterval, TimeSelection, TimestampMs, @@ -3008,9 +3008,7 @@ mod tests { #[test] fn comparison_rejects_different_snapshot_predicate_time_or_horizon() { - use std::rc::Rc; - - use asap_types::pre_asap::query_expr::{Predicate, QueryExpr}; + use asap_types::ir::{Predicate, ScalarExpr}; use asap_types::workload::{DurationMs, TimestampMs}; let raw = comparison_scope(); @@ -3026,7 +3024,7 @@ mod tests { candidate = raw.clone(); candidate.sources[0] .predicates - .push(Predicate(Rc::new(QueryExpr::promql_scalar(1.0)))); + .push(Predicate(ScalarExpr::literal_f64(1.0))); assert_eq!( validate_comparison_scopes(&raw, &candidate), Err(AnalyticalCostError::ComparisonScopeMismatch("sources")) @@ -3246,7 +3244,7 @@ mod tests { operator: PhysicalOperator::Scan, children: vec![], source_coverage: Some(SourceCoverage { - source: asap_types::pre_asap::query_expr::Source::Table { + source: asap_types::ir::operator_properties::Source::Table { table_ref: "other_metrics".into(), }, source_snapshot_id: "catalog-version-42".into(), @@ -3313,7 +3311,7 @@ mod tests { let mut scope = comparison_scope(); let coverage = scope.sources[0].clone(); scope.sources.push(SourceCoverage { - source: asap_types::pre_asap::query_expr::Source::Table { + source: asap_types::ir::operator_properties::Source::Table { table_ref: "auxiliary".into(), }, source_snapshot_id: "catalog-version-42".into(), diff --git a/crates/asap-aware-mapping/src/cost_model.rs b/crates/asap-aware-mapping/src/cost_model.rs index 0bf7c0d80..47195b4c7 100644 --- a/crates/asap-aware-mapping/src/cost_model.rs +++ b/crates/asap-aware-mapping/src/cost_model.rs @@ -26,7 +26,7 @@ //! than overloading these ones across incompatible `Kind`/`Params` types. //! //! Every entry point that doesn't take an explicit `&dyn CostModel` -//! ([`SketchAlgorithmStrategy::default_cost_model`](crate::replacement::SketchAlgorithmStrategy::default_cost_model), +//! ([`ASAPStrategies::default_cost_model`](crate::replacement::ASAPStrategies::default_cost_model), //! [`search_workload`](crate::replacement::search_workload)) runs against //! [`DefaultCostModel`], so a deployment that never plugs in its own cost //! model keeps today's static-preference-order behavior exactly, byte for @@ -35,8 +35,8 @@ //! ## CSE sharing (issue #237, #223 stage 4) //! //! [`CseCandidate`]/[`ShareDecision`]/[`CostModel::cse_share_decision`] below -//! decide whether a CSE-detected shared subtree -//! ([`asap_types::pre_asap::cse::share_common_subtrees`], issue #223 stages +//! decide whether a CSE-detected shared sub-DAG +//! ([`asap_types::ir::cse::share_common_subdags`], issue #223 stages //! 1-2, PR #235) is actually worth sharing, via a real Volcano/Cascades-style //! cost comparison rather than a fixed rule. See //! `docs/design_docs/cse-cost-model-decision.md` for the full design discussion (why @@ -48,14 +48,14 @@ use std::rc::Rc; +use crate::exact_composition::ExactOperation; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; use asap_types::post_asap::{ - ExactOperation, GroupingStrategy, HydraParams, ResultGuarantee, SketchAlgorithm, SketchParams, - SketchQuery, SummaryExpr, SummaryFamilyType, SummaryMaintenanceLifecycleGuarantee, SummaryNode, - SummaryWindowFramework, + FieldDataType, GroupingStrategy, HydraParams, ResultGuarantee, SketchAlgorithm, SketchParams, + SketchStatistic, SummaryMaintenanceLifecycleGuarantee, SummaryWindowFramework, }; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::pre_asap::expr_ir::ColumnRef; -use asap_types::pre_asap::query_expr::QueryExpr; use asap_types::types::AccuracyTarget; use crate::exact_composition::{ExactComposition, OperationPlacement}; @@ -94,7 +94,7 @@ pub struct CostProvenance { /// operator on the update path must never be handed one. #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] pub struct ValueOperationCapabilities { - /// The runtime can apply an exact operator to summary readouts at + /// The runtime can apply an exact operator to summary evaluations at /// query evaluation time. pub query_time: bool, /// The runtime can apply an exact row transform on the update path, @@ -127,17 +127,17 @@ impl ValueOperationCapabilities { /// composes with. #[derive(Debug, Clone, Copy)] pub struct ExactCompositionCostRequest<'a> { - /// The pre-ASAP target the composed candidate replaces. - pub target: &'a QueryExpr, + /// The target the composed candidate replaces. + pub target: &'a OperatorNode, /// The composition itself — placement, operator, child target. pub composition: &'a ExactComposition, /// For [`OperationPlacement::Read`]: the child target's *selected* - /// summary readout candidate the exact operator consumes. For + /// summary evaluation candidate the exact operator consumes. For /// [`OperationPlacement::Maintenance`]: the maintained summary *above* the /// transform that consumes its output (the `SummaryAgg` this transform /// feeds). Either way, the summary whose maintenance/read cost the /// formula charges. - pub summary: &'a SummaryNode, + pub summary: &'a OperatorNode, /// How many times this site actually runs once ancestors' own choices /// are accounted for (see `CandidateLogicalASAPDAGs::global_selection`). pub effective_consumer_count: usize, @@ -147,11 +147,11 @@ pub struct ExactCompositionCostRequest<'a> { /// optional: **an unknown stays `None` — never a zero** — so a formula /// with a missing input yields no rate at all rather than a spuriously /// cheap one, and global selection then keeps the conservative -/// `KeepPreAsap` behavior. A deployment model that wants defaults supplies +/// keep-as-is behavior. A deployment model that wants defaults supplies /// them explicitly by overriding [`CostModel::exact_composition_cost_inputs`]. #[derive(Debug, Clone, PartialEq)] pub struct ExactCompositionCostInputs { - /// Exact operator cost per row it processes — per readout row for a + /// Exact operator cost per row it processes — per evaluation row for a /// read-time operation, per input row for an maintenance-time operation. pub exact_cost_per_row: Option, /// Rows the exact operator consumes per evaluation (read-time operation) or @@ -161,14 +161,14 @@ pub struct ExactCompositionCostInputs { pub expected_output_rows: Option, /// Cost of one update to the composed-with summary's maintained state. pub summary_maintenance_cost_per_update: Option, - /// Cost of one readout of that summary at evaluation time. + /// Cost of one evaluation of that summary at evaluation time. pub summary_read_cost: Option, /// Update (ingest) events per second reaching this site. pub update_rate: Option, /// Evaluations per second across every consumer of this site. pub evaluation_rate: Option, /// Cost of one full raw recompute of the target from pre-ASAP data — - /// the `KeepPreAsap` baseline's per-evaluation cost. + /// the kept-query baseline's per-evaluation cost. pub raw_recompute_cost: Option, /// Recurring formulas require `CostUnitsPerSecond`; totals yield no rate. pub unit: CostUnit, @@ -266,24 +266,24 @@ fn finite_rate(units_per_second: f64) -> Option { .then_some(CostRate(units_per_second)) } -/// A CSE-detected, legality-gated shared subtree with two or more consumers +/// A CSE-detected, legality-gated shared sub-DAG with two or more consumers /// — the unit [`CostModel::cse_share_decision`] decides over. Built by /// [`CandidateLogicalASAPDAGs::cost_sorted`](crate::replacement::CandidateLogicalASAPDAGs::cost_sorted) /// (via [`crate::replacement`]'s own `cse_preference`) the first time it -/// needs a representative bound node for a subtree that -/// [`asap_types::pre_asap::cse::share_common_subtrees`] already collapsed +/// needs a representative bound node for a sub-DAG that +/// [`asap_types::ir::cse::share_common_subdags`] already collapsed /// onto one `Rc` for two or more workload roots. See /// `docs/design_docs/cse-cost-model-decision.md`. pub struct CseCandidate<'a> { - /// The shared pre-ASAP subtree itself. - pub subtree: &'a QueryExpr, - /// The `SummaryNode` this subtree bound to — gives the cost model the - /// concrete `SummaryFamilyType`/`(kind, params)` actually at stake, not - /// just the pre-ASAP shape. - pub bound_summary: &'a SummaryNode, - /// How many workload roots reference this exact shared subtree, counted + /// The shared sub-DAG itself. + pub subtree: &'a Rc, + /// The node this sub-DAG bound to — gives the cost model the + /// concrete `FieldDataType`/`(kind, params)` actually at stake, not + /// just the logical shape. + pub bound_summary: &'a OperatorNode, + /// How many workload roots reference this exact shared sub-DAG, counted /// once up front over the whole workload (always >= 2 — a candidate is - /// only ever constructed for an actually-shared subtree). + /// only ever constructed for an actually-shared sub-DAG). pub consumer_count: usize, } @@ -302,7 +302,7 @@ pub struct Cost(pub f64); /// node. Node identity is preserved so whole-DAG models can bind per-state /// evidence without relying on traversal order. pub struct CostedSummaryDeployment<'a> { - pub summary: &'a SummaryNode, + pub summary: &'a OperatorNode, pub guarantee: &'a SummaryMaintenanceLifecycleGuarantee, pub selected_cost: Cost, } @@ -359,7 +359,7 @@ impl std::ops::Mul for Cost { /// [`CseCandidate`]. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum ShareDecision { - /// Reuse one bound `SummaryNode` across every consumer. + /// Reuse one bound node across every consumer. Share, /// Bind each occurrence independently — the shared-maintenance cost /// isn't worth it for this candidate. @@ -367,27 +367,27 @@ pub enum ShareDecision { } /// Default [`CostModel::cse_recompute_cost`]: a structural-size proxy — the -/// number of *unique* nodes in `subtree`'s DAG -/// ([`asap_types::pre_asap::cse::dag_node_count`], the same module this +/// number of *unique* nodes in `sub-DAG`'s DAG +/// ([`asap_types::ir::cse::dag_node_count`], the same module this /// candidate's sharing was detected in). Deliberately **not** a raw -/// `serde_json` serialization length: after CSE, `subtree` is generally a +/// `serde_json` serialization length: after CSE, `sub-DAG` is generally a /// DAG, not a tree (a `CseCandidate` only exists because something got /// shared), and a naive full serialization re-serializes — over-counts — -/// any descendant `subtree` already shares internally, once per parent +/// any descendant `sub-DAG` already shares internally, once per parent /// that references it, instead of once for the whole DAG. `dag_node_count` /// dedupes by `Rc` pointer identity, so it charges each unique node's /// contribution exactly once regardless of how many places within -/// `subtree` reference it. Cheap to compute (one pass, no serialization), +/// `sub-DAG` reference it. Cheap to compute (one pass, no serialization), /// and still scales with real structural complexity — a genuinely tiny -/// leaf costs little to recompute, a deep multi-join subtree costs a lot. +/// leaf costs little to recompute, a deep multi-join sub-DAG costs a lot. /// A deployment with real per-row/per-update cost knowledge should /// override [`CostModel::cse_recompute_cost`] instead of relying on this. -pub fn default_cse_recompute_cost(subtree: &QueryExpr) -> Cost { - Cost(asap_types::pre_asap::cse::dag_node_count(subtree) as f64) +pub fn default_cse_recompute_cost(subtree: &Rc) -> Cost { + Cost(asap_types::ir::cse::dag_node_count(subtree) as f64) } /// Default [`CostModel::cse_shared_maintenance_cost`]: a small -/// per-[`SummaryFamilyType`] weight, scaled to the same order of magnitude +/// per-[`FieldDataType`] weight, scaled to the same order of magnitude /// as [`default_cse_recompute_cost`]'s typical output (a small node /// count, not a byte length), reflecting that families differ in how /// expensive they are to keep *continuously updated* for the life of a @@ -398,15 +398,15 @@ pub fn default_cse_recompute_cost(subtree: &QueryExpr) -> Cost { /// deployment with real memory/update-cost numbers should override /// [`CostModel::cse_shared_maintenance_cost`] instead of relying on this /// table. -pub fn default_cse_shared_maintenance_cost(family: &SummaryFamilyType) -> Cost { +pub fn default_cse_shared_maintenance_cost(family: &FieldDataType) -> Cost { const UNIT: f64 = 1.0; let weight = match family { - SummaryFamilyType::Plain(_) => 1.0, - SummaryFamilyType::ExactAggregate(..) => 1.0, - SummaryFamilyType::Sketch(..) => 3.0, - SummaryFamilyType::Sample(..) => 3.0, - SummaryFamilyType::Wavelet(..) => 5.0, - SummaryFamilyType::StatModel(..) => 6.0, + FieldDataType::Plain(_) => 1.0, + FieldDataType::ExactAggregate(..) => 1.0, + FieldDataType::Sketch(..) => 3.0, + FieldDataType::Sample(..) => 3.0, + FieldDataType::Wavelet(..) => 5.0, + FieldDataType::StatModel(..) => 6.0, }; Cost(weight * UNIT) } @@ -506,7 +506,7 @@ pub trait CostModel { /// Estimated number of distinct subpopulations produced by `target`'s /// grouping keys. `None` means the deployment has no cardinality estimate; /// grouping alternatives remain legal but keep their discovery order. - fn estimated_subpopulation_count(&self, _target: &QueryExpr) -> Option { + fn estimated_subpopulation_count(&self, _target: &OperatorNode) -> Option { None } @@ -518,7 +518,7 @@ pub trait CostModel { candidate: &ReplacementSubDAG, target: &TargetSubDAG<'_>, ) -> Option { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDag(node) = &candidate.replacement else { return None; }; let (kind, grouping) = sketch_state(node)?; @@ -545,29 +545,29 @@ pub trait CostModel { Realization::PassThrough } - /// Build the `SummaryEstimate` readout for an `Extension` intent this + /// Build the `SummaryEstimate` evaluation for an `Extension` intent this /// same `CostModel` realized as `Realization::Sketch` via /// [`realize_extension`](Self::realize_extension). Only ever called /// when `realize_extension` returned `Sketch` for the same - /// `(ext_kind, payload)` — `replacement::readout` has no other way to build a - /// `SketchQuery` for a shape core doesn't know. A deployment that + /// `(ext_kind, payload)` — `replacement::evaluation` has no other way to build a + /// `SketchStatistic` for a shape core doesn't know. A deployment that /// overrides `realize_extension` to return `Sketch` for some /// `ext_kind` MUST also override this for that same `ext_kind`, or /// this default panics loudly (rather than silently misinterpreting /// `payload`) the first time that intent is actually read out. - fn readout_extension( + fn evaluation_extension( &self, ext_kind: &str, _payload: &serde_json::Value, _col: &ColumnRef, - ) -> SketchQuery { + ) -> SketchStatistic { unimplemented!( "CostModel::realize_extension returned Sketch for ext_kind={ext_kind:?} but \ - readout_extension wasn't overridden to match" + evaluation_extension wasn't overridden to match" ) } - /// Estimate the one-time cost of recomputing `candidate.subtree` + /// Estimate the one-time cost of recomputing `candidate.sub-DAG` /// independently at a single use site. Default: /// [`default_cse_recompute_cost`] (a structural-size proxy). See /// `docs/design_docs/cse-cost-model-decision.md`. @@ -581,7 +581,7 @@ pub trait CostModel { /// weight table), applied to whichever field of /// `candidate.bound_summary`'s output schema actually carries summary /// state (falls back to the cheapest, `Plain`, weight if none does — - /// e.g. `bound_summary` is a passthrough `KeepPreAsap` node with nothing + /// e.g. `bound_summary` is a kept non-ASAP sub-DAG with nothing /// summary-shaped to maintain). See `docs/design_docs/cse-cost-model-decision.md`. fn cse_shared_maintenance_cost(&self, candidate: &CseCandidate) -> Cost { let family = candidate @@ -590,15 +590,15 @@ pub trait CostModel { .fields .iter() .map(|f| &f.dtype) - .find(|dtype| !matches!(dtype, SummaryFamilyType::Plain(_))) + .find(|dtype| !matches!(dtype, FieldDataType::Plain(_))) .cloned() - .unwrap_or(SummaryFamilyType::Plain( + .unwrap_or(FieldDataType::Plain( asap_types::pre_asap::DataType::Float64, )); default_cse_shared_maintenance_cost(&family) } - /// Decide whether to reuse one shared `SummaryNode` across every + /// Decide whether to reuse one shared node across every /// consumer of `candidate`, or bind each occurrence independently — a /// Volcano/Cascades-style cost comparison (issue #237, #223 stage 4; see /// `docs/design_docs/cse-cost-model-decision.md`): share iff the estimated cost of @@ -666,7 +666,7 @@ pub trait CostModel { Cost(1.0) } - /// Cost of recomputing `candidate.subtree` once, from the pre-ASAP/raw + /// Cost of recomputing `candidate.sub-DAG` once, from the pre-ASAP/raw /// path. Units: cost units per recomputation — the `raw_recompute_cost` /// term of `recompute_cost_rate`. Default: delegates to /// [`cse_recompute_cost`](Self::cse_recompute_cost) (the same @@ -738,21 +738,21 @@ pub trait CostModel { /// on [`CandidateLogicalASAPDAGs::cost_sorted`](crate::replacement::CandidateLogicalASAPDAGs::cost_sorted)), /// not just order candidates against each other — that ordering job /// already belongs to [`rank_candidates`](Self::rank_candidates) (for a - /// [`SketchAlgorithmStrategy`](crate::replacement::SketchAlgorithmStrategy) + /// [`ASAPStrategies`](crate::replacement::ASAPStrategies) /// group) and [`cse_share_decision`](Self::cse_share_decision) (for a - /// [`SharedSubtreeStrategy`](crate::replacement::SharedSubtreeStrategy) + /// [`SharedSubDagStrategy`](crate::replacement::SharedSubDagStrategy) /// group). /// /// One method covers both candidate shapes this crate ships: - /// `candidate.replacement`'s [`Replacement::Summary`] arm (a - /// `SketchAlgorithmStrategy` candidate — the bound `SummaryNode` is right - /// there, nothing to reconstruct) and its [`Replacement::Rewrite`] arm - /// (a `SharedSubtreeStrategy` share-vs-recompute candidate — no bound - /// `SummaryNode` of its own, since sharing is a decision about a target + /// `candidate.replacement`'s [`Replacement::SubDag`] from a summary + /// realization (a `ASAPStrategies` candidate — the bound node is + /// right there, nothing to reconstruct) and the same arm from a rewrite + /// (a `SharedSubDagStrategy` share-vs-recompute candidate — no bound + /// summary of its own, since sharing is a decision about a target /// already bound some other way; a representative binding is recovered /// from `target` itself). `target` is threaded through explicitly /// (rather than only ever the target embedded in `candidate` — there - /// isn't one for a `Rewrite`) so both arms have the `consumer_count` + /// isn't one for a rewrite) so both arms have the `consumer_count` /// context a cost estimate needs to be meaningful. /// /// Default: **not a real cost model** — always returns `f64::NAN`. @@ -776,7 +776,7 @@ pub trait CostModel { /// optimistic zeroes. fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs::default() } @@ -786,7 +786,7 @@ pub trait CostModel { /// rate so the horizon integral equals one peak-capacity charge. fn summary_maintenance_lifecycle_cost_inputs_for_horizon( &self, - summary: &SummaryNode, + summary: &OperatorNode, _horizon: Option, ) -> SummaryMaintenanceLifecycleCostInputs { self.summary_maintenance_lifecycle_cost_inputs(summary) @@ -796,7 +796,7 @@ pub trait CostModel { /// conservative default advertises no long-lived maintenance capability. fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities::default() } @@ -807,8 +807,8 @@ pub trait CostModel { /// must not then reuse the partial per-state sum. fn complete_summary_candidate_cost( &self, - _root: &SummaryNode, - _target: Option<&QueryExpr>, + _root: &OperatorNode, + _target: Option<&OperatorNode>, deployments: &[CostedSummaryDeployment<'_>], _horizon: Option, _expected_reads: Option, @@ -827,8 +827,8 @@ pub trait CostModel { /// that do not perform either decision. fn complete_summary_candidate_estimate( &self, - root: &SummaryNode, - target: Option<&QueryExpr>, + root: &OperatorNode, + target: Option<&OperatorNode>, deployments: &[CostedSummaryDeployment<'_>], horizon: Option, expected_reads: Option, @@ -861,7 +861,7 @@ pub trait CostModel { /// Cost of evaluating `target` directly from its logical/raw inputs once. /// When known, lifecycle-aware materialization compares this fallback with /// the aggregate cost of the selected summary deployments. - fn raw_query_recompute_cost(&self, _target: &QueryExpr) -> Option { + fn raw_query_recompute_cost(&self, _target: &OperatorNode) -> Option { None } @@ -870,7 +870,7 @@ pub trait CostModel { /// cardinality changes between evaluations. fn raw_query_recompute_total_cost( &self, - target: &QueryExpr, + target: &OperatorNode, expected_reads: f64, ) -> Option { self.raw_query_recompute_cost(target) @@ -879,7 +879,7 @@ pub trait CostModel { /// Physical feasibility evidence for a complete summary candidate. /// `None` defers admission to physical/deployment compilation; `Some(false)` /// excludes the candidate without changing its computation or parameters. - fn summary_support_evidence(&self, _summary: &SummaryNode) -> Option { + fn summary_support_evidence(&self, _summary: &OperatorNode) -> Option { None } @@ -932,7 +932,7 @@ pub trait CostModel { /// /// Default: every input unknown ([`ExactCompositionCostInputs::unknown`]) /// — unknown is never zero, and with no rate derivable - /// `CandidateLogicalASAPDAGs::global_selection` keeps the conservative `KeepPreAsap` + /// `CandidateLogicalASAPDAGs::global_selection` keeps the conservative keep-as-is /// behavior for the site. A deployment that wants defaults must supply /// them here explicitly. fn exact_composition_cost_inputs( @@ -948,14 +948,16 @@ pub trait CostModel { } fn sketch_state( - node: &SummaryNode, + node: &OperatorNode, ) -> Option<(&asap_types::post_asap::SketchKind, &GroupingStrategy)> { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_state(summary_input), - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, grouping), + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + sketch_state(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, grouping), .. - } => Some((kind, grouping)), + }) => Some((kind, grouping)), _ => None, } } @@ -1027,15 +1029,17 @@ impl CostModel for DefaultCostModel { /// `cse_share_decision`'s default body already composes — rather than a /// second formula: /// - /// - [`Replacement::Summary`]: `cse_recompute_cost` (the one-time + /// - A [`ReplacementProvenance::SummaryRealization`] candidate (a + /// `ASAPStrategies` binding): `cse_recompute_cost` (the one-time /// structural cost of building `target` at all) plus /// `cse_shared_maintenance_cost` of the candidate's own bound family /// (a pricier family — a sketch over an exact accumulator, say — /// costs more here, consistent with the per-family weighting /// [`default_cse_shared_maintenance_cost`] already orders candidates /// by). - /// - [`Replacement::Rewrite`]: recovers one representative bound - /// `SummaryNode` for `target` via `realize_child` (the same + /// - Any other [`Replacement::SubDag`] (a logical rewrite or a CSE + /// share/recompute candidate): recovers one + /// representative bound node for `target` via `realize_child` (the same /// rank-and-take-first helper `replacement::realize_child` reuses for the /// identical need), then charges /// `cse_shared_maintenance_cost` for the candidate that shares @@ -1060,7 +1064,9 @@ impl CostModel for DefaultCostModel { fn estimate_cost(&self, candidate: &ReplacementSubDAG, target: &TargetSubDAG<'_>) -> f64 { let consumer_count = target.consumer_count.max(1); match &candidate.replacement { - Replacement::Summary(node) => { + Replacement::SubDag(node) + if candidate.provenance == ReplacementProvenance::SummaryRealization => + { let cse = CseCandidate { subtree: target.root, bound_summary: node, @@ -1068,7 +1074,7 @@ impl CostModel for DefaultCostModel { }; (self.cse_recompute_cost(&cse) + self.cse_shared_maintenance_cost(&cse)).0 } - Replacement::Rewrite(rc) + Replacement::SubDag(rc) if candidate.provenance == ReplacementProvenance::AccuracyReconciliation => { let Ok(sibling_bound) = realize_child(rc, self) else { @@ -1087,7 +1093,7 @@ impl CostModel for DefaultCostModel { }; self.cse_shared_maintenance_cost(&cse).0 } - Replacement::Rewrite(rc) => { + Replacement::SubDag(rc) => { let Ok(bound) = realize_child(target.root, self) else { return f64::NAN; }; @@ -1320,11 +1326,11 @@ mod tests { assert_eq!( DefaultCostModel.value_operation_support_evidence( &ExactOperation::Aggregate { - reduction: asap_types::pre_asap::query_expr::Reduction::by(vec![]), + reduction: asap_types::ir::operator_properties::Reduction::by(vec![]), measures: vec![AggIntent::Max { col: None }], output_names: vec![], - having: None, filters: vec![], + having: None, }, OperationPlacement::Read, ), @@ -1346,71 +1352,60 @@ mod tests { // ── CSE sharing (issue #237, #223 stage 4) ────────────────────────── + use asap_types::ir::operator_properties::Source; + use asap_types::ir::{NonASAPOp, Predicate, ScalarExpr}; use asap_types::post_asap::{ - ExactKind, ExactParams, GroupingStrategy, SketchKind, SummaryExpr, SummaryField, - SummarySchema, + ExactKind, ExactParams, Field, GroupingStrategy, Schema, SketchKind, }; - use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{Column, DataType, Schema}; + use asap_types::pre_asap::schema::DataType; - fn scan() -> QueryExpr { - QueryExpr::Scan { + fn scan() -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], 0, vec![], ), - } + }) + .unwrap() } - fn summary_node(family: SummaryFamilyType) -> SummaryNode { - SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: std::rc::Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(scan())), - schema: SummarySchema { - fields: vec![], - time_index: None, - }, - guarantee: None, - }), + /// A `SummaryAgg` directly over the kept `scan()` sub-DAG. + fn summary_node(family: FieldDataType) -> Rc { + OperatorNode::asap_node( + ASAPOp::SummaryAgg { + child: scan(), family: family.clone(), input: asap_types::post_asap::SummaryUpdate::column( asap_types::pre_asap::expr_ir::ColumnRef::Named("value".into()), ), - reduction: asap_types::pre_asap::query_expr::Reduction::by(vec![]), + reduction: asap_types::ir::operator_properties::Reduction::by(vec![]), grouping: GroupingStrategy::default(), filter: None, }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "state".into(), - dtype: family, - nullable: false, - }], - time_index: None, - }, - guarantee: None, - } + Schema::lifted(vec![Field::new("state", family, false)], None), + None, + ) } #[test] fn default_recompute_cost_is_positive_and_grows_with_structural_size() { let leaf = scan(); - let nested = QueryExpr::Dedup { + let nested = OperatorNode::non_asap_node(NonASAPOp::Dedup { cols: vec![0], - child: std::rc::Rc::new(leaf.clone()), - }; + child: Rc::clone(&leaf), + }) + .unwrap(); assert!(default_cse_recompute_cost(&leaf) > Cost::ZERO); assert!(default_cse_recompute_cost(&nested) > default_cse_recompute_cost(&leaf)); } - /// The DAG-awareness this proxy exists for: a subtree that internally + /// The DAG-awareness this proxy exists for: a sub-DAG that internally /// re-references one shared descendant (e.g. after single-query CSE, /// `x op x` collapsing both branches onto one `Rc`) must cost the same /// as if that descendant only appeared once — not double, the way a @@ -1418,27 +1413,25 @@ mod tests { /// identity-blind recursive walk) would count it. #[test] fn default_recompute_cost_does_not_double_count_an_internally_shared_descendant() { + use asap_types::ir::operator_properties::JoinKind; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::{JoinKind, Predicate}; - let true_pred = || { - Predicate(std::rc::Rc::new(QueryExpr::Literal(ScalarValue::Boolean( - true, - )))) - }; - let shared_leaf = std::rc::Rc::new(scan()); - let no_sharing = QueryExpr::Join { + let true_pred = || Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))); + let shared_leaf = scan(); + let no_sharing = OperatorNode::non_asap_node(NonASAPOp::Join { kind: JoinKind::Inner, pred: true_pred(), - left: std::rc::Rc::new(scan()), - right: std::rc::Rc::new(scan()), - }; - let with_sharing = QueryExpr::Join { + left: scan(), + right: scan(), + }) + .unwrap(); + let with_sharing = OperatorNode::non_asap_node(NonASAPOp::Join { kind: JoinKind::Inner, pred: true_pred(), - left: std::rc::Rc::clone(&shared_leaf), - right: std::rc::Rc::clone(&shared_leaf), - }; + left: Rc::clone(&shared_leaf), + right: Rc::clone(&shared_leaf), + }) + .unwrap(); assert_eq!( default_cse_recompute_cost(&no_sharing), Cost(3.0), @@ -1455,11 +1448,11 @@ mod tests { #[test] fn default_shared_maintenance_cost_orders_families_cheapest_to_priciest() { - let exact = default_cse_shared_maintenance_cost(&SummaryFamilyType::ExactAggregate( + let exact = default_cse_shared_maintenance_cost(&FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); - let sketch = default_cse_shared_maintenance_cost(&SummaryFamilyType::Sketch( + let sketch = default_cse_shared_maintenance_cost(&FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Hll, SketchParams::Hll { precision: 12 }), GroupingStrategy::default(), )); @@ -1474,7 +1467,7 @@ mod tests { fn cse_share_decision_shares_when_recompute_dominates_maintenance() { let candidate = CseCandidate { subtree: &scan(), - bound_summary: &summary_node(SummaryFamilyType::ExactAggregate( + bound_summary: &summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )), @@ -1492,7 +1485,7 @@ mod tests { fn cse_share_decision_recomputes_when_maintenance_dominates_recompute() { let candidate = CseCandidate { subtree: &scan(), - bound_summary: &summary_node(SummaryFamilyType::StatModel( + bound_summary: &summary_node(FieldDataType::StatModel( asap_types::post_asap::StatModelKind::Parametric, asap_types::post_asap::StatModelParams::Parametric { family: "gaussian_mixture".into(), @@ -1532,7 +1525,7 @@ mod tests { // hardcoding a comparison against its own defaults. let candidate = CseCandidate { subtree: &scan(), - bound_summary: &summary_node(SummaryFamilyType::StatModel( + bound_summary: &summary_node(FieldDataType::StatModel( asap_types::post_asap::StatModelKind::Parametric, asap_types::post_asap::StatModelParams::Parametric { family: "gaussian_mixture".into(), @@ -1566,20 +1559,20 @@ mod tests { } } - let root = Rc::new(scan()); + let root = scan(); let target = TargetSubDAG::new(&root); let candidate = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Summary(Rc::new(summary_node(SummaryFamilyType::Plain( + replacement: Replacement::SubDag(summary_node(FieldDataType::Plain( asap_types::pre_asap::DataType::Float64, - )))), + ))), provenance: crate::replacement::ReplacementProvenance::SummaryRealization, rationale: "whatever".into(), }; assert!(RankOnly.estimate_cost(&candidate, &target).is_nan()); } - /// `DefaultCostModel::estimate_cost` for a [`Replacement::Summary`] + /// `DefaultCostModel::estimate_cost` for a summary-rooted [`Replacement::SubDag`] /// candidate reuses [`default_cse_shared_maintenance_cost`]'s own /// per-family ordering: a candidate bound to a cheap-to-maintain family /// (an exact accumulator) must cost less than one bound to an @@ -1589,25 +1582,26 @@ mod tests { /// above. #[test] fn estimate_cost_for_summary_orders_candidates_by_family_cheapest_to_priciest() { - let root = Rc::new(scan()); + let root = scan(); let target = TargetSubDAG::new(&root); let cheap = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Summary(Rc::new(summary_node( - SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + replacement: Replacement::SubDag(summary_node(FieldDataType::ExactAggregate( + ExactKind::Sum, + ExactParams::Sum, ))), provenance: crate::replacement::ReplacementProvenance::SummaryRealization, rationale: "exact accumulator".into(), }; let pricey = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Summary(Rc::new(summary_node(SummaryFamilyType::StatModel( + replacement: Replacement::SubDag(summary_node(FieldDataType::StatModel( asap_types::post_asap::StatModelKind::Parametric, asap_types::post_asap::StatModelParams::Parametric { family: "gaussian_mixture".into(), }, - )))), + ))), provenance: crate::replacement::ReplacementProvenance::SummaryRealization, rationale: "fitted statistical model".into(), }; @@ -1625,8 +1619,8 @@ mod tests { ); } - /// `DefaultCostModel::estimate_cost` for a [`Replacement::Rewrite`] pair - /// (the `SharedSubtreeStrategy` share-vs-recompute shape) agrees with + /// `DefaultCostModel::estimate_cost` for a relational [`Replacement::SubDag`] pair + /// (the `SharedSubDagStrategy` share-vs-recompute shape) agrees with /// what `cse_share_decision` would already pick for the same target: with /// many consumers of a cheap-to-recompute leaf, the "share" candidate /// (the target's own `Rc`) must cost less than the "recompute @@ -1636,18 +1630,18 @@ mod tests { /// directly. #[test] fn estimate_cost_for_rewrite_prefers_sharing_when_recompute_dominates_maintenance() { - let target_root = Rc::new(scan()); + let target_root = scan(); let target = TargetSubDAG::with_consumer_count(&target_root, 20); let share = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::clone(&target_root)), + replacement: Replacement::SubDag(Rc::clone(&target_root)), provenance: crate::replacement::ReplacementProvenance::CseShare, rationale: "build once and share".into(), }; let recompute = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::new((*target_root).clone())), + replacement: Replacement::SubDag(Rc::new((*target_root).clone())), provenance: crate::replacement::ReplacementProvenance::CseRecompute, rationale: "build independently".into(), }; diff --git a/crates/asap-aware-mapping/src/empirical_comparison.rs b/crates/asap-aware-mapping/src/empirical_comparison.rs index ecd2a6cd4..17e3e9a82 100644 --- a/crates/asap-aware-mapping/src/empirical_comparison.rs +++ b/crates/asap-aware-mapping/src/empirical_comparison.rs @@ -41,7 +41,7 @@ pub struct OfflineExactMeasurement { } /// The companion format binds otherwise query-agnostic sketch primitives to -/// their measured readout and exact reference. Bindings describe state after +/// their measured evaluation and exact reference. Bindings describe state after /// ingestion, without merges or intervening updates during the read sequence. #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] @@ -247,7 +247,9 @@ pub fn recommend_offline( || error.query.get("value_type").and_then(|v| v.as_str()) != Some(request.query.value_type.as_str()) { - return Err("offline error observation has incompatible readout semantics".into()); + return Err( + "offline error observation has incompatible evaluation semantics".into(), + ); } if error.metric != request.accuracy.metric || error.trials < request.accuracy.minimum_trials @@ -802,14 +804,14 @@ mod tests { } assert!(recommend_offline(&evidence, &request).is_err(), "{case}"); } - for case in ["metric", "trials", "binding", "readout"] { + for case in ["metric", "trials", "binding", "evaluation"] { let mut evidence = evidence.clone(); let mut request = request.clone(); match case { "metric" => request.accuracy.metric = "rank_error".into(), "trials" => request.accuracy.minimum_trials = 100, "binding" => evidence.query_bindings.clear(), - "readout" => { + "evaluation" => { for row in &mut evidence.sketch_evidence.records { row.error.as_mut().unwrap().query["kind"] = serde_json::json!("total_count"); diff --git a/crates/asap-aware-mapping/src/empirical_cost.rs b/crates/asap-aware-mapping/src/empirical_cost.rs index ebfbc773c..18d1d7ee4 100644 --- a/crates/asap-aware-mapping/src/empirical_cost.rs +++ b/crates/asap-aware-mapping/src/empirical_cost.rs @@ -2,9 +2,8 @@ //! configuration and environment; they are neither runtime feedback nor proofs //! of an accuracy guarantee. CPU quantities are nanoseconds, never CPU operations. -use asap_types::post_asap::{ - GroupingStrategy, SketchAlgorithm, SketchParams, SummaryExpr, SummaryFamilyType, SummaryNode, -}; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; +use asap_types::post_asap::{FieldDataType, GroupingStrategy, SketchAlgorithm, SketchParams}; use asap_types::pre_asap::AggIntent; use serde::{Deserialize, Serialize}; @@ -214,13 +213,13 @@ impl EmpiricalEvidenceProvider { /// mixed with an existing deployment's unitless or CPU-operation costs. pub fn lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { - let SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, GroupingStrategy::PerSubpopulationInstance), + let Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, GroupingStrategy::PerSubpopulationInstance), grouping: GroupingStrategy::PerSubpopulationInstance, .. - } = &summary.expr + }) = &summary.operator else { return SummaryMaintenanceLifecycleCostInputs::default(); }; @@ -280,7 +279,7 @@ impl CostModel for EmpiricalCostModel { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { self.provider.lifecycle_cost_inputs(summary) } diff --git a/crates/asap-aware-mapping/src/exact_composition.rs b/crates/asap-aware-mapping/src/exact_composition.rs index a4508dc3b..17ff93ca0 100644 --- a/crates/asap-aware-mapping/src/exact_composition.rs +++ b/crates/asap-aware-mapping/src/exact_composition.rs @@ -4,21 +4,23 @@ //! `construct_summary_agg` already nests accumulator realizations, such as //! KLL over exact `Sum` state or a quantile over `Rate` state. This strategy //! covers the more general cases where an exact function must consume a -//! summary readout, or where a maintained summary consumes the values of an +//! summary evaluation, or where a maintained summary consumes the values of an //! exact function that has no accumulator realization. //! -//! Both cases use the general [`SummaryExpr::ValueOperation`] node. Its -//! semantic [`ValueOperation`] is independent from [`ExecutionTiming`], so -//! adding a function does not require adding a new physical node type. +//! Both cases use an ordinary `NonASAPOp::Aggregate` node over the child +//! plan. The node carries no timing: it runs when its consumer runs, so the +//! same operator serves both placements and adding a function does not +//! require adding a new physical node type. [`OperationPlacement`] is the +//! search-time placement choice. //! //! ## Reference, don't select //! //! A composed candidate needs a child plan to compose *with* — the inner -//! quantile's own summary readout, say. This strategy deliberately does +//! quantile's own summary evaluation, say. This strategy deliberately does //! **not** pick that child itself (the way `construct_summary_agg`'s //! `realize_child` takes the head of the child's own ranking): a //! [`Replacement::ExactComposition`] carries only the child *target* -//! (`ExactComposition::child_target`, the same `Rc` whose +//! (`ExactComposition::child_target`, the same `Rc` whose //! `TargetSubDAGCandidates` in `CandidateLogicalASAPDAGs` already holds every candidate for it). It is //! [`CandidateLogicalASAPDAGs::global_selection`](crate::replacement::CandidateLogicalASAPDAGs::global_selection) //! that commits the compatible parent/child pair — so the child's own @@ -34,7 +36,7 @@ //! //! - the target is a single-measure, `HAVING`-free exact aggregate; //! - read-time operation: the child is a bindable aggregate that has at least one -//! readout-producing summary implementation (a sketch/sample/wavelet/ +//! evaluation-producing summary implementation (a sketch/sample/wavelet/ //! model — the shapes a maintained accumulator can't sit above), and the //! target's grouping keys resolve in the child's output schema; //! transform: the target is a per-entity exact function with no @@ -58,18 +60,21 @@ //! - Decide whether a composition is *worth it*: that is //! `global_selection`'s job, using the issue's cost-units-per-second //! formulas (see `crate::cost_model::read_operation_plan_cost_rate` and -//! siblings). Missing statistics keep the conservative `KeepPreAsap`. +//! siblings). Missing statistics keep the conservative kept sub-DAG. +use asap_types::ir::non_asap::any_measure_filtered; use std::rc::Rc; -use asap_types::post_asap::execution_data_state::validate_execution_data_states_at; +use asap_types::ir::aggregate_schema::aggregate_output_schema; +use asap_types::ir::operator_properties::Reduction; +use asap_types::ir::timing::{planned_data_state, validate_default}; +use asap_types::ir::{NonASAPOp, Operator, OperatorNode, Predicate}; +use asap_types::post_asap::execution_data_state::lift_plain; use asap_types::post_asap::{ - exact_operation_output_schema, produced_data_state, AccuracyError, ExactOperation, - ExecutionDataState, ExecutionDataStateError, ResultGuarantee, SummaryExpr, SummaryNode, - SummarySchema, ValueOperation, + AccuracyError, ExactOperationSchemaError, ExecutionDataState, ExecutionDataStateError, + ResultGuarantee, Schema, }; use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{any_measure_filtered, QueryExpr, Reduction}; use asap_types::types::AccuracyTarget; use crate::cost_model::CostModel; @@ -84,9 +89,61 @@ use asap_types::post_asap::ExecutionTiming; /// Which side of the maintenance/read boundary an [`ExactComposition`]'s /// exact function executes on. +/// The exact function an [`ExactComposition`] applies: the parameters of +/// the `NonASAPOp::Aggregate` node the composition builds over its child. +#[derive(Debug, Clone, PartialEq)] +pub enum ExactOperation { + Aggregate { + reduction: Reduction, + measures: Vec, + output_names: Vec, + filters: Vec>, + having: Option, + }, +} + +impl ExactOperation { + /// Output schema of this operation over a child whose edge carries + /// `input` — the same canonical derivation the pre-ASAP `Aggregate` + /// node uses. `Err` when the child carries non-plain state the operator + /// cannot read. + pub fn output_schema(&self, input: &Schema) -> Result { + if !input.is_all_plain() { + return Err(ExactOperationSchemaError::NonPlainInput); + } + let plain = lift_plain(input); + let ExactOperation::Aggregate { + reduction, + measures, + output_names, + .. + } = self; + let out = aggregate_output_schema(&plain, reduction, measures, output_names)?; + Ok(lift_plain(&out)) + } + + fn into_op(self, child: Rc) -> NonASAPOp { + let ExactOperation::Aggregate { + reduction, + measures, + output_names, + filters, + having, + } = self; + NonASAPOp::Aggregate { + reduction, + measures, + output_names, + filters, + having, + child, + } + } +} + #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] pub enum OperationPlacement { - /// After the child's summary readout. + /// After the child's summary evaluation. Read, /// On the maintenance path, feeding /// maintained state above. @@ -120,32 +177,35 @@ pub struct ExactComposition { pub op: ExactOperation, /// The pre-ASAP child the operator consumes; its `TargetSubDAGCandidates` holds the /// candidates `global_selection` may commit this composition with. - pub child_target: Rc, + pub child_target: Rc, /// The composed node's output schema — the target's own pre-ASAP /// output schema, lifted with every column `Plain` (an exact operator /// only ever produces plain values). - pub schema: SummarySchema, + pub schema: Schema, } impl ExactComposition { + /// The data state `child` produces when this operation (its consumer) + /// runs at the placement's timing. + fn child_data_state(&self, child: &Rc) -> ExecutionDataState { + planned_data_state(child, self.placement.data_state().timing) + } + /// Can `child` legally be this composition's input? Phase legality - /// (the child's produced data_state — a `KeepPreAsap` leaf takes the + /// (the child's produced data_state — a kept pre-ASAP sub-DAG takes the /// phase this edge assigns) plus the plain-operand rule, checked /// through the same schema derivation [`Self::compose`] uses. - pub fn accepts_child(&self, child: &SummaryNode) -> bool { - let phase_ok = match produced_data_state(&child.expr) { - None => true, - Some(avail) => avail == self.placement.data_state(), - }; - phase_ok && exact_operation_output_schema(&self.op, &child.schema).is_ok() + pub fn accepts_child(&self, child: &Rc) -> bool { + self.child_data_state(child) == self.placement.data_state() + && self.op.output_schema(&child.schema).is_ok() } /// Build the composed, data_state-validated node over `child`. Every edge of /// the result (including everything beneath `child`) is checked by - /// `asap_types::post_asap::validate_execution_data_states`; an illegal + /// `asap_types::ir::timing::validate_default`; an illegal /// placement is a typed [`RealizationError::ExecutionDataState`], never deferred to a /// runtime. - pub fn compose(&self, child: Rc) -> Result, RealizationError> { + pub fn compose(&self, child: Rc) -> Result, RealizationError> { self.compose_with_accuracy(child, &DefaultAccuracyModel) } @@ -154,24 +214,23 @@ impl ExactComposition { /// unsupported folds fail closed with a typed accuracy error. pub fn compose_with_accuracy( &self, - child: Rc, + child: Rc, accuracy_model: &dyn AccuracyModel, - ) -> Result, RealizationError> { - if let Some(produced) = produced_data_state(&child.expr) { - if produced != self.placement.data_state() { - let edge = match self.placement { - OperationPlacement::Maintenance => "ValueOperation.child (maintenance time)", - OperationPlacement::Read => "ValueOperation.child (read time)", - }; - return Err(RealizationError::ExecutionDataState( - ExecutionDataStateError::IllegalChildDataState { - edge, - child: produced, - }, - )); - } + ) -> Result, RealizationError> { + let produced = self.child_data_state(&child); + if produced != self.placement.data_state() { + let edge = match self.placement { + OperationPlacement::Maintenance => "exact operation child (maintenance time)", + OperationPlacement::Read => "exact operation child (read time)", + }; + return Err(RealizationError::ExecutionDataState( + ExecutionDataStateError::IllegalChildDataState { + edge, + child: produced, + }, + )); } - let schema = exact_operation_output_schema(&self.op, &child.schema)?; + let schema = self.op.output_schema(&child.schema)?; let guarantee = match &child.guarantee { None => None, Some(input) if input.is_exact() => Some(ResultGuarantee::exact(format!( @@ -198,23 +257,11 @@ impl ExactComposition { None => None, }, }; - let timing = match self.placement { - OperationPlacement::Read => asap_types::post_asap::ExecutionTiming::QueryTime, - OperationPlacement::Maintenance => { - asap_types::post_asap::ExecutionTiming::IngestionTime - } - }; - let expr = SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Exact(self.op.clone()), - timing, - }; - let node = Rc::new(SummaryNode { - expr, - schema, - guarantee, - }); - validate_execution_data_states_at(&node, self.placement.data_state())?; + let node = Rc::new( + OperatorNode::with_schema(Operator::NonASAP(self.op.clone().into_op(child)), schema) + .with_guarantee(guarantee), + ); + validate_default(&node, self.placement.data_state().timing)?; Ok(node) } @@ -227,7 +274,7 @@ impl ExactComposition { } } -/// Which exact reducers may run as a query-time fold over readout rows. +/// Which exact reducers may run as a query-time fold over evaluation rows. /// `Count` only at `Exact` accuracy (an approximate count is a sketch /// target, not an exact fold). fn is_query_time_reducer(intent: &AggIntent) -> bool { @@ -245,9 +292,9 @@ fn is_query_time_reducer(intent: &AggIntent) -> bool { ) } -/// Does `implementation` need a `SummaryEstimate` readout to yield a value +/// Does `implementation` need a `SummaryEstimate` evaluation to yield a value /// — i.e. is it a shape a maintained accumulator can't legally sit above? -fn needs_readout(implementation: &Realization) -> bool { +fn needs_evaluation(implementation: &Realization) -> bool { matches!( implementation, Realization::Sketch(_) @@ -259,17 +306,17 @@ fn needs_readout(implementation: &Realization) -> bool { /// The `(op, child)` of a read-time operation-shaped target, or `None`. fn query_time_shape( - root: &QueryExpr, + root: &OperatorNode, cost_model: &dyn CostModel, -) -> Option<(ExactOperation, Rc, AggIntent)> { - let QueryExpr::Aggregate { +) -> Option<(ExactOperation, Rc, AggIntent)> { + let Some(NonASAPOp::Aggregate { reduction, measures, output_names, filters, having: None, child, - } = root + }) = root.non_asap() else { return None; }; @@ -291,13 +338,12 @@ fn query_time_shape( let child_intent = bindable_intent(child)?; if !realizations_for_intent(child_intent, cost_model) .iter() - .any(needs_readout) + .any(needs_evaluation) { return None; } // Grouping keys must resolve in the child's output schema — the same // derivation the composed node's own schema will use. - root.output_schema().ok()?; Some(( ExactOperation::Aggregate { reduction: reduction.clone(), @@ -314,17 +360,17 @@ fn query_time_shape( /// The `(op, child)` of a function-shaped target — a per-entity exact /// transform with no accumulator form — or `None`. fn ingestion_time_shape( - root: &QueryExpr, + root: &OperatorNode, cost_model: &dyn CostModel, -) -> Option<(ExactOperation, Rc, AggIntent)> { - let QueryExpr::Aggregate { +) -> Option<(ExactOperation, Rc, AggIntent)> { + let Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures, output_names, filters, having: None, child, - } = root + }) = root.non_asap() else { return None; }; @@ -346,7 +392,6 @@ fn ingestion_time_shape( { return None; } - root.output_schema().ok()?; Some(( ExactOperation::Aggregate { reduction: Reduction::PerEntity, @@ -386,10 +431,7 @@ impl<'a> ExactCompositionStrategy<'a> { } fn candidates(&self, target: &TargetSubDAG<'_>) -> Vec { - let Ok(schema) = target.root.output_schema() else { - return Vec::new(); - }; - let schema = asap_types::post_asap::execution_data_state::lift_plain(&schema); + let schema = lift_plain(&target.root.schema); let mut out = Vec::new(); if let Some((op, child, intent)) = query_time_shape(target.root, self.cost_model) { @@ -410,10 +452,10 @@ impl<'a> ExactCompositionStrategy<'a> { }), provenance: ReplacementProvenance::ValueOperationAtQueryTime, rationale: format!( - "{} is an exact fold whose input is the readout of {} — a maintained \ - accumulator cannot consume query-time values, so instead of collapsing \ - the whole tree into KeepPreAsap this applies the fold as an \ - ExactRead over whichever summary readout global_selection \ + "{} is an exact fold whose input is the evaluation of {} — a maintained \ + accumulator cannot consume query-time values, so instead of keeping \ + the whole tree pre-ASAP this applies the fold as an \ + ExactRead over whichever summary evaluation global_selection \ commits for the child target (asap_aware_mapping::exact_composition)", describe_intent(&intent), child_desc @@ -441,7 +483,7 @@ impl<'a> ExactCompositionStrategy<'a> { "{} is an exact per-entity function with no accumulator form; as an \ explicit ExactMaintenance on the update path its output can feed a \ maintained summary above it instead of being handed over as an opaque \ - raw KeepPreAsap blob (asap_aware_mapping::exact_composition)", + raw kept subtree (asap_aware_mapping::exact_composition)", describe_intent(&intent) ), }); @@ -465,55 +507,20 @@ impl ReplacementStrategy for ExactCompositionStrategy<'_> { mod tests { use super::*; use crate::cost_model::{DefaultCostModel, ValueOperationCapabilities}; - use crate::replacement::keep_pre_asap; - use asap_types::post_asap::{ExecutionDataStateError, SketchAlgorithm, SummaryFamilyType}; + use crate::replacement::retain_exact; + use crate::test_support::{agg, agg_per_entity as per_entity, metric_scan, timed}; + use asap_types::ir::ASAPOp; + use asap_types::post_asap::{ExecutionDataStateError, FieldDataType, SketchAlgorithm}; use asap_types::pre_asap::agg_intent::default_quantile; - use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{Column, DataType, Schema}; - - fn metric_scan(labels: &[&str]) -> QueryExpr { - let mut columns = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - ]; - columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index(columns, 0, vec![]), - } - } - - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::by(by), - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } - - fn per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } /// `max by (zone) (quantile by (zone, host) (m))`. - fn max_over_quantile() -> Rc { + fn max_over_quantile() -> Rc { let inner = agg( vec![2, 3], default_quantile(0.99), metric_scan(&["zone", "host"]), ); - Rc::new(agg(vec![0], AggIntent::Max { col: None }, inner)) + agg(vec![0], AggIntent::Max { col: None }, inner) } #[test] @@ -535,7 +542,7 @@ mod tests { candidates[0].provenance, ReplacementProvenance::ValueOperationAtQueryTime ); - let QueryExpr::Aggregate { child, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { child, .. }) = root.non_asap() else { unreachable!() }; assert!( @@ -549,7 +556,7 @@ mod tests { #[test] fn proposes_query_time_operation_for_avg_over_quantile_alongside_the_rewrite() { let inner = agg(vec![2], default_quantile(0.99), metric_scan(&["zone"])); - let root = Rc::new(agg(vec![0], AggIntent::Avg { col: None }, inner)); + let root = agg(vec![0], AggIntent::Avg { col: None }, inner); let target = TargetSubDAG::new(&root); assert_eq!( ExactCompositionStrategy::default_cost_model() @@ -563,7 +570,7 @@ mod tests { #[test] fn proposes_ingestion_time_operation_for_a_per_entity_pass_through_over_raw_input() { - let root = Rc::new(per_entity(AggIntent::Deriv, metric_scan(&["zone"]))); + let root = per_entity(AggIntent::Deriv, metric_scan(&["zone"])); let target = TargetSubDAG::new(&root); let candidates = ExactCompositionStrategy::default_cost_model().replacements(&target); assert_eq!(candidates.len(), 1); @@ -575,21 +582,21 @@ mod tests { #[test] fn does_not_propose_for_shapes_already_covered_by_accumulators() { - // sum by (zone) over an exact Sum child: the child has no readout, + // sum by (zone) over an exact Sum child: the child has no evaluation, // so SummaryAgg(Sum) over SummaryAgg(Sum) is already legal. let inner = agg( vec![2, 3], AggIntent::Sum { col: None }, metric_scan(&["zone", "host"]), ); - let root = Rc::new(agg(vec![0], AggIntent::Sum { col: None }, inner)); + let root = agg(vec![0], AggIntent::Sum { col: None }, inner); assert!(!ExactCompositionStrategy::default_cost_model().matches(&TargetSubDAG::new(&root))); // rate is an exact accumulator — directly nestable, no separate value operation. - let rate = Rc::new(per_entity(AggIntent::Rate, metric_scan(&[]))); + let rate = per_entity(AggIntent::Rate, metric_scan(&[])); assert!(!ExactCompositionStrategy::default_cost_model().matches(&TargetSubDAG::new(&rate))); // A sketch-capable outer intent is not an exact fold. let inner = agg(vec![2], default_quantile(0.5), metric_scan(&["zone"])); - let root = Rc::new(agg(vec![0], default_quantile(0.99), inner)); + let root = agg(vec![0], default_quantile(0.99), inner); assert!(!ExactCompositionStrategy::default_cost_model().matches(&TargetSubDAG::new(&root))); } @@ -614,7 +621,7 @@ mod tests { let strategy = ExactCompositionStrategy::new(&NoMixedExecution); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); - let deriv = Rc::new(per_entity(AggIntent::Deriv, metric_scan(&[]))); + let deriv = per_entity(AggIntent::Deriv, metric_scan(&[])); assert!(!strategy.matches(&TargetSubDAG::new(&deriv))); } @@ -626,12 +633,13 @@ mod tests { let Replacement::ExactComposition(comp) = &candidates[0].replacement else { unreachable!() }; - // A bare SummaryAgg (state, no readout) is not a legal read-time operation + // A bare SummaryAgg (state, no evaluation) is not a legal read-time operation // input — the operator would be consuming sketch state. let state_child = crate::replacement::realize_child(&comp.child_target, &DefaultCostModel).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &state_child.expr else { - panic!("expected the child to realize to a readout"); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &state_child.operator + else { + panic!("expected the child to realize to a evaluation"); }; assert!(!comp.accepts_child(summary_input)); assert!(matches!( @@ -640,7 +648,7 @@ mod tests { ExecutionDataStateError::IllegalChildDataState { .. } )) )); - // The readout itself is accepted and composes to a plain schema. + // The evaluation itself is accepted and composes to a plain schema. assert!(comp.accepts_child(&state_child)); let composed = comp.compose(state_child).unwrap(); assert!( @@ -648,46 +656,93 @@ mod tests { "rank error has no registered conversion through max" ); assert!(matches!( - composed.expr, - SummaryExpr::ValueOperation { - timing: ExecutionTiming::QueryTime, - .. - } + composed.operator, + Operator::NonASAP(NonASAPOp::Aggregate { .. }) )); + // Timing is no longer stored by composition: under the default + // lifecycle assignment the composed read-time operation runs at + // query time. + assert_eq!(timed(&composed).timing, Some(ExecutionTiming::QueryTime)); assert!(composed .schema .fields .iter() - .all(|f| matches!(f.dtype, SummaryFamilyType::Plain(_)))); + .all(|f| matches!(f.dtype, FieldDataType::Plain(_)))); } #[test] - fn compose_rejects_a_readout_child_for_a_ingestion_time_operation() { + fn compose_rejects_a_evaluation_child_for_a_ingestion_time_operation() { let inner = agg(vec![2], default_quantile(0.99), metric_scan(&["zone"])); - let root = Rc::new(per_entity(AggIntent::Deriv, inner)); + let root = per_entity(AggIntent::Deriv, inner); let candidates = ExactCompositionStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); let Replacement::ExactComposition(comp) = &candidates[0].replacement else { unreachable!() }; - let readout = + let evaluation = crate::replacement::realize_child(&comp.child_target, &DefaultCostModel).unwrap(); - assert!(!comp.accepts_child(&readout)); + assert!(!comp.accepts_child(&evaluation)); assert!(matches!( - comp.compose(readout), + comp.compose(evaluation), Err(RealizationError::ExecutionDataState( ExecutionDataStateError::IllegalChildDataState { .. } )) )); // Raw update input is fine. - let raw = keep_pre_asap(&comp.child_target).unwrap(); + let raw = retain_exact(&comp.child_target).unwrap(); assert!(comp.accepts_child(&raw)); + // Timing is no longer stored by composition: the composition's + // placement is maintenance time, the composed exact operation is a + // plain Aggregate over the raw rows, and it is legal (and planned to + // run) at ingestion time. + assert_eq!(comp.placement, OperationPlacement::Maintenance); + let composed = comp.compose(raw).unwrap(); assert!(matches!( - comp.compose(raw).unwrap().expr, - SummaryExpr::ValueOperation { - timing: ExecutionTiming::IngestionTime, - .. - } + composed.operator, + Operator::NonASAP(NonASAPOp::Aggregate { .. }) + )); + validate_default(&composed, ExecutionTiming::IngestionTime).unwrap(); + assert_eq!( + planned_data_state(&composed, ExecutionTiming::IngestionTime).timing, + ExecutionTiming::IngestionTime + ); + } + + fn max_op(by: Vec) -> ExactOperation { + ExactOperation::Aggregate { + reduction: Reduction::by(by), + measures: vec![AggIntent::Max { col: None }], + output_names: vec![], + filters: vec![], + having: None, + } + } + + #[test] + fn exact_operator_schema_matches_pre_asap_aggregate_derivation() { + let child = lift_plain(&metric_scan(&["zone"]).schema); + let out = max_op(vec![2]).output_schema(&child).unwrap(); + let names: Vec<_> = out.fields.iter().map(|f| f.name.as_str()).collect(); + assert_eq!(names, vec!["zone", "max"]); + assert!(out.is_all_plain()); + } + + #[test] + fn exact_operator_rejects_non_plain_input() { + let state = Schema::lifted( + vec![asap_types::pre_asap::Field::new( + "state", + FieldDataType::ExactAggregate( + asap_types::post_asap::ExactKind::Sum, + asap_types::post_asap::ExactParams::Sum, + ), + false, + )], + None, + ); + assert!(matches!( + max_op(vec![]).output_schema(&state), + Err(ExactOperationSchemaError::NonPlainInput) )); } } diff --git a/crates/asap-aware-mapping/src/explanation.rs b/crates/asap-aware-mapping/src/explanation.rs index 5f930dbc4..46c2eded8 100644 --- a/crates/asap-aware-mapping/src/explanation.rs +++ b/crates/asap-aware-mapping/src/explanation.rs @@ -31,27 +31,27 @@ //! collapses into a single question this module asks of *that* data instead: //! **for a given `TargetSubDAG`, does its candidate list contain anything //! other than the trivial, no-op realization?** A `TargetSubDAG` whose only -//! candidate is "the one thing `SketchAlgorithmStrategy` would have committed +//! candidate is "the one thing `ASAPStrategies` would have committed //! to anyway, with no alternative" has no optimization to report — that //! candidate isn't an *opportunity*, it's just the target's existing shape //! reflected back. A `TargetSubDAG` with more than one candidate (several //! sketch families to choose between), or one candidate that is itself a -//! genuine alternative to the status quo (share this already-shared subtree +//! genuine alternative to the status quo (share this already-shared sub-DAG //! instead of recomputing it at every consumer), *is* an applicability //! finding — [`explain_replacements`] and //! [`explain_replacements_with`] just translate [`CandidateLogicalASAPDAGs`]'s //! [`TargetSubDAGCandidates`]s into that shape: //! //! - [`ExplanationKind::SketchApproximation`] — the `TargetSubDAG`'s -//! candidate list contains at least one [`Replacement::Summary`] that -//! actually realizes a sketch family (`SummaryFamilyType::Sketch`), i.e. -//! [`SketchAlgorithmStrategy`] found something to offer beyond whatever +//! candidate list contains at least one summary-realization [`Replacement::SubDag`] that +//! actually realizes a sketch family (`FieldDataType::Sketch`), i.e. +//! [`ASAPStrategies`] found something to offer beyond whatever //! exact/pass-through candidate [`crate::replacement`]'s own //! `realizations_for_intent` would have committed to on its own. //! - [`ExplanationKind::CommonSubexpressionReuse`] — the `TargetSubDAG` //! has two or more consumers *and* its candidate list contains the -//! [`SharedSubtreeStrategy`] "build once and share" candidate (the one -//! whose `Rc` is the group's own `target`) — i.e. sharing this subtree +//! [`SharedSubDagStrategy`] "build once and share" candidate (the one +//! whose `Rc` is the group's own `target`) — i.e. sharing this sub-DAG //! instead of recomputing it independently is a real, reported choice, not //! just an accident of how the workload happened to be built. //! @@ -103,7 +103,7 @@ //! [`ReplacementStrategy`] already *is* that extension point, one layer //! down, and [`explain_replacements_with`]'s own `strategies` //! parameter is where a caller plugs in a custom one (or a custom -//! `CostModel`, via [`crate::replacement::SketchAlgorithmStrategy::new`]) — the identical spot +//! `CostModel`, via [`crate::replacement::ASAPStrategies::new`]) — the identical spot //! [`crate::replacement::search_workload_with`] itself exposes. //! //! ## Two guarantees the old traversal made, re-verified against the new one @@ -134,7 +134,7 @@ //! ## One thing [`CandidateLogicalASAPDAGs`] doesn't carry that this module still needs: //! human-readable `location` text //! -//! [`TargetSubDAGCandidates`]/[`CandidateLogicalASAPDAGs`] deliberately track only `Rc` +//! [`TargetSubDAGCandidates`]/[`CandidateLogicalASAPDAGs`] deliberately track only `Rc` //! pointer identity — the currency the search itself needs — not //! caller-facing prose. [`ReplacementExplanation::location`] is prose (a //! breadcrumb like `root "dash_a" > lhs`), so this module keeps one small, @@ -161,11 +161,11 @@ //! //! | Catalog entry | Status | Where a future `ExplanationKind` would come from | //! |---|---|---| -//! | Semantic-equivalent rewriting (e.g. `avg` → `sum`/`count`) | [`AvgToSumOverCountStrategy`](crate::rewrite::AvgToSumOverCountStrategy) exists and is wired into `default_strategies()` (issue #253) — but still no `ExplanationKind` of its own below, since this table is about *direct* findings for a catalog entry, and this strategy's whole point is indirect: its `Replacement::Rewrite` candidate exposes `sum`/`count` as independently bindable discovered targets, which can then earn `CommonSubexpressionReuse` findings when the workload actually reuses them | A dedicated variant would need `findings_from_candidate_logical_asap_dags` to recognize a `LogicalRewrite`-provenance candidate as a finding in its own right, not just rely on what it exposes downstream | -//! | Roll-ups (fine-to-coarse group-by reuse) | [`RollupStrategy`](crate::rollup::RollupStrategy), derived from workload siblings after CSE/target discovery (issue #254) | Any `Replacement::Rewrite` candidate that rolls a coarse aggregate up from a compatible finer aggregate | +//! | Semantic-equivalent rewriting (e.g. `avg` → `sum`/`count`) | [`AvgToSumOverCountStrategy`](crate::rewrite::AvgToSumOverCountStrategy) exists and is wired into `default_strategies()` (issue #253) — but still no `ExplanationKind` of its own below, since this table is about *direct* findings for a catalog entry, and this strategy's whole point is indirect: its `Replacement::SubDag` rewrite candidate exposes `sum`/`count` as independently bindable discovered targets, which can then earn `CommonSubexpressionReuse` findings when the workload actually reuses them | A dedicated variant would need `findings_from_candidate_logical_asap_dags` to recognize a `LogicalRewrite`-provenance candidate as a finding in its own right, not just rely on what it exposes downstream | +//! | Roll-ups (fine-to-coarse group-by reuse) | [`RollupStrategy`](crate::rollup::RollupStrategy), derived from workload siblings after CSE/target discovery (issue #254) | Any `Replacement::SubDag` rewrite candidate that rolls a coarse aggregate up from a compatible finer aggregate | //! | Wavelets/OMP | Params type exists (`WaveletKind`/`WaveletParams`), reachable only via a deployment `CostModel::realize_extension` (no core `AggIntent` dispatch picks it) | A `ReplacementStrategy` that inspects a deployment's own `CostModel`, once some intent shape actually maps to `Realization::Wavelet` | //! | Sampling | Same story as Wavelets: `SamplingKind`/`SamplingParams` exist, unreachable from core dispatch | Same hook as Wavelets, for `Realization::Sample` | -//! | Deep generative compression | No representation at all — no `Realization`/`SummaryFamilyType` variant | Needs a new summary family added to `asap_types::post_asap` first | +//! | Deep generative compression | No representation at all — no `Realization`/`FieldDataType` variant | Needs a new summary family added to `asap_types::post_asap` first | //! | Approximation frameworks for windows | No representation — `TimeRange`/`PromqlSubquery` windows are always evaluated exactly | Would key off those node types once an approximate-window operator exists | //! | Function decomposition | No representation anywhere | No hook point identified yet | //! | Continuous distributed monitoring | No representation — `RepeatingEntry`/`RepetitionInterval` in `asap_types::workload` describe *that* a query repeats, not any monitoring-specific decomposition | Would likely key off `RepeatingEntry` once such logic exists | @@ -175,10 +175,9 @@ //! [`ReplacementStrategy`]: crate::replacement::ReplacementStrategy //! [`ReplacementSubDAG`]: crate::replacement::ReplacementSubDAG //! [`Replacement`]: crate::replacement::Replacement -//! [`Replacement::Summary`]: crate::replacement::Replacement::Summary -//! [`Replacement::Rewrite`]: crate::replacement::Replacement::Rewrite -//! [`SketchAlgorithmStrategy`]: crate::replacement::SketchAlgorithmStrategy -//! [`SharedSubtreeStrategy`]: crate::replacement::SharedSubtreeStrategy +//! [`Replacement::SubDag`]: crate::replacement::Replacement::SubDag +//! [`ASAPStrategies`]: crate::replacement::ASAPStrategies +//! [`SharedSubDagStrategy`]: crate::replacement::SharedSubDagStrategy //! [`CandidateLogicalASAPDAGs`]: crate::replacement::CandidateLogicalASAPDAGs //! [`TargetSubDAGCandidates`]: crate::replacement::TargetSubDAGCandidates @@ -186,9 +185,9 @@ use std::collections::HashMap; use std::fmt::Display; use std::rc::Rc; -use asap_types::post_asap::{SummaryExpr, SummaryFamilyType, SummaryNode}; -use asap_types::pre_asap::cse::{structural_hash, HashCache}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::cse::{structural_hash, HashCache}; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; +use asap_types::post_asap::FieldDataType; use crate::replacement::{ self, CandidateLogicalASAPDAGs, Replacement, ReplacementStrategy, TargetSubDAGCandidates, @@ -206,14 +205,14 @@ use crate::replacement::{ #[non_exhaustive] pub enum ExplanationKind { /// A `TargetSubDAG`'s candidate list contains at least one - /// [`Replacement::Summary`] that realizes a sketch family — - /// [`crate::replacement::SketchAlgorithmStrategy`] found a genuine sketch + /// [`Replacement::SubDag`] that realizes a sketch family — + /// [`crate::replacement::ASAPStrategies`] found a genuine sketch /// alternative for this `Aggregate`, beyond whatever exact/pass-through /// candidate `crate::replacement`'s own `realizations_for_intent` would /// have committed to on its own. SketchApproximation, /// A `TargetSubDAG` has two or more consumers *and* its candidate list - /// contains [`crate::replacement::SharedSubtreeStrategy`]'s "build once + /// contains [`crate::replacement::SharedSubDagStrategy`]'s "build once /// and share" candidate — the catalog's cross-statistic / cross-metrics / /// cross-subpopulation reuse entries, all the same underlying structural /// fact. @@ -222,8 +221,8 @@ pub enum ExplanationKind { /// [`Replacement::ExactComposition`] — /// [`crate::exact_composition::ExactCompositionStrategy`] found an exact /// operator that can be composed with a summary plan across an explicit - /// update/readout boundary instead of collapsing the whole tree into - /// `KeepPreAsap` (issue #171). + /// update/evaluation boundary instead of keeping the whole tree as it is + /// (issue #171). ExactComposition, } @@ -233,11 +232,11 @@ pub enum ExplanationKind { /// not machine parsing — literally the matching candidate's own /// [`crate::replacement::ReplacementSubDAG::rationale`]). /// -/// `node_hash` is [`structural_hash`](asap_types::pre_asap::cse::structural_hash) -/// of the `TargetSubDAG`'s own `target` subtree — the same function, on the -/// same `Rc` shape, that [`asap_types::dag_export::DagNode::hash`] +/// `node_hash` is [`structural_hash`](asap_types::ir::cse::structural_hash) +/// of the `TargetSubDAG`'s own `target` sub-DAG — the same function, on the +/// same `Rc` shape, that [`asap_types::dag_export::DagNode::hash`] /// is computed with. A downstream consumer that independently exported the -/// same `QueryExpr` (e.g. via `asap_types::dag_export::export`) can match +/// same node (e.g. via `asap_types::dag_export::export`) can match /// this explanation to a `DagNode` by first comparing hashes and then /// confirming structural equality with [`ReplacementExplanation::target`]. #[derive(Debug, Clone, PartialEq)] @@ -249,7 +248,7 @@ pub struct ReplacementExplanation { /// The exact target expression the explanation describes. Reporting /// integrations use this together with `node_hash`: the hash narrows the /// search, and structural equality makes the final match collision-safe. - pub target: Rc, + pub target: Rc, } /// Explain every replacement [`crate::replacement::search_workload`] finds @@ -265,7 +264,7 @@ pub struct ReplacementExplanation { /// candidate-plan space, then reads findings off it — see the module docs' /// "The reframing" section for what that translation actually checks. pub fn explain_replacements( - roots: Vec<(Id, QueryExpr)>, + roots: Vec<(Id, Rc)>, ) -> Vec { explain_replacements_with(roots, &replacement::default_strategies()) } @@ -274,17 +273,17 @@ pub fn explain_replacements( /// instead of [`crate::replacement::default_strategies`] — the extension /// point for a deployment-specific [`ReplacementStrategy`], or a custom /// `CostModel` plugged into -/// [`crate::replacement::SketchAlgorithmStrategy::new`] (e.g. via +/// [`crate::replacement::ASAPStrategies::new`] (e.g. via /// [`crate::replacement::default_strategies_with`]). /// /// [`ReplacementStrategy`]: crate::replacement::ReplacementStrategy pub fn explain_replacements_with<'s, Id: Display>( - roots: Vec<(Id, QueryExpr)>, + roots: Vec<(Id, Rc)>, strategies: &[Box], ) -> Vec { - let ided: Vec<(String, Rc)> = roots + let ided: Vec<(String, Rc)> = roots .into_iter() - .map(|(id, expr)| (id.to_string(), Rc::new(expr))) + .map(|(id, expr)| (id.to_string(), expr)) .collect(); let space = replacement::search_workload_with(ided, strategies); findings_from_candidate_logical_asap_dags(&space) @@ -375,7 +374,7 @@ fn sketch_finding_reason(group: &TargetSubDAGCandidates) -> Option { .candidates .iter() .filter( - |c| matches!(&c.replacement, Replacement::Summary(node) if is_sketch_realization(node)), + |c| matches!(&c.replacement, Replacement::SubDag(node) if is_sketch_realization(node)), ) .map(|c| c.rationale.as_str()) .collect(); @@ -387,7 +386,7 @@ fn sketch_finding_reason(group: &TargetSubDAGCandidates) -> Option { } /// Does `group` have two or more consumers *and* a "build once and share" -/// candidate (the [`Replacement::Rewrite`] whose `Rc` is the group's own +/// candidate (the [`Replacement::SubDag`] whose `Rc` is the group's own /// `target`) in its candidate list? If so, the finding's `reason` is that /// candidate's own `rationale`. fn shared_subexpr_finding_reason(group: &TargetSubDAGCandidates) -> Option { @@ -398,18 +397,18 @@ fn shared_subexpr_finding_reason(group: &TargetSubDAGCandidates) -> Option bool { +fn is_sketch_realization(node: &OperatorNode) -> bool { if node .guarantee .as_ref() @@ -417,9 +416,13 @@ fn is_sketch_realization(node: &SummaryNode) -> bool { { return false; } - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => is_sketch_realization(summary_input), - SummaryExpr::SummaryAgg { family, .. } => matches!(family, SummaryFamilyType::Sketch(..)), + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + is_sketch_realization(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) => { + matches!(family, FieldDataType::Sketch(..)) + } _ => false, } } @@ -432,8 +435,10 @@ fn is_sketch_realization(node: &SummaryNode) -> bool { /// every breadcrumb path that reaches a given `Rc`, not just the first: a /// shared node referenced from two workload roots (or two branches of one /// root) needs both breadcrumbs in its finding's `location`, not just one. -fn collect_locations(roots: &[(String, Rc)]) -> HashMap<*const QueryExpr, Vec> { - let mut locations: HashMap<*const QueryExpr, Vec> = HashMap::new(); +fn collect_locations( + roots: &[(String, Rc)], +) -> HashMap<*const OperatorNode, Vec> { + let mut locations: HashMap<*const OperatorNode, Vec> = HashMap::new(); for (id, root) in roots { visit(root, format!("root {id:?}"), &mut locations); } @@ -444,9 +449,9 @@ fn collect_locations(roots: &[(String, Rc)]) -> HashMap<*const QueryE /// through its children. A shared ancestor is intentionally traversed once /// per incoming path so every descendant receives every valid breadcrumb. fn visit( - node: &Rc, + node: &Rc, label: String, - locations: &mut HashMap<*const QueryExpr, Vec>, + locations: &mut HashMap<*const OperatorNode, Vec>, ) { let ptr = Rc::as_ptr(node); locations.entry(ptr).or_default().push(label.clone()); @@ -455,20 +460,24 @@ fn visit( /// `node`'s own **relational-skeleton** operator children — the same scope /// `crate::replacement`'s own target-discovery `walk_children` (and -/// `asap_types::pre_asap::cse::share_common_subtrees`'s `rebuild_children`) -/// use. Exhaustive over every `QueryExpr` variant: a new variant fails to -/// compile here until this match is extended too. +/// `asap_types::ir::cse::share_common_subdags`) use. Exhaustive over every +/// `NonASAPOp` variant: a new variant fails to compile here until this match +/// is extended too. An ASAP node never occurs in a workload root. fn visit_children( - node: &QueryExpr, + node: &OperatorNode, label: &str, - locations: &mut HashMap<*const QueryExpr, Vec>, + locations: &mut HashMap<*const OperatorNode, Vec>, ) { - use QueryExpr::*; - match node { - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => { + use asap_types::ir::{NonASAPOp::*, ScalarExpr}; + let Operator::NonASAP(op) = &node.operator else { + return; + }; + match op { + Scan { .. } | Values { .. } => {} + PromqlVectorFromScalar(ScalarExpr::PromqlScalarFromVector(c)) => { visit(c, format!("{label} > child"), locations) } + PromqlVectorFromScalar(_) => {} PromqlRelabel { child, .. } | PromqlInfoEnrich { child, .. } | PromqlSeriesSample { child, .. } @@ -484,7 +493,7 @@ fn visit_children( | Limit { child, .. } => visit(child, format!("{label} > child"), locations), Concat { children, .. } => { for (i, c) in children.iter().enumerate() { - visit_children(c, &format!("{label} > concat[{i}]"), locations); + visit(c, format!("{label} > concat[{i}]"), locations); } } Join { left, right, .. } | SetOp { left, right, .. } => { @@ -495,52 +504,66 @@ fn visit_children( visit(lhs, format!("{label} > lhs"), locations); visit(rhs, format!("{label} > rhs"), locations); } - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} } } #[cfg(test)] mod tests { use super::*; + use asap_types::ir::operator_properties::{BinaryOpKind, Reduction, Source}; + use asap_types::ir::{BinaryOperator, NonASAPOp, OperatorNode, Predicate, ScalarExpr}; use asap_types::pre_asap::agg_intent::{default_quantile, AggIntent}; - use asap_types::pre_asap::query_expr::{Reduction, Source}; - use asap_types::pre_asap::schema::{Column, DataType, Schema}; + use asap_types::pre_asap::schema::{DataType, Field, Schema}; + use asap_types::types::AccuracyTarget; - fn metric_scan(labels: &[&str]) -> QueryExpr { + fn metric_scan(labels: &[&str]) -> Rc { let mut columns = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ]; - columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); - QueryExpr::Scan { + columns.extend( + labels + .iter() + .map(|n| Field::plain(*n, DataType::Utf8, true)), + ); + OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index(columns, 0, vec![]), - } + }) + .unwrap() } - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], filters: vec![], having: None, - child: Rc::new(child), - } + child, + }) + .unwrap() + } + + fn binary( + kind: BinaryOpKind, + lhs: Rc, + rhs: Rc, + ) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + kind, + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, + lhs, + rhs, + }) + .unwrap() } // ── SketchApproximation ────────────────────────────────────────────── @@ -563,7 +586,7 @@ mod tests { } /// `node_hash` must be the literal `structural_hash` a downstream - /// consumer would compute over the *same* `QueryExpr` subtree via + /// consumer would compute over the *same* `OperatorNode` sub-DAG via /// `asap_types::dag_export::export` — the whole point of carrying it is /// that two independent exports of the same tree agree, with no /// string-matching against `location` required. @@ -582,7 +605,7 @@ mod tests { Some(sketch.node_hash), expected_hash, "ReplacementExplanation::node_hash must match dag_export's DagNode::hash \ - for the same QueryExpr subtree" + for the same OperatorNode subtree" ); } @@ -637,21 +660,18 @@ mod tests { /// A sketch-applicable `Aggregate` reachable via two paths that CSE /// collapses onto one `Rc` — the same `median(x) == median(x)` shape - /// `pre_asap::cse`'s own `single_query_shares_its_own_repeated_subtree` + /// `pre_asap::cse`'s own `single_query_shares_its_own_repeated_sub-DAG` /// test uses — must be reported once, not once per path: it is exactly /// one [`crate::replacement::TargetSubDAGCandidates`], keyed by `Rc` pointer identity, /// not one per path that reaches it. #[test] fn a_shared_sketchable_aggregate_is_reported_only_once() { let quantile = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let root = QueryExpr::BinaryOp { - op: asap_types::pre_asap::query_expr::BinaryOpKind::Compare( - asap_types::pre_asap::expr_ir::CompareOpKind::Eq, - ), - lhs: Rc::new(quantile.clone()), - rhs: Rc::new(quantile), - vector_match: None, - }; + let root = binary( + BinaryOpKind::Compare(asap_types::pre_asap::expr_ir::CompareOpKind::Eq), + Rc::clone(&quantile), + quantile, + ); let findings = explain_replacements(vec![("ratio", root)]); let sketch: Vec<_> = findings .iter() @@ -670,7 +690,7 @@ mod tests { #[test] fn two_roots_with_the_same_grouped_aggregate_share_a_reuse_finding() { // Grouped (`by (job)`), so the shared `Aggregate`'s output schema - // carries a provable unique key — share_common_subtrees's legality + // carries a provable unique key — share_common_subdags's legality // gate — and identical across both roots, so it is shareable. let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); @@ -691,8 +711,8 @@ mod tests { #[test] fn descendant_of_a_shared_root_keeps_every_root_breadcrumb() { let inner = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let outer = agg(vec![2], AggIntent::Sum { col: None }, inner); - let findings = explain_replacements(vec![("dash_a", outer.clone()), ("dash_b", outer)]); + let outer = agg(vec![0], AggIntent::Sum { col: None }, inner); + let findings = explain_replacements(vec![("dash_a", Rc::clone(&outer)), ("dash_b", outer)]); let inner_sketch = findings .iter() .find(|f| { @@ -722,7 +742,7 @@ mod tests { #[test] fn ungrouped_identical_aggregates_are_not_shareable_so_no_finding() { - // Empty `by`: no provable unique key — share_common_subtrees never + // Empty `by`: no provable unique key — share_common_subdags never // hoists these, so consumer_count stays 1 for each and this module // must not report a finding either. let a = agg(vec![], AggIntent::Sum { col: None }, metric_scan(&["job"])); @@ -738,14 +758,11 @@ mod tests { // The same shared branch appearing twice within one query (an `a/a` // shape) — single-query CSE. let branch = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let q = QueryExpr::BinaryOp { - op: asap_types::pre_asap::query_expr::BinaryOpKind::Arithmetic( - asap_types::pre_asap::expr_ir::ArithmeticOpKind::Div, - ), - lhs: Rc::new(branch.clone()), - rhs: Rc::new(branch), - vector_match: None, - }; + let q = binary( + BinaryOpKind::Arithmetic(asap_types::pre_asap::expr_ir::ArithmeticOpKind::Div), + Rc::clone(&branch), + branch, + ); let findings = explain_replacements(vec![("ratio", q)]); let reuse: Vec<_> = findings .iter() @@ -762,7 +779,7 @@ mod tests { /// A shared node nested three levels under two *different*, unshared /// `Filter` parents (mirrors `crate::replacement::tests:: - /// nested_shared_subtree_below_an_unshared_parent_is_still_discovered`) + /// nested_shared_sub-DAG_below_an_unshared_parent_is_still_discovered`) /// must still be exactly one finding — the maximal-`TargetSubDAG` /// guarantee the module docs describe, now provided by /// `crate::replacement`'s own target discovery rather than this module's @@ -770,17 +787,18 @@ mod tests { #[test] fn a_deeply_shared_subtree_under_different_parents_is_reported_once() { use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let root_a = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(1)))), - child: Rc::new(shared.clone()), - }; - let root_b = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(2)))), - child: Rc::new(shared), - }; + let root_a = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(1))), + child: Rc::clone(&shared), + }) + .unwrap(); + let root_b = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(2))), + child: shared, + }) + .unwrap(); let findings = explain_replacements(vec![("a", root_a), ("b", root_b)]); let reuse: Vec<_> = findings .iter() @@ -821,7 +839,7 @@ mod tests { let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let custom_model = AlwaysDDSketch; let strategies: Vec> = vec![Box::new( - crate::replacement::SketchAlgorithmStrategy::new(&custom_model), + crate::replacement::ASAPStrategies::new(&custom_model), )]; let findings = explain_replacements_with(vec![("q", q)], &strategies); assert_eq!(findings.len(), 1); diff --git a/crates/asap-aware-mapping/src/grouping.rs b/crates/asap-aware-mapping/src/grouping.rs index b2899f0d4..48b3efb58 100644 --- a/crates/asap-aware-mapping/src/grouping.rs +++ b/crates/asap-aware-mapping/src/grouping.rs @@ -7,9 +7,9 @@ //! //! ## Placement: planning metadata and edge-state type //! -//! `SummaryExpr::SummaryAgg` carries the grouping choice next to the +//! `ASAPOp::SummaryAgg` carries the grouping choice next to the //! `Reduction` whose `by` keys determine legality. The same choice is also -//! committed to `SummaryFamilyType::Sketch` on the aggregate's output edge. +//! committed to `FieldDataType::Sketch` on the aggregate's output edge. //! That duplication is intentional: the node field makes the choice easy to //! inspect during planning, while the edge type ensures an independent KLL/ //! CMS state and a Hydra-backed state cannot be accepted as compatible inputs @@ -41,7 +41,7 @@ //! An earlier draft of this module (written against the very first draft of //! #251) reused a `CostModel`-wrapping adapter that "steered" a //! whole-recursive-bind decision procedure toward a specific `SketchKind`, -//! the same pattern [`crate::replacement::SketchAlgorithmStrategy`]'s own module +//! the same pattern [`crate::replacement::ASAPStrategies`]'s own module //! docs explain was deliberately deleted from this crate as an anti-pattern: //! forcing a choice via a whole-tree `CostModel` adapter had a real bug where //! the forced choice could leak into a target's own nested aggregates. This @@ -53,7 +53,7 @@ //! passes that exact, //! already-decided `Realization` to //! [`crate::replacement::construct_summary`] — the same first-class, -//! one-candidate-at-a-time primitive [`crate::replacement::SketchAlgorithmStrategy`] +//! one-candidate-at-a-time primitive [`crate::replacement::ASAPStrategies`] //! itself calls once per candidate. No adapter, no steering, no risk of a //! forced choice leaking into nested aggregates. //! @@ -71,13 +71,14 @@ use std::rc::Rc; +use asap_types::ir::operator_properties::Reduction; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; use asap_types::post_asap::{ default_hydra_params, hydra_kind_for, AccuracyError, BoundExpr, CompositionOperator, - GroupingStrategy, GuaranteeSource, HydraKind, ProbabilityExpr, ResultGuarantee, - SketchAlgorithm, SketchParams, SummaryExpr, SummaryFamilyType, SummaryNode, + FieldDataType, GroupingStrategy, GuaranteeSource, HydraKind, ProbabilityExpr, ResultGuarantee, + SketchAlgorithm, SketchParams, }; use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; use crate::accuracy::{ AccuracyBudgetAllocator, AccuracyEvidenceProvider, AccuracyModel, PropagationStats, @@ -110,15 +111,15 @@ pub fn has_subpopulations(reduction: &Reduction) -> bool { /// A single static instance so [`HydraGroupingStrategy::default_cost_model`] /// can hand out a `&'static dyn CostModel` without heap-allocating one — same -/// pattern [`crate::replacement::SketchAlgorithmStrategy`] uses. +/// pattern [`crate::replacement::ASAPStrategies`] uses. static DEFAULT_COST_MODEL: DefaultCostModel = DefaultCostModel; /// Wraps the `GroupingStrategy` axis (issue #256) as a -/// [`ReplacementStrategy`]: for a target [`SketchAlgorithmStrategy`](crate::replacement::SketchAlgorithmStrategy) +/// [`ReplacementStrategy`]: for a target [`ASAPStrategies`](crate::replacement::ASAPStrategies) /// already has an opinion on, offers an additional /// `GroupingStrategy::SharedMultiSubpopulation` candidate wherever the /// legality conditions in the module docs above hold — alongside, not -/// instead of, the per-subpopulation candidates `SketchAlgorithmStrategy` +/// instead of, the per-subpopulation candidates `ASAPStrategies` /// itself enumerates. The workload search composes both strategies over the /// same target, so it sees every summary-family alternative *and* the Hydra /// alternative; the built-in workload search registers both strategies, and @@ -132,7 +133,7 @@ pub struct HydraGroupingStrategy<'a> { impl HydraGroupingStrategy<'static> { /// A strategy that ranks/binds via the built-in [`DefaultCostModel`] — /// what a deployment gets with no custom cost model plugged in, the same - /// default [`crate::replacement::SketchAlgorithmStrategy::default_cost_model`] + /// default [`crate::replacement::ASAPStrategies::default_cost_model`] /// offers. pub fn default_cost_model() -> Self { Self { @@ -144,7 +145,7 @@ impl HydraGroupingStrategy<'static> { impl<'a> HydraGroupingStrategy<'a> { /// A strategy that ranks/binds via `cost_model` instead of the built-in /// static preference order — the same customization point - /// [`crate::replacement::SketchAlgorithmStrategy::new`] already offers. + /// [`crate::replacement::ASAPStrategies::new`] already offers. pub fn new(cost_model: &'a dyn CostModel) -> Self { Self { planning_inputs: CandidatePlanningInputs::with_default_accuracy(cost_model), @@ -173,7 +174,7 @@ impl<'a> HydraGroupingStrategy<'a> { /// variant modeled. fn hydra_proposals(&self, target: &TargetSubDAG<'_>) -> Proposals { let mut proposals = Proposals::default(); - let QueryExpr::Aggregate { reduction, .. } = target.root.as_ref() else { + let Some(NonASAPOp::Aggregate { reduction, .. }) = target.root.non_asap() else { return proposals; }; if !has_subpopulations(reduction) { @@ -208,11 +209,11 @@ impl<'a> HydraGroupingStrategy<'a> { /// `PerSubpopulationInstance` to /// `SharedMultiSubpopulation { kind: hydra_kind, .. }` — reusing the /// entire bind decision procedure (schema derivation, column resolution, - /// readout construction) unchanged, patching only the one field this + /// evaluation construction) unchanged, patching only the one field this /// axis owns. fn build_candidate( &self, - root: &Rc, + root: &Rc, intent: &AggIntent, sketch_kind: SketchAlgorithm, hydra_kind: HydraKind, @@ -233,12 +234,12 @@ impl<'a> HydraGroupingStrategy<'a> { params, }; - let (family, query) = match &node.expr { - SummaryExpr::SummaryEstimate { + let (family, query) = match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } => match &summary_input.expr { - SummaryExpr::SummaryAgg { family, .. } => (family, Some(query)), + }) => match &summary_input.operator { + Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) => (family, Some(query)), _ => return None, }, _ => return None, @@ -295,7 +296,7 @@ impl<'a> HydraGroupingStrategy<'a> { } Some(ReplacementSubDAG { strategy: "HydraGroupingStrategy", - replacement: Replacement::Summary(patched), + replacement: Replacement::SubDag(patched), provenance: crate::replacement::ReplacementProvenance::SummaryRealization, rationale: format!( "{} realizes as a shared {hydra_kind:?} structure over {sketch_kind:?} \ @@ -313,7 +314,7 @@ impl<'a> HydraGroupingStrategy<'a> { impl ReplacementStrategy for HydraGroupingStrategy<'_> { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { - let QueryExpr::Aggregate { reduction, .. } = target.root.as_ref() else { + let Some(NonASAPOp::Aggregate { reduction, .. }) = target.root.non_asap() else { return false; }; if !has_subpopulations(reduction) { @@ -357,15 +358,15 @@ impl ReplacementStrategy for HydraGroupingStrategy<'_> { /// destructures the right variant for `kind`; this function's only job is /// to find whatever `SketchParams` the bind decision already committed to /// and hand the whole thing over unchanged. -fn per_subpopulation_sketch_params(node: &SummaryNode) -> Option { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => { +fn per_subpopulation_sketch_params(node: &OperatorNode) -> Option { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { per_subpopulation_sketch_params(summary_input) } - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), .. - } => Some(kind.params().clone()), + }) => Some(kind.params().clone()), _ => None, } } @@ -373,47 +374,47 @@ fn per_subpopulation_sketch_params(node: &SummaryNode) -> Option { /// Rebuild `node`, replacing its `SummaryAgg`'s `grouping` field with /// `grouping` — patching the one field this axis owns onto an /// already-correctly-bound node rather than re-deriving the rest of it. -/// Recurses through a `SummaryEstimate` readout wrapper (the shape every +/// Recurses through a `SummaryEstimate` evaluation wrapper (the shape every /// sketch candidate this module builds actually has) to reach the /// `SummaryAgg` underneath. fn with_grouping( - node: Rc, + node: Rc, grouping: GroupingStrategy, stats: &PropagationStats, -) -> Rc { - match &node.expr { - SummaryExpr::SummaryEstimate { +) -> Rc { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } => Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { + }) => OperatorNode::asap_node( + ASAPOp::SummaryEstimate { summary_input: with_grouping(Rc::clone(summary_input), grouping, stats), query: query.clone(), }, - schema: node.schema.clone(), - guarantee: node.guarantee.as_ref().map(|g| hydra_guarantee(g, stats)), - }), - SummaryExpr::SummaryAgg { + node.schema.clone(), + node.guarantee.as_ref().map(|g| hydra_guarantee(g, stats)), + ), + Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, reduction, .. - } => { + }) => { let grouped_family = match family { - SummaryFamilyType::Sketch(kind, _) => { - SummaryFamilyType::Sketch(kind.clone(), grouping.clone()) + FieldDataType::Sketch(kind, _) => { + FieldDataType::Sketch(kind.clone(), grouping.clone()) } _ => family.clone(), }; let mut grouped_schema = node.schema.clone(); for field in &mut grouped_schema.fields { - if let SummaryFamilyType::Sketch(kind, _) = &field.dtype { - field.dtype = SummaryFamilyType::Sketch(kind.clone(), grouping.clone()); + if let FieldDataType::Sketch(kind, _) = &field.dtype { + field.dtype = FieldDataType::Sketch(kind.clone(), grouping.clone()); } } - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { + OperatorNode::asap_node( + ASAPOp::SummaryAgg { child: Rc::clone(child), family: grouped_family, input: input.clone(), @@ -421,9 +422,9 @@ fn with_grouping( grouping, filter: None, }, - schema: grouped_schema, - guarantee: None, - }) + grouped_schema, + None, + ) } // Never reached by this module's own callers (they only ever pass a // node `construct_summary_with` just bound for a `Sketch` @@ -492,47 +493,11 @@ fn hydra_guarantee(inner: &ResultGuarantee, stats: &PropagationStats) -> ResultG mod tests { use super::*; use crate::accuracy::{DefaultAccuracyModel, EqualSplitAllocator}; + use crate::test_support::{agg, agg_per_entity, metric_scan}; use asap_types::post_asap::ErrorMetric; use asap_types::pre_asap::agg_intent::{default_cardinality, default_quantile}; - use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{Column, DataType, Schema}; use asap_types::types::AccuracyTarget; - fn metric_scan(labels: &[&str]) -> QueryExpr { - let mut columns = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - ]; - columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index(columns, 0, vec![]), - } - } - - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::by(by), - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } - - fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } - // ── has_subpopulations ──────────────────────────────────────────────── #[test] @@ -552,7 +517,7 @@ mod tests { #[test] fn without_grouping_has_a_subpopulation_concept_even_when_empty() { - use asap_types::pre_asap::query_expr::GroupKeys; + use asap_types::ir::operator_properties::GroupKeys; // `without([])` groups by every remaining label — a real // subpopulation concept, unlike `by([])`'s genuine full reduction. assert!(has_subpopulations(&Reduction::Reduce(GroupKeys::without( @@ -599,7 +564,7 @@ mod tests { delta: 0.01, }, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); assert!(HydraGroupingStrategy::default_cost_model().matches(&target)); } @@ -607,7 +572,7 @@ mod tests { #[test] fn does_not_match_an_empty_by_aggregate() { // Global reduction — no subpopulation concept, no Hydra alternative. - let q = Rc::new(agg(vec![], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let strategy = HydraGroupingStrategy::default_cost_model(); assert!(!strategy.matches(&target)); @@ -616,10 +581,7 @@ mod tests { #[test] fn does_not_match_a_per_entity_aggregate() { - let q = Rc::new(agg_per_entity( - default_quantile(0.99), - metric_scan(&["job"]), - )); + let q = agg_per_entity(default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let strategy = HydraGroupingStrategy::default_cost_model(); assert!(!strategy.matches(&target)); @@ -628,14 +590,14 @@ mod tests { #[test] fn does_not_match_a_non_aggregate_node() { - let scan = Rc::new(metric_scan(&["job"])); + let scan = metric_scan(&["job"]); let target = TargetSubDAG::new(&scan); assert!(!HydraGroupingStrategy::default_cost_model().matches(&target)); } #[test] fn quantile_has_no_hydra_candidate_without_a_modeled_error_bound() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let replacements = HydraGroupingStrategy::default_cost_model().replacements(&target); assert!(replacements.is_empty(), "{replacements:?}"); @@ -649,13 +611,13 @@ mod tests { delta: 0.01, }, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let replacements = HydraGroupingStrategy::default_cost_model().replacements(&target); assert_eq!(replacements.len(), 2, "{replacements:?}"); assert!(replacements.iter().all(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) + Replacement::SubDag(node) if node.guarantee.as_ref().is_some_and(|guarantee| guarantee.bound.evaluate().is_none() && guarantee.failure_probability.evaluate().is_none()) @@ -668,8 +630,8 @@ mod tests { fn propagation_stats( &self, _op: &CompositionOperator, - _family: &SummaryFamilyType, - _query: Option<&asap_types::post_asap::SketchQuery>, + _family: &FieldDataType, + _query: Option<&asap_types::post_asap::SketchStatistic>, ) -> PropagationStats { PropagationStats { hydra_shared_grid_collision_bound: Some(0.0), @@ -687,7 +649,7 @@ mod tests { delta: 0.01, }, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let strategy = HydraGroupingStrategy::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, @@ -698,7 +660,7 @@ mod tests { assert_eq!(replacements.len(), 2, "{replacements:?}"); assert!(replacements.iter().all(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) + Replacement::SubDag(node) if node.guarantee.as_ref().is_some_and(|g| g.bound.evaluate().is_some() && g.failure_probability.evaluate().is_some()) @@ -712,8 +674,8 @@ mod tests { fn propagation_stats( &self, _op: &CompositionOperator, - _family: &SummaryFamilyType, - _query: Option<&asap_types::post_asap::SketchQuery>, + _family: &FieldDataType, + _query: Option<&asap_types::post_asap::SketchStatistic>, ) -> PropagationStats { PropagationStats { hydra_shared_grid_failure_probability: Some(1.5), @@ -727,7 +689,7 @@ mod tests { delta: 0.01, }, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let strategy = HydraGroupingStrategy::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, @@ -758,8 +720,8 @@ mod tests { fn propagation_stats( &self, _op: &CompositionOperator, - _family: &SummaryFamilyType, - _query: Option<&asap_types::post_asap::SketchQuery>, + _family: &FieldDataType, + _query: Option<&asap_types::post_asap::SketchStatistic>, ) -> PropagationStats { PropagationStats { hydra_shared_grid_collision_bound: Some(0.1), @@ -767,7 +729,7 @@ mod tests { } } } - let q = Rc::new(agg( + let q = agg( vec![2], AggIntent::Count { accuracy: AccuracyTarget::EpsilonDelta { @@ -776,7 +738,7 @@ mod tests { }, }, metric_scan(&["job"]), - )); + ); let strategy = HydraGroupingStrategy::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, @@ -797,7 +759,7 @@ mod tests { // summary_candidates(Cardinality) = [Hll, Theta, Kmv] — none have a // modeled Hydra variant, so no candidate at all (not an error, just // an empty result, same conservatism as every other strategy here). - let q = Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))); + let q = agg(vec![2], default_cardinality(), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let strategy = HydraGroupingStrategy::default_cost_model(); assert!(!strategy.matches(&target)); @@ -813,7 +775,7 @@ mod tests { q: 0.99, accuracy: AccuracyTarget::Exact, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let strategy = HydraGroupingStrategy::default_cost_model(); assert!(!strategy.matches(&target)); @@ -824,11 +786,7 @@ mod tests { fn exact_mergeable_intent_has_no_hydra_candidate() { // Sum's exact accumulator has no candidate summary families at all // (summary_candidates only covers approximate-capable intents). - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let strategy = HydraGroupingStrategy::default_cost_model(); assert!(!strategy.matches(&target)); @@ -839,14 +797,15 @@ mod tests { fn does_not_match_a_multi_intent_or_having_aggregate() { let strategy = HydraGroupingStrategy::default_cost_model(); - let multi = Rc::new(QueryExpr::Aggregate { + let multi = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&["job"])), - }); + child: metric_scan(&["job"]), + }) + .unwrap(); let target = TargetSubDAG::new(&multi); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); @@ -855,7 +814,7 @@ mod tests { /// A custom `CostModel` doesn't change *which* candidate is offered — /// only which sketch candidate `realizations_for_intent` itself would /// have ranked first, and how that candidate's own params are sized — - /// same guarantee `SketchAlgorithmStrategy` makes for its own candidates. + /// same guarantee `ASAPStrategies` makes for its own candidates. struct PreferDDSketch; impl CostModel for PreferDDSketch { fn rank_candidates( @@ -874,7 +833,7 @@ mod tests { #[test] fn custom_cost_model_cannot_enable_unproven_hydra_kll() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let custom = PreferDDSketch; let replacements = HydraGroupingStrategy::new(&custom).replacements(&target); diff --git a/crates/asap-aware-mapping/src/lib.rs b/crates/asap-aware-mapping/src/lib.rs index 3af72bf5f..47d11a05a 100644 --- a/crates/asap-aware-mapping/src/lib.rs +++ b/crates/asap-aware-mapping/src/lib.rs @@ -1,18 +1,18 @@ //! `asap-plan` — the cost-aware optimizer layer over the pre-ASAP intent algebra. //! //! This crate sits between the language-agnostic IR ([`asap_ir`]) and -//! any runtime: it consumes pre-ASAP [`QueryExpr`](asap_types::pre_asap::QueryExpr) -//! trees and makes the cost-aware decisions the pre-ASAP IR deliberately +//! any runtime: it consumes pre-ASAP [`OperatorNode`](asap_types::ir::OperatorNode) +//! DAGs and makes the cost-aware decisions the pre-ASAP IR deliberately //! leaves open — which sketch (if any) realises each approximate intent. //! //! **Common sub-expression elimination (CSE) is not this crate's job.** -//! Detection is a primary pass over the pre-ASAP `QueryExpr` IR itself -//! (`asap_types::pre_asap`, design tracked in issue #223), run before a -//! tree ever reaches [`replacement::SketchAlgorithmStrategy`] — see issue #222 +//! Detection is a primary pass over the pre-ASAP operator IR itself +//! (`asap_types::ir::cse`, design tracked in issue #223), run before a +//! tree ever reaches [`replacement::ASAPStrategies`] — see issue #222 //! for why (batch query optimization needs to see shared work across a //! `QueryWorkload` before summary binding, not after). This crate may //! eventually run a second, narrower CSE pass of its own over an -//! already-bound `SummaryExpr`/`SummaryNode` DAG, recognizing sharing that's invisible +//! already-bound post-ASAP `OperatorNode` DAG, recognizing sharing that's invisible //! at the pre-ASAP level by construction — e.g. `Quantile(x, 0.99)` and //! `Quantile(x, 0.95)` are structurally distinct `AggIntent`s but can //! still share one built sketch, read out twice. That post-ASAP pass is @@ -83,7 +83,7 @@ //! re-deriving it from an already-computed, strictly finer sibling //! `Aggregate` over identical child IR instead of an independent pass //! over the raw source — the cross-aggregate sibling of -//! `pre_asap::cse::share_common_subtrees`'s identical-subtree sharing. +//! `pre_asap::cse::share_common_subdags`'s identical-sub-DAG sharing. //! [`rollup::is_legal_rollup_source`] is the standalone legality predicate //! other axes (e.g. issue #256's `GroupingStrategy`) are expected to //! consult directly, so it and this module's `RollupStrategy` can never @@ -92,7 +92,7 @@ //! #33) is an additional `ReplacementStrategy`: the orthogonal //! `GroupingStrategy` axis (one summary instance per `by` subpopulation //! versus one shared Hydra-family structure serving all of them), offered -//! alongside the candidates [`replacement::SketchAlgorithmStrategy`] +//! alongside the candidates [`replacement::ASAPStrategies`] //! enumerates for the same target. //! - [`rewrite`] — the "semantic-equivalent rewriting (e.g. `avg` → //! `sum`/`count`) to increase how often the [sharing/sketch] optimizations @@ -101,7 +101,7 @@ //! is a [`replacement::ReplacementStrategy`] that reshapes a bare `avg` //! node — which [`replacement::realizations_for_intent`] can only //! dispatch to `Realization::PassThrough`, so it can never be a -//! [`replacement::SharedSubtreeStrategy`] target — into a `sum`/`count` +//! [`replacement::SharedSubDagStrategy`] target — into a `sum`/`count` //! pair under the same grouping, re-divided back by a wrapping `Project`, //! so those *are* ordinary mergeable accumulators sharing/sketching can //! reach. It only reshapes; [`replacement::search_workload`]'s cost-based @@ -117,7 +117,7 @@ //! |---|---|---| //! | Schema resolution | Derive input schemas and resolve column names to positions | `asap_types::pre_asap::SchemaResolver::resolve_schema`, `resolve_root` | //! | Realization | Enumerate ranked physical forms for one aggregate intent | `replacement::realizations_for_intent` | -//! | Replacement | Construct each candidate summary sub-DAG | [`replacement::SketchAlgorithmStrategy`] | +//! | Replacement | Construct each candidate summary sub-DAG | [`replacement::ASAPStrategies`] | //! | Search | Enumerate and compare alternatives across a workload | [`replacement::search_workload`] | //! | Runtime placement | Choose deployment locations and concrete executors | Downstream physical plan providers | //! @@ -207,12 +207,12 @@ pub use recurrence::{ UpdateRate, }; pub use replacement::{ - default_strategies, default_strategies_with, search_workload, search_workload_with, - search_workload_with_targets, summary_candidates, CandidateLogicalASAPDAGs, - CompositionDecision, GlobalSelection, Matcher, Proposals, RankedTargetSubDAGCandidates, - Realization, RealizationError, RecurrenceProfileMap, RejectedCandidate, Replacement, - ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, SharedSubtreeStrategy, - SketchAlgorithmStrategy, TargetSubDAG, TargetSubDAGCandidates, TargetSubDAGSelection, + default_strategies, default_strategies_with, is_logical_rewrite, search_workload, + search_workload_with, search_workload_with_targets, summary_candidates, ASAPStrategies, + CandidateLogicalASAPDAGs, CompositionDecision, GlobalSelection, Matcher, Proposals, + RankedTargetSubDAGCandidates, Realization, RealizationError, RecurrenceProfileMap, + RejectedCandidate, Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, + SharedSubDagStrategy, TargetSubDAG, TargetSubDAGCandidates, TargetSubDAGSelection, MAX_SEARCH_ITERATIONS, }; pub use rewrite::{AvgToSumOverCountStrategy, SemanticEquivalentRewriteStrategy}; diff --git a/crates/asap-aware-mapping/src/maintained_population.rs b/crates/asap-aware-mapping/src/maintained_population.rs index 8d9460c14..397de2c89 100644 --- a/crates/asap-aware-mapping/src/maintained_population.rs +++ b/crates/asap-aware-mapping/src/maintained_population.rs @@ -2,42 +2,29 @@ use crate::replacement::{ Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; -use asap_types::post_asap::{ - maintained_population::*, ExecutionTiming, ResultGuarantee, SummaryExpr, SummaryFamilyType, - SummaryField, SummaryNode, SummarySchema, ValueOperation, -}; -use asap_types::pre_asap::{ - any_measure_filtered, AggIntent, CompareOpKind, DataType, QueryExpr, Reduction, ScalarValue, - Schema, Source, -}; +use asap_types::ir::non_asap::any_measure_filtered; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, ScalarExpr}; +use asap_types::post_asap::{maintained_population::*, ResultGuarantee, Schema}; +use asap_types::pre_asap::{AggIntent, CompareOpKind, DataType, Reduction, ScalarValue, Source}; use std::rc::Rc; -fn plain(schema: Schema) -> SummarySchema { - SummarySchema { - time_index: schema.time_index, - fields: schema - .columns - .into_iter() - .map(|c| SummaryField { - name: c.name, - dtype: SummaryFamilyType::Plain(c.dtype), - nullable: c.nullable, - }) - .collect(), - } +fn plain(schema: Schema) -> Schema { + Schema::lifted(schema.fields, schema.time_index) } -fn strip_projection(mut root: &QueryExpr) -> &QueryExpr { - while let QueryExpr::Project { child, .. } = root { +fn strip_projection(mut root: &OperatorNode) -> &OperatorNode { + while let Some(NonASAPOp::Project { child, .. }) = root.non_asap() { root = child; } root } -fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadout, Rc)> { +fn recognize( + root: &OperatorNode, +) -> Option<(MaintainedPopulation, PopulationStatistic, Rc)> { let root = strip_projection(root); - let (source, grouping, readout, value_column) = match root { - QueryExpr::Aggregate { + let (source, grouping, evaluation, value_column) = match root.non_asap()? { + NonASAPOp::Aggregate { child, reduction: Reduction::Reduce(grouping), measures, @@ -51,39 +38,40 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou if any_measure_filtered(filters) { return None; } - let (col, readout) = match intent { + let (col, evaluation) = match intent { AggIntent::Quantile { q, col, .. } if q.is_finite() => { - (*col, PopulationReadout::Quantile { q: *q }) + (*col, PopulationStatistic::Quantile { q: *q }) } - AggIntent::TopK { k, .. } => (None, PopulationReadout::TopK { k: *k }), - AggIntent::Sum { col } => (*col, PopulationReadout::Sum), - AggIntent::Count { .. } => (None, PopulationReadout::Count), - AggIntent::Avg { col } => (*col, PopulationReadout::Average), + AggIntent::TopK { k, .. } => (None, PopulationStatistic::TopK { k: *k }), + AggIntent::Sum { col } => (*col, PopulationStatistic::Sum), + AggIntent::Count { .. } => (None, PopulationStatistic::Count), + AggIntent::Avg { col } => (*col, PopulationStatistic::Average), _ => return None, }; - let schema = child.output_schema().ok()?; - if col.is_some_and(|c| schema.columns.get(c).is_none()) { + let schema = &child.schema; + if col.is_some_and(|c| schema.fields.get(c).is_none()) { return None; } - (child, grouping, readout, col) + (child, grouping, evaluation, col) } - QueryExpr::Limit { - n, + NonASAPOp::Limit { + n: Some(n), offset: 0, child, + .. } => { - let QueryExpr::Sort { + let Some(NonASAPOp::Sort { child, keys, partition_by, - } = child.as_ref() + }) = child.non_asap() else { return None; }; let [key] = keys.as_slice() else { return None; }; - let QueryExpr::Column(col) = &key.expr else { + let ScalarExpr::Column(col) = &key.expr else { return None; }; if key.ascending { @@ -92,21 +80,21 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou ( child, partition_by, - PopulationReadout::TopK { k: *n }, + PopulationStatistic::TopK { k: *n }, Some(*col), ) } _ => return None, }; - if let QueryExpr::Scan { + if let Some(NonASAPOp::Scan { source: Source::Table { .. }, schema, .. - } = source.as_ref() + }) = source.non_asap() { let value_column = value_column.or_else(|| { schema - .columns + .fields .iter() .position(|c| c.dtype == DataType::Float64 && !c.nullable) })?; @@ -119,33 +107,33 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou max_k: 0, quantiles: false, }; - if !schema.closed || !population.matches_input(source) { + if !schema.closed || !population.matches_node(source) { return None; } - return Some((population, readout, Rc::clone(source))); + return Some((population, evaluation, Rc::clone(source))); } // A bare PromQL selector carries the declared ingestion interval as a // temporal input scope. Membership must expire at that horizon; retain // the wrapper as the maintained input so validation can check agreement. - let (series_source, lookback_ms) = match source.as_ref() { - QueryExpr::TimeRange { range, child } => { + let (series_source, lookback_ms) = match source.non_asap() { + Some(NonASAPOp::TimeRange { range, child, .. }) => { let ms = u64::try_from(range.as_millis()).ok()?; if ms == 0 || std::time::Duration::from_millis(ms) != *range { return None; } (child.as_ref(), ms) } - other => (other, 300_000), + _ => (source.as_ref(), 300_000), }; - let QueryExpr::Scan { + let Some(NonASAPOp::Scan { source: Source::TimeSeries { metric }, predicates, schema, - } = series_source + }) = series_source.non_asap() else { return None; }; - if value_column.is_some_and(|c| schema.columns.get(c).is_none_or(|c| c.name != "value")) { + if value_column.is_some_and(|c| schema.fields.get(c).is_none_or(|c| c.name != "value")) { return None; } // PromQL can retain open labels or resolve them into a complete identity column. @@ -156,15 +144,18 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou return None; } let label = |col: usize| -> Option { - let c = schema.columns.get(col)?; + let c = schema.fields.get(col)?; (c.dtype == DataType::Utf8).then(|| c.name.clone()) }; let mut matchers = Vec::new(); for predicate in predicates { - let QueryExpr::Compare { left, op, right } = predicate.0.as_ref() else { + let ScalarExpr::Compare { + left, op, right, .. + } = &predicate.0 + else { return None; }; - let (QueryExpr::Column(col), QueryExpr::Literal(ScalarValue::Utf8(value))) = + let (ScalarExpr::Column(col), ScalarExpr::Literal(ScalarValue::Utf8(value))) = (left.as_ref(), right.as_ref()) else { return None; @@ -203,90 +194,91 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou max_k: 0, quantiles: false, }, - readout, + evaluation, Rc::clone(source), )) } -/// Workload-aware rule: compatible readouts share one retractable population. +/// Workload-aware rule: compatible evaluations share one retractable population. /// Deployments opt in by registering this strategy when they can maintain complete -/// population updates and price the maintenance/readout boundary. -/// The population is exact; max_k bounds the shared readout cache, not its members. +/// population updates and price the maintenance/evaluation boundary. +/// The population is exact; max_k bounds the shared evaluation cache, not its members. pub struct MaintainedPopulationStrategy { - roots: Vec>, + roots: Vec>, } impl MaintainedPopulationStrategy { - pub fn new(roots: &[Rc]) -> Self { + pub fn new(roots: &[Rc]) -> Self { Self { roots: roots.to_vec(), } } - pub fn candidate(&self, root: &Rc) -> Option> { - if let QueryExpr::Project { + pub fn candidate(&self, root: &Rc) -> Option> { + if let Some(NonASAPOp::Project { cols, qualifier, child, - } = root.as_ref() + }) = root.non_asap() { let child = self.candidate(child)?; - return Some(Rc::new(SummaryNode { - guarantee: child.guarantee.clone(), - schema: plain(root.output_schema().ok()?), - expr: SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Project { + let guarantee = child.guarantee.clone(); + return Some(Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Project { cols: cols.clone(), qualifier: qualifier.clone(), - }, - timing: ExecutionTiming::QueryTime, - }, - })); + child, + }), + plain(root.schema.clone()), + ) + .with_guarantee(guarantee), + )); } - let (mut population, readout, source) = recognize(root)?; + let (mut population, evaluation, source) = recognize(root)?; let identity = population.clone(); for other in self.roots.iter().chain(std::iter::once(root)) { if let Some((p, r, _)) = recognize(other) { if p == identity { match r { - PopulationReadout::Quantile { .. } => population.quantiles = true, - PopulationReadout::TopK { k } => population.max_k = population.max_k.max(k), - PopulationReadout::Sum - | PopulationReadout::Count - | PopulationReadout::Average => {} + PopulationStatistic::Quantile { .. } => population.quantiles = true, + PopulationStatistic::TopK { k } => { + population.max_k = population.max_k.max(k) + } + PopulationStatistic::Sum + | PopulationStatistic::Count + | PopulationStatistic::Average => {} } } } } - let input_schema = plain(source.output_schema().ok()?); - let scan = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(source), - schema: input_schema.clone(), - guarantee: Some(ResultGuarantee::exact("source samples")), - }); - // Query time is only the initial layout: whether the population is - // retained at ingestion or rebuilt per query is its lifecycle choice - // (`SummaryMaintenanceLifecyclePlan::execution_timed_dag`). The readout - // and projection above it are query-time by construction. - let maintained = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { + let input_schema = plain(source.schema.clone()); + // The source node itself is the maintained input (a non-ASAP node + // keeps its derived schema), kept with its exact guarantee. + let scan = Rc::new( + source + .as_ref() + .clone() + .with_guarantee(Some(ResultGuarantee::exact("source samples"))), + ); + let maintained = OperatorNode::asap_node( + ASAPOp::MaintainPopulation { child: scan, - operation: ValueOperation::MaintainPopulation { population }, - timing: ExecutionTiming::QueryTime, + population, }, - schema: input_schema, - guarantee: Some(ResultGuarantee::exact( + input_schema, + Some(ResultGuarantee::exact( "exact members under the declared population semantics", )), - }); - Some(Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { + ); + Some(OperatorNode::asap_node( + ASAPOp::EvaluatePopulation { child: maintained, - operation: ValueOperation::ReadPopulation { readout }, - timing: ExecutionTiming::QueryTime, + evaluation, }, - schema: plain(root.output_schema().ok()?), - guarantee: Some(ResultGuarantee::exact("exact current-population readout")), - })) + plain(root.schema.clone()), + Some(ResultGuarantee::exact( + "exact current-population evaluation", + )), + )) } } impl ReplacementStrategy for MaintainedPopulationStrategy { @@ -297,10 +289,10 @@ impl ReplacementStrategy for MaintainedPopulationStrategy { self.candidate(target.root) .map(|node| ReplacementSubDAG { strategy: "MaintainedPopulationStrategy", - replacement: Replacement::Summary(node), + replacement: Replacement::SubDag(node), provenance: ReplacementProvenance::SummaryRealization, rationale: - "share an exact maintained population across compatible aggregate readouts" + "share an exact maintained population across compatible aggregate evaluations" .into(), }) .into_iter() @@ -312,10 +304,26 @@ impl ReplacementStrategy for MaintainedPopulationStrategy { mod tests { use super::*; use crate::test_support::lower_promql; - use asap_types::post_asap::{compile_post_asap_dag, share_common_summary_subtrees}; + use asap_types::ir::cse::share_common_subdags; + use asap_types::ir::export::compile_post_asap_dag as export_timed; + use asap_types::ir::timing::{apply_lifecycle_timings, LifecycleAssignment, TimingMemo}; - fn lower(q: &str) -> Rc { - Rc::new(lower_promql(q, asap_types::types::AccuracyTarget::Exact)) + /// Time `root` under the default lifecycle assignment (which runs the + /// data-state / population-contract validation) and export it. + fn compile_post_asap_dag(root: &Rc) -> Result<(), String> { + root.validate_structure().map_err(|e| e.to_string())?; + let timed = apply_lifecycle_timings( + root, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .map_err(|e| format!("{e:?}"))?; + export_timed(&timed).map_err(|e| format!("{e:?}"))?; + Ok(()) + } + + fn lower(q: &str) -> Rc { + lower_promql(q, asap_types::types::AccuracyTarget::Exact) } // Instant scalar aggregations share the same retractable series population. @@ -340,7 +348,7 @@ mod tests { } } - // Different readout parameters retain one shared maintenance producer in the DAG. + // Different evaluation parameters retain one shared maintenance producer in the DAG. #[test] fn quantiles_and_topk_share_a_planner_population() { let roots: Vec<_> = [ @@ -364,7 +372,7 @@ mod tests { .target_subdag_candidates() .flat_map(|g| &g.candidates) .any(|c| c.strategy == "MaintainedPopulationStrategy")); - let plans = share_common_summary_subtrees( + let plans = share_common_subdags( roots .iter() .enumerate() @@ -374,18 +382,10 @@ mod tests { let mut producers = Vec::new(); for (_, plan) in &plans { compile_post_asap_dag(plan).unwrap(); - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::ReadPopulation { .. }, - .. - } = &plan.expr - else { - panic!("missing typed readout") + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &plan.operator else { + panic!("missing typed evaluation") }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &child.expr + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &child.operator else { panic!("missing maintained population") }; @@ -411,14 +411,10 @@ mod tests { let (p, _, _) = recognize(&roots[0]).unwrap(); assert!(matches!(p.input, PopulationInput::CurrentSeries(ref s) if s.grouping.is_empty())); let candidate = strategy.candidate(&roots[0]).unwrap(); - let SummaryExpr::ValueOperation { child, .. } = &candidate.expr else { + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &candidate.operator else { unreachable!() }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &child.expr - else { + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &child.operator else { unreachable!() }; assert_eq!(population.max_k, 5); @@ -445,23 +441,22 @@ mod tests { assert_eq!(p.matchers[0].operation, CurrentSeriesMatch::Regex); } // Population timing is a lifecycle choice: a retained or rebuilt - // population both validate, while its readout must stay at query time. + // population both validate, while its evaluation must stay at query time. #[test] fn population_timing_is_not_structural() { let root = lower("topk(5,a)"); let candidate = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) .candidate(&root) .unwrap(); - let with_timings = |population: ExecutionTiming, readout: ExecutionTiming| { + use asap_types::post_asap::ExecutionTiming; + let with_timings = |population: ExecutionTiming, evaluation: ExecutionTiming| { let mut node = (*candidate).clone(); - let SummaryExpr::ValueOperation { child, timing, .. } = &mut node.expr else { - unreachable!() - }; - *timing = readout; - let SummaryExpr::ValueOperation { timing, .. } = &mut Rc::make_mut(child).expr else { + node.timing = Some(evaluation); + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &mut node.operator + else { unreachable!() }; - *timing = population; + Rc::make_mut(child).timing = Some(population); compile_post_asap_dag(&Rc::new(node)) }; use ExecutionTiming::{IngestionTime, QueryTime}; @@ -469,34 +464,27 @@ mod tests { assert!(with_timings(QueryTime, QueryTime).is_ok()); assert!(with_timings(IngestionTime, IngestionTime).is_err()); } - // A readout cannot reinterpret arbitrary rows as maintained state or exceed its producer's contract. + // A evaluation cannot reinterpret arbitrary rows as maintained state or exceed its producer's contract. #[test] fn malformed_population_dags_fail_closed() { let root = lower("topk(5,a)"); let strategy = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)); let candidate = strategy.candidate(&root).unwrap(); + compile_post_asap_dag(&candidate).expect("the unmodified candidate is legal"); let mut bad = (*candidate).clone(); - let SummaryExpr::ValueOperation { operation, .. } = &mut bad.expr else { + let Operator::ASAP(ASAPOp::EvaluatePopulation { evaluation, .. }) = &mut bad.operator + else { unreachable!() }; - *operation = ValueOperation::ReadPopulation { - readout: PopulationReadout::TopK { k: 6 }, - }; + *evaluation = PopulationStatistic::TopK { k: 6 }; assert!(compile_post_asap_dag(&Rc::new(bad.clone())).is_err()); - let SummaryExpr::ValueOperation { - child, operation, .. - } = &mut bad.expr + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, evaluation }) = &mut bad.operator else { unreachable!() }; - *operation = ValueOperation::ReadPopulation { - readout: PopulationReadout::TopK { k: 5 }, - }; + *evaluation = PopulationStatistic::TopK { k: 5 }; let producer = Rc::make_mut(child); - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &mut producer.expr + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &mut producer.operator else { unreachable!() }; diff --git a/crates/asap-aware-mapping/src/pane_sharing.rs b/crates/asap-aware-mapping/src/pane_sharing.rs index b5f5bff3a..0d944b2da 100644 --- a/crates/asap-aware-mapping/src/pane_sharing.rs +++ b/crates/asap-aware-mapping/src/pane_sharing.rs @@ -1,6 +1,6 @@ //! Costed reuse of compatible physical pane producers. The executor supplies //! an equality key covering source, state, phase and evidence. This pass never -//! changes logical readout windows or assumes compatibility from metric names. +//! changes logical evaluation windows or assumes compatibility from metric names. /// A concrete mergeable-pane implementation and its horizon costs. #[derive(Debug, Clone)] @@ -9,9 +9,9 @@ pub struct PaneReuseCandidate { pub lookback_ms: u64, /// Build, update, residency and retirement for this producer. Candidates /// with the same key must use the same unit costs and pane width, making - /// the longest-lived producer sufficient for every readout in the group. + /// the longest-lived producer sufficient for every evaluation in the group. pub producer_cost: f64, - /// Readout cost for all consumers of this distinct producer. + /// Evaluation cost for all consumers of this distinct producer. pub read_cost: f64, } @@ -84,7 +84,7 @@ mod tests { read_cost: 2.0, } } - // Share source work once while retaining both readout charges and longest history. + // Share source work once while retaining both evaluation charges and longest history. #[test] fn shares_compatible_windows() { assert_eq!( diff --git a/crates/asap-aware-mapping/src/pass/major.rs b/crates/asap-aware-mapping/src/pass/major.rs index d74091af7..1b0b6ede3 100644 --- a/crates/asap-aware-mapping/src/pass/major.rs +++ b/crates/asap-aware-mapping/src/pass/major.rs @@ -6,10 +6,10 @@ //! it *one* pass rather than *the* algorithm. `ReplacementStrategy` is //! therefore a concept of this pass, not of the optimization interface. +use asap_types::ir::cse::share_common_subdags; use std::rc::Rc; -use asap_types::post_asap::{share_common_summary_subtrees, SummaryNode}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use super::{OptimizationInput, OptimizationPass, OptimizeError, PlanOutput, QueryLifecyclePlan}; @@ -39,10 +39,10 @@ impl OptimizationPass for MajorPass { // search result carries the workload binding the lifecycle stage and // the output both need. CSE may make two identical queries share one // `Rc`, but it never drops or reorders a root, so this stays aligned. - let roots: Vec<(usize, Rc, Option)> = workload + let roots: Vec<(usize, Rc, Option)> = workload .entries() - .enumerate() - .map(|(index, (entry, expr))| { + .zip(workload.operator_indices().iter().copied()) + .map(|((entry, expr), index)| { ( index, Rc::clone(expr), @@ -57,7 +57,7 @@ impl OptimizationPass for MajorPass { // One index per root, in `CandidateLogicalASAPDAGs::roots` order — which is the order // the roots went in, which is `entries()` order. - let entry_indices: Vec = (0..workload.len()).collect(); + let entry_indices = workload.operator_indices().to_vec(); let demand = WorkloadDemand { workload: workload.query_workload(), data_workload: workload.data_workload(), @@ -94,8 +94,7 @@ impl OptimizationPass for MajorPass { .ok_or_else(|| self.missing_group(*entry_index))?; assembled.push(dag); } - let interned = - share_common_summary_subtrees(assembled.iter().cloned().enumerate().collect()); + let interned = share_common_subdags(assembled.iter().cloned().enumerate().collect()); let states: Vec<_> = interned .iter() .map(|(_, dag)| summary_states(dag)) @@ -105,7 +104,7 @@ impl OptimizationPass for MajorPass { // their entries, in every plan that reaches it, so each plan picks // the same lifecycle for it. When that union cannot be costed the // roots keep their own, unshared DAG and entries. - let mut shared_entries: Vec<(Rc, Option>)> = Vec::new(); + let mut shared_entries: Vec<(Rc, Option>)> = Vec::new(); for (position, (entry_index, root)) in space.roots.iter().enumerate() { for state in &states[position] { if shared_entries.iter().any(|(s, _)| Rc::ptr_eq(s, state)) { @@ -183,7 +182,9 @@ impl OptimizationPass for MajorPass { plan, }); } - Ok(PlanOutput::new(plans)) + let mut output = PlanOutput::new(plans); + output.scalar_roots = workload.scalar_roots().to_vec(); + Ok(output) } } diff --git a/crates/asap-aware-mapping/src/pass/mod.rs b/crates/asap-aware-mapping/src/pass/mod.rs index a0b993d97..f9dd0ac2d 100644 --- a/crates/asap-aware-mapping/src/pass/mod.rs +++ b/crates/asap-aware-mapping/src/pass/mod.rs @@ -17,8 +17,8 @@ mod major; use std::collections::BTreeMap; use std::rc::Rc; +use asap_types::ir::OperatorNode; use asap_types::parsed_workload::ParsedWorkload; -use asap_types::post_asap::SummaryNode; use asap_types::workload::WorkloadError; use crate::accuracy::{ @@ -173,7 +173,7 @@ pub struct QueryLifecyclePlan { pub plan: SummaryMaintenanceLifecyclePlan, } -/// One plan per workload entry, in `QueryWorkload::entries()` order; +/// One multi-root workload DAG with query/lifecycle bindings in entry order; /// [`check_contract`] enforces that. /// /// Plans are not deduplicated across entries: a summary state that several @@ -184,25 +184,83 @@ pub struct QueryLifecyclePlan { #[non_exhaustive] pub struct PlanOutput { pub plans: Vec, + /// Exact scalar expressions, keyed by workload entry; embedded plan reads remain visible. + pub scalar_roots: Vec<(usize, asap_types::ir::ScalarExpr)>, } impl PlanOutput { pub fn new(plans: Vec) -> Self { - Self { plans } + Self { + plans, + scalar_roots: Vec::new(), + } } /// Entry indices in output order. pub fn entry_indices(&self) -> Vec { - self.plans.iter().map(|p| p.entry_index).collect() + let mut indices: Vec<_> = self + .plans + .iter() + .map(|p| p.entry_index) + .chain(self.scalar_roots.iter().map(|(i, _)| *i)) + .collect(); + if !self.scalar_roots.is_empty() { + indices.sort_unstable(); + } + indices + } + + /// All query roots in workload order, including standalone scalars. + pub fn roots(&self) -> Vec { + let mut roots: Vec<_> = self + .plans + .iter() + .map(|p| { + ( + p.entry_index, + asap_types::ir::QueryRoot::Operator(Rc::clone(&p.plan.root)), + ) + }) + .chain( + self.scalar_roots + .iter() + .map(|(i, expr)| (*i, asap_types::ir::QueryRoot::Scalar(expr.clone()))), + ) + .collect(); + roots.sort_by_key(|(i, _)| *i); + roots.into_iter().map(|(_, root)| root).collect() } - /// The selected DAG root per query. - pub fn dags(&self) -> Vec> { + /// The selected operator roots. Use `roots()` to include scalar queries. + pub fn operator_roots(&self) -> Vec> { self.plans.iter().map(|p| Rc::clone(&p.plan.root)).collect() } + /// Unique operators in the entire workload DAG, including scalar-plan dependencies. + /// Several query roots can reach the same operator; it is returned once. + pub fn operators(&self) -> Vec> { + let mut seen = std::collections::HashSet::new(); + let mut nodes = Vec::new(); + for root in self.roots() { + let inputs = match root { + asap_types::ir::QueryRoot::Operator(node) => vec![node], + asap_types::ir::QueryRoot::Scalar(expr) => { + expr.operator_refs().into_iter().cloned().collect() + } + }; + for input in inputs { + for node in OperatorNode::reachable(&input) { + if seen.insert(Rc::as_ptr(&node)) { + nodes.push(node); + } + } + } + } + nodes + } + pub fn len(&self) -> usize { - self.plans.len() + self.plans.len() + self.scalar_roots.len() } pub fn is_empty(&self) -> bool { diff --git a/crates/asap-aware-mapping/src/physical_operator_statistics.rs b/crates/asap-aware-mapping/src/physical_operator_statistics.rs index 9cf8045bb..9c40039c4 100644 --- a/crates/asap-aware-mapping/src/physical_operator_statistics.rs +++ b/crates/asap-aware-mapping/src/physical_operator_statistics.rs @@ -6,7 +6,9 @@ use std::collections::HashMap; -use asap_types::pre_asap::query_expr::{InfoMatcher, Predicate, Source}; +use asap_types::ir::operator_properties::{InfoMatcher, Source}; +use asap_types::ir::Predicate; + use asap_types::workload::{ DataArrival, DataWorkload, DurationMs, QueryRecurrence, QueryWorkloadEntry, RepeatedDemand, TimeSelection, TimestampMs, @@ -278,8 +280,8 @@ pub struct PartitionStatistics { /// is the authoritative operator vocabulary: every one of its variants has a /// matching statistics variant here. /// -/// This enum intentionally does not mirror either logical IR. `QueryExpr` and -/// `SummaryExpr` are inputs to physical lowering, and one logical node may +/// This enum intentionally does not mirror the logical IR. `OperatorNode`s +/// are inputs to physical lowering, and one logical node may /// expand into several physical nodes or choose among several algorithms. /// Physical configuration such as a Top-K limit or hash-join build side lives /// on `PhysicalOperator`; this enum contains only workload/catalog evidence diff --git a/crates/asap-aware-mapping/src/physical_plan_cost_model.rs b/crates/asap-aware-mapping/src/physical_plan_cost_model.rs index ffbe08fc4..7a427b51b 100644 --- a/crates/asap-aware-mapping/src/physical_plan_cost_model.rs +++ b/crates/asap-aware-mapping/src/physical_plan_cost_model.rs @@ -2,8 +2,9 @@ use std::{cell::RefCell, rc::Rc}; -use asap_types::post_asap::{SketchAlgorithm, SummaryExpr, SummaryNode}; -use asap_types::pre_asap::{AggIntent, QueryExpr}; +use asap_types::ir::OperatorNode; +use asap_types::post_asap::SketchAlgorithm; +use asap_types::pre_asap::AggIntent; use asap_types::resources::CacheProfile; use crate::analytical_cost::{ @@ -40,7 +41,7 @@ pub struct PhysicalEvidenceSnapshot { /// operator. Post-ASAP summary operators need a physical plan provider because their /// implementation, placement, and retained-state layout are deployment /// choices; that provider must return the complete summary DAG, including any -/// embedded `KeepPreAsap` work. +/// non-ASAP work kept inside it. pub trait PlannerPhysicalPlanProvider { /// Atomically captures the comparison scope and evidence generation. fn capture_evidence_snapshot( @@ -57,7 +58,7 @@ pub trait PlannerPhysicalPlanProvider { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - summary: &Rc, + summary: &Rc, target: &TargetSubDAG<'_>, ) -> Result; } @@ -91,7 +92,7 @@ pub struct PhysicalPlanCostModel<'a> { } struct CachedTargetEvidence { - root: Rc, + root: Rc, consumer_count: usize, snapshot: PhysicalEvidenceSnapshot, raw: PhysicalDag, @@ -188,15 +189,14 @@ impl<'a> PhysicalPlanCostModel<'a> { Replacement::ExactComposition(_) => { return Err(AnalyticalCostError::UnsupportedCandidate) } - Replacement::Rewrite(query) => lower_query_physical_dag(query, scope, &evidence)?, - Replacement::Summary(summary) => match &summary.expr { - SummaryExpr::KeepPreAsap(query) => { - lower_query_physical_dag(query, scope, &evidence)? - } - _ => self - .provider - .summary_physical_dag(&snapshot, summary, target)?, - }, + // A sub-DAG without summary state is the planner's own query + // lowering; anything with summary state is deployment-provided. + Replacement::SubDag(subtree) if !subtree.contains_asap() => { + lower_query_physical_dag(subtree, scope, &evidence)? + } + Replacement::SubDag(summary) => self + .provider + .summary_physical_dag(&snapshot, summary, target)?, }; let resources = estimate_physical_dag_comparison( PhysicalDagEstimateRequest { @@ -359,7 +359,8 @@ mod tests { use std::cell::Cell; use std::collections::HashMap; - use asap_types::pre_asap::{Column, DataType, QueryExpr, Reduction, Schema, Source}; + use asap_types::ir::{NonASAPOp, OperatorNode}; + use asap_types::pre_asap::{DataType, Field, Reduction, Schema, Source}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, @@ -405,8 +406,16 @@ mod tests { } } - fn query() -> Rc { - Rc::new(QueryExpr::Aggregate { + fn query() -> Rc { + let scan = OperatorNode::non_asap_node(NonASAPOp::Scan { + source: Source::Table { + table_ref: "events".into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), + }) + .unwrap(); + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.01), @@ -414,14 +423,9 @@ mod tests { output_names: vec![], filters: vec![], having: None, - child: Rc::new(QueryExpr::Scan { - source: Source::Table { - table_ref: "events".into(), - }, - predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), - }), + child: scan, }) + .unwrap() } fn scope() -> ComparisonScope { @@ -583,7 +587,7 @@ mod tests { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - _summary: &Rc, + _summary: &Rc, _target: &TargetSubDAG<'_>, ) -> Result { assert_eq!(snapshot.version, "test-snapshot-1"); @@ -687,7 +691,7 @@ mod tests { version: "unused-base-v1".into(), }; let candidates = - crate::replacement::SketchAlgorithmStrategy::default_cost_model().replacements(&target); + crate::replacement::ASAPStrategies::default_cost_model().replacements(&target); provider.storage_io = Some(profile.clone()); let model = PhysicalPlanCostModel::new(&provider, base.clone()).unwrap(); let estimate = model.estimate_candidate(&candidates[0], &target).unwrap(); @@ -718,7 +722,7 @@ mod tests { let root = query(); let target = TargetSubDAG::new(&root); let candidates = - crate::replacement::SketchAlgorithmStrategy::default_cost_model().replacements(&target); + crate::replacement::ASAPStrategies::default_cost_model().replacements(&target); let provider = TestProvider::new(true, 800); let model = PhysicalPlanCostModel::new(&provider, calibration()).unwrap(); let estimate = model.estimate_candidate(&candidates[0], &target).unwrap(); @@ -913,7 +917,7 @@ mod tests { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - summary: &Rc, + summary: &Rc, target: &TargetSubDAG<'_>, ) -> Result { self.0.summary_physical_dag(snapshot, summary, target) @@ -1001,7 +1005,7 @@ mod tests { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - summary: &Rc, + summary: &Rc, target: &TargetSubDAG<'_>, ) -> Result { let mut dag = self.0.summary_physical_dag(snapshot, summary, target)?; @@ -1015,7 +1019,7 @@ mod tests { } let root = query(); - let candidates = crate::replacement::SketchAlgorithmStrategy::default_cost_model() + let candidates = crate::replacement::ASAPStrategies::default_cost_model() .replacements(&TargetSubDAG::new(&root)); let provider = WrongScope(TestProvider::new(true, 800)); let model = PhysicalPlanCostModel::new(&provider, calibration()).unwrap(); @@ -1053,7 +1057,7 @@ mod tests { fn summary_physical_dag( &self, _snapshot: &PhysicalEvidenceSnapshot, - _summary: &Rc, + _summary: &Rc, _target: &TargetSubDAG<'_>, ) -> Result { panic!("blank snapshot versions must fail before summary binding") @@ -1061,7 +1065,7 @@ mod tests { } let root = query(); - let candidates = crate::replacement::SketchAlgorithmStrategy::default_cost_model() + let candidates = crate::replacement::ASAPStrategies::default_cost_model() .replacements(&TargetSubDAG::new(&root)); let model = PhysicalPlanCostModel::new(&BlankVersionProvider, calibration()).unwrap(); assert_eq!( @@ -1088,7 +1092,7 @@ mod tests { #[test] fn sibling_candidates_share_one_scope_and_raw_baseline() { let root = query(); - let candidates = crate::replacement::SketchAlgorithmStrategy::default_cost_model() + let candidates = crate::replacement::ASAPStrategies::default_cost_model() .replacements(&TargetSubDAG::new(&root)); assert!(candidates.len() >= 2); let provider = TestProvider::new(true, 800); diff --git a/crates/asap-aware-mapping/src/query_physical_lowering.rs b/crates/asap-aware-mapping/src/query_physical_lowering.rs index aa934e8c9..d911ace38 100644 --- a/crates/asap-aware-mapping/src/query_physical_lowering.rs +++ b/crates/asap-aware-mapping/src/query_physical_lowering.rs @@ -1,7 +1,9 @@ -//! Recursive lowering from the canonical query IR to evidenced physical DAGs. +//! Recursive lowering from the operator IR to evidenced physical DAGs. use std::rc::Rc; +use asap_types::ir::{NonASAPOp, OperatorNode, ScalarExpr}; + use crate::analytical_cost::{ validate_operator_semantics, AnalyticalCostError, EvidenceBackedPhysicalDag, ExecutionMultiplicity, HashJoinBuildSide, PhysicalDagNode, PhysicalNodeEvidence, @@ -13,7 +15,7 @@ use crate::physical_operator_statistics::{ }; pub struct PhysicalNodeRequest<'a> { - pub logical_node: &'a asap_types::pre_asap::QueryExpr, + pub logical_node: &'a OperatorNode, pub operator: PhysicalOperator, pub occurrence: usize, pub synthetic: bool, @@ -40,19 +42,19 @@ where } } -/// Lower a resolved query operator DAG to the physical operators understood by +/// Lower a non-ASAP operator DAG to the physical operators understood by /// this cost model. The authoritative provider supplies statistics by the /// stable physical IDs owned by that provider; missing evidence makes the /// complete query unavailable. Scalar expressions remain part of their -/// containing operator's local cost. +/// containing operator's local cost. An ASAP node is unsupported here. pub fn lower_query_physical_dag( - root: &Rc, + root: &Rc, scope: &ComparisonScope, evidence: &dyn PhysicalNodeEvidenceProvider, ) -> Result { use std::collections::HashMap; - use asap_types::pre_asap::{GroupKeys, QueryExpr, RelationalSetOpKind}; + use asap_types::pre_asap::{GroupKeys, RelationalSetOpKind}; scope.validate()?; @@ -65,7 +67,7 @@ pub fn lower_query_physical_dag( } impl Lowerer<'_> { - fn lower(&mut self, query: &QueryExpr) -> Result { + fn lower(&mut self, query: &OperatorNode) -> Result { let occurrence = self.next_id; self.next_id += 1; self.lower_new(query, occurrence) @@ -73,7 +75,7 @@ pub fn lower_query_physical_dag( fn resolve( &self, - query: &QueryExpr, + query: &OperatorNode, operator: PhysicalOperator, occurrence: usize, synthetic: bool, @@ -128,12 +130,23 @@ pub fn lower_query_physical_dag( fn lower_unary( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, operator: PhysicalOperator, - child: &QueryExpr, + child: &OperatorNode, ) -> Result { let child_id = self.lower(child)?; + self.push_unary(query, occurrence, operator, child_id) + } + + /// `operator` over an already-lowered child. + fn push_unary( + &mut self, + query: &OperatorNode, + occurrence: usize, + operator: PhysicalOperator, + child_id: String, + ) -> Result { let children = vec![child_id.clone()]; let evidence = self.resolve(query, operator, occurrence, false, &children, None)?; let statistics = &evidence.statistics; @@ -151,17 +164,17 @@ pub fn lower_query_physical_dag( fn lower_promql_unary( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, operator: PhysicalOperator, - child: &QueryExpr, + child: &OperatorNode, ) -> Result { self.lower_unary(query, occurrence, operator, child) } fn lower_promql_scalar_leaf( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, ) -> Result { let operator = PhysicalOperator::PromqlScalarLeaf; @@ -171,6 +184,28 @@ pub fn lower_query_physical_dag( self.push(evidence, operator, vec![], None) } + /// Lower the owned scalar operand of `vector(s)`. A literal or + /// `time()` is a physical scalar leaf; `scalar(v)` reads its vector + /// through `PromqlVectorToScalar`. + fn lower_scalar_operand( + &mut self, + query: &OperatorNode, + occurrence: usize, + expr: &ScalarExpr, + ) -> Result { + match expr { + ScalarExpr::Literal(asap_types::pre_asap::ScalarValue::Float64(_)) + | ScalarExpr::EvalTimestamp => self.lower_promql_scalar_leaf(query, occurrence), + ScalarExpr::PromqlScalarFromVector(vector) => self.lower_promql_unary( + query, + occurrence, + PhysicalOperator::PromqlVectorToScalar, + vector, + ), + _ => Err(AnalyticalCostError::UnsupportedQueryOperator), + } + } + fn node_statistics(&self, id: &str) -> Result<&OperatorStatistics, AnalyticalCostError> { self.evidence .get(id) @@ -182,11 +217,14 @@ pub fn lower_query_physical_dag( fn lower_new( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, ) -> Result { - match query { - QueryExpr::Scan { + let Some(op) = query.non_asap() else { + return Err(AnalyticalCostError::UnsupportedQueryOperator); + }; + match op { + NonASAPOp::Scan { source, predicates, .. } => { let coverage = bind_scan_coverage( @@ -252,13 +290,13 @@ pub fn lower_query_physical_dag( require_operator_statistics(filter_operator, &filter_evidence.statistics)?; self.push(filter_evidence, filter_operator, children, None) } - QueryExpr::Filter { pred, child } => { + NonASAPOp::Filter { pred, child } => { let operator = PhysicalOperator::Filter { predicate_operations_per_row: scalar_operation_count(&pred.0)?.max(1), }; self.lower_unary(query, occurrence, operator, child) } - QueryExpr::Project { cols, child, .. } => { + NonASAPOp::Project { cols, child, .. } => { let expression_operations_per_row = cols .iter() .try_fold(0_u64, |total, item| { @@ -280,7 +318,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Aggregate { + NonASAPOp::Aggregate { reduction, measures, filters, @@ -289,7 +327,7 @@ pub fn lower_query_physical_dag( .. } => { if having.is_some() - || asap_types::pre_asap::any_measure_filtered(filters) + || asap_types::ir::non_asap::any_measure_filtered(filters) || measures.is_empty() { return Err(AnalyticalCostError::UnsupportedQueryOperator); @@ -338,13 +376,9 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Dedup { cols, child } => { + NonASAPOp::Dedup { cols, child } => { let key_count = if cols.is_empty() { - child - .output_schema() - .map_err(|_| AnalyticalCostError::UnsupportedQueryOperator)? - .columns - .len() + child.schema.fields.len() } else { cols.len() }; @@ -361,7 +395,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Sort { + NonASAPOp::Sort { keys, partition_by, child, @@ -381,13 +415,26 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Limit { n, offset, child } => { - if let QueryExpr::Sort { + NonASAPOp::Limit { + n, + offset, + partition_by: limit_partition_by, + child, + } => { + // Offset-only and per-group limits have no physical + // operator here. + let Some(n) = n else { + return Err(AnalyticalCostError::UnsupportedQueryOperator); + }; + if limit_partition_by != &GroupKeys::none() { + return Err(AnalyticalCostError::UnsupportedQueryOperator); + } + if let Some(NonASAPOp::Sort { keys, partition_by, child: sorted_child, .. - } = child.as_ref() + }) = child.non_asap() { if !keys.is_empty() && partition_by == &GroupKeys::none() { let child_id = self.lower(sorted_child)?; @@ -442,7 +489,7 @@ pub fn lower_query_physical_dag( require_operator_statistics(operator, statistics)?; self.push(evidence, operator, children, None) } - QueryExpr::SQLWindowFunc { + NonASAPOp::SQLWindowFunc { func, partition_by, order_by, @@ -472,7 +519,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::TimeRange { range, child } => { + NonASAPOp::TimeRange { range, child, .. } => { let range_millis = duration_millis(*range, "range")?; self.lower_promql_unary( query, @@ -481,7 +528,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::PromqlSubquery { + NonASAPOp::PromqlSubquery { range, resolution, child, @@ -513,7 +560,7 @@ pub fn lower_query_physical_dag( } Ok(id) } - QueryExpr::PromqlRelabel { value, child, .. } => self.lower_promql_unary( + NonASAPOp::PromqlRelabel { value, child, .. } => self.lower_promql_unary( query, occurrence, PhysicalOperator::PromqlRelabel { @@ -521,7 +568,7 @@ pub fn lower_query_physical_dag( }, child, ), - QueryExpr::PromqlSeriesSample { + NonASAPOp::PromqlSeriesSample { by, kind, child, .. } => { if by.is_without() { @@ -557,7 +604,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::PromqlInfoEnrich { selector, child } => { + NonASAPOp::PromqlInfoEnrich { selector, child } => { let left_id = self.lower(child)?; let coverage = bind_info_coverage( &format!("occurrence-{occurrence}-info"), @@ -609,34 +656,13 @@ pub fn lower_query_physical_dag( require_operator_statistics(operator, &evidence.statistics)?; self.push(evidence, operator, children, None) } - QueryExpr::BinaryOp { - op, - lhs, - rhs, - vector_match, + NonASAPOp::BinaryOp { + operator, lhs, rhs, .. } => { - let left_scalar = is_promql_scalar(lhs); - let right_scalar = is_promql_scalar(rhs); - if left_scalar && right_scalar { - return Err(AnalyticalCostError::UnsupportedQueryOperator); - } + let op = &operator.kind; + let vector_match = &operator.vector_match; let operation = promql_binary_operation(op); - if (left_scalar || right_scalar) - && !matches!(operation, PromqlBinaryOperation::ArithmeticOrComparison) - { - return Err(AnalyticalCostError::UnsupportedQueryOperator); - } - let operand_mode = match (left_scalar, right_scalar) { - (false, false) => PromqlBinaryOperandMode::VectorVector, - (false, true) => PromqlBinaryOperandMode::VectorScalar, - (true, false) => PromqlBinaryOperandMode::ScalarVector, - (true, true) => unreachable!("scalar/scalar returned above"), - }; - if operand_mode != PromqlBinaryOperandMode::VectorVector - && vector_match.is_some() - { - return Err(AnalyticalCostError::UnsupportedQueryOperator); - } + let operand_mode = PromqlBinaryOperandMode::VectorVector; let cardinality = promql_vector_cardinality(vector_match.as_ref()); let left_id = self.lower(lhs)?; let right_id = self.lower(rhs)?; @@ -670,41 +696,31 @@ pub fn lower_query_physical_dag( require_operator_statistics(operator, &evidence.statistics)?; self.push(evidence, operator, children, None) } - QueryExpr::PromqlVectorFromScalar(child) => self.lower_promql_unary( - query, - occurrence, - PhysicalOperator::PromqlScalarToVector, - child, - ), - QueryExpr::PromqlScalarFromVector(child) => self.lower_promql_unary( - query, - occurrence, - PhysicalOperator::PromqlVectorToScalar, - child, - ), - QueryExpr::PromqlScalarBridge(inner) - if matches!( - inner.as_ref(), - QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Float64(_)) - ) => - { - self.lower_promql_scalar_leaf(query, occurrence) + NonASAPOp::PromqlVectorFromScalar(scalar) => { + let scalar_occurrence = self.next_id; + self.next_id += 1; + let child_id = self.lower_scalar_operand(query, scalar_occurrence, scalar)?; + self.push_unary( + query, + occurrence, + PhysicalOperator::PromqlScalarToVector, + child_id, + ) } - QueryExpr::EvalTimestamp => self.lower_promql_scalar_leaf(query, occurrence), - QueryExpr::TimeShift { shift, child } => { + NonASAPOp::TimeShift { shift, child } => { if !shift.is_identity() { return Err(AnalyticalCostError::UnsupportedQueryOperator); } self.lower_unary(query, occurrence, PhysicalOperator::PassThrough, child) } - QueryExpr::Concat { children, .. } => { + NonASAPOp::Concat { children, .. } => { let child_ids = children .iter() .map(|child| self.lower(child)) .collect::, _>>()?; self.lower_concat(query, occurrence, child_ids) } - QueryExpr::SetOp { + NonASAPOp::SetOp { kind: RelationalSetOpKind::Union, all: true, left, @@ -714,7 +730,7 @@ pub fn lower_query_physical_dag( let right_id = self.lower(right)?; self.lower_concat(query, occurrence, vec![left_id, right_id]) } - QueryExpr::Join { + NonASAPOp::Join { kind, pred, left, @@ -764,7 +780,7 @@ pub fn lower_query_physical_dag( fn lower_concat( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, child_ids: Vec, ) -> Result { @@ -957,7 +973,7 @@ fn require_operator_statistics( fn bind_scan_coverage( node_id: &str, source: &asap_types::pre_asap::Source, - predicates: &[asap_types::pre_asap::Predicate], + predicates: &[asap_types::ir::Predicate], scope: &ComparisonScope, ) -> Result { let mut matches = scope.sources.iter().filter(|coverage| { @@ -1033,7 +1049,7 @@ fn promql_binary_operation( BinaryOpKind::Set(PromQLVectorSetOpKind::And) => PromqlBinaryOperation::And, BinaryOpKind::Set(PromQLVectorSetOpKind::Or) => PromqlBinaryOperation::Or, BinaryOpKind::Set(PromQLVectorSetOpKind::Unless) => PromqlBinaryOperation::Unless, - BinaryOpKind::Arithmetic(_) | BinaryOpKind::Compare(_) | BinaryOpKind::CompareBool(_) => { + BinaryOpKind::Arithmetic(_) | BinaryOpKind::Compare(_) => { PromqlBinaryOperation::ArithmeticOrComparison } } @@ -1051,17 +1067,14 @@ fn promql_vector_cardinality( } fn hash_join_key_count( - expr: &asap_types::pre_asap::QueryExpr, - left: &asap_types::pre_asap::QueryExpr, - right: &asap_types::pre_asap::QueryExpr, + expr: &ScalarExpr, + left: &OperatorNode, + right: &OperatorNode, ) -> Option { - use asap_types::pre_asap::{CompareOpKind, QueryExpr}; + use asap_types::pre_asap::CompareOpKind; - let (Ok(left_schema), Ok(right_schema)) = (left.output_schema(), right.output_schema()) else { - return None; - }; - let left_width = left_schema.columns.len(); - let total_width = left_width.saturating_add(right_schema.columns.len()); + let left_width = left.schema.fields.len(); + let total_width = left_width.saturating_add(right.schema.fields.len()); fn column_side(column: usize, left_width: usize, total_width: usize) -> Option { if column < left_width { @@ -1073,14 +1086,15 @@ fn hash_join_key_count( } } - fn predicate(expr: &QueryExpr, left_width: usize, total_width: usize) -> Option { + fn predicate(expr: &ScalarExpr, left_width: usize, total_width: usize) -> Option { match expr { - QueryExpr::Compare { + ScalarExpr::Compare { left, op: CompareOpKind::Eq, right, + .. } => match (left.as_ref(), right.as_ref()) { - (QueryExpr::Column(left), QueryExpr::Column(right)) => match ( + (ScalarExpr::Column(left), ScalarExpr::Column(right)) => match ( column_side(*left, left_width, total_width), column_side(*right, left_width, total_width), ) { @@ -1089,7 +1103,7 @@ fn hash_join_key_count( }, _ => None, }, - QueryExpr::BoolAnd(parts) if !parts.is_empty() => { + ScalarExpr::BoolAnd(parts) if !parts.is_empty() => { parts.iter().try_fold(0_u64, |count, part| { count.checked_add(predicate(part, left_width, total_width)?) }) @@ -1101,12 +1115,8 @@ fn hash_join_key_count( predicate(expr, left_width, total_width) } -fn scalar_operation_count( - expr: &asap_types::pre_asap::QueryExpr, -) -> Result { - use asap_types::pre_asap::QueryExpr; - - let add = |parts: &[&QueryExpr]| { +fn scalar_operation_count(expr: &ScalarExpr) -> Result { + let add = |parts: &[&ScalarExpr]| { parts.iter().try_fold(0_u64, |total, part| { total .checked_add(scalar_operation_count(part)?) @@ -1119,14 +1129,14 @@ fn scalar_operation_count( .ok_or(AnalyticalCostError::Overflow) }; match expr { - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => Ok(0), - QueryExpr::Compare { left, right, .. } | QueryExpr::Arithmetic { left, right, .. } => { + ScalarExpr::Column(_) + | ScalarExpr::Literal(_) + | ScalarExpr::EvalTimestamp + | ScalarExpr::CurrentTimestamp => Ok(0), + ScalarExpr::Compare { left, right, .. } | ScalarExpr::Arithmetic { left, right, .. } => { with_local(&[left, right]) } - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + ScalarExpr::BoolAnd(parts) | ScalarExpr::BoolOr(parts) => { let children = parts.iter().collect::>(); add(&children)? .checked_add( @@ -1135,12 +1145,11 @@ fn scalar_operation_count( ) .ok_or(AnalyticalCostError::Overflow) } - QueryExpr::Not(child) - | QueryExpr::IsNull(child) - | QueryExpr::IsNotNull(child) - | QueryExpr::PromqlScalarBridge(child) => with_local(&[child]), - QueryExpr::Cast { expr, .. } => with_local(&[expr]), - QueryExpr::InList { expr, list, .. } => { + ScalarExpr::Not(child) | ScalarExpr::IsNull(child) | ScalarExpr::IsNotNull(child) => { + with_local(&[child]) + } + ScalarExpr::Cast { expr, .. } => with_local(&[expr]), + ScalarExpr::InList { expr, list, .. } => { let mut children = Vec::with_capacity(list.len() + 1); children.push(expr.as_ref()); children.extend(list.iter()); @@ -1148,11 +1157,11 @@ fn scalar_operation_count( .checked_add(u64::try_from(list.len()).map_err(|_| AnalyticalCostError::Overflow)?) .ok_or(AnalyticalCostError::Overflow) } - QueryExpr::FunctionCall { args, .. } => { + ScalarExpr::FunctionCall { args, .. } => { let children = args.iter().collect::>(); with_local(&children) } - QueryExpr::Case { + ScalarExpr::Case { operand, branches, else_expr, @@ -1246,16 +1255,6 @@ fn fixed_state_per_series_intent(intent: &asap_types::pre_asap::AggIntent) -> bo ) } -fn is_promql_scalar(query: &asap_types::pre_asap::QueryExpr) -> bool { - use asap_types::pre_asap::QueryExpr; - matches!( - query, - QueryExpr::PromqlScalarBridge(_) - | QueryExpr::PromqlScalarFromVector(_) - | QueryExpr::EvalTimestamp - ) -} - #[cfg(test)] mod tests { use super::*; @@ -1266,6 +1265,7 @@ mod tests { validate_comparison_scopes, BinaryEdgeStatistics, PartitionStatistics, PromqlEdgeStatistics, PromqlUnaryEdgeStatistics, PromqlValueKind, UnaryEdgeStatistics, }; + use asap_types::ir::{BinaryOperator, ExprSemantics, Predicate, SortKey, TimeRangeKind}; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, }; @@ -1403,7 +1403,7 @@ mod tests { fn coverage( source: asap_types::pre_asap::Source, - predicates: Vec, + predicates: Vec, ) -> SourceCoverage { SourceCoverage { source, @@ -1451,27 +1451,27 @@ mod tests { // Correlation can be costed as an exact hash aggregate using provider-supplied state size. #[test] fn correlation_lowers_to_physical_hash_aggregate() { - use asap_types::pre_asap::{ - AggIntent, Column, DataType, QueryExpr, Reduction, Schema, Source, - }; + use asap_types::pre_asap::{AggIntent, DataType, Field, Reduction, Schema, Source}; let source = Source::Table { table_ref: "pairs".into(), }; - let root = Rc::new(QueryExpr::Aggregate { + let root = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::PearsonCorr { left: 0, right: 1 }], output_names: vec!["r".into()], filters: vec![], having: None, - child: Rc::new(QueryExpr::Scan { + child: OperatorNode::non_asap_node(NonASAPOp::Scan { source: source.clone(), predicates: vec![], schema: Schema::new(vec![ - Column::new("x", DataType::Float64, true), - Column::new("y", DataType::Float64, true), + Field::plain("x", DataType::Float64, true), + Field::plain("y", DataType::Float64, true), ]), - }), - }); + }) + .unwrap(), + }) + .unwrap(); let scope = scope(vec![coverage(source, vec![])]); let provided = HashMap::from([ ( @@ -1501,51 +1501,56 @@ mod tests { #[test] fn query_lowering_recurses_and_fuses_global_sort_limit() { - use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr, Reduction, SortKey, Source}; - use asap_types::pre_asap::{Column, DataType, Schema}; + use asap_types::pre_asap::{AggIntent, GroupKeys, Reduction, Source}; + use asap_types::pre_asap::{DataType, Field, Schema}; use std::rc::Rc; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, - predicates: vec![asap_types::pre_asap::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Boolean(true)), + predicates: vec![Predicate(ScalarExpr::Literal( + asap_types::pre_asap::ScalarValue::Boolean(true), ))], schema: Schema::new(vec![ - Column::new("service", DataType::Utf8, false), - Column::new("value", DataType::Float64, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("value", DataType::Float64, false), ]), - }); - let aggregate = Rc::new(QueryExpr::Aggregate { + }) + .unwrap(); + let aggregate = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![0]), measures: vec![AggIntent::Sum { col: Some(1) }], output_names: vec![], filters: vec![], having: None, child: Rc::clone(&scan), - }); - let sort = Rc::new(QueryExpr::Sort { + }) + .unwrap(); + let sort = OperatorNode::non_asap_node(NonASAPOp::Sort { keys: vec![SortKey { - expr: QueryExpr::Column(0), + expr: ScalarExpr::Column(0), ascending: false, nulls_first: false, }], partition_by: GroupKeys::none(), child: aggregate, - }); - let root = Rc::new(QueryExpr::Limit { - n: 10, + }) + .unwrap(); + let root = OperatorNode::non_asap_node(NonASAPOp::Limit { + n: Some(10), offset: 5, + partition_by: GroupKeys::none(), child: sort, - }); + }) + .unwrap(); let scan_coverage = coverage( Source::Table { table_ref: "events".into(), }, - vec![asap_types::pre_asap::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Boolean(true)), + vec![Predicate(ScalarExpr::Literal( + asap_types::pre_asap::ScalarValue::Boolean(true), ))], ); let scope = scope(vec![scan_coverage]); @@ -1640,27 +1645,30 @@ mod tests { #[test] fn query_lowering_shares_only_provider_identified_physical_nodes() { - use asap_types::pre_asap::{Column, CompareOpKind, DataType, Schema}; - use asap_types::pre_asap::{JoinKind, Predicate, QueryExpr, Source}; + use asap_types::pre_asap::{CompareOpKind, DataType, Field, Schema}; + use asap_types::pre_asap::{JoinKind, Source}; use std::rc::Rc; - let shared = Rc::new(QueryExpr::Scan { + let shared = OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::Table { table_ref: "dimensions".into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), - }); - let root = Rc::new(QueryExpr::Join { + schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), + }) + .unwrap(); + let root = OperatorNode::non_asap_node(NonASAPOp::Join { kind: JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), - })), + right: Box::new(ScalarExpr::Column(1)), + semantics: ExprSemantics::Sql, + }), left: Rc::clone(&shared), right: Rc::clone(&shared), - }); + }) + .unwrap(); let source_coverage = coverage( Source::Table { table_ref: "dimensions".into(), @@ -1784,16 +1792,18 @@ mod tests { )) ); - let invalid = Rc::new(QueryExpr::Join { + let invalid = OperatorNode::non_asap_node(NonASAPOp::Join { kind: JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(0)), - })), + right: Box::new(ScalarExpr::Column(0)), + semantics: ExprSemantics::Sql, + }), left: Rc::clone(&shared), right: Rc::clone(&shared), - }); + }) + .unwrap(); assert_eq!( lower_query_physical_dag(&invalid, &shared_scope, &shared_provider), Err(AnalyticalCostError::UnsupportedQueryOperator) @@ -1802,63 +1812,69 @@ mod tests { #[test] fn query_lowering_covers_relational_unary_operators() { - use asap_types::pre_asap::{Column, DataType, ScalarValue, Schema}; - use asap_types::pre_asap::{ - GroupKeys, Predicate, QueryExpr, SortKey, Source, TimeShift, WindowFuncKind, - }; - use std::rc::Rc; + use asap_types::pre_asap::{DataType, Field, ScalarValue, Schema}; + use asap_types::pre_asap::{GroupKeys, Source, TimeShift, WindowFuncKind}; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), - }); - let filter = Rc::new(QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), + }) + .unwrap(); + let filter = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), child: scan, - }); - let project = Rc::new(QueryExpr::Project { + }) + .unwrap(); + let project = OperatorNode::non_asap_node(NonASAPOp::Project { cols: vec![], qualifier: None, child: filter, - }); - let dedup = Rc::new(QueryExpr::Dedup { + }) + .unwrap(); + let dedup = OperatorNode::non_asap_node(NonASAPOp::Dedup { cols: vec![0], child: project, - }); - let window = Rc::new(QueryExpr::SQLWindowFunc { + }) + .unwrap(); + let window = OperatorNode::non_asap_node(NonASAPOp::SQLWindowFunc { func: WindowFuncKind::RowNumber, args: vec![], partition_by: GroupKeys::none(), order_by: vec![SortKey { - expr: QueryExpr::Column(0), + expr: ScalarExpr::Column(0), ascending: true, nulls_first: false, }], frame: None, output_name: "rn".into(), child: dedup, - }); - let sort = Rc::new(QueryExpr::Sort { + }) + .unwrap(); + let sort = OperatorNode::non_asap_node(NonASAPOp::Sort { keys: vec![SortKey { - expr: QueryExpr::Column(0), + expr: ScalarExpr::Column(0), ascending: true, nulls_first: false, }], partition_by: GroupKeys::by(vec![0]), child: window, - }); - let limit = Rc::new(QueryExpr::Limit { - n: 20, + }) + .unwrap(); + let limit = OperatorNode::non_asap_node(NonASAPOp::Limit { + n: Some(20), offset: 0, + partition_by: GroupKeys::none(), child: sort, - }); - let root = Rc::new(QueryExpr::TimeShift { + }) + .unwrap(); + let root = OperatorNode::non_asap_node(NonASAPOp::TimeShift { shift: TimeShift::default(), child: limit, - }); + }) + .unwrap(); let source_coverage = coverage( Source::Table { @@ -1976,23 +1992,26 @@ mod tests { #[test] fn query_lowering_maps_concat_and_union_all_but_rejects_distinct_set_ops() { - use asap_types::pre_asap::{Column, DataType, Schema}; - use asap_types::pre_asap::{QueryExpr, RelationalSetOpKind, Source}; - use std::rc::Rc; + use asap_types::pre_asap::{DataType, Field, Schema}; + use asap_types::pre_asap::{RelationalSetOpKind, Source}; - let scan = |name: &str| QueryExpr::Scan { - source: Source::Table { - table_ref: name.into(), - }, - predicates: vec![], - schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), + let scan = |name: &str| { + OperatorNode::non_asap_node(NonASAPOp::Scan { + source: Source::Table { + table_ref: name.into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), + }) + .unwrap() }; - let union = Rc::new(QueryExpr::SetOp { + let union = OperatorNode::non_asap_node(NonASAPOp::SetOp { kind: RelationalSetOpKind::Union, all: true, - left: Rc::new(scan("a")), - right: Rc::new(scan("b")), - }); + left: scan("a"), + right: scan("b"), + }) + .unwrap(); let scope = scope(vec![ coverage( Source::Table { @@ -2036,19 +2055,21 @@ mod tests { )) ); - let concat = Rc::new(QueryExpr::Concat { + let concat = OperatorNode::non_asap_node(NonASAPOp::Concat { children: vec![scan("a"), scan("b")], discriminator_unique_key: None, - }); + }) + .unwrap(); let dag = lower_query_physical_dag(&concat, &scope, &scripted(&provided)).unwrap(); assert_eq!(dag.nodes.last().unwrap().operator, PhysicalOperator::Concat); - let distinct_union = Rc::new(QueryExpr::SetOp { + let distinct_union = OperatorNode::non_asap_node(NonASAPOp::SetOp { kind: RelationalSetOpKind::Union, all: false, - left: Rc::new(scan("a")), - right: Rc::new(scan("b")), - }); + left: scan("a"), + right: scan("b"), + }) + .unwrap(); assert_eq!( lower_query_physical_dag(&distinct_union, &scope, &scripted(&provided)), Err(AnalyticalCostError::UnsupportedQueryOperator) @@ -2057,22 +2078,23 @@ mod tests { #[test] fn query_lowering_fails_closed_for_missing_or_inconsistent_statistics() { - use asap_types::pre_asap::{Column, DataType, Schema}; - use asap_types::pre_asap::{QueryExpr, Source}; - use std::rc::Rc; + use asap_types::pre_asap::Source; + use asap_types::pre_asap::{DataType, Field, Schema}; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), - }); - let root = Rc::new(QueryExpr::Project { + schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), + }) + .unwrap(); + let root = OperatorNode::non_asap_node(NonASAPOp::Project { cols: vec![], qualifier: None, child: scan, - }); + }) + .unwrap(); let comparison_scope = scope(vec![coverage( Source::Table { @@ -2173,26 +2195,29 @@ mod tests { #[test] fn query_lowering_accepts_a_consistently_empty_edge() { - use asap_types::pre_asap::{Column, DataType, ScalarValue, Schema}; - use asap_types::pre_asap::{Predicate, QueryExpr, Source}; - use std::rc::Rc; + use asap_types::pre_asap::{DataType, Field, ScalarValue, Schema}; + use asap_types::pre_asap::{GroupKeys, Source}; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), - }); - let filter = Rc::new(QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(false)))), + schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), + }) + .unwrap(); + let filter = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(false))), child: scan, - }); - let root = Rc::new(QueryExpr::Limit { - n: 10, + }) + .unwrap(); + let root = OperatorNode::non_asap_node(NonASAPOp::Limit { + n: Some(10), offset: 0, + partition_by: GroupKeys::none(), child: filter, - }); + }) + .unwrap(); let scope = scope(vec![coverage( Source::Table { @@ -2233,23 +2258,21 @@ mod tests { #[test] fn query_lowering_rejects_aggregates_without_a_hash_implementation() { - use asap_types::pre_asap::{ - AggIntent, GroupKeys, QueryExpr, Reduction, Source, WindowFuncKind, - }; - use asap_types::pre_asap::{Column, DataType, Schema}; + use asap_types::pre_asap::{AggIntent, GroupKeys, Reduction, Source, WindowFuncKind}; + use asap_types::pre_asap::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; - use std::rc::Rc; let scan = || { - Rc::new(QueryExpr::Scan { + OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), }) + .unwrap() }; - let exact_quantile = Rc::new(QueryExpr::Aggregate { + let exact_quantile = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Quantile { col: Some(0), @@ -2260,32 +2283,38 @@ mod tests { filters: vec![], having: None, child: scan(), - }); - let empty_sort_limit = Rc::new(QueryExpr::Limit { - n: 10, + }) + .unwrap(); + let empty_sort_limit = OperatorNode::non_asap_node(NonASAPOp::Limit { + n: Some(10), offset: 0, - child: Rc::new(QueryExpr::Sort { + partition_by: GroupKeys::none(), + child: OperatorNode::non_asap_node(NonASAPOp::Sort { keys: vec![], partition_by: GroupKeys::none(), child: scan(), - }), - }); - let unsupported_window = Rc::new(QueryExpr::SQLWindowFunc { + }) + .unwrap(), + }) + .unwrap(); + let unsupported_window = OperatorNode::non_asap_node(NonASAPOp::SQLWindowFunc { func: WindowFuncKind::Lag, - args: vec![QueryExpr::Column(0)], + args: vec![ScalarExpr::Column(0)], partition_by: GroupKeys::none(), order_by: vec![], frame: None, output_name: "lag".into(), child: scan(), - }); - let shifted = Rc::new(QueryExpr::TimeShift { + }) + .unwrap(); + let shifted = OperatorNode::non_asap_node(NonASAPOp::TimeShift { shift: asap_types::pre_asap::TimeShift { offset_ms: 60_000, at: None, }, child: scan(), - }); + }) + .unwrap(); let scope = scope(vec![coverage( Source::Table { table_ref: "events".into(), @@ -2308,40 +2337,41 @@ mod tests { #[test] fn scalar_work_counts_every_local_predicate_operation() { - use asap_types::pre_asap::{CompareOpKind, QueryExpr, ScalarValue}; + use asap_types::pre_asap::{CompareOpKind, ScalarValue}; - let comparison = || QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let comparison = || ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + right: Box::new(ScalarExpr::Literal(ScalarValue::Int64(1))), + semantics: ExprSemantics::Sql, }; - let predicate = QueryExpr::BoolAnd(vec![comparison(), comparison()]); + let predicate = ScalarExpr::BoolAnd(vec![comparison(), comparison()]); assert_eq!(scalar_operation_count(&predicate), Ok(3)); } #[test] fn promql_presence_is_lowered_with_a_per_step_output_bound() { - use asap_types::pre_asap::{ - AggIntent, Column, DataType, QueryExpr, Reduction, Schema, Source, - }; + use asap_types::pre_asap::{AggIntent, DataType, Field, Reduction, Schema, Source}; let source = Source::TimeSeries { metric: "missing".into(), }; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::non_asap_node(NonASAPOp::Scan { source: source.clone(), predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), - }); - let root = Rc::new(QueryExpr::Aggregate { + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), + }) + .unwrap(); + let root = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Absent], output_names: vec![], filters: vec![], having: None, child: scan, - }); + }) + .unwrap(); let vector = promql_edge(0, 2, PromqlValueKind::Vector); let scan_statistics = OperatorStatistics::Scan { edges: promql_unary_edges(edge(0, 0), edge(0, 0), vector, vector), @@ -2390,24 +2420,28 @@ mod tests { #[test] fn promql_range_and_subquery_preserve_internal_steps() { - use asap_types::pre_asap::{Column, DataType, QueryExpr, Schema, Source}; + use asap_types::pre_asap::{DataType, Field, Schema, Source}; use std::time::Duration; let source = Source::TimeSeries { metric: "m".into() }; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::non_asap_node(NonASAPOp::Scan { source: source.clone(), predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), - }); - let range = Rc::new(QueryExpr::TimeRange { + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), + }) + .unwrap(); + let range = OperatorNode::non_asap_node(NonASAPOp::TimeRange { range: Duration::from_secs(300), + kind: TimeRangeKind::Range, child: scan, - }); - let root = Rc::new(QueryExpr::PromqlSubquery { + }) + .unwrap(); + let root = OperatorNode::non_asap_node(NonASAPOp::PromqlSubquery { range: Duration::from_secs(300), resolution: Some(Duration::from_secs(60)), child: range, - }); + }) + .unwrap(); let vector = promql_edge(10, 6, PromqlValueKind::Vector); let range_vector = promql_edge(10, 6, PromqlValueKind::RangeVector); let outer_range = promql_edge(10, 1, PromqlValueKind::RangeVector); @@ -2465,32 +2499,39 @@ mod tests { #[test] fn promql_binary_lowering_keeps_operation_and_matching_cardinality() { use asap_types::pre_asap::{ - ArithmeticOpKind, BinaryOpKind, Column, DataType, GroupSide, QueryExpr, Schema, Source, + ArithmeticOpKind, BinaryOpKind, DataType, Field, GroupSide, Schema, Source, VectorGrouping, VectorMatch, VectorMatchKind, }; let left_source = Source::TimeSeries { metric: "a".into() }; let right_source = Source::TimeSeries { metric: "b".into() }; let scan = |source| { - Rc::new(QueryExpr::Scan { + OperatorNode::non_asap_node(NonASAPOp::Scan { source, predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), }) + .unwrap() }; - let root = Rc::new(QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + let root = OperatorNode::non_asap_node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + vector_match: Some(VectorMatch { + kind: VectorMatchKind::On, + labels: vec!["service".into()], + grouping: Some(VectorGrouping { + side: GroupSide::Left, + labels: vec!["region".into()], + }), + }), + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, lhs: scan(left_source.clone()), rhs: scan(right_source.clone()), - vector_match: Some(VectorMatch { - kind: VectorMatchKind::On, - labels: vec!["service".into()], - grouping: Some(VectorGrouping { - side: GroupSide::Left, - labels: vec!["region".into()], - }), - }), - }); + }) + .unwrap(); let left_promql = promql_edge(10, 10, PromqlValueKind::Vector); let right_promql = promql_edge(5, 10, PromqlValueKind::Vector); let output_promql = promql_edge(8, 10, PromqlValueKind::Vector); @@ -2548,36 +2589,40 @@ mod tests { #[test] fn promql_relabel_sample_and_per_series_lower_as_a_complete_chain() { use asap_types::pre_asap::{ - AggIntent, Column, DataType, GroupKeys, QueryExpr, Reduction, SampleKind, ScalarValue, - Schema, Source, + AggIntent, DataType, Field, GroupKeys, Reduction, SampleKind, ScalarValue, Schema, + Source, }; let source = Source::TimeSeries { metric: "requests".into(), }; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::non_asap_node(NonASAPOp::Scan { source: source.clone(), predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), - }); - let relabel = Rc::new(QueryExpr::PromqlRelabel { + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), + }) + .unwrap(); + let relabel = OperatorNode::non_asap_node(NonASAPOp::PromqlRelabel { dst: "service".into(), - value: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("api".into()))), + value: ScalarExpr::Literal(ScalarValue::Utf8("api".into())), child: scan, - }); - let sample = Rc::new(QueryExpr::PromqlSeriesSample { + }) + .unwrap(); + let sample = OperatorNode::non_asap_node(NonASAPOp::PromqlSeriesSample { by: GroupKeys::none(), kind: SampleKind::LimitK(5), child: relabel, - }); - let root = Rc::new(QueryExpr::Aggregate { + }) + .unwrap(); + let root = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Sum { col: None }], output_names: vec![], filters: vec![], having: None, child: sample, - }); + }) + .unwrap(); let input = edge(100, 1_600); let sampled = edge(50, 800); diff --git a/crates/asap-aware-mapping/src/recurrence.rs b/crates/asap-aware-mapping/src/recurrence.rs index 48e2ba033..b17716e68 100644 --- a/crates/asap-aware-mapping/src/recurrence.rs +++ b/crates/asap-aware-mapping/src/recurrence.rs @@ -6,7 +6,7 @@ //! neither reached [`CostModel`]'s CSE share-vs-recompute decision //! ([`CostModel::cse_share_decision`]): that decision only ever compared a //! *structural* consumer count (how many workload locations reference a -//! shared subtree) against a flat per-family maintenance weight — it had no +//! shared sub-DAG) against a flat per-family maintenance weight — it had no //! notion of how *often* those consumers actually run. //! //! This module adds that notion as a generic cost context, not a scheduler: @@ -394,7 +394,7 @@ pub enum RootRecurrence { // ── Explanation ────────────────────────────────────────────────────────── -/// The full readout [`CostModel::cse_share_decision_with_recurrence`] +/// The full evaluation [`CostModel::cse_share_decision_with_recurrence`] /// returns: which alternative was selected, both compared cost rates /// (and, when a [`Horizon`] was supplied, both compared totals), every /// input that went into them, their units, and provenance — meant to be @@ -779,41 +779,45 @@ mod tests { // ── decide (structural fallback) ───────────────────────────────────── use crate::cost_model::CseCandidate; + use asap_types::ir::operator_properties::{Reduction, Source}; + use asap_types::ir::{ + ASAPOp, BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, Predicate, ScalarExpr, + }; use asap_types::post_asap::{ - ExactKind, ExactParams, GroupingStrategy, ResultGuarantee, SummaryExpr, SummaryFamilyType, - SummaryField, SummaryNode, SummarySchema, + ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, ResultGuarantee, Schema, }; use asap_types::pre_asap::expr_ir::ColumnRef; - use asap_types::pre_asap::query_expr::{QueryExpr, Reduction, Source}; - use asap_types::pre_asap::schema::{Column, DataType, Schema}; + use asap_types::pre_asap::schema::DataType; + use std::rc::Rc; - fn scan() -> QueryExpr { - QueryExpr::Scan { + fn scan() -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], 0, vec![], ), - } + }) + .unwrap() } - fn summary_node(family: SummaryFamilyType) -> SummaryNode { - SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(scan())), - schema: SummarySchema { - fields: vec![], - time_index: None, - }, - guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), - }), + /// A summary of `family` over the kept pre-ASAP scan. + fn summary_node(family: FieldDataType) -> Rc { + let kept = Rc::new( + scan() + .as_ref() + .clone() + .with_guarantee(Some(ResultGuarantee::exact("RetainedExact"))), + ); + OperatorNode::asap_node( + ASAPOp::SummaryAgg { + child: kept, family: family.clone(), input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Named( "value".into(), @@ -822,22 +826,15 @@ mod tests { grouping: GroupingStrategy::default(), filter: None, }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "state".into(), - dtype: family, - nullable: false, - }], - time_index: None, - }, - guarantee: None, - } + Schema::lifted(vec![Field::new("state", family, false)], None), + None, + ) } #[test] fn decide_falls_back_to_structural_decision_when_profile_is_empty() { let subtree = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -863,7 +860,7 @@ mod tests { #[test] fn decide_rejects_mixed_one_shot_and_repeating_without_horizon() { let subtree = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -882,7 +879,7 @@ mod tests { #[test] fn decide_accepts_mixed_one_shot_and_repeating_with_an_explicit_horizon() { let subtree = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -943,7 +940,7 @@ mod tests { #[test] fn high_frequency_selects_maintained_low_frequency_selects_recompute() { let subtree = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -999,7 +996,7 @@ mod tests { #[test] fn update_rate_only_affects_maintained_cost_evaluation_rate_affects_both() { let subtree = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -1039,7 +1036,7 @@ mod tests { #[test] fn one_shot_only_consumer_decides_without_an_explicit_horizon() { let subtree = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -1072,7 +1069,7 @@ mod tests { #[test] fn one_shot_only_single_consumer_does_not_unconditionally_prefer_share() { let subtree = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -1099,7 +1096,7 @@ mod tests { #[test] fn batch_only_workload_does_not_unconditionally_prefer_share_under_default_cost_model() { let subtree = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -1130,9 +1127,9 @@ mod tests { // ── multiple roots sharing a sub-DAG, via CandidateLogicalASAPDAGs ────────────────── use crate::replacement::search_workload; + use asap_types::ir::operator_properties::Reduction as QueryReduction; use asap_types::pre_asap::agg_intent::AggIntent; - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::{Predicate, Reduction as QueryReduction}; + use asap_types::pre_asap::expr_ir::{CompareOpKind, ScalarValue}; /// Like `scan()`, plus a "job" label column to group by — CSE's /// sharing legality gate requires a provable unique key @@ -1142,31 +1139,33 @@ mod tests { /// real one, matching the pattern /// `replacement.rs`'s own CSE fixtures already use (`metric_scan`/`agg` /// grouped by a label column). - fn labeled_scan() -> QueryExpr { - QueryExpr::Scan { + fn labeled_scan() -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - Column::new("job", DataType::Utf8, true), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("job", DataType::Utf8, true), ], 0, vec![], ), - } + }) + .unwrap() } - fn sum_agg() -> QueryExpr { - QueryExpr::Aggregate { + fn sum_agg() -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: QueryReduction::by(vec![2]), measures: vec![AggIntent::Sum { col: Some(1) }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(labeled_scan()), - } + child: labeled_scan(), + }) + .unwrap() } /// A root wrapping a fresh, independently-built (but structurally @@ -1175,15 +1174,21 @@ mod tests { /// themselves structurally distinct (so they don't collapse into one /// root the way whole-root-identical fixtures do — see /// `shared_aggregate_across_two_roots_gets_both_strategies_candidates`'s - /// own doc) while letting `share_common_subtrees` unify their + /// own doc) while letting `share_common_subdags` unify their /// identical `sum_agg()` children onto one shared `Rc`. - fn filtered_root(distinguishing_literal: i64) -> QueryExpr { - QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64( - distinguishing_literal, - )))), - child: Rc::new(sum_agg()), - } + fn filtered_root(distinguishing_literal: i64) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(1)), + op: CompareOpKind::Gt, + right: Box::new(ScalarExpr::Literal(ScalarValue::Int64( + distinguishing_literal, + ))), + semantics: ExprSemantics::Sql, + }), + child: sum_agg(), + }) + .unwrap() } /// Three workload roots share one underlying `sum_agg()` sub-DAG: two @@ -1195,10 +1200,10 @@ mod tests { /// roots sharing a sub-DAG" acceptance criteria. #[test] fn recurrence_profiles_aggregates_mixed_intervals_across_roots_sharing_a_subdag() { - let roots: Vec<(&str, Rc)> = vec![ - ("root_a", Rc::new(filtered_root(1))), - ("root_b", Rc::new(filtered_root(2))), - ("root_c", Rc::new(filtered_root(3))), + let roots: Vec<(&str, Rc)> = vec![ + ("root_a", filtered_root(1)), + ("root_b", filtered_root(2)), + ("root_c", filtered_root(3)), ]; let space = search_workload(roots); @@ -1216,7 +1221,7 @@ mod tests { ); let shared_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("the shared sum_agg() is a discovered target"); assert_eq!(shared_group.consumer_count, 3, "shared by all 3 roots"); @@ -1260,14 +1265,11 @@ mod tests { #[test] fn plan_selection_uses_recurrence_profiles_for_cse_choices() { - let roots = vec![ - ("a", Rc::new(filtered_root(1))), - ("b", Rc::new(filtered_root(2))), - ]; + let roots = vec![("a", filtered_root(1)), ("b", filtered_root(2))]; let space = search_workload(roots); let shared = space .target_subdag_candidates() - .find(|group| matches!(group.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|group| matches!(group.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("the aggregate is shared by both roots"); let update_rate = Some(UpdateRate(10.0)); @@ -1325,8 +1327,8 @@ mod tests { #[test] fn recurrence_profiles_rejects_an_invalid_evaluation_rate() { - let root = Rc::new(scan()); - let roots: Vec<(&str, Rc)> = vec![("only", root)]; + let root = scan(); + let roots: Vec<(&str, Rc)> = vec![("only", root)]; let space = search_workload(roots); let err = space .recurrence_profiles(&[RootRecurrence::Repeating(EvaluationRate(f64::NAN))], None) @@ -1339,8 +1341,8 @@ mod tests { /// signature promises a `Result`. #[test] fn recurrence_profiles_reports_a_root_count_mismatch_as_an_error_not_a_panic() { - let root = Rc::new(scan()); - let roots: Vec<(&str, Rc)> = vec![("only", root)]; + let root = scan(); + let roots: Vec<(&str, Rc)> = vec![("only", root)]; let space = search_workload(roots); let err = space.recurrence_profiles(&[], None).unwrap_err(); assert_eq!( @@ -1354,8 +1356,8 @@ mod tests { #[test] fn recurrence_profiles_rejects_an_invalid_update_rate() { - let root = Rc::new(scan()); - let roots: Vec<(&str, Rc)> = vec![("only", root)]; + let root = scan(); + let roots: Vec<(&str, Rc)> = vec![("only", root)]; let space = search_workload(roots); let err = space .recurrence_profiles( @@ -1383,23 +1385,24 @@ mod tests { /// `consumer_count`. #[test] fn recurrence_profiles_does_not_stamp_update_rate_on_a_site_unreachable_from_any_root() { - let avg_root = QueryExpr::Aggregate { + let avg_root = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: QueryReduction::by(vec![]), measures: vec![AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(scan()), - }; - let roots: Vec<(&str, Rc)> = vec![("q", Rc::new(avg_root))]; + child: scan(), + }) + .unwrap(); + let roots: Vec<(&str, Rc)> = vec![("q", avg_root)]; let space = search_workload(roots); let count_group = space .target_subdag_candidates() .find(|g| { matches!( - g.target.as_ref(), - QueryExpr::Aggregate { measures, .. } + g.target.non_asap(), + Some(NonASAPOp::Aggregate { measures, .. }) if measures.iter().any(|m| matches!(m, AggIntent::Count { .. })) ) }) @@ -1438,19 +1441,23 @@ mod tests { /// reachability-set walk would (wrongly) collapse it to. #[test] fn recurrence_profiles_credits_a_direct_repeated_reference_by_its_multiplicity() { - let root = QueryExpr::BinaryOp { - op: asap_types::pre_asap::query_expr::BinaryOpKind::Compare( - asap_types::pre_asap::expr_ir::CompareOpKind::Eq, - ), - lhs: Rc::new(sum_agg()), - rhs: Rc::new(sum_agg()), - vector_match: None, - }; - let space = search_workload(vec![("q", Rc::new(root))]); + let root = OperatorNode::non_asap_node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: asap_types::ir::operator_properties::BinaryOpKind::Compare(CompareOpKind::Eq), + vector_match: None, + }, + return_bool: false, + lhs: sum_agg(), + rhs: sum_agg(), + }) + .unwrap(); + let space = search_workload(vec![("q", root)]); let shared_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("sum_agg() should merge onto one shared Rc, referenced twice from BinaryOp"); assert_eq!( shared_group.consumer_count, 2, @@ -1473,7 +1480,7 @@ mod tests { let scan_group = space .target_subdag_candidates() - .find(|group| matches!(group.target.as_ref(), QueryExpr::Scan { .. })) + .find(|group| matches!(group.target.non_asap(), Some(NonASAPOp::Scan { .. }))) .expect("the shared aggregate has a scan descendant"); assert_eq!( profiles @@ -1490,7 +1497,7 @@ mod tests { #[test] fn decide_rejects_a_zero_or_negative_horizon() { let subtree = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 809955996..6164638ea 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -3,9 +3,9 @@ //! under "Key concepts (not yet implemented)", implemented for real (issue //! #251, part of #33). //! -//! ## One step, not two: `SketchAlgorithmStrategy::replacements()` decides *and* builds +//! ## One step, not two: `ASAPStrategies::replacements()` decides *and* builds //! -//! For a bindable `Aggregate`, `SketchAlgorithmStrategy::replacements()` is the +//! For a bindable `Aggregate`, `ASAPStrategies::replacements()` is the //! single place this crate both decides what an `AggIntent` may become and //! turns each of those candidates into a real, executable //! [`ReplacementSubDAG`]: @@ -18,8 +18,8 @@ //! `CostModel::size_params`, not a placeholder filled in later). //! 2. **Build**: for each candidate in that list, [`construct_summary`] //! mechanically turns the already-decided `(kind, params)` into a real -//! [`SummaryNode`] — derives the child schema, resolves the summarized -//! column, builds the readout query, recurses into the child (via +//! [`OperatorNode`] — derives the child schema, resolves the summarized +//! column, builds the evaluation query, recurses into the child (via //! [`realize_child`], so a nested aggregate gets its own //! independent enumeration, never the outer target's forced choice), and //! assembles the `SummaryAgg`/`SummaryEstimate` node. @@ -31,13 +31,13 @@ //! has to run regardless of how `(kind, params)` were chosen, so it lives //! directly inside the one method that needs it. //! -//! - [`TargetSubDAG`] — a reference to a pre-ASAP [`QueryExpr`] node that is a +//! - [`TargetSubDAG`] — a reference to a pre-ASAP [`OperatorNode`] that is a //! candidate for replacement, plus how many places in the workload already //! reference it (its `consumer_count`) — the one piece of cross-node -//! context [`SharedSubtreeStrategy`] needs that a bare node reference alone +//! context [`SharedSubDagStrategy`] needs that a bare node reference alone //! doesn't carry. //! - [`ReplacementSubDAG`] — one candidate replacement for a `TargetSubDAG`: -//! either a fully bound [`SummaryNode`] or a pre-ASAP [`QueryExpr`] rewrite +//! either a fully bound summary sub-DAG or a pre-ASAP logical rewrite //! (still logical, structurally different from the target but semantically //! equivalent) — see [`Replacement`] — plus a human-readable `rationale`. //! - [`ReplacementStrategy`] — `matches` + `replacements`, the same @@ -72,20 +72,20 @@ //! //! ## The two strategies, and why these two //! -//! - [`SketchAlgorithmStrategy`] wraps [`realizations_for_intent`]'s exhaustive, +//! - [`ASAPStrategies`] wraps [`realizations_for_intent`]'s exhaustive, //! ranked list directly: for the same bindable-`Aggregate` shape this crate //! binds (single intent, no `HAVING`), every entry becomes its own bound //! candidate. -//! - [`SharedSubtreeStrategy`] wraps -//! `asap_types::pre_asap::cse::share_common_subtrees`'s sharing decision. +//! - [`SharedSubDagStrategy`] wraps +//! `asap_types::pre_asap::cse::share_common_subdags`'s sharing decision. //! Wherever a [`TargetSubDAG`] already has two or more consumers (i.e. -//! `share_common_subtrees` already collapsed two or more workload -//! locations onto the same `Rc` — [`discover_targets`] below +//! `share_common_subdags` already collapsed two or more workload +//! locations onto the same `Rc` — [`discover_targets`] below //! does the identical workload-wide discovery for [`search_workload_with`]; //! this module's own tests reuse the same dedup logic to build realistic //! fixtures), it reports the two-way candidate CSE's own detection pass //! deliberately declines to pick between on its own: build once and share -//! the already-interned subtree, or build it independently at each +//! the already-interned sub-DAG, or build it independently at each //! consumer. [`crate::cost_model::CostModel::cse_share_decision`] is where //! that choice actually gets made *today* (a fixed comparison, not a //! search) — this strategy exposes the same two-way choice as an explicit, @@ -98,7 +98,7 @@ //! unchanged.** Same inputs still produce the same exhaustive, ranked //! list — only its home moved (from a separate `implementation` module //! into this one) and its own visibility dropped to module-private, since -//! [`SketchAlgorithmStrategy`] is now its only caller. +//! [`ASAPStrategies`] is now its only caller. //! //! ## Workload-wide search — merged in from the former `search.rs` (issue #252, part of #33) //! @@ -136,19 +136,19 @@ //! ``` //! //! Read literally, this enumerates whole *plans* — full copies of the -//! workload's tree, one per combination of per-target choices. A workload +//! workload's DAG, one per combination of per-target choices. A workload //! with `N` independently-choosable targets would produce up to `2^N` flat -//! plans, each one duplicating every untouched sibling subtree. This module +//! plans, each one duplicating every untouched sibling sub-DAG. This module //! does not do that: //! //! 1. **Per-target candidates, not flat plans.** [`TargetSubDAGCandidates`] //! stores the alternatives for one distinct [`TargetSubDAG`] (identified by -//! its own `Rc` pointer identity — the same currency -//! [`asap_types::pre_asap::cse::share_common_subtrees`] already +//! its own `Rc` pointer identity — the same currency +//! [`asap_types::pre_asap::cse::share_common_subdags`] already //! established across the workload) holding every //! [`ReplacementSubDAG`] alternative discovered for it. [`CandidateLogicalASAPDAGs`] is //! a collection of these groups, keyed by `TargetSubDAG` — a candidate -//! "plan" is never materialized as a distinct top-level `Rc` +//! "plan" is never materialized as a distinct top-level `Rc` //! at all; two logically-different overall choices at two different //! targets are just two different entries in two different groups, //! sharing every other node in the workload by construction (they *are* @@ -157,7 +157,7 @@ //! discipline.** [`asap_types::pre_asap::cse::structural_hash`] (made //! `pub` for exactly this reuse) is only ever a candidate-narrowing //! filter; [`TargetSubDAGCandidates::add_candidate`]'s actual duplicate check is -//! `QueryExpr`'s derived `PartialEq` — the same "hash is a filter, +//! `OperatorNode`'s derived `PartialEq` — the same "hash is a filter, //! `PartialEq` is the decision, no exceptions" rule `cse.rs`'s own //! "Correctness" section states and this module inherits rather than //! reinvents. See [`is_duplicate_rewrite`] for the one deliberate @@ -175,12 +175,12 @@ //! line above stands for: every `TargetSubDAG` this pass discovers is one //! iteration of that loop. It walks every workload root's whole DAG (the //! same **relational-skeleton** operator-child scope -//! `asap_types::pre_asap::cse::share_common_subtrees` itself uses — see +//! `asap_types::pre_asap::cse::share_common_subdags` itself uses — see //! that module's "Algorithm" section), discovering one `TargetSubDAG` per //! distinct `Rc` and a *real* `consumer_count`: how many operator-child //! positions anywhere in the workload reference that exact `Rc`, not just //! how many of the workload's own top-level roots happen to be it — a -//! `SharedSubtreeStrategy` candidate three levels under an unshared +//! `SharedSubDagStrategy` candidate three levels under an unshared //! `Filter` is exactly as real a target as a shared whole root, so this //! module's discovery can't stop at the top level. //! @@ -212,10 +212,10 @@ //! an alternative *for* the target just processed, not a new target of its //! own; see [`discover_new_descendant_targets`]) are scanned for pointers //! not already known, and any found become next round's frontier. Both shipped -//! strategies are idempotent in exactly this sense: [`SketchAlgorithmStrategy`] -//! produces terminal [`Replacement::Summary`] candidates (no `QueryExpr` -//! children to scan at all), and [`SharedSubtreeStrategy`]'s two -//! [`Replacement::Rewrite`] candidates both reuse the target's own +//! strategies are idempotent in exactly this sense: [`ASAPStrategies`] +//! produces terminal bound-summary [`Replacement::SubDag`] candidates (no +//! logical-rewrite children to scan at all), and [`SharedSubDagStrategy`]'s +//! two logical-rewrite [`Replacement::SubDag`] candidates both reuse the target's own //! already-known child `Rc`s verbatim (`Rc::clone`/a shallow top-level //! `.clone()` — see that strategy's own doc). So for both, the frontier is //! always empty after round one: real workloads converge in exactly one @@ -244,11 +244,11 @@ //! being the whole story; it isn't a contradiction of #237, it's the scope //! change #237 itself named). Concretely, per [`TargetSubDAGCandidates`]: //! -//! - A group whose candidates are the [`SharedSubtreeStrategy`] +//! - A group whose candidates are the [`SharedSubDagStrategy`] //! share-vs-recompute pair is ranked by calling //! [`CostModel::cse_share_decision`] via this module's own //! [`cse_preference`] — rather than re-deriving a competing comparison. -//! - A group whose candidates are [`SketchAlgorithmStrategy`]'s sketch-family +//! - A group whose candidates are [`ASAPStrategies`]'s sketch-family //! candidates is ranked via [`CostModel::rank_candidates`] (the same hook //! `realizations_for_intent` itself consults), applied to the //! candidates' own [`SketchAlgorithm`]s. @@ -265,9 +265,9 @@ //! interact — which both shipped strategies' one-round convergence (see //! "Termination" above) makes the common case — but it's the wrong answer //! whenever they do. Concretely: [`CostModel::cse_share_decision`] costs a -//! [`SharedSubtreeStrategy`] group by comparing a `consumer_count`-scaled +//! [`SharedSubDagStrategy`] group by comparing a `consumer_count`-scaled //! recompute cost against a fixed maintenance cost — but a **nested** -//! `SharedSubtreeStrategy` group's *true* recompute burden isn't its own +//! `SharedSubDagStrategy` group's *true* recompute burden isn't its own //! raw [`TargetSubDAGCandidates::consumer_count`] (how many operator-child positions //! directly reference it) whenever an ancestor on the path to it is //! *itself* being recomputed independently rather than shared: recomputing @@ -285,7 +285,7 @@ //! from *already-decided* ancestors). For every site it computes the //! **effective consumer count** — how many times that site actually runs //! once every ancestor's own selected candidate is accounted for — and, for -//! every [`SharedSubtreeStrategy`]-shaped group, re-decides +//! every [`SharedSubDagStrategy`]-shaped group, re-decides //! [`CostModel::cse_share_decision`] against *that* corrected count instead //! of the group's raw structural one. When that group also contains a //! non-CSE alternative such as a semantic rewrite, the chosen CSE candidate @@ -296,7 +296,7 @@ //! multiplicity to exactly `1` for everything beneath it (one shared //! execution backs every use of it); a group that chooses //! `RecomputeIndependently` — or has no Share/Recompute decision of its own -//! at all, i.e. isn't itself a `SharedSubtreeStrategy` shape — passes its +//! at all, i.e. isn't itself a `SharedSubDagStrategy` shape — passes its //! *own* effective count straight through to whatever it references, //! transitively composing contributions from every ancestor on the path, //! not just the immediate parent. @@ -317,8 +317,8 @@ //! documented follow-up rather than silently overclaimed: //! //! - [`CostModel::rank_candidates`]/[`CostModel::size_params`] — the hooks -//! [`SketchAlgorithmStrategy`] groups rank by — take no `consumer_count` -//! parameter at all today, so a `SketchAlgorithmStrategy` group's selection +//! [`ASAPStrategies`] groups rank by — take no `consumer_count` +//! parameter at all today, so a `ASAPStrategies` group's selection //! here still falls back to [`rank_group`]'s ordinary (consumer-count- //! blind) local ranking, even though its own //! [`TargetSubDAGSelection::effective_consumer_count`] is computed and exposed @@ -330,10 +330,10 @@ //! groups included. //! - This is not an exhaustive search over combinations of choices for a //! provably-global optimum in every case. [`CostModel::cse_share_decision`] -//! is still a *local*, pairwise comparison at each `SharedSubtreeStrategy` +//! is still a *local*, pairwise comparison at each `SharedSubDagStrategy` //! site (recompute-total vs. one fixed maintenance cost) — this module //! just now feeds it a *correct* input instead of an *incorrect* one. Two -//! sibling `SharedSubtreeStrategy` groups that could trade off against +//! sibling `SharedSubDagStrategy` groups that could trade off against //! each other under some shared resource budget (memory, say) still //! aren't jointly optimized here — this crate has no //! cardinality/statistics estimation to bound a combinatorial search like @@ -345,30 +345,33 @@ use crate::accuracy::estimators::{ cms::{cms_depth, cms_width}, saturating_ceil, }; +use asap_types::ir::non_asap::any_measure_filtered; +use asap_types::pre_asap::resolve_column_ref; use std::cell::RefCell; use std::collections::{HashMap, HashSet, VecDeque}; -use asap_types::post_asap::{ - validate_execution_data_states_at, EntityIdentity, ExactKind, ExactOperation, - ExactOperationSchemaError, ExactParams, ExecutionDataState, ExecutionDataStateError, - ExecutionTiming, GroupingStrategy, NonNegativeWeightProof, SamplingKind, SamplingParams, - SketchAlgorithm, SketchKind, SketchParams, SketchQuery as PostAsapSketchQuery, StatModelKind, - StatModelParams, SummaryExpr, SummaryFamilyType, SummaryField, SummaryInputExpr, SummaryNode, - SummarySchema, SummaryUpdate, ValueOperation, WaveletKind, WaveletParams, WeightDomain, +use asap_types::ir::cse::{share_common_subdags, structural_hash, HashCache}; +use asap_types::ir::operator_properties::{BinaryOpKind, JoinKind, Reduction}; +use asap_types::ir::timing::validate_default; +use asap_types::ir::SchemaDerivationError; +use asap_types::ir::{ + ASAPOp, BinaryOperator, NonASAPOp, Operator, OperatorNode, Predicate, ProjectItem, ScalarExpr, + SortKey, }; use asap_types::post_asap::{AccuracyError, CompositionOperator, GuaranteeSource, ResultGuarantee}; +use asap_types::post_asap::{ + EntityIdentity, ExactKind, ExactOperationSchemaError, ExactParams, ExecutionDataStateError, + ExecutionTiming, Field, FieldDataType, GroupingStrategy, NonNegativeWeightProof, SamplingKind, + SamplingParams, Schema, SketchAlgorithm, SketchKind, SketchParams, + SketchStatistic as PostAsapSketchStatistic, StatModelKind, StatModelParams, SummaryInputExpr, + SummaryUpdate, WaveletKind, WaveletParams, WeightDomain, +}; use asap_types::pre_asap::agg_intent::{agg_is_mergeable, AggIntent}; -use asap_types::pre_asap::column_resolution::resolve_column_ref; -use asap_types::pre_asap::cse::{share_common_subtrees, structural_hash, HashCache}; use asap_types::pre_asap::expr_ir::{ArithmeticOpKind, ColumnRef}; -use asap_types::pre_asap::query_expr::any_measure_filtered; -use asap_types::pre_asap::query_expr::{ - BinaryOpKind, Predicate, QueryExpr, QueryExprError, Reduction, -}; -use asap_types::pre_asap::schema::{ColumnId, Schema}; +use asap_types::pre_asap::schema::ColumnId; use asap_types::types::AccuracyTarget; use asap_types::workload::{DataWorkload, QueryRecurrence, QueryWorkload, RepeatedDemand}; -use std::rc::Rc; +use std::rc::{Rc, Weak}; use thiserror::Error; use crate::accuracy::reconciliation::AccuracyReconciliationStrategy; @@ -390,20 +393,20 @@ use crate::rollup::RollupStrategy; use crate::topk_reuse::TopKLimitReuseStrategy; /// Errors from the pre-ASAP → post-ASAP replacement/construction path -/// ([`realize_child`] and [`keep_pre_asap`]). Moved here from the former +/// ([`realize_child`] and [`retain_exact`]). Moved here from the former /// `bind.rs` (issue #251): this is what a [`ReplacementStrategy`] /// implementor's own construction path can realistically fail with — -/// schema derivation over a pre-ASAP [`QueryExpr`] — not something specific -/// to workload-wide orchestration. +/// schema derivation over a pre-ASAP [`OperatorNode`] sub-DAG — not +/// something specific to workload-wide orchestration. #[derive(Debug, Error)] pub enum RealizationError { - /// Schema derivation failed while lifting an edge to `SummarySchema`. + /// Schema derivation failed while lifting an edge to `Schema`. #[error("schema derivation failed during pre-ASAP → post-ASAP binding: {0}")] - Schema(#[from] QueryExprError), + Schema(#[from] SchemaDerivationError), /// The candidate is accuracy-illegal (issue #172): its composed /// guarantee has no sound propagation rule, or misses the applicable /// `AccuracyTarget`. Fail-closed — the candidate is never constructed - /// with the child "treated as exact". [`SketchAlgorithmStrategy::propose`] + /// with the child "treated as exact". [`ASAPStrategies::propose`] /// records it as a [`RejectedCandidate`] instead of a candidate. #[error("accuracy-illegal candidate: {0}")] Accuracy(#[from] AccuracyError), @@ -413,8 +416,8 @@ pub enum RealizationError { /// would change its semantics. #[error("unsupported physical summary realization: {0}")] PhysicalRealization(&'static str), - /// A constructed plan violates the update/readout phase contract - /// (issue #171) — e.g. a summary readout placed beneath a maintained + /// A constructed plan violates the update/evaluation phase contract + /// (issue #171) — e.g. a summary evaluation placed beneath a maintained /// `SummaryAgg`. Detected at construction, never at runtime. #[error("execution-data_state violation in post-ASAP plan: {0}")] ExecutionDataState(#[from] ExecutionDataStateError), @@ -426,28 +429,28 @@ pub enum RealizationError { /// A pre-ASAP sub-DAG a [`ReplacementStrategy`] knows how to replace. /// -/// `root` is a reference into the workload's own [`QueryExpr`] tree (an -/// `Rc`, the same currency [`search_workload`] and -/// `asap_types::pre_asap::cse::share_common_subtrees` already thread through -/// this crate's public API — not a bare `&QueryExpr` — so a strategy that +/// `root` is a reference into the workload's own [`OperatorNode`] DAG (an +/// `Rc`, the same currency [`search_workload`] and +/// `asap_types::ir::cse::share_common_subdags` already thread through +/// this crate's public API — not a bare `&OperatorNode` — so a strategy that /// needs the node's own `Rc` identity, not just its shape, has it available /// without the caller re-deriving it). /// /// `consumer_count` is how many locations across the workload reference this -/// exact `Rc` — 1 for an ordinary single-use node and 2+ for a shared subtree. +/// exact `Rc` — 1 for an ordinary single-use node and 2+ for a shared sub-DAG. /// [`search_workload_with`] computes the workload-wide value during target /// discovery. [`TargetSubDAG::new`] defaults it to `1` for callers invoking a /// strategy against one node in isolation. A strategy that only cares about -/// `root`'s shape (for example, [`SketchAlgorithmStrategy`]) can ignore the -/// count; [`SharedSubtreeStrategy`] consults it directly. +/// `root`'s shape (for example, [`ASAPStrategies`]) can ignore the +/// count; [`SharedSubDagStrategy`] consults it directly. /// /// `strictest_sibling_accuracy` is the strictest accuracy among workload /// siblings that read the same summary input as `root`, when stricter than -/// `root`'s own. [`search_workload_with`] sets it; [`SketchAlgorithmStrategy`] +/// `root`'s own. [`search_workload_with`] sets it; [`ASAPStrategies`] /// also sizes a candidate to it. #[derive(Debug, Clone, Copy)] pub struct TargetSubDAG<'a> { - pub root: &'a Rc, + pub root: &'a Rc, pub consumer_count: usize, pub strictest_sibling_accuracy: Option<&'a AccuracyTarget>, } @@ -455,7 +458,7 @@ pub struct TargetSubDAG<'a> { impl<'a> TargetSubDAG<'a> { /// A target assumed to have exactly one consumer — the common case for a /// caller that isn't already tracking cross-workload sharing. - pub fn new(root: &'a Rc) -> Self { + pub fn new(root: &'a Rc) -> Self { Self { root, consumer_count: 1, @@ -465,7 +468,7 @@ impl<'a> TargetSubDAG<'a> { /// A target with an explicit `consumer_count`, used by workload discovery /// and by callers that already know how many locations reference `root`. - pub fn with_consumer_count(root: &'a Rc, consumer_count: usize) -> Self { + pub fn with_consumer_count(root: &'a Rc, consumer_count: usize) -> Self { Self { root, consumer_count, @@ -481,25 +484,35 @@ impl<'a> TargetSubDAG<'a> { /// — into "one candidate among several", each with its own /// [`ReplacementSubDAG`]. #[derive(Debug, Clone)] +#[allow(clippy::large_enum_variant)] // Keep the public strategy API value-based. pub enum Replacement { - /// A fully bound post-ASAP summary decision, for one particular - /// candidate realization of the target. - Summary(Rc), - /// A pre-ASAP rewrite: still a logical [`QueryExpr`], structurally - /// different from the target's own `root` (e.g. sharing vs. not sharing - /// a subtree) but semantically equivalent to it. - Rewrite(Rc), + /// A sub-DAG that replaces the target: either a bound summary decision + /// (a DAG containing ASAP operators, for one particular candidate + /// realization of the target) or a pre-ASAP rewrite (a logical sub-DAG + /// with no ASAP operator, structurally different from the target's own + /// `root` — e.g. sharing vs. not sharing a sub-DAG — but semantically + /// equivalent to it). [`is_logical_rewrite`] tells the two apart. + SubDag(Rc), /// An exact operator composed over another target's *own* selected - /// decision across an explicit update/readout boundary (issue #171): - /// `ValueOperationAtQueryTime` over a child's summary readout, or + /// decision across an explicit update/evaluation boundary (issue #171): + /// `ValueOperationAtQueryTime` over a child's summary evaluation, or /// `ValueOperationAtIngestionTime` feeding a maintained summary above. Carries only a /// reference to the child target — [`CandidateLogicalASAPDAGs::global_selection`] /// commits the compatible parent/child pair and /// [`GlobalSelection::assemble_selected_dag`] links it into one validated - /// `SummaryNode`. See [`crate::exact_composition`]. + /// `OperatorNode` DAG. See [`crate::exact_composition`]. ExactComposition(ExactComposition), } +/// Whether a [`Replacement::SubDag`] is a pure logical rewrite: a sub-DAG +/// with no ASAP operator and no guarantee established yet (the shape every +/// front end emits and every rewrite strategy builds). A bound summary +/// decision contains an ASAP operator, or is a kept pre-ASAP sub-DAG that +/// already carries its exact guarantee. +pub fn is_logical_rewrite(node: &OperatorNode) -> bool { + node.guarantee.is_none() && !node.contains_asap() +} + /// One candidate replacement for a [`TargetSubDAG`], plus a human-readable /// `rationale` explaining why it's a valid candidate (meant for a /// report/log/debugging a search engine's choices, not machine parsing — @@ -522,11 +535,12 @@ pub struct ReplacementSubDAG { impl ReplacementSubDAG { /// Whether this summary still needs accuracy/domain evidence before it can /// be treated as certified. A missing guarantee on any summary candidate - /// is unknown; exact `KeepPreAsap` carries an explicit exact guarantee. + /// (a sub-DAG whose root is an ASAP operator) is unknown; a kept + /// pre-ASAP sub-DAG carries an explicit exact guarantee. pub fn has_missing_accuracy_evidence(&self) -> bool { matches!( &self.replacement, - Replacement::Summary(node) if has_missing_accuracy_evidence(node) + Replacement::SubDag(node) if !is_logical_rewrite(node) && has_missing_accuracy_evidence(node) ) } @@ -538,8 +552,12 @@ impl ReplacementSubDAG { Replacement::ExactComposition(composition) => { cost_model.value_operation_support_evidence(&composition.op, composition.placement) } - Replacement::Summary(node) => cost_model.summary_support_evidence(node), - Replacement::Rewrite(_) => Some(true), + // Any summary decision, including one rooted in a relational + // operator above its evaluations, asks the deployment for support. + Replacement::SubDag(node) if !is_logical_rewrite(node) => { + cost_model.summary_support_evidence(node) + } + Replacement::SubDag(_) => Some(true), } } } @@ -573,7 +591,7 @@ pub enum ReplacementProvenance { /// A finalized whole-query result over rows carrying the PromQL series /// identity, which the logical root does not expose (see /// [`ReplacementStrategy::propose_for_root`]). Default selection never - /// commits it, because its readout must be validated and priced by + /// commits it, because its evaluation must be validated and priced by /// deployment; otherwise it would silently replace the logical plan. RootPhysicalRealization, } @@ -614,7 +632,7 @@ pub struct Proposals { /// of this trait or any existing strategy required. /// /// `replacements` is only meaningful when `matches` would return `true` for -/// the same target; both [`SketchAlgorithmStrategy`] and [`SharedSubtreeStrategy`] +/// the same target; both [`ASAPStrategies`] and [`SharedSubDagStrategy`] /// return an empty `Vec` rather than panicking when called on a target they /// don't match, so a caller that skips the `matches` check first still gets a /// safe (merely uninformative) answer instead of a crash. @@ -656,7 +674,7 @@ pub trait ReplacementStrategy { /// (for example, the PromQL series identity), so /// [`search_workload_with_targets`] asks only workload roots, once each. /// They decide what to compute, never placement. Default: none. - fn propose_for_root(&self, _root: &Rc, _target: &AccuracyTarget) -> Proposals { + fn propose_for_root(&self, _root: &Rc, _target: &AccuracyTarget) -> Proposals { Proposals::default() } } @@ -673,41 +691,41 @@ pub trait ReplacementStrategy { /// [`realizations_for_intent`] is where every valid realization gets /// enumerated, exhaustive and ranked (most-preferred first) — this crate has /// no separate function that computes just "the one" `Realization` -/// independently of that list. [`SketchAlgorithmStrategy`] is the sole +/// independently of that list. [`ASAPStrategies`] is the sole /// consumer: it wraps every entry of this list into its own bound -/// [`SummaryNode`] and returns all of them, ranked — a caller wanting a +/// [`OperatorNode`] and returns all of them, ranked — a caller wanting a /// single answer keeps the first one itself (see the module docs above). #[derive(Debug, Clone, PartialEq)] pub enum Realization { /// An exact **mergeable** accumulator (partial state ≡ the value /// itself: `Sum` / `Count` / `Min` / `Max` / `Rate` / `Increase`). The - /// built state *is* the answer already — no `SummaryEstimate` readout + /// built state *is* the answer already — no `SummaryEstimate` evaluation /// step. ExactAggregate { kind: ExactKind, params: ExactParams, }, /// An approximate sketch sized to the intent's [`AccuracyTarget`]. - /// Needs a `SummaryEstimate` readout to recover a value. Already + /// Needs a `SummaryEstimate` evaluation to recover a value. Already /// classified into its [`SketchKind`] category (`SketchKind::new` /// having been called) — construction always goes through that /// classifier, never this variant directly. Sketch(SketchKind), /// A sampling-based summary (a retained row subset). Needs a - /// `SummaryEstimate` readout. Not chosen by any core `AggIntent` + /// `SummaryEstimate` evaluation. Not chosen by any core `AggIntent` /// dispatch today — see the module docs. Sample { kind: SamplingKind, params: SamplingParams, }, - /// A wavelet-transform summary. Needs a `SummaryEstimate` readout. Not + /// A wavelet-transform summary. Needs a `SummaryEstimate` evaluation. Not /// chosen by any core `AggIntent` dispatch today — see the module docs. Wavelet { kind: WaveletKind, params: WaveletParams, }, /// A fitted statistical/parametric-model summary. Needs a - /// `SummaryEstimate` readout. Not chosen by any core `AggIntent` + /// `SummaryEstimate` evaluation. Not chosen by any core `AggIntent` /// dispatch today — see the module docs. StatModel { kind: StatModelKind, @@ -826,7 +844,7 @@ pub fn accuracy_target(intent: &AggIntent) -> Option<&AccuracyTarget> { /// (most-preferred first via `cost_model`) — the *only* place this crate /// decides what an `AggIntent` may become. Nothing in this crate computes /// "the one" `Realization` independently of this list: -/// [`SketchAlgorithmStrategy`] keeps every entry as a candidate, and a caller +/// [`ASAPStrategies`] keeps every entry as a candidate, and a caller /// that wants a single executable answer takes the head of *that* strategy's /// output itself. /// @@ -834,7 +852,7 @@ pub fn accuracy_target(intent: &AggIntent) -> Option<&AccuracyTarget> { /// explicit realization is a compile error, and the coverage-matrix test pins /// each variant's category. /// -/// `pub(crate)`: [`SketchAlgorithmStrategy::replacements`] is this module's +/// `pub(crate)`: [`ASAPStrategies::replacements`] is this module's /// own caller; `grouping::HydraGroupingStrategy` (issue #256) is the one /// caller outside it, needing the exact same already-ranked candidate list /// to find the `Realization::Sketch` matching the Hydra-eligible kind it @@ -1189,9 +1207,9 @@ pub fn posterior_aware_size_params( } } -// ── SketchAlgorithmStrategy ───────────────────────────────────────────────── +// ── ASAPStrategies ───────────────────────────────────────────────── -/// A single static instance so [`SketchAlgorithmStrategy::default_cost_model`] +/// A single static instance so [`ASAPStrategies::default_cost_model`] /// can hand out a `&'static dyn CostModel` without heap-allocating one — /// `DefaultCostModel` is a unit struct with no state, so one instance serves /// every caller. @@ -1226,14 +1244,17 @@ impl<'a> CandidatePlanningInputs<'a> { } } -/// Wraps [`realizations_for_intent`]'s exhaustive, ranked list directly: for -/// a bindable `Aggregate`, every valid candidate summary realization as its -/// own [`ReplacementSubDAG`]. +/// Proposes the supported ASAP realizations for a bindable aggregate, including +/// exact accumulators, approximate sketches, and supported maintained populations. +/// Each valid realization becomes its own [`ReplacementSubDAG`]. +/// +/// [`realizations_for_intent`] enumerates summary families; extension hooks can +/// supply additional supported families. This is not limited to sketch algorithms. /// /// Ranked (only to *order the enumeration*, never to drop a candidate) via a /// [`CostModel`] — [`DefaultCostModel`] unless constructed with -/// [`SketchAlgorithmStrategy::new`] — so a deployment-specific cost model's -/// other hooks (`size_params`, `realize_extension`, `readout_extension`) are +/// [`ASAPStrategies::new`] — so a deployment-specific cost model's +/// other hooks (`size_params`, `realize_extension`, `evaluation_extension`) are /// still consulted while binding each candidate. /// /// The one thing that *does* drop a candidate is accuracy legality (issue @@ -1244,11 +1265,11 @@ impl<'a> CandidatePlanningInputs<'a> { /// [`ReplacementStrategy::propose`] as a [`RejectedCandidate`]. See /// [`crate::accuracy`]'s module docs for the rules and the precedence /// between root and per-node targets. -pub struct SketchAlgorithmStrategy<'a> { +pub struct ASAPStrategies<'a> { planning_inputs: CandidatePlanningInputs<'a>, } -impl SketchAlgorithmStrategy<'static> { +impl ASAPStrategies<'static> { /// A strategy that ranks/binds via the built-in [`DefaultCostModel`] — /// what a deployment gets with no custom cost model plugged in. pub fn default_cost_model() -> Self { @@ -1258,7 +1279,7 @@ impl SketchAlgorithmStrategy<'static> { } } -impl<'a> SketchAlgorithmStrategy<'a> { +impl<'a> ASAPStrategies<'a> { /// A strategy that ranks/binds via `cost_model` instead of the built-in /// static preference order — the same customization point /// [`realizations_for_intent`] already offers. Accuracy legality stays @@ -1313,47 +1334,46 @@ impl<'a> SketchAlgorithmStrategy<'a> { /// treats a range of historical samples as the instant vector. pub fn current_series_topk_candidates( &self, - root: &Rc, + root: &Rc, accuracy: &AccuracyTarget, ) -> Proposals { - let QueryExpr::Limit { - n, + let Some(NonASAPOp::Limit { + n: Some(n), offset: 0, child, - } = root.as_ref() + .. + }) = root.non_asap() else { return Proposals::default(); }; - let QueryExpr::Sort { + let Some(NonASAPOp::Sort { keys, partition_by, child, - } = child.as_ref() + }) = child.non_asap() else { return Proposals::default(); }; let [key] = keys.as_slice() else { return Proposals::default(); }; - let QueryExpr::Column(value) = key.expr else { - return Proposals::default(); - }; - let Ok(schema) = child.output_schema() else { + let ScalarExpr::Column(value) = key.expr else { return Proposals::default(); }; + let schema = &child.schema; if key.ascending || key.nulls_first || partition_by.is_without() || !schema.has_promql_series_identity() || !schema - .columns + .fields .get(value) .is_some_and(|column| column.name == "value") || !is_current_series_source(child) { return Proposals::default(); } - let ranked = Rc::new(QueryExpr::Aggregate { + let Ok(ranked) = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::Reduce(partition_by.clone()), measures: vec![AggIntent::TopK { k: *n, @@ -1363,10 +1383,89 @@ impl<'a> SketchAlgorithmStrategy<'a> { filters: vec![], having: None, child: Rc::clone(child), - }); + }) else { + return Proposals::default(); + }; self.propose_with(&ranked, None, None) } + /// Fixed-window maintenance can finalize each series' counter state and + /// build a fresh heap or grouped Sum for that evaluation window. Deployment must provide + /// a complete, synchronized population and bind the matching window; this + /// candidate never incrementally adds one window's rates to another. + /// + pub fn fixed_window_rate_candidates(&self, root: &Rc) -> Proposals { + fn place(node: &Rc) -> Option> { + retime_rate_finalize(node, ExecutionTiming::IngestionTime, true) + } + let timed = |node: &Rc| { + asap_types::ir::timing::apply_lifecycle_timings( + node, + &asap_types::ir::timing::LifecycleAssignment::default_maintained(), + &mut asap_types::ir::timing::TimingMemo::new(), + ) + .ok() + .and_then(|timed| asap_types::ir::export::compile_post_asap_dag(&timed).ok()) + }; + let mut proposals = self.propose_with(root, None, None); + proposals.candidates.retain_mut(|candidate| { + let Replacement::SubDag(node) = &candidate.replacement else { + return false; + }; + let Some(dag) = timed(node) else { + return false; + }; + if !dag.nodes.iter().any(|node| match &node.payload { + asap_types::ir::export::PostAsapOperatorPayload::SummaryAgg { + family: FieldDataType::Sketch(kind, _), + .. + } => matches!( + kind.algorithm(), + SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap + ), + asap_types::ir::export::PostAsapOperatorPayload::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Sum, _), + .. + } => true, + _ => false, + }) { + return false; + } + let Some(placed) = place(node) else { + return false; + }; + if timed(&placed).is_none() { + return false; + } + let Ok(placed) = finalize_query_candidate(placed, root) else { + return false; + }; + candidate.replacement = Replacement::SubDag(placed); + candidate + .rationale + .push_str("; fixed-window precompute over complete per-series counter states"); + true + }); + proposals + } + + /// Retain grouped Sum after a per-series Rate evaluation as a query-time + /// candidate alongside its complete-window maintenance placement. + /// + pub fn query_time_rate_aggregation_candidates(&self, root: &Rc) -> Proposals { + let mut proposals = self.fixed_window_rate_candidates(root); + proposals.candidates.retain_mut(|candidate| { + let Replacement::SubDag(node) = &candidate.replacement else { return false }; + if !matches!(&node.operator, Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) + if matches!(&child.operator, Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Sum, _), .. }))) { return false; } + let Some(query_time) = retime_rate_finalize(node, ExecutionTiming::QueryTime, false) else { return false }; + candidate.replacement = Replacement::SubDag(query_time); + candidate.rationale = "query-time grouped Sum over complete per-series Rate evaluations".into(); + true + }); + proposals + } + pub(crate) fn from_planning_inputs(planning_inputs: CandidatePlanningInputs<'a>) -> Self { Self { planning_inputs } } @@ -1379,20 +1478,21 @@ impl<'a> SketchAlgorithmStrategy<'a> { /// with the sibling that needs it. fn propose_with( &self, - root: &Rc, + root: &Rc, intent_override: Option<&AggIntent>, strictest_sibling: Option<&AccuracyTarget>, ) -> Proposals { let mut proposals = Proposals::default(); - // A selected logical rewrite otherwise remains KeepPreAsap during DAG - // assembly. Also expose its concrete summary realization for selection. + // A selected logical rewrite otherwise stays a kept pre-ASAP sub-DAG + // during DAG assembly. Also expose its concrete summary realization + // for selection. if intent_override.is_none() { if let Some(rewritten) = crate::rewrite::composed_aggregate_rewrite(root) { if let Ok(node) = realize_child_with(&rewritten, self.planning_inputs, None) { - if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) { + if node.contains_asap() { proposals.candidates.push(ReplacementSubDAG { - replacement: Replacement::Summary(node), - strategy: "SketchAlgorithmStrategy", + replacement: Replacement::SubDag(node), + strategy: "ASAPStrategies", provenance: ReplacementProvenance::SummaryRealization, rationale: "realize a schema-preserving composition of temporal and grouped accumulators".into(), }); @@ -1402,8 +1502,8 @@ impl<'a> SketchAlgorithmStrategy<'a> { } if let Ok(Some(node)) = exact_topk_over_temporal_values(root, self.planning_inputs) { proposals.candidates.push(ReplacementSubDAG { - replacement: Replacement::Summary(node), - strategy: "SketchAlgorithmStrategy", + replacement: Replacement::SubDag(node), + strategy: "ASAPStrategies", provenance: ReplacementProvenance::SummaryRealization, rationale: "select exact Top-K from independently maintained temporal values" .into(), @@ -1412,8 +1512,8 @@ impl<'a> SketchAlgorithmStrategy<'a> { if intent_override.is_none() { if let Ok(Some(node)) = realize_temporal_average(root, self.planning_inputs, None) { proposals.candidates.push(ReplacementSubDAG { - replacement: Replacement::Summary(node), - strategy: "SketchAlgorithmStrategy", + replacement: Replacement::SubDag(node), + strategy: "ASAPStrategies", provenance: ReplacementProvenance::SummaryRealization, rationale: "read temporal average from sum/count only within the finite arithmetic domain; otherwise execute the original average".into(), }); @@ -1427,8 +1527,8 @@ impl<'a> SketchAlgorithmStrategy<'a> { "preserve exact PromQL arithmetic over independently realized summary operands" }; proposals.candidates.push(ReplacementSubDAG { - replacement: Replacement::Summary(node), - strategy: "SketchAlgorithmStrategy", + replacement: Replacement::SubDag(node), + strategy: "ASAPStrategies", provenance: ReplacementProvenance::SummaryRealization, rationale: rationale.into(), }); @@ -1511,21 +1611,21 @@ impl<'a> SketchAlgorithmStrategy<'a> { else { continue; }; - let family = SummaryFamilyType::Sketch(kind.clone(), GroupingStrategy::default()); + let family = FieldDataType::Sketch(kind.clone(), GroupingStrategy::default()); let Some(child) = aggregate_child(root) else { continue; }; - let QueryExpr::Aggregate { reduction, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { reduction, .. }) = root.non_asap() else { continue; }; let Ok(input) = realize_physical_summary_input(intent, &family, reduction, child) else { continue; }; - let readout_query = readout(intent, &input.input, planning_inputs.cost); + let evaluation_query = evaluation(intent, &input.input, planning_inputs.cost); let Some(local) = planning_inputs .accuracy - .local_guarantee(&family, &readout_query) + .local_guarantee(&family, &evaluation_query) else { continue; }; @@ -1536,7 +1636,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { let allocations = planning_inputs.allocator.allocations(target, &shape); if allocations.is_empty() { proposals.rejected.push(RejectedCandidate { - strategy: "SketchAlgorithmStrategy", + strategy: "ASAPStrategies", description: rationale.clone(), error: AccuracyError::NoLegalAllocation { target: target.clone(), @@ -1590,10 +1690,10 @@ impl<'a> SketchAlgorithmStrategy<'a> { } if proposals.candidates.is_empty() { if let Some(error) = &proposals.domain_error { - if let Ok(node) = keep_pre_asap(root) { + if let Ok(node) = retain_exact(root) { proposals.candidates.push(ReplacementSubDAG { - strategy: "SketchAlgorithmStrategy", - replacement: Replacement::Summary(node), + strategy: "ASAPStrategies", + replacement: Replacement::SubDag(node), provenance: ReplacementProvenance::SummaryRealization, rationale: format!( "{} stays pre-ASAP because summary construction crosses an illegal \ @@ -1612,16 +1712,16 @@ impl Proposals { /// File one construction attempt: a legal node becomes a candidate, an /// [`RealizationError::Accuracy`] becomes a [`RejectedCandidate`], and a /// schema-derivation failure is skipped exactly as it always was. - fn record(&mut self, rationale: String, built: Result, RealizationError>) { + fn record(&mut self, rationale: String, built: Result, RealizationError>) { match built { Ok(node) => self.candidates.push(ReplacementSubDAG { - strategy: "SketchAlgorithmStrategy", - replacement: Replacement::Summary(node), + strategy: "ASAPStrategies", + replacement: Replacement::SubDag(node), provenance: ReplacementProvenance::SummaryRealization, rationale, }), Err(RealizationError::Accuracy(error)) => self.rejected.push(RejectedCandidate { - strategy: "SketchAlgorithmStrategy", + strategy: "ASAPStrategies", description: rationale, error, }), @@ -1638,14 +1738,14 @@ impl Proposals { } /// The `child` of a [`bindable_intent`]-shaped `Aggregate`. -fn aggregate_child(node: &QueryExpr) -> Option<&Rc> { - match node { - QueryExpr::Aggregate { child, .. } => Some(child), +fn aggregate_child(node: &OperatorNode) -> Option<&Rc> { + match node.non_asap() { + Some(NonASAPOp::Aggregate { child, .. }) => Some(child), _ => None, } } -impl ReplacementStrategy for SketchAlgorithmStrategy<'_> { +impl ReplacementStrategy for ASAPStrategies<'_> { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { bindable_intent(target.root).is_some() || is_supported_exact_binary(target.root) } @@ -1664,24 +1764,24 @@ impl ReplacementStrategy for SketchAlgorithmStrategy<'_> { /// for the identity-carrying root. Placement variants (for example, /// fixed-window or query-time Rate aggregation) are not listed here: the /// lifecycle assigns timing and the physical compiler reads it. - fn propose_for_root(&self, root: &Rc, target: &AccuracyTarget) -> Proposals { + fn propose_for_root(&self, root: &Rc, target: &AccuracyTarget) -> Proposals { let Ok(typed) = asap_types::pre_asap::schema::with_promql_series_identity(root) else { return Proposals::default(); }; - let typed = Rc::new(typed); + let mut proposals = self.current_series_topk_candidates(&typed, target); for mut candidate in std::mem::take(&mut proposals.candidates) { - let Replacement::Summary(node) = candidate.replacement else { + let Replacement::SubDag(node) = candidate.replacement else { continue; }; let Ok(node) = finalize_query_candidate(node, &typed) else { continue; }; let duplicate = proposals.candidates.iter().any(|existing| { - matches!(&existing.replacement, Replacement::Summary(other) if *other == node) + matches!(&existing.replacement, Replacement::SubDag(other) if *other == node) }); if !duplicate { - candidate.replacement = Replacement::Summary(node); + candidate.replacement = Replacement::SubDag(node); candidate.provenance = ReplacementProvenance::RootPhysicalRealization; proposals.candidates.push(candidate); } @@ -1754,11 +1854,11 @@ pub(crate) fn describe_intent(intent: &AggIntent) -> String { } } -// ── realize_child / keep_pre_asap: rank-and-take-first, and its fallback ── +// ── realize_child / retain_exact: rank-and-take-first, and its fallback ── -/// Rank-and-take-first selector for a single [`QueryExpr`] node: enumerate -/// every candidate via [`SketchAlgorithmStrategy::replacements`], keep the -/// `cost_model`-preferred (first) one, and fall back to [`keep_pre_asap`] +/// Rank-and-take-first selector for a single [`OperatorNode`]: enumerate +/// every candidate via [`ASAPStrategies::replacements`], keep the +/// `cost_model`-preferred (first) one, and fall back to [`retain_exact`] /// when there's no candidate at all — **not** a general single-answer API /// for a whole workload. Use [`CandidateLogicalASAPDAGs::global_selection`] and DAG assembly /// for coordinated logical selection; physical deployment remains downstream. @@ -1770,16 +1870,16 @@ pub(crate) fn describe_intent(intent: &AggIntent) -> String { /// ([`construct_summary_agg`], so a nested aggregate gets its own /// independent enumeration instead of inheriting the parent's forced /// candidate), from this module's own [`realize_one`] (the representative -/// bound `SummaryNode` [`cse_preference`] needs for a +/// bound `OperatorNode` [`cse_preference`] needs for a /// [`CostModel::cse_share_decision`] comparison), and from /// [`crate::cost_model::DefaultCostModel::estimate_cost`] (the same /// representative-node need, for a [`Replacement::Rewrite`] candidate's own /// cost estimate). Every other caller goes through -/// [`SketchAlgorithmStrategy::replacements`] directly and decides for itself. +/// [`ASAPStrategies::replacements`] directly and decides for itself. pub(crate) fn realize_child( - root: &Rc, + root: &Rc, cost_model: &dyn CostModel, -) -> Result, RealizationError> { +) -> Result, RealizationError> { realize_child_with( root, CandidatePlanningInputs::with_default_accuracy(cost_model), @@ -1796,17 +1896,17 @@ pub(crate) fn realize_child( /// budget. A child whose declared target is `Exact` keeps it: an allocation /// never approximates something the caller declared exact. fn exact_topk_over_temporal_values( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, -) -> Result>, RealizationError> { - let QueryExpr::Aggregate { +) -> Result>, RealizationError> { + let Some(NonASAPOp::Aggregate { reduction, measures, output_names: _, filters, having: None, child, - } = root.as_ref() + }) = root.non_asap() else { return Ok(None); }; @@ -1816,19 +1916,19 @@ fn exact_topk_over_temporal_values( let [AggIntent::TopK { k, .. }] = measures.as_slice() else { return Ok(None); }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, child: input, .. - } = child.as_ref() + }) = child.non_asap() else { return Ok(None); }; - if !matches!(input.as_ref(), QueryExpr::TimeRange { .. }) { + if !matches!(input.non_asap(), Some(NonASAPOp::TimeRange { .. })) { return Ok(None); } let values = realize_child_with(child, planning_inputs, Some(&AccuracyTarget::Exact))?; - if matches!(values.expr, SummaryExpr::KeepPreAsap(_)) + if !values.contains_asap() || !values .guarantee .as_ref() @@ -1844,61 +1944,63 @@ fn exact_topk_over_temporal_values( ))? .clone(); let score = ranking_score_index(child, &values.schema)?; - let sorted = Rc::new(SummaryNode { - guarantee: values.guarantee.clone(), - schema: values.schema.clone(), - expr: SummaryExpr::ValueOperation { - child: values, - operation: ValueOperation::Sort { - keys: vec![asap_types::pre_asap::SortKey { - expr: QueryExpr::Column(score), + let guarantee = values.guarantee.clone(); + let schema = values.schema.clone(); + let sorted = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Sort { + keys: vec![SortKey { + expr: ScalarExpr::Column(score), ascending: false, nulls_first: false, }], partition_by: partition_by.clone(), - }, - timing: ExecutionTiming::QueryTime, - }, - }); - let node = Rc::new(SummaryNode { - guarantee: sorted.guarantee.clone(), - schema: sorted.schema.clone(), - expr: SummaryExpr::ValueOperation { - child: sorted, - operation: ValueOperation::Limit { - n: *k, + child: values, + }), + schema.clone(), + ) + .with_guarantee(guarantee.clone()), + ); + let node = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Limit { + n: Some(*k), offset: 0, partition_by, - }, - timing: ExecutionTiming::QueryTime, - }, - }); - validate_execution_data_states_at(&node, ExecutionDataState::QUERY_ROWS)?; + child: sorted, + }), + schema, + ) + .with_guarantee(guarantee), + ); + validate_default(&node, ExecutionTiming::QueryTime)?; Ok(Some(node)) } fn realize_temporal_average( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, target: Option<&AccuracyTarget>, -) -> Result>, RealizationError> { +) -> Result>, RealizationError> { let Some(components) = crate::rewrite::temporal_average_components(root) else { return Ok(None); }; let mut node = realize_child_with(&components, planning_inputs, target)?; - let SummaryExpr::BinaryOp { operator, .. } = &mut Rc::make_mut(&mut node).expr else { + let Operator::NonASAP(NonASAPOp::BinaryOp { operator, .. }) = + &mut Rc::make_mut(&mut node).operator + else { return Ok(None); }; operator.checked_finite_division = true; - validate_execution_data_states_at(&node, ExecutionDataState::QUERY_ROWS)?; + validate_default(&node, ExecutionTiming::QueryTime)?; Ok(Some(node)) } pub(crate) fn realize_child_with( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, end_to_end_target: Option<&AccuracyTarget>, -) -> Result, RealizationError> { +) -> Result, RealizationError> { if let Some(node) = realize_temporal_average(root, planning_inputs, end_to_end_target)? { return Ok(node); } @@ -1912,29 +2014,29 @@ pub(crate) fn realize_child_with( Some(_) => Some(override_accuracy(declared, target)), } }); - match SketchAlgorithmStrategy::from_planning_inputs(planning_inputs) + match ASAPStrategies::from_planning_inputs(planning_inputs) .propose_with(root, overridden.as_ref(), None) .candidates .into_iter() .next() { Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDag(node), .. }) => Ok(node), Some(ReplacementSubDAG { - replacement: Replacement::Rewrite(_) | Replacement::ExactComposition(_), + replacement: Replacement::ExactComposition(_), .. }) => { - unreachable!("SketchAlgorithmStrategy never returns a Rewrite/composition candidate") + unreachable!("ASAPStrategies never returns a composition candidate") } // No candidate at all: `root` isn't `bindable_intent` shape (or its // intent has no realization `realizations_for_intent` can't // produce — never happens, that match is exhaustive), or every // candidate was accuracy-illegal — either way the same conservative - // fallback `SketchAlgorithmStrategy::matches` uses: keep the - // pre-ASAP subtree, executed exactly. - None => keep_pre_asap(root), + // fallback `ASAPStrategies::matches` uses: keep the + // pre-ASAP sub-DAG, executed exactly. + None => retain_exact(root), } } @@ -1943,29 +2045,24 @@ pub(crate) fn realize_child_with( /// accelerated, return `None` so the caller keeps the whole query exact; /// mixed raw/summary snapshots are never constructed. fn realize_binary( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, end_to_end_target: Option<&AccuracyTarget>, -) -> Result>, RealizationError> { - let QueryExpr::BinaryOp { - op, +) -> Result>, RealizationError> { + let Some(NonASAPOp::BinaryOp { + operator, + return_bool, lhs, rhs, - vector_match, - } = root.as_ref() + }) = root.non_asap() else { return Ok(None); }; + let (op, vector_match) = (&operator.kind, &operator.vector_match); if !matches!(op, BinaryOpKind::Arithmetic(_)) || vector_match.is_some() { return Ok(None); } - let lhs_scalar = is_promql_scalar(lhs); - let rhs_scalar = is_promql_scalar(rhs); - if lhs_scalar && rhs_scalar { - return Ok(None); - } - let mut lhs_node = realize_binary_operand(lhs, planning_inputs, None)?; let mut rhs_node = realize_binary_operand(rhs, planning_inputs, None)?; @@ -2084,8 +2181,8 @@ fn realize_binary( return Ok(None); } - let lhs_accelerated = lhs_scalar || !matches!(lhs_node.expr, SummaryExpr::KeepPreAsap(_)); - let rhs_accelerated = rhs_scalar || !matches!(rhs_node.expr, SummaryExpr::KeepPreAsap(_)); + let lhs_accelerated = lhs_node.contains_asap(); + let rhs_accelerated = rhs_node.contains_asap(); if !lhs_accelerated || !rhs_accelerated { return Ok(None); } @@ -2132,47 +2229,107 @@ fn realize_binary( return Ok(None); } - Ok(Some(Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - timing: ExecutionTiming::QueryTime, - lhs: lhs_node, - rhs: rhs_node, - operator: asap_types::post_asap::BinaryOperator { - checked_relative_division: false, - checked_finite_division: false, - kind: op.clone(), - vector_match: vector_match.clone(), - }, - }, - schema: lift(&root.output_schema()?), + Ok(Some(Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: op.clone(), + vector_match: vector_match.clone(), + }, + return_bool: *return_bool, + lhs: lhs_node, + rhs: rhs_node, + }), + lift(&root.schema), + ) // Exact arithmetic does not erase approximation error. Until the // accuracy algebra has an operator-specific rule (and any value-range // evidence needed by multiplication/division), unknown stays unknown. - guarantee, - }))) + .with_guarantee(guarantee), + ))) +} + +/// Rebuild the summary chain above a per-series `Rate` accumulator with its +/// `FinalizeExactAccumulator` placed at `timing`. `strict` additionally +/// requires the fixed-window shape (a `PerEntity` Rate over a `TimeRange`); +/// `None` when no such boundary exists (strict only). +fn retime_rate_finalize( + node: &Rc, + timing: ExecutionTiming, + strict: bool, +) -> Option> { + let is_rate_boundary = |child: &OperatorNode| match &child.operator { + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), + reduction, + child: source, + .. + }) => { + !strict + || (matches!(reduction, Reduction::PerEntity) + && matches!(source.non_asap(), Some(NonASAPOp::TimeRange { .. }))) + } + _ => false, + }; + let rebuilt = |operator: Operator, timing: Option| { + Rc::new(OperatorNode { + operator, + timing, + ..node.as_ref().clone() + }) + }; + match &node.operator { + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) if is_rate_boundary(child) => { + Some(rebuilt(node.operator.clone(), Some(timing))) + } + Operator::ASAP( + ASAPOp::FinalizeExactAccumulator { child } + | ASAPOp::SummaryAgg { child, .. } + | ASAPOp::SummaryEstimate { + summary_input: child, + .. + }, + ) => { + let placed = match retime_rate_finalize(child, timing, strict) { + Some(placed) => placed, + None if strict => return None, + None => return Some(Rc::clone(node)), + }; + let operator = node.operator.map_children(|_| Rc::clone(&placed)); + Some(rebuilt(operator, node.timing)) + } + _ if strict => None, + _ => Some(Rc::clone(node)), + } } /// Put an explicit read boundary between maintained exact state and a /// query-time value consumer. Approximate summaries must already carry a /// `SummaryEstimate`, so they deliberately do not pass this predicate. pub fn finalize_query_candidate( - node: Rc, - logical_output: &QueryExpr, -) -> Result, RealizationError> { - finalize_exact_accumulator_at(node, logical_output, ExecutionTiming::QueryTime) -} - -fn finalize_exact_accumulator_at( - node: Rc, - logical_output: &QueryExpr, - timing: ExecutionTiming, -) -> Result, RealizationError> { + node: Rc, + logical_output: &OperatorNode, +) -> Result, RealizationError> { + finalize_exact_accumulator(node, logical_output, ExecutionTiming::QueryTime) +} + +/// The read boundary's placement is fixed here, where the candidate's +/// semantics decide it (a fresh query-time summary over this evaluation's +/// finalized values vs. finalized values feeding maintenance); the lifecycle +/// timing pass honors it. +fn finalize_exact_accumulator( + node: Rc, + logical_output: &OperatorNode, + placement: ExecutionTiming, +) -> Result, RealizationError> { let is_exact_state = matches!( - node.expr, - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(..), + node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(..), .. - } + }) ); if !is_exact_state { return Ok(node); @@ -2181,41 +2338,36 @@ fn finalize_exact_accumulator_at( // boundary produces the logical operator's ordinary values. Preserve the // canonical pre-ASAP output types instead of leaking ExactAggregate into // query-time operators that follow this node. - let schema = lift(&logical_output.output_schema()?); + let schema = lift(&logical_output.schema); let guarantee = node.guarantee.clone(); - Ok(Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: node, - operation: ValueOperation::FinalizeExactAccumulator, - timing, - }, - schema, - guarantee, - })) + Ok(Rc::new( + OperatorNode::with_schema( + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: node }), + schema, + ) + .with_guarantee(guarantee) + .with_timing(Some(placement)), + )) } -fn is_supported_exact_binary(root: &QueryExpr) -> bool { +fn is_supported_exact_binary(root: &OperatorNode) -> bool { matches!( - root, - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(_), - vector_match: None, + root.non_asap(), + Some(NonASAPOp::BinaryOp { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(_), + vector_match: None, + .. + }, .. - } - ) -} - -fn is_promql_scalar(expr: &QueryExpr) -> bool { - matches!( - expr, - QueryExpr::PromqlScalarBridge(_) | QueryExpr::Literal(_) + }) ) } /// Quantile operands inherit one workload target. A temporal mean is exact /// on its checked finite domain and needs no approximation budget. -fn shared_quantile_target(lhs: &QueryExpr, rhs: &QueryExpr) -> Option { - let quantile_target = |expr: &QueryExpr| match bindable_intent(expr) { +fn shared_quantile_target(lhs: &OperatorNode, rhs: &OperatorNode) -> Option { + let quantile_target = |expr: &OperatorNode| match bindable_intent(expr) { Some(AggIntent::Quantile { accuracy, q, .. }) if q.is_finite() && (0.0..=1.0).contains(q) => { @@ -2253,18 +2405,18 @@ fn ddsketch_ratio_operand_target(target: &AccuracyTarget) -> Option Option { - let SummaryExpr::SummaryEstimate { +fn ddsketch_quantile_alpha(node: &OperatorNode) -> Option { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, - query: PostAsapSketchQuery::Quantile { .. }, - } = &node.expr + query: PostAsapSketchStatistic::Quantile { .. }, + }) = &node.operator else { return None; }; - let SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + let Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), .. - } = &summary_input.expr + }) = &summary_input.operator else { return None; }; @@ -2274,7 +2426,7 @@ fn ddsketch_quantile_alpha(node: &SummaryNode) -> Option { } } -fn has_missing_accuracy_evidence(node: &SummaryNode) -> bool { +fn has_missing_accuracy_evidence(node: &OperatorNode) -> bool { node.guarantee .as_ref() .is_none_or(ResultGuarantee::has_unknown) @@ -2283,10 +2435,10 @@ fn has_missing_accuracy_evidence(node: &SummaryNode) -> bool { /// A direct ratio has an operator-specific DDSketch proof, so it must select /// DDSketch rather than the cost model's generally preferred KLL candidate. fn realize_ddsketch_quantile_operand( - operand: &Rc, + operand: &Rc, planning_inputs: CandidatePlanningInputs<'_>, target: &AccuracyTarget, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let intent = bindable_intent(operand).and_then(|intent| match intent { AggIntent::Quantile { .. } => Some(override_accuracy(intent, target)), _ => None, @@ -2305,20 +2457,10 @@ fn realize_ddsketch_quantile_operand( } fn realize_binary_operand( - operand: &Rc, + operand: &Rc, planning_inputs: CandidatePlanningInputs<'_>, end_to_end_target: Option<&AccuracyTarget>, -) -> Result, RealizationError> { - if is_promql_scalar(operand) { - return Ok(Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::clone(operand)), - schema: SummarySchema { - fields: Vec::new(), - time_index: None, - }, - guarantee: Some(ResultGuarantee::exact("PromQL scalar")), - })); - } +) -> Result, RealizationError> { realize_child_with(operand, planning_inputs, end_to_end_target) } @@ -2336,43 +2478,77 @@ fn override_accuracy(intent: &AggIntent, target: &AccuracyTarget) -> AggIntent { out } -/// Wrap an unrewritten pre-ASAP subtree, lifting its schema with every column -/// `SummaryFamilyType::Plain`. `pub` so a caller can fall back to this -/// explicitly — e.g. when `SketchAlgorithmStrategy::replacements()` returns no -/// candidate for a target, or a deployment wants to force a node its own -/// runtime can't actually implement — through the same fallback this -/// crate's own dispatch uses, without duplicating the schema-lift logic. -pub fn keep_pre_asap(expr: &Rc) -> Result, RealizationError> { - keep_pre_asap_rc(Rc::clone(expr)) -} - -fn keep_pre_asap_rc(expr: Rc) -> Result, RealizationError> { - let schema = expr.output_schema()?; - Ok(Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(expr), - schema: lift(&schema), - // A kept pre-ASAP subtree is executed exactly by the runtime - // (`Realization::PassThrough`'s contract) — zero error. - guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), - })) +/// Keep an unrewritten pre-ASAP sub-DAG as it is. There is no wrapper node: +/// the sub-DAG itself is the plan, carrying an exact guarantee. The same +/// `Rc` is returned when the node already has a guarantee; otherwise a copy +/// with `guarantee = exact("RetainedExact")` — only for a sub-DAG with no +/// ASAP operator (a sub-DAG containing one keeps whatever its construction +/// established). `pub` so a caller can fall back to this explicitly — e.g. +/// when `ASAPStrategies::replacements()` returns no candidate for a +/// target, or a deployment wants to force a node its own runtime can't +/// actually implement — through the same fallback this crate's own dispatch +/// uses. +pub fn retain_exact(expr: &Rc) -> Result, RealizationError> { + retain_exact_rc(Rc::clone(expr)) +} + +fn retain_exact_rc(expr: Rc) -> Result, RealizationError> { + if expr.guarantee.is_some() || expr.contains_asap() { + return Ok(expr); + } + // Keeping the same sub-DAG twice (e.g. one `Scan` read by an exact + // aggregate and by a sketch, or by two candidates) must yield one node: + // sharing is pointer identity. Memoize the kept copy per input node while + // both are alive; weak references keep the memo from extending lifetimes + // or matching a reused address. + type KeptMemo = HashMap<*const OperatorNode, (Weak, Weak)>; + thread_local! { + static KEPT: RefCell = RefCell::new(HashMap::new()); + } + let key = Rc::as_ptr(&expr); + if let Some(kept) = KEPT.with(|memo| { + memo.borrow().get(&key).and_then(|(input, kept)| { + input + .upgrade() + .filter(|input| Rc::ptr_eq(input, &expr)) + .and_then(|_| kept.upgrade()) + }) + }) { + return Ok(kept); + } + let kept = Rc::new( + expr.as_ref() + .clone() + // A kept pre-ASAP sub-DAG is executed exactly by the runtime + // (`Realization::PassThrough`'s contract) — zero error. + .with_guarantee(Some(ResultGuarantee::exact("RetainedExact"))), + ); + KEPT.with(|memo| { + let mut memo = memo.borrow_mut(); + if memo.len() > 4096 { + memo.retain(|_, (input, kept)| input.strong_count() > 0 && kept.strong_count() > 0); + } + memo.insert(key, (Rc::downgrade(&expr), Rc::downgrade(&kept))); + }); + Ok(kept) } -// ── Construction: turn one already-decided Realization into a SummaryNode ─ +// ── Construction: turn one already-decided Realization into an OperatorNode ─ -/// The bindable shape [`SketchAlgorithmStrategy`] targets: a single intent, no +/// The bindable shape [`ASAPStrategies`] targets: a single intent, no /// `HAVING`. A multi-intent node (SQL `SELECT SUM(a), AVG(b)`), or one with a /// `HAVING` predicate (the filter would need the estimate first), stays -/// logical. Unsupported logical parents still conservatively become one -/// [`SummaryExpr::KeepPreAsap`] subtree. Composable query-time value -/// operators (`Project`, `Filter`, `Sort`, and `Limit`) are retained during final -/// DAG assembly so their independently planned children remain visible. -pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { - if let QueryExpr::Aggregate { +/// logical. Unsupported logical parents are conservatively kept as pre-ASAP +/// sub-DAGs ([`retain_exact`]). Relational operators are retained during +/// final DAG assembly so their independently planned children remain +/// visible. +pub fn bindable_intent(node: &OperatorNode) -> Option<&AggIntent> { + if let Some(NonASAPOp::Aggregate { measures, filters, having, .. - } = node + }) = node.non_asap() { if let ([intent], None) = (measures.as_slice(), having) { if !any_measure_filtered(filters) { @@ -2384,7 +2560,7 @@ pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { } /// `expr` must still be the [`bindable_intent`] shape for `realization` to -/// have any effect; anything else falls back to [`keep_pre_asap`]. +/// have any effect; anything else falls back to [`retain_exact`]. /// Only `expr`'s own top-level decision is forced — recursion into `expr`'s /// child goes back through [`realize_child`] (fresh candidate /// enumeration, not a forced pick), so choosing one candidate for a target @@ -2392,27 +2568,27 @@ pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { /// /// `pub(crate)`: `grouping::HydraGroupingStrategy` (issue #256) is the one /// caller outside this module — the same first-class, -/// one-candidate-at-a-time primitive [`SketchAlgorithmStrategy`] itself +/// one-candidate-at-a-time primitive [`ASAPStrategies`] itself /// calls once per candidate, reused rather than duplicated so a Hydra /// candidate gets exactly the same schema derivation/column -/// resolution/readout construction as every other candidate, patching only +/// resolution/evaluation construction as every other candidate, patching only /// the `grouping` field this axis owns. /// Construct a summary with every model explicit (issue #172). `intent` /// is `expr`'s own [`bindable_intent`], or a copy of it with an allocated /// `AccuracyTarget` substituted (see [`realize_child_with`]). -/// `child_target`, when set, is the end-to-end budget the child subtree is +/// `child_target`, when set, is the end-to-end budget the child sub-DAG is /// re-enumerated under; `allocation` is the provenance note recording the /// split that produced both. `Err(RealizationError::Accuracy)` is the /// fail-closed answer for a composition with no sound rule or one that /// misses `intent`'s target. pub(crate) fn construct_summary_with( - expr: &QueryExpr, + expr: &OperatorNode, intent: &AggIntent, realization: Realization, planning_inputs: CandidatePlanningInputs<'_>, child_target: Option<&AccuracyTarget>, allocation: Option, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let local_target = match allocation.as_ref() { Some(GuaranteeSource::BudgetAllocation { local_target, .. }) => Some(local_target), _ => accuracy_target(intent), @@ -2429,9 +2605,9 @@ pub(crate) fn construct_summary_with( }, other => other, }; - if let QueryExpr::Aggregate { + if let Some(NonASAPOp::Aggregate { reduction, child, .. - } = expr + }) = expr.non_asap() { // `bindable_intent` already established the shape: exactly one // intent, no HAVING. (Multi-intent nodes and HAVING stay logical.) @@ -2456,28 +2632,28 @@ pub(crate) fn construct_summary_with( } } } - keep_pre_asap_rc(Rc::new(expr.clone())) + retain_exact_rc(Rc::new(expr.clone())) } fn finish_weighted_topk( - candidate: Rc, - logical: &QueryExpr, + candidate: Rc, + logical: &OperatorNode, intent: &AggIntent, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let AggIntent::TopK { k, .. } = intent else { unreachable!() }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(groups), child, .. - } = logical + }) = logical.non_asap() else { return Err(RealizationError::PhysicalRealization( "TopK requires explicit grouping", )); }; - let schema = lift(&child.output_schema()?); + let schema = lift(&child.schema); let score = ranking_score_index(child, &schema)?; let cols = schema .fields @@ -2504,100 +2680,95 @@ fn finish_weighted_topk( } } }; - Ok(asap_types::pre_asap::query_expr::ProjectItem { + Ok(ProjectItem { alias: Some(field.name.clone()), - expr: QueryExpr::Column(source), + expr: ScalarExpr::Column(source), }) }) .collect::, _>>()?; let guarantee = candidate.guarantee.clone(); - let projected = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: candidate, - operation: ValueOperation::Project { + let projected = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Project { cols, qualifier: None, - }, - timing: ExecutionTiming::QueryTime, - }, - schema: schema.clone(), - guarantee: guarantee.clone(), - }); - let sorted = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: projected, - operation: ValueOperation::Sort { - keys: vec![asap_types::pre_asap::SortKey { - expr: QueryExpr::Column(score), + child: candidate, + }), + schema.clone(), + ) + .with_guarantee(guarantee.clone()), + ); + let sorted = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Sort { + keys: vec![SortKey { + expr: ScalarExpr::Column(score), ascending: false, nulls_first: false, }], partition_by: groups.clone(), - }, - timing: ExecutionTiming::QueryTime, - }, - schema: schema.clone(), - guarantee: guarantee.clone(), - }); - let result = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: sorted, - operation: ValueOperation::Limit { - n: *k, + child: projected, + }), + schema.clone(), + ) + .with_guarantee(guarantee.clone()), + ); + let result = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Limit { + n: Some(*k), offset: 0, partition_by: groups.clone(), - }, - timing: ExecutionTiming::QueryTime, - }, - schema, - guarantee, - }); - validate_execution_data_states_at(&result, ExecutionDataState::QUERY_ROWS)?; + child: sorted, + }), + schema, + ) + .with_guarantee(guarantee), + ); + validate_default(&result, ExecutionTiming::QueryTime)?; Ok(result) } -fn is_current_series_source(child: &QueryExpr) -> bool { - let source = match child { - QueryExpr::TimeRange { child, .. } => child.as_ref(), - source => source, +fn is_current_series_source(child: &OperatorNode) -> bool { + let source = match child.non_asap() { + Some(NonASAPOp::TimeRange { child, .. }) => child.as_ref(), + _ => child, }; - matches!(source, QueryExpr::Scan { + matches!(source.non_asap(), Some(NonASAPOp::Scan { source: asap_types::pre_asap::Source::TimeSeries { .. }, schema, .. - } if schema.has_promql_series_identity()) + }) if schema.has_promql_series_identity()) } -fn is_snapshot_weighted_topk(intent: &AggIntent, child: &QueryExpr) -> bool { +fn is_snapshot_weighted_topk(intent: &AggIntent, child: &OperatorNode) -> bool { matches!(intent, AggIntent::TopK { .. }) && (is_current_series_source(child) - || matches!(child, - QueryExpr::Aggregate { measures, child, .. } + || matches!(child.non_asap(), + Some(NonASAPOp::Aggregate { measures, child, .. }) if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase]) || (matches!(measures.as_slice(), [AggIntent::Sum { .. }]) - && matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + && matches!(child.non_asap(), Some(NonASAPOp::Aggregate { measures, .. }) if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase]))))) } /// Translate an [`Realization`] into the `(family, needs a -/// SummaryEstimate readout)` pair [`construct_summary_agg`] needs, or `None` -/// for `PassThrough` (the caller falls back to [`keep_pre_asap`]). +/// SummaryEstimate evaluation)` pair [`construct_summary_agg`] needs, or `None` +/// for `PassThrough` (the caller falls back to [`retain_exact`]). /// -/// Every family's partial state needs a readout to recover a value, except +/// Every family's partial state needs a evaluation to recover a value, except /// `ExactAggregate` — its partial state *is* the value already, so no /// estimate step follows it. -fn summary_family(realization: Realization) -> Option<(SummaryFamilyType, bool)> { +fn summary_family(realization: Realization) -> Option<(FieldDataType, bool)> { Some(match realization { Realization::ExactAggregate { kind, params } => { - (SummaryFamilyType::ExactAggregate(kind, params), false) + (FieldDataType::ExactAggregate(kind, params), false) } Realization::Sketch(kind) => ( - SummaryFamilyType::Sketch(kind, GroupingStrategy::default()), + FieldDataType::Sketch(kind, GroupingStrategy::default()), true, ), - Realization::Sample { kind, params } => (SummaryFamilyType::Sample(kind, params), true), - Realization::Wavelet { kind, params } => (SummaryFamilyType::Wavelet(kind, params), true), - Realization::StatModel { kind, params } => { - (SummaryFamilyType::StatModel(kind, params), true) - } + Realization::Sample { kind, params } => (FieldDataType::Sample(kind, params), true), + Realization::Wavelet { kind, params } => (FieldDataType::Wavelet(kind, params), true), + Realization::StatModel { kind, params } => (FieldDataType::StatModel(kind, params), true), Realization::PassThrough => return None, }) } @@ -2607,7 +2778,7 @@ fn summary_family(realization: Realization) -> Option<(SummaryFamilyType, bool)> /// input value. Composite realizations can instead consume a larger /// logical sub-DAG and bind a different key or value. struct PhysicalSummaryInput { - child: Rc, + child: Rc, input: SummaryUpdate, } @@ -2617,12 +2788,8 @@ enum PhysicalSummaryInputRuleResult { Unsupported(&'static str), } -type PhysicalSummaryInputRule = fn( - &AggIntent, - &SummaryFamilyType, - &Reduction, - &Rc, -) -> PhysicalSummaryInputRuleResult; +type PhysicalSummaryInputRule = + fn(&AggIntent, &FieldDataType, &Reduction, &Rc) -> PhysicalSummaryInputRuleResult; /// Ordered physical-realization rules for realizations that consume more /// than the immediate logical input. New composite primitives add a rule here @@ -2637,24 +2804,20 @@ const PHYSICAL_SUMMARY_INPUT_RULES: &[PhysicalSummaryInputRule] = &[ fn realize_value_frequency_summary_input( intent: &AggIntent, - family: &SummaryFamilyType, + family: &FieldDataType, _reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { // Frequency counts hash sample values as items but add one per observation. // Using the sample as a weight would turn counts into sums and admit signed CMS updates. - if !matches!(family, SummaryFamilyType::Sketch(kind, _) + if !matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon || (matches!(intent, AggIntent::Count { .. }) && matches!(kind.algorithm(), SketchAlgorithm::Cms | SketchAlgorithm::CountSketch))) { return PhysicalSummaryInputRuleResult::NotApplicable; } - let Ok(schema) = child.output_schema() else { - return PhysicalSummaryInputRuleResult::Unsupported( - "value frequency input needs a valid schema", - ); - }; + let schema = &child.schema; // One item per observation is a single value stream. `summary_candidates` // already withholds UnivMon from a distinct-tuple count; refused here too // so the invariant does not rest on that table alone. @@ -2666,7 +2829,7 @@ fn realize_value_frequency_summary_input( PhysicalSummaryInputRuleResult::Realized(PhysicalSummaryInput { child: Rc::clone(child), input: SummaryUpdate { - item: Some(SummaryInputExpr::Column(summarised_column(intent, &schema))), + item: Some(SummaryInputExpr::Column(summarised_column(intent, schema))), weight: SummaryInputExpr::Constant(1.0), weight_domain: WeightDomain::NonNegative { proof: NonNegativeWeightProof::UnitCount, @@ -2677,9 +2840,9 @@ fn realize_value_frequency_summary_input( fn realize_physical_summary_input( intent: &AggIntent, - family: &SummaryFamilyType, + family: &FieldDataType, reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> Result { for rule in PHYSICAL_SUMMARY_INPUT_RULES { match rule(intent, family, reduction, child) { @@ -2691,7 +2854,7 @@ fn realize_physical_summary_input( } } - let child_schema = child.output_schema()?; + let child_schema = &child.schema; if matches!(intent, AggIntent::TopK { .. }) { return Err(RealizationError::PhysicalRealization( "Top-K needs an explicit item identity and additive update input", @@ -2701,95 +2864,89 @@ fn realize_physical_summary_input( child: Rc::clone(child), input: SummaryUpdate { item: None, - weight: summarised_input(intent, &child_schema)?, + weight: summarised_input(intent, child_schema)?, weight_domain: WeightDomain::UnknownOrSigned, }, }) } /// Emit `SummaryAgg` (recursively binding the child), plus the -/// `SummaryEstimate` readout when `estimate` is set. +/// `SummaryEstimate` evaluation when `estimate` is set. // Retain the exact expression and schema while placing its value production -// on the update path. This is the initial layout for values feeding a summary; -// lifecycle timing is authoritative. Read-time consumers keep their original -// shared nodes. -fn maintenance_exact_values(node: Rc) -> Option> { - let expr = match &node.expr { +// on the update path (a node runs when its consumer runs, so beneath a +// maintained summary this value production is ingestion-time work). +// Read-time consumers keep their original shared nodes. +fn maintenance_exact_values(node: Rc) -> Option> { + let operator = match &node.operator { // These guards can fall back at read time, but cannot recover a parent // sketch after an invalid value has entered its maintained state. - SummaryExpr::BinaryOp { operator, .. } + Operator::NonASAP(NonASAPOp::BinaryOp { operator, .. }) if operator.checked_finite_division || operator.checked_relative_division => { return None; } - SummaryExpr::BinaryOp { - lhs, rhs, operator, .. - } if operator.vector_match.is_none() - && matches!( - operator.kind, - asap_types::pre_asap::BinaryOpKind::Arithmetic(_) - ) + Operator::NonASAP(NonASAPOp::BinaryOp { + lhs, + rhs, + operator, + return_bool, + }) if operator.vector_match.is_none() + && matches!(operator.kind, BinaryOpKind::Arithmetic(_)) && node .guarantee .as_ref() .is_some_and(ResultGuarantee::is_exact) => { - SummaryExpr::BinaryOp { + Operator::NonASAP(NonASAPOp::BinaryOp { lhs: maintenance_exact_values(lhs.clone())?, rhs: maintenance_exact_values(rhs.clone())?, operator: operator.clone(), - timing: ExecutionTiming::IngestionTime, - } + return_bool: *return_bool, + }) } - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - .. - } if matches!( - child.expr, - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(..), - .. - } - ) => + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) + if matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(..), + .. + }) + ) => { - SummaryExpr::ValueOperation { + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: child.clone(), - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::IngestionTime, - } + }) } _ => return Some(node), }; - Some(Rc::new(SummaryNode { - expr, - schema: node.schema.clone(), - guarantee: node.guarantee.clone(), - })) + Some(Rc::new( + OperatorNode::with_schema(operator, node.schema.clone()) + .with_guarantee(node.guarantee.clone()), + )) } #[allow(clippy::too_many_arguments)] fn construct_summary_agg( - node: &QueryExpr, + node: &OperatorNode, reduction: &Reduction, intent: &AggIntent, input: PhysicalSummaryInput, - family: SummaryFamilyType, + family: FieldDataType, estimate: bool, planning_inputs: CandidatePlanningInputs<'_>, child_target: Option<&AccuracyTarget>, allocation: Option, -) -> Result, RealizationError> { +) -> Result, RealizationError> { // The single canonical pre-ASAP derivation (per-series vs cross-series, // name overrides) already computes the row shape; binding only retypes // the summary state column. let keyed_heap = input.input.item.is_some() && matches!( &family, - SummaryFamilyType::Sketch(kind, _) + FieldDataType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap) ); - let snapshot_weighted = matches!(node, QueryExpr::Aggregate { child, .. } + let snapshot_weighted = matches!(node.non_asap(), Some(NonASAPOp::Aggregate { child, .. }) if is_snapshot_weighted_topk(intent, child)); let mut family = family; let score_population = if snapshot_weighted { @@ -2799,7 +2956,7 @@ fn construct_summary_agg( "invalid weighted TopK distinct-item bound", )); } - if let (Some(n), SummaryFamilyType::Sketch(kind, grouping)) = (bound, &family) { + if let (Some(n), FieldDataType::Sketch(kind, grouping)) = (bound, &family) { let (eps, delta) = accuracy_budget(accuracy_target(intent).expect("TopK target")); let params = default_size_params( kind.algorithm().clone(), @@ -2807,7 +2964,7 @@ fn construct_summary_agg( eps, delta / (2.0 * n as f64), ); - family = SummaryFamilyType::Sketch( + family = FieldDataType::Sketch( SketchKind::new(kind.algorithm().clone(), params), grouping.clone(), ); @@ -2817,10 +2974,10 @@ fn construct_summary_agg( None }; let physical_reduction = if snapshot_weighted { - let QueryExpr::Aggregate { child, .. } = node else { + let Some(NonASAPOp::Aggregate { child, .. }) = node.non_asap() else { unreachable!() }; - let source = input.child.output_schema()?; + let source = &input.child.schema; let Reduction::Reduce(keys) = reduction else { return Err(RealizationError::PhysicalRealization( "TopK requires explicit partitions", @@ -2838,7 +2995,7 @@ fn construct_summary_agg( RealizationError::PhysicalRealization("invalid TopK partition key"), )?; let matches = source - .columns + .fields .iter() .enumerate() .filter(|(_, column)| column_ref(column) == reference) @@ -2858,66 +3015,64 @@ fn construct_summary_agg( } else { reduction.clone() }; - let out_schema = node.output_schema()?; - let measures = match node { - QueryExpr::Aggregate { measures, .. } => measures.len(), + let out_schema = &node.schema; + let measures = match node.non_asap() { + Some(NonASAPOp::Aggregate { measures, .. }) => measures.len(), _ => 1, }; - let state_idx = summary_col_index(&out_schema, reduction, measures); + let state_idx = summary_col_index(out_schema, reduction, measures); - let readout_schema = if keyed_heap - && matches!(node, QueryExpr::Aggregate { child, .. } if is_snapshot_weighted_topk(intent, child)) + let evaluation_schema = if keyed_heap + && matches!(node.non_asap(), Some(NonASAPOp::Aggregate { child, .. }) if is_snapshot_weighted_topk(intent, child)) { - keyed_heap_readout_schema(&input, node)? + keyed_heap_evaluation_schema(&input, node)? } else { - lift(&out_schema) + lift(out_schema) }; let summary_input = input.input; let query = estimate.then(|| { if snapshot_weighted { - if let SummaryFamilyType::Sketch(kind, _) = &family { + if let FieldDataType::Sketch(kind, _) = &family { let capacity = match kind.params() { SketchParams::CmsWithHeap { heap_size, .. } | SketchParams::CountSketchWithHeap { heap_size, .. } => *heap_size, _ => unreachable!(), }; - return PostAsapSketchQuery::TopK { + return PostAsapSketchStatistic::TopK { k: capacity as usize, }; } } - readout(intent, &summary_input, planning_inputs.cost) + evaluation(intent, &summary_input, planning_inputs.cost) }); - let mut state_schema = lift(&out_schema); + let mut state_schema = lift(out_schema); if keyed_heap { let mut state = state_schema.fields[state_idx].clone(); state.dtype = family.clone(); let mut fields = if snapshot_weighted { - readout_schema.fields[..reduction.group_keys().map_or(0, |keys| keys.len())].to_vec() + evaluation_schema.fields[..reduction.group_keys().map_or(0, |keys| keys.len())].to_vec() } else { Vec::new() }; fields.push(state); - state_schema = SummarySchema { - fields, - time_index: None, - }; + state_schema = Schema::lifted(fields, None); } else if let Some(field) = state_schema.fields.get_mut(state_idx) { field.dtype = family.clone(); - if matches!(&family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) + field.nullable = false; + if matches!(&family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) { // State identity is independent of which statistic reads it. field.name = "univmon".into(); } else if let (AggIntent::Quantile { .. }, SummaryInputExpr::Column(col)) = (intent, &summary_input.weight) { - // The quantile is a readout parameter: name the state after the + // The quantile is a evaluation parameter: name the state after the // column it summarizes, not after the query's output column. - let child_schema = input.child.output_schema()?; + let child_schema = input.child.schema.clone(); if let Ok(i) = resolve_column_ref(col, &child_schema) { - field.name = child_schema.columns[i].name.clone(); + field.name = child_schema.fields[i].name.clone(); } } } @@ -2942,9 +3097,9 @@ fn construct_summary_agg( .ok_or(RealizationError::PhysicalRealization( "snapshot ranking requires a supported current-series population", ))?; - let SummaryExpr::ValueOperation { child, .. } = &population.expr else { + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &population.operator else { return Err(RealizationError::PhysicalRealization( - "missing population readout", + "missing population evaluation", )); }; Rc::clone(child) @@ -2954,27 +3109,14 @@ fn construct_summary_agg( // initial layout; a retained summary's lifecycle moves it to ingestion. finalize_query_candidate(bound_child, &input.child)? } else { - let child = finalize_exact_accumulator_at( - bound_child, - &input.child, - ExecutionTiming::IngestionTime, - )?; - let child = maintenance_exact_values(child).unwrap_or(keep_pre_asap(&input.child)?); - // Maintenance arithmetic must satisfy the ingestion contract; e.g. a - // per-series sum over different selectors has no exact aligned - // layout, so this candidate fails closed and exact execution remains. - // Unlike checked division, it does not fall back to `keep_pre_asap`: - // that retains the range expression at ingestion time, where range - // functions cannot run (they need a query evaluation time). - if matches!(child.expr, SummaryExpr::BinaryOp { .. }) { - validate_execution_data_states_at(&child, ExecutionDataState::INGESTION_ROWS)?; - } - child + let child = + finalize_exact_accumulator(bound_child, &input.child, ExecutionTiming::IngestionTime)?; + maintenance_exact_values(child).unwrap_or(retain_exact(&input.child)?) }; // ── Guarantee (issue #172) ────────────────────────────────────────── // Derived *before* the node exists, so an illegal composition is never - // materialized: the local guarantee of this family's readout (or exact + // materialized: the local guarantee of this family's evaluation (or exact // accumulator) composed over the child's, under the operator this // family applies to the child's values. let local_target = match allocation.as_ref() { @@ -2987,7 +3129,7 @@ fn construct_summary_agg( local_target, ); let membership_query = if snapshot_weighted { - Some(readout(intent, &summary_input, planning_inputs.cost)) + Some(evaluation(intent, &summary_input, planning_inputs.cost)) } else { query.clone() }; @@ -3065,8 +3207,8 @@ fn construct_summary_agg( // a genuine empty-`by` reduction apart from a per-entity shape with no // grouping concept at all (issue #163). `construct_summary_agg` is the // single place that decides this; nothing downstream re-derives it. - let agg = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { + let agg = OperatorNode::asap_node( + ASAPOp::SummaryAgg { child: bound_child, family, input: summary_input, @@ -3074,47 +3216,47 @@ fn construct_summary_agg( grouping: GroupingStrategy::default(), filter: None, }, - schema: state_schema, + state_schema, // Summary *state* carries no caller-visible guarantee; only a // finalized value does. An exact accumulator's state is its value. - guarantee: if estimate { None } else { guarantee.clone() }, - }); + if estimate { None } else { guarantee.clone() }, + ); match query { - // The readout: downstream of the estimate the schema is the plain + // The evaluation: downstream of the estimate the schema is the plain // pre-ASAP row shape again (the summary-state type does not // propagate). - Some(query) => Ok(Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { + Some(query) => Ok(OperatorNode::asap_node( + ASAPOp::SummaryEstimate { summary_input: agg, query, }, - schema: readout_schema, + evaluation_schema, guarantee, - })), + )), None => Ok(agg), } } -// Heap readout rows contain the encoded item identity, subpopulation keys, +// Heap evaluation rows contain the encoded item identity, subpopulation keys, // and an estimated score. They never inherit the exact-value producer's schema. -fn keyed_heap_readout_schema( +fn keyed_heap_evaluation_schema( input: &PhysicalSummaryInput, - node: &QueryExpr, -) -> Result { - let source = input.child.output_schema()?; + node: &OperatorNode, +) -> Result { + let source = &input.child.schema; let mut refs = Vec::new(); - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, child, .. - } = node + }) = node.non_asap() else { return Err(RealizationError::PhysicalRealization( - "heap readout requires an aggregate", + "heap evaluation requires an aggregate", )); }; if let Reduction::Reduce(groups) = reduction { if groups.is_without() { return Err(RealizationError::PhysicalRealization( - "heap readout requires explicit grouping", + "heap evaluation requires explicit grouping", )); } for index in groups.iter() { @@ -3141,7 +3283,7 @@ fn keyed_heap_readout_schema( "dynamic label identity requires an explicit row representation", )); } - for (index, column) in schema.columns.iter().enumerate() { + for (index, column) in schema.fields.iter().enumerate() { if Some(index) != schema.time_index && column.name != "value" { let reference = match &column.table { Some(table) => ColumnRef::Qualified { @@ -3172,13 +3314,13 @@ fn keyed_heap_readout_schema( .ok_or(RealizationError::PhysicalRealization( "heap item identity is missing", ))?, - &source, + source, &mut refs, )?; - let mut fields = Vec::::new(); + let mut fields = Vec::::new(); for reference in refs { let matches: Vec<_> = source - .columns + .fields .iter() .filter(|column| match &reference { ColumnRef::Named(name) => &column.name == name, @@ -3200,50 +3342,43 @@ fn keyed_heap_readout_schema( "heap keys must have distinct output names", )); } - fields.push(asap_types::post_asap::SummaryField { - name: column.name.clone(), - dtype: SummaryFamilyType::Plain(column.dtype.clone()), - nullable: column.nullable, - }); + fields.push(Field::new( + column.name.clone(), + column.dtype.clone(), + column.nullable, + )); } if fields.is_empty() { return Err(RealizationError::PhysicalRealization( - "heap readout has no identity columns", + "heap evaluation has no identity columns", )); } - fields.push(asap_types::post_asap::SummaryField { - name: "__asap_estimate".into(), - dtype: SummaryFamilyType::Plain(asap_types::pre_asap::DataType::Float64), - nullable: false, - }); - Ok(SummarySchema { - fields, - time_index: None, - }) + fields.push(Field::new( + "__asap_estimate", + FieldDataType::Plain(asap_types::pre_asap::DataType::Float64), + false, + )); + Ok(Schema::lifted(fields, None)) } -fn ranking_score_index( - logical: &QueryExpr, - values: &SummarySchema, -) -> Result { +fn ranking_score_index(logical: &OperatorNode, values: &Schema) -> Result { if is_current_series_source(logical) { return values .fields .iter() .position(|field| { field.name == "value" - && field.dtype - == SummaryFamilyType::Plain(asap_types::pre_asap::DataType::Float64) + && field.dtype == FieldDataType::Plain(asap_types::pre_asap::DataType::Float64) }) .ok_or(RealizationError::PhysicalRealization( "snapshot ranking requires the sample value column", )); } - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, measures, .. - } = logical + }) = logical.non_asap() else { return Err(RealizationError::PhysicalRealization( "ranking requires an explicit aggregate score", @@ -3273,7 +3408,7 @@ fn ranking_score_index( || !values.fields.get(index).is_some_and(|field| { matches!( field.dtype, - SummaryFamilyType::Plain( + FieldDataType::Plain( asap_types::pre_asap::DataType::Int64 | asap_types::pre_asap::DataType::Float64 ) ) @@ -3290,21 +3425,17 @@ fn ranking_score_index( /// The rate window is preserved; raw counter samples never become CMS weights. fn realize_counter_value_summary_input( intent: &AggIntent, - family: &SummaryFamilyType, + family: &FieldDataType, output_reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) - || !matches!(family, SummaryFamilyType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap)) - || !matches!(child.as_ref(), QueryExpr::Aggregate { reduction: Reduction::PerEntity, measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])) + || !matches!(family, FieldDataType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap)) + || !matches!(child.non_asap(), Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures, .. }) if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])) { return PhysicalSummaryInputRuleResult::NotApplicable; } - let Ok(schema) = child.output_schema() else { - return PhysicalSummaryInputRuleResult::Unsupported( - "counter ranking needs a valid value schema", - ); - }; + let schema = &child.schema; if !schema.closed { return PhysicalSummaryInputRuleResult::Unsupported( "counter ranking needs the complete resolved series identity", @@ -3323,7 +3454,7 @@ fn realize_counter_value_summary_input( // Retain the evaluation timestamp in each returned row. This sketch is a // snapshot, not an additive history of successive rate evaluations. let items = schema - .columns + .fields .iter() .enumerate() .filter(|(index, column)| column.name != "value" && !groups.contains(index)) @@ -3348,14 +3479,14 @@ fn realize_counter_value_summary_input( /// Rebuild the state for each evaluation; historical samples are not updates. fn realize_current_series_summary_input( intent: &AggIntent, - family: &SummaryFamilyType, + family: &FieldDataType, output_reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) || !is_current_series_source(child) { return PhysicalSummaryInputRuleResult::NotApplicable; } - let SummaryFamilyType::Sketch(kind, _) = family else { + let FieldDataType::Sketch(kind, _) = family else { return PhysicalSummaryInputRuleResult::NotApplicable; }; match kind.algorithm() { @@ -3377,13 +3508,9 @@ fn realize_current_series_summary_input( "snapshot ranking requires resolved partitions", ); } - let Ok(schema) = child.output_schema() else { - return PhysicalSummaryInputRuleResult::Unsupported( - "snapshot ranking requires a valid source schema", - ); - }; + let schema = &child.schema; let items = schema - .columns + .fields .iter() .enumerate() .filter(|(index, column)| column.name != "value" && !groups.contains(index)) @@ -3409,14 +3536,14 @@ fn realize_current_series_summary_input( /// it does not consume an independently materialized Count result. fn realize_keyed_additive_summary_input( intent: &AggIntent, - family: &SummaryFamilyType, + family: &FieldDataType, output_reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) { return PhysicalSummaryInputRuleResult::NotApplicable; } - let SummaryFamilyType::Sketch(kind, _) = family else { + let FieldDataType::Sketch(kind, _) = family else { return PhysicalSummaryInputRuleResult::NotApplicable; }; let heap_algorithm = kind.algorithm(); @@ -3426,18 +3553,18 @@ fn realize_keyed_additive_summary_input( ) { return PhysicalSummaryInputRuleResult::NotApplicable; } - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, measures, having: None, child: raw_child, .. - } = child.as_ref() + }) = child.non_asap() else { return PhysicalSummaryInputRuleResult::NotApplicable; }; let counter_input = matches!(measures.as_slice(), [AggIntent::Sum { .. }]) - && matches!(raw_child.as_ref(), QueryExpr::Aggregate { measures, .. } + && matches!(raw_child.non_asap(), Some(NonASAPOp::Aggregate { measures, .. }) if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])); let weight = match measures.as_slice() { [AggIntent::Count { .. }] => SummaryInputExpr::Constant(1.0), @@ -3529,9 +3656,8 @@ fn realize_keyed_additive_summary_input( }) } -fn schema_column_ref(child: &QueryExpr, index: usize) -> Option { - let schema = child.output_schema().ok()?; - let column = schema.columns.get(index)?; +fn schema_column_ref(child: &OperatorNode, index: usize) -> Option { + let column = child.schema.fields.get(index)?; Some(match &column.table { Some(table) => ColumnRef::Qualified { table: table.clone(), @@ -3551,16 +3677,16 @@ fn schema_column_ref(child: &QueryExpr, index: usize) -> Option { /// (an approximate family the model has no local guarantee for, over an /// exact child) — unknown, never exact. fn compose_guarantee( - family: &SummaryFamilyType, - query: Option<&PostAsapSketchQuery>, - child: &SummaryNode, + family: &FieldDataType, + query: Option<&PostAsapSketchStatistic>, + child: &OperatorNode, intent: &AggIntent, accuracy: &dyn AccuracyModel, evidence: &dyn AccuracyEvidenceProvider, allocation: Option, ) -> Result, AccuracyError> { let (op, local) = match (family, query) { - (SummaryFamilyType::ExactAggregate(kind, _), _) => { + (FieldDataType::ExactAggregate(kind, _), _) => { let op = match kind { // A row count does not depend on the rows' values: exact // regardless of the child's own error. @@ -3584,7 +3710,7 @@ fn compose_guarantee( ) } (_, Some(query)) => ( - if matches!(query, PostAsapSketchQuery::TopK { .. }) { + if matches!(query, PostAsapSketchStatistic::TopK { .. }) { CompositionOperator::TopKSelection } else { CompositionOperator::ApproximateAggregate @@ -3638,10 +3764,10 @@ fn summary_col_index(out_schema: &Schema, reduction: &Reduction, measures: usize match reduction { Reduction::PerEntity => out_schema .column_id("value") - .or_else(|| (0..out_schema.columns.len()).find(|&i| Some(i) != out_schema.time_index)) + .or_else(|| (0..out_schema.fields.len()).find(|&i| Some(i) != out_schema.time_index)) .unwrap_or(0), Reduction::Reduce(keys) if keys.is_without() => { - out_schema.columns.len().saturating_sub(measures) + out_schema.fields.len().saturating_sub(measures) } Reduction::Reduce(keys) => keys.len(), } @@ -3655,14 +3781,14 @@ fn summarised_column(intent: &AggIntent, child_schema: &Schema) -> ColumnRef { match intent .input_cols() .first() - .and_then(|id| child_schema.columns.get(*id)) + .and_then(|id| child_schema.fields.get(*id)) { Some(c) => column_ref(c), None => ColumnRef::SampleValue, } } -fn column_ref(column: &asap_types::pre_asap::Column) -> ColumnRef { +fn column_ref(column: &Field) -> ColumnRef { match &column.table { Some(t) => ColumnRef::Qualified { table: t.clone(), @@ -3693,7 +3819,7 @@ fn summarised_input( } let legs = cols .iter() - .map(|id| child_schema.columns.get(*id).map(column_ref)) + .map(|id| child_schema.fields.get(*id).map(column_ref)) .collect::>>() .ok_or(RealizationError::PhysicalRealization( "a tuple column is outside the input schema", @@ -3703,19 +3829,19 @@ fn summarised_input( )) } -/// The `SummaryEstimate` readout for a summary-bound intent. -fn readout( +/// The `SummaryEstimate` evaluation for a summary-bound intent. +fn evaluation( intent: &AggIntent, input: &SummaryUpdate, cost_model: &dyn CostModel, -) -> PostAsapSketchQuery { +) -> PostAsapSketchStatistic { match intent { - AggIntent::Quantile { q, .. } => PostAsapSketchQuery::Quantile { q: *q }, - AggIntent::Cardinality { .. } => PostAsapSketchQuery::Cardinality, - AggIntent::FrequencyL2 { .. } => PostAsapSketchQuery::FrequencyL2, - AggIntent::FrequencyEntropy { .. } => PostAsapSketchQuery::FrequencyEntropy, - AggIntent::TopK { k, .. } => PostAsapSketchQuery::TopK { k: *k }, - AggIntent::Count { .. } => PostAsapSketchQuery::PointCount { + AggIntent::Quantile { q, .. } => PostAsapSketchStatistic::Quantile { q: *q }, + AggIntent::Cardinality { .. } => PostAsapSketchStatistic::Cardinality, + AggIntent::FrequencyL2 { .. } => PostAsapSketchStatistic::FrequencyL2, + AggIntent::FrequencyEntropy { .. } => PostAsapSketchStatistic::FrequencyEntropy, + AggIntent::TopK { k, .. } => PostAsapSketchStatistic::TopK { k: *k }, + AggIntent::Count { .. } => PostAsapSketchStatistic::PointCount { key: match &input.weight { SummaryInputExpr::Column(col) => col.clone(), SummaryInputExpr::Constant(1.0) => ColumnRef::SampleValue, @@ -3724,13 +3850,15 @@ fn readout( value: None, }, // Core doesn't know the shape of a deployment-specific `Extension` - // intent, so it can't build its readout either — delegate to the + // intent, so it can't build its evaluation either — delegate to the // same `CostModel` that decided (via `realize_extension`) this - // intent gets a summary realization at all. See `readout_extension`'s + // intent gets a summary realization at all. See `evaluation_extension`'s // doc for the invariant this depends on. AggIntent::Extension { ext_kind, payload } => match &input.weight { - SummaryInputExpr::Column(col) => cost_model.readout_extension(ext_kind, payload, col), - _ => unreachable!("extension readout requires one column"), + SummaryInputExpr::Column(col) => { + cost_model.evaluation_extension(ext_kind, payload, col) + } + _ => unreachable!("extension evaluation requires one column"), }, other => { unreachable!("no summary realization for {other:?} (realizations_for_intent)") @@ -3738,33 +3866,22 @@ fn readout( } } -/// Lift a pre-ASAP [`Schema`] to a [`SummarySchema`] with every column -/// `SummaryFamilyType::Plain` — shared by [`construct_summary_agg`] and -/// [`keep_pre_asap`], both in this module. -fn lift(schema: &Schema) -> SummarySchema { - SummarySchema { - fields: schema - .columns - .iter() - .map(|c| SummaryField { - name: c.name.clone(), - dtype: SummaryFamilyType::Plain(c.dtype.clone()), - nullable: c.nullable, - }) - .collect(), - time_index: schema.time_index, - } +/// Lift a pre-ASAP [`Schema`] to a [`Schema`] with every column +/// `FieldDataType::Plain` — shared by [`construct_summary_agg`] and +/// [`retain_exact`], both in this module. +fn lift(schema: &Schema) -> Schema { + Schema::lifted(schema.fields.clone(), schema.time_index) } -// ── SharedSubtreeStrategy ──────────────────────────────────────────────── +// ── SharedSubDagStrategy ──────────────────────────────────────────────── -/// Wraps `asap_types::pre_asap::cse::share_common_subtrees`'s sharing +/// Wraps `asap_types::ir::cse::share_common_subdags`'s sharing /// decision as an explicit candidate pair, wherever a [`TargetSubDAG`] /// already has two or more consumers. /// /// This strategy does not decide sharing itself, nor does it discover which /// nodes are shared — by the time a caller builds a `TargetSubDAG` with -/// `consumer_count >= 2`, `share_common_subtrees` has already made that +/// `consumer_count >= 2`, `share_common_subdags` has already made that /// (legality-gated, `PartialEq`-checked) call; [`discover_targets`] below /// discovers real consumer counts across a workload the same way for /// [`search_workload_with`] (this module's own tests reuse the identical @@ -3774,9 +3891,9 @@ fn lift(schema: &Schema) -> SummarySchema { /// `Rc`" as the two-way choice a downstream cost model (today, /// [`CostModel::cse_share_decision`]) picks between: build once and share, or /// build independently at each consumer. -pub struct SharedSubtreeStrategy; +pub struct SharedSubDagStrategy; -impl ReplacementStrategy for SharedSubtreeStrategy { +impl ReplacementStrategy for SharedSubDagStrategy { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { target.consumer_count >= 2 } @@ -3788,26 +3905,26 @@ impl ReplacementStrategy for SharedSubtreeStrategy { let count = target.consumer_count; vec![ ReplacementSubDAG { - strategy: "SharedSubtreeStrategy", + strategy: "SharedSubDagStrategy", // The already-interned `Rc` itself: reusing it verbatim *is* // "build once and share" — no new node to construct. - replacement: Replacement::Rewrite(Rc::clone(target.root)), + replacement: Replacement::SubDag(Rc::clone(target.root)), provenance: ReplacementProvenance::CseShare, rationale: format!( - "build once and share: share_common_subtrees already interned this \ + "build once and share: share_common_subdags already interned this \ subtree once and reused it across {count} consumers — one build can \ answer all of them instead of computing it {count} times" ), }, ReplacementSubDAG { - strategy: "SharedSubtreeStrategy", + strategy: "SharedSubDagStrategy", // A structurally-identical but freshly-allocated `Rc`: same // value (`PartialEq`), deliberately *not* the same pointer, // representing "undo the sharing and recompute independently". - replacement: Replacement::Rewrite(Rc::new((**target.root).clone())), + replacement: Replacement::SubDag(Rc::new((**target.root).clone())), provenance: ReplacementProvenance::CseRecompute, rationale: format!( - "build independently: undo the sharing share_common_subtrees found and \ + "build independently: undo the sharing share_common_subdags found and \ recompute this subtree separately at each of its {count} consumers — \ worth it only when independence outweighs the shared-maintenance cost, \ a CostModel's call (e.g. CostModel::cse_share_decision) and not this \ @@ -3827,14 +3944,14 @@ impl ReplacementStrategy for SharedSubtreeStrategy { /// A generous, documented backstop against a hypothetically ill-behaved /// future [`ReplacementStrategy`] (see the module docs' "Termination" /// section) — not a bound either shipped strategy could ever approach. -/// [`SketchAlgorithmStrategy`] and [`SharedSubtreeStrategy`] both converge in +/// [`ASAPStrategies`] and [`SharedSubDagStrategy`] both converge in /// exactly 2 passes over a fixed target set, regardless of workload size. pub const MAX_SEARCH_ITERATIONS: usize = 1_000; // ── TargetSubDAGCandidates ────────────────────────────────────────────── /// Candidates for one distinct [`TargetSubDAG`] (its -/// own `target` `Rc`, keyed by pointer identity in +/// own `target` `Rc`, keyed by pointer identity in /// [`CandidateLogicalASAPDAGs`]'s internal map — never re-derived by value) plus every /// [`ReplacementSubDAG`] alternative any registered [`ReplacementStrategy`] /// proposed for it. @@ -3847,7 +3964,7 @@ pub const MAX_SEARCH_ITERATIONS: usize = 1_000; #[derive(Debug, Clone)] pub struct TargetSubDAGCandidates { /// The target sub-DAG this group is for. - pub target: Rc, + pub target: Rc, /// How many operator-child positions across the whole workload /// reference this exact `Rc` — see [`discover_targets`]. pub consumer_count: usize, @@ -3864,7 +3981,7 @@ pub struct TargetSubDAGCandidates { } impl TargetSubDAGCandidates { - fn new(target: Rc, consumer_count: usize) -> Self { + fn new(target: Rc, consumer_count: usize) -> Self { Self { target, consumer_count, @@ -3881,10 +3998,12 @@ impl TargetSubDAGCandidates { fn add_candidate(&mut self, candidate: ReplacementSubDAG) -> bool { let is_duplicate = self.candidates.iter().any(|existing| { match (&existing.replacement, &candidate.replacement) { - (Replacement::Rewrite(existing_rc), Replacement::Rewrite(rc)) => { + (Replacement::SubDag(existing_rc), Replacement::SubDag(rc)) + if is_logical_rewrite(existing_rc) && is_logical_rewrite(rc) => + { is_duplicate_rewrite(existing_rc, rc, &self.target) } - (Replacement::Summary(existing_node), Replacement::Summary(node)) => { + (Replacement::SubDag(existing_node), Replacement::SubDag(node)) => { is_duplicate_summary(existing_node, node) } ( @@ -3905,12 +4024,12 @@ impl TargetSubDAGCandidates { } } -/// Are `existing` and `candidate` the same [`Replacement::Rewrite`] -/// candidate for a group targeting `target`? +/// Are `existing` and `candidate` the same logical-rewrite +/// [`Replacement::SubDag`] candidate for a group targeting `target`? /// -/// Structural (`QueryExpr`) value equality alone is *not* enough here: this -/// module's one shipped multi-candidate `Replacement::Rewrite` source, -/// [`SharedSubtreeStrategy`], deliberately returns **two** candidates that +/// Structural (`OperatorNode`) value equality alone is *not* enough here: +/// this module's one shipped multi-candidate logical-rewrite source, +/// [`SharedSubDagStrategy`], deliberately returns **two** candidates that /// are value-equal to each other (`build once and share` vs. `build /// independently` — see that strategy's own doc) but represent genuinely /// different physical choices, distinguished *only* by whether the @@ -3927,7 +4046,7 @@ impl TargetSubDAGCandidates { /// So: two candidates whose "is this the target's own `Rc`?" bit disagrees /// are never duplicates of each other, full stop. Only when that bit /// *agrees* does this fall through to the real dedup discipline — -/// [`structural_hash`] as a candidate-narrowing filter, `QueryExpr`'s +/// [`structural_hash`] as a candidate-narrowing filter, `OperatorNode`'s /// derived `PartialEq` as the actual decision — protecting against the /// (currently hypothetical, since neither shipped strategy causes it) /// case of the exact same alternative being proposed twice. A fresh @@ -3936,9 +4055,9 @@ impl TargetSubDAGCandidates { /// wider traversal to amortize the cache across the way `InternTable`'s own /// use of `structural_hash` does. fn is_duplicate_rewrite( - existing: &Rc, - candidate: &Rc, - target: &Rc, + existing: &Rc, + candidate: &Rc, + target: &Rc, ) -> bool { let existing_is_target = Rc::ptr_eq(existing, target); let candidate_is_target = Rc::ptr_eq(candidate, target); @@ -3950,16 +4069,16 @@ fn is_duplicate_rewrite( && existing == candidate } -/// Are `existing` and `candidate` the same [`Replacement::Summary`] -/// candidate? +/// Are `existing` and `candidate` the same bound-summary +/// [`Replacement::SubDag`] candidate? /// -/// [`SummaryNode`] derives neither `PartialEq` nor `Hash` (it embeds -/// `SketchParams`/`f64`-bearing accuracy targets deep inside `SummaryExpr`, -/// the same reason `QueryExpr` can't derive `Hash` either — see -/// [`structural_hash`]'s own doc). Per this module's inherited "hash is a -/// filter, `PartialEq` is the decision, no exceptions" rule, there is no -/// real equality check to back a dedup *decision* here — and skipping the -/// check is the only choice that rule permits: never merging two candidates +/// A bound summary embeds `SketchParams`/`f64`-bearing accuracy targets and +/// guarantees, so value equality is not a dedup decision this module is +/// willing to make (see [`structural_hash`]'s own doc on `f64` hashing). +/// Per this module's inherited "hash is a filter, `PartialEq` is the +/// decision, no exceptions" rule, there is no real equality check to back a +/// dedup *decision* here — and skipping the check is the only choice that +/// rule permits: never merging two candidates /// is harmless (at worst, a redundant entry in a group's candidate list), /// while comparing by some proxy this module can't actually verify (e.g. /// `Debug` text, or `ReplacementSubDAG::rationale` — documented elsewhere in @@ -3968,7 +4087,7 @@ fn is_duplicate_rewrite( /// shipped today already return a structurally distinct candidate for every /// entry of one `replacements()` call, so this is future-proofing against a /// hypothetical repeat call, not a gap either strategy's own tests exercise. -fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc) -> bool { +fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc) -> bool { false } @@ -3977,34 +4096,34 @@ fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc` whose -/// group holds its alternatives. +/// caller can still map a `Root`'s `Id` back to the `Rc` whose +/// group holds its alternatives. Memos are keyed by `*const OperatorNode`. pub struct CandidateLogicalASAPDAGs { - /// The workload's roots, after the one `share_common_subtrees` pass + /// The workload's roots, after the one `share_common_subdags` pass /// [`search_workload_with`] runs up front — the same post-CSE roots /// every `TargetSubDAG` in `groups` was discovered from. - pub roots: Vec<(Id, Rc)>, - groups: HashMap<*const QueryExpr, TargetSubDAGCandidates>, + pub roots: Vec<(Id, Rc)>, + groups: HashMap<*const OperatorNode, TargetSubDAGCandidates>, /// Discovery order — stable iteration for [`CandidateLogicalASAPDAGs::target_subdag_candidates`]/ /// [`CandidateLogicalASAPDAGs::cost_sorted`], since `HashMap` iteration order isn't. - order: Vec<*const QueryExpr>, + order: Vec<*const OperatorNode>, /// Composition proofs are computed with the search model, then retained /// through costing and DAG assembly so no later default can replace it. composition_plans: Vec, } struct PreparedComposition { - target: *const QueryExpr, + target: *const OperatorNode, operation: ExactComposition, - child: Rc, - plan: Rc, + child: Rc, + plan: Rc, } impl CandidateLogicalASAPDAGs { fn prepare_compositions( &mut self, accuracy: &dyn AccuracyModel, - targets: &HashMap<*const QueryExpr, Vec>, + targets: &HashMap<*const OperatorNode, Vec>, ) { self.composition_plans.clear(); for group in self.groups.values() { @@ -4019,14 +4138,16 @@ impl CandidateLogicalASAPDAGs { .into_iter() .flat_map(|g| &g.candidates) .filter_map(|c| match &c.replacement { - Replacement::Summary(child) if operation.accepts_child(child) => { + Replacement::SubDag(child) + if !is_logical_rewrite(child) && operation.accepts_child(child) => + { Some(Rc::clone(child)) } _ => None, }) .collect(), OperationPlacement::Maintenance => { - keep_pre_asap(&operation.child_target).into_iter().collect() + retain_exact(&operation.child_target).into_iter().collect() } }; for child in children { @@ -4064,11 +4185,11 @@ impl CandidateLogicalASAPDAGs { /// never a silently truncated inventory presented as exhaustive. #[derive(Debug)] pub struct CandidateDagInventory { - pub candidates: Vec)>>, + pub candidates: Vec)>>, pub rejected_assemblies: Vec, } -type CandidateDagChoice<'a> = (Option<&'a ReplacementSubDAG>, Option>); +type CandidateDagChoice<'a> = (Option<&'a ReplacementSubDAG>, Option>); impl CandidateLogicalASAPDAGs { pub fn enumerate_candidate_dags( @@ -4103,7 +4224,7 @@ impl CandidateLogicalASAPDAGs { fn enumerate_candidate_roots( &self, - roots: &[(Id, Rc)], + roots: &[(Id, Rc)], expansion_limit: usize, ) -> Result, RealizationError> { let mut reachable = Vec::new(); @@ -4119,8 +4240,10 @@ impl CandidateLogicalASAPDAGs { cursor += 1; if let Some(group) = self.groups.get(&ptr) { for candidate in &group.candidates { - if let Replacement::Rewrite(rewritten) = &candidate.replacement { - walk(rewritten, &mut reachable, &mut nodes, &mut counts); + if let Replacement::SubDag(rewritten) = &candidate.replacement { + if is_logical_rewrite(rewritten) { + walk(rewritten, &mut reachable, &mut nodes, &mut counts); + } } } } @@ -4214,77 +4337,12 @@ impl CandidateLogicalASAPDAGs { .collect::, _>>(); match roots { Ok(roots) => { - let roots = asap_types::post_asap::share_common_summary_subtrees(roots); + let roots = share_common_subdags(roots); use std::hash::{Hash, Hasher}; let mut hash = std::collections::hash_map::DefaultHasher::new(); - let mut pending = roots - .iter() - .map(|(_, node)| node.as_ref()) - .collect::>(); - while let Some(node) = pending.pop() { - std::mem::discriminant(&node.expr).hash(&mut hash); - let raw = match &node.expr { - SummaryExpr::KeepPreAsap(raw) => Some(raw.as_ref()), - _ => None, - }; - let operation = match &node.expr { - SummaryExpr::ValueOperation { - timing, operation, .. - } => serde_json::json!((timing, operation)), - SummaryExpr::BinaryOp { - timing, operator, .. - } => serde_json::json!((timing, operator)), - SummaryExpr::SummaryMerge { timing, .. } => serde_json::json!(timing), - _ => serde_json::Value::Null, - }; - let mut value = - serde_json::to_value((&node.schema, &node.guarantee, raw, operation)) - .map_err(|_| { - RealizationError::PhysicalRealization( - "candidate identity serialization failed", - ) - })?; - fn normalize(value: &mut serde_json::Value) { - match value { - serde_json::Value::Number(number) - if number.as_f64() == Some(0.0) => - { - *value = serde_json::json!(0); - } - serde_json::Value::Array(values) => { - values.iter_mut().for_each(normalize) - } - serde_json::Value::Object(values) => { - values.values_mut().for_each(normalize) - } - _ => {} - } - } - normalize(&mut value); - value.sort_all_objects(); - value.to_string().hash(&mut hash); - match &node.expr { - SummaryExpr::KeepPreAsap(_) => {} - SummaryExpr::BinaryOp { lhs, rhs, .. } => { - pending.extend([lhs.as_ref(), rhs.as_ref()]) - } - SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::SummarySubtract { left, right } => { - pending.extend([left.as_ref(), right.as_ref()]) - } - SummaryExpr::ValueOperation { child, .. } - | SummaryExpr::SummaryAgg { child, .. } => pending.push(child.as_ref()), - SummaryExpr::SummaryJoin { outer, inner, .. } => { - pending.extend([outer.as_ref(), inner.as_ref()]) - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - pending.push(summary_input.as_ref()) - } - SummaryExpr::SummaryMerge { children, .. } => { - pending.extend(children.iter().map(|child| child.as_ref())) - } - } + let mut cache = HashCache::new(); + for (_, node) in &roots { + structural_hash(node, &mut cache).hash(&mut hash); } let bucket = seen.entry(hash.finish()).or_default(); if !bucket @@ -4310,25 +4368,25 @@ impl CandidateLogicalASAPDAGs { /// Lifecycle-aware whole-subplan costs keyed by target and candidate identity. #[derive(Default, Clone)] pub(crate) struct CandidateCostOverrides { - costs: HashMap<(*const QueryExpr, *const ReplacementSubDAG), Cost>, - raw_costs: HashMap<*const QueryExpr, Cost>, + costs: HashMap<(*const OperatorNode, *const ReplacementSubDAG), Cost>, + raw_costs: HashMap<*const OperatorNode, Cost>, /// Targets for which the caller requested an atomic raw-vs-summary /// decision. Other memo groups continue through ordinary CSE selection. - finalized_targets: HashSet<*const QueryExpr>, + finalized_targets: HashSet<*const OperatorNode>, } impl CandidateCostOverrides { - pub(crate) fn finalize_target(&mut self, target: &Rc) { + pub(crate) fn finalize_target(&mut self, target: &Rc) { self.finalized_targets.insert(Rc::as_ptr(target)); } - fn finalizes(&self, target: &Rc) -> bool { + fn finalizes(&self, target: &Rc) -> bool { self.finalized_targets.contains(&Rc::as_ptr(target)) } pub(crate) fn insert( &mut self, - target: &Rc, + target: &Rc, candidate: &ReplacementSubDAG, cost: Cost, ) { @@ -4336,17 +4394,17 @@ impl CandidateCostOverrides { .insert((Rc::as_ptr(target), candidate as *const _), cost); } - fn get(&self, target: &Rc, candidate: &ReplacementSubDAG) -> Option { + fn get(&self, target: &Rc, candidate: &ReplacementSubDAG) -> Option { self.costs .get(&(Rc::as_ptr(target), candidate as *const _)) .copied() } - pub(crate) fn insert_raw(&mut self, target: &Rc, cost: Cost) { + pub(crate) fn insert_raw(&mut self, target: &Rc, cost: Cost) { self.raw_costs.insert(Rc::as_ptr(target), cost); } - fn raw(&self, target: &Rc) -> Option { + fn raw(&self, target: &Rc) -> Option { self.raw_costs.get(&Rc::as_ptr(target)).copied() } } @@ -4363,7 +4421,7 @@ impl CandidateLogicalASAPDAGs { } /// Whether no targets were discovered at all (an empty workload, or one - /// with no `QueryExpr` nodes reachable from any root — never true for a + /// with no `OperatorNode`s reachable from any root — never true for a /// non-empty `roots`, since every root is itself a target). pub fn is_empty(&self) -> bool { self.groups.is_empty() @@ -4372,7 +4430,10 @@ impl CandidateLogicalASAPDAGs { /// The candidate set for `target`, if `target`'s own `Rc` is a discovered /// `TargetSubDAG` (i.e. `Rc::ptr_eq` to some node reachable from /// `roots`). - pub fn candidates_for_target(&self, target: &Rc) -> Option<&TargetSubDAGCandidates> { + pub fn candidates_for_target( + &self, + target: &Rc, + ) -> Option<&TargetSubDAGCandidates> { self.groups.get(&Rc::as_ptr(target)) } @@ -4494,17 +4555,17 @@ impl CandidateLogicalASAPDAGs { /// half of issue #287. Looked up by `Rc` pointer identity, the same /// currency [`CandidateLogicalASAPDAGs::candidates_for_target`]/[`GlobalSelection::for_target`] already /// use. -/// Holds an owned `Rc` clone alongside each profile (not just its -/// raw pointer) so this map keeps every node it describes alive for as long -/// as the map itself lives — a `RecurrenceProfileMap` is safe to outlive the -/// `CandidateLogicalASAPDAGs` it was built from. Without this, a raw `*const QueryExpr` key +/// Holds an owned `Rc` clone alongside each profile (not just +/// its raw pointer) so this map keeps every node it describes alive for as +/// long as the map itself lives — a `RecurrenceProfileMap` is safe to outlive +/// the `CandidateLogicalASAPDAGs` it was built from. Without this, a raw `*const OperatorNode` key /// could, after the originating `CandidateLogicalASAPDAGs` (the only other owner of those /// `Rc`s) is dropped, collide with an unrelated, later allocation that /// happens to reuse the same freed address — silently returning a stale /// profile for the wrong node (issue #287 review, bug 4). #[derive(Debug, Clone)] pub struct RecurrenceProfileMap { - profiles: HashMap<*const QueryExpr, (Rc, RecurrenceProfile)>, + profiles: HashMap<*const OperatorNode, (Rc, RecurrenceProfile)>, } impl RecurrenceProfileMap { @@ -4513,7 +4574,7 @@ impl RecurrenceProfileMap { /// in the [`CandidateLogicalASAPDAGs`] this map was built from (or carried no /// recurring/one-shot/update-rate metadata at all) — always a valid, /// "no metadata" answer, never a panic. - pub fn for_target(&self, target: &Rc) -> RecurrenceProfile { + pub fn for_target(&self, target: &Rc) -> RecurrenceProfile { self.profiles .get(&Rc::as_ptr(target)) .map(|(_, profile)| *profile) @@ -4533,7 +4594,7 @@ impl CandidateLogicalASAPDAGs { /// `self.roots[i]` — the same order [`search_workload`]/ /// [`search_workload_with`] were originally called with (post-CSE /// dedup preserves both root count and order — see - /// `asap_types::pre_asap::cse::share_common_subtrees`'s own + /// `asap_types::ir::cse::share_common_subdags`'s own /// `.map(...).collect()` body). This keeps `Id` fully opaque (no `Eq`/ /// `Hash`/`Clone` bound needed on it at all — issue #287's "keep /// caller/query identifiers opaque" requirement) at the cost of the @@ -4610,12 +4671,12 @@ impl CandidateLogicalASAPDAGs { } } - let mut rates: HashMap<*const QueryExpr, f64> = HashMap::new(); - let mut one_shot_counts: HashMap<*const QueryExpr, usize> = HashMap::new(); + let mut rates: HashMap<*const OperatorNode, f64> = HashMap::new(); + let mut one_shot_counts: HashMap<*const OperatorNode, usize> = HashMap::new(); // Sites actually reached by at least one root's own recurrence tag // during the walk below — see this method's own "Unreachable // sites" doc. - let mut reached: HashSet<*const QueryExpr> = HashSet::new(); + let mut reached: HashSet<*const OperatorNode> = HashSet::new(); for ((_, root), recurrence) in self.roots.iter().zip(root_recurrence) { let recurrence = *recurrence; @@ -4625,7 +4686,7 @@ impl CandidateLogicalASAPDAGs { // recomputed occurrence is evaluated twice as well; stopping // expansion after the first pointer visit undercounts exactly // the effective-consumer rate recurrence-aware costing needs. - let mut queue: VecDeque<(*const QueryExpr, usize)> = VecDeque::new(); + let mut queue: VecDeque<(*const OperatorNode, usize)> = VecDeque::new(); queue.push_back((root_ptr, 1)); while let Some((ptr, path_count)) = queue.pop_front() { @@ -4769,7 +4830,7 @@ impl CandidateLogicalASAPDAGs { &self, workload: &QueryWorkload, root_workload_entries: &[usize], - ) -> Result>, RecurrenceError> { + ) -> Result>, RecurrenceError> { let entry_count = workload.entries().count(); if root_workload_entries.len() != self.roots.len() { return Err(RecurrenceError::RootCountMismatch { @@ -4777,7 +4838,7 @@ impl CandidateLogicalASAPDAGs { got: root_workload_entries.len(), }); } - let mut bindings: HashMap<*const QueryExpr, HashSet> = HashMap::new(); + let mut bindings: HashMap<*const OperatorNode, HashSet> = HashMap::new(); for ((_, root), &entry_index) in self.roots.iter().zip(root_workload_entries) { if entry_index >= entry_count { return Err(RecurrenceError::InvalidWorkloadEntry { @@ -4819,12 +4880,12 @@ impl CandidateLogicalASAPDAGs { /// child always has `edge_count >= 1` in practice, but this keeps the /// helper correct regardless). fn contribute( - ptr: *const QueryExpr, + ptr: *const OperatorNode, times: usize, recurrence: RootRecurrence, - rates: &mut HashMap<*const QueryExpr, f64>, - one_shot_counts: &mut HashMap<*const QueryExpr, usize>, - reached: &mut HashSet<*const QueryExpr>, + rates: &mut HashMap<*const OperatorNode, f64>, + one_shot_counts: &mut HashMap<*const OperatorNode, usize>, + reached: &mut HashSet<*const OperatorNode>, ) { if times == 0 { return; @@ -4845,7 +4906,7 @@ fn contribute( /// [`CandidateLogicalASAPDAGs::cost_sorted`]. #[derive(Debug)] pub struct RankedTargetSubDAGCandidates<'a> { - pub target: &'a Rc, + pub target: &'a Rc, pub consumer_count: usize, pub candidates: Vec<&'a ReplacementSubDAG>, /// `costs[i]` is `candidates[i]`'s own grouping-state cost when available, @@ -4873,7 +4934,7 @@ fn rank_group<'a>( return ranked; } - // Shape 1: the exact `SharedSubtreeStrategy` share-vs-recompute pair — + // Shape 1: the exact `SharedSubDagStrategy` share-vs-recompute pair — // rank via `CostModel::cse_share_decision`, the same comparison // the local CSE ranking path already uses. if cse_candidate_pair(group).is_some() { @@ -4893,7 +4954,7 @@ fn rank_group<'a>( // estimate, compare N independent states with the shared grid directly. let target = TargetSubDAG::with_consumer_count(&group.target, group.consumer_count); let has_hydra = ranked.iter().any(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDag(node) = &candidate.replacement else { return false; }; summary_grouping(node).is_some_and(|grouping| { @@ -4925,7 +4986,7 @@ fn rank_group<'a>( return ranked; } - // Shape 3: `SketchAlgorithmStrategy`'s sketch-family candidates (every + // Shape 3: `ASAPStrategies`'s sketch-family candidates (every // candidate is a `Summary` that realizes a `SketchAlgorithm`) — rank via // `CostModel::rank_candidates`, the same hook `realizations_for_intent` // itself consults. @@ -4933,16 +4994,16 @@ fn rank_group<'a>( let kinds: Option> = ranked .iter() .map(|c| match &c.replacement { - Replacement::Summary(node) => sketch_kind_of(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => None, + Replacement::SubDag(node) => sketch_kind_of(node), + Replacement::ExactComposition(_) => None, }) .collect(); if let Some(kinds) = kinds { let order = crate::cost_model::validated_candidate_ranking(cost_model, intent, &kinds); ranked.sort_by_key(|c| { let kind = match &c.replacement { - Replacement::Summary(node) => sketch_kind_of(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => None, + Replacement::SubDag(node) => sketch_kind_of(node), + Replacement::ExactComposition(_) => None, }; kind.and_then(|k| order.iter().position(|o| *o == k)) .unwrap_or(usize::MAX) @@ -4974,11 +5035,11 @@ fn rank_group<'a>( } /// For a group whose candidates are all [`Replacement::Rewrite`] (the -/// [`SharedSubtreeStrategy`] shape): does [`CostModel::cse_share_decision`] +/// [`SharedSubDagStrategy`] shape): does [`CostModel::cse_share_decision`] /// prefer the candidate that shares `group.target`'s own `Rc` (`true`), or /// the one that recomputes independently (`false`)? `None` when there's no /// real comparison to make — fewer than 2 consumers (mirrors -/// [`SharedSubtreeStrategy::matches`]'s own gate), or `group.target` can't +/// [`SharedSubDagStrategy::matches`]'s own gate), or `group.target` can't /// actually be bound at all (no candidate and no logical fallback — never /// expected in practice for a target that's already part of a legitimate /// workload tree, but this degrades to "keep discovery order" rather than @@ -4999,21 +5060,21 @@ fn cse_preference(group: &TargetSubDAGCandidates, cost_model: &dyn CostModel) -> }) } -/// [`cse_preference`] only needs one representative bound [`SummaryNode`] +/// [`cse_preference`] only needs one representative bound [`OperatorNode`] /// for `target` (to build a [`CseCandidate`] for /// [`CostModel::cse_share_decision`]), not the full ranked candidate list -/// [`SketchAlgorithmStrategy::replacements`] returns — so this just reuses +/// [`ASAPStrategies::replacements`] returns — so this just reuses /// [`realize_child`], the same rank-and-take-first helper /// `construct_summary_agg`'s own recursion and /// [`crate::cost_model::DefaultCostModel::estimate_cost`] already use, /// wrapped to swallow the (here, uninteresting) error into `None`. -fn realize_one(target: &Rc, cost_model: &dyn CostModel) -> Option> { +fn realize_one(target: &Rc, cost_model: &dyn CostModel) -> Option> { realize_child(target, cost_model).ok() } -/// The `SketchAlgorithm` a bound [`Replacement::Summary`] candidate ultimately +/// The `SketchAlgorithm` a bound [`Replacement::SubDag`] candidate ultimately /// realizes, if any (`None` for an `ExactAggregate`/pass-through -/// `Summary` — nothing to rank against another `SketchAlgorithm`). +/// sub-DAG — nothing to rank against another `SketchAlgorithm`). /// /// Mirrors this module's own `#[cfg(test)]`-only `summary_family_algorithm` /// helper (in the test module below), which does the identical @@ -5022,23 +5083,27 @@ fn realize_one(target: &Rc, cost_model: &dyn CostModel) -> Option Option { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_kind_of(summary_input), - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), +fn sketch_kind_of(node: &OperatorNode) -> Option { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + sketch_kind_of(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), .. - } => Some(kind.algorithm().clone()), + }) => Some(kind.algorithm().clone()), _ => None, } } /// The grouping strategy used by a bound summary candidate, unwrapping its -/// readout node when necessary. -fn summary_grouping(node: &SummaryNode) -> Option<&GroupingStrategy> { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => summary_grouping(summary_input), - SummaryExpr::SummaryAgg { grouping, .. } => Some(grouping), +/// evaluation node when necessary. +fn summary_grouping(node: &OperatorNode) -> Option<&GroupingStrategy> { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + summary_grouping(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { grouping, .. }) => Some(grouping), _ => None, } } @@ -5047,7 +5112,7 @@ fn summary_grouping(node: &SummaryNode) -> Option<&GroupingStrategy> { /// One target sub-DAG's selected choice and usage information — the answer /// [`CandidateLogicalASAPDAGs::global_selection`] commits to for one site, after folding in -/// every ancestor [`SharedSubtreeStrategy`] decision on the path from a +/// every ancestor [`SharedSubDagStrategy`] decision on the path from a /// workload root to this site. See the module docs' "Whole-plan /// (cross-group) selection" section for the full recurrence. /// @@ -5063,7 +5128,7 @@ fn summary_grouping(node: &SummaryNode) -> Option<&GroupingStrategy> { #[derive(Debug)] pub struct TargetSubDAGSelection<'a> { /// The target sub-DAG this selection is for. - pub target: &'a Rc, + pub target: &'a Rc, /// [`TargetSubDAGCandidates::consumer_count`] — how many operator-child positions /// directly reference `target`, ignoring every ancestor's own choice. pub consumer_count: usize, @@ -5071,7 +5136,7 @@ pub struct TargetSubDAGSelection<'a> { /// ancestor's own selected candidate is accounted for — see /// [`multiplier`]'s doc for the exact recurrence. Equal to /// `consumer_count` unless some ancestor on a path from a root to this - /// site has a [`SharedSubtreeStrategy`] alternative that chose + /// site has a [`SharedSubDagStrategy`] alternative that chose /// [`ShareDecision::RecomputeIndependently`]. pub effective_consumer_count: usize, /// The candidate chosen for this target, or `None` when no replacement @@ -5094,18 +5159,18 @@ pub struct TargetSubDAGSelection<'a> { #[derive(Debug)] pub struct CompositionDecision<'a> { /// The exact child/operation pair validated by the search accuracy model. - pub plan: Rc, + pub plan: Rc, /// The child target the composed operator consumes. - pub child_target: &'a Rc, + pub child_target: &'a Rc, /// For a read-time operation: the child's own candidate committed alongside - /// (the summary readout the operator folds). `None` for an update-path + /// (the summary evaluation the operator folds). `None` for an update-path /// transform, whose input is raw update data — its cost is charged to /// the maintained summary *above* it instead. pub child_candidate: Option<&'a ReplacementSubDAG>, /// The composed plan's recurring rate — `read_operation_plan_cost_rate` /// or `maintenance_operation_plan_cost_rate`. pub cost_rate: CostRate, - /// `raw_recompute_cost_rate` — the `KeepPreAsap` baseline it beat. + /// `raw_recompute_cost_rate` — the kept-sub-DAG baseline it beat. pub baseline_rate: CostRate, /// The statistics (and their provenance) both rates were computed from. pub inputs: ExactCompositionCostInputs, @@ -5116,12 +5181,13 @@ pub struct CompositionDecision<'a> { /// [`CandidateLogicalASAPDAGs::cost_sorted`] use. #[derive(Debug)] pub struct GlobalSelection<'a> { - order: Vec<*const QueryExpr>, - groups: HashMap<*const QueryExpr, TargetSubDAGSelection<'a>>, + order: Vec<*const OperatorNode>, + groups: HashMap<*const OperatorNode, TargetSubDAGSelection<'a>>, /// [`Self::assemble_selected_dag`]'s memo — one bound node per target for the /// life of this selection, so two parents composing over one shared - /// child get the *same* `Rc`. - assembled_nodes: RefCell>>, + /// child get the *same* `Rc` (a kept pre-ASAP sub-DAG + /// shared by two parents stays one `Rc` the same way). + assembled_nodes: RefCell>>, } fn normalize_cross_input_equi_predicate( @@ -5129,15 +5195,17 @@ fn normalize_cross_input_equi_predicate( left_width: usize, total_width: usize, ) -> Option { - let QueryExpr::Compare { + let ScalarExpr::Compare { left, op: asap_types::pre_asap::CompareOpKind::Eq, right, - } = pred.0.as_ref() + semantics, + } = &pred.0 else { return None; }; - let (QueryExpr::Column(left_id), QueryExpr::Column(right_id)) = (left.as_ref(), right.as_ref()) + let (ScalarExpr::Column(left_id), ScalarExpr::Column(right_id)) = + (left.as_ref(), right.as_ref()) else { return None; }; @@ -5150,20 +5218,12 @@ fn normalize_cross_input_equi_predicate( } else { return None; }; - Some(Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(left_id)), + Some(Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(left_id)), op: asap_types::pre_asap::CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(right_id)), - }))) -} - -fn relational_join_guarantee( - left: Option<&ResultGuarantee>, - right: Option<&ResultGuarantee>, -) -> Option { - left.zip(right) - .filter(|(left, right)| left.is_exact() && right.is_exact()) - .map(|_| ResultGuarantee::exact("RelationalJoin over exact inputs")) + right: Box::new(ScalarExpr::Column(right_id)), + semantics: *semantics, + })) } impl<'a> GlobalSelection<'a> { @@ -5175,47 +5235,52 @@ impl<'a> GlobalSelection<'a> { /// The selection for `target`, if `target`'s own `Rc` is a discovered /// site (i.e. `Rc::ptr_eq` to some node reachable from the workload's /// roots). - pub fn for_target(&self, target: &Rc) -> Option<&TargetSubDAGSelection<'a>> { + pub fn for_target(&self, target: &Rc) -> Option<&TargetSubDAGSelection<'a>> { self.groups.get(&Rc::as_ptr(target)) } /// Link this selection's per-site decisions into one data_state-validated /// post-ASAP DAG rooted at `target` — the one place a committed - /// composition's child *reference* becomes an actual `Rc` + /// composition's child *reference* becomes an actual `Rc` /// edge (issue #171). `None` if `target` is not a discovered site. /// /// Per site: a [`Replacement::ExactComposition`] uses its validated /// operation/child plan, retaining the search model's guarantee; - /// a [`Replacement::Summary`] is + /// a bound-summary [`Replacement::SubDag`] is /// re-linked so its `SummaryAgg` child is the child target's own /// DAG assembly whenever that is phase-legal beneath maintenance /// (so a child that chose an `ValueOperationAtIngestionTime` actually ends up under - /// the summary); a [`Replacement::Rewrite`] or an unmatched site stays - /// the conservative `KeepPreAsap`. Memoized by target identity, so a - /// shared inner summary is one `Rc` no matter how many roots reach it. + /// the summary); a logical-rewrite [`Replacement::SubDag`] is kept + /// as it is (exact); an unmatched site keeps its own operator with each + /// child assembled independently ([`Self::assemble_residual`]). + /// Memoized by target identity, so a shared inner summary is one `Rc` + /// no matter how many roots reach it. pub fn assemble_selected_dag( &self, - target: &Rc, - ) -> Result>, RealizationError> { + target: &Rc, + ) -> Result>, RealizationError> { if !self.groups.contains_key(&Rc::as_ptr(target)) { return Ok(None); } self.assemble_target(target).map(Some) } - /// Assemble a complete query result, including an exact-state readout when + /// Assemble a complete query result, including an exact-state evaluation when /// needed. `assemble_selected_dag` also serves internal state frontiers; /// callers exposing query results must use this boundary instead. pub fn assemble_selected_query( &self, - target: &Rc, - ) -> Result>, RealizationError> { + target: &Rc, + ) -> Result>, RealizationError> { self.assemble_selected_dag(target)? .map(|node| finalize_query_candidate(node, target)) .transpose() } - fn assemble_target(&self, target: &Rc) -> Result, RealizationError> { + fn assemble_target( + &self, + target: &Rc, + ) -> Result, RealizationError> { let ptr = Rc::as_ptr(target); if let Some(node) = self.assembled_nodes.borrow().get(&ptr) { return Ok(Rc::clone(node)); @@ -5227,10 +5292,12 @@ impl<'a> GlobalSelection<'a> { .groups .get(&ptr) .and_then(|sel| sel.chosen) - .is_some_and(|candidate| matches!(&candidate.replacement, - Replacement::Summary(node) if matches!(&node.expr, - SummaryExpr::SummaryAgg { child, .. } - if !matches!(&child.expr, SummaryExpr::KeepPreAsap(raw) if contains_aggregate(raw))))); + .is_some_and(|candidate| { + matches!(&candidate.replacement, + Replacement::SubDag(node) if matches!(&node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) + if child.contains_asap() || !contains_aggregate(child))) + }); let node = if query_time_nested_sum(target) && !selected_composed_summary { self.assemble_residual(target)? } else { @@ -5241,8 +5308,10 @@ impl<'a> GlobalSelection<'a> { .map(|c| &c.replacement) { None => self.assemble_residual(target)?, - Some(Replacement::Rewrite(rewritten)) => keep_pre_asap(rewritten)?, - Some(Replacement::Summary(node)) => self.relink_summary(node, target)?, + Some(Replacement::SubDag(node)) if node.contains_asap() => { + self.relink_summary(node, target)? + } + Some(Replacement::SubDag(kept)) => retain_exact(kept)?, Some(Replacement::ExactComposition(_)) => Rc::clone( &self.groups[&ptr] .composition @@ -5258,130 +5327,124 @@ impl<'a> GlobalSelection<'a> { Ok(node) } - /// Preserve composable query-time value operators in post-ASAP form even - /// when the operator itself has no summary realization. Its child is - /// assembled independently, so a selected summary remains visible - /// beneath `Project`/`Filter`/`Sort`/`Limit` instead of being swallowed by - /// one opaque `KeepPreAsap` subtree. + /// Keep `target`'s own operator and assemble each child independently, + /// so a selected summary remains visible beneath a relational operator + /// that has no summary realization of its own instead of being + /// swallowed by one opaque kept sub-DAG. Every child that is a + /// discovered target is assembled (and finalized to query-time values); + /// any other child is kept as it is. The guarantee is composed from the + /// assembled children: all exact → exact; exactly one child → that + /// child's guarantee; otherwise unknown. An inner `Join` first has its + /// cross-input equi-predicate normalized; any other join is kept whole. fn assemble_residual( &self, - target: &Rc, - ) -> Result, RealizationError> { - if let QueryExpr::Join { + target: &Rc, + ) -> Result, RealizationError> { + if target.children().is_empty() { + // A leaf has nothing to assemble beneath it: keep it as it is. + return retain_exact(target); + } + let mut operator = target.operator.clone(); + if let Operator::NonASAP(NonASAPOp::Join { left, right, kind, pred, - } = target.as_ref() + }) = &mut operator { - let left_width = left.output_schema()?.columns.len(); - let total_width = left_width + right.output_schema()?.columns.len(); - let normalized_pred = matches!(kind, asap_types::pre_asap::JoinKind::Inner) + let left_width = left.schema.fields.len(); + let total_width = left_width + right.schema.fields.len(); + let normalized_pred = matches!(kind, JoinKind::Inner) .then(|| normalize_cross_input_equi_predicate(pred, left_width, total_width)) .flatten(); - let Some(pred) = normalized_pred else { - return keep_pre_asap(target); + let Some(normalized) = normalized_pred else { + return retain_exact(target); }; - let left = finalize_query_candidate(self.assemble_target(left)?, left)?; - let right = finalize_query_candidate(self.assemble_target(right)?, right)?; - let guarantee = - relational_join_guarantee(left.guarantee.as_ref(), right.guarantee.as_ref()); - let node = Rc::new(SummaryNode { - expr: SummaryExpr::RelationalJoin { - left, - right, - kind: kind.clone(), - pred, - pruning: None, - }, - schema: lift(&target.output_schema()?), - guarantee, - }); - validate_execution_data_states_at(&node, ExecutionDataState::QUERY_ROWS)?; - return Ok(node); + *pred = normalized; } - let (child_target, operation) = match target.as_ref() { - QueryExpr::Project { - cols, - qualifier, - child, - } => ( - child, - ValueOperation::Project { - cols: cols.clone(), - qualifier: qualifier.clone(), - }, - ), - QueryExpr::Filter { pred, child } => { - (child, ValueOperation::Filter { pred: pred.clone() }) + let mut failure = None; + let mut children = Vec::new(); + let operator = operator.map_children(|child| { + if failure.is_some() { + return Rc::clone(child); } - QueryExpr::Sort { - keys, - partition_by, - child, - } => ( - child, - ValueOperation::Sort { - keys: keys.clone(), - partition_by: partition_by.clone(), - }, - ), - QueryExpr::Limit { n, offset, child } => ( - child, - ValueOperation::Limit { - n: *n, - offset: *offset, - partition_by: match child.as_ref() { - QueryExpr::Sort { partition_by, .. } => partition_by.clone(), - _ => Default::default(), - }, - }, - ), - QueryExpr::Aggregate { - reduction, - measures, - output_names, - filters, - having, - child, - } if query_time_nested_sum(target) => ( - child, - ValueOperation::Exact(ExactOperation::Aggregate { - reduction: reduction.clone(), - measures: measures.clone(), - output_names: output_names.clone(), - filters: filters.clone(), - having: having.clone(), - }), - ), - _ => return keep_pre_asap(target), - }; - let child = finalize_query_candidate(self.assemble_target(child_target)?, child_target)?; - let guarantee = child.guarantee.clone(); - let node = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child, - operation, - timing: ExecutionTiming::QueryTime, - }, - schema: lift(&target.output_schema()?), - guarantee, + let assembled = if self.groups.contains_key(&Rc::as_ptr(child)) { + self.assemble_target(child) + .and_then(|node| finalize_query_candidate(node, child)) + } else { + Ok(Rc::clone(child)) + }; + match assembled { + Ok(node) => { + children.push(Rc::clone(&node)); + node + } + Err(error) => { + failure = Some(error); + Rc::clone(child) + } + } + }); + if let Some(error) = failure { + return Err(error); + } + // An operator that computes new values from its input rows has no + // sound accuracy composition over an approximate input (e.g. `max` + // over a quantile evaluation's rank error). Without a selected + // composition such a node stays an exact pre-ASAP sub-DAG; only the + // read-time nested SUM keeps its assembled children. + let computes_values = matches!( + target.non_asap(), + Some( + NonASAPOp::Aggregate { .. } + | NonASAPOp::BinaryOp { .. } + | NonASAPOp::SQLWindowFunc { .. } + ) + ) && !query_time_nested_sum(target); + let approximate_input = children.iter().any(|child| { + !child + .guarantee + .as_ref() + .is_some_and(ResultGuarantee::is_exact) }); - validate_execution_data_states_at(&node, ExecutionDataState::QUERY_ROWS)?; + if computes_values && approximate_input { + return retain_exact(target); + } + let guarantee = match children.as_slice() { + [child] => child.guarantee.clone(), + children + if children.iter().all(|child| { + child + .guarantee + .as_ref() + .is_some_and(ResultGuarantee::is_exact) + }) => + { + Some(ResultGuarantee::exact(format!( + "{} over exact inputs", + target.operator.kind_name() + ))) + } + _ => None, + }; + let node = Rc::new( + OperatorNode::with_schema(operator, lift(&target.schema)).with_guarantee(guarantee), + ); + validate_default(&node, ExecutionTiming::QueryTime)?; Ok(node) } - /// Re-link a bound `Summary` candidate's `SummaryAgg` child to the + /// Re-link a bound summary candidate's `SummaryAgg` child to the /// child target's own DAG assembly when that is legal beneath /// maintenance; otherwise keep the candidate exactly as constructed. fn relink_summary( &self, - node: &Rc, - target: &Rc, - ) -> Result, RealizationError> { - let QueryExpr::Aggregate { + node: &Rc, + target: &Rc, + ) -> Result, RealizationError> { + let Some(NonASAPOp::Aggregate { child: pre_child, .. - } = target.as_ref() + }) = target.non_asap() else { return Ok(Rc::clone(node)); }; @@ -5406,16 +5469,16 @@ impl<'a> GlobalSelection<'a> { /// A mergeable outer SUM over a relationally wrapped aggregate is a read-time /// reduction of the inner summary values. Maintaining the outer SUM directly -/// would hide that inner temporal aggregate inside `KeepPreAsap` and lose its -/// independently selected summary. -fn query_time_nested_sum(target: &QueryExpr) -> bool { - let QueryExpr::Aggregate { +/// would hide that inner temporal aggregate inside one kept sub-DAG and lose +/// its independently selected summary. +fn query_time_nested_sum(target: &OperatorNode) -> bool { + let Some(NonASAPOp::Aggregate { measures, filters, having: None, child, .. - } = target + }) = target.non_asap() else { return false; }; @@ -5424,13 +5487,15 @@ fn query_time_nested_sum(target: &QueryExpr) -> bool { && contains_aggregate(child) } -fn contains_aggregate(expr: &QueryExpr) -> bool { - match expr { - QueryExpr::Aggregate { .. } => true, - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => contains_aggregate(child), +fn contains_aggregate(expr: &OperatorNode) -> bool { + match expr.non_asap() { + Some(NonASAPOp::Aggregate { .. }) => true, + Some( + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. }, + ) => contains_aggregate(child), _ => false, } } @@ -5438,38 +5503,38 @@ fn contains_aggregate(expr: &QueryExpr) -> bool { /// Rebuild `node` (a `SummaryAgg`, possibly under a `SummaryEstimate`) with /// `new_child` as the `SummaryAgg`'s child, if the result still validates /// as maintained state; otherwise return `node` unchanged. -fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc { - match &node.expr { - SummaryExpr::SummaryEstimate { +fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } => { + }) => { let inner = relink_agg_child(summary_input, new_child); if Rc::ptr_eq(&inner, summary_input) { return Rc::clone(node); } - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { + OperatorNode::asap_node( + ASAPOp::SummaryEstimate { summary_input: inner, query: query.clone(), }, - schema: node.schema.clone(), - guarantee: node.guarantee.clone(), - }) + node.schema.clone(), + node.guarantee.clone(), + ) } - SummaryExpr::SummaryAgg { + Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, reduction, grouping, filter, - } => { + }) => { if Rc::ptr_eq(child, new_child) { return Rc::clone(node); } - let rebuilt = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { + let rebuilt = OperatorNode::asap_node( + ASAPOp::SummaryAgg { child: Rc::clone(new_child), family: family.clone(), input: input.clone(), @@ -5477,11 +5542,10 @@ fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc rebuilt, Err(_) => Rc::clone(node), } @@ -5490,13 +5554,15 @@ fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc) -> Option<&Rc> { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => maintained_summary(summary_input), - SummaryExpr::SummaryAgg { .. } => Some(node), +fn maintained_summary(node: &Rc) -> Option<&Rc> { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + maintained_summary(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => Some(node), _ => None, } } @@ -5513,10 +5579,10 @@ struct CompositionContext { /// child target ptr → the child's candidate an ancestor's composition /// already committed to (a later parent must compose with the *same* /// one, and the child's own selection is forced to it). - committed_child: HashMap<*const QueryExpr, *const ReplacementSubDAG>, + committed_child: HashMap<*const OperatorNode, *const ReplacementSubDAG>, /// site ptr → the maintained `SummaryAgg` directly above it, when its - /// parent chose a bound `Summary` — what an `ValueOperationAtIngestionTime` here feeds. - maintaining_parent: HashMap<*const QueryExpr, Rc>, + /// parent chose a bound summary — what an `ValueOperationAtIngestionTime` here feeds. + maintaining_parent: HashMap<*const OperatorNode, Rc>, } /// One eligible composed alternative at a site, before the cheapest wins. @@ -5529,10 +5595,10 @@ struct CompositionOption<'a> { /// composed-plan rate is *known* and beats the raw-recompute baseline — /// costed against each compatible child candidate already in `CandidateLogicalASAPDAGs` /// (or the one an earlier parent committed). Unknown statistics yield no -/// option at all: the conservative `KeepPreAsap` path stays. +/// option at all: the conservative kept-sub-DAG path stays. fn composition_options<'a>( group: &'a TargetSubDAGCandidates, - groups: &'a HashMap<*const QueryExpr, TargetSubDAGCandidates>, + groups: &'a HashMap<*const OperatorNode, TargetSubDAGCandidates>, effective: usize, cost_model: &dyn CostModel, context: &CompositionContext, @@ -5551,7 +5617,7 @@ fn composition_options<'a>( continue; }; let already_committed = context.committed_child.get(&child_ptr).copied(); - let cost = |summary: &SummaryNode, shared: bool| { + let cost = |summary: &OperatorNode, shared: bool| { let request = ExactCompositionCostRequest { target: &group.target, composition, @@ -5587,10 +5653,10 @@ fn composition_options<'a>( if !is_automatically_selectable(child_candidate, cost_model) { continue; } - let Replacement::Summary(summary) = &child_candidate.replacement else { + let Replacement::SubDag(summary) = &child_candidate.replacement else { continue; }; - if !composition.accepts_child(summary) { + if is_logical_rewrite(summary) || !composition.accepts_child(summary) { continue; } let Some(prepared) = plans.iter().find(|p| { @@ -5655,7 +5721,7 @@ impl CandidateLogicalASAPDAGs { /// "Whole-plan (cross-group) selection" section describes: one /// [`TargetSubDAGSelection`] per discovered site, each ranked against an /// `effective_consumer_count` that accounts for every ancestor - /// [`SharedSubtreeStrategy`] decision on the path to it — unlike + /// [`SharedSubDagStrategy`] decision on the path to it — unlike /// [`Self::cost_sorted`], whose per-group ranking only ever sees a /// group's own raw [`TargetSubDAGCandidates::consumer_count`]. /// Uncertified DDSketch ratios remain in [`CandidateLogicalASAPDAGs`] for downstream @@ -5699,8 +5765,8 @@ impl CandidateLogicalASAPDAGs { let topo = topological_order(&self.order, &graph); let mut effective_uses = graph.external_root_uses.clone(); - let mut chosen_share: HashMap<*const QueryExpr, ShareDecision> = HashMap::new(); - let mut groups: HashMap<*const QueryExpr, TargetSubDAGSelection<'_>> = HashMap::new(); + let mut chosen_share: HashMap<*const OperatorNode, ShareDecision> = HashMap::new(); + let mut groups: HashMap<*const OperatorNode, TargetSubDAGSelection<'_>> = HashMap::new(); let mut context = CompositionContext::default(); for ptr in &topo { @@ -5927,8 +5993,8 @@ impl CandidateLogicalASAPDAGs { // Record the maintained summary this site's bound candidate // builds, for a child that may compose an `ValueOperationAtIngestionTime` // beneath it. - if let (Some(Replacement::Summary(node)), QueryExpr::Aggregate { child, .. }) = - (chosen.map(|c| &c.replacement), group.target.as_ref()) + if let (Some(Replacement::SubDag(node)), Some(NonASAPOp::Aggregate { child, .. })) = + (chosen.map(|c| &c.replacement), group.target.non_asap()) { if let Some(summary) = maintained_summary(node) { context @@ -5940,7 +6006,7 @@ impl CandidateLogicalASAPDAGs { let outgoing_multiplier = multiplier(*ptr, &effective_uses, &chosen_share); match chosen { Some(ReplacementSubDAG { - replacement: Replacement::Rewrite(source), + replacement: Replacement::SubDag(source), provenance: ReplacementProvenance::AccuracyReconciliation, .. }) => { @@ -5953,8 +6019,10 @@ impl CandidateLogicalASAPDAGs { } _ => { let selected_rewrite = match chosen.map(|candidate| &candidate.replacement) { - Some(Replacement::Rewrite(rewrite)) => rewrite, - Some(Replacement::Summary(_) | Replacement::ExactComposition(_)) | None => { + Some(Replacement::SubDag(rewrite)) if is_logical_rewrite(rewrite) => { + rewrite + } + Some(Replacement::SubDag(_) | Replacement::ExactComposition(_)) | None => { &group.target } }; @@ -6009,7 +6077,7 @@ fn is_automatically_selectable(candidate: &ReplacementSubDAG, cost_model: &dyn C /// chose [`ShareDecision::RecomputeIndependently`] (each of its own uses /// gets its own independent execution, so referencing it costs as much as /// its *own* full multiplicity), or it has no Share/Recompute decision at -/// all (not a [`SharedSubtreeStrategy`] shape — nothing here collapses +/// all (not a [`SharedSubDagStrategy`] shape — nothing here collapses /// its multiplicity to one, so whatever multiplicity *its* ancestors /// established simply passes through). /// @@ -6020,9 +6088,9 @@ fn is_automatically_selectable(candidate: &ReplacementSubDAG, cost_model: &dyn C /// ancestor sits anywhere on the path from a root to a site — see the /// module docs' "Whole-plan (cross-group) selection" section. fn multiplier( - parent_ptr: *const QueryExpr, - effective_uses: &HashMap<*const QueryExpr, usize>, - chosen_share: &HashMap<*const QueryExpr, ShareDecision>, + parent_ptr: *const OperatorNode, + effective_uses: &HashMap<*const OperatorNode, usize>, + chosen_share: &HashMap<*const OperatorNode, ShareDecision>, ) -> usize { let effective = *effective_uses.get(&parent_ptr).expect( "topological_order guarantees a parent is processed (and its effective_consumer_count \ @@ -6046,7 +6114,7 @@ fn cse_candidate_pair( for candidate in &group.candidates { match candidate.provenance { ReplacementProvenance::CseShare => { - let Replacement::Rewrite(rc) = &candidate.replacement else { + let Replacement::SubDag(rc) = &candidate.replacement else { return None; }; if !Rc::ptr_eq(rc, &group.target) || share.replace(candidate).is_some() { @@ -6054,7 +6122,7 @@ fn cse_candidate_pair( } } ReplacementProvenance::CseRecompute => { - let Replacement::Rewrite(rc) = &candidate.replacement else { + let Replacement::SubDag(rc) = &candidate.replacement else { return None; }; if Rc::ptr_eq(rc, &group.target) @@ -6111,7 +6179,7 @@ fn decide_group_with_recurrence( )) } -/// The [`SharedSubtreeStrategy`] candidate matching `decision`: the one +/// The [`SharedSubDagStrategy`] candidate matching `decision`: the one /// that shares `group.target`'s own `Rc` for [`ShareDecision::Share`], the /// freshly-allocated one for [`ShareDecision::RecomputeIndependently`] — /// the same `Rc`-identity distinction [`is_duplicate_rewrite`]'s own doc @@ -6132,26 +6200,25 @@ fn pick_shared_subtree_candidate( /// The parent/child structure [`CandidateLogicalASAPDAGs::global_selection`]'s DP walks — /// built separately from [`discover_targets`]'s own `order`/`nodes`/`counts` /// maps (which only track *aggregate* reference counts, not per-parent -/// breakdown or direction) rather than extending that already-reviewed, -/// already-tested pass. Same "small duplicated traversal over reshaping -/// proven code" call as [`is_shared_subtree_group`]. +/// breakdown or direction). Selection needs per-parent edge counts to +/// distinguish shared producers from repeated uses within one consumer. struct ReferenceGraph { /// child ptr -> `(parent ptr, edge count from that one parent)`, for /// every direct operator-child edge in the relational-skeleton scope /// [`walk_children`] itself uses (an edge count above 1 happens when /// one parent references the same child from two different fields, /// e.g. a `Join`'s `left`/`right` both being the same `Rc`). - parents_of: HashMap<*const QueryExpr, Vec<(*const QueryExpr, usize)>>, + parents_of: HashMap<*const OperatorNode, Vec<(*const OperatorNode, usize)>>, /// parent ptr -> every distinct child ptr it directly references — the /// reverse of `parents_of`, for [`topological_order`]'s Kahn's-algorithm /// traversal. - children_of: HashMap<*const QueryExpr, Vec<*const QueryExpr>>, + children_of: HashMap<*const OperatorNode, Vec<*const OperatorNode>>, /// How many of the workload's own `roots` point directly at each node — /// a node's "external" use. Nothing inside the tree decides this (it /// isn't a reference from another discovered site), so it's never /// subject to any ancestor's Share/Recompute choice — it's the base /// case [`CandidateLogicalASAPDAGs::global_selection`]'s recurrence starts from. - external_root_uses: HashMap<*const QueryExpr, usize>, + external_root_uses: HashMap<*const OperatorNode, usize>, } /// Build an ordering graph containing every edge that could be selected: @@ -6177,7 +6244,10 @@ fn reference_graph(space: &CandidateLogicalASAPDAGs) -> ReferenceGraph { let group = &space.groups[ptr]; record_possible_edges(*ptr, &group.target, &mut graph); for candidate in &group.candidates { - if let Replacement::Rewrite(rewrite) = &candidate.replacement { + if let Replacement::SubDag(rewrite) = &candidate.replacement { + if !is_logical_rewrite(rewrite) { + continue; + } if candidate.provenance == ReplacementProvenance::AccuracyReconciliation { add_edge(*ptr, Rc::as_ptr(rewrite), 1, &mut graph); } else { @@ -6193,8 +6263,8 @@ fn reference_graph(space: &CandidateLogicalASAPDAGs) -> ReferenceGraph { /// [`ReferenceGraph`]'s fields), retaining the greatest multiplicity seen /// when the target and alternative rewrites expose the same edge. fn add_edge( - parent_ptr: *const QueryExpr, - child_ptr: *const QueryExpr, + parent_ptr: *const OperatorNode, + child_ptr: *const OperatorNode, edge_count: usize, graph: &mut ReferenceGraph, ) { @@ -6210,8 +6280,8 @@ fn add_edge( } fn record_possible_edges( - parent_ptr: *const QueryExpr, - node: &QueryExpr, + parent_ptr: *const OperatorNode, + node: &OperatorNode, graph: &mut ReferenceGraph, ) { for (child_ptr, edge_count) in direct_child_counts(node) { @@ -6221,8 +6291,8 @@ fn record_possible_edges( /// Direct relational-skeleton children and their edge multiplicities. /// `Concat` is transparent, matching [`walk_children`]'s site scope. -fn direct_child_counts(node: &QueryExpr) -> Vec<(*const QueryExpr, usize)> { - fn push(children: &mut Vec<(*const QueryExpr, usize)>, child: &Rc) { +fn direct_child_counts(node: &OperatorNode) -> Vec<(*const OperatorNode, usize)> { + fn push(children: &mut Vec<(*const OperatorNode, usize)>, child: &Rc) { let ptr = Rc::as_ptr(child); match children.iter_mut().find(|(existing, _)| *existing == ptr) { Some((_, count)) => *count += 1, @@ -6230,57 +6300,19 @@ fn direct_child_counts(node: &QueryExpr) -> Vec<(*const QueryExpr, usize)> { } } - fn collect(node: &QueryExpr, children: &mut Vec<(*const QueryExpr, usize)>) { - use QueryExpr::*; - match node { - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => { - push(children, c); - } - PromqlRelabel { child, .. } - | PromqlInfoEnrich { child, .. } - | PromqlSeriesSample { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | Sort { child, .. } - | Limit { child, .. } => { - push(children, child); - } - Concat { - children: concat_children, - .. - } => { - for c in concat_children { - collect(c, children); - } - } - Join { left, right, .. } | SetOp { left, right, .. } => { - push(children, left); - push(children, right); + fn collect(node: &OperatorNode, children: &mut Vec<(*const OperatorNode, usize)>) { + if let Some(NonASAPOp::Concat { + children: concat_children, + .. + }) = node.non_asap() + { + for c in concat_children { + collect(c, children); } - BinaryOp { lhs, rhs, .. } => { - push(children, lhs); - push(children, rhs); - } - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} + return; + } + for child in node.children() { + push(children, child); } } @@ -6296,14 +6328,17 @@ fn direct_child_counts(node: &QueryExpr) -> Vec<(*const QueryExpr, usize)> { /// different root paths can have a parent that's discovered *after* it (see /// this function's own test for a worked diamond example), which is exactly /// backwards for [`CandidateLogicalASAPDAGs::global_selection`]'s recurrence. -fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec<*const QueryExpr> { - let mut in_degree: HashMap<*const QueryExpr, usize> = HashMap::new(); +fn topological_order( + order: &[*const OperatorNode], + graph: &ReferenceGraph, +) -> Vec<*const OperatorNode> { + let mut in_degree: HashMap<*const OperatorNode, usize> = HashMap::new(); for ptr in order { let degree = graph.parents_of.get(ptr).map(Vec::len).unwrap_or(0); in_degree.insert(*ptr, degree); } - let mut queue: VecDeque<*const QueryExpr> = order + let mut queue: VecDeque<*const OperatorNode> = order .iter() .copied() .filter(|ptr| in_degree[ptr] == 0) @@ -6327,9 +6362,9 @@ fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec< assert_eq!( topo.len(), order.len(), - "topological_order: the discovered-site reference graph has a cycle — every QueryExpr \ - node is built from Rc children, which can't form one, so this indicates a bug in \ - reference_graph rather than a real cyclic workload", + "topological_order: the discovered-site reference graph has a cycle — every \ + OperatorNode is built from Rc children, which can't form one, so this indicates a bug \ + in reference_graph rather than a real cyclic workload", ); topo } @@ -6352,33 +6387,33 @@ fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec< /// included here (issue #253) even though it's a /// [`Replacement::Rewrite`]-only strategy with no [`CostModel`] of its own to /// plug in — it's context-free (`matches`/`replacements` need nothing beyond -/// the target itself) exactly like [`SharedSubtreeStrategy`], so it belongs +/// the target itself) exactly like [`SharedSubDagStrategy`], so it belongs /// in this list rather than being derived per-workload the way /// [`RollupStrategy`] is. Rewriting `avg` into `sum`/`count` upfront is what -/// lets [`SketchAlgorithmStrategy`] and [`SharedSubtreeStrategy`] see a +/// lets [`ASAPStrategies`] and [`SharedSubDagStrategy`] see a /// mergeable accumulator to sketch or share at all — see that module's own /// doc comment for why a bare `avg` node otherwise never becomes a /// [`ReplacementStrategy`] target for anything. pub fn default_strategies() -> Vec> { vec![ - Box::new(SketchAlgorithmStrategy::default_cost_model()), + Box::new(ASAPStrategies::default_cost_model()), Box::new(HydraGroupingStrategy::default_cost_model()), - Box::new(SharedSubtreeStrategy), + Box::new(SharedSubDagStrategy), Box::new(crate::rewrite::AvgToSumOverCountStrategy), Box::new(ExactCompositionStrategy::default_cost_model()), ] } -/// Like [`default_strategies`], but [`SketchAlgorithmStrategy`] ranks/binds via +/// Like [`default_strategies`], but [`ASAPStrategies`] ranks/binds via /// `cost_model` instead of the built-in [`DefaultCostModel`] — the same -/// customization point [`SketchAlgorithmStrategy::new`] itself offers. +/// customization point [`ASAPStrategies::new`] itself offers. pub fn default_strategies_with<'a>( cost_model: &'a dyn CostModel, ) -> Vec> { vec![ - Box::new(SketchAlgorithmStrategy::new(cost_model)), + Box::new(ASAPStrategies::new(cost_model)), Box::new(HydraGroupingStrategy::new(cost_model)), - Box::new(SharedSubtreeStrategy), + Box::new(SharedSubDagStrategy), Box::new(crate::rewrite::SemanticEquivalentRewriteStrategy), Box::new(ExactCompositionStrategy::new(cost_model)), ] @@ -6386,21 +6421,19 @@ pub fn default_strategies_with<'a>( /// Default context-free strategies with both deployment costing and typed /// planning-time accuracy evidence. This is the production counterpart of -/// constructing [`SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence`] and +/// constructing [`ASAPStrategies::new_with_planning_inputs_and_evidence`] and /// [`HydraGroupingStrategy::new_with_planning_inputs_and_evidence`] separately. pub fn default_strategies_with_evidence<'a>( cost_model: &'a dyn CostModel, evidence: &'a dyn AccuracyEvidenceProvider, ) -> Vec> { vec![ - Box::new( - SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - cost_model, - &DEFAULT_ACCURACY_MODEL, - &DEFAULT_ALLOCATOR, - evidence, - ), - ), + Box::new(ASAPStrategies::new_with_planning_inputs_and_evidence( + cost_model, + &DEFAULT_ACCURACY_MODEL, + &DEFAULT_ALLOCATOR, + evidence, + )), Box::new( HydraGroupingStrategy::new_with_planning_inputs_and_evidence( cost_model, @@ -6409,7 +6442,7 @@ pub fn default_strategies_with_evidence<'a>( evidence, ), ), - Box::new(SharedSubtreeStrategy), + Box::new(SharedSubDagStrategy), Box::new(crate::rewrite::AvgToSumOverCountStrategy), Box::new(ExactCompositionStrategy::new(cost_model)), ] @@ -6420,12 +6453,12 @@ pub fn default_strategies_with_evidence<'a>( /// Search a whole workload's pre-ASAP roots for every candidate replacement /// [`default_strategies`] can find, deduped into a [`CandidateLogicalASAPDAGs`]. Candidate /// *generation* uses the built-in [`DefaultCostModel`] (via -/// [`default_strategies`], the same way [`SketchAlgorithmStrategy::default_cost_model`] +/// [`default_strategies`], the same way [`ASAPStrategies::default_cost_model`] /// does); call [`CandidateLogicalASAPDAGs::cost_sorted`] on the result for the final /// `sorted_by(cost_model)` step. Use [`search_workload_with`] to plug in a /// custom strategy set (e.g. built via [`default_strategies_with`] for a /// deployment-specific [`CostModel`]). -pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidateLogicalASAPDAGs { +pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidateLogicalASAPDAGs { search_workload_with(roots, &default_strategies()) } @@ -6435,7 +6468,7 @@ pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidateLogicalA /// [`RollupStrategy`] is derived and added automatically after CSE for both /// entry points, because only this function owns the post-CSE sibling set. /// -/// Runs [`share_common_subtrees`] once over `roots` first — so every +/// Runs [`share_common_subdags`] once over `roots` first — so every /// strategy (and, transitively, every /// [`crate::explanation::ReplacementExplanation`] a caller reads off the /// result) sees the same already-deduplicated tree — then discovers every @@ -6445,10 +6478,10 @@ pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidateLogicalA /// section). Deduping candidate plans this way needs no /// [`CostModel`] at all — that only enters at two well-defined points: each /// [`ReplacementStrategy`] in `strategies` may already carry its own (e.g. -/// [`SketchAlgorithmStrategy::new`]'s), and [`CandidateLogicalASAPDAGs::cost_sorted`]'s final +/// [`ASAPStrategies::new`]'s), and [`CandidateLogicalASAPDAGs::cost_sorted`]'s final /// ranking step takes one explicitly. pub fn search_workload_with<'s, Id>( - roots: Vec<(Id, Rc)>, + roots: Vec<(Id, Rc)>, strategies: &[Box], ) -> CandidateLogicalASAPDAGs { let mut space = search_cse_workload_with(cse_workload(roots), strategies); @@ -6459,7 +6492,7 @@ pub fn search_workload_with<'s, Id>( /// [`search_workload_with`] plus a per-root end-to-end `AccuracyTarget` /// (issue #172) — the workload's `QueryRequirements.accuracy`, threaded /// alongside each root. After the search, every root that carries a target -/// has its group's bound [`Replacement::Summary`] candidates checked with +/// has its group's bound-summary [`Replacement::SubDag`] candidates checked with /// `accuracy_model`'s [`AccuracyModel::satisfies`]: a candidate whose /// guarantee is fully known and misses the target is moved from /// [`TargetSubDAGCandidates::candidates`] to [`TargetSubDAGCandidates::rejected`] *before* @@ -6467,16 +6500,16 @@ pub fn search_workload_with<'s, Id>( /// group. A constructible candidate with unknown accuracy remains visible for /// downstream review under an approximate target, but default whole-plan /// selection does not commit it. An exact target cannot accept an unknown -/// approximate summary. A `KeepPreAsap` candidate is +/// approximate summary. A kept pre-ASAP candidate is /// exact and always survives — the raw/pre-ASAP alternative is what an -/// unsatisfiable root keeps. Logical [`Replacement::Rewrite`] candidates -/// are not bound values and are left alone; the targets *inside* a rewrite -/// are their own groups. +/// unsatisfiable root keeps. Logical-rewrite [`Replacement::SubDag`] +/// candidates are not bound values and are left alone; the targets *inside* +/// a rewrite are their own groups. /// /// Precedence against per-node `AggIntent.accuracy` is documented in /// [`crate::accuracy`]'s module docs. pub fn search_workload_with_targets<'s, Id>( - roots: Vec<(Id, Rc, Option)>, + roots: Vec<(Id, Rc, Option)>, strategies: &[Box], accuracy_model: &dyn AccuracyModel, ) -> CandidateLogicalASAPDAGs { @@ -6490,7 +6523,7 @@ pub fn search_workload_with_targets<'s, Id>( .collect(); let mut space = search_cse_workload_with(cse_workload(roots), strategies); // `cse_workload` preserves root order, so targets zip by position. - let root_ptrs: Vec<(*const QueryExpr, AccuracyTarget)> = space + let root_ptrs: Vec<(*const OperatorNode, AccuracyTarget)> = space .roots .iter() .zip(targets) @@ -6532,11 +6565,11 @@ pub fn search_workload_with_targets<'s, Id>( .candidates .drain(..) .partition(|candidate| match &candidate.replacement { - Replacement::Summary(node) => node.guarantee.as_ref().map_or_else( + Replacement::SubDag(node) if is_logical_rewrite(node) => true, + Replacement::SubDag(node) => node.guarantee.as_ref().map_or_else( || !matches!(target, AccuracyTarget::Exact), |g| accuracy_model.satisfies(&g.optimistic_floor(), &target), ), - Replacement::Rewrite(_) => true, // A composition's guarantee depends on the concrete child; // prepare_compositions checks those pairs after all roots. Replacement::ExactComposition(_) => true, @@ -6544,7 +6577,7 @@ pub fn search_workload_with_targets<'s, Id>( group.candidates = legal; group.rejected.extend(illegal.into_iter().map(|candidate| { let (metric, bound, failure_probability) = match &candidate.replacement { - Replacement::Summary(node) => node + Replacement::SubDag(node) => node .guarantee .as_ref() .map(|g| { @@ -6559,7 +6592,6 @@ pub fn search_workload_with_targets<'s, Id>( None, None, )), - Replacement::Rewrite(_) => unreachable!("rewrites are never rejected here"), Replacement::ExactComposition(_) => ( asap_types::post_asap::ErrorMetric::AbsoluteValue, None, @@ -6584,22 +6616,22 @@ pub fn search_workload_with_targets<'s, Id>( /// The strictest accuracy among `siblings` that read the same summary input /// as `root` — same child, grouping and filters, and the same intent apart -/// from its accuracy (and a quantile's rank, a readout parameter) — when +/// from its accuracy (and a quantile's rank, a evaluation parameter) — when /// stricter than `root`'s own. One summary sized for the strictest consumer /// serves every sibling: #509's summary-capability rule. fn strictest_sibling_accuracy( - root: &QueryExpr, - siblings: &[Rc], + root: &OperatorNode, + siblings: &[Rc], ) -> Option { fn approximate(intent: &AggIntent) -> Option<&AccuracyTarget> { accuracy_target(intent).filter(|accuracy| !matches!(accuracy, AccuracyTarget::Exact)) } - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, filters, child, .. - } = root + }) = root.non_asap() else { return None; }; @@ -6607,12 +6639,12 @@ fn strictest_sibling_accuracy( let own = accuracy_budget(approximate(intent)?); let (mut eps, mut delta) = own; for sibling in siblings { - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: sibling_reduction, filters: sibling_filters, child: sibling_child, .. - } = sibling.as_ref() + }) = sibling.non_asap() else { continue; }; @@ -6650,48 +6682,45 @@ fn strictest_sibling_accuracy( } } -fn cse_workload(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> { - // `share_common_subtrees` wants owned `QueryExpr`s, not already-`Rc` - // roots — the same `Rc::try_unwrap`-with-clone-fallback pattern - // `asap_types::pre_asap::cse::intern_child` itself uses to recover an - // owned node without cloning in the common (uniquely-owned) case. - let owned_roots: Vec<(Id, QueryExpr)> = roots - .into_iter() - .map(|(id, rc)| { - let expr = Rc::try_unwrap(rc).unwrap_or_else(|shared| (*shared).clone()); - (id, expr) - }) - .collect(); - share_common_subtrees(owned_roots) +fn cse_workload(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> { + share_common_subdags(roots) } fn search_cse_workload_with<'s, Id>( - cse_roots: Vec<(Id, Rc)>, + cse_roots: Vec<(Id, Rc)>, strategies: &[Box], ) -> CandidateLogicalASAPDAGs { + for (_, root) in &cse_roots { + assert!( + !root.contains_asap(), + "search_workload: a workload root already contains an ASAP operator \ + ({}); replacement search takes the front end's pre-ASAP DAG only", + root.operator.kind_name() + ); + } let mut order = Vec::new(); let mut nodes = HashMap::new(); - let mut counts: HashMap<*const QueryExpr, usize> = HashMap::new(); + let mut counts: HashMap<*const OperatorNode, usize> = HashMap::new(); discover_targets(&cse_roots, &mut order, &mut nodes, &mut counts); - let siblings: Vec> = order + let siblings: Vec> = order .iter() .filter_map(|ptr| { let node = &nodes[ptr]; - matches!(node.as_ref(), QueryExpr::Aggregate { .. }).then(|| Rc::clone(node)) + matches!(node.non_asap(), Some(NonASAPOp::Aggregate { .. })).then(|| Rc::clone(node)) }) .collect(); let rollup_strategy = RollupStrategy::new(&siblings); let accuracy_reconciliation_strategy = AccuracyReconciliationStrategy::new(&siblings); - let limits: Vec> = order + let limits: Vec> = order .iter() .filter_map(|ptr| { let node = &nodes[ptr]; - matches!(node.as_ref(), QueryExpr::Limit { .. }).then(|| Rc::clone(node)) + matches!(node.non_asap(), Some(NonASAPOp::Limit { .. })).then(|| Rc::clone(node)) }) .collect(); let topk_reuse_strategy = TopKLimitReuseStrategy::new(&limits); - let mut groups: HashMap<*const QueryExpr, TargetSubDAGCandidates> = HashMap::new(); + let mut groups: HashMap<*const OperatorNode, TargetSubDAGCandidates> = HashMap::new(); for ptr in &order { groups.insert( *ptr, @@ -6714,7 +6743,7 @@ fn search_cse_workload_with<'s, Id>( "search_workload: fixpoint search did not converge within {MAX_SEARCH_ITERATIONS} \ rounds — a registered ReplacementStrategy's Replacement::Rewrite candidates keep \ exposing new, never-before-seen descendant structure every round. \ - SketchAlgorithmStrategy/SharedSubtreeStrategy never do this (see replacement.rs's \ + ASAPStrategies/SharedSubDagStrategy never do this (see replacement.rs's \ module docs' \"Termination\" section); check any custom strategies passed to \ search_workload_with.", ); @@ -6777,8 +6806,10 @@ fn search_cse_workload_with<'s, Id>( } for candidate in &proposed { - if let Replacement::Rewrite(rc) = &candidate.replacement { - discover_new_descendant_targets(rc, &mut order, &mut nodes, &mut counts); + if let Replacement::SubDag(rc) = &candidate.replacement { + if is_logical_rewrite(rc) { + discover_new_descendant_targets(rc, &mut order, &mut nodes, &mut counts); + } } } @@ -6816,12 +6847,13 @@ fn search_cse_workload_with<'s, Id>( /// Materialize share/recompute alternatives for descendants whose raw edge /// count is one but whose effective count can exceed one when a repeated /// ancestor is recomputed. We only do this when an ordinary repeated group -/// proves that `SharedSubtreeStrategy` is part of this search's strategy set. +/// proves that `SharedSubDagStrategy` is part of this search's strategy set. fn add_effective_count_cse_candidates( - order: &[*const QueryExpr], - groups: &mut HashMap<*const QueryExpr, TargetSubDAGCandidates>, + order: &[*const OperatorNode], + groups: &mut HashMap<*const OperatorNode, TargetSubDAGCandidates>, ) { - let mut possible_children: HashMap<*const QueryExpr, Vec<*const QueryExpr>> = HashMap::new(); + let mut possible_children: HashMap<*const OperatorNode, Vec<*const OperatorNode>> = + HashMap::new(); for ptr in order { let group = &groups[ptr]; let children = possible_children.entry(*ptr).or_default(); @@ -6831,7 +6863,10 @@ fn add_effective_count_cse_candidates( } } for candidate in &group.candidates { - if let Replacement::Rewrite(rewrite) = &candidate.replacement { + if let Replacement::SubDag(rewrite) = &candidate.replacement { + if !is_logical_rewrite(rewrite) { + continue; + } for (child, _) in direct_child_counts(rewrite) { if !children.contains(&child) { children.push(child); @@ -6867,14 +6902,14 @@ fn add_effective_count_cse_candidates( if potentially_repeated.contains(ptr) && cse_candidate_pair(group).is_none() { let target = Rc::clone(&group.target); let site = TargetSubDAG::with_consumer_count(&target, 2); - for mut candidate in SharedSubtreeStrategy.replacements(&site) { + for mut candidate in SharedSubDagStrategy.replacements(&site) { candidate.rationale = format!( "{}: this subtree can become repeated when a repeated ancestor is recomputed; \ global_selection decides using its effective consumer count", match candidate.provenance { ReplacementProvenance::CseShare => "build once and share", ReplacementProvenance::CseRecompute => "recompute independently", - _ => unreachable!("SharedSubtreeStrategy only emits CSE candidates"), + _ => unreachable!("SharedSubDagStrategy only emits CSE candidates"), } ); group.add_candidate(candidate); @@ -6889,10 +6924,10 @@ fn add_effective_count_cse_candidates( /// `Rc` and its real `consumer_count` — see the module docs' "Where /// `TargetSubDAG` discovery comes from" section for the full rationale. fn discover_targets( - roots: &[(Id, Rc)], - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, + roots: &[(Id, Rc)], + order: &mut Vec<*const OperatorNode>, + nodes: &mut HashMap<*const OperatorNode, Rc>, + counts: &mut HashMap<*const OperatorNode, usize>, ) { for (_, root) in roots { walk(root, order, nodes, counts); @@ -6901,17 +6936,17 @@ fn discover_targets( /// Scan `candidate`'s **children** (deliberately never `candidate`'s own /// top-level pointer — see the module docs' "Termination" section: a -/// [`Replacement::Rewrite`]'s value is an alternative *for* the target that +/// logical rewrite's value is an alternative *for* the target that /// proposed it, never a new target of its own) for any `Rc` not already /// known, appending each to `order`/`nodes`/`counts` so /// [`search_workload_with`]'s next round processes it. A no-op when every /// child is already known — the case both shipped strategies always produce /// (see that section). fn discover_new_descendant_targets( - candidate: &Rc, - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, + candidate: &Rc, + order: &mut Vec<*const OperatorNode>, + nodes: &mut HashMap<*const OperatorNode, Rc>, + counts: &mut HashMap<*const OperatorNode, usize>, ) { walk_children(candidate, order, nodes, counts); } @@ -6919,10 +6954,10 @@ fn discover_new_descendant_targets( /// Visit `node`: count this occurrence, and — the first time this exact /// `Rc` is seen — record it as a target and recurse into its children. fn walk( - node: &Rc, - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, + node: &Rc, + order: &mut Vec<*const OperatorNode>, + nodes: &mut HashMap<*const OperatorNode, Rc>, + counts: &mut HashMap<*const OperatorNode, usize>, ) { let ptr = Rc::as_ptr(node); let already_visited = counts.contains_key(&ptr); @@ -6934,61 +6969,25 @@ fn walk( } } -/// `node`'s own **relational-skeleton** operator children — the same scope -/// `asap_types::pre_asap::cse::share_common_subtrees`/`rebuild_children` -/// itself uses (see that module's "Algorithm" section) and -/// `tests::count_consumers` mirrors for its own fixtures. Exhaustive over -/// every `QueryExpr` variant: a new variant fails to compile here until this -/// match is extended too. +/// `node`'s own operator children ([`OperatorNode::children`]: operator +/// inputs plus the operator nodes its scalar expressions read), the same +/// scope `asap_types::ir::cse::share_common_subdags` itself uses and +/// `tests::count_consumers` mirrors for its own fixtures. `Concat` is +/// transparent: its branches are walked in place of it. fn walk_children( - node: &QueryExpr, - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, + node: &OperatorNode, + order: &mut Vec<*const OperatorNode>, + nodes: &mut HashMap<*const OperatorNode, Rc>, + counts: &mut HashMap<*const OperatorNode, usize>, ) { - use QueryExpr::*; - match node { - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => walk(c, order, nodes, counts), - PromqlRelabel { child, .. } - | PromqlInfoEnrich { child, .. } - | PromqlSeriesSample { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | Sort { child, .. } - | Limit { child, .. } => walk(child, order, nodes, counts), - Concat { children, .. } => { - for c in children { - walk_children(c, order, nodes, counts); - } - } - Join { left, right, .. } | SetOp { left, right, .. } => { - walk(left, order, nodes, counts); - walk(right, order, nodes, counts); - } - BinaryOp { lhs, rhs, .. } => { - walk(lhs, order, nodes, counts); - walk(rhs, order, nodes, counts); - } - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} + if let Some(NonASAPOp::Concat { children, .. }) = node.non_asap() { + for c in children { + walk_children(c, order, nodes, counts); + } + return; + } + for child in node.children() { + walk(child, order, nodes, counts); } } @@ -6997,17 +6996,19 @@ mod tests { use super::*; use crate::accuracy::PropagationStats; use crate::cost_model::Cost; - use crate::test_support::lower_promql; + use crate::test_support::{agg, agg_per_entity, lower_promql, metric_scan, timed}; + use asap_types::ir::operator_properties::{Reduction as ReductionTy, Source}; + use asap_types::ir::TimeRangeKind; use asap_types::pre_asap::agg_intent::{ agg_is_exact, default_cardinality, default_quantile, MathFunc, TimeFunc, }; - use asap_types::pre_asap::query_expr::{Reduction as ReductionTy, Source}; - use asap_types::pre_asap::schema::{Column, DataType, Schema as SchemaTy}; + use asap_types::pre_asap::schema::{DataType, Field, Schema as SchemaTy}; + use asap_types::types::AccuracyTarget; use std::collections::HashMap; // Candidate shape without execution timing: what is computed, not where. - fn timing_free_shape(node: &Rc) -> serde_json::Value { + fn timing_free_shape(node: &Rc) -> serde_json::Value { fn strip(value: &mut serde_json::Value) { match value { serde_json::Value::Object(fields) => { @@ -7018,9 +7019,10 @@ mod tests { _ => {} } } - let mut shape = - serde_json::to_value(asap_types::post_asap::compile_post_asap_dag(node).unwrap()) - .unwrap(); + let mut shape = serde_json::to_value( + asap_types::ir::export::compile_post_asap_dag(&timed(node)).unwrap(), + ) + .unwrap(); strip(&mut shape); shape } @@ -7032,7 +7034,7 @@ mod tests { ("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact), ("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)), ] { - let root = Rc::new(lower_promql(query, accuracy)); + let root = lower_promql(query, accuracy); let inventory = search_workload(vec![(0usize, root)]) .enumerate_candidate_dags(4096) .unwrap(); @@ -7047,37 +7049,32 @@ mod tests { } } - // Grouped Sum over Rate readouts stays a summary state in the inventory, + // Grouped Sum over Rate evaluations stays a summary state in the inventory, // so lifecycle assignment can place it in precompute or at query time. #[test] fn grouped_rate_sum_inventory_keeps_sum_state_for_lifecycle_placement() { - let root = Rc::new(lower_promql( - "sum by(job)(rate(m[1m]))", - AccuracyTarget::Exact, - )); + let root = lower_promql("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact); let inventory = search_workload(vec![(0usize, root)]) .enumerate_candidate_dags(4096) .unwrap(); - let is_exact = |node: &SummaryNode, kind: ExactKind| { - matches!(&node.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(k, _), .. - } if *k == kind) + let is_exact = |node: &OperatorNode, kind: ExactKind| { + matches!(&node.operator, Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(k, _), .. + }) if *k == kind) }; assert!(inventory.candidates.iter().any(|forest| { - let SummaryExpr::ValueOperation { child: sum, .. } = &forest[0].1.expr else { + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: sum }) = &forest[0].1.operator else { return false; }; - let SummaryExpr::SummaryAgg { child: rate, .. } = &sum.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child: rate, .. }) = &sum.operator else { return false; }; is_exact(sum, ExactKind::Sum) - && matches!(&rate.expr, SummaryExpr::ValueOperation { - child, operation: ValueOperation::FinalizeExactAccumulator, .. - } if is_exact(child, ExactKind::Rate)) + && matches!(&rate.operator, Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) if is_exact(child, ExactKind::Rate)) })); } - // Every exposed query result has a readout; internal accumulator frontiers stay states. + // Every exposed query result has a evaluation; internal accumulator frontiers stay states. #[test] fn query_candidate_roots_do_not_leak_exact_accumulator_state() { for query in [ @@ -7085,20 +7082,20 @@ mod tests { "sum by(job)(m)", "sum_over_time(m[1m])", ] { - let root = Rc::new(lower_promql(query, AccuracyTarget::Exact)); + let root = lower_promql(query, AccuracyTarget::Exact); let space = search_workload(vec![(0usize, root.clone())]); let inventory = space.enumerate_candidate_dags(4096).unwrap(); assert!(!inventory.candidates.is_empty()); - let strategy = SketchAlgorithmStrategy::new(&DefaultCostModel); + let strategy = ASAPStrategies::new(&DefaultCostModel); for candidate in strategy.propose(&TargetSubDAG::new(&root)).candidates { - if let Replacement::Summary(node) = candidate.replacement { + if let Replacement::SubDag(node) = candidate.replacement { let output = finalize_query_candidate(node, &root).unwrap(); assert!( output .schema .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_))), + .all(|field| matches!(field.dtype, FieldDataType::Plain(_))), "direct candidate {query} leaks state" ); } @@ -7118,7 +7115,7 @@ mod tests { node.schema .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_))), + .all(|field| matches!(field.dtype, FieldDataType::Plain(_))), "{query}: query root leaks state: {:?}", node.schema ); @@ -7128,7 +7125,7 @@ mod tests { #[test] fn unpriced_inventory_retains_quantile_families_and_raw_execution() { - let query = Rc::new(agg(vec![2], default_quantile(0.9), metric_scan(&["job"]))); + let query = agg(vec![2], default_quantile(0.9), metric_scan(&["job"])); let space = search_workload(vec![(0usize, query)]); let inventory = space.enumerate_candidate_dags(4096).unwrap(); let roots = inventory @@ -7141,7 +7138,7 @@ mod tests { assert!(inventory .candidates .iter() - .any(|forest| matches!(forest[0].1.expr, SummaryExpr::KeepPreAsap(_)))); + .any(|forest| !forest[0].1.contains_asap())); } // Independent roots must not require materializing their Cartesian product. @@ -7151,11 +7148,11 @@ mod tests { .map(|id| { ( id, - Rc::new(agg( + agg( vec![2], default_quantile((id + 1) as f64 / 25.0), metric_scan(&["job"]), - )), + ), ) }) .collect(); @@ -7177,7 +7174,7 @@ mod tests { assert!(inventory .candidates .iter() - .any(|forest| matches!(forest[0].1.expr, SummaryExpr::KeepPreAsap(_)))); + .any(|forest| !forest[0].1.contains_asap())); } assert!(space.enumerate_candidate_dags_for_root(&24, 4096).is_err()); assert!(space.enumerate_candidate_dags_for_root(&0, 0).is_err()); @@ -7190,11 +7187,11 @@ mod tests { .map(|id| { ( id, - Rc::new(agg( + agg( vec![2], default_quantile(0.5 + id as f64 * 0.4), metric_scan(&["job"]), - )), + ), ) }) .collect(); @@ -7216,30 +7213,31 @@ mod tests { #[test] fn inventory_budget_never_returns_a_silent_partial_search() { - let query = Rc::new(agg(vec![2], default_quantile(0.9), metric_scan(&["job"]))); + let query = agg(vec![2], default_quantile(0.9), metric_scan(&["job"])); let space = search_workload(vec![(0usize, query)]); assert!(space.enumerate_candidate_dags(0).is_err()); } fn equi_pred(left: ColumnId, right: ColumnId) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(left)), + Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(left)), op: asap_types::pre_asap::CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(right)), - })) + right: Box::new(ScalarExpr::Column(right)), + semantics: asap_types::ir::ExprSemantics::Sql, + }) } // Finite samples can overflow a sum although their native average is finite. #[test] fn temporal_average_requires_finite_division_guard() { - let root = Rc::new(lower_promql("avg_over_time(a[5m])", AccuracyTarget::Exact)); + let root = lower_promql("avg_over_time(a[5m])", AccuracyTarget::Exact); let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); let operator = candidates .iter() .find_map(|c| match &c.replacement { - Replacement::Summary(node) => match &node.expr { - SummaryExpr::BinaryOp { operator, .. } => Some(operator), + Replacement::SubDag(node) => match &node.operator { + Operator::NonASAP(NonASAPOp::BinaryOp { operator, .. }) => Some(operator), _ => None, }, _ => None, @@ -7257,20 +7255,20 @@ mod tests { // Approximate requests also admit exact temporal ranking candidates. #[test] fn approximate_temporal_topk_admits_exact_maintained_values() { - let root = Rc::new(lower_promql( + let root = lower_promql( "topk by(job)(1,count_over_time(a[5m]))", AccuracyTarget::EpsilonDelta { epsilon: 0.01, delta: 0.01, }, - )); + ); let planning_inputs = CandidatePlanningInputs::with_default_accuracy(&crate::cost_model::DefaultCostModel); let node = exact_topk_over_temporal_values(&root, planning_inputs) .unwrap() .expect("exact ranking is legal for an approximate request"); assert!(node.guarantee.as_ref().unwrap().is_exact()); - asap_types::post_asap::compile_post_asap_dag(&node).unwrap(); + crate::test_support::time_and_export(&node).unwrap(); } // Exact Top-K consumes the Planner's maintained temporal values. @@ -7280,7 +7278,7 @@ mod tests { "topk(5, sum_over_time(a[5m]))", "topk by(job)(5, count_over_time(a[5m]))", ] { - let root = Rc::new(lower_promql(query, AccuracyTarget::Exact)); + let root = lower_promql(query, AccuracyTarget::Exact); let planning_inputs = CandidatePlanningInputs::with_default_accuracy( &crate::cost_model::DefaultCostModel, ); @@ -7288,29 +7286,21 @@ mod tests { .unwrap() .expect("exact Top-K candidate"); assert!(node.guarantee.as_ref().unwrap().is_exact()); - let SummaryExpr::ValueOperation { + let Operator::NonASAP(NonASAPOp::Limit { child: sorted, - operation: - ValueOperation::Limit { - n, - offset, - partition_by, - }, - .. - } = &node.expr + n, + offset, + partition_by, + }) = &node.operator else { panic!("temporal TopK must compose Sort and Limit"); }; - assert_eq!((*n, *offset), (5, 0)); - let SummaryExpr::ValueOperation { - operation: - ValueOperation::Sort { - keys, - partition_by: sort_groups, - }, + assert_eq!((*n, *offset), (Some(5), 0)); + let Operator::NonASAP(NonASAPOp::Sort { + keys, + partition_by: sort_groups, child: values, - .. - } = &sorted.expr + }) = &sorted.operator else { panic!("Limit must consume sorted temporal values"); }; @@ -7322,7 +7312,7 @@ mod tests { assert_eq!(keys.len(), 1); assert!(!keys[0].ascending); assert_eq!(node.schema, values.schema); - asap_types::post_asap::compile_post_asap_dag(&node).unwrap(); + crate::test_support::time_and_export(&node).unwrap(); } } @@ -7333,7 +7323,7 @@ mod tests { impl AccuracyEvidenceProvider for Domain { fn quantile_input_domain( &self, - _: &QueryExpr, + _: &OperatorNode, ) -> Option { Some(crate::accuracy::QuantileInputDomain { lower: 1.0, @@ -7355,7 +7345,7 @@ mod tests { "avg_over_time(a[5m]) / quantile_over_time(0.5,a[5m])", "quantile_over_time(0.5,a[5m]) / avg_over_time(a[5m])", ] { - let root = Rc::new(lower_promql(query, target.clone())); + let root = lower_promql(query, target.clone()); let node = realize_binary(&root, inputs, Some(&target)) .unwrap() .expect("bounded ratio candidate"); @@ -7370,10 +7360,10 @@ mod tests { epsilon: 0.01, delta: 0.01, }; - let root = Rc::new(lower_promql( + let root = lower_promql( "quantile_over_time(0.5,a[5m]) / quantile_over_time(0.9,a[5m])", target.clone(), - )); + ); let planning_inputs = CandidatePlanningInputs::with_default_accuracy(&crate::cost_model::DefaultCostModel); let candidate = realize_binary(&root, planning_inputs, Some(&target)) @@ -7381,10 +7371,10 @@ mod tests { .expect("direct quantile ratio candidate"); assert!(candidate.guarantee.is_none()); - let other = Rc::new(lower_promql( + let other = lower_promql( "avg_over_time(a[5m]) / quantile_over_time(0.5,a[5m])", target.clone(), - )); + ); assert!(realize_binary(&other, planning_inputs, Some(&target)) .unwrap() .is_none()); @@ -7403,11 +7393,65 @@ mod tests { #[test] fn relational_join_is_exact_only_when_both_inputs_are_exact() { - let exact = ResultGuarantee::exact("test exact input"); - assert!(relational_join_guarantee(Some(&exact), Some(&exact)) - .is_some_and(|guarantee| guarantee.is_exact())); - assert!(relational_join_guarantee(Some(&exact), None).is_none()); - assert!(relational_join_guarantee(None, Some(&exact)).is_none()); + // `relational_join_guarantee` folded into assembly's generic + // "keep the operator, assemble its children" branch: an assembled + // inner equi-`Join` is exact exactly when both assembled inputs are. + let join = |left_intent: AggIntent, right_intent: AggIntent| { + let left = agg(vec![2], left_intent, metric_scan(&["job"])); + let right = agg( + vec![2], + right_intent, + crate::test_support::scan("n", metric_scan(&["job"]).schema.clone()), + ); + OperatorNode::non_asap_node(NonASAPOp::Join { + kind: asap_types::ir::operator_properties::JoinKind::Inner, + pred: equi_pred(0, 2), + left, + right, + }) + .unwrap() + }; + let is_exact = |node: &OperatorNode| { + node.guarantee + .as_ref() + .is_some_and(ResultGuarantee::is_exact) + }; + for (root, both_exact_expected) in [ + ( + join(AggIntent::Sum { col: None }, AggIntent::Sum { col: None }), + true, + ), + ( + join(AggIntent::Sum { col: None }, quantile_eps_intent(0.5, 0.05)), + false, + ), + ] { + let space = search_workload(vec![(0usize, Rc::clone(&root))]); + let assembled = space + .global_selection(&DefaultCostModel) + .assemble_selected_query(&space.roots[0].1) + .unwrap() + .unwrap(); + let Some(NonASAPOp::Join { left, right, .. }) = assembled.non_asap() else { + panic!("the join is kept and its inputs assembled: {assembled:?}"); + }; + assert_eq!( + is_exact(&assembled), + is_exact(left) && is_exact(right), + "join guarantee must be exact iff both inputs are exact" + ); + if both_exact_expected { + assert!(is_exact(&assembled), "exact inputs give an exact join"); + } + } + } + + fn quantile_eps_intent(q: f64, e: f64) -> AggIntent { + AggIntent::Quantile { + col: None, + q, + accuracy: AccuracyTarget::Epsilon(e), + } } fn eps(e: f64) -> AccuracyTarget { @@ -7957,64 +8001,43 @@ mod tests { ); } - // ── SketchAlgorithmStrategy / SharedSubtreeStrategy fixtures ─────────── - - fn metric_scan(labels: &[&str]) -> QueryExpr { - let mut columns = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - ]; - columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: SchemaTy::with_time_index(columns, 0, vec![]), - } - } - - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: ReductionTy::by(by), - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } + // ── ASAPStrategies / SharedSubDagStrategy fixtures ─────────── - // ── SketchAlgorithmStrategy ───────────────────────────────────────────── + // ── ASAPStrategies ───────────────────────────────────────────── #[test] fn matches_a_bindable_aggregate() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - assert!(SketchAlgorithmStrategy::default_cost_model().matches(&target)); + assert!(ASAPStrategies::default_cost_model().matches(&target)); } #[test] fn does_not_match_a_multi_intent_or_having_aggregate() { - let strategy = SketchAlgorithmStrategy::default_cost_model(); + let strategy = ASAPStrategies::default_cost_model(); - let multi = Rc::new(QueryExpr::Aggregate { + let multi = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: ReductionTy::by(vec![2]), measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&["job"])), - }); + child: metric_scan(&["job"]), + }) + .unwrap(); let target = TargetSubDAG::new(&multi); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); - let mut having_q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - if let QueryExpr::Aggregate { having, .. } = &mut having_q { - *having = Some(asap_types::pre_asap::query_expr::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true)), - ))); - } - let having_q = Rc::new(having_q); + let having_q = crate::test_support::aggregate( + ReductionTy::by(vec![2]), + vec![default_quantile(0.99)], + vec![], + Some(asap_types::ir::Predicate(ScalarExpr::Literal( + asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true), + ))), + metric_scan(&["job"]), + ); let target = TargetSubDAG::new(&having_q); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); @@ -8022,10 +8045,10 @@ mod tests { #[test] fn does_not_match_a_non_aggregate_node() { - let scan = Rc::new(metric_scan(&["job"])); + let scan = metric_scan(&["job"]); let target = TargetSubDAG::new(&scan); - assert!(!SketchAlgorithmStrategy::default_cost_model().matches(&target)); - assert!(SketchAlgorithmStrategy::default_cost_model() + assert!(!ASAPStrategies::default_cost_model().matches(&target)); + assert!(ASAPStrategies::default_cost_model() .replacements(&target) .is_empty()); } @@ -8033,11 +8056,11 @@ mod tests { #[test] fn approximate_quantile_enumerates_every_summary_candidate() { // Quantile's candidate list is [Kll, DDSketch] (summary_candidates) — - // every entry must come back as its own bound SummaryNode candidate, + // every entry must come back as its own bound summary candidate, // not just Kll (the CostModel-ranked head realizations_for_intent commits to). - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let replacements = SketchAlgorithmStrategy::default_cost_model().replacements(&target); + let replacements = ASAPStrategies::default_cost_model().replacements(&target); assert_eq!( replacements.len(), 2, @@ -8047,8 +8070,8 @@ mod tests { let kinds: Vec = replacements .iter() .map(|r| match &r.replacement { - Replacement::Summary(node) => summary_family_algorithm(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => { + Replacement::SubDag(node) => summary_family_algorithm(node), + Replacement::ExactComposition(_) => { panic!("expected a Summary replacement") } }) @@ -8063,14 +8086,14 @@ mod tests { #[test] fn cardinality_epsilon_delta_keeps_unknown_accuracy_candidates() { - let q = Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))); + let q = agg(vec![2], default_cardinality(), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let replacements = SketchAlgorithmStrategy::default_cost_model().replacements(&target); + let replacements = ASAPStrategies::default_cost_model().replacements(&target); let kinds: Vec = replacements .iter() .map(|r| match &r.replacement { - Replacement::Summary(node) => summary_family_algorithm(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => { + Replacement::SubDag(node) => summary_family_algorithm(node), + Replacement::ExactComposition(_) => { panic!("expected a Summary replacement") } }) @@ -8085,7 +8108,7 @@ mod tests { ] ); - let q = Rc::new(agg( + let q = agg( vec![2], AggIntent::Cardinality { cols: vec![], @@ -8095,13 +8118,13 @@ mod tests { }, }, metric_scan(&["job"]), - )); - let kinds: Vec<_> = SketchAlgorithmStrategy::default_cost_model() + ); + let kinds: Vec<_> = ASAPStrategies::default_cost_model() .replacements(&TargetSubDAG::new(&q)) .iter() .map(|r| match &r.replacement { - Replacement::Summary(node) => summary_family_algorithm(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => { + Replacement::SubDag(node) => summary_family_algorithm(node), + Replacement::ExactComposition(_) => { panic!("expected a Summary replacement") } }) @@ -8126,35 +8149,28 @@ mod tests { q: 0.99, accuracy: AccuracyTarget::Exact, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let replacements = SketchAlgorithmStrategy::default_cost_model().replacements(&target); + let replacements = ASAPStrategies::default_cost_model().replacements(&target); assert_eq!(replacements.len(), 1, "{replacements:?}"); assert!(matches!( &replacements[0].replacement, - Replacement::Summary(node) if matches!( - node.expr, - asap_types::post_asap::SummaryExpr::KeepPreAsap(_) - ) + Replacement::SubDag(node) if !node.contains_asap() )); assert!(replacements[0].rationale.contains("only realization")); } #[test] fn exact_mergeable_intent_yields_exactly_one_accumulator_candidate() { - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let replacements = SketchAlgorithmStrategy::default_cost_model().replacements(&target); + let replacements = ASAPStrategies::default_cost_model().replacements(&target); assert_eq!(replacements.len(), 1, "{replacements:?}"); assert!(matches!( &replacements[0].replacement, - Replacement::Summary(node) if matches!( - node.expr, - asap_types::post_asap::SummaryExpr::SummaryAgg { .. } + Replacement::SubDag(node) if matches!( + node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { .. }) ) )); } @@ -8181,15 +8197,15 @@ mod tests { #[test] fn custom_cost_model_still_enumerates_every_candidate_not_just_its_own_pick() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let custom = PreferDDSketch; - let replacements = SketchAlgorithmStrategy::new(&custom).replacements(&target); + let replacements = ASAPStrategies::new(&custom).replacements(&target); let kinds: Vec = replacements .iter() .map(|r| match &r.replacement { - Replacement::Summary(node) => summary_family_algorithm(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => { + Replacement::SubDag(node) => summary_family_algorithm(node), + Replacement::ExactComposition(_) => { panic!("expected a Summary replacement") } }) @@ -8213,9 +8229,9 @@ mod tests { // so this test injects `RankAdditiveModel` to admit the composition // and keep exercising the per-node enumeration property it is about. let inner = agg(vec![2], default_quantile(0.5), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], default_quantile(0.99), inner)); + let outer = agg(vec![], default_quantile(0.99), inner); let target = TargetSubDAG::new(&outer); - let replacements = SketchAlgorithmStrategy::new_with_planning_inputs( + let replacements = ASAPStrategies::new_with_planning_inputs( &DefaultCostModel, &RankAdditiveModel, &EqualSplitAllocator, @@ -8223,9 +8239,9 @@ mod tests { .replacements(&target); assert_eq!(replacements.len(), 2, "{replacements:?}"); - assert!(replacements - .iter() - .all(|candidate| { matches!(candidate.replacement, Replacement::Summary(_)) })); + assert!(replacements.iter().all(|candidate| { + matches!(&candidate.replacement, Replacement::SubDag(n) if n.contains_asap()) + })); // The inner target is still independently enumerated and ranked — // a custom cost model that prefers DDSketch for it is honored, and // nothing about the outer target's choice reaches it. @@ -8233,7 +8249,7 @@ mod tests { vec![("q", Rc::clone(&outer))], &default_strategies_with(&PreferDDSketchViaCostModel), ); - let QueryExpr::Aggregate { child, .. } = space.roots[0].1.as_ref() else { + let Some(NonASAPOp::Aggregate { child, .. }) = space.roots[0].1.non_asap() else { unreachable!() }; let inner_group = space @@ -8243,7 +8259,7 @@ mod tests { .candidates .iter() .filter_map(|c| match &c.replacement { - Replacement::Summary(node) => sketch_kind_of(node), + Replacement::SubDag(node) => sketch_kind_of(node), _ => None, }) .collect(); @@ -8254,54 +8270,44 @@ mod tests { ); } - /// The `SummaryFamilyType`'s committed `SketchAlgorithm`, from the top + /// The `FieldDataType`'s committed `SketchAlgorithm`, from the top /// `SummaryAgg` reachable under a (possibly `SummaryEstimate`-wrapped) /// bound root. - fn summary_family_algorithm(node: &SummaryNode) -> SketchAlgorithm { - match &node.expr { - asap_types::post_asap::SummaryExpr::SummaryEstimate { summary_input, .. } => { + fn summary_family_algorithm(node: &OperatorNode) -> SketchAlgorithm { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { summary_family_algorithm(summary_input) } - asap_types::post_asap::SummaryExpr::SummaryAgg { family, .. } => match family { - asap_types::post_asap::SummaryFamilyType::Sketch(kind, _) => { - kind.algorithm().clone() - } + Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) => match family { + asap_types::post_asap::FieldDataType::Sketch(kind, _) => kind.algorithm().clone(), other => panic!("expected a Sketch family, got {other:?}"), }, other => panic!("expected SummaryAgg/SummaryEstimate, got {other:?}"), } } - // ── SharedSubtreeStrategy ──────────────────────────────────────────── + // ── SharedSubDagStrategy ──────────────────────────────────────────── #[test] fn does_not_match_a_single_consumer_target() { - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); assert_eq!(target.consumer_count, 1); - assert!(!SharedSubtreeStrategy.matches(&target)); - assert!(SharedSubtreeStrategy.replacements(&target).is_empty()); + assert!(!SharedSubDagStrategy.matches(&target)); + assert!(SharedSubDagStrategy.replacements(&target).is_empty()); } #[test] fn two_or_more_consumers_yields_the_share_vs_independent_pair() { - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let target = TargetSubDAG::with_consumer_count(&q, 2); - assert!(SharedSubtreeStrategy.matches(&target)); + assert!(SharedSubDagStrategy.matches(&target)); - let replacements = SharedSubtreeStrategy.replacements(&target); + let replacements = SharedSubDagStrategy.replacements(&target); assert_eq!(replacements.len(), 2, "{replacements:?}"); let shared = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDag(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; assert!( @@ -8311,7 +8317,7 @@ mod tests { assert!(replacements[0].rationale.contains("build once and share")); let independent = match &replacements[1].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDag(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; assert!( @@ -8327,13 +8333,9 @@ mod tests { #[test] fn three_consumers_are_reported_verbatim_in_both_rationales() { - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let target = TargetSubDAG::with_consumer_count(&q, 3); - let replacements = SharedSubtreeStrategy.replacements(&target); + let replacements = SharedSubDagStrategy.replacements(&target); assert!(replacements[0].rationale.contains('3')); assert!(replacements[1].rationale.contains('3')); } @@ -8341,13 +8343,13 @@ mod tests { /// Builds realistic multi-consumer `TargetSubDAG`s the same way this /// module's own [`discover_targets`]/`walk` does: dedup by `Rc::as_ptr`, /// walking only the relational-skeleton operator children - /// `asap_types::pre_asap::cse::share_common_subtrees` itself scopes to, + /// `asap_types::ir::cse::share_common_subdags` itself scopes to, /// so a shared node nested below another shared node is only ever /// counted at the highest (maximal) point sharing starts. Test-only: /// this module deliberately does not ship a workload-wide discovery /// pass of its own (see the module docs' "Non-goals"). - fn count_consumers(roots: &[Rc]) -> HashMap<*const QueryExpr, usize> { - fn walk(node: &Rc, counts: &mut HashMap<*const QueryExpr, usize>) { + fn count_consumers(roots: &[Rc]) -> HashMap<*const OperatorNode, usize> { + fn walk(node: &Rc, counts: &mut HashMap<*const OperatorNode, usize>) { let ptr = Rc::as_ptr(node); let already_visited = counts.contains_key(&ptr); *counts.entry(ptr).or_insert(0) += 1; @@ -8355,50 +8357,15 @@ mod tests { walk_children(node, counts); } } - fn walk_children(node: &QueryExpr, counts: &mut HashMap<*const QueryExpr, usize>) { - use QueryExpr::*; - match node { - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => walk(c, counts), - PromqlRelabel { child, .. } - | PromqlInfoEnrich { child, .. } - | PromqlSeriesSample { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | Sort { child, .. } - | Limit { child, .. } => walk(child, counts), - Concat { children, .. } => { - for c in children { - walk_children(c, counts); - } - } - Join { left, right, .. } | SetOp { left, right, .. } => { - walk(left, counts); - walk(right, counts); + fn walk_children(node: &OperatorNode, counts: &mut HashMap<*const OperatorNode, usize>) { + if let Some(NonASAPOp::Concat { children, .. }) = node.non_asap() { + for c in children { + walk_children(c, counts); } - BinaryOp { lhs, rhs, .. } => { - walk(lhs, counts); - walk(rhs, counts); - } - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} + return; + } + for child in node.children() { + walk(child, counts); } } @@ -8411,25 +8378,25 @@ mod tests { #[test] fn realistic_cse_output_produces_a_two_consumer_target() { - // Two workload roots that `share_common_subtrees` collapses onto one + // Two workload roots that `share_common_subdags` collapses onto one // Rc (mirrors `explanation`'s and `cse`'s own fixtures): a grouped // Sum aggregate over the same scan, built independently at each root. let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let shared = asap_types::pre_asap::cse::share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = asap_types::ir::cse::share_common_subdags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; assert!(Rc::ptr_eq(ra, rb), "fixture sanity: the two roots merged"); - let roots: Vec> = shared.into_iter().map(|(_, rc)| rc).collect(); + let roots: Vec> = shared.into_iter().map(|(_, rc)| rc).collect(); let counts = count_consumers(&roots); let count = counts[&Rc::as_ptr(&roots[0])]; assert_eq!(count, 2); let target = TargetSubDAG::with_consumer_count(&roots[0], count); - assert!(SharedSubtreeStrategy.matches(&target)); - assert_eq!(SharedSubtreeStrategy.replacements(&target).len(), 2); + assert!(SharedSubDagStrategy.matches(&target)); + assert_eq!(SharedSubDagStrategy.replacements(&target).len(), 2); } // ── search_workload / CandidateLogicalASAPDAGs / TargetSubDAGCandidates (merged from search.rs) ── @@ -8450,7 +8417,7 @@ mod tests { delta: 0.01, }, }; - let root = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let root = agg(vec![2], intent, metric_scan(&["job"])); let space = search_workload(vec![("q", root)]); // One group for the Aggregate, one for its Scan child. @@ -8458,7 +8425,7 @@ mod tests { let agg_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("an Aggregate group must be discovered"); assert_eq!(agg_group.consumer_count, 1); assert_eq!( @@ -8470,24 +8437,26 @@ mod tests { assert!(agg_group .candidates .iter() - .all(|c| matches!(c.replacement, Replacement::Summary(_)))); + .all(|c| matches!(&c.replacement, Replacement::SubDag(n) if n.contains_asap()))); assert_eq!( agg_group .candidates .iter() .filter(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDag(node) = &candidate.replacement else { return false; }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = + &node.operator + else { return false; }; matches!( - &summary_input.expr, - SummaryExpr::SummaryAgg { + &summary_input.operator, + Operator::ASAP(ASAPOp::SummaryAgg { grouping: GroupingStrategy::SharedMultiSubpopulation { .. }, .. - } + }) ) }) .count(), @@ -8511,7 +8480,7 @@ mod tests { let scan_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Scan { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Scan { .. }))) .expect("a Scan group must be discovered"); assert_eq!(scan_group.consumer_count, 1); assert!( @@ -8522,20 +8491,20 @@ mod tests { #[test] fn cardinality_group_keeps_all_four_candidates() { - let root = Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))); + let root = agg(vec![2], default_cardinality(), metric_scan(&["job"])); let space = search_workload(vec![("q", root)]); let agg_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .unwrap(); assert_eq!(agg_group.candidates.len(), 4); assert!(agg_group.candidates.iter().any(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) if node.guarantee.is_none() + Replacement::SubDag(node) if node.guarantee.is_none() && candidate.has_missing_accuracy_evidence() ))); - let root = Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))); + let root = agg(vec![2], default_cardinality(), metric_scan(&["job"])); let targeted = search_workload_with_targets( vec![( "q", @@ -8556,7 +8525,7 @@ mod tests { .iter() .any(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) if node.guarantee.is_none() + Replacement::SubDag(node) if node.guarantee.is_none() && candidate.has_missing_accuracy_evidence() ))); assert!(!targeted @@ -8569,7 +8538,7 @@ mod tests { let exact_target = search_workload_with_targets( vec![( "q", - Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))), + agg(vec![2], default_cardinality(), metric_scan(&["job"])), Some(AccuracyTarget::Exact), )], &default_strategies(), @@ -8586,13 +8555,13 @@ mod tests { #[test] fn shared_aggregate_across_two_roots_gets_both_strategies_candidates() { // Two independently-built, structurally identical Sum aggregates: - // share_common_subtrees (run inside search_workload) collapses them + // share_common_subdags (run inside search_workload) collapses them // onto one Rc with consumer_count 2, so this single group should - // carry SketchAlgorithmStrategy's one ExactAggregate candidate *and* - // SharedSubtreeStrategy's share-vs-recompute pair. + // carry ASAPStrategies's one ExactAggregate candidate *and* + // SharedSubDagStrategy's share-vs-recompute pair. let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); + let space = search_workload(vec![("a", a), ("b", b)]); // roots[0] and roots[1] must have merged onto the same Rc. assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); @@ -8606,15 +8575,17 @@ mod tests { group.candidates ); + // Old `Replacement::Summary` ↔ a `Subtree` containing an ASAP node; + // old `Replacement::Rewrite` ↔ a pure pre-ASAP `Subtree`. let summary_count = group .candidates .iter() - .filter(|c| matches!(c.replacement, Replacement::Summary(_))) + .filter(|c| matches!(&c.replacement, Replacement::SubDag(n) if n.contains_asap())) .count(); let rewrite_count = group .candidates .iter() - .filter(|c| matches!(c.replacement, Replacement::Rewrite(_))) + .filter(|c| matches!(&c.replacement, Replacement::SubDag(n) if !n.contains_asap())) .count(); assert_eq!(summary_count, 1); assert_eq!(rewrite_count, 2); @@ -8623,11 +8594,11 @@ mod tests { // (the "false-positive dedup" failure mode `is_duplicate_rewrite` // exists to prevent). let one_is_the_target = group.candidates.iter().any( - |c| matches!(&c.replacement, Replacement::Rewrite(rc) if Rc::ptr_eq(rc, &group.target)), + |c| matches!(&c.replacement, Replacement::SubDag(rc) if Rc::ptr_eq(rc, &group.target)), + ); + let one_is_not = group.candidates.iter().any( + |c| matches!(&c.replacement, Replacement::SubDag(rc) if !Rc::ptr_eq(rc, &group.target)), ); - let one_is_not = group.candidates.iter().any(|c| { - matches!(&c.replacement, Replacement::Rewrite(rc) if !Rc::ptr_eq(rc, &group.target)) - }); assert!(one_is_the_target && one_is_not); } @@ -8638,51 +8609,49 @@ mod tests { // walking the whole DAG, not just root-level pointer identity // (a naive whole-root-only consumer-count pass would miss this; // this module's discover_targets must not). + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - let shared = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); // Different predicates so the two Filter *parents* stay distinct // (don't themselves merge under CSE) — only their shared `child` // should collapse onto one `Rc`. - let root_a = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(1)))), + let root_a = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(1))), child: Rc::clone(&shared), - }; - let root_b = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(2)))), + }) + .unwrap(); + let root_b = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(2))), child: Rc::clone(&shared), - }; + }) + .unwrap(); - let space = search_workload(vec![("a", Rc::new(root_a)), ("b", Rc::new(root_b))]); + let space = search_workload(vec![("a", root_a), ("b", root_b)]); assert_eq!( space.len(), 4, "2 distinct Filters + 1 shared Aggregate + 1 shared Scan" ); - // `share_common_subtrees` re-clones+re-interns anything that already + // `share_common_subdags` re-clones+re-interns anything that already // had more than one owner going in (see `cse.rs`'s own doc on // `intern_child`'s clone-fallback path) — so the post-CSE shared // node is a *fresh* Rc, structurally equal to (but not the same // pointer as) the pre-search `shared` variable. Recover it from the // post-CSE root's own `child` field instead of the stale `shared` // handle. - let QueryExpr::Filter { + let Some(NonASAPOp::Filter { child: post_cse_shared_a, .. - } = space.roots[0].1.as_ref() + }) = space.roots[0].1.non_asap() else { panic!("expected a Filter root"); }; - let QueryExpr::Filter { + let Some(NonASAPOp::Filter { child: post_cse_shared_b, .. - } = space.roots[1].1.as_ref() + }) = space.roots[1].1.non_asap() else { panic!("expected a Filter root"); }; @@ -8696,7 +8665,7 @@ mod tests { .expect("shared node must be a discovered target"); assert_eq!(group.consumer_count, 2); assert!( - SharedSubtreeStrategy.matches(&TargetSubDAG::with_consumer_count( + SharedSubDagStrategy.matches(&TargetSubDAG::with_consumer_count( post_cse_shared, group.consumer_count )) @@ -8707,19 +8676,15 @@ mod tests { #[test] fn add_candidate_rejects_a_true_rewrite_duplicate() { - // SharedSubtreeStrategy's `Replacement::Rewrite` candidates are - // real `QueryExpr` values with `PartialEq`, so `add_candidate` can + // SharedSubDagStrategy's `Replacement::Rewrite` candidates are + // real `OperatorNode` values with `PartialEq`, so `add_candidate` can // (and must) actually reject a genuine repeat — unlike the // `Replacement::Summary` case (see the test below). - let root = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let root = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let mut group = TargetSubDAGCandidates::new(Rc::clone(&root), 2); let target = TargetSubDAG::with_consumer_count(&root, 2); let mut inserted = 0; - for candidate in SharedSubtreeStrategy.replacements(&target) { + for candidate in SharedSubDagStrategy.replacements(&target) { if group.add_candidate(candidate) { inserted += 1; } @@ -8732,7 +8697,7 @@ mod tests { // structurally identical value, both already covered by // `is_duplicate_rewrite`. let mut re_inserted = 0; - for candidate in SharedSubtreeStrategy.replacements(&target) { + for candidate in SharedSubDagStrategy.replacements(&target) { if group.add_candidate(candidate) { re_inserted += 1; } @@ -8746,8 +8711,8 @@ mod tests { #[test] fn add_candidate_never_dedups_summary_candidates() { - // Documented, deliberate consequence of `SummaryNode` deriving no - // `PartialEq` (see `is_duplicate_summary`'s own doc): re-proposing + // Documented, deliberate consequence of `is_duplicate_summary` + // refusing value equality on `f64`-bearing summaries: re-proposing // the same `Replacement::Summary` candidates DOES grow the group — // this module refuses to guess at an equality check it can't back // with a real `PartialEq`. `search_workload_with` never actually @@ -8755,9 +8720,9 @@ mod tests { // module docs' "Termination" section), so this test exists to pin // the documented behavior, not to endorse calling `replacements` // twice for the same target. - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let mut group = TargetSubDAGCandidates::new(Rc::clone(&root), 1); - let strategy = SketchAlgorithmStrategy::default_cost_model(); + let strategy = ASAPStrategies::default_cost_model(); let target = TargetSubDAG::new(&root); for candidate in strategy.replacements(&target) { group.add_candidate(candidate); @@ -8776,11 +8741,7 @@ mod tests { #[test] fn is_duplicate_rewrite_never_merges_share_with_recompute() { - let target = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let target = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let share = Rc::clone(&target); let recompute = Rc::new((*target).clone()); assert!(!Rc::ptr_eq(&share, &recompute)); @@ -8794,11 +8755,7 @@ mod tests { #[test] fn is_duplicate_rewrite_catches_a_real_repeat() { - let target = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let target = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let first_recompute = Rc::new((*target).clone()); let second_recompute = Rc::new((*target).clone()); assert!(!Rc::ptr_eq(&first_recompute, &second_recompute)); @@ -8819,7 +8776,7 @@ mod tests { let mut roots = Vec::new(); let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); for i in 0..20 { - roots.push((i, Rc::new(shared.clone()))); + roots.push((i, Rc::new((*shared).clone()))); } let space = search_workload(roots); let group = space.candidates_for_target(&space.roots[0].1).unwrap(); @@ -8832,18 +8789,18 @@ mod tests { .unwrap(); assert!(matches!( &ranked_group.candidates[0].replacement, - Replacement::Rewrite(rc) if Rc::ptr_eq(rc, &group.target) + Replacement::SubDag(rc) if Rc::ptr_eq(rc, &group.target) )); let rewrites: Vec<&ReplacementSubDAG> = ranked_group .candidates .iter() - .filter(|c| matches!(c.replacement, Replacement::Rewrite(_))) + .filter(|c| matches!(&c.replacement, Replacement::SubDag(n) if !n.contains_asap())) .copied() .collect(); assert_eq!(rewrites.len(), 2); let first_shares_target = match &rewrites[0].replacement { - Replacement::Rewrite(rc) => Rc::ptr_eq(rc, &group.target), - Replacement::Summary(_) | Replacement::ExactComposition(_) => false, + Replacement::SubDag(rc) => Rc::ptr_eq(rc, &group.target), + Replacement::ExactComposition(_) => false, }; assert!( first_shares_target, @@ -8869,17 +8826,17 @@ mod tests { } } - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let space = search_workload(vec![("q", root)]); let ranked = space.cost_sorted(&PreferDDSketch); let agg_group = ranked .iter() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .unwrap(); assert_eq!(agg_group.candidates.len(), 2); let first_kind = match &agg_group.candidates[0].replacement { - Replacement::Summary(node) => sketch_kind_of(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => None, + Replacement::SubDag(node) => sketch_kind_of(node), + Replacement::ExactComposition(_) => None, }; assert_eq!(first_kind, Some(SketchAlgorithm::DDSketch)); } @@ -8897,7 +8854,7 @@ mod tests { candidates.to_vec() } - fn estimated_subpopulation_count(&self, _target: &QueryExpr) -> Option { + fn estimated_subpopulation_count(&self, _target: &OperatorNode) -> Option { Some(self.0) } } @@ -8910,19 +8867,15 @@ mod tests { delta: 0.01, }, }; - let root = Rc::new(agg( - vec![2, 3], - intent, - metric_scan(&["tenant_id", "endpoint"]), - )); + let root = agg(vec![2, 3], intent, metric_scan(&["tenant_id", "endpoint"])); let strategies = default_strategies_with(&model); let space = search_workload_with(vec![("tenant_endpoint_count", root)], &strategies); let ranked = space.cost_sorted(&model); let aggregate = ranked .iter() - .find(|group| matches!(group.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|group| matches!(group.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("aggregate group"); - let Replacement::Summary(node) = &aggregate.candidates[0].replacement else { + let Replacement::SubDag(node) = &aggregate.candidates[0].replacement else { panic!("grouping candidate must be a summary") }; summary_grouping(node) @@ -8946,12 +8899,12 @@ mod tests { /// and target produces, not some other (or stale) number. #[test] fn cost_sorted_pairs_each_candidate_with_its_own_estimate_cost() { - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let space = search_workload(vec![("q", root)]); let ranked = space.cost_sorted(&DefaultCostModel); let agg_group = ranked .iter() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .unwrap(); assert_eq!( agg_group.costs.len(), @@ -8973,9 +8926,9 @@ mod tests { // ── global_selection (issue #271) ─────────────────────────────────── - /// A `CostModel` with a constant, `subtree`-independent recompute cost + /// A `CostModel` with a constant, `sub-DAG`-independent recompute cost /// and shared-maintenance cost, chosen (40 recompute-per-use, 100 - /// maintenance) so that a `SharedSubtreeStrategy` group's + /// maintenance) so that a `SharedSubDagStrategy` group's /// `cse_share_decision` flips exactly between a consumer count of 2 /// (recompute total 80, below maintenance: `RecomputeIndependently`) /// and a consumer count of 3 (recompute total 120, above @@ -9030,7 +8983,7 @@ mod tests { } } - let aggregate = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let aggregate = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let space = search_workload(vec![("left", Rc::clone(&aggregate)), ("right", aggregate)]); let root = &space.roots[0].1; assert!(cse_candidate_pair(space.candidates_for_target(root).unwrap()).is_some()); @@ -9045,7 +8998,7 @@ mod tests { // must equal the group's own raw consumer_count, and its `chosen` // candidate must be cost_sorted's top pick, for both the sketch // group and its child Scan. - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let space = search_workload(vec![("q", root)]); let ranked = space.cost_sorted(&DefaultCostModel); @@ -9072,12 +9025,12 @@ mod tests { // A bare Scan: no registered strategy has an opinion on it, so it // gets a group with an empty candidate list (see TargetSubDAGCandidates's own // doc) — global_selection must not invent a candidate for it. - let root = Rc::new(metric_scan(&["job"])); + let root = metric_scan(&["job"]); let space = search_workload(vec![("q", root)]); let selected = space.global_selection(&DefaultCostModel); let scan_group = selected .target_selections() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Scan { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Scan { .. }))) .unwrap(); assert!(scan_group.chosen.is_none()); assert_eq!(scan_group.effective_consumer_count, 1); @@ -9085,7 +9038,7 @@ mod tests { #[test] fn global_selection_falls_back_to_local_ranking_for_sketch_family_groups() { - // SketchAlgorithmStrategy groups have no cross-group-aware cost hook + // ASAPStrategies groups have no cross-group-aware cost hook // (rank_candidates takes no consumer_count) — global_selection must // still return cost_sorted's own top pick for them (documented in // the module docs' "Whole-plan (cross-group) selection" section), @@ -9110,16 +9063,16 @@ mod tests { } } - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let space = search_workload(vec![("q", root)]); let selected = space.global_selection(&PreferDDSketch); let agg_group = selected .target_selections() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .unwrap(); let kind = match &agg_group.chosen.unwrap().replacement { - Replacement::Summary(node) => sketch_kind_of(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => None, + Replacement::SubDag(node) => sketch_kind_of(node), + Replacement::ExactComposition(_) => None, }; assert_eq!(kind, Some(SketchAlgorithm::DDSketch)); @@ -9143,24 +9096,29 @@ mod tests { #[test] fn mixed_rewrite_group_keeps_and_selects_its_explicit_cse_pair() { - let target = Rc::new(metric_scan(&["job"])); + let target = metric_scan(&["job"]); let mut group = TargetSubDAGCandidates::new(Rc::clone(&target), 2); group.candidates = vec![ ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::clone(&target)), + replacement: Replacement::SubDag(Rc::clone(&target)), provenance: ReplacementProvenance::CseShare, rationale: "share".into(), }, ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::new(target.as_ref().clone())), + replacement: Replacement::SubDag(Rc::new(target.as_ref().clone())), provenance: ReplacementProvenance::CseRecompute, rationale: "recompute".into(), }, ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::new(QueryExpr::CurrentTimestamp)), + replacement: Replacement::SubDag( + OperatorNode::non_asap_node(NonASAPOp::PromqlVectorFromScalar( + ScalarExpr::EvalTimestamp, + )) + .unwrap(), + ), provenance: ReplacementProvenance::LogicalRewrite, rationale: "different rewrite strategy".into(), }, @@ -9186,9 +9144,9 @@ mod tests { #[test] fn effective_consumer_count_corrects_a_nested_groups_share_decision() { - // The interaction issue #271 describes: an outer shared subtree `a` + // The interaction issue #271 describes: an outer shared sub-DAG `a` // (referenced by 2 roots, so consumer_count == 2) wraps an inner - // shared subtree `c` (referenced once through `a`'s own child edge, + // shared sub-DAG `c` (referenced once through `a`'s own child edge, // plus once more directly by a third, separate root — so `c`'s own // *raw* structural consumer_count is also 2, independent of `a`). // @@ -9199,11 +9157,11 @@ mod tests { // // `a` and `c` are both non-`Aggregate` nodes (`Filter`/`Dedup`) so // neither is bindable — each group is a *clean* two-candidate - // SharedSubtreeStrategy share-vs-recompute pair, with no - // SketchAlgorithmStrategy `Summary` candidate mixed in to complicate + // SharedSubDagStrategy share-vs-recompute pair, with no + // ASAPStrategies `Summary` candidate mixed in to complicate // ranking (see `shared_aggregate_across_two_roots_gets_both_strategies_candidates` // for what a *mixed*-shape group looks like — deliberately avoided - // here to isolate the SharedSubtreeStrategy-only interaction). + // here to isolate the SharedSubDagStrategy-only interaction). // // Under ConstantCseCost, consumer_count == 2 loses to maintenance // (2 * 40 = 80 < 100 ⇒ RecomputeIndependently); consumer_count == 3 wins @@ -9216,31 +9174,33 @@ mod tests { // which flips its own decision to Share. Only global_selection, // which folds `a`'s decision into `c`'s effective_consumer_count // before deciding `c`, gets this right. + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - let c = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["job"])), + let c = || { + OperatorNode::non_asap_node(NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["job"]), + }) + .unwrap() }; - let a = || QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(c()), + let a = || { + OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: c(), + }) + .unwrap() }; - let space = search_workload(vec![ - ("root1", Rc::new(a())), - ("root2", Rc::new(a())), - ("root3", Rc::new(c())), - ]); + let space = search_workload(vec![("root1", a()), ("root2", a()), ("root3", c())]); // Fixture sanity: root1/root2 merged onto one shared `a`, and `c` // (root1/root2's shared child, and root3 itself) merged onto one // shared `c` with raw consumer_count 2, and both groups are clean - // (non-mixed) two-candidate SharedSubtreeStrategy pairs. + // (non-mixed) two-candidate SharedSubDagStrategy pairs. assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); let a_rc = &space.roots[0].1; - let QueryExpr::Filter { child: c_via_a, .. } = a_rc.as_ref() else { + let Some(NonASAPOp::Filter { child: c_via_a, .. }) = a_rc.non_asap() else { panic!("expected root1/root2 to still be a Filter"); }; assert!(Rc::ptr_eq(c_via_a, &space.roots[2].1)); @@ -9274,7 +9234,7 @@ mod tests { .unwrap(); let c_top_shares = matches!( &c_ranked.candidates[0].replacement, - Replacement::Rewrite(rc) if Rc::ptr_eq(rc, c_via_a) + Replacement::SubDag(rc) if Rc::ptr_eq(rc, c_via_a) ); assert!( !c_top_shares, @@ -9295,7 +9255,7 @@ mod tests { ); let a_shares = matches!( &a_selected.chosen.unwrap().replacement, - Replacement::Rewrite(rc) if Rc::ptr_eq(rc, a_rc) + Replacement::SubDag(rc) if Rc::ptr_eq(rc, a_rc) ); assert!( !a_shares, @@ -9308,7 +9268,7 @@ mod tests { ); let c_shares = matches!( &c_selected.chosen.unwrap().replacement, - Replacement::Rewrite(rc) if Rc::ptr_eq(rc, c_via_a) + Replacement::SubDag(rc) if Rc::ptr_eq(rc, c_via_a) ); assert!( c_shares, @@ -9346,10 +9306,11 @@ mod tests { } } - let shared = Rc::new(QueryExpr::Dedup { + let shared = OperatorNode::non_asap_node(NonASAPOp::Dedup { cols: vec![0], - child: Rc::new(metric_scan(&["job"])), - }); + child: metric_scan(&["job"]), + }) + .unwrap(); let space = search_workload(vec![ ("left", Rc::clone(&shared)), ("right", Rc::clone(&shared)), @@ -9362,20 +9323,26 @@ mod tests { #[test] fn effective_repetition_materializes_a_cse_choice_for_a_single_edge_child() { + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - let c = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["job"])), + let c = || { + OperatorNode::non_asap_node(NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["job"]), + }) + .unwrap() }; - let a = || QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(c()), + let a = || { + OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: c(), + }) + .unwrap() }; - let space = search_workload(vec![("root1", Rc::new(a())), ("root2", Rc::new(a()))]); + let space = search_workload(vec![("root1", a()), ("root2", a())]); let a_rc = &space.roots[0].1; - let QueryExpr::Filter { child: c_rc, .. } = a_rc.as_ref() else { + let Some(NonASAPOp::Filter { child: c_rc, .. }) = a_rc.non_asap() else { panic!("expected Filter root"); }; @@ -9390,8 +9357,8 @@ mod tests { #[test] fn shared_ancestor_keeps_a_single_use_cse_descendant_selected() { + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; struct AlwaysShare; impl CostModel for AlwaysShare { @@ -9412,22 +9379,25 @@ mod tests { } } - let child = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["job"])), + let child = || { + OperatorNode::non_asap_node(NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["job"]), + }) + .unwrap() }; - let parent = || QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(child()), + let parent = || { + OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: child(), + }) + .unwrap() }; - let space = search_workload(vec![ - ("root1", Rc::new(parent())), - ("root2", Rc::new(parent())), - ]); + let space = search_workload(vec![("root1", parent()), ("root2", parent())]); let parent_rc = &space.roots[0].1; - let QueryExpr::Filter { + let Some(NonASAPOp::Filter { child: child_rc, .. - } = parent_rc.as_ref() + }) = parent_rc.non_asap() else { panic!("expected Filter root"); }; @@ -9451,38 +9421,42 @@ mod tests { #[test] fn global_selection_propagates_uses_through_the_selected_rewrite() { + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; struct ReplaceFilterChild; impl ReplacementStrategy for ReplaceFilterChild { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { - matches!(target.root.as_ref(), QueryExpr::Filter { .. }) + matches!(target.root.non_asap(), Some(NonASAPOp::Filter { .. })) } fn replacements(&self, _target: &TargetSubDAG<'_>) -> Vec { vec![ReplacementSubDAG { strategy: "ReplaceFilterChild", - replacement: Replacement::Rewrite(Rc::new(QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["replacement"])), - })), + replacement: Replacement::SubDag( + OperatorNode::non_asap_node(NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["replacement"]), + }) + .unwrap(), + ), provenance: ReplacementProvenance::LogicalRewrite, rationale: "replace the Filter and its input".into(), }] } } - let original_child = Rc::new(metric_scan(&["original"])); - let root = Rc::new(QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + let original_child = metric_scan(&["original"]); + let root = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), child: Rc::clone(&original_child), - }); + }) + .unwrap(); let strategies: Vec> = vec![Box::new(ReplaceFilterChild)]; let space = search_workload_with(vec![("q", root)], &strategies); let root = &space.roots[0].1; let selected = space.global_selection(&DefaultCostModel); - let Replacement::Rewrite(rewrite) = &selected + let Replacement::SubDag(rewrite) = &selected .for_target(root) .unwrap() .chosen @@ -9491,17 +9465,17 @@ mod tests { else { panic!("expected logical rewrite"); }; - let QueryExpr::Dedup { + let Some(NonASAPOp::Dedup { child: replacement_child, .. - } = rewrite.as_ref() + }) = rewrite.non_asap() else { panic!("expected Dedup rewrite"); }; - let QueryExpr::Filter { + let Some(NonASAPOp::Filter { child: original_child, .. - } = root.as_ref() + }) = root.non_asap() else { panic!("expected Filter root"); }; @@ -9537,11 +9511,11 @@ mod tests { fn candidate_cost(&self, _: &ReplacementSubDAG, _: &TargetSubDAG<'_>) -> Option { Some(Cost(1.0)) } - fn summary_support_evidence(&self, _: &SummaryNode) -> Option { + fn summary_support_evidence(&self, _: &OperatorNode) -> Option { Some(false) } } - let root = Rc::new(lower_promql("sum_over_time(a[1m])", AccuracyTarget::Exact)); + let root = lower_promql("sum_over_time(a[1m])", AccuracyTarget::Exact); let space = search_workload(vec![("q", root)]); let selected = space.global_selection(&Unsupported); assert!(selected @@ -9554,18 +9528,15 @@ mod tests { // Composable temporal/grouped Sum must be executable as one producer. #[test] fn grouped_temporal_sum_has_one_summary_producer_candidate() { - let root = Rc::new(lower_promql( - "sum by(job)(sum_over_time(a[1m]))", - AccuracyTarget::Exact, - )); + let root = lower_promql("sum by(job)(sum_over_time(a[1m]))", AccuracyTarget::Exact); let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); assert!(candidates .iter() .any(|candidate| matches!(&candidate.replacement, - Replacement::Summary(node) if matches!(&node.expr, - SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))))); + Replacement::SubDag(node) if matches!(&node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { reduction: Reduction::Reduce(_), child, .. }) + if !child.contains_asap())))); struct PreferComposed; impl CostModel for PreferComposed { fn rank_candidates( @@ -9582,9 +9553,9 @@ mod tests { ) -> Option { Some(Cost( if matches!(&candidate.replacement, - Replacement::Summary(node) if matches!(&node.expr, - SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))) + Replacement::SubDag(node) if matches!(&node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { reduction: Reduction::Reduce(_), child, .. }) + if !child.contains_asap())) { 1.0 } else { @@ -9596,9 +9567,9 @@ mod tests { let space = search_workload(vec![("q", root.clone())]); let selected = space.global_selection(&PreferComposed); let node = selected.assemble_target(&space.roots[0].1).unwrap(); - assert!(matches!(&node.expr, - SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))); + assert!(matches!(&node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { reduction: Reduction::Reduce(_), child, .. }) + if !child.contains_asap())); } // Mixed candidate ranking must honor explicit costs, not legacy estimates. @@ -9627,10 +9598,7 @@ mod tests { )) } } - let root = Rc::new(lower_promql( - "sum by(job)(sum_over_time(a[1m]))", - AccuracyTarget::Exact, - )); + let root = lower_promql("sum by(job)(sum_over_time(a[1m]))", AccuracyTarget::Exact); let space = search_workload(vec![("q", root)]); let selection = space.global_selection(&ExplicitCosts); let selected = selection @@ -9666,16 +9634,8 @@ mod tests { } } - let a = Rc::new(agg( - vec![2], - AggIntent::Avg { col: None }, - metric_scan(&["job"]), - )); - let b = Rc::new(agg( - vec![2], - AggIntent::Avg { col: None }, - metric_scan(&["job"]), - )); + let a = agg(vec![2], AggIntent::Avg { col: None }, metric_scan(&["job"])); + let b = agg(vec![2], AggIntent::Avg { col: None }, metric_scan(&["job"])); let space = search_workload(vec![("a", a), ("b", b)]); let root = &space.roots[0].1; let selected = space.global_selection(&PreferLogicalRewrite); @@ -9691,26 +9651,28 @@ mod tests { #[test] fn topological_order_puts_a_later_discovered_parent_before_its_child() { - // Mirrors nested_shared_subtree_below_an_unshared_parent_is_still_discovered's + // Mirrors nested_shared_sub-DAG_below_an_unshared_parent_is_still_discovered's // diamond fixture: discover_targets's own `order` visits root_b (a // parent of `shared`) *after* `shared` itself, because `shared` was // already fully walked via root_a first. A naive "process // discover_targets's own order" DP would see root_b's child edge // after already processing `shared` — topological_order must not // make that mistake. + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let root_a = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(1)))), - child: Rc::new(shared.clone()), - }; - let root_b = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(2)))), - child: Rc::new(shared), - }; - let roots = vec![("a", Rc::new(root_a)), ("b", Rc::new(root_b))]; + let root_a = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(1))), + child: shared.clone(), + }) + .unwrap(); + let root_b = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(2))), + child: shared, + }) + .unwrap(); + let roots = vec![("a", root_a), ("b", root_b)]; let mut order = Vec::new(); let mut nodes = HashMap::new(); @@ -9736,10 +9698,10 @@ mod tests { // Discovery-order sanity: root_b comes after the shared child in // discover_targets's own order (the exact non-topological case this // test exists to cover). - let QueryExpr::Filter { + let Some(NonASAPOp::Filter { child: shared_via_a, .. - } = space.roots[0].1.as_ref() + }) = space.roots[0].1.non_asap() else { panic!("expected a Filter root"); }; @@ -9772,7 +9734,7 @@ mod tests { // module docs — so this always converges in exactly 2 passes). let a = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let b = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); + let space = search_workload(vec![("a", a), ("b", b)]); assert!(!space.is_empty()); } @@ -9798,19 +9760,21 @@ mod tests { fn replacements(&self, target: &TargetSubDAG<'_>) -> Vec { let n = self.next.get(); self.next.set(n + 1); + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - let fresh_inner_layer = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(n)))), + let fresh_inner_layer = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(n))), child: Rc::clone(target.root), - }; - let outer_wrapper = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(fresh_inner_layer), - }; + }) + .unwrap(); + let outer_wrapper = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: fresh_inner_layer, + }) + .unwrap(); vec![ReplacementSubDAG { strategy: "AlwaysGrowingStrategy", - replacement: Replacement::Rewrite(Rc::new(outer_wrapper)), + replacement: Replacement::SubDag(outer_wrapper), provenance: ReplacementProvenance::LogicalRewrite, rationale: format!("pathological candidate #{n}"), }] @@ -9820,13 +9784,13 @@ mod tests { #[test] #[should_panic(expected = "did not converge")] fn a_pathologically_growing_strategy_trips_the_iteration_cap() { - let root = Rc::new(metric_scan(&["job"])); + let root = metric_scan(&["job"]); let strategies: Vec> = vec![Box::new(AlwaysGrowingStrategy { next: std::cell::Cell::new(0), })]; let _ = search_workload_with(vec![("q", root)], &strategies); } - // ── realize_child / keep_pre_asap: end-to-end single-target realization ── + // ── realize_child / retain_exact: end-to-end single-target realization ── // // Moved from the former `bind.rs` (issue #251): `bind.rs`'s own // workload-wide orchestration (`implement_workload`/ @@ -9839,18 +9803,7 @@ mod tests { // pattern by hand since `realize_child` is `pub(crate)`), these tests // call `realize_child` directly. - fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: ReductionTy::PerEntity, - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } - - fn field<'a>(schema: &'a SummarySchema, name: &str) -> &'a SummaryField { + fn field<'a>(schema: &'a Schema, name: &str) -> &'a Field { schema .fields .iter() @@ -9859,54 +9812,54 @@ mod tests { } fn realize_first( - expr: &QueryExpr, + expr: &OperatorNode, cost_model: &dyn CostModel, - ) -> Result, RealizationError> { + ) -> Result, RealizationError> { realize_child(&Rc::new(expr.clone()), cost_model) } - fn realize(expr: &QueryExpr) -> Result, RealizationError> { + fn realize(expr: &OperatorNode) -> Result, RealizationError> { realize_first(expr, &DefaultCostModel) } #[test] fn quantile_realizes_kll_wrapped_in_estimate() { // quantile by (job) (m) at ε=0.01 → Estimate(Quantile) over - // SummaryAgg(Kll{k:269}) over KeepPreAsap(Scan). job = col 2. + // SummaryAgg(Kll{k:269}) over the kept Scan. job = col 2. let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; - assert!(matches!(query, PostAsapSketchQuery::Quantile { q } if *q == 0.99)); + assert!(matches!(query, PostAsapSketchStatistic::Quantile { q } if *q == 0.99)); // Estimate edge: plain row shape — group key + Float64 answer. assert_eq!( field(&root.schema, "quantile_0_99").dtype, - SummaryFamilyType::Plain(DataType::Float64) + FieldDataType::Plain(DataType::Float64) ); assert_eq!( field(&root.schema, "job").dtype, - SummaryFamilyType::Plain(DataType::Utf8) + FieldDataType::Plain(DataType::Utf8) ); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, reduction, .. - } = &summary_input.expr + }) = &summary_input.operator else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) @@ -9917,13 +9870,14 @@ mod tests { // the committed family. assert_eq!( field(&summary_input.schema, "value").dtype, - SummaryFamilyType::Sketch( + FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) ); - assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(ref e) - if matches!(**e, QueryExpr::Scan { .. }))); + // The kept pre-ASAP leaf is the Scan node itself (no wrapper). + assert!(matches!(child.non_asap(), Some(NonASAPOp::Scan { .. }))); + assert!(!child.contains_asap()); } /// A deployment-supplied [`CostModel`] can override the default KLL @@ -9953,28 +9907,36 @@ mod tests { // Default: KLL (see `quantile_realizes_kll_wrapped_in_estimate` above). let default_root = realize(&q).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &default_root.expr else { - panic!("expected SummaryEstimate root, got {:?}", default_root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &default_root.operator + else { + panic!( + "expected SummaryEstimate root, got {:?}", + default_root.operator + ); }; - let SummaryExpr::SummaryAgg { family, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert!(matches!( family, - SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll + FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll )); // With `PreferDDSketchViaCostModel`: DDSketch instead, same query. let custom_root = realize_first(&q, &PreferDDSketchViaCostModel).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &custom_root.expr else { - panic!("expected SummaryEstimate root, got {:?}", custom_root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &custom_root.operator + else { + panic!( + "expected SummaryEstimate root, got {:?}", + custom_root.operator + ); }; - let SummaryExpr::SummaryAgg { family, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::DDSketch, SketchParams::DDSketch { alpha: 0.01 } @@ -9987,8 +9949,8 @@ mod tests { /// A deployment-supplied `CostModel` can realize an `AggIntent::Extension` /// intent as a real sketch instead of the default `PassThrough` (issue /// #150) — `realizations_for_intent` must consult `realize_extension` - /// for the `Extension` arm, and `readout` must consult - /// `readout_extension` to build its `SketchQuery` without panicking. + /// for the `Extension` arm, and `evaluation` must consult + /// `evaluation_extension` to build its `SketchStatistic` without panicking. struct FrequencyCostModel; impl CostModel for FrequencyCostModel { @@ -10014,15 +9976,15 @@ mod tests { } } - fn readout_extension( + fn evaluation_extension( &self, ext_kind: &str, payload: &serde_json::Value, _col: &ColumnRef, - ) -> PostAsapSketchQuery { + ) -> PostAsapSketchStatistic { assert_eq!(ext_kind, "frequency"); let value = payload["item"].as_str().map(str::to_string); - PostAsapSketchQuery::PointCount { + PostAsapSketchStatistic::PointCount { key: ColumnRef::Named("item".into()), value, } @@ -10040,7 +10002,7 @@ mod tests { }; let q = agg(vec![], intent, metric_scan(&[])); let root = realize(&q).unwrap(); - assert!(matches!(root.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!root.contains_asap()); } #[test] @@ -10052,25 +10014,25 @@ mod tests { let q = agg(vec![], intent, metric_scan(&[])); let root = realize_first(&q, &FrequencyCostModel).unwrap(); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; assert!(matches!( query, - PostAsapSketchQuery::PointCount { key: ColumnRef::Named(k), value: Some(v) } + PostAsapSketchStatistic::PointCount { key: ColumnRef::Named(k), value: Some(v) } if k == "item" && v == "checkout" )); - let SummaryExpr::SummaryAgg { family, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::CountSketch, SketchParams::CountSketch { @@ -10087,19 +10049,19 @@ mod tests { fn exact_sum_realizes_accumulator_without_estimate() { let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryAgg { family, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &root.operator else { panic!( "expected bare SummaryAgg (no estimate), got {:?}", - root.expr + root.operator ); }; assert_eq!( family, - &SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + &FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) ); assert_eq!( field(&root.schema, "sum").dtype, - SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) ); } @@ -10110,18 +10072,20 @@ mod tests { use std::time::Duration; let q = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { + OperatorNode::non_asap_node(NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["job"])), - }, + child: metric_scan(&["job"]), + }) + .unwrap(), ); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryAgg { family, .. } = &root.expr else { - panic!("expected SummaryAgg, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &root.operator else { + panic!("expected SummaryAgg, got {:?}", root.operator); }; assert_eq!( family, - &SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) + &FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) ); assert_eq!( root.schema @@ -10133,7 +10097,7 @@ mod tests { ); assert_eq!( field(&root.schema, "value").dtype, - SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) + FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) ); assert_eq!(root.schema.time_index, Some(0)); } @@ -10148,17 +10112,19 @@ mod tests { use std::time::Duration; let q = agg_per_entity( default_quantile(0.99), - QueryExpr::TimeRange { + OperatorNode::non_asap_node(NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, range: Duration::from_secs(10), - child: Rc::new(metric_scan(&["job"])), - }, + child: metric_scan(&["job"]), + }) + .unwrap(), ); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("expected estimate root, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { + panic!("expected estimate root, got {:?}", root.operator); }; - let SummaryExpr::SummaryAgg { reduction, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { reduction, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!(reduction, &ReductionTy::PerEntity); } @@ -10176,11 +10142,11 @@ mod tests { }; let q = agg(vec![], intent, metric_scan(&["job"])); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("expected estimate root, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { + panic!("expected estimate root, got {:?}", root.operator); }; - let SummaryExpr::SummaryAgg { reduction, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { reduction, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!(reduction, &ReductionTy::by(vec![])); } @@ -10192,39 +10158,41 @@ mod tests { // accumulator. let inner = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let outer = agg(vec![], default_quantile(0.9), inner); - let root = realize(&outer).unwrap(); + // Timing is not stored during realization: time the candidate under + // the default lifecycle assignment to read the maintenance boundary. + let root = timed(&realize(&outer).unwrap()); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("expected estimate root, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { + panic!("expected estimate root, got {:?}", root.operator); }; - let SummaryExpr::SummaryAgg { child, family, .. } = &summary_input.expr else { - panic!("expected outer SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { child, family, .. }) = &summary_input.operator + else { + panic!( + "expected outer SummaryAgg, got {:?}", + summary_input.operator + ); }; assert!(matches!( family, - SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll + FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll )); - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::IngestionTime, - } = &child.expr - else { - panic!("expected explicit maintenance readout"); + assert_eq!(child.timing, Some(ExecutionTiming::IngestionTime)); + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) = &child.operator else { + panic!("expected explicit maintenance evaluation"); }; - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family: inner_family, child: leaf, .. - } = &child.expr + }) = &child.operator else { - panic!("expected inner SummaryAgg, got {:?}", child.expr); + panic!("expected inner SummaryAgg, got {:?}", child.operator); }; assert_eq!( inner_family, - &SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + &FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) ); - assert!(matches!(leaf.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!leaf.contains_asap()); } /// Issue #115: the summary is built over the intent's own input columns. @@ -10260,12 +10228,14 @@ mod tests { } /// The update expression of the first `SummaryAgg` in the tree. - fn find_summary_input(node: &SummaryNode) -> Option { - match &node.expr { - SummaryExpr::SummaryAgg { input, .. } if input.item.is_none() => { + fn find_summary_input(node: &OperatorNode) -> Option { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) if input.item.is_none() => { Some(input.weight.clone()) } - SummaryExpr::SummaryEstimate { summary_input, .. } => find_summary_input(summary_input), + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + find_summary_input(summary_input) + } _ => None, } } @@ -10274,7 +10244,7 @@ mod tests { fn pass_through_intents_stay_logical() { // avg is exact but non-mergeable; histogram_quantile (classic // buckets, #79) is never sketchable; exact quantile is exact by - // decree. All three stay whole logical subtrees. + // decree. All three stay whole logical sub-DAGs. for intent in [ AggIntent::Avg { col: None }, AggIntent::HistogramQuantile { q: 0.99, le: 0 }, @@ -10286,59 +10256,61 @@ mod tests { ] { let q = agg(vec![2], intent.clone(), metric_scan(&["job"])); let root = realize(&q).unwrap(); + // Kept pass-through: the pre-ASAP node itself, not a wrapper. assert!( - matches!(root.expr, SummaryExpr::KeepPreAsap(ref e) if **e == q), - "expected KeepPreAsap passthrough for {intent:?}" + !root.contains_asap() && root.operator == q.operator && root.schema == q.schema, + "expected kept pre-ASAP passthrough for {intent:?}" ); } } #[test] fn logical_parent_subsumes_bindable_child() { - // Filter over a bindable quantile: `KeepPreAsap` has no post-ASAP - // children, so the conservative fallback keeps the whole subtree + // Filter over a bindable quantile: a kept non-ASAP sub-DAG has no + // summary children, so the conservative fallback keeps the whole sub-DAG // logical. + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::{CompareOpKind, ScalarValue}; - use asap_types::pre_asap::query_expr::Predicate; - let q = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let q = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(0.5))), - })), - child: Rc::new(agg(vec![], default_quantile(0.99), metric_scan(&[]))), - }; + right: Box::new(ScalarExpr::Literal(ScalarValue::Float64(0.5))), + semantics: asap_types::ir::ExprSemantics::Promql, + }), + child: agg(vec![], default_quantile(0.99), metric_scan(&[])), + }) + .unwrap(); let root = realize(&q).unwrap(); - assert!(matches!(root.expr, SummaryExpr::KeepPreAsap(ref e) if **e == q)); + assert!( + !root.contains_asap() && root.operator == q.operator && root.schema == q.schema, + "expected the whole Filter subtree kept pre-ASAP" + ); } #[test] fn having_and_multi_intent_stay_logical() { + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - let mut q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - if let QueryExpr::Aggregate { having, .. } = &mut q { - *having = Some(Predicate(Rc::new(QueryExpr::Literal( - ScalarValue::Boolean(true), - )))); - } - assert!(matches!( - realize(&q).unwrap().expr, - SummaryExpr::KeepPreAsap(_) - )); + let q = crate::test_support::aggregate( + ReductionTy::by(vec![2]), + vec![default_quantile(0.99)], + vec![], + Some(Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true)))), + metric_scan(&["job"]), + ); + assert!(!realize(&q).unwrap().contains_asap()); - let multi = QueryExpr::Aggregate { + let multi = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: ReductionTy::by(vec![2]), measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&["job"])), - }; - assert!(matches!( - realize(&multi).unwrap().expr, - SummaryExpr::KeepPreAsap(_) - )); + child: metric_scan(&["job"]), + }) + .unwrap(); + assert!(!realize(&multi).unwrap().contains_asap()); } // No binding rule applies a per-measure `FILTER` (#466), so the @@ -10346,18 +10318,16 @@ mod tests { #[test] fn filtered_measure_stays_logical() { use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; let mut q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - if let QueryExpr::Aggregate { filters, .. } = &mut q { - *filters = vec![Some(Predicate(Rc::new(QueryExpr::Literal( - ScalarValue::Boolean(true), + if let Operator::NonASAP(NonASAPOp::Aggregate { filters, .. }) = + &mut Rc::make_mut(&mut q).operator + { + *filters = vec![Some(Predicate(ScalarExpr::Literal(ScalarValue::Boolean( + true, ))))]; } assert!(bindable_intent(&q).is_none()); - assert!(matches!( - realize(&q).unwrap().expr, - SummaryExpr::KeepPreAsap(_) - )); + assert!(!realize(&q).unwrap().contains_asap()); } #[test] @@ -10371,7 +10341,7 @@ mod tests { metric_scan(&["job"]), ); let root = realize(&q).unwrap(); - assert!(matches!(root.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!root.contains_asap()); } #[test] @@ -10383,20 +10353,20 @@ mod tests { }, metric_scan(&["job"]), ); - let root = Rc::new(agg( + let root = agg( vec![], AggIntent::TopK { k: 5, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); + ); let proposals = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); assert!(!proposals.is_empty()); assert!(proposals.iter().any(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) if node.guarantee.as_ref().is_some_and(|g| + Replacement::SubDag(node) if node.guarantee.as_ref().is_some_and(|g| g.bound.evaluate().is_none() && g.failure_probability.evaluate().is_none()) ))); @@ -10408,8 +10378,8 @@ mod tests { fn propagation_stats( &self, op: &CompositionOperator, - _family: &SummaryFamilyType, - _query: Option<&PostAsapSketchQuery>, + _family: &FieldDataType, + _query: Option<&PostAsapSketchStatistic>, ) -> PropagationStats { if matches!(op, CompositionOperator::TopKSelection) { PropagationStats { @@ -10436,7 +10406,7 @@ mod tests { }, metric_scan(&["job"]), ); - let q = Rc::new(agg( + let q = agg( vec![], AggIntent::TopK { k: 5, @@ -10446,8 +10416,8 @@ mod tests { }, }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -10457,7 +10427,7 @@ mod tests { assert!(!replacements.is_empty()); assert!(replacements.iter().all(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) + Replacement::SubDag(node) if node.guarantee.as_ref().is_some_and(|g| g.metric == ErrorMetric::TopKMembership && g.failure_probability.evaluate() == Some(0.005)) @@ -10473,15 +10443,15 @@ mod tests { }, metric_scan(&["service"]), ); - let outer = Rc::new(agg( + let outer = agg( vec![], AggIntent::TopK { k: 10, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -10491,26 +10461,26 @@ mod tests { let node = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CmsWithHeap") => { + Replacement::SubDag(node) if candidate.rationale.contains("CmsWithHeap") => { Some(node) } _ => None, }) .expect("CmsWithHeap candidate"); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &node.expr + }) = &node.operator else { - panic!("expected Top-K readout") + panic!("expected Top-K evaluation") }; - assert!(matches!(query, PostAsapSketchQuery::TopK { k: 10 })); - let SummaryExpr::SummaryAgg { + assert!(matches!(query, PostAsapSketchStatistic::TopK { k: 10 })); + let Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, .. - } = &summary_input.expr + }) = &summary_input.operator else { panic!("expected fused summary aggregation") }; @@ -10527,10 +10497,10 @@ mod tests { ); assert!(matches!( family, - SummaryFamilyType::Sketch(kind, _) + FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::CmsWithHeap )); - assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!child.contains_asap()); } #[test] @@ -10540,15 +10510,15 @@ mod tests { AggIntent::Sum { col: None }, metric_scan(&["service"]), ); - let outer = Rc::new(agg( + let outer = agg( vec![], AggIntent::TopK { k: 5, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -10565,7 +10535,7 @@ mod tests { let node = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) + Replacement::SubDag(node) if candidate.rationale.contains("CountSketchWithHeap") => { Some(node) @@ -10573,15 +10543,16 @@ mod tests { _ => None, }) .expect("CountSketchWithHeap candidate"); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &node.expr + }) = &node.operator else { - panic!("expected Top-K readout") + panic!("expected Top-K evaluation") }; - assert!(matches!(query, PostAsapSketchQuery::TopK { k: 5 })); - let SummaryExpr::SummaryAgg { child, input, .. } = &summary_input.expr else { + assert!(matches!(query, PostAsapSketchStatistic::TopK { k: 5 })); + let Operator::ASAP(ASAPOp::SummaryAgg { child, input, .. }) = &summary_input.operator + else { panic!("expected fused summary aggregation") }; assert!(matches!( @@ -10592,31 +10563,34 @@ mod tests { input.weight, SummaryInputExpr::Column(ColumnRef::SampleValue) ); - assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!child.contains_asap()); } #[test] fn temporal_per_entity_topk_uses_series_identity_and_sample_value() { - let inner = QueryExpr::Aggregate { + let inner = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: ReductionTy::PerEntity, measures: vec![AggIntent::Sum { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: OperatorNode::non_asap_node(NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, range: std::time::Duration::from_secs(60), - child: Rc::new(metric_scan(&["service"])), - }), - }; - let outer = Rc::new(agg( + child: metric_scan(&["service"]), + }) + .unwrap(), + }) + .unwrap(); + let outer = agg( vec![2], AggIntent::TopK { k: 5, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -10624,13 +10598,14 @@ mod tests { ); let candidates = strategy.replacements(&TargetSubDAG::new(&outer)); let input = candidates.iter().find_map(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDag(node) = &candidate.replacement else { return None; }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator + else { return None; }; - let SummaryExpr::SummaryAgg { input, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) = &summary_input.operator else { return None; }; input.item.is_some().then_some(input) @@ -10659,15 +10634,15 @@ mod tests { }, metric_scan(&["service", "region"]), ); - let outer = Rc::new(agg( + let outer = agg( vec![], AggIntent::TopK { k: 10, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -10678,10 +10653,10 @@ mod tests { .replacements(&TargetSubDAG::new(&outer)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) => match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => { - match &summary_input.expr { - SummaryExpr::SummaryAgg { input, .. } => Some(input.clone()), + Replacement::SubDag(node) => match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + match &summary_input.operator { + Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) => Some(input.clone()), _ => None, } } @@ -10712,15 +10687,15 @@ mod tests { ); // The inner aggregate outputs its grouping keys first, so column 2 is // `region`. Each region is a separate Top-K subpopulation. - let outer = Rc::new(agg( + let outer = agg( vec![2], AggIntent::TopK { k: 10, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -10730,12 +10705,12 @@ mod tests { .replacements(&TargetSubDAG::new(&outer)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) => match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => { - match &summary_input.expr { - SummaryExpr::SummaryAgg { + Replacement::SubDag(node) => match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + match &summary_input.operator { + Operator::ASAP(ASAPOp::SummaryAgg { input, reduction, .. - } => Some((input.clone(), reduction.clone())), + }) => Some((input.clone(), reduction.clone())), _ => None, } } @@ -10760,25 +10735,26 @@ mod tests { fn sql_reducer_resolves_named_input_column() { // SUM(bytes) over a tabular scan: `col` resolves positionally to the // named column, not the PromQL sample value. - let scan = QueryExpr::Scan { + let scan = OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::Table { table_ref: "t".into(), }, predicates: vec![], schema: SchemaTy { - columns: vec![ - Column::new("host", DataType::Utf8, false), - Column::new("bytes", DataType::Int64, false), + fields: vec![ + Field::plain("host", DataType::Utf8, false), + Field::plain("bytes", DataType::Int64, false), ], time_index: None, unique_keys: vec![], closed: true, }, - }; + }) + .unwrap(); let q = agg(vec![0], AggIntent::Sum { col: Some(1) }, scan); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryAgg { input, .. } = &root.expr else { - panic!("expected SummaryAgg, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) = &root.operator else { + panic!("expected SummaryAgg, got {:?}", root.operator); }; let SummaryInputExpr::Column(col) = &input.weight else { panic!("expected observation column") @@ -10800,8 +10776,8 @@ mod tests { impl AccuracyModel for RankAdditiveModel { fn local_guarantee( &self, - family: &SummaryFamilyType, - query: &PostAsapSketchQuery, + family: &FieldDataType, + query: &PostAsapSketchStatistic, ) -> Option { DefaultAccuracyModel.local_guarantee(family, query) } @@ -10843,10 +10819,12 @@ mod tests { } } - fn summary_child(node: &SummaryNode) -> &Rc { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => summary_child(summary_input), - SummaryExpr::SummaryAgg { child, .. } => child, + fn summary_child(node: &OperatorNode) -> &Rc { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + summary_child(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) => child, other => panic!("expected a SummaryAgg, got {other:?}"), } } @@ -10857,9 +10835,8 @@ mod tests { // registered rule, so every outer sketch candidate is refused with a // typed reason and the raw/pre-ASAP alternative is what remains. let inner = agg(vec![2], default_quantile(0.5), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], default_quantile(0.99), inner)); - let proposals = - SketchAlgorithmStrategy::default_cost_model().propose(&TargetSubDAG::new(&outer)); + let outer = agg(vec![], default_quantile(0.99), inner); + let proposals = ASAPStrategies::default_cost_model().propose(&TargetSubDAG::new(&outer)); assert!( proposals.candidates.is_empty(), "no outer sketch may be proposed over an approximate child without a rule: {:?}", @@ -10880,9 +10857,9 @@ mod tests { rejection.error ); } - // Fallback keeps the whole subtree pre-ASAP — executed exactly. + // Fallback keeps the whole sub-DAG pre-ASAP — executed exactly. let realized = realize_child(&outer, &DefaultCostModel).unwrap(); - assert!(matches!(realized.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!realized.contains_asap()); assert!(realized .guarantee .as_ref() @@ -10890,9 +10867,8 @@ mod tests { // Cross-metric: a quantile over a cardinality estimate. let inner = agg(vec![2], default_cardinality(), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], default_quantile(0.99), inner)); - let proposals = - SketchAlgorithmStrategy::default_cost_model().propose(&TargetSubDAG::new(&outer)); + let outer = agg(vec![], default_quantile(0.99), inner); + let proposals = ASAPStrategies::default_cost_model().propose(&TargetSubDAG::new(&outer)); assert!(proposals.candidates.is_empty()); assert!(proposals.rejected.iter().all(|r| matches!( &r.error, @@ -10904,14 +10880,14 @@ mod tests { #[test] fn exact_child_contributes_zero_error() { // quantile(0.9, sum by (job) (m)): KLL over an exact Sum accumulator - // — the readout's guarantee is exactly KLL's own local guarantee. + // — the evaluation's guarantee is exactly KLL's own local guarantee. let inner = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let outer = agg(vec![], default_quantile(0.9), inner); let root = realize(&outer).unwrap(); let guarantee = root .guarantee .as_ref() - .expect("a readout carries a guarantee"); + .expect("a evaluation carries a guarantee"); assert_eq!(guarantee.metric, ErrorMetric::Rank); assert_eq!( guarantee.bound.evaluate(), @@ -10928,7 +10904,7 @@ mod tests { ))); // The sketch *state* node carries no guarantee; the exact // accumulator's state is its value and does. - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { panic!() }; assert!(summary_input.guarantee.is_none()); @@ -10939,16 +10915,19 @@ mod tests { } #[test] - fn exact_sum_can_consume_an_approximate_readout() { + fn exact_sum_can_consume_an_approximate_evaluation() { // sum(count_distinct by (job) (m)) is an outer exact summary over - // the inner HLL readout. Both summary levels remain explicit. + // the inner HLL evaluation. Both summary levels remain explicit. let inner = agg(vec![2], default_cardinality(), metric_scan(&["job"])); let outer = agg(vec![], AggIntent::Sum { col: None }, inner); let root = realize(&outer).unwrap(); - let SummaryExpr::SummaryAgg { child, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &root.operator else { panic!("outer exact sum should remain a SummaryAgg") }; - assert!(matches!(child.expr, SummaryExpr::SummaryEstimate { .. })); + assert!(matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + )); assert!(root.guarantee.is_some()); // count(...) over the same child is exact: a row count does not @@ -10969,12 +10948,12 @@ mod tests { } #[test] - fn equal_split_allocation_supports_nested_summary_readouts() { + fn equal_split_allocation_supports_nested_summary_evaluations() { // A registered rank-additive rule and valid budget split make both // summary levels explicit while preserving the composed guarantee. let inner = agg(vec![2], quantile_eps(0.5, 0.1), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], quantile_eps(0.99, 0.1), inner)); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs( + let outer = agg(vec![], quantile_eps(0.99, 0.1), inner); + let strategy = ASAPStrategies::new_with_planning_inputs( &DefaultCostModel, &RankAdditiveModel, &EqualSplitAllocator, @@ -10983,13 +10962,15 @@ mod tests { assert!(!proposals.candidates.is_empty()); assert!(proposals.candidates.iter().all(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDag(node) = &candidate.replacement else { return false; }; - matches!(node.expr, SummaryExpr::SummaryEstimate { .. }) - && node.guarantee.as_ref().is_some_and(|guarantee| { - DefaultAccuracyModel.satisfies(guarantee, &AccuracyTarget::Epsilon(0.1)) - }) + matches!( + node.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + ) && node.guarantee.as_ref().is_some_and(|guarantee| { + DefaultAccuracyModel.satisfies(guarantee, &AccuracyTarget::Epsilon(0.1)) + }) })); } @@ -10998,9 +10979,9 @@ mod tests { // The same nested summary remains available through workload search // and global cost ranking. let inner = agg(vec![2], quantile_eps(0.5, 0.1), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], quantile_eps(0.99, 0.1), inner)); + let outer = agg(vec![], quantile_eps(0.99, 0.1), inner); let strategies: Vec> = - vec![Box::new(SketchAlgorithmStrategy::new_with_planning_inputs( + vec![Box::new(ASAPStrategies::new_with_planning_inputs( &DefaultCostModel, &RankAdditiveModel, &EqualSplitAllocator, @@ -11010,10 +10991,14 @@ mod tests { let group = space.candidates_for_target(root).unwrap(); assert!(!group.rejected.is_empty()); assert!(group.candidates.iter().all(|c| match &c.replacement { - Replacement::Summary(node) => node.guarantee.as_ref().is_some_and(|g| { - DefaultAccuracyModel.satisfies(g, &AccuracyTarget::Epsilon(0.1)) - }), - Replacement::Rewrite(_) => false, + // A summary candidate (old `Replacement::Summary`) contains an + // ASAP node; a logical rewrite (old `Replacement::Rewrite`) does not. + Replacement::SubDag(node) if node.contains_asap() => { + node.guarantee.as_ref().is_some_and(|g| { + DefaultAccuracyModel.satisfies(g, &AccuracyTarget::Epsilon(0.1)) + }) + } + Replacement::SubDag(_) => false, Replacement::ExactComposition(_) => false, })); let ranked = space.cost_sorted(&DefaultCostModel); @@ -11026,15 +11011,18 @@ mod tests { .unwrap() .chosen .expect("a nested summary candidate wins"); - let Replacement::Summary(node) = &chosen.replacement else { + let Replacement::SubDag(node) = &chosen.replacement else { panic!() }; - assert!(matches!(node.expr, SummaryExpr::SummaryEstimate { .. })); + assert!(matches!( + node.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + )); } #[test] fn root_target_check_removes_candidates_before_cost_ranking() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); // A root target tighter than the node's own ε=0.01: every sketch // candidate misses it and is moved to `rejected`; nothing is left // for the cost model to rank. @@ -11048,7 +11036,7 @@ mod tests { assert!(group .candidates .iter() - .all(|c| matches!(c.replacement, Replacement::Rewrite(_)))); + .all(|c| matches!(&c.replacement, Replacement::SubDag(n) if !n.contains_asap()))); assert!(group.rejected.iter().all(|r| matches!( r.error, AccuracyError::TargetNotSatisfied { target: AccuracyTarget::Epsilon(e), .. } if e == 0.001 @@ -11067,7 +11055,7 @@ mod tests { assert!(group .candidates .iter() - .any(|c| matches!(c.replacement, Replacement::Summary(_)))); + .any(|c| matches!(&c.replacement, Replacement::SubDag(n) if n.contains_asap()))); // An `Exact` root target admits only exact candidates. let space = search_workload_with_targets( @@ -11077,11 +11065,11 @@ mod tests { ); let group = space.candidates_for_target(&space.roots[0].1).unwrap(); assert!(group.candidates.iter().all(|c| match &c.replacement { - Replacement::Summary(node) => node + Replacement::SubDag(node) if node.contains_asap() => node .guarantee .as_ref() .is_some_and(ResultGuarantee::is_exact), - Replacement::Rewrite(_) => true, + Replacement::SubDag(_) => true, Replacement::ExactComposition(_) => false, })); } @@ -11095,14 +11083,14 @@ mod tests { }, metric_scan(&["job"]), ); - let q = Rc::new(agg( + let q = agg( vec![], AggIntent::TopK { k: 10, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); + ); let space = search_workload_with_targets( vec![("q", Rc::clone(&q), Some(AccuracyTarget::Epsilon(0.01)))], &default_strategies(), @@ -11112,14 +11100,14 @@ mod tests { assert!(group.candidates.iter().any(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) if node.guarantee.as_ref().is_some_and(ResultGuarantee::has_unknown) + Replacement::SubDag(node) if node.guarantee.as_ref().is_some_and(ResultGuarantee::has_unknown) ))); let candidate = group .candidates .iter() .find(|candidate| candidate.has_missing_accuracy_evidence()) .unwrap(); - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDag(node) = &candidate.replacement else { unreachable!() }; let exported = asap_types::dag_export::export_summary(node); @@ -11143,13 +11131,13 @@ mod tests { fn scoped_hll_evidence_sizes_and_certifies_without_a_deployment_model() { use crate::accuracy::EstimatorContract; struct SourceEvidence { - expression: QueryExpr, + expression: OperatorNode, max_distinct: u32, } impl AccuracyEvidenceProvider for SourceEvidence { - fn estimator_contract(&self, expression: &QueryExpr) -> Option { + fn estimator_contract(&self, expression: &OperatorNode) -> Option { (expression == &self.expression).then_some(EstimatorContract::ClassicHll { - max_distinct_per_readout: self.max_distinct, + max_distinct_per_evaluation: self.max_distinct, }) } } @@ -11157,19 +11145,19 @@ mod tests { epsilon: 0.05, delta: 0.01, }; - let root = Rc::new(agg( + let root = agg( vec![], AggIntent::Cardinality { cols: vec![], accuracy: target.clone(), }, metric_scan(&[]), - )); + ); let evidence = SourceEvidence { expression: (*root).clone(), max_distinct: 128, }; - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -11179,7 +11167,7 @@ mod tests { let hll = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) + Replacement::SubDag(node) if summary_family_algorithm(node) == SketchAlgorithm::Hll => { Some(node) @@ -11189,13 +11177,13 @@ mod tests { .expect("HLL candidate"); assert!(DefaultAccuracyModel .satisfies(hll.guarantee.as_ref().expect("HLL confidence"), &target)); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &hll.expr else { - panic!("readout") + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &hll.operator else { + panic!("evaluation") }; - let SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + let Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), .. - } = &summary_input.expr + }) = &summary_input.operator else { panic!("HLL state") }; @@ -11209,9 +11197,8 @@ mod tests { precision: expected } ); - let absent = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); - assert!(!absent.iter().any(|candidate| matches!(&candidate.replacement, Replacement::Summary(node) + let absent = ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); + assert!(!absent.iter().any(|candidate| matches!(&candidate.replacement, Replacement::SubDag(node) if summary_family_algorithm(node) == SketchAlgorithm::Hll && node.guarantee.as_ref().is_some_and(|g| DefaultAccuracyModel.satisfies(g, &target))))); // Invalid contracts, infeasible targets and evidence for another source // must never authorize a confidence-bearing HLL candidate. @@ -11225,30 +11212,30 @@ mod tests { epsilon: 0.05, delta, }; - let query = Rc::new(agg( + let query = agg( vec![], AggIntent::Cardinality { cols: vec![], accuracy: target.clone(), }, metric_scan(&[]), - )); + ); let evidence = SourceEvidence { expression: if wrong_scope { - metric_scan(&["other"]) + (*metric_scan(&["other"])).clone() } else { (*query).clone() }, max_distinct, }; - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, &evidence, ); assert!(!strategy.replacements(&TargetSubDAG::new(&query)).iter().any(|candidate| - matches!(&candidate.replacement, Replacement::Summary(node) + matches!(&candidate.replacement, Replacement::SubDag(node) if summary_family_algorithm(node) == SketchAlgorithm::Hll && node.guarantee.as_ref().is_some_and(|g| DefaultAccuracyModel.satisfies(g, &target))))); } } @@ -11256,63 +11243,62 @@ mod tests { // A value projection cannot consume an opaque exact accumulator edge. #[test] fn residual_projection_finalizes_selected_exact_state() { - let inner = Rc::new(agg(vec![], AggIntent::Sum { col: None }, metric_scan(&[]))); - let root = Rc::new(QueryExpr::Project { - cols: vec![asap_types::pre_asap::ProjectItem { - expr: QueryExpr::Column(0), + let inner = agg(vec![], AggIntent::Sum { col: None }, metric_scan(&[])); + let root = OperatorNode::non_asap_node(NonASAPOp::Project { + cols: vec![ProjectItem { + expr: ScalarExpr::Column(0), alias: Some("result".into()), }], qualifier: None, child: inner.clone(), - }); + }) + .unwrap(); let space = search_workload_with_targets( vec![("q", root.clone(), Some(AccuracyTarget::Exact))], &default_strategies(), &DefaultAccuracyModel, ); let selected = space.global_selection(&DefaultCostModel); + // CSE re-interns the workload, so the space's root/child `Rc`s are not + // the fixture's. Assembly only assembles children that are discovered + // targets, so seed the memo under the space's own child pointer. + let root = Rc::clone(&space.roots[0].1); + let Some(NonASAPOp::Project { child: inner, .. }) = root.non_asap() else { + unreachable!() + }; + assert!(space.candidates_for_target(inner).is_some()); selected .assembled_nodes .borrow_mut() - .insert(Rc::as_ptr(&inner), realize(inner.as_ref()).unwrap()); + .insert(Rc::as_ptr(inner), realize(inner.as_ref()).unwrap()); let node = selected.assemble_target(&root).unwrap(); - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Project { .. }, - .. - } = &node.expr - else { + let Operator::NonASAP(NonASAPOp::Project { child, .. }) = &node.operator else { panic!("expected Project"); }; assert!(matches!( - child.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } + child.operator, + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { .. }) )); assert!(child .schema .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); } // A numeric group key must not be mistaken for the ranked aggregate score. #[test] fn ranking_uses_aggregate_output_position_not_first_numeric_column() { let logical = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["id"])); - let mut values = lift(&logical.output_schema().unwrap()); - values.fields[0].dtype = SummaryFamilyType::Plain(DataType::Int64); + let mut values = lift(&logical.schema.clone()); + values.fields[0].dtype = FieldDataType::Plain(DataType::Int64); assert_eq!(ranking_score_index(&logical, &values).unwrap(), 1); } // A heap's key schema is derived from its encoded item, not all label columns. #[test] - fn heap_readout_preserves_numeric_item_identity() { - let mut raw = metric_scan(&["id", "description"]); - let QueryExpr::Scan { schema, .. } = &mut raw else { - unreachable!() - }; - schema.columns[2].dtype = DataType::Int64; + fn heap_evaluation_preserves_numeric_item_identity() { + let mut schema = metric_scan(&["id", "description"]).schema.clone(); + schema.fields[2].dtype = FieldDataType::Plain(DataType::Int64); + let raw = crate::test_support::scan("m", schema); let node = agg( vec![], AggIntent::TopK { @@ -11322,7 +11308,7 @@ mod tests { agg(vec![2], AggIntent::Sum { col: None }, raw.clone()), ); let input = PhysicalSummaryInput { - child: Rc::new(raw), + child: raw, input: SummaryUpdate { item: Some(SummaryInputExpr::Column(ColumnRef::Named("id".into()))), weight: SummaryInputExpr::Constant(1.0), @@ -11331,7 +11317,7 @@ mod tests { }, }, }; - let schema = keyed_heap_readout_schema(&input, &node).unwrap(); + let schema = keyed_heap_evaluation_schema(&input, &node).unwrap(); assert_eq!( schema .fields @@ -11342,7 +11328,7 @@ mod tests { ); assert_eq!( schema.fields[0].dtype, - SummaryFamilyType::Plain(DataType::Int64) + FieldDataType::Plain(DataType::Int64) ); } } diff --git a/crates/asap-aware-mapping/src/rewrite.rs b/crates/asap-aware-mapping/src/rewrite.rs index 06d1724fd..ae2b9d59f 100644 --- a/crates/asap-aware-mapping/src/rewrite.rs +++ b/crates/asap-aware-mapping/src/rewrite.rs @@ -12,12 +12,12 @@ //! comment on why: `Avg`/`StdDev`/`Variance` "need richer partial state" //! than a bare sketch/exact accumulator gives, so there is no summary //! realization for a bare `avg` node to bind to at all. A logical `avg` -//! node therefore can never be a [`SharedSubtreeStrategy`] target either: +//! node therefore can never be a [`SharedSubDagStrategy`] target either: //! CSE-style sharing needs *some* mergeable accumulator underneath, and //! `PassThrough` has none. //! //! `Sum` and `Count` are both ordinary mergeable accumulators -//! (`agg_is_mergeable`) — exactly the shape [`SharedSubtreeStrategy`] and a +//! (`agg_is_mergeable`) — exactly the shape [`SharedSubDagStrategy`] and a //! future sketch-family search already know how to reuse across a //! workload. Rewriting `Aggregate{ measures: [Avg{col}], .. }` into two //! independent single-measure `Sum` and `Count` aggregates, divided with a @@ -35,7 +35,7 @@ //! - **`without(...)` grouping** leaves an `Aggregate`'s own output schema //! *open* (`closed: false`, see `without_output_schema`), while the //! `Project` this strategy always wraps the rewrite in forces -//! `closed: true` (see `QueryExpr::output_schema`'s `Project` arm). Under +//! `closed: true` (see `NonASAPOp::output_schema`'s `Project` arm). Under //! `without(...)` the rewritten form's `closed` flag would silently flip //! relative to the original — exactly the kind of schema drift this //! module exists to avoid. @@ -43,7 +43,7 @@ //! Both are follow-ups (issue #253 itself scopes to "the concrete case in //! Peilin's comment"), not correctness bugs in what ships here — a node //! outside this scope simply doesn't `match`, the same "safe but -//! uninformative" fallback [`SketchAlgorithmStrategy`]/[`SharedSubtreeStrategy`] +//! uninformative" fallback [`ASAPStrategies`]/[`SharedSubDagStrategy`] //! already use for shapes they don't have an opinion on. //! //! ## Non-goals (mirrors [`replacement`]'s own discipline) @@ -57,14 +57,15 @@ //! the rewritten form is actually worth picking, by letting the original //! and rewritten forms compete on cost — not this strategy. +use asap_types::ir::non_asap::any_measure_filtered; use std::rc::Rc; +use asap_types::ir::operator_properties::{BinaryOpKind, Reduction}; +use asap_types::ir::{BinaryOperator, NonASAPOp, OperatorNode, ProjectItem, ScalarExpr}; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::pre_asap::expr_ir::ArithmeticOpKind; -use asap_types::pre_asap::query_expr::{ - any_measure_filtered, BinaryOpKind, ProjectItem, QueryExpr, Reduction, -}; use asap_types::pre_asap::schema::{ColumnId, DataType}; + use asap_types::types::AccuracyTarget; use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG}; @@ -74,15 +75,35 @@ use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, Ta /// the module docs' "Scope" for why `without(...)`/`PerEntity` are /// excluded). Returns the grouping key count and the summed column so /// [`build_rewrite`] doesn't have to re-match. -fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { - let QueryExpr::Aggregate { +/// `a / b` with PromQL arithmetic semantics and no vector matching. +fn arithmetic( + op: ArithmeticOpKind, + lhs: Rc, + rhs: Rc, +) -> Option> { + OperatorNode::non_asap_node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: BinaryOpKind::Arithmetic(op), + vector_match: None, + }, + return_bool: false, + lhs, + rhs, + }) + .ok() +} + +fn avg_rewrite_target(node: &OperatorNode) -> Option<(usize, Option)> { + let Some(NonASAPOp::Aggregate { reduction, measures, filters, having: None, child, .. - } = node + }) = node.non_asap() else { return None; }; @@ -102,11 +123,11 @@ fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { // therefore be decomposed through it only when the averaged input is // provably non-null; otherwise NULL rows would incorrectly contribute to // the denominator. - let input_schema = child.output_schema().ok()?; + let input_schema = &child.schema; let value_col = col .or_else(|| input_schema.column_id("value")) - .or_else(|| (0..input_schema.columns.len()).find(|i| !by.contains(i)))?; - if input_schema.columns.get(value_col)?.nullable { + .or_else(|| (0..input_schema.fields.len()).find(|i| !by.contains(i)))?; + if input_schema.fields.get(value_col)?.nullable { return None; } Some((by.keys().len(), *col)) @@ -129,21 +150,21 @@ fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { /// exactly regardless of the summed column's own type (integer division /// would otherwise silently reappear whenever the input column is itself /// integer-typed: `Sum`'s output type tracks its input, `Count`'s is always -/// `Int64`, and `QueryExpr::output_schema`'s own `Arithmetic` type inference +/// `Int64`, and `ScalarExpr::scalar_type`'s own `Arithmetic` type inference /// types a `Div` of two `Int64` operands as `Int64` — the explicit operand /// `Cast` is what keeps both the division and rewritten `avg` column /// `Float64` the way the original always was, not an incidental extra step). // These are conditional physical components, never an unconditional Rewrite. // The caller must attach the finite-division execution guard before admission. -pub(crate) fn temporal_average_components(root: &Rc) -> Option> { - let QueryExpr::Aggregate { +pub(crate) fn temporal_average_components(root: &Rc) -> Option> { + let Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures, filters, child, having: None, .. - } = root.as_ref() + }) = root.non_asap() else { return None; }; @@ -153,18 +174,18 @@ pub(crate) fn temporal_average_components(root: &Rc) -> Option) -> Option) -> Option> { +fn build_rewrite(root: &Rc) -> Option> { let (group_count, col) = avg_rewrite_target(root)?; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, output_names, child, .. - } = root.as_ref() + }) = root.non_asap() else { unreachable!("avg_rewrite_target already confirmed an Aggregate shape"); }; // The original `avg` column's own name: `output_names[0]` if the // producing front end overrode it (SQL threading DataFusion's own - // generated name — see `QueryExpr::Aggregate::output_names`'s docs), + // generated name — see `NonASAPOp::Aggregate::output_names`'s docs), // else `AggIntent::Avg`'s synthetic default. Either way this is the // *only* thing about the original output column this rewrite needs to // reproduce — `AggIntent::Avg::output_column`'s `(Float64, nullable: @@ -210,15 +231,16 @@ fn build_rewrite(root: &Rc) -> Option> { .cloned() .unwrap_or_else(|| "avg".to_string()); - let sum_agg = Rc::new(QueryExpr::Aggregate { + let sum_agg = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: reduction.clone(), measures: vec![AggIntent::Sum { col }], output_names: Vec::new(), filters: vec![], having: None, child: Rc::clone(child), - }); - let count_agg = Rc::new(QueryExpr::Aggregate { + }) + .ok()?; + let count_agg = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: reduction.clone(), measures: vec![AggIntent::Count { accuracy: AccuracyTarget::Exact, @@ -227,60 +249,57 @@ fn build_rewrite(root: &Rc) -> Option> { filters: vec![], having: None, child: Rc::clone(child), - }); + }) + .ok()?; let sum_idx = group_count; let mut cols: Vec = (0..group_count) .map(|i| ProjectItem { alias: None, - expr: QueryExpr::Column(i), + expr: ScalarExpr::Column(i), }) .collect(); cols.push(ProjectItem { alias: Some(avg_name), - expr: QueryExpr::Cast { - expr: Rc::new(QueryExpr::Column(sum_idx)), + expr: ScalarExpr::Cast { + expr: Box::new(ScalarExpr::Column(sum_idx)), to: DataType::Float64, try_cast: false, }, }); - let float_sum = Rc::new(QueryExpr::Project { + let float_sum = OperatorNode::non_asap_node(NonASAPOp::Project { cols, qualifier: None, child: sum_agg, - }); - Some(Rc::new(QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: float_sum, - rhs: count_agg, - vector_match: None, - })) + }) + .ok()?; + arithmetic(ArithmeticOpKind::Div, float_sum, count_agg) } /// Compose adjacent per-entity and cross-entity accumulators when their /// algebra, rather than a query-language spelling, proves equivalence. -pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option> { - let original_schema = root.output_schema().ok()?; - let QueryExpr::Aggregate { +pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option> { + let original_schema = &root.schema; + let Some(NonASAPOp::Aggregate { reduction: outer_reduction @ Reduction::Reduce(_), measures: outer_measures, output_names, filters: outer_filters, having: None, child, - } = root.as_ref() + }) = root.non_asap() else { return None; }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: inner_measures, filters: inner_filters, having: None, child: inner_child, .. - } = child.as_ref() + }) = child.non_asap() else { return None; }; @@ -297,14 +316,15 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option inner.clone(), _ => return None, }; - let aggregate = Rc::new(QueryExpr::Aggregate { + let aggregate = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: outer_reduction.clone(), measures: vec![composed], output_names: output_names.clone(), filters: vec![], having: None, child: Rc::clone(inner_child), - }); + }) + .ok()?; // The outer Sum sees PromQL's Float64 sample value, whereas the composed // Count accumulator is Int64. Keep the original observable type. @@ -321,7 +341,7 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option = (0..by.keys().len()) .map(|i| ProjectItem { alias: None, - expr: QueryExpr::Column(i), + expr: ScalarExpr::Column(i), }) .collect(); cols.push(ProjectItem { @@ -329,20 +349,21 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option) -> Option) -> Option QueryExpr { - let mut columns = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - ]; - columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index(columns, 0, vec![]), - } + use crate::test_support::metric_scan; + use asap_types::ir::TimeRangeKind; + + fn avg_agg( + by: Vec, + col: Option, + child: Rc, + ) -> Rc { + avg_agg_with(by, col, vec![], None, child) } - fn avg_agg(by: Vec, col: Option, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn avg_agg_with( + by: Vec, + col: Option, + output_names: Vec, + having: Option, + child: Rc, + ) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![AggIntent::Avg { col }], - output_names: vec![], + output_names, filters: vec![], - having: None, - child: Rc::new(child), - } + having, + child, + }) + .unwrap() } // Temporal averages expose two single-measure children without closing labels. #[test] fn temporal_average_components_preserves_schema_and_exposes_sum_count() { - let root = Rc::new(lower_promql( - "avg_over_time(a{job=\"api\"}[5m])", - AccuracyTarget::Exact, - )); + let root = lower_promql("avg_over_time(a{job=\"api\"}[5m])", AccuracyTarget::Exact); assert!(SemanticEquivalentRewriteStrategy .replacements(&TargetSubDAG::new(&root)) .is_empty()); let rewritten = temporal_average_components(&root).expect("conditional sum/count components"); - assert_eq!( - root.output_schema().unwrap(), - rewritten.output_schema().unwrap() - ); - assert!(matches!(rewritten.as_ref(), QueryExpr::BinaryOp { .. })); + assert_eq!(root.schema.clone(), rewritten.schema.clone()); + assert!(matches!( + rewritten.non_asap(), + Some(NonASAPOp::BinaryOp { .. }) + )); } // ── matches ────────────────────────────────────────────────────────── #[test] fn matches_a_bare_avg_aggregate() { - let q = Rc::new(avg_agg(vec![], None, metric_scan(&[]))); + let q = avg_agg(vec![], None, metric_scan(&[])); let target = TargetSubDAG::new(&q); assert!(AvgToSumOverCountStrategy.matches(&target)); } #[test] fn matches_a_grouped_avg_aggregate() { - let q = Rc::new(avg_agg(vec![2], None, metric_scan(&["job"]))); + let q = avg_agg(vec![2], None, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); assert!(AvgToSumOverCountStrategy.matches(&target)); } #[test] fn does_not_match_a_multi_measure_aggregate() { - let q = Rc::new(QueryExpr::Aggregate { + let q = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&["job"])), - }); + child: metric_scan(&["job"]), + }) + .unwrap(); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -490,13 +514,15 @@ mod tests { #[test] fn does_not_match_a_having_bearing_avg_aggregate() { - let mut q = avg_agg(vec![2], None, metric_scan(&["job"])); - if let QueryExpr::Aggregate { having, .. } = &mut q { - *having = Some(asap_types::pre_asap::query_expr::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true)), - ))); - } - let q = Rc::new(q); + let q = avg_agg_with( + vec![2], + None, + vec![], + Some(asap_types::ir::Predicate(ScalarExpr::Literal( + asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true), + ))), + metric_scan(&["job"]), + ); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -511,14 +537,15 @@ mod tests { }, AggIntent::Min { col: None }, ] { - let q = Rc::new(QueryExpr::Aggregate { + let q = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![intent.clone()], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&["job"])), - }); + child: metric_scan(&["job"]), + }) + .unwrap(); let target = TargetSubDAG::new(&q); assert!( !AvgToSumOverCountStrategy.matches(&target), @@ -530,16 +557,17 @@ mod tests { #[test] fn does_not_match_a_without_grouped_avg_aggregate() { - let q = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::Reduce(asap_types::pre_asap::query_expr::GroupKeys::without( + let q = OperatorNode::non_asap_node(NonASAPOp::Aggregate { + reduction: Reduction::Reduce(asap_types::ir::operator_properties::GroupKeys::without( vec![2], )), measures: vec![AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&["job"])), - }); + child: metric_scan(&["job"]), + }) + .unwrap(); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -547,14 +575,15 @@ mod tests { #[test] fn does_not_match_a_per_entity_avg_aggregate() { - let q = Rc::new(QueryExpr::Aggregate { + let q = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&[])), - }); + child: metric_scan(&[]), + }) + .unwrap(); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -562,7 +591,7 @@ mod tests { #[test] fn does_not_match_a_non_aggregate_node() { - let scan = Rc::new(metric_scan(&["job"])); + let scan = metric_scan(&["job"]); let target = TargetSubDAG::new(&scan); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -575,7 +604,7 @@ mod tests { #[test] fn avg_rewrites_and_schema_matches_exactly_when_ungrouped() { let original = avg_agg(vec![], None, metric_scan(&[])); - let original_rc = Rc::new(original.clone()); + let original_rc = Rc::clone(&original); let target = TargetSubDAG::new(&original_rc); let replacements = AvgToSumOverCountStrategy.replacements(&target); @@ -583,28 +612,26 @@ mod tests { assert!(!replacements[0].rationale.is_empty()); let rewritten = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDag(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let QueryExpr::BinaryOp { lhs, rhs, .. } = rewritten.as_ref() else { + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = rewritten.non_asap() else { panic!("expected sum/count BinaryOp, got {rewritten:?}"); }; - let QueryExpr::Project { child: sum, .. } = lhs.as_ref() else { + let Some(NonASAPOp::Project { child: sum, .. }) = lhs.non_asap() else { panic!("expected cast Project above Sum, got {lhs:?}"); }; - assert!(matches!( - sum.as_ref(), - QueryExpr::Aggregate { measures, .. } + assert!(matches!(sum.non_asap(), + Some(NonASAPOp::Aggregate { measures, .. }) if matches!(measures.as_slice(), [AggIntent::Sum { col: None }]) )); - assert!(matches!( - rhs.as_ref(), - QueryExpr::Aggregate { measures, .. } + assert!(matches!(rhs.non_asap(), + Some(NonASAPOp::Aggregate { measures, .. }) if matches!(measures.as_slice(), [AggIntent::Count { accuracy: AccuracyTarget::Exact }]) )); - let original_schema = original.output_schema().unwrap(); - let rewritten_schema = rewritten.output_schema().unwrap(); + let original_schema = original.schema.clone(); + let rewritten_schema = rewritten.schema.clone(); assert_eq!( original_schema, rewritten_schema, "the rewritten tree must report exactly the same output schema as the original avg" @@ -616,22 +643,24 @@ mod tests { /// synthetic `"avg"` default. #[test] fn preserves_an_explicit_output_name_override() { - let mut q = avg_agg(vec![], None, metric_scan(&[])); - if let QueryExpr::Aggregate { output_names, .. } = &mut q { - *output_names = vec!["avg_latency".to_string()]; - } - let original_schema = q.output_schema().unwrap(); - let q = Rc::new(q); + let q = avg_agg_with( + vec![], + None, + vec!["avg_latency".to_string()], + None, + metric_scan(&[]), + ); + let original_schema = q.schema.clone(); let target = TargetSubDAG::new(&q); let replacements = AvgToSumOverCountStrategy.replacements(&target); let rewritten = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDag(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let rewritten_schema = rewritten.output_schema().unwrap(); + let rewritten_schema = rewritten.schema.clone(); assert_eq!(original_schema, rewritten_schema); - assert_eq!(rewritten_schema.columns[0].name, "avg_latency"); + assert_eq!(rewritten_schema.fields[0].name, "avg_latency"); } /// A grouped rewrite must preserve the aggregate's grouping-key metadata; @@ -639,23 +668,23 @@ mod tests { #[test] fn grouped_avg_rewrite_preserves_the_whole_schema() { let original = avg_agg(vec![2], None, metric_scan(&["job"])); - let original_schema = original.output_schema().unwrap(); - let original_rc = Rc::new(original); + let original_schema = original.schema.clone(); + let original_rc = Rc::clone(&original); let target = TargetSubDAG::new(&original_rc); let replacements = AvgToSumOverCountStrategy.replacements(&target); let rewritten = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDag(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let rewritten_schema = rewritten.output_schema().unwrap(); + let rewritten_schema = rewritten.schema.clone(); assert_eq!(rewritten_schema, original_schema); } #[test] fn default_search_discovers_bindable_sum_and_count_targets() { - let root = Rc::new(avg_agg(vec![2], None, metric_scan(&["job"]))); + let root = avg_agg(vec![2], None, metric_scan(&["job"])); let space = crate::replacement::search_workload(vec![("avg", Rc::clone(&root))]); let avg_group = space @@ -668,7 +697,7 @@ mod tests { let mut found_sum = false; let mut found_count = false; for group in space.target_subdag_candidates() { - let QueryExpr::Aggregate { measures, .. } = group.target.as_ref() else { + let Some(NonASAPOp::Aggregate { measures, .. }) = group.target.non_asap() else { continue; }; let expected = matches!(measures.as_slice(), [AggIntent::Sum { .. }]) @@ -685,7 +714,8 @@ mod tests { group .candidates .iter() - .any(|candidate| matches!(candidate.replacement, Replacement::Summary(_))), + .any(|candidate| matches!(&candidate.replacement, + Replacement::SubDag(node) if node.contains_asap())), "rewritten accumulator must be independently bindable: {measures:?}" ); found_sum |= matches!(measures.as_slice(), [AggIntent::Sum { .. }]); @@ -702,50 +732,51 @@ mod tests { #[test] fn works_with_a_bound_column_not_just_the_sample_value() { let mut schema_cols = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("job", DataType::Utf8, true), - Column::new("bytes", DataType::Int64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("job", DataType::Utf8, true), + Field::plain("bytes", DataType::Int64, false), ]; - let child = QueryExpr::Scan { + let child = OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: { let cols = std::mem::take(&mut schema_cols); Schema::with_time_index(cols, 0, vec![]) }, - }; + }) + .unwrap(); let original = avg_agg(vec![1], Some(2), child); - let original_schema = original.output_schema().unwrap(); - let original_rc = Rc::new(original); + let original_schema = original.schema.clone(); + let original_rc = Rc::clone(&original); let target = TargetSubDAG::new(&original_rc); let replacements = AvgToSumOverCountStrategy.replacements(&target); let rewritten = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDag(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let rewritten_schema = rewritten.output_schema().unwrap(); + let rewritten_schema = rewritten.schema.clone(); // The whole reason for the explicit `Cast` in `build_rewrite`: an // `Int64` input column (`bytes`) makes `Sum`'s own output `Int64` // too, and a bare (uncast) `Int64 / Int64` would type the avg // column `Int64` — this assertion is what would catch that // regression. - assert_eq!(rewritten_schema.columns, original_schema.columns); + assert_eq!(rewritten_schema.fields, original_schema.fields); assert_eq!( - rewritten_schema.columns.last().unwrap().dtype, + rewritten_schema.fields.last().unwrap().dtype, DataType::Float64 ); - let QueryExpr::BinaryOp { lhs, .. } = rewritten.as_ref() else { + let Some(NonASAPOp::BinaryOp { lhs, .. }) = rewritten.non_asap() else { panic!("expected sum/count BinaryOp"); }; - let QueryExpr::Project { cols, .. } = lhs.as_ref() else { + let Some(NonASAPOp::Project { cols, .. }) = lhs.non_asap() else { panic!("expected cast Project above Sum"); }; assert!(matches!( &cols.last().unwrap().expr, - QueryExpr::Cast { + ScalarExpr::Cast { to: DataType::Float64, .. } @@ -754,39 +785,43 @@ mod tests { #[test] fn does_not_rewrite_avg_of_a_nullable_column_via_count_star() { - let child = QueryExpr::Scan { + let child = OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - Column::new("latency", DataType::Float64, true), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("latency", DataType::Float64, true), ], 0, vec![], ), - }; - let q = Rc::new(avg_agg(vec![], Some(2), child)); + }) + .unwrap(); + let q = avg_agg(vec![], Some(2), child); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); } - fn nested_aggregate(outer: AggIntent, inner: AggIntent) -> Rc { - let temporal = QueryExpr::Aggregate { + fn nested_aggregate(outer: AggIntent, inner: AggIntent) -> Rc { + let temporal = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![inner], output_names: vec![], filters: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: OperatorNode::non_asap_node(NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["service"])), - }), - }; - Rc::new(QueryExpr::Aggregate { + child: metric_scan(&["service"]), + }) + .unwrap(), + }) + .unwrap(); + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![outer], // Match the PromQL front end: an empty entry selects the intent's @@ -794,8 +829,9 @@ mod tests { output_names: vec![String::new()], filters: vec![], having: None, - child: Rc::new(temporal), + child: temporal, }) + .unwrap() } #[test] @@ -816,34 +852,30 @@ mod tests { let [candidate] = candidates.as_slice() else { panic!("supported pair should produce exactly one rewrite") }; - let Replacement::Rewrite(rewritten) = &candidate.replacement else { + let Replacement::SubDag(rewritten) = &candidate.replacement else { panic!("expected a logical rewrite") }; - assert_eq!( - original.output_schema().unwrap(), - rewritten.output_schema().unwrap() - ); - let aggregate = match rewritten.as_ref() { - QueryExpr::Aggregate { .. } => rewritten.as_ref(), - QueryExpr::Project { child, .. } => child.as_ref(), + assert_eq!(original.schema.clone(), rewritten.schema.clone()); + let aggregate = match rewritten.non_asap() { + Some(NonASAPOp::Aggregate { .. }) => rewritten.as_ref(), + Some(NonASAPOp::Project { child, .. }) => child.as_ref(), other => panic!("expected Aggregate or cast Project, got {other:?}"), }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), measures, child, .. - } = aggregate + }) = aggregate.non_asap() else { panic!("expected composed cross-entity aggregate") }; assert_eq!(by.keys(), &[2]); assert_eq!(measures, &[expected]); - assert!(matches!( - child.as_ref(), - QueryExpr::TimeRange { range, child } + assert!(matches!(child.non_asap(), + Some(NonASAPOp::TimeRange { range, child, .. }) if *range == Duration::from_secs(300) - && matches!(child.as_ref(), QueryExpr::Scan { .. }) + && matches!(child.non_asap(), Some(NonASAPOp::Scan { .. })) )); } } @@ -876,9 +908,9 @@ mod tests { .iter() .find(|candidate| candidate.strategy == "SemanticEquivalentRewriteStrategy") .expect("default search should run semantic rewrites"); - let Replacement::Rewrite(rewritten) = &candidate.replacement else { + let Replacement::SubDag(rewritten) = &candidate.replacement else { panic!("expected logical rewrite") }; - assert_eq!(rewritten.output_schema().unwrap().columns[1].name, "sum"); + assert_eq!(rewritten.schema.clone().fields[1].name, "sum"); } } diff --git a/crates/asap-aware-mapping/src/rollup.rs b/crates/asap-aware-mapping/src/rollup.rs index a8bd0312f..60d5573a9 100644 --- a/crates/asap-aware-mapping/src/rollup.rs +++ b/crates/asap-aware-mapping/src/rollup.rs @@ -20,7 +20,7 @@ //! "Rolling up aggregations on a fine-grained group by to get a //! coarse-grained group by (like AHA)," alongside "CSE across aggregations, //! and group by key management" — this strategy is the *cross-aggregate* -//! sibling of `pre_asap::cse::share_common_subtrees`'s *identical*-subtree +//! sibling of `pre_asap::cse::share_common_subdags`'s *identical*-sub-DAG //! sharing: CSE shares two structurally-*equal* aggregates onto one `Rc`; //! this strategy relates two structurally-*different* (differently grouped) //! aggregates over the same shared source. @@ -77,7 +77,7 @@ //! this module never reconciles `ColumnId`s across distinct schemas. //! //! ## Non-goals (tracked separately, not attempted here — same split -//! `replacement.rs`'s own module docs draw for `SharedSubtreeStrategy`'s +//! `replacement.rs`'s own module docs draw for `SharedSubDagStrategy`'s //! `consumer_count`) //! //! - **No sibling discovery inside this strategy.** Finding every aggregate @@ -88,7 +88,7 @@ //! - **No materialized roll-up operator.** Actually building a pre-aggregated //! summary/scan leaf at execution time is separate, larger work outside //! `asap-aware-mapping`'s scope (see issue #254's own "Non-goal" section) -//! — this module only constructs the pre-ASAP [`QueryExpr::Aggregate`] +//! — this module only constructs the pre-ASAP `NonASAPOp::Aggregate` //! rewrite; a `CostModel`/search engine decides whether to prefer it. //! - **No cross-schema reconciliation** (see "`ColumnId` comparability" //! above) and **no `without(...)` grouping support** — `without`'s kept @@ -97,34 +97,37 @@ //! against a superset/subset relationship at all; [`is_legal_rollup_source`] //! declines both directions. +use asap_types::ir::non_asap::any_measure_filtered; use std::collections::HashSet; use std::rc::Rc; +use asap_types::ir::operator_properties::{GroupKeys, Reduction}; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{any_measure_filtered, GroupKeys, QueryExpr, Reduction}; use asap_types::pre_asap::schema::{ColumnId, Schema}; + use asap_types::types::AccuracyTarget; use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG}; /// The `(by, intent, child)` shape this strategy operates on: a single /// measure, no `HAVING` — the same bindable shape -/// [`crate::replacement::SketchAlgorithmStrategy`] requires (see that module's +/// [`crate::replacement::ASAPStrategies`] requires (see that module's /// private `bindable_intent`) — **plus** a genuine [`Reduction::Reduce`] /// grouping to compare (not [`Reduction::PerEntity`], which has no `by` set /// at all). `None` for anything else, including a multi-measure or `HAVING` /// aggregate, a non-`Aggregate` node, or a `PerEntity` reduction. fn bindable_grouped_aggregate( - node: &QueryExpr, -) -> Option<(&GroupKeys, &AggIntent, &Rc)> { - let QueryExpr::Aggregate { + node: &OperatorNode, +) -> Option<(&GroupKeys, &AggIntent, &Rc)> { + let Some(NonASAPOp::Aggregate { reduction, measures, filters, having, child, .. - } = node + }) = node.non_asap() else { return None; }; @@ -204,16 +207,16 @@ fn rollup_combinator(intent: &AggIntent, finer_measure_col: ColumnId) -> Option< /// 4. `finer_output_schema` (the finer aggregate's own *output* schema, not /// the shared child's) carries a provable unique key /// ([`Schema::has_unique_key`]) — **the exact legality gate -/// `pre_asap::cse::share_common_subtrees` already applies to its own +/// `pre_asap::cse::share_common_subdags` already applies to its own /// sharing decisions**, reused verbatim here rather than re-invented: -/// `share_common_subtrees`'s own doc ("Legality: gated by +/// `share_common_subdags`'s own doc ("Legality: gated by /// `Schema::unique_keys`") states a producer's output is only safely /// reusable across consumers when its row identity is provably stable — /// exactly the property re-aggregating over `finer` as if it were a /// fresh source requires. /// 5. `coarser_by` is a **strict, proper** subset of `finer_by` (same /// `ColumnId`s, finer strictly more of them) — an *equal* `by` is -/// `SharedSubtreeStrategy`'s CSE-sharing question, not a roll-up, so +/// `SharedSubDagStrategy`'s CSE-sharing question, not a roll-up, so /// equality is deliberately excluded here, not treated as a degenerate /// roll-up. pub fn is_legal_rollup_source( @@ -263,7 +266,7 @@ fn is_strict_column_superset(finer: &[ColumnId], coarser: &[ColumnId]) -> bool { /// docs' "Non-goals" on why finding the full sibling set across a workload /// is a workload-wide traversal this strategy does not own. pub struct RollupStrategy { - siblings: Vec>, + siblings: Vec>, } impl RollupStrategy { @@ -271,7 +274,7 @@ impl RollupStrategy { /// each as a candidate roll-up source (or target) — typically the full set of `Aggregate` /// nodes a workload-wide discovery pass (issue #252) already found /// sharing at least one child `Rc` with something else. - pub fn new(siblings: &[Rc]) -> Self { + pub fn new(siblings: &[Rc]) -> Self { Self { siblings: siblings.to_vec(), } @@ -280,7 +283,7 @@ impl RollupStrategy { /// Every sibling that is a legal, strictly finer roll-up source for /// `target` — shared between `matches` and `replacements` so the two /// can never disagree about which siblings qualify. - fn finer_sources(&self, target: &TargetSubDAG<'_>) -> Vec<&Rc> { + fn finer_sources(&self, target: &TargetSubDAG<'_>) -> Vec<&Rc> { let Some((coarser_by, coarser_intent, coarser_child)) = bindable_grouped_aggregate(target.root) else { @@ -301,12 +304,9 @@ impl RollupStrategy { if !Rc::ptr_eq(finer_child, coarser_child) && finer_child != coarser_child { return false; } - let Ok(finer_schema) = candidate.output_schema() else { - return false; - }; is_legal_rollup_source( finer_by, - &finer_schema, + &candidate.schema, finer_intent, coarser_by, coarser_intent, @@ -325,7 +325,7 @@ impl ReplacementStrategy for RollupStrategy { let Some((coarser_by, coarser_intent, _)) = bindable_grouped_aggregate(target.root) else { return Vec::new(); }; - let QueryExpr::Aggregate { output_names, .. } = target.root.as_ref() else { + let Some(NonASAPOp::Aggregate { output_names, .. }) = target.root.non_asap() else { unreachable!("bindable_grouped_aggregate already confirmed Aggregate"); }; self.finer_sources(target) @@ -335,7 +335,7 @@ impl ReplacementStrategy for RollupStrategy { } } -/// Build the coarser replacement: a new `QueryExpr::Aggregate` grouped by +/// Build the coarser replacement: a new `NonASAPOp::Aggregate` grouped by /// `coarser_by`'s columns (repositioned into `finer`'s own output schema — /// see below), computing `rollup_combinator(intent, ..)` over `finer`'s own /// measure column, with `child = finer` instead of the original shared @@ -350,7 +350,7 @@ impl ReplacementStrategy for RollupStrategy { /// position in the shared child to its position in `finer`'s output: the /// index its `ColumnId` occupies within `finer_by`'s own ordered list. fn build_rollup( - finer: &Rc, + finer: &Rc, coarser_by: &GroupKeys, intent: &AggIntent, output_names: &[String], @@ -367,18 +367,19 @@ fn build_rollup( .map(|id| finer_by.keys().iter().position(|f| f == id)) .collect::>>()?; - let rewritten = QueryExpr::Aggregate { + let rewritten = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(remapped_by), measures: vec![combinator], output_names: output_names.to_vec(), filters: vec![], having: None, child: Rc::clone(finer), - }; + }) + .ok()?; Some(ReplacementSubDAG { strategy: "RollupStrategy", - replacement: Replacement::Rewrite(Rc::new(rewritten)), + replacement: Replacement::SubDag(rewritten), provenance: crate::replacement::ReplacementProvenance::LogicalRewrite, rationale: format!( "rolls up from the finer Aggregate grouped by {:?} (a strict superset of this \ @@ -394,30 +395,31 @@ fn build_rollup( #[cfg(test)] mod tests { use super::*; - use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{Column, DataType}; + use asap_types::ir::operator_properties::Source; + use asap_types::pre_asap::schema::{DataType, Field}; use asap_types::types::AccuracyTarget; /// `[ts(0), value(1), job(2), region(3)]`. - fn metric_scan() -> QueryExpr { - QueryExpr::Scan { + fn metric_scan() -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - Column::new("job", DataType::Utf8, true), - Column::new("region", DataType::Utf8, true), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("job", DataType::Utf8, true), + Field::plain("region", DataType::Utf8, true), ], 0, vec![], ), - } + }) + .unwrap() } - fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], @@ -425,14 +427,15 @@ mod tests { having: None, child: Rc::clone(child), }) + .unwrap() } fn without_agg( excluded: Vec, intent: AggIntent, - child: &Rc, - ) -> Rc { - Rc::new(QueryExpr::Aggregate { + child: &Rc, + ) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(excluded)), measures: vec![intent], output_names: vec![], @@ -440,6 +443,7 @@ mod tests { having: None, child: Rc::clone(child), }) + .unwrap() } // ── is_legal_rollup_source (the standalone predicate) ─────────────── @@ -511,7 +515,7 @@ mod tests { #[test] fn predicate_rejects_equal_by_sets() { - // Equality is `SharedSubtreeStrategy`'s question, not a roll-up. + // Equality is `SharedSubDagStrategy`'s question, not a roll-up. let finer_schema = Schema::with_time_index(vec![], 0, vec![vec![0]]); assert!(!is_legal_rollup_source( &GroupKeys::by(vec![2]), @@ -569,7 +573,7 @@ mod tests { #[test] fn superset_by_over_identical_mergeable_intent_and_shared_child_rolls_up() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); let coarse = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); @@ -581,16 +585,16 @@ mod tests { let replacements = strategy.replacements(&target); assert_eq!(replacements.len(), 1, "{replacements:?}"); - let Replacement::Rewrite(rewritten) = &replacements[0].replacement else { + let Replacement::SubDag(rewritten) = &replacements[0].replacement else { panic!("expected a Rewrite replacement"); }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, measures, child, having, .. - } = rewritten.as_ref() + }) = rewritten.non_asap() else { panic!("expected an Aggregate rewrite, got {rewritten:?}"); }; @@ -620,7 +624,7 @@ mod tests { // Count is not self-combining (see the module docs) — the rewritten // measure must be Sum over the finer Count's own output column, not // Count reapplied. - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg( vec![2, 3], AggIntent::Count { @@ -642,10 +646,10 @@ mod tests { let replacements = strategy.replacements(&target); assert_eq!(replacements.len(), 1, "{replacements:?}"); - let Replacement::Rewrite(rewritten) = &replacements[0].replacement else { + let Replacement::SubDag(rewritten) = &replacements[0].replacement else { panic!("expected a Rewrite replacement"); }; - let QueryExpr::Aggregate { measures, .. } = rewritten.as_ref() else { + let Some(NonASAPOp::Aggregate { measures, .. }) = rewritten.non_asap() else { panic!("expected an Aggregate rewrite"); }; assert_eq!( @@ -657,7 +661,7 @@ mod tests { #[test] fn approximate_count_does_not_roll_up_via_sum() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let intent = AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.01), }; @@ -673,8 +677,8 @@ mod tests { #[test] fn default_workload_search_adds_rollup_for_two_query_workload() { - let fine_scan = Rc::new(metric_scan()); - let coarse_scan = Rc::new(metric_scan()); + let fine_scan = metric_scan(); + let coarse_scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &fine_scan); let coarse = agg(vec![2], AggIntent::Sum { col: Some(1) }, &coarse_scan); @@ -682,12 +686,11 @@ mod tests { let coarse_group = space .target_subdag_candidates() .find(|group| { - matches!( - group.target.as_ref(), - QueryExpr::Aggregate { + matches!(group.target.non_asap(), + Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), .. - } if by.keys() == [2] + }) if by.keys() == [2] ) }) .expect("coarser aggregate group"); @@ -696,19 +699,19 @@ mod tests { .candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Rewrite(rewrite) => Some(rewrite), - Replacement::Summary(_) | Replacement::ExactComposition(_) => None, + // Old `Replacement::Rewrite`: a pure pre-ASAP sub-DAG. + Replacement::SubDag(rewrite) if !rewrite.contains_asap() => Some(rewrite), + Replacement::SubDag(_) | Replacement::ExactComposition(_) => None, }) .expect("default search must include the roll-up rewrite"); - let QueryExpr::Aggregate { child, .. } = rewrite.as_ref() else { + let Some(NonASAPOp::Aggregate { child, .. }) = rewrite.non_asap() else { panic!("expected aggregate rewrite, got {rewrite:?}"); }; - assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { + assert!(matches!(child.non_asap(), + Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), .. - } if by.keys() == [2, 3] + }) if by.keys() == [2, 3] )); } @@ -717,18 +720,17 @@ mod tests { let intent = AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.01), }; - let fine = agg(vec![2, 3], intent.clone(), &Rc::new(metric_scan())); - let coarse = agg(vec![2], intent, &Rc::new(metric_scan())); + let fine = agg(vec![2, 3], intent.clone(), &metric_scan()); + let coarse = agg(vec![2], intent, &metric_scan()); let space = crate::replacement::search_workload(vec![("fine", fine), ("coarse", coarse)]); let coarse_group = space .target_subdag_candidates() .find(|group| { - matches!( - group.target.as_ref(), - QueryExpr::Aggregate { + matches!(group.target.non_asap(), + Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), .. - } if by.keys() == [2] + }) if by.keys() == [2] ) }) .expect("coarser aggregate group"); @@ -736,32 +738,34 @@ mod tests { assert!(coarse_group .candidates .iter() - .all(|candidate| !matches!(candidate.replacement, Replacement::Rewrite(_)))); + .all(|candidate| !matches!(&candidate.replacement, + Replacement::SubDag(rewrite) if !rewrite.contains_asap()))); } #[test] fn rollup_preserves_the_coarser_output_name() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); - let coarse = Rc::new(QueryExpr::Aggregate { + let coarse = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![AggIntent::Sum { col: Some(1) }], output_names: vec!["total_requests".into()], filters: vec![], having: None, child: Rc::clone(&scan), - }); - let original_schema = coarse.output_schema().unwrap(); + }) + .unwrap(); + let original_schema = coarse.schema.clone(); let siblings = vec![Rc::clone(&fine), Rc::clone(&coarse)]; let strategy = RollupStrategy::new(&siblings); let replacements = strategy.replacements(&TargetSubDAG::new(&coarse)); - let Replacement::Rewrite(rewritten) = &replacements[0].replacement else { + let Replacement::SubDag(rewritten) = &replacements[0].replacement else { panic!("expected a Rewrite replacement"); }; - assert_eq!(rewritten.output_schema().unwrap(), original_schema); - let QueryExpr::Aggregate { output_names, .. } = rewritten.as_ref() else { + assert_eq!(rewritten.schema.clone(), original_schema); + let Some(NonASAPOp::Aggregate { output_names, .. }) = rewritten.non_asap() else { unreachable!(); }; assert_eq!(output_names, &vec!["total_requests".to_string()]); @@ -769,7 +773,7 @@ mod tests { #[test] fn non_mergeable_intent_does_not_roll_up() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Avg { col: Some(1) }, &scan); let coarse = agg(vec![2], AggIntent::Avg { col: Some(1) }, &scan); @@ -789,12 +793,12 @@ mod tests { // numerically a superset of the coarser side's *kept* positions — // `is_legal_rollup_source` rejects any `without` grouping outright, // and would reject on the missing unique key regardless. - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = without_agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); let coarse = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); assert!( - !fine.output_schema().unwrap().has_unique_key(), + !fine.schema.clone().has_unique_key(), "fixture sanity: a without(...) aggregate has no provable unique key" ); @@ -809,7 +813,7 @@ mod tests { #[test] fn unrelated_by_sets_do_not_roll_up() { // Neither `[job]` nor `[region]` is a superset of the other. - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let a = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); let b = agg(vec![3], AggIntent::Sum { col: Some(1) }, &scan); @@ -827,10 +831,10 @@ mod tests { #[test] fn equal_by_sets_do_not_roll_up() { - // Equal groupings are `SharedSubtreeStrategy`'s CSE-sharing + // Equal groupings are `SharedSubDagStrategy`'s CSE-sharing // question (build once and share, or build independently) — a // roll-up requires a *strict* superset, not equality. - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let a = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); let b = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); @@ -845,16 +849,8 @@ mod tests { // Scans without unique keys are deliberately not pointer-aliased by // CSE. Structural equality still proves identical schemas and makes // the two aggregates' positional ColumnIds comparable. - let fine = agg( - vec![2, 3], - AggIntent::Sum { col: Some(1) }, - &Rc::new(metric_scan()), - ); - let coarse = agg( - vec![2], - AggIntent::Sum { col: Some(1) }, - &Rc::new(metric_scan()), - ); + let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &metric_scan()); + let coarse = agg(vec![2], AggIntent::Sum { col: Some(1) }, &metric_scan()); let siblings = vec![Rc::clone(&fine), Rc::clone(&coarse)]; let strategy = RollupStrategy::new(&siblings); @@ -865,9 +861,9 @@ mod tests { #[test] fn does_not_match_a_multi_measure_or_having_aggregate() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); - let multi = Rc::new(QueryExpr::Aggregate { + let multi = OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![ AggIntent::Sum { col: Some(1) }, @@ -879,7 +875,8 @@ mod tests { filters: vec![], having: None, child: Rc::clone(&scan), - }); + }) + .unwrap(); let siblings = vec![Rc::clone(&fine), Rc::clone(&multi)]; let strategy = RollupStrategy::new(&siblings); diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs index e31606404..2ea1137f9 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs @@ -1,7 +1,7 @@ use super::*; pub(super) fn estimate_heterogeneous_summary( - root: &SummaryNode, + root: &OperatorNode, deployments: &[CostedSummaryDeployment<'_>], evidence: &SummaryNodeEvidence, scope: &ComparisonScope, @@ -19,46 +19,6 @@ pub(super) fn estimate_heterogeneous_summary( .map(|(deployment, framework)| (deployment.summary as *const _, framework)) .collect(); validate_summary_edges_and_physical_ids(root, evidence, &frameworks_by_node)?; - fn summary_source_selections( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - out: &mut Vec, - ) -> Result<(), AnalyticalCostError> { - if !seen.insert(node as *const _) { - return Ok(()); - } - match &node.expr { - SummaryExpr::KeepPreAsap(query) => query_source_selections(query, out)?, - SummaryExpr::SummaryAgg { child, .. } | SummaryExpr::ValueOperation { child, .. } => { - summary_source_selections(child, seen, out)? - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - summary_source_selections(child, seen, out)?; - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - summary_source_selections(left, seen, out)?; - summary_source_selections(right, seen, out)?; - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - summary_source_selections(summary_input, seen, out)? - } - } - Ok(()) - } let evaluation_count = scope.validate()?; let by_node: HashMap<_, _> = deployments .iter() @@ -81,7 +41,7 @@ pub(super) fn estimate_heterogeneous_summary( let node_evidence = evidence .aggregation(deployment.summary) .ok_or(AnalyticalCostError::MissingOrStale("summary_agg"))?; - let SummaryExpr::SummaryAgg { child, .. } = &deployment.summary.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &deployment.summary.operator else { return Err(AnalyticalCostError::UnsupportedCandidate); }; let inputs = node_evidence.inputs.validate()?; @@ -95,7 +55,7 @@ pub(super) fn estimate_heterogeneous_summary( .ok_or(AnalyticalCostError::MissingComparisonScope( "summary source coverage", ))?; - if !matches!(&child.expr, SummaryExpr::KeepPreAsap(_)) + if !has_retained_subdag_evidence(child, evidence) || inputs.initial_input_rows != raw.planning_time_input_rows || inputs.initial_input_bytes != raw.planning_time_input_bytes || inputs.initial_source_scan_bytes != raw.planning_time_source_scan_bytes @@ -106,7 +66,7 @@ pub(super) fn estimate_heterogeneous_summary( )); } let mut actual_selections = Vec::new(); - summary_source_selections(child, &mut HashSet::new(), &mut actual_selections)?; + query_source_selections(child, &mut HashSet::new(), &mut actual_selections)?; let actual_selections = deduplicate_source_selections(actual_selections); let expected = ( declared.source.clone(), @@ -120,7 +80,7 @@ pub(super) fn estimate_heterogeneous_summary( } } None => { - if matches!(&child.expr, SummaryExpr::KeepPreAsap(_)) + if has_retained_subdag_evidence(child, evidence) || inputs.initial_source_scan_bytes != 0 || !node_evidence.bootstrap_read_identity.is_empty() { @@ -131,7 +91,7 @@ pub(super) fn estimate_heterogeneous_summary( } } validate_guarantee(deployment.guarantee, scope.data_arrival)?; - let logical_state = format!("{:?}", deployment.summary.expr); + let logical_state = format!("{:?}", deployment.summary.operator); let window_framework = (*frameworks_by_node .get(&(deployment.summary as *const _)) .ok_or(AnalyticalCostError::MissingOrStale( @@ -253,9 +213,9 @@ pub(super) fn estimate_heterogeneous_summary( #[expect(clippy::too_many_arguments, reason = "CPU and I/O traversal state")] fn visit_ops( - node: &SummaryNode, + node: &OperatorNode, seen: &mut HashSet, - by_node: &HashMap<*const SummaryNode, &CostedSummaryDeployment<'_>>, + by_node: &HashMap<*const OperatorNode, &CostedSummaryDeployment<'_>>, evidence: &SummaryNodeEvidence, scope: &ComparisonScope, evaluation_count: u64, @@ -266,13 +226,29 @@ pub(super) fn estimate_heterogeneous_summary( if !seen.insert(physical_id) { return Ok(()); } - match &node.expr { - SummaryExpr::BinaryOp { lhs, rhs, .. } - | SummaryExpr::RelationalJoin { + if has_retained_subdag_evidence(node, evidence) { + let retained = evidence + .retained_queries + .get(&(node as *const _)) + .ok_or(AnalyticalCostError::MissingOrStale("retain_exact"))?; + if !retained.preprocessing_cpu_ops_over_horizon.is_finite() + || retained.preprocessing_cpu_ops_over_horizon < 0.0 + { + return Err(AnalyticalCostError::InvalidOperationCost( + "retain_exact", + retained.preprocessing_cpu_ops_over_horizon, + )); + } + *cpu_ops += retained.preprocessing_cpu_ops_over_horizon; + return Ok(()); + } + match &node.operator { + Operator::NonASAP(NonASAPOp::BinaryOp { lhs, rhs, .. }) + | Operator::NonASAP(NonASAPOp::Join { left: lhs, right: rhs, .. - } => { + }) => { let operation = summary_operation_evidence(node, evidence)?.resource(); *cpu_ops += evaluation_count as f64 * validated_operator_executions("exact_binary", operation)? as f64 @@ -300,39 +276,31 @@ pub(super) fn estimate_heterogeneous_summary( )?; } - SummaryExpr::ValueOperation { child, .. } => { + Operator::NonASAP(_) + | Operator::ASAP( + ASAPOp::FinalizeExactAccumulator { .. } + | ASAPOp::MaintainPopulation { .. } + | ASAPOp::EvaluatePopulation { .. }, + ) => { let operation = summary_operation_evidence(node, evidence)?.resource(); *cpu_ops += evaluation_count as f64 * validated_operator_executions("value_operation", operation)? as f64 * validated_operator_cpu("value_operation", operation.cpu_ops)?; add_operator_io(io_bytes, operation, evaluation_count)?; - visit_ops( - child, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - } - SummaryExpr::KeepPreAsap(_) => { - let retained = evidence - .retained_queries - .get(&(node as *const _)) - .ok_or(AnalyticalCostError::MissingOrStale("keep_pre_asap"))?; - if !retained.preprocessing_cpu_ops_over_horizon.is_finite() - || retained.preprocessing_cpu_ops_over_horizon < 0.0 - { - return Err(AnalyticalCostError::InvalidOperationCost( - "keep_pre_asap", - retained.preprocessing_cpu_ops_over_horizon, - )); + for child in node.children() { + visit_ops( + child, + seen, + by_node, + evidence, + scope, + evaluation_count, + cpu_ops, + io_bytes, + )?; } - *cpu_ops += retained.preprocessing_cpu_ops_over_horizon; } - SummaryExpr::SummaryAgg { child, .. } => { + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) => { visit_ops( child, seen, @@ -344,7 +312,7 @@ pub(super) fn estimate_heterogeneous_summary( io_bytes, )?; } - SummaryExpr::SummaryMerge { children, .. } => { + Operator::ASAP(ASAPOp::SummaryMerge { children }) => { let operation = summary_operation_evidence(node, evidence)?.resource(); let merge = validated_operator_cpu("summary_merge", operation.cpu_ops)?; *cpu_ops += evaluation_count as f64 @@ -364,7 +332,7 @@ pub(super) fn estimate_heterogeneous_summary( )?; } } - SummaryExpr::SummarySubtract { left, right } => { + Operator::ASAP(ASAPOp::SummarySubtract { left, right }) => { let operation = summary_operation_evidence(node, evidence)?.resource(); *cpu_ops += evaluation_count as f64 * validated_operator_executions("summary_subtract", operation)? as f64 @@ -391,7 +359,7 @@ pub(super) fn estimate_heterogeneous_summary( io_bytes, )?; } - SummaryExpr::SummaryDelete { summary_input, .. } => { + Operator::ASAP(ASAPOp::SummaryDelete { summary_input, .. }) => { let delete = summary_operation_evidence(node, evidence)?; let SummaryOperatorEvidence::Delete { resource: operation, @@ -406,44 +374,18 @@ pub(super) fn estimate_heterogeneous_summary( .get(&(node as *const _)) .ok_or(AnalyticalCostError::MissingOrStale("summary_delete_owner"))?; fn collect_aggs( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - out: &mut Vec<*const SummaryNode>, + node: &OperatorNode, + seen: &mut HashSet<*const OperatorNode>, + out: &mut Vec<*const OperatorNode>, ) { if !seen.insert(node as *const _) { return; } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } => { - out.push(node as *const _); - collect_aggs(child, seen, out); - } - SummaryExpr::ValueOperation { child, .. } => collect_aggs(child, seen, out), - SummaryExpr::SummaryMerge { children, .. } => { - children - .iter() - .for_each(|child| collect_aggs(child, seen, out)); - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - collect_aggs(left, seen, out); - collect_aggs(right, seen, out); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - collect_aggs(summary_input, seen, out) - } - SummaryExpr::KeepPreAsap(_) => {} + if matches!(node.operator, Operator::ASAP(ASAPOp::SummaryAgg { .. })) { + out.push(node as *const _); + } + for child in node.children() { + collect_aggs(child, seen, out); } } let mut reachable = Vec::new(); @@ -491,11 +433,11 @@ pub(super) fn estimate_heterogeneous_summary( io_bytes, )?; } - SummaryExpr::SummaryEstimate { summary_input, .. } => { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { let operation = summary_operation_evidence(node, evidence)?.resource(); *cpu_ops += evaluation_count as f64 - * validated_operator_executions("summary_readout", operation)? as f64 - * validated_operator_cpu("summary_readout", operation.cpu_ops)?; + * validated_operator_executions("summary_evaluation", operation)? as f64 + * validated_operator_cpu("summary_evaluation", operation.cpu_ops)?; add_operator_io(io_bytes, operation, evaluation_count)?; visit_ops( summary_input, @@ -508,7 +450,7 @@ pub(super) fn estimate_heterogeneous_summary( io_bytes, )?; } - SummaryExpr::SummaryJoin { outer, inner, .. } => { + Operator::ASAP(ASAPOp::SummaryJoin { outer, inner, .. }) => { let join = evidence .joins .get(&(node as *const _)) @@ -555,6 +497,9 @@ pub(super) fn estimate_heterogeneous_summary( io_bytes, )?; } + Operator::ASAP(ASAPOp::Extension { .. }) => { + return Err(AnalyticalCostError::UnsupportedCandidate); + } } Ok(()) } @@ -613,55 +558,37 @@ fn add_operator_io( } fn validate_summary_edges_and_physical_ids( - root: &SummaryNode, + root: &OperatorNode, evidence: &SummaryNodeEvidence, - frameworks_by_node: &HashMap<*const SummaryNode, &Option>, + frameworks_by_node: &HashMap<*const OperatorNode, &Option>, ) -> Result<(), AnalyticalCostError> { - fn children(node: &SummaryNode) -> Vec<&SummaryNode> { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::SummaryAgg { child, .. } | SummaryExpr::ValueOperation { child, .. } => { - vec![child] - } - SummaryExpr::SummaryMerge { children, .. } => { - children.iter().map(|child| child.as_ref()).collect() - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => vec![left, right], - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => vec![summary_input], - } + fn children<'a>( + node: &'a OperatorNode, + evidence: &SummaryNodeEvidence, + ) -> Vec<&'a OperatorNode> { + summary_children(node, evidence) } fn metadata( - node: &SummaryNode, + node: &OperatorNode, evidence: &SummaryNodeEvidence, ) -> Result<(String, Vec, EdgeStatistics), AnalyticalCostError> { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => { - let retained = evidence - .retained_queries - .get(&(node as *const _)) - .ok_or(AnalyticalCostError::MissingOrStale("keep_pre_asap"))?; - Ok((retained.physical_id.clone(), vec![], retained.output)) + if let Some(retained) = evidence.retained_queries.get(&(node as *const _)) { + if node.contains_asap() { + return Err(AnalyticalCostError::InvalidPhysicalDag( + "retained sub-DAG evidence covers summary operators", + )); } - SummaryExpr::SummaryAgg { .. } => { + return Ok((retained.physical_id.clone(), vec![], retained.output)); + } + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => { let value = evidence .aggregations .get(&(node as *const _)) .ok_or(AnalyticalCostError::MissingOrStale("summary_agg"))?; Ok((value.physical_id.clone(), vec![value.input], value.output)) } - SummaryExpr::SummaryJoin { .. } => { + Operator::ASAP(ASAPOp::SummaryJoin { .. }) => { let value = evidence .joins .get(&(node as *const _)) @@ -683,16 +610,16 @@ fn validate_summary_edges_and_physical_ids( } } fn visit( - node: &SummaryNode, + node: &OperatorNode, evidence: &SummaryNodeEvidence, - frameworks_by_node: &HashMap<*const SummaryNode, &Option>, - seen: &mut HashSet<*const SummaryNode>, + frameworks_by_node: &HashMap<*const OperatorNode, &Option>, + seen: &mut HashSet<*const OperatorNode>, physical: &mut HashMap, EdgeStatistics, String)>, ) -> Result { if !seen.insert(node as *const _) { return metadata(node, evidence).map(|(_, _, output)| output); } - let child_nodes = children(node); + let child_nodes = children(node, evidence); let child_outputs = child_nodes .iter() .map(|child| visit(child, evidence, frameworks_by_node, seen, physical)) @@ -702,17 +629,18 @@ fn validate_summary_edges_and_physical_ids( .map(|child| summary_physical_id(child, evidence)) .collect::, _>>()?; let (id, inputs, output) = metadata(node, evidence)?; - let local_fingerprint = match &node.expr { - SummaryExpr::KeepPreAsap(_) => { - format!("{:?}", evidence.retained_queries.get(&(node as *const _))) - } - SummaryExpr::SummaryAgg { .. } => { - format!("{:?}", evidence.aggregations.get(&(node as *const _))) - } - SummaryExpr::SummaryJoin { .. } => { - format!("{:?}", evidence.joins.get(&(node as *const _))) + let local_fingerprint = if has_retained_subdag_evidence(node, evidence) { + format!("{:?}", evidence.retained_queries.get(&(node as *const _))) + } else { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => { + format!("{:?}", evidence.aggregations.get(&(node as *const _))) + } + Operator::ASAP(ASAPOp::SummaryJoin { .. }) => { + format!("{:?}", evidence.joins.get(&(node as *const _))) + } + _ => format!("{:?}", evidence.operations.get(&(node as *const _))), } - _ => format!("{:?}", evidence.operations.get(&(node as *const _))), }; // A provider identity names the complete physical operator, including // its inputs. Equal local widths/costs do not make operators consuming @@ -720,7 +648,7 @@ fn validate_summary_edges_and_physical_ids( let framework = frameworks_by_node.get(&(node as *const _)); let fingerprint = format!( "logical={:?}|framework={framework:?}|{local_fingerprint}|children={child_physical_ids:?}", - node.expr + node.operator ); if id.is_empty() || inputs != child_outputs @@ -756,20 +684,49 @@ fn validate_summary_edges_and_physical_ids( .map(|_| ()) } +/// Whether `node` is a retained non-ASAP sub-DAG costed as one unit: the +/// provider bound retained-query evidence to it instead of per-operator +/// evidence. Its children are then not visited. No ASAP descendant may be +/// hidden by this boundary; graph validation rejects such evidence. +fn has_retained_subdag_evidence(node: &OperatorNode, evidence: &SummaryNodeEvidence) -> bool { + evidence.retained_queries.contains_key(&(node as *const _)) && !node.contains_asap() +} + +/// The inputs the estimator visits below `node`: none for a retained +/// sub-DAG, every direct input otherwise. +fn summary_children<'a>( + node: &'a OperatorNode, + evidence: &SummaryNodeEvidence, +) -> Vec<&'a OperatorNode> { + if has_retained_subdag_evidence(node, evidence) { + vec![] + } else { + node.children() + .into_iter() + .map(|child| child.as_ref()) + .collect() + } +} + fn summary_physical_id( - node: &SummaryNode, + node: &OperatorNode, evidence: &SummaryNodeEvidence, ) -> Result { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => evidence + if has_retained_subdag_evidence(node, evidence) { + return evidence .retained_queries .get(&(node as *const _)) - .map(|value| value.physical_id.clone()), - SummaryExpr::SummaryAgg { .. } => evidence + .map(|value| value.physical_id.clone()) + .ok_or(AnalyticalCostError::MissingOrStale( + "summary physical identity", + )); + } + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => evidence .aggregations .get(&(node as *const _)) .map(|value| value.physical_id.clone()), - SummaryExpr::SummaryJoin { .. } => evidence + Operator::ASAP(ASAPOp::SummaryJoin { .. }) => evidence .joins .get(&(node as *const _)) .map(|value| value.physical_id.clone()), @@ -786,45 +743,26 @@ fn summary_physical_id( /// child output buffers remain live until their final consumer executes; /// operator workspace and its output buffer coexist during that execution. pub(super) fn estimate_transient_liveness( - root: &SummaryNode, + root: &OperatorNode, evidence: &SummaryNodeEvidence, ) -> Result { - fn children(node: &SummaryNode) -> Vec<&SummaryNode> { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::SummaryAgg { child, .. } | SummaryExpr::ValueOperation { child, .. } => { - vec![child] - } - SummaryExpr::SummaryMerge { children, .. } => { - children.iter().map(|child| child.as_ref()).collect() - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => vec![left, right], - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => vec![summary_input], - } + fn children<'a>( + node: &'a OperatorNode, + evidence: &SummaryNodeEvidence, + ) -> Vec<&'a OperatorNode> { + summary_children(node, evidence) } fn visit<'a>( - node: &'a SummaryNode, + node: &'a OperatorNode, evidence: &SummaryNodeEvidence, seen: &mut HashSet, uses: &mut HashMap, - order: &mut Vec<&'a SummaryNode>, + order: &mut Vec<&'a OperatorNode>, ) -> Result<(), AnalyticalCostError> { if !seen.insert(summary_physical_id(node, evidence)?) { return Ok(()); } - for child in children(node) { + for child in children(node, evidence) { *uses .entry(summary_physical_id(child, evidence)?) .or_default() += 1; @@ -834,28 +772,24 @@ pub(super) fn estimate_transient_liveness( Ok(()) } fn memory( - node: &SummaryNode, + node: &OperatorNode, evidence: &SummaryNodeEvidence, ) -> Result<(u64, u64), AnalyticalCostError> { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => evidence + if has_retained_subdag_evidence(node, evidence) { + return evidence .retained_queries .get(&(node as *const _)) .map(|value| (value.working_memory_bytes, value.output_buffer_bytes)) - .ok_or(AnalyticalCostError::MissingOrStale("keep_pre_asap")), - SummaryExpr::SummaryAgg { .. } => Ok((0, 0)), - SummaryExpr::SummaryJoin { .. } => evidence + .ok_or(AnalyticalCostError::MissingOrStale("retain_exact")); + } + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => Ok((0, 0)), + Operator::ASAP(ASAPOp::SummaryJoin { .. }) => evidence .joins .get(&(node as *const _)) .map(|value| (value.working_memory_bytes, value.output_buffer_bytes)) .ok_or(AnalyticalCostError::MissingOrStale("summary_join")), - SummaryExpr::SummaryMerge { .. } - | SummaryExpr::BinaryOp { .. } - | SummaryExpr::RelationalJoin { .. } - | SummaryExpr::ValueOperation { .. } - | SummaryExpr::SummarySubtract { .. } - | SummaryExpr::SummaryDelete { .. } - | SummaryExpr::SummaryEstimate { .. } => { + _ => { let value = summary_operation_evidence(node, evidence)?.resource(); Ok((value.working_memory_bytes, value.output_buffer_bytes)) } @@ -884,7 +818,7 @@ pub(super) fn estimate_transient_liveness( live = live .checked_add(output) .ok_or(AnalyticalCostError::Overflow)?; - for child in children(node) { + for child in children(node, evidence) { let child_id = summary_physical_id(child, evidence)?; let remaining = uses.get_mut(&child_id) @@ -902,50 +836,23 @@ pub(super) fn estimate_transient_liveness( Ok(peak) } #[cfg(test)] -pub(super) fn evidence_nodes(root: &SummaryNode) -> (Vec<&SummaryNode>, Vec<&SummaryNode>) { +pub(super) fn evidence_nodes(root: &OperatorNode) -> (Vec<&OperatorNode>, Vec<&OperatorNode>) { fn visit<'a>( - node: &'a SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - aggregations: &mut Vec<&'a SummaryNode>, - joins: &mut Vec<&'a SummaryNode>, + node: &'a OperatorNode, + seen: &mut HashSet<*const OperatorNode>, + aggregations: &mut Vec<&'a OperatorNode>, + joins: &mut Vec<&'a OperatorNode>, ) { if !seen.insert(node as *const _) { return; } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } => { - aggregations.push(node); - visit(child, seen, aggregations, joins); - } - SummaryExpr::ValueOperation { child, .. } => visit(child, seen, aggregations, joins), - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - visit(child, seen, aggregations, joins); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - if matches!(&node.expr, SummaryExpr::SummaryJoin { .. }) { - joins.push(node); - } - visit(left, seen, aggregations, joins); - visit(right, seen, aggregations, joins); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - visit(summary_input, seen, aggregations, joins); - } - SummaryExpr::KeepPreAsap(_) => {} + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => aggregations.push(node), + Operator::ASAP(ASAPOp::SummaryJoin { .. }) => joins.push(node), + _ => {} + } + for child in node.children() { + visit(child, seen, aggregations, joins); } } let mut aggregations = Vec::new(); @@ -961,7 +868,7 @@ struct SummaryOperationCounts { merges_per_read: u64, subtracts_per_read: u64, deletes_per_update: u64, - readouts_per_read: u64, + evaluations_per_read: u64, joins_per_read: u64, } @@ -970,7 +877,7 @@ struct SummaryOperationCounts { /// once; explicit delete frequency comes from deletion evidence. #[cfg(test)] pub(super) fn estimate_incremental_summary_maintenance( - root: &SummaryNode, + root: &OperatorNode, guarantee: &SummaryMaintenanceLifecycleGuarantee, inputs: SummaryMaintenanceInputs, cpu: SummaryOperationCpuEvidence, @@ -980,7 +887,7 @@ pub(super) fn estimate_incremental_summary_maintenance( } #[cfg(test)] pub(super) fn estimate_incremental_summary_maintenance_with_join( - root: &SummaryNode, + root: &OperatorNode, guarantee: &SummaryMaintenanceLifecycleGuarantee, inputs: SummaryMaintenanceInputs, cpu: SummaryOperationCpuEvidence, @@ -1030,10 +937,10 @@ pub(super) fn estimate_incremental_summary_maintenance_with_join( .checked_mul(fanout) .ok_or(AnalyticalCostError::Overflow)? }; - let readout = required_cpu_when( - counts.readouts_per_read, - "readout_cpu_ops", - cpu.readout_cpu_ops, + let evaluation = required_cpu_when( + counts.evaluations_per_read, + "evaluation_cpu_ops", + cpu.evaluation_cpu_ops, )?; let join_cpu = match (counts.joins_per_read, join.as_ref()) { (0, _) => 0.0, @@ -1071,7 +978,7 @@ pub(super) fn estimate_incremental_summary_maintenance_with_join( + evaluations * counts.merges_per_read as f64 * instances * merge + evaluations * counts.subtracts_per_read as f64 * instances * subtract + delete_events as f64 * counts.deletes_per_update as f64 * delete - + evaluations * counts.readouts_per_read as f64 * instances * readout + + evaluations * counts.evaluations_per_read as f64 * instances * evaluation + evaluations * counts.joins_per_read as f64 * join_cpu; if !cpu_ops.is_finite() { return Err(AnalyticalCostError::Overflow); @@ -1238,25 +1145,23 @@ fn required_cpu_when( } #[cfg(test)] -fn count_operations(root: &SummaryNode) -> Result { +fn count_operations(root: &OperatorNode) -> Result { fn visit( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, + node: &OperatorNode, + seen: &mut HashSet<*const OperatorNode>, counts: &mut SummaryOperationCounts, ) -> Result<(), AnalyticalCostError> { - if !seen.insert(node as *const SummaryNode) { + if !seen.insert(node as *const OperatorNode) { return Ok(()); } - match &node.expr { - SummaryExpr::KeepPreAsap(_) => {} - SummaryExpr::SummaryAgg { child, .. } => { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => { counts.state_builds = counts .state_builds .checked_add(1) .ok_or(AnalyticalCostError::Overflow)?; - visit(child, seen, counts)?; } - SummaryExpr::SummaryMerge { children, .. } => { + Operator::ASAP(ASAPOp::SummaryMerge { children }) => { if children.is_empty() { return Err(AnalyticalCostError::InvalidPhysicalDag( "summary merge has no children", @@ -1266,50 +1171,44 @@ fn count_operations(root: &SummaryNode) -> Result { + Operator::ASAP(ASAPOp::SummarySubtract { .. }) => { counts.subtracts_per_read = counts .subtracts_per_read .checked_add(1) .ok_or(AnalyticalCostError::Overflow)?; - visit(left, seen, counts)?; - visit(right, seen, counts)?; - } - SummaryExpr::BinaryOp { lhs, rhs, .. } => { - visit(lhs, seen, counts)?; - visit(rhs, seen, counts)?; } - SummaryExpr::RelationalJoin { left, right, .. } => { - visit(left, seen, counts)?; - visit(right, seen, counts)?; - } - - SummaryExpr::ValueOperation { child, .. } => visit(child, seen, counts)?, - SummaryExpr::SummaryDelete { summary_input, .. } => { + Operator::ASAP(ASAPOp::SummaryDelete { .. }) => { counts.deletes_per_update = counts .deletes_per_update .checked_add(1) .ok_or(AnalyticalCostError::Overflow)?; - visit(summary_input, seen, counts)?; } - SummaryExpr::SummaryEstimate { summary_input, .. } => { - counts.readouts_per_read = counts - .readouts_per_read + Operator::ASAP( + ASAPOp::SummaryEstimate { .. } | ASAPOp::FinalizeExactAccumulator { .. }, + ) => { + counts.evaluations_per_read = counts + .evaluations_per_read .checked_add(1) .ok_or(AnalyticalCostError::Overflow)?; - visit(summary_input, seen, counts)?; } - SummaryExpr::SummaryJoin { outer, inner, .. } => { + Operator::ASAP(ASAPOp::SummaryJoin { .. }) => { counts.joins_per_read = counts .joins_per_read .checked_add(1) .ok_or(AnalyticalCostError::Overflow)?; - visit(outer, seen, counts)?; - visit(inner, seen, counts)?; } + // Retained relational work, accumulator/population boundaries and + // exact query-time operators add no summary operation. + Operator::NonASAP(_) + | Operator::ASAP( + ASAPOp::MaintainPopulation { .. } + | ASAPOp::EvaluatePopulation { .. } + | ASAPOp::Extension { .. }, + ) => {} + } + for child in node.children() { + visit(child, seen, counts)?; } Ok(()) } diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs index de20a94a0..6b0bca24f 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs @@ -134,7 +134,7 @@ pub struct SummaryOperationCpuEvidence { pub delete_events_per_second: Option, /// Concrete state instances touched by one delete event. pub delete_routing_fanout: Option, - pub readout_cpu_ops: Option, + pub evaluation_cpu_ops: Option, } /// Physical evidence for one `SummaryJoin` implementation. Total work, @@ -185,7 +185,7 @@ pub struct SummaryOperatorResourceEvidence { } /// Evidence is structured by logical summary operation so delete-only facts -/// cannot be attached to merge, subtract, or readout nodes. +/// cannot be attached to merge, subtract, or evaluation nodes. #[derive(Debug, Clone, PartialEq)] pub enum SummaryOperatorEvidence { /// Exact query-time arithmetic over two independently realized operands. @@ -201,7 +201,7 @@ pub enum SummaryOperatorEvidence { events_per_second: f64, routing_fanout: u64, }, - Readout(SummaryOperatorResourceEvidence), + Evaluation(SummaryOperatorResourceEvidence), } impl SummaryOperatorEvidence { @@ -212,7 +212,7 @@ impl SummaryOperatorEvidence { | Self::Merge(resource) | Self::Subtract(resource) | Self::Delete { resource, .. } - | Self::Readout(resource) => resource, + | Self::Evaluation(resource) => resource, } } @@ -224,12 +224,12 @@ impl SummaryOperatorEvidence { | Self::Merge(resource) | Self::Subtract(resource) | Self::Delete { resource, .. } - | Self::Readout(resource) => resource, + | Self::Evaluation(resource) => resource, } } } -/// Non-aggregation work for a retained pre-ASAP subtree over the comparison +/// Non-aggregation work for a retained pre-ASAP sub-DAG over the comparison /// horizon. Bootstrap/source I/O belongs exclusively to the owning aggregate, /// and summary insertion belongs exclusively to its insert evidence. #[derive(Debug, Clone, PartialEq)] @@ -247,27 +247,27 @@ pub struct RetainedSubDagEvidence { /// structurally equal node is not silently treated as the same deployment. #[derive(Debug, Clone, Default)] pub struct SummaryNodeEvidence { - pub(super) aggregations: HashMap<*const SummaryNode, SummaryAggregateEvidence>, - pub(super) joins: HashMap<*const SummaryNode, SummaryJoinEvidence>, - pub(super) operations: HashMap<*const SummaryNode, SummaryOperatorEvidence>, - pub(super) operation_state_owners: HashMap<*const SummaryNode, *const SummaryNode>, - pub(super) retained_queries: HashMap<*const SummaryNode, RetainedSubDagEvidence>, + pub(super) aggregations: HashMap<*const OperatorNode, SummaryAggregateEvidence>, + pub(super) joins: HashMap<*const OperatorNode, SummaryJoinEvidence>, + pub(super) operations: HashMap<*const OperatorNode, SummaryOperatorEvidence>, + pub(super) operation_state_owners: HashMap<*const OperatorNode, *const OperatorNode>, + pub(super) retained_queries: HashMap<*const OperatorNode, RetainedSubDagEvidence>, } impl SummaryNodeEvidence { pub fn insert_aggregation( &mut self, - node: &Rc, + node: &Rc, evidence: SummaryAggregateEvidence, ) { self.aggregations.insert(Rc::as_ptr(node), evidence); } - pub fn insert_join(&mut self, node: &Rc, evidence: SummaryJoinEvidence) { + pub fn insert_join(&mut self, node: &Rc, evidence: SummaryJoinEvidence) { self.joins.insert(Rc::as_ptr(node), evidence); } - pub fn insert_operation(&mut self, node: &Rc, evidence: SummaryOperatorEvidence) { + pub fn insert_operation(&mut self, node: &Rc, evidence: SummaryOperatorEvidence) { self.operations.insert(Rc::as_ptr(node), evidence); } @@ -275,8 +275,8 @@ impl SummaryNodeEvidence { /// aggregation deployment whose active interval it follows. pub fn insert_state_operation( &mut self, - node: &Rc, - state: &Rc, + node: &Rc, + state: &Rc, evidence: SummaryOperatorEvidence, ) { self.operations.insert(Rc::as_ptr(node), evidence); @@ -286,52 +286,57 @@ impl SummaryNodeEvidence { pub fn insert_retained_query( &mut self, - node: &Rc, + node: &Rc, evidence: RetainedSubDagEvidence, ) { self.retained_queries.insert(Rc::as_ptr(node), evidence); } - pub(super) fn aggregation(&self, node: &SummaryNode) -> Option { + pub(super) fn aggregation(&self, node: &OperatorNode) -> Option { self.aggregations.get(&(node as *const _)).cloned() } } pub(super) fn summary_operation_evidence<'a>( - node: &SummaryNode, + node: &OperatorNode, evidence: &'a SummaryNodeEvidence, ) -> Result<&'a SummaryOperatorEvidence, AnalyticalCostError> { let operation = evidence .operations .get(&(node as *const _)) .ok_or(AnalyticalCostError::MissingOrStale("summary operation"))?; - let matches = matches!( - (&node.expr, operation), + // A binary operator or join over two inputs is `Binary` evidence; every + // other non-ASAP operator, and the accumulator/population boundaries, + // is a `ValueOperation`. + let matches = match (&node.operator, operation) { ( - SummaryExpr::BinaryOp { .. }, - SummaryOperatorEvidence::Binary(_) - ) | ( - SummaryExpr::ValueOperation { .. }, - SummaryOperatorEvidence::ValueOperation(_) - ) | ( - SummaryExpr::SummaryMerge { .. }, - SummaryOperatorEvidence::Merge(_) - ) | ( - SummaryExpr::SummarySubtract { .. }, - SummaryOperatorEvidence::Subtract(_) - ) | ( - SummaryExpr::SummaryDelete { .. }, - SummaryOperatorEvidence::Delete { .. } - ) | ( - SummaryExpr::SummaryEstimate { .. }, - SummaryOperatorEvidence::Readout(_) - ) - ); + Operator::NonASAP(NonASAPOp::BinaryOp { .. } | NonASAPOp::Join { .. }), + SummaryOperatorEvidence::Binary(_), + ) => true, + (Operator::NonASAP(NonASAPOp::BinaryOp { .. } | NonASAPOp::Join { .. }), _) => false, + ( + Operator::NonASAP(_) + | Operator::ASAP( + ASAPOp::FinalizeExactAccumulator { .. } + | ASAPOp::MaintainPopulation { .. } + | ASAPOp::EvaluatePopulation { .. }, + ), + SummaryOperatorEvidence::ValueOperation(_), + ) => true, + (Operator::ASAP(ASAPOp::SummaryMerge { .. }), SummaryOperatorEvidence::Merge(_)) + | (Operator::ASAP(ASAPOp::SummarySubtract { .. }), SummaryOperatorEvidence::Subtract(_)) + | (Operator::ASAP(ASAPOp::SummaryDelete { .. }), SummaryOperatorEvidence::Delete { .. }) + | ( + Operator::ASAP(ASAPOp::SummaryEstimate { .. }), + SummaryOperatorEvidence::Evaluation(_), + ) => true, + _ => false, + }; if matches { Ok(operation) } else { Err(AnalyticalCostError::InconsistentOperatorStatistics( - "summary operation evidence kind does not match SummaryExpr", + "summary operation evidence kind does not match the operator", )) } } diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs index 6e4901f48..934fc4b82 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs @@ -1,4 +1,4 @@ -//! Analytical resource cost for at-rest and incrementally maintained summary deployments. +//! Analytical resource cost for at-rest and at-rest and incrementally maintained summary deployments. //! //! The canonical workload and lifecycle types own deployment semantics. This //! module only adds physical evidence absent from those schemas: state size, @@ -7,14 +7,13 @@ use std::collections::{HashMap, HashSet}; use std::rc::Rc; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, Predicate}; use asap_types::post_asap::{ - BoundExpr, ErrorMetric, ExactKind, GuaranteeSource, ProbabilityExpr, ResultGuarantee, - SketchAlgorithm, SummaryExpr, SummaryFamilyType, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryNode, SummaryWindowFramework, -}; -use asap_types::pre_asap::{ - agg_intent::AggIntent, CompareOpKind, InfoMatcher, Predicate, QueryExpr, Source, + BoundExpr, ErrorMetric, ExactKind, FieldDataType, GuaranteeSource, ProbabilityExpr, + ResultGuarantee, SketchAlgorithm, SummaryMaintenanceLifecycle, + SummaryMaintenanceLifecycleGuarantee, SummaryWindowFramework, }; +use asap_types::pre_asap::{agg_intent::AggIntent, CompareOpKind, InfoMatcher, Source}; use asap_types::types::AccuracyTarget; use asap_types::workload::{DataArrival, DataWorkload, QueryRecurrence, RepeatedDemand}; use serde::{Deserialize, Serialize}; diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs index 8bbca6cc4..e7c9aa34c 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs @@ -7,7 +7,7 @@ pub struct SummaryMaintenanceCostModel { pub node_evidence: SummaryNodeEvidence, pub calibration: ResourceCalibration, pub capabilities: SummaryMaintenanceCapabilities, - target_comparisons: HashMap<*const QueryExpr, SummaryTargetComparison>, + target_comparisons: HashMap<*const OperatorNode, SummaryTargetComparison>, candidate_comparisons: HashMap, physical_plan_alternatives: HashMap>, @@ -15,17 +15,17 @@ pub struct SummaryMaintenanceCostModel { HashMap>, } -type CandidateComparisonKey = (*const QueryExpr, *const SummaryNode); +type CandidateComparisonKey = (*const OperatorNode, *const OperatorNode); #[derive(Debug, Clone)] struct BoundCandidateIdentity { - _target: Rc, - _root: Rc, + _target: Rc, + _root: Rc, } #[derive(Debug, Clone)] struct SummaryTargetComparison { - _target: Rc, + _target: Rc, scope: ComparisonScope, raw: RawInputEvidence, } @@ -59,73 +59,40 @@ fn info_source(selector: &[InfoMatcher]) -> Result }) } +/// Collect the source selections (scan sources with their predicates, and +/// info-metric selectors) of every leaf reachable from `node`, visiting a +/// shared node once. pub(super) fn query_source_selections( - query: &QueryExpr, + node: &OperatorNode, + seen: &mut HashSet<*const OperatorNode>, out: &mut Vec, ) -> Result<(), AnalyticalCostError> { - use QueryExpr::*; - match query { - Scan { + if !seen.insert(node as *const _) { + return Ok(()); + } + match &node.operator { + Operator::NonASAP(NonASAPOp::Scan { source, predicates, .. - } => out.push((source.clone(), predicates.clone(), vec![])), - PromqlVectorFromScalar(child) | PromqlScalarFromVector(child) => { - query_source_selections(child, out)? - } - PromqlInfoEnrich { selector, child } => { - query_source_selections(child, out)?; + }) => out.push((source.clone(), predicates.clone(), vec![])), + Operator::NonASAP(NonASAPOp::PromqlInfoEnrich { selector, child }) => { + query_source_selections(child, seen, out)?; out.push((info_source(selector)?, vec![], selector.clone())); } - PromqlRelabel { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | PromqlSeriesSample { child, .. } - | Sort { child, .. } - | Limit { child, .. } => query_source_selections(child, out)?, - Concat { children, .. } => { - for child in children { - query_source_selections(child, out)?; + _ => { + for child in node.children() { + query_source_selections(child, seen, out)?; } } - Join { left, right, .. } | SetOp { left, right, .. } => { - query_source_selections(left, out)?; - query_source_selections(right, out)?; - } - BinaryOp { lhs, rhs, .. } => { - query_source_selections(lhs, out)?; - query_source_selections(rhs, out)?; - } - PromqlScalarBridge(_) - | EvalTimestamp - | CurrentTimestamp - | Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} } Ok(()) } fn validate_query_scope( - target: &QueryExpr, + target: &OperatorNode, scope: &ComparisonScope, ) -> Result<(), AnalyticalCostError> { let mut actual = Vec::new(); - query_source_selections(target, &mut actual)?; + query_source_selections(target, &mut HashSet::new(), &mut actual)?; let actual = deduplicate_source_selections(actual); let mut declared: Vec<_> = scope .sources @@ -428,8 +395,8 @@ impl SummaryMaintenanceCostModel { /// rejected rather than silently replacing the canonical context. pub fn bind_candidate_comparison( &mut self, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, scope: ComparisonScope, raw: RawInputEvidence, ) -> Result<(), AnalyticalCostError> { @@ -474,8 +441,8 @@ impl SummaryMaintenanceCostModel { /// candidate. Duplicate or empty provider identities are rejected. pub fn bind_physical_plan_alternative( &mut self, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, alternative: SummaryPhysicalPlanAlternative, ) -> Result<(), AnalyticalCostError> { let key = (Rc::as_ptr(target), Rc::as_ptr(root)); @@ -512,8 +479,8 @@ impl SummaryMaintenanceCostModel { /// evidence keep the implementations distinct during ranking. pub fn bind_window_framework_candidate( &mut self, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, candidate: SummaryWindowFrameworkCandidate, ) -> Result<(), AnalyticalCostError> { let key = (Rc::as_ptr(target), Rc::as_ptr(root)); @@ -565,8 +532,8 @@ impl SummaryMaintenanceCostModel { fn comparison_context( &self, - root: &SummaryNode, - target: Option<&QueryExpr>, + root: &OperatorNode, + target: Option<&OperatorNode>, horizon: Option, expected_reads: Option, ) -> Option<(CandidateComparisonKey, &SummaryTargetComparison)> { @@ -600,7 +567,7 @@ impl SummaryMaintenanceCostModel { fn complete_cost_with_evidence( &self, - root: &SummaryNode, + root: &OperatorNode, deployments: &[CostedSummaryDeployment<'_>], comparison: &SummaryTargetComparison, evidence: &SummaryNodeEvidence, @@ -619,7 +586,7 @@ impl SummaryMaintenanceCostModel { ) } - fn canonical_inputs(&self, summary: &SummaryNode) -> Option { + fn canonical_inputs(&self, summary: &OperatorNode) -> Option { let evidence = self.node_evidence.aggregation(summary)?; evidence.inputs.validate().ok()?; Some(evidence) @@ -631,7 +598,7 @@ impl SummaryMaintenanceCostModel { fn lifecycle_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, horizon: Option, ) -> Option { let evidence = self.canonical_inputs(summary)?; @@ -657,8 +624,8 @@ impl SummaryMaintenanceCostModel { Some(SummaryMaintenanceLifecycleCostInputs { build_cost: Some(build), maintenance_cost_per_update: Some(maintenance), - // Readout is a separate physical operator in the complete DAG. - // A state-only candidate therefore does not fabricate readout + // Evaluation is a separate physical operator in the complete DAG. + // A state-only candidate therefore does not fabricate evaluation // evidence merely to keep a lifecycle alternative selectable. summary_read_cost: Some(Cost::ZERO), retention_cost_rate: Some(CostRate(retention_total.0 / horizon_seconds)), @@ -681,8 +648,7 @@ impl CostModel for SummaryMaintenanceCostModel { // Lifecycle selection supplies a complete override. If it cannot, // the candidate remains unavailable rather than receiving this // trait's structural fallback. - Replacement::Summary(_) => None, - Replacement::Rewrite(_) => None, + Replacement::SubDag(_) => None, } } @@ -700,14 +666,14 @@ impl CostModel for SummaryMaintenanceCostModel { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs::default() } fn summary_maintenance_lifecycle_cost_inputs_for_horizon( &self, - summary: &SummaryNode, + summary: &OperatorNode, horizon: Option, ) -> SummaryMaintenanceLifecycleCostInputs { self.lifecycle_inputs(summary, horizon).unwrap_or_default() @@ -715,15 +681,15 @@ impl CostModel for SummaryMaintenanceCostModel { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { self.capabilities } fn complete_summary_candidate_cost( &self, - root: &SummaryNode, - target: Option<&QueryExpr>, + root: &OperatorNode, + target: Option<&OperatorNode>, deployments: &[CostedSummaryDeployment<'_>], horizon: Option, expected_reads: Option, @@ -742,8 +708,8 @@ impl CostModel for SummaryMaintenanceCostModel { fn complete_summary_candidate_estimate( &self, - root: &SummaryNode, - target: Option<&QueryExpr>, + root: &OperatorNode, + target: Option<&OperatorNode>, deployments: &[CostedSummaryDeployment<'_>], horizon: Option, expected_reads: Option, @@ -847,14 +813,14 @@ impl CostModel for SummaryMaintenanceCostModel { true } - fn raw_query_recompute_cost(&self, target: &QueryExpr) -> Option { + fn raw_query_recompute_cost(&self, target: &OperatorNode) -> Option { let _ = target; None } fn raw_query_recompute_total_cost( &self, - target: &QueryExpr, + target: &OperatorNode, expected_reads: f64, ) -> Option { let target_ptr = target as *const _; @@ -880,13 +846,14 @@ use super::estimator::*; mod tests { use std::rc::Rc; + use asap_types::ir::{ASAPOp, BinaryOperator, NonASAPOp, Operator, OperatorNode}; use asap_types::post_asap::{ - EvaluationSchedule, ExactKind, ExactParams, GroupingStrategy, OutputRepresentation, - SummaryExpr, SummaryFamilyType, SummaryField, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, SummarySchema, + EvaluationSchedule, ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, + OutputRepresentation, Schema, SummaryMaintenanceLifecycle, + SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, SummaryUpdate, }; use asap_types::pre_asap::{ - agg_intent::AggIntent, Column, ColumnRef, DataType, QueryExpr, Reduction, Schema, Source, + agg_intent::AggIntent, ArithmeticOpKind, BinaryOpKind, DataType, Reduction, Source, }; use asap_types::workload::{ DataWorkload, Evidence, EvidenceSource, Predictability, Query, QueryLanguage, @@ -903,7 +870,7 @@ mod tests { }; fn estimate_test( - root: &SummaryNode, + root: &OperatorNode, guarantee: &SummaryMaintenanceLifecycleGuarantee, inputs: SummaryMaintenanceInputs, cpu: SummaryOperationCpuEvidence, @@ -912,7 +879,7 @@ mod tests { } fn estimate_join_test( - root: &SummaryNode, + root: &OperatorNode, guarantee: &SummaryMaintenanceLifecycleGuarantee, inputs: SummaryMaintenanceInputs, cpu: SummaryOperationCpuEvidence, @@ -1200,7 +1167,7 @@ mod tests { inputs, SummaryOperationCpuEvidence { insert_cpu_ops: Some(2.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ) @@ -1253,7 +1220,7 @@ mod tests { inputs, SummaryOperationCpuEvidence { insert_cpu_ops: Some(2.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ) @@ -1442,7 +1409,10 @@ mod tests { let mut model = streaming_model(); for group in space.target_subdag_candidates() { for candidate in &group.candidates { - if let Replacement::Summary(root) = &candidate.replacement { + if let Replacement::SubDag(root) = &candidate.replacement { + if !root.contains_asap() { + continue; + } bind_aggregations( &mut model, &group.target, @@ -1713,10 +1683,11 @@ mod tests { op: CompareOpKind::Eq, value: "api".into(), }]; - let info_target = QueryExpr::PromqlInfoEnrich { + let info_target = OperatorNode::non_asap_node(NonASAPOp::PromqlInfoEnrich { selector: selector.clone(), child: target, - }; + }) + .unwrap(); let mut info_scope = streaming_scope(); info_scope .sources @@ -1739,8 +1710,9 @@ mod tests { } #[test] - fn delete_owner_must_be_the_unique_state_reachable_from_its_input() { - let workload = streaming_workload(); + fn summary_delete_dag_fails_closed_before_costing() { + // SummaryDelete is reserved: planning rejects the DAG even with full + // delete evidence, instead of costing (or owner-checking) the delete. let target = streaming_sum_query(); let root = summary_with_operations(false, false, true); let mut cpu = streaming_cpu(); @@ -1750,34 +1722,32 @@ mod tests { let mut model = streaming_model(); model.capabilities.delete = true; bind_aggregations(&mut model, &target, &root, streaming_inputs(), cpu); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - unreachable!(); - }; - let delete_ptr = Rc::as_ptr(summary_input); - let unrelated = summary_with_operations(false, false, false); - let unrelated_agg = evidence_nodes(&unrelated).0[0] as *const _; - model - .node_evidence - .operation_state_owners - .insert(delete_ptr, unrelated_agg); - let plan = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(plan.summary_total_cost, None); + // Under a evaluation, the reserved delete surfaces as an illegal child. + assert!(matches!( + streaming_planning_error(Rc::clone(&root), &model), + asap_types::post_asap::ExecutionDataStateError::IllegalChildDataState { + edge: "FinalizeExactAccumulator.child", + child: asap_types::post_asap::ExecutionDataState::QUERY_ROWS, + } + )); + // As the root, it is reported as the unimplemented operator itself. + assert!(matches!( + streaming_planning_error(evaluation_state(&root), &model), + asap_types::post_asap::ExecutionDataStateError::UnimplementedOperator { + operator: "SummaryDelete" + } + )); } #[test] fn summary_edge_and_io_evidence_fail_closed() { + // Over two independent summaries combined by a BinaryOp: a parent + // input edge that disagrees with its child's output, or missing I/O + // evidence on the root, leaves the whole-DAG cost unset. let workload = streaming_workload(); let target = streaming_sum_query(); - let root = summary_join(); + let root = add_independent_summary_results(); let mut model = streaming_model(); bind_aggregations( &mut model, @@ -1786,19 +1756,15 @@ mod tests { streaming_inputs(), streaming_cpu(), ); - let join = evidence_nodes(&root).1[0]; - model.node_evidence.joins.insert( - join as *const _, - SummaryJoinEvidence { - physical_id: "join-edge".into(), - inputs: vec![test_edge(), EdgeStatistics { rows: 2, bytes: 16 }], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: 1, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, + model.node_evidence.insert_operation( + &root, + SummaryOperatorEvidence::Binary(test_resource( + "binary-edge", + vec![test_edge(), EdgeStatistics { rows: 2, bytes: 16 }], + 1.0, + 1, + 0, + )), ); let bad_edge = plan_summary_maintenance_lifecycles( Rc::clone(&root), @@ -1811,20 +1777,34 @@ mod tests { .unwrap(); assert_eq!(bad_edge.summary_total_cost, None); - model + let root_evidence = model .node_evidence - .joins - .get_mut(&(join as *const _)) + .operations + .get_mut(&Rc::as_ptr(&root)) .unwrap() - .inputs = vec![test_edge(), test_edge()]; + .resource_mut(); + root_evidence.inputs = vec![test_edge(), test_edge()]; + root_evidence.io_bytes_per_execution = None; + let missing_io = plan_summary_maintenance_lifecycles( + Rc::clone(&root), + WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), + 0, + Some(Horizon(5.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &model, + ) + .unwrap(); + assert_eq!(missing_io.summary_total_cost, None); + + // Control: the same evidence with I/O restored is costable. model .node_evidence .operations .get_mut(&Rc::as_ptr(&root)) .unwrap() .resource_mut() - .io_bytes_per_execution = None; - let missing_io = plan_summary_maintenance_lifecycles( + .io_bytes_per_execution = Some(0); + let complete = plan_summary_maintenance_lifecycles( root, WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), 0, @@ -1833,14 +1813,17 @@ mod tests { &model, ) .unwrap(); - assert_eq!(missing_io.summary_total_cost, None); + assert!(complete.summary_total_cost.is_some()); } #[test] fn summary_edges_io_and_physical_identity_fail_closed() { + // Evidence bound to a structurally equal clone of the BinaryOp does not + // count for the real node; a bad input edge or missing I/O on the real + // node still fails closed. let workload = streaming_workload(); let target = streaming_sum_query(); - let root = summary_join(); + let root = add_independent_summary_results(); let mut model = streaming_model(); bind_aggregations( &mut model, @@ -1849,34 +1832,27 @@ mod tests { streaming_inputs(), streaming_cpu(), ); - let (_, joins) = evidence_nodes(&root); - model.node_evidence.insert_join( - &Rc::new(joins[0].clone()), - SummaryJoinEvidence { - physical_id: "unused".into(), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: 1, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, + model.node_evidence.insert_operation( + &Rc::new((*root).clone()), + SummaryOperatorEvidence::Binary(test_resource( + "unused", + vec![test_edge(), test_edge()], + 1.0, + 1, + 0, + )), ); - // Bind the actual join, then make one parent input disagree with its - // child's output. - model.node_evidence.joins.insert( - joins[0] as *const _, - SummaryJoinEvidence { - physical_id: "join-edge".into(), - inputs: vec![test_edge(), EdgeStatistics { rows: 2, bytes: 16 }], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: 1, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, + // Bind the actual BinaryOp, then make one parent input disagree with + // its child's output. + model.node_evidence.insert_operation( + &root, + SummaryOperatorEvidence::Binary(test_resource( + "binary-edge", + vec![test_edge(), EdgeStatistics { rows: 2, bytes: 16 }], + 1.0, + 1, + 0, + )), ); let bad_edge = plan_summary_maintenance_lifecycles( Rc::clone(&root), @@ -1889,19 +1865,14 @@ mod tests { .unwrap(); assert_eq!(bad_edge.summary_total_cost, None); - model - .node_evidence - .joins - .get_mut(&(joins[0] as *const _)) - .unwrap() - .inputs = vec![test_edge(), test_edge()]; - model + let root_evidence = model .node_evidence .operations .get_mut(&Rc::as_ptr(&root)) .unwrap() - .resource_mut() - .io_bytes_per_execution = None; + .resource_mut(); + root_evidence.inputs = vec![test_edge(), test_edge()]; + root_evidence.io_bytes_per_execution = None; let missing_io = plan_summary_maintenance_lifecycles( root, WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), @@ -1955,9 +1926,11 @@ mod tests { #[test] fn conflicting_evidence_cannot_alias_one_provider_physical_identity() { + // Two independent summary states (combined by a BinaryOp) that claim + // one physical id but carry different evidence leave the cost unset. let workload = streaming_workload(); let target = streaming_sum_query(); - let root = summary_join(); + let root = add_independent_summary_results(); let mut model = streaming_model(); bind_aggregations( &mut model, @@ -1967,6 +1940,7 @@ mod tests { streaming_cpu(), ); let aggregations = evidence_nodes(&root).0; + assert_eq!(aggregations.len(), 2); let first = aggregations[0] as *const _; let second = aggregations[1] as *const _; model @@ -1992,8 +1966,9 @@ mod tests { } #[test] - fn lifecycle_plan_does_not_fall_back_to_partial_agg_cost_for_a_join_root() { - let workload = streaming_workload(); + fn summary_join_root_fails_closed_even_with_join_evidence() { + // SummaryJoin is reserved: planning rejects the DAG whether or not + // join evidence is bound, so no partial or join cost is produced. let root = summary_join(); let target = streaming_sum_query(); let mut model = streaming_model(); @@ -2004,21 +1979,15 @@ mod tests { streaming_inputs(), streaming_cpu(), ); - let plan = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(plan.deployments.len(), 2); - assert_eq!(plan.summary_total_cost, None); + assert!(matches!( + streaming_planning_error(Rc::clone(&root), &model), + asap_types::post_asap::ExecutionDataStateError::UnimplementedOperator { + operator: "SummaryJoin" + } + )); - let mut costed = model; let join_node = evidence_nodes(&root).1[0]; - costed.node_evidence.joins.insert( + model.node_evidence.joins.insert( join_node as *const _, SummaryJoinEvidence { physical_id: "costed-join".into(), @@ -2031,28 +2000,28 @@ mod tests { io_bytes_per_execution: Some(0), }, ); - let costed_plan = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &costed, - ) - .unwrap(); - assert!(costed_plan.summary_total_cost.is_some()); + assert!(matches!( + streaming_planning_error(root, &model), + asap_types::post_asap::ExecutionDataStateError::UnimplementedOperator { + operator: "SummaryJoin" + } + )); } #[test] fn whole_dag_cost_requires_and_uses_each_rc_bound_state_evidence() { + // Over two independent summaries combined by a BinaryOp: the cost needs + // evidence for each Rc-bound state, charges peak transient memory, and + // de-duplicates bootstrap scans only on a shared provider read id. let workload = streaming_workload(); - let root = summary_join(); + let root = add_independent_summary_results(); + let (left, right) = binary_operands(&root); let target = streaming_sum_query(); - let (aggregations, joins) = evidence_nodes(&root); + let aggregations = [evaluation_state(&left), evaluation_state(&right)]; let mut model = streaming_model(); bind_comparison(&mut model, &target, &root); - model.node_evidence.aggregations.insert( - aggregations[0] as *const _, + model.node_evidence.insert_aggregation( + &aggregations[0], SummaryAggregateEvidence { physical_id: "left-state".into(), input: test_edge(), @@ -2078,8 +2047,8 @@ mod tests { second_inputs.state_bytes_per_summary = 250; let mut second_cpu = streaming_cpu(); second_cpu.insert_cpu_ops = Some(5.0); - model.node_evidence.aggregations.insert( - aggregations[1] as *const _, + model.node_evidence.insert_aggregation( + &aggregations[1], SummaryAggregateEvidence { physical_id: "right-state".into(), input: test_edge(), @@ -2090,31 +2059,37 @@ mod tests { insert_cpu_ops: second_cpu.insert_cpu_ops.unwrap(), }, ); - model.node_evidence.joins.insert( - joins[0] as *const _, - SummaryJoinEvidence { - physical_id: "join".into(), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops_per_execution: 6.0, - working_memory_bytes: 64, - output_buffer_bytes: 64, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, + // The left evaluation plays the old join's role: a 64-byte workspace and a + // 64-byte output that stays live until the root BinaryOp consumes it. + model.node_evidence.insert_operation( + &left, + SummaryOperatorEvidence::ValueOperation(test_resource( + "left-evaluation", + vec![test_edge()], + 6.0, + 64, + 64, + )), ); - model.node_evidence.operations.insert( - Rc::as_ptr(&root), - SummaryOperatorEvidence::Readout(SummaryOperatorResourceEvidence { - physical_id: "root-readout".into(), - inputs: vec![test_edge()], - output: test_edge(), - cpu_ops: 3.0, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }), + model.node_evidence.insert_operation( + &right, + SummaryOperatorEvidence::ValueOperation(test_resource( + "right-evaluation", + vec![test_edge()], + 3.0, + 0, + 0, + )), + ); + model.node_evidence.insert_operation( + &root, + SummaryOperatorEvidence::Binary(test_resource( + "root-binary", + vec![test_edge(), test_edge()], + 3.0, + 0, + 0, + )), ); let complete = plan_summary_maintenance_lifecycles( Rc::clone(&root), @@ -2143,8 +2118,9 @@ mod tests { &model, ) .unwrap(); - // The join's 64-byte output remains live while the readout's workspace - // is active. The join's execution workspace is released first. + // The left evaluation's 64-byte output remains live while the binary's + // workspace is active (64 + 128 = 192); the evaluation's own workspace is + // released first, so the old peak was 64 + 64 = 128. assert_eq!( larger_workspace.summary_total_cost.unwrap().0 - complete.summary_total_cost.unwrap().0, 64.0 @@ -2184,7 +2160,10 @@ mod tests { }); let target = streaming_sum_query(); let root = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: summary_input, + }) = &root.operator + else { unreachable!(); }; let windowed_summary = Rc::clone(summary_input); @@ -2326,7 +2305,10 @@ mod tests { fn window_framework_candidates_require_unique_nonempty_planner_primitives() { let target = streaming_sum_query(); let root = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: summary_input, + }) = &root.operator + else { unreachable!(); }; let windowed_summary = Rc::clone(summary_input); @@ -2365,9 +2347,12 @@ mod tests { #[test] fn one_physical_identity_cannot_alias_different_window_frameworks() { + // Two independent summaries (combined by a BinaryOp) sharing one + // physical state id but assigned different window frameworks leave + // the cost unset. let workload = streaming_workload(); let target = streaming_sum_query(); - let root = summary_join(); + let root = add_independent_summary_results(); let mut model = streaming_model(); bind_aggregations( &mut model, @@ -2376,44 +2361,22 @@ mod tests { streaming_inputs(), streaming_cpu(), ); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - unreachable!(); - }; - let SummaryExpr::SummaryJoin { outer, inner, .. } = &summary_input.expr else { - unreachable!(); - }; - let aggregation_nodes = [Rc::clone(outer), Rc::clone(inner)]; - let (aggregations, joins) = evidence_nodes(&root); - model.node_evidence.joins.insert( - joins[0] as *const _, - SummaryJoinEvidence { - physical_id: "joined-readout".into(), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: 8, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, - ); + let (left, right) = binary_operands(&root); + let aggregation_nodes = [evaluation_state(&left), evaluation_state(&right)]; let mut shared_aggregation = - model.node_evidence.aggregations[&(aggregations[0] as *const _)].clone(); + model.node_evidence.aggregations[&Rc::as_ptr(&aggregation_nodes[0])].clone(); shared_aggregation.physical_id = "shared-window-state".into(); - model - .node_evidence - .aggregations - .insert(aggregations[0] as *const _, shared_aggregation.clone()); - model - .node_evidence - .aggregations - .insert(aggregations[1] as *const _, shared_aggregation); + for aggregate in &aggregation_nodes { + model + .node_evidence + .insert_aggregation(aggregate, shared_aggregation.clone()); + } let retained_children: Vec<_> = aggregation_nodes .iter() - .map(|aggregate| match &aggregate.expr { - SummaryExpr::SummaryAgg { child, .. } => Rc::clone(child), + .map(|aggregate| match &aggregate.operator { + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) => Rc::clone(child), _ => unreachable!(), }) .collect(); @@ -2428,7 +2391,7 @@ mod tests { } let candidate = SummaryWindowFrameworkCandidate { - physical_plan_id: "mixed-framework-join".into(), + physical_plan_id: "mixed-framework-binary".into(), assignments: vec![ SummaryWindowFrameworkAssignment { summary: Rc::clone(&aggregation_nodes[0]), @@ -2544,7 +2507,10 @@ mod tests { asap_types::workload::AccuracyRequirement::Explicit(AccuracyTarget::Epsilon(1.0)); let target = streaming_sum_query(); let root = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: summary_input, + }) = &root.operator + else { unreachable!(); }; let mut model = streaming_model(); @@ -2583,6 +2549,42 @@ mod tests { assert_eq!(plan.summary_total_cost, None); } + /// Bulk relational evidence cannot hide summary operators below an ordinary root. + #[test] + fn retained_subdag_evidence_cannot_hide_summary_work() { + let workload = streaming_workload(); + let target = streaming_sum_query(); + let root = add_shared_summary_result(); + let mut model = streaming_model(); + bind_aggregations( + &mut model, + &target, + &root, + streaming_inputs(), + streaming_cpu(), + ); + model.node_evidence.insert_retained_query( + &root, + RetainedSubDagEvidence { + physical_id: "false-retained-root".into(), + output: test_edge(), + preprocessing_cpu_ops_over_horizon: 0.0, + working_memory_bytes: 0, + output_buffer_bytes: 0, + }, + ); + let plan = plan_summary_maintenance_lifecycles( + root, + WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), + 0, + Some(Horizon(5.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &model, + ) + .unwrap(); + assert_eq!(plan.summary_total_cost, None); + } + #[test] fn whole_dag_fails_closed_for_missing_retained_work_or_false_source_lineage() { let workload = streaming_workload(); @@ -2631,23 +2633,22 @@ mod tests { } #[test] - fn aggregate_recurses_into_child_operations_and_state_only_needs_no_readout() { + fn state_only_needs_no_evaluation_and_summary_merge_child_fails_closed() { + // A state-only root is costable without evaluation evidence; a + // SummaryAgg over a reserved SummaryMerge is rejected at planning. let workload = streaming_workload(); let target = streaming_sum_query(); let estimated = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &estimated.expr else { - unreachable!(); - }; - let state_only = Rc::clone(summary_input); - let mut no_readout_cpu = streaming_cpu(); - no_readout_cpu.readout_cpu_ops = None; + let state_only = evaluation_state(&estimated); + let mut no_evaluation_cpu = streaming_cpu(); + no_evaluation_cpu.evaluation_cpu_ops = None; let mut state_model = streaming_model(); bind_aggregations( &mut state_model, &target, &state_only, streaming_inputs(), - no_readout_cpu, + no_evaluation_cpu, ); let state_plan = plan_summary_maintenance_lifecycles( state_only, @@ -2660,27 +2661,22 @@ mod tests { .unwrap(); assert!(state_plan.summary_total_cost.is_some()); - let child_readout = summary_with_operations(true, false, false); - let SummaryExpr::SummaryEstimate { - summary_input: child, - .. - } = &child_readout.expr - else { - unreachable!(); - }; - let child = Rc::clone(child); - let nested = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child, - family: SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count), - input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Wildcard), + let nested = OperatorNode::asap_node( + ASAPOp::SummaryAgg { + child: evaluation_state(&summary_with_operations(true, false, false)), + family: FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), + input: SummaryUpdate { + item: None, + weight: asap_types::post_asap::SummaryInputExpr::Constant(1.0), + weight_domain: Default::default(), + }, reduction: Reduction::by(vec![]), grouping: GroupingStrategy::PerSubpopulationInstance, filter: None, }, - schema: estimated.schema.clone(), - guarantee: None, - }); + count_state_schema(), + None, + ); let mut nested_cpu = streaming_cpu(); nested_cpu.merge_cpu_ops = Some(1.0); let mut nested_model = streaming_model(); @@ -2691,23 +2687,12 @@ mod tests { streaming_inputs(), nested_cpu, ); - nested_model - .node_evidence - .operations - .retain(|_, operation| { - operation.resource().cpu_ops != 1.0 - || operation.resource().working_memory_bytes == 0 - }); - let nested_plan = plan_summary_maintenance_lifecycles( - nested, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &nested_model, - ) - .unwrap(); - assert_eq!(nested_plan.summary_total_cost, None); + assert!(matches!( + streaming_planning_error(nested, &nested_model), + asap_types::post_asap::ExecutionDataStateError::UnimplementedOperator { + operator: "SummaryMerge" + } + )); } #[test] @@ -2744,7 +2729,7 @@ mod tests { }, SummaryOperationCpuEvidence { insert_cpu_ops: Some(2.0), - readout_cpu_ops: Some(3.0), + evaluation_cpu_ops: Some(3.0), ..SummaryOperationCpuEvidence::default() }, ) @@ -2778,7 +2763,7 @@ mod tests { delete_cpu_ops: Some(5.0), delete_events_per_second: Some(4.0), delete_routing_fanout: Some(2), - readout_cpu_ops: Some(7.0), + evaluation_cpu_ops: Some(7.0), }, ) .unwrap(); @@ -2808,7 +2793,7 @@ mod tests { }, SummaryOperationCpuEvidence { insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ), @@ -2835,7 +2820,7 @@ mod tests { }, SummaryOperationCpuEvidence { insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ), @@ -2866,7 +2851,7 @@ mod tests { }, SummaryOperationCpuEvidence { insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ), @@ -2901,7 +2886,7 @@ mod tests { }, SummaryOperationCpuEvidence { insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ) @@ -2937,7 +2922,7 @@ mod tests { }, SummaryOperationCpuEvidence { insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ) @@ -2981,7 +2966,7 @@ mod tests { }; let cpu = SummaryOperationCpuEvidence { insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }; assert_eq!( @@ -3009,164 +2994,239 @@ mod tests { assert_eq!(estimate.peak_memory_bytes(), 64); // 4 persistent states + join memory. } - fn summary_with_operations(merge: bool, subtract: bool, delete: bool) -> Rc { - let state_type = SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count); - let schema = SummarySchema { - fields: vec![SummaryField { - name: "count".into(), - dtype: state_type.clone(), - nullable: false, - }], - time_index: None, - }; - let leaf = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: "metrics".into(), - }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - ], - 0, - vec![], - ), - })), - schema: schema.clone(), - guarantee: None, - }); - let agg = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: leaf, + fn count_state_schema() -> Schema { + Schema::lifted( + vec![Field::new( + "count", + FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), + false, + )], + None, + ) + } + + fn count_evaluation_schema() -> Schema { + Schema::lifted(vec![Field::plain("count", DataType::Int64, false)], None) + } + + /// The retained relational input of every test summary: a bare scan of + /// the `metrics` series. + fn metrics_scan() -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Scan { + source: Source::TimeSeries { + metric: "metrics".into(), + }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + ], + 0, + vec![], + ), + }) + .unwrap() + } + + fn summary_with_operations(merge: bool, subtract: bool, delete: bool) -> Rc { + let state_type = FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count); + let schema = count_state_schema(); + let agg = OperatorNode::asap_node( + ASAPOp::SummaryAgg { + child: metrics_scan(), family: state_type, - input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Wildcard), + input: SummaryUpdate { + item: None, + weight: asap_types::post_asap::SummaryInputExpr::Constant(1.0), + weight_domain: Default::default(), + }, reduction: Reduction::by(vec![]), grouping: GroupingStrategy::PerSubpopulationInstance, filter: None, }, - schema: schema.clone(), - guarantee: None, - }); + schema.clone(), + None, + ); let mut root = Rc::clone(&agg); if merge { - root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: asap_types::post_asap::ExecutionTiming::IngestionTime, + root = OperatorNode::asap_node( + ASAPOp::SummaryMerge { children: vec![Rc::clone(&agg), Rc::clone(&agg)], }, - schema: schema.clone(), - guarantee: None, - }); + schema.clone(), + None, + ); } if subtract { - root = Rc::new(SummaryNode { - expr: SummaryExpr::SummarySubtract { + root = OperatorNode::asap_node( + ASAPOp::SummarySubtract { left: Rc::clone(&root), right: Rc::clone(&agg), }, - schema: schema.clone(), - guarantee: None, - }); + schema.clone(), + None, + ); } if delete { - root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryDelete { + root = OperatorNode::asap_node( + ASAPOp::SummaryDelete { summary_input: root, - key: ColumnRef::Wildcard, + key: 0, }, - schema: schema.clone(), - guarantee: None, - }); + schema.clone(), + None, + ); } - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: root, - query: asap_types::post_asap::SketchQuery::PointCount { - key: ColumnRef::Wildcard, - value: None, - }, - }, - schema, - guarantee: Some(ResultGuarantee::exact("exact count readout")), - }) + OperatorNode::asap_node( + ASAPOp::FinalizeExactAccumulator { child: root }, + count_evaluation_schema(), + Some(ResultGuarantee::exact("exact count evaluation")), + ) } - fn summary_join() -> Rc { + fn summary_join() -> Rc { let left = summary_with_operations(false, false, false); let right = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { - summary_input: left, - .. - } = &left.expr + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: left }) = &left.operator else { unreachable!() }; - let SummaryExpr::SummaryEstimate { - summary_input: right, - .. - } = &right.expr + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: right }) = &right.operator else { unreachable!() }; - let schema = left.schema.clone(); - let join = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryJoin { + let join = OperatorNode::asap_node( + ASAPOp::SummaryJoin { outer: Rc::clone(left), inner: Rc::clone(right), - key: ColumnRef::Wildcard, - family: SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count), - }, - schema: schema.clone(), - guarantee: None, - }); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: join, - query: asap_types::post_asap::SketchQuery::PointCount { - key: ColumnRef::Wildcard, - value: None, - }, + key: 0, + family: FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), }, - schema, - guarantee: None, - }) + left.schema.clone(), + None, + ); + OperatorNode::asap_node( + ASAPOp::FinalizeExactAccumulator { child: join }, + count_evaluation_schema(), + None, + ) } - fn summary_binary() -> Rc { + fn add_shared_summary_result() -> Rc { let operand = summary_with_operations(false, false, false); - Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - timing: asap_types::post_asap::ExecutionTiming::QueryTime, + Rc::new( + OperatorNode::new(Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + }, + return_bool: false, lhs: Rc::clone(&operand), rhs: operand, - operator: asap_types::post_asap::BinaryOperator { + })) + .unwrap() + .with_guarantee(Some(ResultGuarantee::exact("test binary"))), + ) + } + + /// Two independent summary states, each read out, combined by an ordinary + /// `BinaryOp`: the non-reserved replacement for a `SummaryJoin` fixture. + #[test] + fn independent_summary_results_form_a_valid_dag() { + add_independent_summary_results() + .validate_structure() + .unwrap(); + } + + fn add_independent_summary_results() -> Rc { + Rc::new( + OperatorNode::new(Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { checked_relative_division: false, checked_finite_division: false, - kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( - asap_types::pre_asap::ArithmeticOpKind::Add, - ), + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), vector_match: None, }, - }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), - nullable: false, - }], - time_index: None, - }, - guarantee: Some(ResultGuarantee::exact("test binary")), - }) + return_bool: false, + lhs: summary_with_operations(false, false, false), + rhs: summary_with_operations(false, false, false), + })) + .unwrap() + .with_guarantee(Some(ResultGuarantee::exact("test binary"))), + ) + } + + /// The `(lhs, rhs)` evaluations of [`add_independent_summary_results`]. + fn binary_operands(root: &OperatorNode) -> (Rc, Rc) { + let Operator::NonASAP(NonASAPOp::BinaryOp { lhs, rhs, .. }) = &root.operator else { + unreachable!(); + }; + (Rc::clone(lhs), Rc::clone(rhs)) + } + + /// The `SummaryAgg` under one `SummaryEstimate` evaluation. + fn evaluation_state(evaluation: &OperatorNode) -> Rc { + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: summary_input, + }) = &evaluation.operator + else { + unreachable!(); + }; + Rc::clone(summary_input) + } + + /// The DAG-validation error that streaming lifecycle planning of `root` + /// fails closed with. + fn streaming_planning_error( + root: Rc, + model: &SummaryMaintenanceCostModel, + ) -> asap_types::post_asap::ExecutionDataStateError { + let workload = streaming_workload(); + match plan_summary_maintenance_lifecycles( + root, + WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), + 0, + Some(Horizon(5.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + model, + ) { + Err( + crate::summary_maintenance_lifecycle::SummaryMaintenanceLifecyclePlanError::InvalidPostAsapDag( + error, + ), + ) => error, + Err(other) => panic!("unexpected planning error: {other}"), + Ok(_) => panic!("planning must fail closed"), + } + } + + fn test_resource( + physical_id: &str, + inputs: Vec, + cpu_ops: f64, + working_memory_bytes: u64, + output_buffer_bytes: u64, + ) -> SummaryOperatorResourceEvidence { + SummaryOperatorResourceEvidence { + physical_id: physical_id.into(), + inputs, + output: test_edge(), + cpu_ops, + working_memory_bytes, + output_buffer_bytes, + executions_per_evaluation: 1, + io_bytes_per_execution: Some(0), + } } #[test] fn exact_binary_is_costable_with_explicit_physical_evidence() { let workload = streaming_workload(); let target = streaming_sum_query(); - let root = summary_binary(); + let root = add_shared_summary_result(); let mut model = streaming_model(); bind_aggregations( &mut model, @@ -3193,29 +3253,16 @@ mod tests { assert!(plan.summary_total_cost.is_some()); } - fn streaming_sum_query() -> Rc { - let scan = Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: "metrics".into(), - }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - ], - 0, - vec![], - ), - }); - Rc::new(QueryExpr::Aggregate { + fn streaming_sum_query() -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Sum { col: None }], output_names: vec![], filters: vec![], having: None, - child: scan, + child: metrics_scan(), }) + .unwrap() } fn streaming_workload() -> QueryWorkload { @@ -3310,61 +3357,43 @@ mod tests { } } + /// A retained relational sub-DAG: a non-ASAP node with no summary below + /// it, costed as one unit through retained-query evidence. + fn is_retained(node: &OperatorNode) -> bool { + !node.contains_asap() + } + fn bind_comparison( model: &mut SummaryMaintenanceCostModel, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, ) { model .bind_candidate_comparison(target, root, streaming_scope(), streaming_raw()) .unwrap(); fn retained( model: &mut SummaryMaintenanceCostModel, - node: &Rc, - seen: &mut HashSet<*const SummaryNode>, + node: &Rc, + seen: &mut HashSet<*const OperatorNode>, ) { if !seen.insert(Rc::as_ptr(node)) { return; } - match &node.expr { - SummaryExpr::KeepPreAsap(_) => { - model.node_evidence.insert_retained_query( - node, - RetainedSubDagEvidence { - physical_id: format!("retained-{node:p}"), - output: test_edge(), - preprocessing_cpu_ops_over_horizon: 1.0, - working_memory_bytes: 8, - output_buffer_bytes: 0, - }, - ); - } - SummaryExpr::SummaryAgg { child, .. } - | SummaryExpr::ValueOperation { child, .. } => retained(model, child, seen), - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - retained(model, child, seen); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - retained(model, left, seen); - retained(model, right, seen); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - retained(model, summary_input, seen) - } + if is_retained(node) { + model.node_evidence.insert_retained_query( + node, + RetainedSubDagEvidence { + physical_id: format!("retained-{node:p}"), + output: test_edge(), + preprocessing_cpu_ops_over_horizon: 1.0, + working_memory_bytes: 8, + output_buffer_bytes: 0, + }, + ); + return; + } + for child in node.children() { + retained(model, child, seen); } } retained(model, root, &mut HashSet::new()); @@ -3391,24 +3420,23 @@ mod tests { fn streaming_cpu() -> SummaryOperationCpuEvidence { SummaryOperationCpuEvidence { insert_cpu_ops: Some(2.0), - readout_cpu_ops: Some(3.0), + evaluation_cpu_ops: Some(3.0), ..SummaryOperationCpuEvidence::default() } } fn bind_aggregations( model: &mut SummaryMaintenanceCostModel, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, inputs: SummaryMaintenanceInputs, cpu: SummaryOperationCpuEvidence, ) { bind_comparison(model, target, root); for node in evidence_nodes(root).0 { let source_root = matches!( - &node.expr, - SummaryExpr::SummaryAgg { child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)) + &node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) if is_retained(child) ); let mut node_inputs = inputs; if !source_root { @@ -3433,147 +3461,131 @@ mod tests { }, ); } + fn resource( + physical_id: String, + inputs: Vec, + cpu_ops: f64, + working_memory_bytes: u64, + ) -> SummaryOperatorResourceEvidence { + SummaryOperatorResourceEvidence { + physical_id, + inputs, + output: test_edge(), + cpu_ops, + working_memory_bytes, + output_buffer_bytes: 0, + executions_per_evaluation: 1, + io_bytes_per_execution: Some(0), + } + } fn bind_ops( model: &mut SummaryMaintenanceCostModel, - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, + node: &OperatorNode, + seen: &mut HashSet<*const OperatorNode>, inputs: SummaryMaintenanceInputs, cpu: SummaryOperationCpuEvidence, ) { if !seen.insert(node as *const _) { return; } - let operation = match &node.expr { - SummaryExpr::BinaryOp { .. } => cpu.readout_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::Binary(SummaryOperatorResourceEvidence { - physical_id: format!("binary-{node:p}"), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }) - }), - SummaryExpr::ValueOperation { .. } => cpu.readout_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::ValueOperation(SummaryOperatorResourceEvidence { - physical_id: format!("value-operation-{node:p}"), - inputs: vec![test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), + if is_retained(node) { + return; + } + let operation = match &node.operator { + Operator::NonASAP(NonASAPOp::BinaryOp { .. }) => { + cpu.evaluation_cpu_ops.map(|cpu_ops| { + SummaryOperatorEvidence::Binary(resource( + format!("binary-{node:p}"), + vec![test_edge(), test_edge()], + cpu_ops, + 0, + )) }) - }), - SummaryExpr::SummaryMerge { .. } => cpu.merge_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::Merge(SummaryOperatorResourceEvidence { - physical_id: format!("merge-{node:p}"), - inputs: match &node.expr { - SummaryExpr::SummaryMerge { children, .. } => { - vec![test_edge(); children.len()] - } - _ => unreachable!(), - }, - output: test_edge(), + } + Operator::NonASAP(NonASAPOp::Join { .. }) => None, + Operator::NonASAP(_) + | Operator::ASAP( + ASAPOp::FinalizeExactAccumulator { .. } + | ASAPOp::MaintainPopulation { .. } + | ASAPOp::EvaluatePopulation { .. }, + ) => cpu.evaluation_cpu_ops.map(|cpu_ops| { + SummaryOperatorEvidence::ValueOperation(resource( + format!("value-operation-{node:p}"), + vec![test_edge()], cpu_ops, - working_memory_bytes: inputs.state_bytes_per_summary, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }) + 0, + )) }), - SummaryExpr::SummarySubtract { .. } => cpu.subtract_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::Subtract(SummaryOperatorResourceEvidence { - physical_id: format!("subtract-{node:p}"), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: inputs.state_bytes_per_summary, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), + Operator::ASAP(ASAPOp::SummaryMerge { children }) => { + cpu.merge_cpu_ops.map(|cpu_ops| { + SummaryOperatorEvidence::Merge(resource( + format!("merge-{node:p}"), + vec![test_edge(); children.len()], + cpu_ops, + inputs.state_bytes_per_summary, + )) }) - }), - SummaryExpr::SummaryDelete { .. } => cpu.delete_cpu_ops.and_then(|cpu_ops| { - Some(SummaryOperatorEvidence::Delete { - resource: SummaryOperatorResourceEvidence { - physical_id: format!("delete-{node:p}"), - inputs: vec![test_edge()], - output: test_edge(), + } + Operator::ASAP(ASAPOp::SummarySubtract { .. }) => { + cpu.subtract_cpu_ops.map(|cpu_ops| { + SummaryOperatorEvidence::Subtract(resource( + format!("subtract-{node:p}"), + vec![test_edge(), test_edge()], cpu_ops, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, - events_per_second: cpu.delete_events_per_second?, - routing_fanout: cpu.delete_routing_fanout?, + inputs.state_bytes_per_summary, + )) }) - }), - SummaryExpr::SummaryEstimate { .. } => cpu.readout_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::Readout(SummaryOperatorResourceEvidence { - physical_id: format!("readout-{node:p}"), - inputs: vec![test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), + } + Operator::ASAP(ASAPOp::SummaryDelete { .. }) => { + cpu.delete_cpu_ops.and_then(|cpu_ops| { + Some(SummaryOperatorEvidence::Delete { + resource: resource( + format!("delete-{node:p}"), + vec![test_edge()], + cpu_ops, + 0, + ), + events_per_second: cpu.delete_events_per_second?, + routing_fanout: cpu.delete_routing_fanout?, + }) }) - }), - _ => None, + } + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) => { + cpu.evaluation_cpu_ops.map(|cpu_ops| { + SummaryOperatorEvidence::Evaluation(resource( + format!("evaluation-{node:p}"), + vec![test_edge()], + cpu_ops, + 0, + )) + }) + } + Operator::ASAP( + ASAPOp::SummaryAgg { .. } + | ASAPOp::SummaryJoin { .. } + | ASAPOp::Extension { .. }, + ) => None, }; if let Some(operation) = operation { model .node_evidence .operations .insert(node as *const _, operation); - if let SummaryExpr::SummaryDelete { summary_input, .. } = &node.expr { + if let Operator::ASAP(ASAPOp::SummaryDelete { summary_input, .. }) = &node.operator + { fn owning_aggs( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - owners: &mut Vec<*const SummaryNode>, + node: &OperatorNode, + seen: &mut HashSet<*const OperatorNode>, + owners: &mut Vec<*const OperatorNode>, ) { if !seen.insert(node as *const _) { return; } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } => { - owners.push(node as *const _); - owning_aggs(child, seen, owners); - } - SummaryExpr::ValueOperation { child, .. } => { - owning_aggs(child, seen, owners) - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - owning_aggs(child, seen, owners); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - owning_aggs(left, seen, owners); - owning_aggs(right, seen, owners); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - owning_aggs(summary_input, seen, owners); - } - SummaryExpr::KeepPreAsap(_) => {} + if matches!(node.operator, Operator::ASAP(ASAPOp::SummaryAgg { .. })) { + owners.push(node as *const _); + } + for child in node.children() { + owning_aggs(child, seen, owners); } } let mut owners = Vec::new(); @@ -3588,36 +3600,8 @@ mod tests { } } } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } - | SummaryExpr::ValueOperation { child, .. } => { - bind_ops(model, child, seen, inputs, cpu) - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - bind_ops(model, child, seen, inputs, cpu); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - bind_ops(model, left, seen, inputs, cpu); - bind_ops(model, right, seen, inputs, cpu); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - bind_ops(model, summary_input, seen, inputs, cpu) - } - SummaryExpr::KeepPreAsap(_) => {} + for child in node.children() { + bind_ops(model, child, seen, inputs, cpu); } } bind_ops(model, root, &mut HashSet::new(), inputs, cpu); diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs index 867a60e58..fb80d5ee0 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs @@ -3,7 +3,7 @@ use super::*; /// One per-state window choice within a complete Planner candidate. #[derive(Debug, Clone)] pub struct SummaryWindowFrameworkAssignment { - pub summary: Rc, + pub summary: Rc, /// `None` explicitly means that this state is not window-organized. pub framework: Option, } @@ -29,46 +29,20 @@ pub struct SummaryWindowFrameworkCandidate { pub node_evidence: SummaryNodeEvidence, } -pub(super) fn summary_aggregation_identities(root: &SummaryNode) -> HashSet<*const SummaryNode> { +pub(super) fn summary_aggregation_identities(root: &OperatorNode) -> HashSet<*const OperatorNode> { fn visit( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - out: &mut HashSet<*const SummaryNode>, + node: &OperatorNode, + seen: &mut HashSet<*const OperatorNode>, + out: &mut HashSet<*const OperatorNode>, ) { if !seen.insert(node as *const _) { return; } - match &node.expr { - SummaryExpr::KeepPreAsap(_) => {} - SummaryExpr::SummaryAgg { child, .. } => { - out.insert(node as *const _); - visit(child, seen, out); - } - SummaryExpr::ValueOperation { child, .. } => visit(child, seen, out), - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - visit(child, seen, out); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - visit(left, seen, out); - visit(right, seen, out); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - visit(summary_input, seen, out); - } + if matches!(node.operator, Operator::ASAP(ASAPOp::SummaryAgg { .. })) { + out.insert(node as *const _); + } + for child in node.children() { + visit(child, seen, out); } } @@ -146,11 +120,11 @@ impl SummaryWindowAccuracyEvidence { eh_summaries.len() == 1 && eh_summaries.iter().all(|assignment| { matches!( - &assignment.summary.expr, - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + &assignment.summary.operator, + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), .. - } if kind.algorithm() == &SketchAlgorithm::Kll + }) if kind.algorithm() == &SketchAlgorithm::Kll ) }) } @@ -160,14 +134,14 @@ impl SummaryWindowAccuracyEvidence { eh_summaries.len() == 1 && eh_summaries.iter().all(|assignment| { matches!( - &assignment.summary.expr, - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate( + &assignment.summary.operator, + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate( ExactKind::Count | ExactKind::Sum, _ ), .. - } + }) ) }) } diff --git a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs index 129d06ed3..7667238e8 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs @@ -8,12 +8,14 @@ use std::collections::HashMap; use std::rc::Rc; +use asap_types::ir::OperatorNode; use serde::Serialize; use asap_types::dag_export::{self, SummaryDagGraph}; +use asap_types::ir::export::PostAsapNodeId; use asap_types::post_asap::{ - PostAsapNodeId, ResultGuarantee, SummaryExpr, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryNode, SummaryWindowFramework, + ResultGuarantee, SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, + SummaryWindowFramework, }; use crate::summary_maintenance_lifecycle::{ @@ -93,13 +95,7 @@ pub fn export_summary_maintenance_plan( .zip(&deployments) .map(|(deployment, export)| (Rc::as_ptr(&deployment.summary), export)) .collect(); - let mut next_node_id = 0; - annotate_lifecycle_deployments( - &plan.root, - &mut graph, - &deployment_by_summary, - &mut next_node_id, - ); + annotate_lifecycle_deployments(&mut graph, &deployment_by_summary); SummaryMaintenanceDagExport { graph, @@ -116,47 +112,21 @@ pub fn export_summary_maintenance_plan( } } -/// Walk in the same post-order as `dag_export::export_summary` and attach a -/// deployment directly to every flattened occurrence of its state node. -/// This makes the decision visible to graph consumers without asking them to -/// reconstruct pointer identity from graph position. +/// Attach a deployment directly to the exported node of its `SummaryAgg`, +/// matched by the `Rc` identity every exported node carries. This makes the +/// decision visible to graph consumers without asking them to reconstruct +/// pointer identity from graph position. fn annotate_lifecycle_deployments( - node: &SummaryNode, graph: &mut SummaryDagGraph, - deployments: &HashMap<*const SummaryNode, &SummaryMaintenanceDeploymentExport>, - next_node_id: &mut usize, + deployments: &HashMap<*const OperatorNode, &SummaryMaintenanceDeploymentExport>, ) { - if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) { - for child in summary_children(&node.expr) { - annotate_lifecycle_deployments(child, graph, deployments, next_node_id); + for graph_node in &mut graph.nodes { + let Some(source) = &graph_node.source_node else { + continue; + }; + if let Some(deployment) = deployments.get(&Rc::as_ptr(source)) { + graph_node.detail["summary_maintenance"] = + serde_json::to_value(deployment).expect("lifecycle export is serializable"); } } - let graph_node = &mut graph.nodes[*next_node_id]; - if let Some(deployment) = deployments.get(&(node as *const SummaryNode)) { - graph_node.detail["summary_maintenance"] = - serde_json::to_value(deployment).expect("lifecycle export is serializable"); - } - *next_node_id += 1; -} - -fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { - match expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::BinaryOp { lhs, rhs, .. } => vec![lhs, rhs], - SummaryExpr::SummaryAgg { child, .. } => vec![child], - SummaryExpr::ValueOperation { child, .. } => vec![child], - SummaryExpr::SummaryJoin { outer, inner, .. } - | SummaryExpr::RelationalJoin { - left: outer, - right: inner, - .. - } - | SummaryExpr::SummarySubtract { - left: outer, - right: inner, - } => vec![outer, inner], - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => vec![summary_input], - SummaryExpr::SummaryMerge { children, .. } => children.iter().collect(), - } } diff --git a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs index 6bc5b13c7..4b5eb8a6a 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs @@ -17,17 +17,20 @@ //! incrementally. Unknown evidence stays unknown and therefore cannot make a //! long-lived alternative win. +use asap_types::ir::cse::share_common_subdags; use std::collections::{HashMap, HashSet}; use std::rc::Rc; +use asap_types::ir::export::{ + compile_post_asap_dag_with_node_ids, PostAsapDag, PostAsapDagValidationError, PostAsapNodeId, +}; +use asap_types::ir::timing::{apply_lifecycle_timings, LifecycleAssignment, TimingMemo}; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; use asap_types::post_asap::{ - compile_post_asap_dag_with_node_ids, share_common_summary_subtrees, EvaluationSchedule, - ExecutionDataStateError, ExecutionTiming, OutputRepresentation, PostAsapDag, - PostAsapDagValidationError, PostAsapNodeId, ResultGuarantee, SummaryExpr, - SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, - SummaryNode, SummaryWindowFramework, ValueOperation, + EvaluationSchedule, ExecutionDataStateError, ExecutionTiming, OutputRepresentation, + ResultGuarantee, SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, + SummaryMaintenanceMode, SummaryWindowFramework, }; -use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; use asap_types::workload::{ DataArrival, DataWorkload, Predictability, QueryRecurrence, QueryWorkload, RepeatedDemand, @@ -44,7 +47,7 @@ use crate::recurrence::{ }; use crate::replacement::{ CandidateCostOverrides, CandidateLogicalASAPDAGs, GlobalSelection, RealizationError, - Replacement, + Replacement, ReplacementProvenance, }; /// Summary-maintenance lifecycle shapes supported by the target runtime. @@ -168,10 +171,8 @@ pub struct SummaryMaintenanceDeployment { /// It is scoped to one plan version and is not a summary definition or /// summary instance identity. pub post_asap_node_id: PostAsapNodeId, - /// The unique materialized `SummaryAgg`, or maintained population - /// (`MaintainPopulation`) not consumed by a `SummaryAgg`, represented by - /// this deployment. Cost-model lifecycle hooks receive this node. - pub summary: Rc, + /// The unique materialized `SummaryAgg` represented by this deployment. + pub summary: Rc, /// Lifecycle, evaluation, and representation commitment selected for this /// state, or `None` when no alternative is selectable. pub summary_maintenance_lifecycle_guarantee: Option, @@ -187,10 +188,9 @@ pub struct SummaryMaintenanceDeployment { #[derive(Debug, Clone)] pub struct SummaryMaintenanceLifecyclePlan { /// Root of the materialized post-ASAP DAG being deployed. - pub root: Rc, - /// One entry per unique reachable `SummaryAgg`, then per unique - /// maintained population outside any `SummaryAgg`'s inputs; shared `Rc` - /// nodes appear only once. + pub root: Rc, + /// One entry per unique reachable `SummaryAgg`; shared `Rc` nodes appear + /// only once. pub deployments: Vec, /// Caller-supplied optimization horizon used to turn rates into total /// costs. `None` keeps horizon-dependent alternatives unselectable. @@ -239,19 +239,25 @@ impl SummaryMaintenanceLifecyclePlan { /// /// A retained (non-`Ephemeral`) state outlives one query, so it and every /// input it consumes run at ingestion time. Every other node runs at query - /// time: readouts and consumers of retained state, and each `Ephemeral` + /// time: evaluations and consumers of retained state, and each `Ephemeral` /// state not consumed by retained state together with its inputs, whose /// raw data the deployment must supply as a query source. This applies to /// maintained populations as to `SummaryAgg` states; a population feeding /// a `SummaryAgg` is one of its inputs. Timings already on the root are /// ignored. pub fn execution_timed_dag(&self) -> Result { - let compiled = compile_post_asap_dag_with_node_ids(&self.root)?; + let mut memo = TimingMemo::new(); + let timed = apply_lifecycle_timings( + &self.root, + &LifecycleAssignment::default_maintained(), + &mut memo, + )?; + let compiled = compile_post_asap_dag_with_node_ids(&timed)?; let dag = compiled.dag; for population in &standalone_populations(&self.root) { let id = compiled .node_ids - .node_id(population) + .node_id(memo.timed(population).expect("population was timed")) .expect("collected population belongs to the compiled DAG"); if !self .deployments @@ -390,7 +396,7 @@ pub struct SummaryMaintenanceLifecycleCandidates<'a> { arrival: DataArrival, required_accuracy: Vec, cost_model: &'a dyn CostModel, - comparison_target: Option<&'a QueryExpr>, + comparison_target: Option<&'a OperatorNode>, } /// Why an explicit per-state lifecycle choice cannot be bound. @@ -580,7 +586,7 @@ struct SummaryMaintenanceWorkloadFacts { /// unique summary state, and select the cheapest legal alternative whose cost /// is fully known. pub fn plan_summary_maintenance_lifecycles( - root: Rc, + root: Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, @@ -603,7 +609,7 @@ pub fn plan_summary_maintenance_lifecycles( /// alternatives itself binds its choice with /// [`SummaryMaintenanceLifecycleCandidates::select`]. pub fn enumerate_summary_maintenance_lifecycles<'a>( - root: Rc, + root: Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, @@ -627,14 +633,14 @@ pub fn enumerate_summary_maintenance_lifecycles<'a>( /// DAG path multiplicity has been propagated by `CandidateLogicalASAPDAGs`. #[expect(clippy::too_many_arguments, reason = "internal bound planning context")] fn enumerate_with_profile<'a>( - root: Rc, + root: Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &'a dyn CostModel, profile: Option, - comparison_target: Option<&'a QueryExpr>, + comparison_target: Option<&'a OperatorNode>, ) -> Result, SummaryMaintenanceLifecyclePlanError> { demand.workload.validate()?; if let Some(data) = demand.data_workload { @@ -673,11 +679,33 @@ fn enumerate_with_profile<'a>( StateKind::SummaryAgg, ); summaries.extend(standalone_populations(&root)); - let node_ids = compile_post_asap_dag_with_node_ids(&root)?.node_ids; + let mut timing_memo = TimingMemo::new(); + let timed_root = apply_lifecycle_timings( + &root, + &LifecycleAssignment::default_maintained(), + &mut timing_memo, + )?; + let node_ids = compile_post_asap_dag_with_node_ids(&timed_root)?.node_ids; let components = summary_state_components(&summaries); let deployments: Vec = summaries .into_iter() .map(|summary| { + let capabilities = if OperatorNode::reachable(&summary).iter().any(|node| { + matches!( + node.non_asap(), + Some(asap_types::ir::NonASAPOp::BinaryOp { .. }) + ) && asap_types::ir::timing::validate_default(node, ExecutionTiming::IngestionTime) + .is_err() + }) { + SummaryMaintenanceLifecycleCapabilities { + supports_ephemeral: capabilities.supports_ephemeral, + supports_prepared: false, + supports_shared: false, + supports_continuously_maintained: false, + } + } else { + capabilities + }; let alternatives = alternatives_for( &facts, horizon, @@ -686,8 +714,9 @@ fn enumerate_with_profile<'a>( cost_model.summary_maintenance_lifecycle_cost_inputs_for_horizon(&summary, horizon), ); SummaryMaintenanceDeployment { - post_asap_node_id: node_ids - .node_id(&summary) + post_asap_node_id: timing_memo + .timed(&summary) + .and_then(|timed| node_ids.node_id(timed)) .expect("collected summary belongs to the compiled DAG"), summary, summary_maintenance_lifecycle_guarantee: None, @@ -696,7 +725,7 @@ fn enumerate_with_profile<'a>( } }) .collect(); - let selected_raw_recompute = matches!(root.expr, SummaryExpr::KeepPreAsap(_)); + let selected_raw_recompute = !root.contains_asap(); Ok(SummaryMaintenanceLifecycleCandidates { plan: SummaryMaintenanceLifecyclePlan { root, @@ -761,9 +790,14 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( continue; }; for candidate in &group.candidates { - let Replacement::Summary(summary) = &candidate.replacement else { + // Only summary realizations carry a maintenance lifecycle; a + // logical rewrite or CSE share/recompute candidate does not. + let Replacement::SubDag(summary) = &candidate.replacement else { continue; }; + if candidate.provenance != ReplacementProvenance::SummaryRealization { + continue; + } costs.finalize_target(&group.target); let plan = enumerate_with_profile( Rc::clone(summary), @@ -800,14 +834,14 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( // Intern every member once; members whose outermost state (the // `SummaryAgg` every other state of the candidate feeds) interns to the // same node share it. Classes are kept in first-member order. - let interned = share_common_summary_subtrees( + let interned = share_common_subdags( members .iter() .enumerate() .map(|(index, (_, _, summary))| (index, Rc::clone(summary))) .collect(), ); - let mut classes: Vec<(Rc, Vec)> = Vec::new(); + let mut classes: Vec<(Rc, Vec)> = Vec::new(); for (index, root) in interned { let states = summary_states(&root); let Some(state) = states @@ -826,7 +860,7 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( } let mut shared = Vec::new(); for (state, class) in classes { - let mut targets: Vec<&Rc> = Vec::new(); + let mut targets: Vec<&Rc> = Vec::new(); for &index in &class { let target = &members[index].0.target; if !targets.iter().any(|t| Rc::ptr_eq(t, target)) { @@ -902,7 +936,7 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( /// `demand`, or `None` when no lifecycle alternative is selectable for it. /// No comparison target is supplied: the state serves several queries. pub(crate) fn shared_state_cost( - state: &Rc, + state: &Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, @@ -924,7 +958,7 @@ pub(crate) fn shared_state_cost( } /// Every unique `SummaryAgg` reachable from `root`. -pub(crate) fn summary_states(root: &Rc) -> Vec> { +pub(crate) fn summary_states(root: &Rc) -> Vec> { let mut states = Vec::new(); collect_states( root, @@ -939,7 +973,7 @@ pub(crate) fn summary_states(root: &Rc) -> Vec> { /// summary maintenance decisions. This does not create or maintain runtime state. pub fn assemble_selected_dag_with_summary_maintenance_lifecycles( selection: &GlobalSelection<'_>, - target: &Rc, + target: &Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, @@ -966,8 +1000,8 @@ pub fn assemble_selected_dag_with_summary_maintenance_lifecycles( /// [`assemble_selected_dag_with_summary_maintenance_lifecycles`], for a root /// the caller already assembled (and possibly interned across queries). pub(crate) fn plan_assembled_dag( - root: Rc, - target: &Rc, + root: Rc, + target: &Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, @@ -994,7 +1028,7 @@ pub(crate) fn plan_assembled_dag( .is_none_or(|summary| raw.0 <= summary.0) }) { - plan.root = crate::replacement::keep_pre_asap(target)?; + plan.root = crate::replacement::retain_exact(target)?; plan.deployments.clear(); plan.selected_raw_recompute = true; plan.selected_window_implementation_id = None; @@ -1445,66 +1479,35 @@ enum StateKind { /// Collect every unique node of `kind` reachable from `node`. fn collect_states( - node: &Rc, - seen: &mut HashSet<*const SummaryNode>, - output: &mut Vec>, + node: &Rc, + seen: &mut HashSet<*const OperatorNode>, + output: &mut Vec>, kind: StateKind, ) { if !seen.insert(Rc::as_ptr(node)) { return; } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } => { - if kind == StateKind::SummaryAgg { - output.push(Rc::clone(node)); - } - collect_states(child, seen, output, kind); - } - SummaryExpr::ValueOperation { - child, operation, .. - } => { - if kind == StateKind::Population - && matches!(operation, ValueOperation::MaintainPopulation { .. }) - { - output.push(Rc::clone(node)); - } - collect_states(child, seen, output, kind) - } - SummaryExpr::SummaryJoin { outer, inner, .. } - | SummaryExpr::RelationalJoin { - left: outer, - right: inner, - .. - } - | SummaryExpr::BinaryOp { - lhs: outer, - rhs: inner, - .. - } - | SummaryExpr::SummarySubtract { - left: outer, - right: inner, - } => { - collect_states(outer, seen, output, kind); - collect_states(inner, seen, output, kind); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - collect_states(summary_input, seen, output, kind) - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - collect_states(child, seen, output, kind); - } - } - SummaryExpr::KeepPreAsap(_) => {} + if matches!( + (&node.operator, kind), + ( + Operator::ASAP(ASAPOp::SummaryAgg { .. }), + StateKind::SummaryAgg + ) | ( + Operator::ASAP(ASAPOp::MaintainPopulation { .. }), + StateKind::Population + ) + ) { + output.push(Rc::clone(node)); + } + for child in node.children() { + collect_states(child, seen, output, kind); } } /// Maintained populations that are not an input of any `SummaryAgg`. A /// population feeding summary state is on that state's maintenance path, so -/// that state's lifecycle times it, even when a readout also reads it directly. -fn standalone_populations(root: &Rc) -> Vec> { +/// that state's lifecycle times it, even when a evaluation also reads it directly. +fn standalone_populations(root: &Rc) -> Vec> { let mut summaries = Vec::new(); collect_states( root, @@ -1552,7 +1555,7 @@ pub(crate) fn evaluation_schedule( /// Summary states composed on one maintenance path must be produced on the /// same schedule. Return a component id for each collected state. -fn summary_state_components(summaries: &[Rc]) -> Vec { +fn summary_state_components(summaries: &[Rc]) -> Vec { let indices: HashMap<_, _> = summaries .iter() .enumerate() @@ -1568,16 +1571,18 @@ fn summary_state_components(summaries: &[Rc]) -> Vec { } for (parent_index, summary) in summaries.iter().enumerate() { - let SummaryExpr::SummaryAgg { child, .. } = &summary.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary.operator else { continue; }; if !matches!( - child.expr, - SummaryExpr::SummaryAgg { .. } - | SummaryExpr::SummaryJoin { .. } - | SummaryExpr::SummarySubtract { .. } - | SummaryExpr::SummaryDelete { .. } - | SummaryExpr::SummaryMerge { .. } + child.operator, + Operator::ASAP( + ASAPOp::SummaryAgg { .. } + | ASAPOp::SummaryJoin { .. } + | ASAPOp::SummarySubtract { .. } + | ASAPOp::SummaryDelete { .. } + | ASAPOp::SummaryMerge { .. } + ) ) { continue; } @@ -1603,10 +1608,10 @@ fn summary_state_components(summaries: &[Rc]) -> Vec { /// Inputs shared by every complete lifecycle-combination evaluation of one /// root, whether Planner searches combinations or a caller supplies one. struct CompleteCostContext<'a> { - root: &'a SummaryNode, + root: &'a OperatorNode, components: &'a [usize], cost_model: &'a dyn CostModel, - comparison_target: Option<&'a QueryExpr>, + comparison_target: Option<&'a OperatorNode>, horizon: Option, expected_reads: Option, required_accuracy: &'a [AccuracyTarget], @@ -1695,12 +1700,12 @@ fn apply_selection( #[expect(clippy::too_many_arguments, reason = "complete combination context")] fn select_complete_lifecycle_combination( - root: &SummaryNode, + root: &OperatorNode, deployments: &mut [SummaryMaintenanceDeployment], components: &[usize], arrival: DataArrival, cost_model: &dyn CostModel, - comparison_target: Option<&QueryExpr>, + comparison_target: Option<&OperatorNode>, horizon: Option, expected_reads: Option, required_accuracy: &[AccuracyTarget], @@ -1836,12 +1841,16 @@ mod tests { } } use super::*; + use asap_types::ir::export::{NonASAPOpKind, PostAsapOperatorPayload}; + use asap_types::ir::{BinaryOperator, NonASAPOp}; use asap_types::post_asap::{ - ExactKind, ExactParams, GroupingStrategy, PostAsapOperatorPayload, ResultGuarantee, - SketchAlgorithm, SummaryFamilyType, SummaryField, SummarySchema, + ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, ResultGuarantee, Schema, + SketchAlgorithm, }; use asap_types::pre_asap::AggIntent; - use asap_types::pre_asap::{Column, ColumnRef, DataType, QueryExpr, Reduction, Schema, Source}; + use asap_types::pre_asap::{ + ArithmeticOpKind, BinaryOpKind, ColumnRef, DataType, Reduction, Source, + }; use asap_types::types::AccuracyTarget; use asap_types::workload::{ BatchEntry, DataWorkload, DurationMs, Evidence, EvidenceSource, Predictability, Query, @@ -1861,7 +1870,7 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(10.0)), @@ -1874,7 +1883,7 @@ mod tests { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -1897,19 +1906,19 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { UnitCosts.summary_maintenance_lifecycle_cost_inputs(summary) } fn summary_maintenance_capabilities( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { UnitCosts.summary_maintenance_capabilities(summary) } - fn raw_query_recompute_cost(&self, _target: &QueryExpr) -> Option { + fn raw_query_recompute_cost(&self, _target: &OperatorNode) -> Option { Some(Cost(1.0)) } } @@ -1927,14 +1936,14 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { UnitCosts.summary_maintenance_lifecycle_cost_inputs(summary) } fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -1949,7 +1958,7 @@ mod tests { impl CostModel for SummaryMaintenancePrefersDdSketch { fn raw_query_recompute_total_cost( &self, - _target: &QueryExpr, + _target: &OperatorNode, _expected_reads: f64, ) -> Option { Some(Cost(1_000.0)) @@ -1967,7 +1976,7 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { let build = match sketch_algorithm(summary) { Some(SketchAlgorithm::Kll) => 100.0, @@ -1999,8 +2008,8 @@ mod tests { fn complete_summary_candidate_cost( &self, - _root: &SummaryNode, - _target: Option<&QueryExpr>, + _root: &OperatorNode, + _target: Option<&OperatorNode>, deployments: &[CostedSummaryDeployment<'_>], _horizon: Option, _expected_reads: Option, @@ -2032,12 +2041,13 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { + // A leaf summary is one built directly over kept pre-ASAP rows + // (its child is not an ASAP node); a nested one reads state. let is_leaf = matches!( - summary.expr, - SummaryExpr::SummaryAgg { ref child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)) + &summary.operator, + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) if !child.is_asap() ); SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(if is_leaf { 1.0 } else { 100.0 })), @@ -2050,7 +2060,7 @@ mod tests { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -2060,40 +2070,45 @@ mod tests { } } - fn sketch_algorithm(node: &SummaryNode) -> Option { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_algorithm(summary_input), - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), - .. - } => Some(kind.algorithm().clone()), - _ => None, + /// The sketch algorithm of the first `SummaryAgg` reachable from `node` + /// (through a evaluation or any relational operator kept above it). + fn sketch_algorithm(node: &OperatorNode) -> Option { + if let Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), + .. + }) = &node.operator + { + return Some(kind.algorithm().clone()); } + node.children() + .into_iter() + .find_map(|child| sketch_algorithm(child)) } - fn query_root() -> Rc { + fn query_root() -> Rc { query_root_for("m") } - fn query_root_for(metric: &str) -> Rc { - Rc::new(QueryExpr::Scan { + fn query_root_for(metric: &str) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], 0, vec![], ), }) + .unwrap() } - fn sum_query() -> Rc { - Rc::new(QueryExpr::Aggregate { + fn sum_query() -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Sum { col: None }], output_names: vec![], @@ -2101,10 +2116,11 @@ mod tests { having: None, child: query_root(), }) + .unwrap() } - fn quantile_query() -> Rc { - Rc::new(QueryExpr::Aggregate { + fn quantile_query() -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Quantile { col: None, @@ -2116,20 +2132,20 @@ mod tests { having: None, child: query_root(), }) + .unwrap() } - fn summary() -> Rc { - let child = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(query_root()), - schema: SummarySchema { - fields: vec![], - time_index: None, - }, - guarantee: Some(ResultGuarantee::exact("raw")), - }); - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { + /// An exact sum accumulator over the kept pre-ASAP scan. + fn summary() -> Rc { + let child = Rc::new( + query_root() + .as_ref() + .clone() + .with_guarantee(Some(ResultGuarantee::exact("raw"))), + ); + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + OperatorNode::asap_node( + ASAPOp::SummaryAgg { child, family: family.clone(), input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Named( @@ -2139,23 +2155,16 @@ mod tests { grouping: GroupingStrategy::default(), filter: None, }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "state".into(), - dtype: family, - nullable: false, - }], - time_index: None, - }, - guarantee: Some(ResultGuarantee::exact("sum")), - }) + Schema::lifted(vec![Field::new("state", family, false)], None), + Some(ResultGuarantee::exact("sum")), + ) } - fn nested_summary() -> Rc { + fn nested_summary() -> Rc { let child = summary(); - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + OperatorNode::asap_node( + ASAPOp::SummaryAgg { child, family: family.clone(), input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Named( @@ -2165,16 +2174,9 @@ mod tests { grouping: GroupingStrategy::default(), filter: None, }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "state".into(), - dtype: family, - nullable: false, - }], - time_index: None, - }, - guarantee: Some(ResultGuarantee::exact("nested sum")), - }) + Schema::lifted(vec![Field::new("state", family, false)], None), + Some(ResultGuarantee::exact("nested sum")), + ) } fn batch(predictability: Predictability) -> BatchEntry { @@ -2666,7 +2668,13 @@ mod tests { assert_eq!(plan.raw_recompute_total_cost, Some(Cost(1.0))); assert_eq!(plan.summary_total_cost, None); assert!(plan.deployments.is_empty()); - assert!(matches!(plan.root.expr, SummaryExpr::KeepPreAsap(_))); + // The logical query stays exact; deployment assigns execution timing. + assert!(!plan.root.contains_asap()); + assert!(matches!( + plan.root.non_asap(), + Some(NonASAPOp::Aggregate { .. }) + )); + assert!(plan.root.timing.is_none()); let exported = crate::summary_maintenance_dag_export::export_summary_maintenance_plan(&plan); @@ -2788,7 +2796,11 @@ mod tests { .unwrap(); assert!(plan.selected_raw_recompute); assert!(plan.raw_recompute_total_cost.is_none()); - assert!(matches!(plan.root.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!plan.root.contains_asap()); + assert!(matches!( + plan.root.non_asap(), + Some(NonASAPOp::Aggregate { .. }) + )); } #[test] @@ -2813,7 +2825,8 @@ mod tests { assert_eq!(plan.raw_recompute_total_cost, Some(Cost(1.0))); assert_eq!(plan.summary_total_cost, None); assert!(plan.deployments.is_empty()); - assert!(matches!(plan.root.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!plan.root.contains_asap()); + assert!(matches!(plan.root.non_asap(), Some(NonASAPOp::Scan { .. }))); } #[test] @@ -2844,15 +2857,29 @@ mod tests { #[test] fn lifecycle_cost_counts_one_shared_summary_node_once() { + // One shared exact accumulator read twice by the same root: a + // query-time `sum + sum` over one finalized state. (`SummaryMerge` + // is reserved in the unified IR, so the sharing is expressed through + // a relational consumer instead.) let shared = summary(); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: asap_types::post_asap::ExecutionTiming::IngestionTime, - children: vec![Rc::clone(&shared), Rc::clone(&shared)], + let finalized = Rc::new( + OperatorNode::new(Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: Rc::clone(&shared), + })) + .unwrap(), + ); + let root = OperatorNode::non_asap_node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, }, - schema: shared.schema.clone(), - guarantee: None, - }); + return_bool: false, + lhs: Rc::clone(&finalized), + rhs: finalized, + }) + .unwrap(); let workload = workload( vec![batch(Predictability::AdHoc), batch(Predictability::AdHoc)], vec![], @@ -2890,7 +2917,7 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(10.0)), @@ -2903,18 +2930,18 @@ mod tests { fn summary_maintenance_capabilities( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { UnitCosts.summary_maintenance_capabilities(summary) } fn raw_query_recompute_total_cost( &self, - target: &QueryExpr, + target: &OperatorNode, _expected_reads: f64, ) -> Option { - match target { - QueryExpr::Aggregate { measures, .. } => match measures[..] { + match target.non_asap() { + Some(NonASAPOp::Aggregate { measures, .. }) => match measures[..] { [AggIntent::Quantile { q: 0.5, .. }] => Some(Cost(1.0)), _ => Some(Cost(8.0)), }, @@ -2931,7 +2958,7 @@ mod tests { #[test] fn sharing_class_reverts_when_a_member_selects_elsewhere() { let quantile = |q| { - Rc::new(QueryExpr::Aggregate { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Quantile { col: None, @@ -2944,6 +2971,7 @@ mod tests { having: None, child: query_root(), }) + .unwrap() }; let workload = workload(vec![], vec![repeating(), repeating()], at_rest()); // Whether each root selected a summary rather than raw recompute. @@ -3236,14 +3264,10 @@ mod tests { #[test] fn enumeration_lists_each_unique_summary_state_once() { let shared = summary(); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: asap_types::post_asap::ExecutionTiming::IngestionTime, - children: vec![Rc::clone(&shared), Rc::clone(&shared), summary()], - }, - schema: shared.schema.clone(), - guarantee: None, - }); + let root = test_binary( + test_binary(evaluation(&shared), evaluation(&shared)), + evaluation(&summary()), + ); let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); let candidates = enumerate_summary_maintenance_lifecycles( root, @@ -3267,23 +3291,39 @@ mod tests { .any(|deployment| Rc::ptr_eq(&deployment.summary, &shared))); } - fn readout(state: &Rc) -> Rc { - Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { + fn evaluation(state: &Rc) -> Rc { + OperatorNode::asap_node( + ASAPOp::FinalizeExactAccumulator { child: Rc::clone(state), - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::QueryTime, }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), - nullable: false, - }], - time_index: None, - }, - guarantee: Some(ResultGuarantee::exact("sum")), - }) + Schema::lifted( + vec![Field::new( + "value", + FieldDataType::Plain(DataType::Float64), + false, + )], + None, + ), + Some(ResultGuarantee::exact("sum")), + ) + } + + fn test_binary(lhs: Rc, rhs: Rc) -> Rc { + let schema = lhs.schema.clone(); + Rc::new(OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::BinaryOp { + lhs, + rhs, + return_bool: false, + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + }), + schema, + )) } fn lifecycle_matching( @@ -3301,7 +3341,7 @@ mod tests { /// Bind the lifecycle `choose` picks for every state of `root`, then /// derive the timed DAG. fn timed_dag( - root: Rc, + root: Rc, workload: &QueryWorkload, data: &DataWorkload, horizon: Option, @@ -3336,10 +3376,14 @@ mod tests { .iter() .map(|node| { let kind = match node.payload { - PostAsapOperatorPayload::Fallback { .. } => "raw", + PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::BinaryOp { .. }, + } => "binary", + PostAsapOperatorPayload::Relational { .. } => "raw", PostAsapOperatorPayload::SummaryAgg { .. } => "state", - PostAsapOperatorPayload::Value { .. } => "readout", - PostAsapOperatorPayload::Binary { .. } => "binary", + PostAsapOperatorPayload::FinalizeExactAccumulator + | PostAsapOperatorPayload::EvaluatePopulation { .. } => "evaluation", + _ => "other", }; (kind, node.output_state.timing) @@ -3351,7 +3395,7 @@ mod tests { const QUERY: ExecutionTiming = ExecutionTiming::QueryTime; // Every retained lifecycle kind runs its state and inputs at ingestion - // time and its readout at query time. + // time and its evaluation at query time. #[test] fn retained_lifecycles_time_state_and_inputs_at_ingestion() { let mut scheduled = batch(Predictability::Predictable { @@ -3391,7 +3435,7 @@ mod tests { ]; for (workload, data, horizon, kind) in cases { let dag = timed_dag( - readout(&summary()), + evaluation(&summary()), &workload, &data, horizon, @@ -3399,21 +3443,21 @@ mod tests { ); assert_eq!( timings(&dag), - [("raw", INGEST), ("state", INGEST), ("readout", QUERY)] + [("raw", INGEST), ("state", INGEST), ("evaluation", QUERY)] ); } } - // An Ephemeral state, its raw input, and its readout all run at query time. + // An Ephemeral state, its raw input, and its evaluation all run at query time. #[test] fn ephemeral_lifecycle_times_state_and_downstream_at_query() { let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); - let dag = timed_dag(readout(&summary()), &workload, &at_rest(), None, |_| { + let dag = timed_dag(evaluation(&summary()), &workload, &at_rest(), None, |_| { SummaryMaintenanceLifecycle::Ephemeral }); assert_eq!( timings(&dag), - [("raw", QUERY), ("state", QUERY), ("readout", QUERY)] + [("raw", QUERY), ("state", QUERY), ("evaluation", QUERY)] ); } @@ -3422,25 +3466,9 @@ mod tests { #[test] fn shared_state_is_timed_once_for_all_consumers() { let state = summary(); - let lhs = readout(&state); + let lhs = evaluation(&state); let rhs = Rc::new(lhs.as_ref().clone()); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - lhs, - rhs, - operator: asap_types::post_asap::BinaryOperator { - kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( - asap_types::pre_asap::ArithmeticOpKind::Add, - ), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }, - timing: QUERY, - }, - schema: readout(&state).schema.clone(), - guarantee: None, - }); + let root = test_binary(lhs, rhs); let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); let dag = timed_dag(root, &workload, &data, Some(Horizon(10.0)), |deployment| { @@ -3452,8 +3480,8 @@ mod tests { [ ("raw", INGEST), ("state", INGEST), - ("readout", QUERY), - ("readout", QUERY), + ("evaluation", QUERY), + ("evaluation", QUERY), ("binary", QUERY), ] ); @@ -3498,7 +3526,7 @@ mod tests { let data = at_rest(); let demand = WorkloadDemand::new_with_data(&workload, &data, &[0]); let plan = plan_summary_maintenance_lifecycles( - readout(&summary()), + evaluation(&summary()), demand, 1_000, None, @@ -3513,7 +3541,7 @@ mod tests { ) ); let raw = plan_summary_maintenance_lifecycles( - crate::replacement::keep_pre_asap(&sum_query()).unwrap(), + crate::replacement::retain_exact(&sum_query()).unwrap(), demand, 1_000, None, @@ -3523,16 +3551,13 @@ mod tests { .unwrap(); assert_eq!( timings(&raw.execution_timed_dag().unwrap()), - [("raw", QUERY)] + [("raw", QUERY), ("raw", QUERY)] ); } /// A strategy-built `sum(a)` over one maintained current-series population. - fn population_readout() -> Rc { - let target = Rc::new(crate::test_support::lower_promql( - "sum(a)", - AccuracyTarget::Exact, - )); + fn population_evaluation() -> Rc { + let target = crate::test_support::lower_promql("sum(a)", AccuracyTarget::Exact); crate::maintained_population::MaintainedPopulationStrategy::new(std::slice::from_ref( &target, )) @@ -3540,13 +3565,10 @@ mod tests { .unwrap() } - fn is_population(node: &SummaryNode) -> bool { + fn is_population(node: &OperatorNode) -> bool { matches!( - node.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { .. }, - .. - } + node.operator, + Operator::ASAP(ASAPOp::MaintainPopulation { .. }) ) } @@ -3555,9 +3577,7 @@ mod tests { .iter() .zip(timings(dag)) .map(|(node, (kind, timing))| match node.payload { - PostAsapOperatorPayload::Value { - operation: ValueOperation::MaintainPopulation { .. }, - } => ("population", timing), + PostAsapOperatorPayload::MaintainPopulation { .. } => ("population", timing), _ => (kind, timing), }) .collect() @@ -3570,7 +3590,7 @@ mod tests { let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); let candidates = enumerate_summary_maintenance_lifecycles( - population_readout(), + population_evaluation(), WorkloadDemand::new_with_data(&workload, &data, &[0]), 1_000, Some(Horizon(10.0)), @@ -3608,7 +3628,7 @@ mod tests { let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); let plan = plan_summary_maintenance_lifecycles( - population_readout(), + population_evaluation(), WorkloadDemand::new_with_data(&workload, &data, &[0]), 1_000, Some(Horizon(10.0)), @@ -3638,7 +3658,7 @@ mod tests { let workload = workload(vec![], vec![repeating()], data.clone()); let timed = |lifecycle: SummaryMaintenanceLifecycle| { population_timings(&timed_dag( - population_readout(), + population_evaluation(), &workload, &data, Some(Horizon(10.0)), @@ -3647,11 +3667,21 @@ mod tests { }; assert_eq!( timed(SummaryMaintenanceLifecycle::ContinuouslyMaintained), - [("raw", INGEST), ("population", INGEST), ("readout", QUERY)] + [ + ("raw", INGEST), + ("raw", INGEST), + ("population", INGEST), + ("evaluation", QUERY) + ] ); assert_eq!( timed(SummaryMaintenanceLifecycle::Ephemeral), - [("raw", QUERY), ("population", QUERY), ("readout", QUERY)] + [ + ("raw", QUERY), + ("raw", QUERY), + ("population", QUERY), + ("evaluation", QUERY) + ] ); } @@ -3663,7 +3693,7 @@ mod tests { let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); let plan = plan_summary_maintenance_lifecycles( - population_readout(), + population_evaluation(), WorkloadDemand::new_with_data(&workload, &data, &[0]), 1_000, Some(Horizon(10.0)), @@ -3678,7 +3708,12 @@ mod tests { )); assert_eq!( population_timings(&plan.execution_timed_dag().unwrap()), - [("raw", INGEST), ("population", INGEST), ("readout", QUERY)] + [ + ("raw", INGEST), + ("raw", INGEST), + ("population", INGEST), + ("evaluation", QUERY) + ] ); } @@ -3686,39 +3721,39 @@ mod tests { // deployment: the state's lifecycle times it. #[test] fn population_feeding_summary_state_follows_that_state() { - let SummaryExpr::ValueOperation { + let Operator::ASAP(ASAPOp::EvaluatePopulation { child: population, .. - } = &population_readout().expr + }) = &population_evaluation().operator else { unreachable!() }; let state = summary(); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, input, reduction, grouping, .. - } = &state.expr + }) = &state.operator else { unreachable!() }; - let state = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { + let state = Rc::new(OperatorNode { + operator: Operator::ASAP(ASAPOp::SummaryAgg { child: Rc::clone(population), family: family.clone(), input: input.clone(), reduction: reduction.clone(), grouping: grouping.clone(), filter: None, - }, + }), ..state.as_ref().clone() }); let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); let timed = |lifecycle: SummaryMaintenanceLifecycle| { population_timings(&timed_dag( - readout(&state), + evaluation(&state), &workload, &data, Some(Horizon(10.0)), @@ -3731,19 +3766,21 @@ mod tests { assert_eq!( timed(SummaryMaintenanceLifecycle::ContinuouslyMaintained), [ + ("raw", INGEST), ("raw", INGEST), ("population", INGEST), ("state", INGEST), - ("readout", QUERY) + ("evaluation", QUERY) ] ); assert_eq!( timed(SummaryMaintenanceLifecycle::Ephemeral), [ + ("raw", QUERY), ("raw", QUERY), ("population", QUERY), ("state", QUERY), - ("readout", QUERY) + ("evaluation", QUERY) ] ); } @@ -3753,59 +3790,41 @@ mod tests { // timed by the state's lifecycle. #[test] fn shared_population_follows_its_summary_consumer() { - let direct = population_readout(); - let SummaryExpr::ValueOperation { + let direct = population_evaluation(); + let Operator::ASAP(ASAPOp::EvaluatePopulation { child: population, .. - } = &direct.expr + }) = &direct.operator else { unreachable!() }; let state = summary(); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, input, reduction, grouping, .. - } = &state.expr + }) = &state.operator else { unreachable!() }; - let state = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { + let state = Rc::new(OperatorNode { + operator: Operator::ASAP(ASAPOp::SummaryAgg { child: Rc::clone(population), family: family.clone(), input: input.clone(), reduction: reduction.clone(), grouping: grouping.clone(), filter: None, - }, + }), ..state.as_ref().clone() }); - let binary = |lhs: Rc, rhs: Rc| { - Rc::new(SummaryNode { - schema: lhs.schema.clone(), - expr: SummaryExpr::BinaryOp { - lhs, - rhs, - operator: asap_types::post_asap::BinaryOperator { - kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( - asap_types::pre_asap::ArithmeticOpKind::Add, - ), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }, - timing: QUERY, - }, - guarantee: None, - }) - }; + let binary = test_binary; let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); for root in [ - binary(Rc::clone(&direct), readout(&state)), - binary(readout(&state), Rc::clone(&direct)), + binary(Rc::clone(&direct), evaluation(&state)), + binary(evaluation(&state), Rc::clone(&direct)), ] { for lifecycle in [ SummaryMaintenanceLifecycle::ContinuouslyMaintained, @@ -3844,7 +3863,7 @@ mod tests { let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); let mut plan = plan_summary_maintenance_lifecycles( - population_readout(), + population_evaluation(), WorkloadDemand::new_with_data(&workload, &data, &[0]), 1_000, Some(Horizon(10.0)), diff --git a/crates/asap-aware-mapping/src/test_support.rs b/crates/asap-aware-mapping/src/test_support.rs index 612cf7c0d..60fdbca83 100644 --- a/crates/asap-aware-mapping/src/test_support.rs +++ b/crates/asap-aware-mapping/src/test_support.rs @@ -1,11 +1,16 @@ -use asap_types::pre_asap::QueryExpr; +// Shared fixture helpers; not every test module uses every helper. +#![allow(dead_code)] + +use std::rc::Rc; + +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, }; -pub(crate) fn lower_promql(query: &str, accuracy: AccuracyTarget) -> QueryExpr { +pub(crate) fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Rc { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -35,3 +40,159 @@ pub(crate) fn lower_promql(query: &str, accuracy: AccuracyTarget) -> QueryExpr { .pop() .unwrap() } + +// ── Shared pre-ASAP fixture builders ───────────────────────────────────── +// +// Every builder returns an `Rc` whose schema is derived by +// `OperatorNode::non_asap_node`, so a fixture is exactly what a front end +// would hand the planner. Added by the test migration; only add here, never +// rename or remove (several test modules share these). + +use std::time::Duration; + +use asap_types::ir::operator_properties::{GroupKeys, Reduction, Source}; +use asap_types::ir::timing::{apply_lifecycle_timings, LifecycleAssignment, TimingMemo}; +use asap_types::ir::{NonASAPOp, Predicate, ScalarExpr, TimeRangeKind}; +use asap_types::pre_asap::agg_intent::AggIntent; +use asap_types::pre_asap::schema::{ColumnId, DataType, Field, Schema}; + +/// A `TimeSeries("m")` scan over `[ts(0), value(1), labels...]`, time index 0, +/// no unique key. +pub(crate) fn metric_scan(labels: &[&str]) -> Rc { + metric_scan_with_keys(labels, vec![]) +} + +/// [`metric_scan`] with explicit `unique_keys` (a `[[0]]` key makes CSE +/// willing to hoist the scan). +pub(crate) fn metric_scan_with_keys( + labels: &[&str], + unique_keys: Vec>, +) -> Rc { + let mut columns = vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + ]; + columns.extend( + labels + .iter() + .map(|n| Field::plain(*n, DataType::Utf8, true)), + ); + scan("m", Schema::with_time_index(columns, 0, unique_keys)) +} + +/// A predicate-free `TimeSeries(metric)` scan with the given schema. +pub(crate) fn scan(metric: &str, schema: Schema) -> Rc { + scan_from( + Source::TimeSeries { + metric: metric.into(), + }, + schema, + ) +} + +pub(crate) fn scan_from(source: Source, schema: Schema) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Scan { + source, + predicates: vec![], + schema, + }) + .unwrap() +} + +/// A general aggregate node. +pub(crate) fn aggregate( + reduction: Reduction, + measures: Vec, + output_names: Vec, + having: Option, + child: Rc, +) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { + reduction, + measures, + output_names, + filters: vec![], + having, + child, + }) + .unwrap() +} + +/// `intent by (by)` — a single-measure, `HAVING`-free grouped aggregate. +pub(crate) fn agg( + by: Vec, + intent: AggIntent, + child: Rc, +) -> Rc { + aggregate(Reduction::by(by), vec![intent], vec![], None, child) +} + +/// `intent without (excluded)`. +pub(crate) fn without_agg( + excluded: Vec, + intent: AggIntent, + child: Rc, +) -> Rc { + aggregate( + Reduction::Reduce(GroupKeys::without(excluded)), + vec![intent], + vec![], + None, + child, + ) +} + +/// A per-entity (per-series) single-measure aggregate. +pub(crate) fn agg_per_entity(intent: AggIntent, child: Rc) -> Rc { + aggregate(Reduction::PerEntity, vec![intent], vec![], None, child) +} + +pub(crate) fn filter(pred: ScalarExpr, child: Rc) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(pred), + child, + }) + .unwrap() +} + +pub(crate) fn dedup(cols: Vec, child: Rc) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Dedup { cols, child }).unwrap() +} + +/// An explicit range selector `child[range]`. +pub(crate) fn time_range(range: Duration, child: Rc) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::TimeRange { + range, + kind: TimeRangeKind::Range, + child, + }) + .unwrap() +} + +/// `root` timed under the default (every summary maintained) lifecycle +/// assignment — the shape export and the post-ASAP validators consume. +pub(crate) fn timed(root: &Rc) -> Rc { + apply_lifecycle_timings( + root, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .expect("default lifecycle timings apply") +} + +/// Time `root` under the default lifecycle assignment (which runs every +/// data-state / population-contract check) and export it as a post-ASAP DAG — +/// the replacement for the old one-step `post_asap::compile_post_asap_dag`. +pub(crate) fn time_and_export( + root: &Rc, +) -> Result< + asap_types::ir::export::PostAsapDag, + asap_types::post_asap::execution_data_state::ExecutionDataStateError, +> { + let timed = apply_lifecycle_timings( + root, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + )?; + asap_types::ir::export::compile_post_asap_dag(&timed) +} diff --git a/crates/asap-aware-mapping/src/topk_reuse.rs b/crates/asap-aware-mapping/src/topk_reuse.rs index 8329569d1..288762cbe 100644 --- a/crates/asap-aware-mapping/src/topk_reuse.rs +++ b/crates/asap-aware-mapping/src/topk_reuse.rs @@ -7,7 +7,7 @@ use std::rc::Rc; -use asap_types::pre_asap::QueryExpr; +use asap_types::ir::{NonASAPOp, OperatorNode}; use crate::replacement::{ Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, @@ -15,22 +15,23 @@ use crate::replacement::{ /// Derives a smaller top-k result from a compatible larger top-k sibling. pub struct TopKLimitReuseStrategy { - limits: Vec>, + limits: Vec>, } impl TopKLimitReuseStrategy { - pub fn new(limits: &[Rc]) -> Self { + pub fn new(limits: &[Rc]) -> Self { Self { limits: limits.to_vec(), } } - fn larger_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { - let QueryExpr::Limit { - n: target_n, + fn larger_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { + let Some(NonASAPOp::Limit { + n: Some(target_n), offset: 0, child: target_child, - } = target.root.as_ref() + .. + }) = target.root.non_asap() else { return Vec::new(); }; @@ -42,11 +43,12 @@ impl TopKLimitReuseStrategy { if Rc::ptr_eq(candidate, target.root) { return false; } - let QueryExpr::Limit { - n, + let Some(NonASAPOp::Limit { + n: Some(n), offset: 0, child, - } = candidate.as_ref() + .. + }) = candidate.non_asap() else { return false; }; @@ -56,8 +58,8 @@ impl TopKLimitReuseStrategy { .collect(); // Prefer the smallest sufficient materialized top-k when several // larger siblings are available. - sources.sort_by_key(|source| match source.as_ref() { - QueryExpr::Limit { n, .. } => *n, + sources.sort_by_key(|source| match source.non_asap() { + Some(NonASAPOp::Limit { n: Some(n), .. }) => *n, _ => unreachable!(), }); sources @@ -70,34 +72,38 @@ impl ReplacementStrategy for TopKLimitReuseStrategy { } fn replacements(&self, target: &TargetSubDAG<'_>) -> Vec { - let QueryExpr::Limit { - n: target_n, + let Some(NonASAPOp::Limit { + n: Some(target_n), offset: 0, + partition_by, .. - } = target.root.as_ref() + }) = target.root.non_asap() else { return Vec::new(); }; self.larger_sources(target) .into_iter() - .map(|source| { - let source_n = match source.as_ref() { - QueryExpr::Limit { n, .. } => *n, + .filter_map(|source| { + let source_n = match source.non_asap() { + Some(NonASAPOp::Limit { n: Some(n), .. }) => *n, _ => unreachable!(), }; - ReplacementSubDAG { + let rewritten = OperatorNode::non_asap_node(NonASAPOp::Limit { + n: Some(*target_n), + offset: 0, + partition_by: partition_by.clone(), + child: Rc::clone(source), + }) + .ok()?; + Some(ReplacementSubDAG { strategy: "TopKLimitReuseStrategy", - replacement: Replacement::Rewrite(Rc::new(QueryExpr::Limit { - n: *target_n, - offset: 0, - child: Rc::clone(source), - })), + replacement: Replacement::SubDag(rewritten), provenance: ReplacementProvenance::LogicalRewrite, rationale: format!( "derives top-{target_n} from the compatible shared top-{source_n} result; both rank the identical input with the same ordering" ), - } + }) }) .collect() } @@ -106,38 +112,39 @@ impl ReplacementStrategy for TopKLimitReuseStrategy { #[cfg(test)] mod tests { use super::*; - use asap_types::pre_asap::{Schema, Source}; + use crate::test_support::scan; + use asap_types::ir::operator_properties::GroupKeys; + use asap_types::pre_asap::Schema; + + fn scan_named(metric: &str) -> Rc { + scan(metric, Schema::with_time_index(vec![], 0, vec![])) + } - fn scan_named(metric: &str) -> Rc { - Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: metric.into(), - }, - predicates: vec![], - schema: Schema::with_time_index(vec![], 0, vec![]), + fn limit(n: usize, offset: usize, child: Rc) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Limit { + n: Some(n), + offset, + partition_by: GroupKeys::none(), + child, }) + .unwrap() } #[test] fn smaller_limit_reuses_larger_compatible_limit() { let child = scan_named("m"); - let small = Rc::new(QueryExpr::Limit { - n: 5, - offset: 0, - child: Rc::clone(&child), - }); - let large = Rc::new(QueryExpr::Limit { - n: 10, - offset: 0, - child, - }); + let small = limit(5, 0, Rc::clone(&child)); + let large = limit(10, 0, child); let strategy = TopKLimitReuseStrategy::new(&[Rc::clone(&small), Rc::clone(&large)]); let replacements = strategy.replacements(&TargetSubDAG::new(&small)); assert_eq!(replacements.len(), 1); - let Replacement::Rewrite(rewrite) = &replacements[0].replacement else { + let Replacement::SubDag(rewrite) = &replacements[0].replacement else { panic!() }; - let QueryExpr::Limit { n: 5, child, .. } = rewrite.as_ref() else { + let Some(NonASAPOp::Limit { + n: Some(5), child, .. + }) = rewrite.non_asap() + else { panic!() }; assert!(Rc::ptr_eq(child, &large)); @@ -147,21 +154,9 @@ mod tests { fn offset_or_different_input_is_not_reused() { let a = scan_named("a"); let b = scan_named("b"); - let small = Rc::new(QueryExpr::Limit { - n: 5, - offset: 0, - child: a, - }); - let large = Rc::new(QueryExpr::Limit { - n: 10, - offset: 0, - child: b, - }); - let offset = Rc::new(QueryExpr::Limit { - n: 20, - offset: 1, - child: scan_named("a"), - }); + let small = limit(5, 0, a); + let large = limit(10, 0, b); + let offset = limit(20, 1, scan_named("a")); let strategy = TopKLimitReuseStrategy::new(&[Rc::clone(&small), large, offset]); assert!(!strategy.matches(&TargetSubDAG::new(&small))); } diff --git a/crates/asap-aware-mapping/tests/physical_handoff_cost.rs b/crates/asap-aware-mapping/tests/physical_handoff_cost.rs index 0cf89169e..4e8877964 100644 --- a/crates/asap-aware-mapping/tests/physical_handoff_cost.rs +++ b/crates/asap-aware-mapping/tests/physical_handoff_cost.rs @@ -5,7 +5,7 @@ use asap_aware_mapping::analytical_cost::{ use asap_aware_mapping::physical_operator_statistics::{ ComparisonScope, EdgeStatistics, OperatorStatistics, SourceCoverage, UnaryEdgeStatistics, }; -use asap_types::pre_asap::query_expr::Source; +use asap_types::ir::operator_properties::Source; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, }; diff --git a/crates/asap-aware-mapping/tests/storage_io.rs b/crates/asap-aware-mapping/tests/storage_io.rs index d8fd86143..04b0f6a1e 100644 --- a/crates/asap-aware-mapping/tests/storage_io.rs +++ b/crates/asap-aware-mapping/tests/storage_io.rs @@ -5,7 +5,7 @@ use asap_aware_mapping::analytical_cost::{ use asap_aware_mapping::physical_operator_statistics::{ ComparisonScope, EdgeStatistics, OperatorStatistics, SourceCoverage, UnaryEdgeStatistics, }; -use asap_types::pre_asap::query_expr::Source; +use asap_types::ir::operator_properties::Source; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, }; diff --git a/crates/asap-physical-operators/Cargo.toml b/crates/asap-physical-operators/Cargo.toml index 193707c48..3e49d61b8 100644 --- a/crates/asap-physical-operators/Cargo.toml +++ b/crates/asap-physical-operators/Cargo.toml @@ -12,6 +12,7 @@ serde_json = "1" tracing = "0.1" thiserror = "2" regex = "1" +chrono = { version = "=0.4.39", default-features = false, features = ["std"] } [dev-dependencies] diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index 5f6384659..f8055a8c6 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -14,7 +14,7 @@ thread pool. Poll multiple root streams concurrently when they share inputs. `operators::Operator` implements native batch sources, scalar values, projection, filtering, grouped exact aggregation, semi-join, grouped Sort and -Limit, vector-to-scalar conversion, Union, and summary construction/merge/readout. +Limit, vector-to-scalar conversion, Union, and summary construction/merge/evaluation. Sort followed by Limit implements grouped ranking; no dedicated TopK physical operator is needed. Summary construction updates state batch by batch. End of input means the supplied query range or ingestion window is complete. @@ -59,7 +59,7 @@ Plain values preserve Planner scalar/collection types and nullability. Numeric arithmetic uses matching Int64 or Float64 inputs; integer overflow is an error. Boolean predicates use three-valued logic. Native summary states currently cover exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS and CountSketch with candidate heaps. Binding checks family, -parameters and readout compatibility; source batches also validate state payloads. +parameters and evaluation compatibility; source batches also validate state payloads. Existing accumulator algorithms are reused as kernels behind these operators. This crate is owned by ASAPPlanner. Its `planner-types` dependency is the local @@ -78,9 +78,9 @@ See [the design](../../docs/design_docs/physical-planning-and-deployment.md). - `operators`: projection, filter, joins, aggregate/window, sort, limit and summary implementations. - `sources`: raw-source interface, Scan and the memory connector. - `physical_planner`: native operator lowering, typed input contracts and checked instantiation. -- `summary_kernels`: in-memory summary state over `asap_sketchlib` and exact Planner state: merge, typed readout and update adapters. -- `readout`: readouts over merged exact summary states. -- `capability`: explicit kernel, native-batch and typed readout validation. +- `summary_kernels`: in-memory summary state over `asap_sketchlib` and exact Planner state: merge, typed evaluation and update adapters. +- `evaluation`: evaluations over merged exact summary states. +- `capability`: explicit kernel, native-batch and typed evaluation validation. The `dag`, `factory`, `traits` and `arithmetic` paths are re-exports. They contain no alternative execution implementations. diff --git a/crates/asap-physical-operators/src/capability.rs b/crates/asap-physical-operators/src/capability.rs index e2e69be06..8bdb1891e 100644 --- a/crates/asap-physical-operators/src/capability.rs +++ b/crates/asap-physical-operators/src/capability.rs @@ -2,16 +2,16 @@ //! //! `validate_summary_kernel` checks update kernels, including families without a //! native batch representation. `validate_native_family` and -//! `validate_sketch_readout` / `validate_exact_readout` check native state and readout support. -//! Keyed weighted-frequency readouts are checked by `Operator::keyed_readout`. +//! `validate_sketch_evaluation` / `validate_exact_evaluation` check native state and evaluation support. +//! Keyed weighted-frequency evaluations are checked by `Operator::keyed_evaluation`. //! A successful kernel check alone does not mean a physical DAG will bind. //! //! Stored-state encodings belong to deployments. Full plan acceptance is //! owned by `binding`, which also validates schemas, expressions and inputs. use crate::Error; use planner_types::post_asap::{ - ExactKind, ExactParams, GroupingStrategy, SketchAlgorithm, SketchParams, SketchQuery, - SummaryFamilyType, SummaryUpdate, + ExactKind, ExactParams, FieldDataType as SummaryFamilyType, GroupingStrategy, SketchAlgorithm, + SketchParams, SketchStatistic, SummaryUpdate, }; /// Check the same contract used by `create_planner_accumulator` before a plan @@ -184,30 +184,31 @@ pub fn validate_native_family(family: &SummaryFamilyType) -> Result<(), Error> { .map_err(Error::Invalid) } -/// A sketch readout is native only for the families Planner can read directly. -pub fn validate_sketch_readout( +/// A sketch evaluation is native only for the families Planner can read directly. +pub fn validate_sketch_evaluation( family: &SummaryFamilyType, - query: &SketchQuery, + query: &SketchStatistic, ) -> Result<(), Error> { validate_native_family(family)?; use planner_types::post_asap::SketchAlgorithm as A; // A point count without an item value reads the total count. - let bare_count = matches!(query, SketchQuery::PointCount { value: None, .. }); + let bare_count = matches!(query, SketchStatistic::PointCount { value: None, .. }); let supported = match family { SummaryFamilyType::Sketch(kind, _) => match (kind.algorithm(), query) { - (A::Kll, SketchQuery::Quantile { q }) | (A::DDSketch, SketchQuery::Quantile { q }) => { + (A::Kll, SketchStatistic::Quantile { q }) + | (A::DDSketch, SketchStatistic::Quantile { q }) => { if !(0.0..=1.0).contains(q) { return Err(Error::Invalid( - "quantile readout requires quantile in [0,1]".into(), + "quantile evaluation requires quantile in [0,1]".into(), )); } true } (A::DDSketch, _) => bare_count, - (A::Hll, SketchQuery::Cardinality) => true, + (A::Hll, SketchStatistic::Cardinality) => true, (A::Hll, _) => bare_count, // Only count intents read a Count-Min bare count, and their - // updates have unit weight; the readout is typed Int64 on that basis. + // updates have unit weight; the evaluation is typed Int64 on that basis. (A::Cms, _) => bare_count, _ => false, }, @@ -215,22 +216,22 @@ pub fn validate_sketch_readout( }; if !supported { return Err(Error::Invalid( - "readout is not implemented for this summary family".into(), + "evaluation is not implemented for this summary family".into(), )); } Ok(()) } -/// An exact readout must match the exact family it reads. -pub fn validate_exact_readout( +/// An exact evaluation must match the exact family it reads. +pub fn validate_exact_evaluation( family: &SummaryFamilyType, - readout: &crate::summary_kernels::exact::ExactReadout, + evaluation: &crate::summary_kernels::exact::ExactEvaluation, ) -> Result<(), Error> { validate_native_family(family)?; use crate::Statistic as S; use planner_types::post_asap::ExactKind as E; let supported = matches!( - (family, readout.statistic), + (family, evaluation.statistic), (SummaryFamilyType::ExactAggregate(E::Sum, _), S::Sum) | (SummaryFamilyType::ExactAggregate(E::Count, _), S::Count) | (SummaryFamilyType::ExactAggregate(E::Min, _), S::Min) @@ -243,11 +244,11 @@ pub fn validate_exact_readout( ); if !supported { return Err(Error::Invalid( - "readout is not implemented for this summary family".into(), + "evaluation is not implemented for this summary family".into(), )); } - if readout.lookback_ms.is_some_and(|lookback| { - lookback <= 0 || !matches!(readout.statistic, S::Rate | S::Increase) + if evaluation.lookback_ms.is_some_and(|lookback| { + lookback <= 0 || !matches!(evaluation.statistic, S::Rate | S::Increase) }) { return Err(Error::Invalid("invalid exact counter lookback".into())); } diff --git a/crates/asap-physical-operators/src/readout.rs b/crates/asap-physical-operators/src/evaluation.rs similarity index 83% rename from crates/asap-physical-operators/src/readout.rs rename to crates/asap-physical-operators/src/evaluation.rs index e884ecc3f..843145ce2 100644 --- a/crates/asap-physical-operators/src/readout.rs +++ b/crates/asap-physical-operators/src/evaluation.rs @@ -1,4 +1,4 @@ -//! Readouts over merged exact summary states. +//! Evaluations over merged exact summary states. use crate::summary_kernels::exact::ExactAccumulator; use crate::{AggregateCore, KeyByLabelValues, Statistic}; use std::sync::Arc; @@ -12,7 +12,7 @@ fn merge_exact_states( .as_any() .downcast_ref::() .cloned() - .ok_or_else(|| "readout requires Planner exact state".to_string()) + .ok_or_else(|| "evaluation requires Planner exact state".to_string()) }; let mut merged = exact(&states.next().ok_or("empty exact state input")?)?; for state in states { @@ -23,7 +23,7 @@ fn merge_exact_states( Ok(merged) } -/// PromQL counter readouts omit a series with fewer than two samples. Other +/// PromQL counter evaluations omit a series with fewer than two samples. Other /// state/type/range failures remain errors rather than empty results. pub fn insufficient_counter_samples(state: &dyn AggregateCore, statistic: Statistic) -> bool { matches!(statistic, Statistic::Rate | Statistic::Increase) @@ -36,7 +36,7 @@ pub fn insufficient_counter_samples(state: &dyn AggregateCore, statistic: Statis /// Merge already selected exact panes and read one population. `None` means /// the population is absent from the result: a counter with too few samples, /// or an empty MIN/MAX. -pub fn exact_readout( +pub fn exact_evaluation( states: impl IntoIterator>, statistic: Statistic, range_ms: Option<(i64, i64)>, @@ -47,14 +47,14 @@ pub fn exact_readout( return Ok(None); } merged - .readout(statistic, range_ms, key) + .evaluation(statistic, range_ms, key) .map_err(|error| error.to_string()) } #[cfg(test)] mod counter_tests { use super::*; - use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; + use planner_types::post_asap::{ExactKind, ExactParams, FieldDataType as SummaryFamilyType}; fn counter(kind: ExactKind, params: ExactParams, keyed: bool) -> ExactAccumulator { ExactAccumulator::new(SummaryFamilyType::ExactAggregate(kind, params), keyed).unwrap() @@ -76,7 +76,7 @@ mod counter_tests { let key = keyed.then(|| KeyByLabelValues::new_with_labels(vec!["checkout".into()])); state.update(key.as_ref(), 10., 10_000); assert_eq!( - exact_readout( + exact_evaluation( [Arc::new(state) as Arc], statistic, None, @@ -97,15 +97,15 @@ mod counter_tests { let rate = Statistic::Rate; let one = [Arc::new(state.clone()) as Arc]; assert_eq!( - exact_readout(one, rate, Some((0, 60_000)), None).unwrap(), + exact_evaluation(one, rate, Some((0, 60_000)), None).unwrap(), None ); state.update(None, 20., 20_000); let two = || [Arc::new(state.clone()) as Arc]; - assert!(exact_readout(two(), rate, Some((0, 60_000)), None) + assert!(exact_evaluation(two(), rate, Some((0, 60_000)), None) .unwrap() .is_some()); - assert!(exact_readout(two(), rate, Some((60_000, 0)), None).is_err()); - assert!(exact_readout([], rate, Some((0, 60_000)), None).is_err()); + assert!(exact_evaluation(two(), rate, Some((60_000, 0)), None).is_err()); + assert!(exact_evaluation([], rate, Some((0, 60_000)), None).is_err()); } } diff --git a/crates/asap-physical-operators/src/expressions/arithmetic.rs b/crates/asap-physical-operators/src/expressions/arithmetic.rs index e0766763d..64dbb8301 100644 --- a/crates/asap-physical-operators/src/expressions/arithmetic.rs +++ b/crates/asap-physical-operators/src/expressions/arithmetic.rs @@ -20,12 +20,13 @@ pub fn evaluate_float64_arithmetic( /// Execute the Planner binary contract after a deployment has resolved matching rows. pub fn evaluate_binary( - operator: &planner_types::post_asap::BinaryOperator, + operator: &crate::expressions::binary::BinaryOperator, left: f64, right: f64, ) -> Result { + use crate::expressions::binary::BinaryOpKind; use crate::{values::Value, Error}; - use planner_types::pre_asap::{ArithmeticOpKind, BinaryOpKind}; + use planner_types::pre_asap::ArithmeticOpKind; let invalid = || Error::Invalid("unsupported binary operation or invalid checked-division domain".into()); if operator.vector_match.is_some() { diff --git a/crates/asap-physical-operators/src/expressions/binary.rs b/crates/asap-physical-operators/src/expressions/binary.rs new file mode 100644 index 000000000..0c3c5b3b1 --- /dev/null +++ b/crates/asap-physical-operators/src/expressions/binary.rs @@ -0,0 +1,34 @@ +//! Execution configuration for a binary kernel, including comparison evaluation mode. +use planner_types::pre_asap::{ + ArithmeticOpKind, CompareOpKind, PromQLVectorSetOpKind, VectorMatch, +}; +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +pub enum BinaryOpKind { + Arithmetic(ArithmeticOpKind), + Compare(CompareOpKind), + CompareBool(CompareOpKind), + Set(PromQLVectorSetOpKind), +} +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +pub struct BinaryOperator { + pub kind: BinaryOpKind, + pub vector_match: Option, + pub checked_relative_division: bool, + pub checked_finite_division: bool, +} +impl BinaryOperator { + pub fn from_logical(operator: &planner_types::ir::BinaryOperator, return_bool: bool) -> Self { + use planner_types::pre_asap::BinaryOpKind as L; + Self { + kind: match &operator.kind { + L::Arithmetic(op) => BinaryOpKind::Arithmetic(op.clone()), + L::Compare(op) if return_bool => BinaryOpKind::CompareBool(op.clone()), + L::Compare(op) => BinaryOpKind::Compare(op.clone()), + L::Set(op) => BinaryOpKind::Set(op.clone()), + }, + vector_match: operator.vector_match.clone(), + checked_relative_division: operator.checked_relative_division, + checked_finite_division: operator.checked_finite_division, + } + } +} diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs index 627fd9314..d36fd1f5b 100644 --- a/crates/asap-physical-operators/src/expressions/mod.rs +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -5,12 +5,13 @@ use crate::{ }; use planner_types::pre_asap::{ArithmeticOpKind, DataType}; pub mod arithmetic; +pub mod binary; mod planner; pub use planner::CompiledExpression; #[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] pub enum Expression { Binary { - operator: planner_types::post_asap::BinaryOperator, + operator: crate::expressions::binary::BinaryOperator, left: Box, right: Box, }, @@ -64,7 +65,8 @@ impl Expression { left, right, } => { - use planner_types::pre_asap::{BinaryOpKind, CompareOpKind}; + use crate::expressions::binary::BinaryOpKind; + use planner_types::pre_asap::CompareOpKind; let (a, n) = left.dtype(input)?; let (b, m) = right.dtype(input)?; if a != DataType::Float64 || b != a || operator.vector_match.is_some() { diff --git a/crates/asap-physical-operators/src/expressions/planner.rs b/crates/asap-physical-operators/src/expressions/planner.rs index 2130a2f70..1b5f2f4cb 100644 --- a/crates/asap-physical-operators/src/expressions/planner.rs +++ b/crates/asap-physical-operators/src/expressions/planner.rs @@ -3,20 +3,22 @@ use crate::{ values::{Schema, Value}, Error, }; -use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind, DataType, QueryExpr, ScalarValue}; +use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind, DataType, ScalarValue}; + +use planner_types::ir::ScalarExpr; use std::{cmp::Ordering, sync::Arc}; pub(super) fn evaluate( - expr: &QueryExpr, + expr: &ScalarExpr, row: &[Value], schema: &planner_types::pre_asap::Schema, ) -> Result { match expr { - QueryExpr::Column(index) => row.get(*index).cloned().ok_or(Error::Invalid(format!( + ScalarExpr::Column(index) => row.get(*index).cloned().ok_or(Error::Invalid(format!( "column {index} outside row width {}", row.len() ))), - QueryExpr::Literal(value) => Ok(match value { + ScalarExpr::Literal(value) => Ok(match value { ScalarValue::Interval { months, days, @@ -32,18 +34,62 @@ pub(super) fn evaluate( ScalarValue::Boolean(value) => Value::Bool(*value), ScalarValue::Null => Value::Null, }), - QueryExpr::Compare { left, op, right } => { + ScalarExpr::Cast { expr, to, .. } => { + let value = evaluate(expr, row, schema)?; + match (value, to) { + (Value::Null, _) => Ok(Value::Null), + (Value::Int64(value), DataType::Float64) => Ok(Value::Float64(value as f64)), + (value, _) + if expr + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0 + == *to => + { + Ok(value) + } + _ => Err(Error::Invalid("unsupported cast".into())), + } + } + ScalarExpr::Negative { expr, .. } => match evaluate(expr, row, schema)? { + Value::Float64(v) => Ok(Value::Float64(-v)), + Value::Int64(v) => v + .checked_neg() + .map(Value::Int64) + .ok_or_else(|| Error::Invalid("integer negation overflow".into())), + Value::Null => Ok(Value::Null), + _ => Err(Error::Invalid("invalid negation input".into())), + }, + ScalarExpr::Compare { + left, op, right, .. + } => { let left = evaluate(left, row, schema)?; let right = evaluate(right, row, schema)?; compare(op, left, right) } - QueryExpr::Arithmetic { op, left, right } => arithmetic( + ScalarExpr::Arithmetic { + op, left, right, .. + } => arithmetic( op, evaluate(left, row, schema)?, evaluate(right, row, schema)?, ), - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { - let and = matches!(expr, QueryExpr::BoolAnd(_)); + ScalarExpr::Case { + operand: None, + branches, + else_expr, + } => { + for (condition, value) in branches { + if matches!(evaluate(condition, row, schema)?, Value::Bool(true)) { + return evaluate(value, row, schema); + } + } + else_expr + .as_ref() + .map_or(Ok(Value::Null), |e| evaluate(e, row, schema)) + } + ScalarExpr::BoolAnd(parts) | ScalarExpr::BoolOr(parts) => { + let and = matches!(expr, ScalarExpr::BoolAnd(_)); let mut null = false; for part in parts { match evaluate(part, row, schema)? { @@ -55,21 +101,44 @@ pub(super) fn evaluate( } Ok(if null { Value::Null } else { Value::Bool(and) }) } - QueryExpr::Not(value) => match evaluate(value, row, schema)? { + ScalarExpr::Not(value) => match evaluate(value, row, schema)? { Value::Bool(value) => Ok(Value::Bool(!value)), Value::Null => Ok(Value::Null), _ => Err(Error::Invalid("boolean predicate required".into())), }, - QueryExpr::IsNull(value) => Ok(Value::Bool(matches!( + ScalarExpr::IsNull(value) => Ok(Value::Bool(matches!( evaluate(value, row, schema)?, Value::Null ))), - QueryExpr::IsNotNull(value) => Ok(Value::Bool(!matches!( + ScalarExpr::IsNotNull(value) => Ok(Value::Bool(!matches!( evaluate(value, row, schema)?, Value::Null ))), - QueryExpr::FunctionCall { name, args } => { + ScalarExpr::FunctionCall { name, args } => { use planner_types::pre_asap::scalar_signature::MapScalarFunction; + if planner_types::pre_asap::scalar_signature::promql_function_arity(name).is_some() { + let values = args + .iter() + .map(|arg| match evaluate(arg, row, schema)? { + Value::Float64(v) => Ok(v), + _ => Err(Error::Invalid("PromQL function requires floats".into())), + }) + .collect::, _>>()?; + return Ok(Value::Float64(promql_function(name, &values)?)); + } + if name == "promql_drop_metric_name" { + let Value::Utf8(encoded) = evaluate(&args[0], row, schema)? else { + return Err(Error::Invalid("series identity must be Utf8".into())); + }; + let mut labels: std::collections::BTreeMap = + serde_json::from_str(&encoded).map_err(|e| Error::Invalid(e.to_string()))?; + labels.remove("__name__"); + return Ok(Value::Utf8( + serde_json::to_string(&labels) + .map_err(|e| Error::Invalid(e.to_string()))? + .into(), + )); + } if name.eq_ignore_ascii_case("asap_struct_field") { expr.scalar_type(schema) .map_err(|error| Error::Invalid(error.to_string()))?; @@ -81,10 +150,10 @@ pub(super) fn evaluate( unreachable!() }; let offset = match &args[1] { - QueryExpr::Literal(ScalarValue::Int64(index)) => { + ScalarExpr::Literal(ScalarValue::Int64(index)) => { usize::try_from(index - 1).ok() } - QueryExpr::Literal(ScalarValue::Utf8(name)) => { + ScalarExpr::Literal(ScalarValue::Utf8(name)) => { fields.iter().position(|field| &field.name == name) } _ => None, @@ -329,35 +398,120 @@ fn cell_cmp(left: &Value, right: &Value) -> Option { } } +fn promql_function(name: &str, args: &[f64]) -> Result { + let x = args[0]; + Ok(match &name[7..] { + "abs" => x.abs(), + "ceil" => x.ceil(), + "floor" => x.floor(), + "exp" => x.exp(), + "ln" => x.ln(), + "log2" => x.log2(), + "log10" => x.log10(), + "sqrt" => x.sqrt(), + "sgn" => { + if x.is_nan() { + f64::NAN + } else if x == 0.0 { + 0.0 + } else { + x.signum() + } + } + "sin" => x.sin(), + "cos" => x.cos(), + "tan" => x.tan(), + "asin" => x.asin(), + "acos" => x.acos(), + "atan" => x.atan(), + "sinh" => x.sinh(), + "cosh" => x.cosh(), + "tanh" => x.tanh(), + "asinh" => x.asinh(), + "acosh" => x.acosh(), + "atanh" => x.atanh(), + "deg" => x.to_degrees(), + "rad" => x.to_radians(), + "round" => { + let inverse = 1.0 / args[1]; + (x * inverse + 0.5).floor() / inverse + } + "clamp_min" => { + if x.is_nan() || args[1].is_nan() { + f64::NAN + } else { + x.max(args[1]) + } + } + "clamp_max" => { + if x.is_nan() || args[1].is_nan() { + f64::NAN + } else { + x.min(args[1]) + } + } + "clamp" => { + if args.iter().any(|x| x.is_nan()) { + f64::NAN + } else { + x.max(args[1]).min(args[2]) + } + } + part => { + use chrono::{Datelike, Timelike}; + if !x.is_finite() || x < i64::MIN as f64 || x >= i64::MAX as f64 { + return Ok(f64::NAN); + } + let Some(date) = chrono::DateTime::from_timestamp(x as i64, 0) else { + return Ok(f64::NAN); + }; + match part { + "minute" => date.minute() as f64, + "hour" => date.hour() as f64, + "day_of_week" => date.weekday().num_days_from_sunday() as f64, + "day_of_month" => date.day() as f64, + "day_of_year" => date.ordinal() as f64, + "month" => date.month() as f64, + "year" => date.year() as f64, + "days_in_month" => { + let year = date.year(); + let leap = year % 4 == 0 && (year % 100 != 0 || year % 400 == 0); + match date.month() { + 2 => { + if leap { + 29.0 + } else { + 28.0 + } + } + 4 | 6 | 9 | 11 => 30.0, + _ => 31.0, + } + } + _ => return Err(Error::Invalid("unregistered PromQL function".into())), + } + } + }) +} + #[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] pub struct CompiledExpression { - expression: QueryExpr, + expression: ScalarExpr, schema: planner_types::pre_asap::Schema, output: (DataType, bool), } impl CompiledExpression { - pub(crate) fn expression(&self) -> &QueryExpr { + pub(crate) fn expression(&self) -> &ScalarExpr { &self.expression } - pub fn compile(expression: &QueryExpr, input: &Schema) -> Result { - let schema = input - .fields - .iter() - .map(|field| { - let planner_types::post_asap::SummaryFamilyType::Plain(dtype) = &field.dtype else { - return Err(Error::Invalid( - "scalar expression cannot consume opaque summary state".into(), - )); - }; - Ok(planner_types::pre_asap::Column::new( - field.name.clone(), - dtype.clone(), - field.nullable, - )) - }) - .collect::, Error>>()?; - let schema = planner_types::pre_asap::Schema::new(schema); + pub fn compile(expression: &ScalarExpr, input: &Schema) -> Result { + if !input.is_all_plain() { + return Err(Error::Invalid( + "scalar expression cannot consume summary state".into(), + )); + } + let schema = input.as_ref().clone(); validate(expression, &schema)?; let output = expression .scalar_type(&schema) @@ -378,15 +532,13 @@ impl CompiledExpression { "persisted expression type differs from its semantics".into(), )); } - if input.fields.len() != self.schema.columns.len() + if input.fields.len() != self.schema.fields.len() || input .fields .iter() - .zip(&self.schema.columns) + .zip(&self.schema.fields) .any(|(field, column)| { - field.dtype - != planner_types::post_asap::SummaryFamilyType::Plain(column.dtype.clone()) - || field.nullable != column.nullable + field.dtype != column.dtype.clone() || field.nullable != column.nullable }) { return Err(Error::Invalid( @@ -397,11 +549,12 @@ impl CompiledExpression { } /// Evaluate a row under the same typed schema used when binding the expression. pub fn evaluate(&self, row: &[Value]) -> Result { - if row.len() != self.schema.columns.len() - || row - .iter() - .zip(&self.schema.columns) - .any(|(value, column)| !value.matches(&column.dtype, column.nullable)) + if row.len() != self.schema.fields.len() + || row.iter().zip(&self.schema.fields).any(|(value, column)| { + !column + .plain_dtype() + .is_some_and(|dtype| value.matches(dtype, column.nullable)) + }) { return Err(Error::Invalid( "expression input differs from its bound schema".into(), @@ -410,13 +563,27 @@ impl CompiledExpression { evaluate(&self.expression, row, &self.schema) } } -fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Result<(), Error> { +fn validate(expr: &ScalarExpr, schema: &planner_types::pre_asap::Schema) -> Result<(), Error> { let invalid = || Error::Invalid(format!("unsupported scalar expression: {expr:?}")); expr.scalar_type(schema) .map_err(|e| Error::Invalid(e.to_string()))?; match expr { - QueryExpr::Column(_) | QueryExpr::Literal(_) => Ok(()), - QueryExpr::Arithmetic { left, right, .. } => { + ScalarExpr::Column(_) | ScalarExpr::Literal(_) => Ok(()), + ScalarExpr::Cast { expr, to, .. } => { + let source = expr + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0; + if source != *to + && source != DataType::Null + && !(source == DataType::Int64 && *to == DataType::Float64) + { + return Err(invalid()); + } + validate(expr, schema) + } + ScalarExpr::Negative { expr, .. } => validate(expr, schema), + ScalarExpr::Arithmetic { left, right, .. } => { for value in [left, right] { validate(value, schema)?; if !matches!( @@ -431,7 +598,9 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::Compare { left, right, op } => { + ScalarExpr::Compare { + left, right, op, .. + } => { if !matches!( op, CompareOpKind::Eq @@ -475,8 +644,10 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::FunctionCall { name, args } => { - if name != "asap_struct_field" + ScalarExpr::FunctionCall { name, args } => { + if name != "promql_drop_metric_name" + && planner_types::pre_asap::scalar_signature::promql_function_arity(name).is_none() + && name != "asap_struct_field" && name != "asap_element_access" && planner_types::pre_asap::scalar_signature::MapScalarFunction::from_name(name) .is_none() @@ -488,7 +659,29 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + ScalarExpr::Case { + operand: None, + branches, + else_expr, + } => { + for (condition, value) in branches { + validate(condition, schema)?; + if condition + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0 + != DataType::Bool + { + return Err(invalid()); + } + validate(value, schema)?; + } + if let Some(value) = else_expr { + validate(value, schema)?; + } + Ok(()) + } + ScalarExpr::BoolAnd(parts) | ScalarExpr::BoolOr(parts) => { for part in parts { validate(part, schema)?; if !matches!( @@ -502,7 +695,7 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::Not(value) => { + ScalarExpr::Not(value) => { validate(value, schema)?; if !matches!( value @@ -515,7 +708,7 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::IsNull(value) | QueryExpr::IsNotNull(value) => validate(value, schema), + ScalarExpr::IsNull(value) | ScalarExpr::IsNotNull(value) => validate(value, schema), _ => Err(invalid()), } } diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs index 345ee762d..ea6b73eaf 100644 --- a/crates/asap-physical-operators/src/lib.rs +++ b/crates/asap-physical-operators/src/lib.rs @@ -20,7 +20,7 @@ pub use planner_types as planner; pub mod dag; -pub mod readout; +pub mod evaluation; mod error; pub use error::Error; diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs index 89bd9ec5e..87d60c531 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/mod.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -24,7 +24,8 @@ impl Operator { } else { t.clone() }, - false, + !input.has_promql_series_identity() + && (groups.is_empty() || plain(&input, *i)?.1), ) } Reduction::Quantile { column, q } => { @@ -183,7 +184,8 @@ async fn reduce( let mut work = Cooperative::new(context); let mut workspace = Workspace::new(context)?; let mut grouped = BTreeMap::>, Vec>>::new(); - if rows.is_empty() && groups.is_empty() { + // PromQL sums over an empty vector emit no sample. + if rows.is_empty() && groups.is_empty() && !input.has_promql_series_identity() { grouped.insert(vec![], vec![]); } for row in rows { @@ -318,6 +320,9 @@ async fn reduce_one( .ok_or_else(|| invalid("integer aggregate overflow"))?; count += 1; } + if count == 0 && !input.has_promql_series_identity() { + return Ok(Value::Null); + } return if matches!(measure, Reduction::Avg(_)) { Ok(Value::Float64(sum as f64 / count as f64)) } else { @@ -334,6 +339,9 @@ async fn reduce_one( }; floats.push(*v); } + if floats.is_empty() && !input.has_promql_series_identity() { + return Ok(Value::Null); + } Ok(Value::Float64(if matches!(measure, Reduction::Avg(_)) { promql_avg(&floats) } else { diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs index 1d694d9d4..24ffdad99 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -308,7 +308,9 @@ mod tests { values::Batch, }; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{ + Field as SummaryField, FieldDataType as SummaryFamilyType, Schema as SummarySchema, + }, pre_asap::DataType, types::AccuracyTarget, }; @@ -323,14 +325,18 @@ mod tests { name: "time".into(), dtype: SummaryFamilyType::Plain(DataType::Timestamp), nullable: false, + table: None, }, SummaryField { name: "value".into(), dtype: SummaryFamilyType::Plain(DataType::Float64), nullable: false, + table: None, }, ], time_index: Some(0), + unique_keys: vec![], + closed: false, }); for scope in [ Scope::Query { diff --git a/crates/asap-physical-operators/src/operators/aligned_binary.rs b/crates/asap-physical-operators/src/operators/aligned_binary.rs index 941a9f61e..1d0b4be34 100644 --- a/crates/asap-physical-operators/src/operators/aligned_binary.rs +++ b/crates/asap-physical-operators/src/operators/aligned_binary.rs @@ -1,6 +1,8 @@ //! Arithmetic on complete, aligned population/window rows used by precomputation. use super::*; -use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; +use crate::expressions::binary::BinaryOpKind; +use crate::expressions::binary::BinaryOperator; + use std::collections::BTreeSet; impl Operator { diff --git a/crates/asap-physical-operators/src/operators/common.rs b/crates/asap-physical-operators/src/operators/common.rs index 9f1709da4..2daa16f1a 100644 --- a/crates/asap-physical-operators/src/operators/common.rs +++ b/crates/asap-physical-operators/src/operators/common.rs @@ -5,6 +5,8 @@ pub(super) fn invalid(message: &str) -> Error { pub(super) fn schema(fields: Vec) -> Schema { Arc::new(SummarySchema { fields, + unique_keys: vec![], + closed: false, time_index: None, }) } @@ -13,6 +15,7 @@ pub(super) fn result_field(name: &str, dtype: DataType, nullable: bool) -> Summa name: name.into(), dtype: SummaryFamilyType::Plain(dtype), nullable, + table: None, } } diff --git a/crates/asap-physical-operators/src/operators/joins/mod.rs b/crates/asap-physical-operators/src/operators/joins/mod.rs index a43970337..1e111bc15 100644 --- a/crates/asap-physical-operators/src/operators/joins/mod.rs +++ b/crates/asap-physical-operators/src/operators/joins/mod.rs @@ -22,6 +22,15 @@ impl Operator { output: left, }) } + /// Require every candidate key to have an authoritative value at execution. + pub fn certified_semi_join( + left: Schema, + right: Schema, + keys: Vec<(usize, usize)>, + ) -> Result { + Ok(Self::semi_join(left, right, keys)?.require_complete_right()) + } + pub(crate) fn require_complete_right(mut self) -> Self { if let Kind::SemiJoin { require_complete_right, @@ -45,7 +54,7 @@ impl Operator { left: Schema, right: Schema, kind: planner_types::pre_asap::JoinKind, - predicate: &planner_types::pre_asap::Predicate, + predicate: &planner_types::ir::Predicate, output: Schema, ) -> Result { use planner_types::pre_asap::JoinKind; diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index df6ae0796..d4fb7de1a 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -8,7 +8,10 @@ use crate::{ }; use futures::StreamExt; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema, SummaryUpdate}, + post_asap::{ + Field as SummaryField, FieldDataType as SummaryFamilyType, Schema as SummarySchema, + SummaryUpdate, + }, pre_asap::{ColumnRef, DataType}, }; use std::{collections::BTreeMap, sync::Arc}; @@ -34,7 +37,7 @@ pub(crate) mod vector_window; pub use aggregate::Reduction; pub use series_window::SubquerySteps; pub use sort::SortKey; -pub use summary::ReadoutQuery; +pub use summary::SummaryEvaluation; #[derive(Clone, serde::Serialize, serde::Deserialize)] enum Kind { #[serde(skip)] @@ -58,13 +61,13 @@ enum Kind { column: usize, }, VectorBinary { - operator: planner_types::post_asap::BinaryOperator, + operator: crate::expressions::binary::BinaryOperator, return_bool: bool, }, AlignedBinary { keys: Vec<(usize, usize)>, values: (usize, usize), - operator: planner_types::post_asap::BinaryOperator, + operator: crate::expressions::binary::BinaryOperator, }, RangeWindow { intent: Box>, @@ -89,7 +92,7 @@ enum Kind { unique: bool, }, SeriesBinary { - operator: planner_types::post_asap::BinaryOperator, + operator: crate::expressions::binary::BinaryOperator, scalars: [bool; 2], }, SeriesRelabel { @@ -144,7 +147,7 @@ enum Kind { items: Vec, groups: Vec, }, - KeyedReadout { + KeyedEvaluation { state: usize, k: usize, }, @@ -152,9 +155,9 @@ enum Kind { state: usize, groups: Vec, }, - Readout { + Evaluation { state: usize, - query: ReadoutQuery, + query: SummaryEvaluation, }, } /// A bound operation has a fully checked input/output contract before execution. @@ -175,11 +178,11 @@ impl Operator { } } - pub(crate) fn is_counter_readout(&self) -> bool { + pub(crate) fn is_counter_evaluation(&self) -> bool { matches!( self.kind, - Kind::Readout { - query: ReadoutQuery::Exact(crate::summary_kernels::exact::ExactReadout { + Kind::Evaluation { + query: SummaryEvaluation::Exact(crate::summary_kernels::exact::ExactEvaluation { statistic: crate::Statistic::Rate | crate::Statistic::Increase, .. }), @@ -192,21 +195,24 @@ impl Operator { if lookback <= 0 { return Err(invalid("counter lookback must be positive")); } - if let Kind::Readout { - query: ReadoutQuery::Exact(readout), + if let Kind::Evaluation { + query: SummaryEvaluation::Exact(evaluation), .. } = &mut self.kind { - readout.lookback_ms = Some(lookback); + evaluation.lookback_ms = Some(lookback); } Ok(self) } - /// Resolve a counter readout's logical lookback to this run's evaluation range. - pub(super) fn readout_range(&self, context: &RunContext) -> Result, Error> { - let Kind::Readout { + /// Resolve a counter evaluation's logical lookback to this run's evaluation range. + pub(super) fn evaluation_range( + &self, + context: &RunContext, + ) -> Result, Error> { + let Kind::Evaluation { query: - ReadoutQuery::Exact(crate::summary_kernels::exact::ExactReadout { + SummaryEvaluation::Exact(crate::summary_kernels::exact::ExactEvaluation { lookback_ms: Some(lookback), .. }), @@ -335,15 +341,15 @@ impl PhysicalOperator for Operator { Kind::SemiJoin { .. } => "SemiJoin", Kind::Join { .. } => "RelationalJoin", Kind::SummaryBuild { .. } | Kind::KeyedSummaryBuild { .. } => "SummaryAgg", - Kind::KeyedReadout { .. } => "SummaryEstimate", + Kind::KeyedEvaluation { .. } => "SummaryEstimate", Kind::SummaryMerge { .. } => "SummaryMerge", - Kind::Readout { .. } => "SummaryReadout", + Kind::Evaluation { .. } => "SummaryEvaluation", } } fn validate_context(&self, context: &RunContext) -> Result<(), Error> { current_series::validate_context(self, context)?; series_window::validate_context(self, context)?; - self.readout_range(context).map(|_| ()) + self.evaluation_range(context).map(|_| ()) } fn input_schemas(&self) -> Vec { self.inputs.clone() @@ -387,9 +393,9 @@ impl PhysicalOperator for Operator { Kind::Join { .. } | Kind::SemiJoin { .. } => joins::execute(self, inputs, context), Kind::SummaryMerge { .. } => summary::execute_merge(self, inputs, context), Kind::SummaryBuild { .. } - | Kind::Readout { .. } + | Kind::Evaluation { .. } | Kind::KeyedSummaryBuild { .. } - | Kind::KeyedReadout { .. } => summary::execute(self, inputs, context), + | Kind::KeyedEvaluation { .. } => summary::execute(self, inputs, context), } } } diff --git a/crates/asap-physical-operators/src/operators/series_labels.rs b/crates/asap-physical-operators/src/operators/series_labels.rs index 7f468964d..9e29ed026 100644 --- a/crates/asap-physical-operators/src/operators/series_labels.rs +++ b/crates/asap-physical-operators/src/operators/series_labels.rs @@ -1,10 +1,9 @@ //! PromQL label-set rewriting and binary operators over rows that carry a //! series identity or plain label columns. use super::*; -use planner_types::{ - post_asap::BinaryOperator, - pre_asap::{schema::PROMQL_SERIES_IDENTITY, BinaryOpKind, VectorMatchKind}, -}; +use crate::expressions::binary::BinaryOpKind; +use crate::expressions::binary::BinaryOperator; +use planner_types::pre_asap::{schema::PROMQL_SERIES_IDENTITY, VectorMatchKind}; type Labels = BTreeMap; diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs index ef5c49cb5..88cc47db2 100644 --- a/crates/asap-physical-operators/src/operators/summary/mod.rs +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -1,9 +1,9 @@ use super::*; -/// A summary readout: a sketch query, or an exact readout with typed parameters. +/// A summary evaluation: a sketch query, or an exact evaluation with typed parameters. #[derive(Clone, Debug, PartialEq, serde::Serialize, serde::Deserialize)] -pub enum ReadoutQuery { - Sketch(planner_types::post_asap::SketchQuery), - Exact(crate::summary_kernels::exact::ExactReadout), +pub enum SummaryEvaluation { + Sketch(planner_types::post_asap::SketchStatistic), + Exact(crate::summary_kernels::exact::ExactEvaluation), } impl Operator { @@ -47,6 +47,7 @@ impl Operator { name: "state".into(), dtype: family.clone(), nullable: false, + table: None, }); Ok(Self { kind: Kind::KeyedSummaryBuild { @@ -59,7 +60,7 @@ impl Operator { output: schema(fields), }) } - pub fn keyed_readout( + pub fn keyed_evaluation( input: Schema, state: usize, k: usize, @@ -68,23 +69,23 @@ impl Operator { use crate::summary_kernels::weighted_frequency::WeightedFrequency; crate::values::validate_family(&field(&input, state)?.dtype)?; let SummaryFamilyType::Sketch(kind, _) = &field(&input, state)?.dtype else { - return Err(invalid("keyed readout requires summary state")); + return Err(invalid("keyed evaluation requires summary state")); }; let (_, _, _, capacity) = WeightedFrequency::configuration(kind)?; if k > capacity || output.fields.len() <= input.fields.len() { - return Err(invalid("invalid keyed readout shape or capacity")); + return Err(invalid("invalid keyed evaluation shape or capacity")); } if state + 1 != input.fields.len() || output.fields[..state] != input.fields[..state] || output.fields.last().unwrap().dtype != SummaryFamilyType::Plain(DataType::Float64) { return Err(invalid( - "keyed readout must preserve partitions and return a Float64 score", + "keyed evaluation must preserve partitions and return a Float64 score", )); } crate::values::validate_schema(&output)?; Ok(Self { - kind: Kind::KeyedReadout { state, k }, + kind: Kind::KeyedEvaluation { state, k }, inputs: vec![input], output, }) @@ -132,6 +133,7 @@ impl Operator { name: "state".into(), dtype: family.clone(), nullable: false, + table: None, }); Ok(Self { kind: Kind::SummaryBuild { @@ -161,15 +163,19 @@ impl Operator { output: schema(fields), }) } - pub fn readout(input: Schema, state: usize, query: ReadoutQuery) -> Result { + pub fn evaluation( + input: Schema, + state: usize, + query: SummaryEvaluation, + ) -> Result { let family = &field(&input, state)?.dtype; crate::values::validate_family(family)?; match &query { - ReadoutQuery::Sketch(query) => { - crate::capability::validate_sketch_readout(family, query)? + SummaryEvaluation::Sketch(query) => { + crate::capability::validate_sketch_evaluation(family, query)? } - ReadoutQuery::Exact(readout) => { - crate::capability::validate_exact_readout(family, readout)? + SummaryEvaluation::Exact(evaluation) => { + crate::capability::validate_exact_evaluation(family, evaluation)? } } let mut fields = input.fields.clone(); @@ -195,7 +201,7 @@ impl Operator { ); fields[state] = result_field("value", result_type, nullable); Ok(Self { - kind: Kind::Readout { state, query }, + kind: Kind::Evaluation { state, query }, inputs: vec![input], output: schema(fields), }) @@ -204,12 +210,12 @@ impl Operator { /// The Planner reads a Count-Min bare count only for count intents, whose /// output is Int64 and whose updates have unit weight; execution rejects a /// non-integral total rather than rounding it. -fn integral_count(family: &SummaryFamilyType, query: &ReadoutQuery) -> bool { +fn integral_count(family: &SummaryFamilyType, query: &SummaryEvaluation) -> bool { matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::Cms) && matches!( query, - ReadoutQuery::Sketch(planner_types::post_asap::SketchQuery::PointCount { + SummaryEvaluation::Sketch(planner_types::post_asap::SketchStatistic::PointCount { value: None, .. }) @@ -220,7 +226,7 @@ pub(super) fn execute<'a>( mut inputs: Vec>, context: RunContext, ) -> Result, Error> { - let range_ms = operator.readout_range(&context)?; + let range_ms = operator.evaluation_range(&context)?; let output = operator.output.clone(); let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; match &operator.kind { @@ -232,7 +238,7 @@ pub(super) fn execute<'a>( } => Ok(futures::stream::once(async move { Batch::try_new( output, - build_summary(input, family, *value, *time, groups, &context).await?, + build_summary(input, family, *value, *time, groups, !operator.inputs[0].has_promql_series_identity(), &context).await?, ) }) .boxed_local()), @@ -248,7 +254,7 @@ pub(super) fn execute<'a>( ) }) .boxed_local()), - Kind::KeyedReadout { state, k } => Ok(input + Kind::KeyedEvaluation { state, k } => Ok(input .map(move |batch| { let batch = batch?; let mut rows = Vec::new(); @@ -278,20 +284,20 @@ pub(super) fn execute<'a>( Batch::try_new(output.clone(), rows) }) .boxed_local()), - Kind::Readout { state, query } => Ok(input + Kind::Evaluation { state, query } => Ok(input .map(move |batch| { let batch = batch?; let mut rows = batch.rows().to_vec(); - if let ReadoutQuery::Exact(readout) = query { + if let SummaryEvaluation::Exact(evaluation) = query { rows.retain(|row| !matches!(&row[*state], Value::Summary { state: summary, .. } - if crate::readout::insufficient_counter_samples(summary.as_ref(), readout.statistic))); + if crate::evaluation::insufficient_counter_samples(summary.as_ref(), evaluation.statistic))); } for row in &mut rows { let Value::Summary { state: summary, .. } = &row[*state] else { return Err(invalid("summary value required")); }; row[*state] = match query { - ReadoutQuery::Sketch(query) => { + SummaryEvaluation::Sketch(query) => { let value = summary .estimate(query) .map_err(|e| Error::Operator(e.to_string()))?; @@ -309,12 +315,13 @@ pub(super) fn execute<'a>( Value::Float64(value) } } - ReadoutQuery::Exact(readout) => { + SummaryEvaluation::Exact(evaluation) => { let exact = summary .as_any() .downcast_ref::() - .ok_or_else(|| invalid("exact readout requires exact state"))?; - if output.fields[*state].dtype == SummaryFamilyType::Plain(DataType::Int64) { + .ok_or_else(|| invalid("exact evaluation requires exact state"))?; + if output.fields[*state].nullable && exact.is_empty_sum() { Value::Null } + else if output.fields[*state].dtype == SummaryFamilyType::Plain(DataType::Int64) { let count = exact.count().ok_or_else(|| { Error::Operator("exact count state lacks an integer count".into()) })?; @@ -323,7 +330,7 @@ pub(super) fn execute<'a>( })?) } else { match exact - .readout(readout.statistic, range_ms, None) + .evaluation(evaluation.statistic, range_ms, None) .map_err(|e| Error::Operator(e.to_string()))? { Some(value) => Value::Float64(value), @@ -370,6 +377,7 @@ async fn build_summary( value: usize, time: Option, groups: &[usize], + emit_empty_global: bool, context: &RunContext, ) -> Result>, Error> { type State = ( @@ -392,7 +400,8 @@ async fn build_summary( }; let mut work = Cooperative::new(context); let mut states = BTreeMap::>, State>::new(); - if groups.is_empty() { + // PromQL aggregation of an empty vector produces no sample. + if groups.is_empty() && emit_empty_global { states.insert(vec![], create(vec![], 0)?); } let ordered_time = matches!( diff --git a/crates/asap-physical-operators/src/operators/unchecked.rs b/crates/asap-physical-operators/src/operators/unchecked.rs index 7a8d9d95f..06dc072c0 100644 --- a/crates/asap-physical-operators/src/operators/unchecked.rs +++ b/crates/asap-physical-operators/src/operators/unchecked.rs @@ -138,9 +138,7 @@ impl TryFrom for Operator { input(0)?, input(1)?, kind, - &planner_types::pre_asap::Predicate(std::rc::Rc::new( - predicate.expression().clone(), - )), + &planner_types::ir::Predicate(predicate.expression().clone()), output.clone(), )?, Kind::SummaryBuild { @@ -155,13 +153,13 @@ impl TryFrom for Operator { items, groups, } => Operator::keyed_summary_build(input(0)?, family, value, items, groups)?, - Kind::KeyedReadout { state, k } => { - Operator::keyed_readout(input(0)?, state, k, output.clone())? + Kind::KeyedEvaluation { state, k } => { + Operator::keyed_evaluation(input(0)?, state, k, output.clone())? } Kind::SummaryMerge { state, groups } => { Operator::summary_merge(input(0)?, state, groups)? } - Kind::Readout { state, query } => Operator::readout(input(0)?, state, query)?, + Kind::Evaluation { state, query } => Operator::evaluation(input(0)?, state, query)?, } .with_output_schema(output)?; if serde_json::to_value(&op.kind).map_err(|error| invalid(&error.to_string()))? diff --git a/crates/asap-physical-operators/src/operators/vector_binary.rs b/crates/asap-physical-operators/src/operators/vector_binary.rs index 88f019c25..aa34b2f36 100644 --- a/crates/asap-physical-operators/src/operators/vector_binary.rs +++ b/crates/asap-physical-operators/src/operators/vector_binary.rs @@ -1,6 +1,7 @@ //! Label matching and scalar broadcasting are physical computation, not source binding. use super::*; -use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; +use crate::expressions::binary::BinaryOpKind; +use crate::expressions::binary::BinaryOperator; pub(crate) fn value_schema(scalar: bool) -> Schema { let mut fields = Vec::new(); diff --git a/crates/asap-physical-operators/src/operators/vector_window.rs b/crates/asap-physical-operators/src/operators/vector_window.rs index 7592d0b54..14cad166b 100644 --- a/crates/asap-physical-operators/src/operators/vector_window.rs +++ b/crates/asap-physical-operators/src/operators/vector_window.rs @@ -9,6 +9,8 @@ pub(crate) fn matrix_schema() -> Schema { fields.push(result_field("window_end", DataType::Timestamp, false)); Arc::new(SummarySchema { fields, + unique_keys: vec![], + closed: false, time_index: Some(1), }) } diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs index 7cac8ef2b..16bd74d2d 100644 --- a/crates/asap-physical-operators/src/physical_planner/candidates.rs +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -14,7 +14,7 @@ pub struct PhysicalASAPDAG { /// Compile an explicit materialization frontier selected by Planner maintenance /// search. Operators upstream of that frontier run in precompute, including -/// readouts/reductions; query execution receives their typed output values. +/// evaluations/reductions; query execution receives their typed output values. /// Empty frontiers retain the full computation in the query DAG. /// /// Repeated windows must be instantiated with the same evaluation/population @@ -361,14 +361,20 @@ mod tests { let root = asap_frontend_promql::lower_promql_workload(&workload, 0) .unwrap() .remove(0); - let root = std::rc::Rc::new(promql_rows::with_series_identity(&root).unwrap()); + let root = promql_rows::with_series_identity(&root).unwrap(); let space = asap_aware_mapping::search_workload(vec![("q", root)]); let selected = space .global_selection(&asap_aware_mapping::cost_model::DefaultCostModel) .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let dag = planner_types::post_asap::compile_post_asap_dag(&selected).unwrap(); + let selected = planner_types::ir::apply_lifecycle_timings( + &selected, + &Default::default(), + &mut Default::default(), + ) + .unwrap(); + let dag = planner_types::ir::export::compile_post_asap_dag(&selected).unwrap(); let state = dag .nodes .iter() @@ -417,7 +423,14 @@ mod tests { let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, Payload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + Payload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::TimeRange { .. } + } + ) + }) .unwrap(); BTreeMap::from([( u64::from(raw.id.0), diff --git a/crates/asap-physical-operators/src/physical_planner/logical.rs b/crates/asap-physical-operators/src/physical_planner/logical.rs new file mode 100644 index 000000000..8aaa934d6 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/logical.rs @@ -0,0 +1,374 @@ +//! Reconstruct shared operator references from the transport DAG for native lowering. +use super::*; +use planner_types::ir::export::{EdgeRole, NonASAPOpKind as N, PostAsapNodeId, WireScalarExpr}; +use planner_types::ir::{ + ASAPOp, NonASAPOp, Operator as LogicalOperator, OperatorNode, Predicate, ProjectItem, + ScalarExpr, SortKey as LogicalSortKey, +}; +use std::rc::Rc; +pub(super) fn scalar( + expr: &WireScalarExpr, + id_of: &mut impl FnMut(PostAsapNodeId) -> Rc, +) -> ScalarExpr { + fn boxed( + e: &WireScalarExpr, + id_of: &mut impl FnMut(PostAsapNodeId) -> Rc, + ) -> Box { + Box::new(scalar(e, id_of)) + } + fn list( + es: &[WireScalarExpr], + id_of: &mut impl FnMut(PostAsapNodeId) -> Rc, + ) -> Vec { + es.iter().map(|e| scalar(e, id_of)).collect() + } + match expr { + WireScalarExpr::Column(id) => ScalarExpr::Column(*id), + WireScalarExpr::Literal(v) => ScalarExpr::Literal(v.clone()), + WireScalarExpr::Negative { expr, semantics } => ScalarExpr::Negative { + expr: boxed(expr, id_of), + semantics: *semantics, + }, + WireScalarExpr::Compare { + left, + op, + right, + semantics, + } => ScalarExpr::Compare { + left: boxed(left, id_of), + op: op.clone(), + right: boxed(right, id_of), + semantics: *semantics, + }, + WireScalarExpr::BoolAnd(parts) => ScalarExpr::BoolAnd(list(parts, id_of)), + WireScalarExpr::BoolOr(parts) => ScalarExpr::BoolOr(list(parts, id_of)), + WireScalarExpr::Not(e) => ScalarExpr::Not(boxed(e, id_of)), + WireScalarExpr::IsNull(e) => ScalarExpr::IsNull(boxed(e, id_of)), + WireScalarExpr::IsNotNull(e) => ScalarExpr::IsNotNull(boxed(e, id_of)), + WireScalarExpr::Cast { expr, to, try_cast } => ScalarExpr::Cast { + expr: boxed(expr, id_of), + to: to.clone(), + try_cast: *try_cast, + }, + WireScalarExpr::InList { + expr, + list: items, + negated, + } => ScalarExpr::InList { + expr: boxed(expr, id_of), + list: list(items, id_of), + negated: *negated, + }, + WireScalarExpr::FunctionCall { name, args } => ScalarExpr::FunctionCall { + name: name.clone(), + args: list(args, id_of), + }, + WireScalarExpr::Arithmetic { + op, + left, + right, + semantics, + } => ScalarExpr::Arithmetic { + op: op.clone(), + left: boxed(left, id_of), + right: boxed(right, id_of), + semantics: *semantics, + }, + WireScalarExpr::Case { + operand, + branches, + else_expr, + } => ScalarExpr::Case { + operand: operand.as_ref().map(|e| boxed(e, id_of)), + branches: branches + .iter() + .map(|(w, t)| (scalar(w, id_of), scalar(t, id_of))) + .collect(), + else_expr: else_expr.as_ref().map(|e| boxed(e, id_of)), + }, + WireScalarExpr::CurrentTimestamp => ScalarExpr::CurrentTimestamp, + WireScalarExpr::EvalTimestamp => ScalarExpr::EvalTimestamp, + WireScalarExpr::PromqlScalarFromVector(node) => { + ScalarExpr::PromqlScalarFromVector(id_of(*node)) + } + WireScalarExpr::ScalarSubquery(node) => ScalarExpr::ScalarSubquery(id_of(*node)), + WireScalarExpr::Exists { subquery, negated } => ScalarExpr::Exists { + subquery: id_of(*subquery), + negated: *negated, + }, + WireScalarExpr::InSubquery { + expr, + subquery, + negated, + } => ScalarExpr::InSubquery { + expr: boxed(expr, id_of), + subquery: id_of(*subquery), + negated: *negated, + }, + } +} + +pub(super) fn restore(dag: &PostAsapDag) -> Result>, Error> { + dag.validate().map_err(|e| invalid(e.to_string()))?; + let mut done = BTreeMap::new(); + let mut remaining: Vec<_> = dag.nodes.iter().collect(); + while !remaining.is_empty() { + let before = remaining.len(); + let mut next = Vec::new(); + for node in remaining { + let mut edges: Vec<_> = dag.edges.iter().filter(|e| e.consumer == node.id).collect(); + if edges + .iter() + .any(|e| !done.contains_key(&u64::from(e.producer.0))) + { + next.push(node); + continue; + } + edges.sort_by_key(|e| match e.role { + EdgeRole::Left => 0, + EdgeRole::Input => 1, + EdgeRole::Right => 2, + EdgeRole::ScalarRef => 3, + }); + let inputs: Vec<_> = edges + .iter() + .filter(|e| e.role != EdgeRole::ScalarRef) + .map(|e| Rc::clone(&done[&u64::from(e.producer.0)])) + .collect(); + let input = |index: usize| { + inputs + .get(index) + .cloned() + .ok_or_else(|| invalid("operator is missing an input")) + }; + let mut missing = false; + let mut ref_node = |id: PostAsapNodeId| { + if let Some(node) = done.get(&u64::from(id.0)) { + Rc::clone(node) + } else { + missing = true; + Rc::new(OperatorNode::with_schema( + LogicalOperator::NonASAP(NonASAPOp::Values { + rows: vec![], + schema: Default::default(), + }), + Default::default(), + )) + } + }; + let mut value = |expr: &WireScalarExpr| scalar(expr, &mut ref_node); + let operator = match &node.payload { + Payload::Relational { operator } => LogicalOperator::NonASAP(match operator { + N::Scan { + source, + predicates, + schema, + } => NonASAPOp::Scan { + source: source.clone(), + predicates: predicates.iter().map(|p| Predicate(value(&p.0))).collect(), + schema: schema.clone(), + }, + N::Values { rows, schema } => NonASAPOp::Values { + rows: rows + .iter() + .map(|r| r.iter().map(&mut value).collect()) + .collect(), + schema: schema.clone(), + }, + N::Filter { pred } => NonASAPOp::Filter { + pred: Predicate(value(&pred.0)), + child: input(0)?, + }, + N::Project { cols, qualifier } => NonASAPOp::Project { + cols: cols + .iter() + .map(|c| ProjectItem { + alias: c.alias.clone(), + expr: value(&c.expr), + }) + .collect(), + qualifier: qualifier.clone(), + child: input(0)?, + }, + N::Aggregate { + reduction, + measures, + output_names, + filters, + having, + } => NonASAPOp::Aggregate { + reduction: reduction.clone(), + measures: measures.clone(), + output_names: output_names.clone(), + filters: filters + .iter() + .map(|p| p.as_ref().map(|p| Predicate(value(&p.0)))) + .collect(), + having: having.as_ref().map(|p| Predicate(value(&p.0))), + child: input(0)?, + }, + N::Join { join_kind, pred } => NonASAPOp::Join { + kind: join_kind.clone(), + pred: Predicate(value(&pred.0)), + left: input(0)?, + right: input(1)?, + }, + N::SetOp { set_kind, all } => NonASAPOp::SetOp { + kind: set_kind.clone(), + all: *all, + left: input(0)?, + right: input(1)?, + }, + N::Concat { + discriminator_unique_key, + } => NonASAPOp::Concat { + children: inputs.clone(), + discriminator_unique_key: discriminator_unique_key.clone(), + }, + N::Dedup { cols } => NonASAPOp::Dedup { + cols: cols.clone(), + child: input(0)?, + }, + N::Sort { keys, partition_by } => NonASAPOp::Sort { + keys: keys + .iter() + .map(|k| LogicalSortKey { + expr: value(&k.expr), + ascending: k.ascending, + nulls_first: k.nulls_first, + }) + .collect(), + partition_by: partition_by.clone(), + child: input(0)?, + }, + N::Limit { + n, + offset, + partition_by, + } => NonASAPOp::Limit { + n: *n, + offset: *offset, + partition_by: partition_by.clone(), + child: input(0)?, + }, + N::BinaryOp { + operator, + return_bool, + } => NonASAPOp::BinaryOp { + operator: operator.clone(), + return_bool: *return_bool, + lhs: input(0)?, + rhs: input(1)?, + }, + N::SQLWindowFunc { + func, + args, + partition_by, + order_by, + frame, + output_name, + } => NonASAPOp::SQLWindowFunc { + func: func.clone(), + args: args.iter().map(&mut value).collect(), + partition_by: partition_by.clone(), + order_by: order_by + .iter() + .map(|k| LogicalSortKey { + expr: value(&k.expr), + ascending: k.ascending, + nulls_first: k.nulls_first, + }) + .collect(), + frame: frame.clone(), + output_name: output_name.clone(), + child: input(0)?, + }, + N::TimeRange { range, range_kind } => NonASAPOp::TimeRange { + range: *range, + kind: *range_kind, + child: input(0)?, + }, + N::TimeShift { shift } => NonASAPOp::TimeShift { + shift: *shift, + child: input(0)?, + }, + N::PromqlVectorFromScalar { expr } => { + NonASAPOp::PromqlVectorFromScalar(value(expr)) + } + N::PromqlRelabel { dst, value: expr } => NonASAPOp::PromqlRelabel { + dst: dst.clone(), + value: value(expr), + child: input(0)?, + }, + N::PromqlInfoEnrich { selector } => NonASAPOp::PromqlInfoEnrich { + selector: selector.clone(), + child: input(0)?, + }, + N::PromqlSeriesSample { by, sample_kind } => NonASAPOp::PromqlSeriesSample { + by: by.clone(), + kind: *sample_kind, + child: input(0)?, + }, + N::PromqlSubquery { range, resolution } => NonASAPOp::PromqlSubquery { + range: *range, + resolution: *resolution, + child: input(0)?, + }, + }), + Payload::SummaryAgg { + family, + input: update, + reduction, + grouping, + filter, + } => LogicalOperator::ASAP(ASAPOp::SummaryAgg { + child: input(0)?, + family: family.clone(), + input: update.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + filter: filter.as_ref().map(|p| Predicate(value(&p.0))), + }), + Payload::SummaryEstimate { query } => { + LogicalOperator::ASAP(ASAPOp::SummaryEstimate { + summary_input: input(0)?, + query: query.clone(), + }) + } + Payload::FinalizeExactAccumulator => { + LogicalOperator::ASAP(ASAPOp::FinalizeExactAccumulator { child: input(0)? }) + } + Payload::MaintainPopulation { population } => { + LogicalOperator::ASAP(ASAPOp::MaintainPopulation { + child: input(0)?, + population: population.clone(), + }) + } + Payload::EvaluatePopulation { evaluation } => { + LogicalOperator::ASAP(ASAPOp::EvaluatePopulation { + child: input(0)?, + evaluation: evaluation.clone(), + }) + } + Payload::SummaryMerge => LogicalOperator::ASAP(ASAPOp::SummaryMerge { + children: inputs.clone(), + }), + _ => return Err(invalid("reserved ASAP operation has no native lowering")), + }; + if missing { + return Err(invalid( + "scalar reference is not a preceding DAG dependency", + )); + } + let mut rebuilt = OperatorNode::with_schema(operator, node.output_schema.clone()); + rebuilt.guarantee = node.guarantee.clone(); + rebuilt.timing = Some(node.output_state.timing); + done.insert(u64::from(node.id.0), Rc::new(rebuilt)); + } + if next.len() == before { + return Err(invalid("operator DAG is cyclic")); + } + remaining = next; + } + Ok(done) +} diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 72aed2720..0cabbb897 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -1,23 +1,24 @@ //! Compile logical computation to native operators with typed external inputs. //! Compilation needs no readers; deployment resolves inputs after selection. -use crate::operators::ReadoutQuery; -use crate::summary_kernels::exact::ExactReadout; +use crate::operators::SummaryEvaluation; +use crate::summary_kernels::exact::ExactEvaluation; use crate::{ operators::{Expression, Operator, Reduction, SortKey}, plan::{Boundedness, Emission, NodeId, PhysicalDag, PhysicalOperator, PlanProperties}, values::{Batch, Schema}, Error, }; +use planner_types::ir::export::{ + NonASAPOpKind, PostAsapDag, PostAsapDagNode, PostAsapOperatorPayload as Payload, WireScalarExpr, +}; +use planner_types::ir::{ASAPOp, NonASAPOp, Operator as LogicalOperator, OperatorNode, ScalarExpr}; use planner_types::{ - post_asap::{ - ExactOperation, PostAsapDag, PostAsapDagNode, PostAsapOperatorPayload as Payload, - SketchQuery, SummaryFamilyType, SummaryInputExpr, ValueOperation, - }, + post_asap::{FieldDataType, SketchStatistic, SummaryInputExpr}, pre_asap::{ - AggIntent, ColumnRef, CompareOpKind, DataType, GroupKeys, QueryExpr, - Reduction as PlannerReduction, + AggIntent, ColumnRef, CompareOpKind, DataType, GroupKeys, Reduction as PlannerReduction, }, }; +mod logical; use std::{ collections::{BTreeMap, BTreeSet}, sync::Arc, @@ -78,6 +79,7 @@ pub fn bind_with_data_sources<'a>( roots: &[NodeId], data_sources: &crate::sources::DataSources, ) -> Result, Error> { + let restored = logical::restore(dag)?; // Only resolve scans reachable below the selected input boundaries. let mut pending = roots.to_vec(); let mut seen = BTreeSet::new(); @@ -85,16 +87,13 @@ pub fn bind_with_data_sources<'a>( if !seen.insert(id) || sources.contains_key(&id) { continue; } - let node = dag + let _node = dag .nodes .iter() .find(|n| u64::from(n.id.0) == id) .ok_or_else(|| invalid(format!("missing node {id}")))?; - if let Payload::Fallback { - expression: expression @ QueryExpr::Scan { .. }, - } = &node.payload - { - sources.insert(id, Box::new(data_sources.bind(expression)?)); + if matches!(restored[&id].non_asap(), Some(NonASAPOp::Scan { .. })) { + sources.insert(id, Box::new(data_sources.bind(&restored[&id])?)); } else { pending.extend( dag.edges @@ -128,6 +127,7 @@ fn compile_internal( roots: &[NodeId], ) -> Result { preflight_depth(dag)?; + let restored = logical::restore(dag)?; dag.validate().map_err(|e| invalid(e.to_string()))?; let nodes = dag .nodes @@ -141,46 +141,43 @@ fn compile_internal( ( edge.consumer.0, match edge.role { - planner_types::post_asap::EdgeRole::Left => 0, - planner_types::post_asap::EdgeRole::Input => 1, - planner_types::post_asap::EdgeRole::Right => 2, + planner_types::ir::export::EdgeRole::Left => 0, + planner_types::ir::export::EdgeRole::Input => 1, + planner_types::ir::export::EdgeRole::Right => 2, + planner_types::ir::export::EdgeRole::ScalarRef => 3, }, ) }); - // Scalar literal operands of query-time arithmetic are folded into the consumer. - let mut literals = BTreeMap::::new(); + let literals = BTreeMap::::new(); for edge in edges { - let consumer = u64::from(edge.consumer.0); - if let ( - Payload::Fallback { expression }, - Some(PostAsapDagNode { - payload: Payload::Binary { .. }, - .. - }), - ) = ( - &nodes[&u64::from(edge.producer.0)].payload, - nodes.get(&consumer), - ) { - if let Some(value) = row_values::scalar_literal(expression) { - let left = edge.role == planner_types::post_asap::EdgeRole::Left; - if literals.insert(consumer, (value, left)).is_some() { - return Err(invalid("binary with two scalar literals is not folded")); - } - continue; - } - } dependencies .entry(u64::from(edge.consumer.0)) .or_default() .push(u64::from(edge.producer.0)); } + let mut fallback = BTreeMap::new(); + for (&id, root) in &restored { + let raw_summary_input = matches!(root.non_asap(), Some(NonASAPOp::TimeRange { .. })) + && dag.edges.iter().any(|e| { + u64::from(e.producer.0) == id + && matches!( + nodes[&u64::from(e.consumer.0)].payload, + Payload::SummaryAgg { .. } + ) + }); + if !root.contains_asap() && !raw_summary_input { + if let Ok(lowered) = promql_fallback::lower(root) { + fallback.insert(id, lowered); + } + } + } let known = |id: &NodeId| { nodes.contains_key(id) || promql_fallback::raw_series_owner(*id).is_some_and(|owner| { matches!( nodes.get(&owner), Some(PostAsapDagNode { - payload: Payload::Fallback { .. }, + payload: Payload::Relational { .. }, .. }) ) @@ -204,7 +201,7 @@ fn compile_internal( return Err(invalid(format!("missing root {id}"))); } pending.push((id, true)); - if !sources.contains_key(&id) { + if !sources.contains_key(&id) && !fallback.contains_key(&id) { for &input in dependencies.get(&id).into_iter().flatten() { pending.push((input, false)); } @@ -241,20 +238,11 @@ fn compile_internal( inputs = vec![auxiliary]; schemas.truncate(1); } - // A consumed bare selector supplies raw range rows (e.g. to a - // per-entity summary), not an instant vector, so only its consumer computes. - let raw_rows = matches!( - &node.payload, - Payload::Fallback { - expression: QueryExpr::TimeRange { .. } - } - ) && dag.edges.iter().any(|e| u64::from(e.producer.0) == id); - if let (Payload::Fallback { expression }, false) = (&node.payload, raw_rows) { - let promql_fallback::Lowering { - selectors, - mut steps, - } = promql_fallback::lower(expression) - .map_err(|error| invalid(format!("node {id}: {error}")))?; + if let Some(promql_fallback::Lowering { + selectors, + mut steps, + }) = fallback.remove(&id) + { let mut slots = Vec::new(); for (i, (_, schema)) in selectors.iter().enumerate() { let slot = promql_fallback::raw_series_input(id, i); @@ -300,10 +288,7 @@ fn compile_internal( )?; continue; } - if let Payload::Value { - operation: ValueOperation::MaintainPopulation { population }, - } = &node.payload - { + if let Payload::MaintainPopulation { population } = &node.payload { use planner_types::post_asap::maintained_population::PopulationInput; let PopulationInput::CurrentSeries(spec) = &population.input else { return Err(invalid( @@ -336,22 +321,16 @@ fn compile_internal( )?; continue; } - if let Payload::Value { - operation: ValueOperation::ReadPopulation { readout }, - } = &node.payload - { + if let Payload::EvaluatePopulation { evaluation } = &node.payload { use planner_types::post_asap::maintained_population::{ - PopulationInput, PopulationReadout, + PopulationInput, PopulationStatistic, }; let [producer] = inputs.as_slice() else { - return Err(invalid("population readout requires one input")); + return Err(invalid("population evaluation requires one input")); }; - let Payload::Value { - operation: ValueOperation::MaintainPopulation { population }, - } = &nodes[producer].payload - else { + let Payload::MaintainPopulation { population } = &nodes[producer].payload else { return Err(invalid( - "population readout requires its declared population", + "population evaluation requires its declared population", )); }; let PopulationInput::CurrentSeries(spec) = &population.input else { @@ -363,9 +342,9 @@ fn compile_internal( )); } let input = schemas[0].clone(); - let PopulationReadout::TopK { k } = readout else { + let PopulationStatistic::TopK { k } = evaluation else { let mut chain = - row_values::population_aggregate(&input, &spec.grouping, readout)?; + row_values::population_aggregate(&input, &spec.grouping, evaluation)?; let last = chain.pop().expect("nonempty chain"); let mut inputs = inputs; for operator in chain { @@ -415,15 +394,12 @@ fn compile_internal( let [input_id] = inputs.as_slice() else { return Err(invalid("per-entity summary requires one input")); }; - let Payload::Fallback { - expression: QueryExpr::TimeRange { child, .. }, - } = &nodes[input_id].payload - else { + let Some(NonASAPOp::TimeRange { child, .. }) = restored[input_id].non_asap() else { return Err(invalid( "per-entity summary requires a resolved raw time range", )); }; - let QueryExpr::Scan { schema, .. } = child.as_ref() else { + let Some(NonASAPOp::Scan { schema, .. }) = child.non_asap() else { return Err(invalid("per-entity summary requires a resolved source")); }; if !schema.closed || update.item.is_some() { @@ -462,7 +438,18 @@ fn compile_internal( )?; continue; } - if let Payload::Binary { operator } = &node.payload { + if let Payload::Relational { + operator: + NonASAPOpKind::BinaryOp { + operator, + return_bool, + }, + } = &node.payload + { + let operator = crate::expressions::binary::BinaryOperator::from_logical( + operator, + *return_bool, + ); let query_time = node.output_state.timing == planner_types::post_asap::ExecutionTiming::QueryTime; if let Some(&(value, left)) = literals.get(&id) { @@ -499,25 +486,17 @@ fn compile_internal( schema .fields .iter() - .any(|f| matches!(f.dtype, SummaryFamilyType::Plain(DataType::Map { .. }))) + .any(|f| matches!(f.dtype, FieldDataType::Plain(DataType::Map { .. }))) }; // Grouped rows carry their labels as columns; per-series rows // carry the series identity. if let (true, [left, right]) = (query_time, schemas.as_slice()) { if !label_map(left) && !label_map(right) { - // A scalar-valued Fallback operand, such as `scalar(x)`, has no labels. - let scalar = |input: &NodeId| { - matches!( - nodes.get(input).map(|node| &node.payload), - Some(Payload::Fallback { expression }) - if promql_fallback::scalar(expression) - ) - }; let binary = Operator::series_binary( left.clone(), right.clone(), operator.clone(), - [scalar(&inputs[0]), scalar(&inputs[1])], + [false, false], ) .map_err(|error| invalid(format!("node {id}: {error}")))?; graph.add(id, inputs, binary.with_output_schema(output)?)?; @@ -525,17 +504,14 @@ fn compile_internal( } } } - if let Payload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - } = &node.payload - { + if let Payload::FinalizeExactAccumulator = &node.payload { // Exact counts read out as Int64; PromQL declares a Float64 sample. - let readout = bind_operation(node, &schemas) + let evaluation = bind_operation(node, &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; - let actual = readout.schema(); + let actual = evaluation.schema(); let converted = actual.fields.iter().zip(&output.fields).position(|(a, d)| { - a.dtype == SummaryFamilyType::Plain(DataType::Int64) - && d.dtype == SummaryFamilyType::Plain(DataType::Float64) + a.dtype == FieldDataType::Plain(DataType::Int64) + && d.dtype == FieldDataType::Plain(DataType::Float64) }); if let Some(column) = converted { let columns = actual @@ -555,8 +531,8 @@ fn compile_internal( .collect(); let project = Operator::project(actual, columns)?.with_output_schema(output.clone())?; - graph.add(auxiliary, inputs, readout)?; - if temporal_readout_drops_name(node) { + graph.add(auxiliary, inputs, evaluation)?; + if temporal_evaluation_drops_name(node) { graph.add(auxiliary - 1, vec![auxiliary], project)?; graph.add( id, @@ -572,7 +548,7 @@ fn compile_internal( } let mut operator = compile_node(node, &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; - if operator.is_counter_readout() { + if operator.is_counter_evaluation() { let mut pending = vec![id]; let mut visited = BTreeSet::new(); let mut ranges = BTreeSet::new(); @@ -580,8 +556,8 @@ fn compile_internal( if !visited.insert(ancestor) { continue; } - if let Payload::Fallback { - expression: QueryExpr::TimeRange { range, .. }, + if let Payload::Relational { + operator: NonASAPOpKind::TimeRange { range, .. }, } = &nodes[&ancestor].payload { ranges.insert( @@ -593,13 +569,13 @@ fn compile_internal( pending.extend(dependencies.get(&ancestor).into_iter().flatten().copied()); } if ranges.len() > 1 { - return Err(invalid("counter readout has ambiguous logical windows")); + return Err(invalid("counter evaluation has ambiguous logical windows")); } if let Some(lookback) = ranges.into_iter().next() { operator = operator.with_counter_lookback(lookback)?; } } - if temporal_readout_drops_name(node) { + if temporal_evaluation_drops_name(node) { graph.add(auxiliary, inputs, operator)?; graph.add(id, vec![auxiliary], Operator::series_without_name(output)?)?; } else { @@ -611,24 +587,23 @@ fn compile_internal( Ok(graph) } -// Temporal summary readouts produce PromQL vectors, whose range functions drop +// Temporal summary evaluations produce PromQL vectors, whose range functions drop // the metric name before matching/filtering. Stored state retains its full identity. -fn temporal_readout_drops_name(node: &PostAsapDagNode) -> bool { +fn temporal_evaluation_drops_name(node: &PostAsapDagNode) -> bool { node.output_schema .fields .iter() .any(|field| field.name == promql_rows::SERIES_IDENTITY_COLUMN) && matches!( &node.payload, - Payload::Value { - operation: ValueOperation::FinalizeExactAccumulator - } | Payload::SummaryEstimate { - query: SketchQuery::Quantile { .. } - | SketchQuery::Cardinality - | SketchQuery::PointCount { .. } - | SketchQuery::FrequencyL2 - | SketchQuery::FrequencyEntropy - } + Payload::FinalizeExactAccumulator + | Payload::SummaryEstimate { + query: SketchStatistic::Quantile { .. } + | SketchStatistic::Cardinality + | SketchStatistic::PointCount { .. } + | SketchStatistic::FrequencyL2 + | SketchStatistic::FrequencyEntropy + } ) } @@ -642,7 +617,15 @@ pub fn compile_node(node: &PostAsapDagNode, inputs: &[Schema]) -> Result Result { - if let Payload::Binary { operator } = &node.payload { + if let Payload::Relational { + operator: NonASAPOpKind::BinaryOp { + operator, + return_bool, + }, + } = &node.payload + { + let operator = + crate::expressions::binary::BinaryOperator::from_logical(operator, *return_bool); let [left, right] = inputs else { return Err(invalid("binary requires two inputs")); }; @@ -654,7 +637,7 @@ fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result>(); @@ -688,54 +671,83 @@ fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result, Error>>() + }) + .collect::, Error>>()?; + let schema = Arc::new(schema.clone()); + return Operator::source( + schema.clone(), + vec![crate::values::Batch::try_new(schema, rows)?], + ); + } let [input] = inputs else { return Err(invalid( "native Planner binding currently requires a unary operation or an explicit source", )); }; match &node.payload { - Payload::Value { operation, .. } => match operation { - ValueOperation::Project { cols, .. } => Operator::project( + Payload::FinalizeExactAccumulator => { + let state = summary_column(input)?; + use crate::Statistic as S; + use planner_types::post_asap::ExactKind as E; + let statistic = match &input.fields[state].dtype { + FieldDataType::ExactAggregate(kind, _) => match kind { + E::Sum => S::Sum, + E::Count => S::Count, + E::Min => S::Min, + E::Max => S::Max, + E::Rate => S::Rate, + E::Increase => S::Increase, + _ => return Err(invalid("exact family evaluation is unsupported")), + }, + _ => return Err(invalid("exact finalization requires exact state")), + }; + Operator::evaluation( + input.clone(), + state, + SummaryEvaluation::Exact(ExactEvaluation { + statistic, + lookback_ms: None, + }), + ) + } + + Payload::Relational { operator } => match operator { + NonASAPOpKind::Project { cols, .. } => Operator::project( input.clone(), cols.iter() .enumerate() @@ -748,21 +760,21 @@ fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result Expression::Column(*index), + WireScalarExpr::Column(index) => Expression::Column(*index), expr => expression(expr, input)?, }, )) }) .collect::>()?, ), - ValueOperation::Filter { pred } => { + NonASAPOpKind::Filter { pred } => { Operator::filter(input.clone(), expression(&pred.0, input)?) } - ValueOperation::Sort { keys, partition_by } => Operator::sort( + NonASAPOpKind::Sort { keys, partition_by } => Operator::sort( input.clone(), keys.iter() .map(|key| { - let QueryExpr::Column(column) = key.expr else { + let WireScalarExpr::Column(column) = key.expr else { return Err(invalid( "sort expression must be projected before sorting", )); @@ -776,23 +788,23 @@ fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result>()?, groups(input, partition_by)?, ), - ValueOperation::Limit { + NonASAPOpKind::Limit { n, offset, partition_by, } => Operator::limit( input.clone(), - *n as u64, + n.unwrap_or(usize::MAX) as u64, *offset as u64, groups(input, partition_by)?, ), - ValueOperation::Exact(ExactOperation::Aggregate { + NonASAPOpKind::Aggregate { reduction, measures, output_names, filters, having: None, - }) => { + } => { if filters.iter().any(Option::is_some) { return Err(invalid("filtered aggregate has no native implementation")); } @@ -829,31 +841,6 @@ fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result>()?; Operator::aggregate(input.clone(), groups(input, keys)?, measures) } - ValueOperation::FinalizeExactAccumulator => { - let state = summary_column(input)?; - use crate::Statistic as S; - use planner_types::post_asap::ExactKind as E; - let statistic = match &input.fields[state].dtype { - SummaryFamilyType::ExactAggregate(kind, _) => match kind { - E::Sum => S::Sum, - E::Count => S::Count, - E::Min => S::Min, - E::Max => S::Max, - E::Rate => S::Rate, - E::Increase => S::Increase, - _ => return Err(invalid("exact family readout is unsupported")), - }, - _ => return Err(invalid("exact finalization requires exact state")), - }; - Operator::readout( - input.clone(), - state, - ReadoutQuery::Exact(ExactReadout { - statistic, - lookback_ms: None, - }), - ) - } _ => Err(invalid("value operation has no native implementation")), }, Payload::SummaryAgg { @@ -877,7 +864,7 @@ fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result Result { - if let SketchQuery::TopK { k } = query { - return Operator::keyed_readout( + if let SketchStatistic::TopK { k } = query { + return Operator::keyed_evaluation( input.clone(), summary_column(input)?, *k, Arc::new(node.output_schema.clone()), ); } - Operator::readout( + Operator::evaluation( input.clone(), summary_column(input)?, - ReadoutQuery::Sketch(query.clone()), + SummaryEvaluation::Sketch(query.clone()), ) } _ => Err(invalid( @@ -968,7 +955,7 @@ fn summary_column(input: &Schema) -> Result { .fields .iter() .enumerate() - .filter(|(_, f)| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + .filter(|(_, f)| !matches!(f.dtype, FieldDataType::Plain(_))) .map(|(i, _)| i) .collect::>(); match columns.as_slice() { @@ -978,7 +965,7 @@ fn summary_column(input: &Schema) -> Result { } fn named_column(input: &Schema, column: &ColumnRef) -> Result { let name = match column { - // Executable SummarySchema retains column names, not table qualifiers. + // Executable Schema retains column names, not table qualifiers. // Frontend binding has resolved the qualifier; still reject ambiguous // names here rather than guessing a join side. ColumnRef::Named(name) | ColumnRef::Qualified { name, .. } => name.as_str(), @@ -1010,9 +997,10 @@ fn groups(input: &Schema, groups: &GroupKeys) -> Result, Error> { } Ok(groups.keys().to_vec()) } -fn expression(expr: &QueryExpr, input: &Schema) -> Result { +fn expression(expr: &WireScalarExpr, input: &Schema) -> Result { + let expr = local_scalar(expr)?; Ok(Expression::planner( - crate::expressions::CompiledExpression::compile(expr, input)?, + crate::expressions::CompiledExpression::compile(&expr, input)?, )) } @@ -1111,23 +1099,24 @@ fn preflight_depth(dag: &PostAsapDag) -> Result<(), Error> { /// Join predicates address the concatenated left/right schema. fn semi_join_keys( - expr: &QueryExpr, + expr: &ScalarExpr, left: usize, right: usize, keys: &mut Vec<(usize, usize)>, ) -> Result<(), Error> { match expr { - QueryExpr::BoolAnd(parts) => { + ScalarExpr::BoolAnd(parts) => { for part in parts { semi_join_keys(part, left, right, keys)?; } } - QueryExpr::Compare { + ScalarExpr::Compare { left: a, op: CompareOpKind::Eq, right: b, + .. } => { - let (QueryExpr::Column(a), QueryExpr::Column(b)) = (a.as_ref(), b.as_ref()) else { + let (ScalarExpr::Column(a), ScalarExpr::Column(b)) = (a.as_ref(), b.as_ref()) else { return Err(invalid("semi-join requires column equality keys")); }; let (a, b) = if a < b { (*a, *b) } else { (*b, *a) }; @@ -1144,9 +1133,9 @@ fn semi_join_keys( /// Resolve equality keys against the Planner join's concatenated input schema. /// Deployments may use these positions to bind their source columns. pub fn equijoin_keys( - pred: &planner_types::pre_asap::Predicate, - left: &planner_types::post_asap::SummarySchema, - right: &planner_types::post_asap::SummarySchema, + pred: &planner_types::ir::Predicate, + left: &planner_types::post_asap::Schema, + right: &planner_types::post_asap::Schema, ) -> Result, Error> { let mut keys = Vec::new(); semi_join_keys(&pred.0, left.fields.len(), right.fields.len(), &mut keys)?; @@ -1155,3 +1144,24 @@ pub fn equijoin_keys( } Ok(keys) } + +fn local_scalar(expr: &WireScalarExpr) -> Result { + let mut missing = false; + let result = logical::scalar(expr, &mut |_| { + missing = true; + std::rc::Rc::new(OperatorNode::with_schema( + LogicalOperator::NonASAP(NonASAPOp::Values { + rows: vec![], + schema: Default::default(), + }), + Default::default(), + )) + }); + if missing { + Err(invalid( + "scalar plan reads require explicit execution bindings", + )) + } else { + Ok(result) + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index 0674ff4c2..2df9058af 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -1,8 +1,9 @@ //! Compile immutable summary-input computation with explicit population and pane identity. use super::promql_rows::SERIES_IDENTITY_COLUMN as SERIES_IDENTITY; use super::*; +use planner_types::post_asap::FieldDataType as SummaryFamilyType; use planner_types::{ - post_asap::{ExecutionTiming, GroupingStrategy, SummarySchema}, + post_asap::{ExecutionTiming, GroupingStrategy, Schema as SummarySchema}, pre_asap::DataType, }; @@ -11,7 +12,7 @@ use planner_types::{ pub fn population_schema(family: SummaryFamilyType) -> Schema { Arc::new(SummarySchema { fields: vec![ - planner_types::post_asap::SummaryField { + planner_types::post_asap::Field { name: "$population".into(), dtype: SummaryFamilyType::Plain(DataType::Map { key: Box::new(DataType::Utf8), @@ -19,19 +20,24 @@ pub fn population_schema(family: SummaryFamilyType) -> Schema { value_nullable: false, }), nullable: false, + table: None, }, - planner_types::post_asap::SummaryField { + planner_types::post_asap::Field { name: "$window_end".into(), dtype: SummaryFamilyType::Plain(DataType::Timestamp), nullable: false, + table: None, }, - planner_types::post_asap::SummaryField { + planner_types::post_asap::Field { name: "value".into(), dtype: family, nullable: false, + table: None, }, ], time_index: Some(1), + unique_keys: vec![], + closed: false, }) } @@ -78,18 +84,13 @@ pub fn raw_sample_row( /// Input contract of a precompute boundary: raw sample rows for a raw time /// series scan, otherwise the stored population of its summary state. pub fn boundary_schema(node: &PostAsapDagNode) -> Result { - let Payload::Fallback { expression } = &node.payload else { - return source_schema(&node.output_schema); - }; - let scan = match expression { - planner_types::pre_asap::QueryExpr::TimeRange { child, .. } => child.as_ref(), - expression => expression, - }; if !matches!( - scan, - planner_types::pre_asap::QueryExpr::Scan { - source: planner_types::pre_asap::Source::TimeSeries { .. }, - .. + &node.payload, + Payload::Relational { + operator: NonASAPOpKind::Scan { + source: planner_types::pre_asap::Source::TimeSeries { .. }, + .. + } | NonASAPOpKind::TimeRange { .. } } ) { return source_schema(&node.output_schema); @@ -177,9 +178,10 @@ pub fn compile( ( edge.consumer.0, match edge.role { - planner_types::post_asap::EdgeRole::Left => 0, - planner_types::post_asap::EdgeRole::Input => 1, - planner_types::post_asap::EdgeRole::Right => 2, + planner_types::ir::export::EdgeRole::Left => 0, + planner_types::ir::export::EdgeRole::Input => 1, + planner_types::ir::export::EdgeRole::Right => 2, + planner_types::ir::export::EdgeRole::ScalarRef => 3, }, ) }); @@ -307,7 +309,15 @@ fn fragment( Ok(id) }; let root = match &node.payload { - Payload::Binary { operator } => { + Payload::Relational { + operator: + NonASAPOpKind::BinaryOp { + operator, + return_bool, + }, + } => { + let operator = + crate::expressions::binary::BinaryOperator::from_logical(operator, *return_bool); validate_value_output(node)?; if node.output_schema.time_index.is_none() || parents.iter().any(|p| p.output_schema.time_index.is_none()) @@ -330,9 +340,7 @@ fn fragment( )?, )? } - Payload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - } => { + Payload::FinalizeExactAccumulator => { let [input] = schemas else { return Err(invalid("finalize requires one state input")); }; @@ -351,10 +359,10 @@ fn fragment( )) } }; - let read = Operator::readout( + let read = Operator::evaluation( input.clone(), 2, - ReadoutQuery::Exact(ExactReadout { + SummaryEvaluation::Exact(ExactEvaluation { statistic, lookback_ms: None, }), @@ -393,7 +401,7 @@ fn fragment( return Err(invalid("summary update requires one input")); }; // Item identities resolve against the complete label set of raw - // samples; finalized readouts carry no such identity. + // samples; finalized evaluations carry no such identity. let raw = *input == raw_sample_schema(); // A unit-frequency summary (HLL) observes each raw sample value. let unit_frequency = raw @@ -499,10 +507,11 @@ fn fragment( )?; for (index, (expression, dtype)) in items.into_iter().enumerate() { let name = format!("$item{index}"); - fields.push(planner_types::post_asap::SummaryField { + fields.push(planner_types::post_asap::Field { name: name.clone(), dtype: SummaryFamilyType::Plain(dtype), nullable: false, + table: None, }); columns.push((name, expression)); } @@ -511,6 +520,8 @@ fn fragment( let project = Operator::project(input.clone(), columns)?.with_output_schema( Arc::new(SummarySchema { fields, + unique_keys: vec![], + closed: false, time_index: Some(1), }), )?; diff --git a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs index 3d92f0278..b5d88d9c8 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs @@ -1,4 +1,4 @@ -//! Compile a retained PromQL subtree (`Fallback`) from its typed expression. +//! Compile a retained PromQL sub-DAG (`Fallback`) from its typed expression. //! The deployment supplies the raw series of each selector; the Planner //! computes selection, range functions, subqueries, matching and aggregation. use super::*; @@ -19,14 +19,14 @@ pub(super) fn raw_series_owner(slot: NodeId) -> Option { } /// A selector expression and its raw-series row schema. -pub type Selector = (QueryExpr, Schema); +pub type Selector = (OperatorNode, Schema); /// The selectors a Fallback expression reads, left to right, and the row /// schema of the raw series the deployment supplies for each at /// [`raw_series_input`]. The rows must cover the selector's window at every /// evaluation instant `T`, or at its `@` time: `(T - offset - range, T - offset]`; /// under a subquery `[R:S] offset O` that is `(T - O - R - offset - range, T - O - offset]`. -pub fn raw_series(expression: &QueryExpr) -> Result, Error> { +pub fn raw_series(expression: &OperatorNode) -> Result, Error> { Ok(lower(expression)?.selectors) } @@ -43,16 +43,47 @@ pub(super) struct Lowering { pub steps: Vec<(Operator, Vec)>, } -pub(super) fn lower(expression: &QueryExpr) -> Result { +pub(super) fn lower(expression: &OperatorNode) -> Result { let mut lowering = Lowering::default(); lowering.value(expression)?; Ok(lowering) } -fn declared(expression: &QueryExpr) -> Result { - let schema = expression - .output_schema() - .map_err(|error| invalid(error.to_string()))?; +/// Compile a standalone scalar expression and expose its real series dependencies. +/// Input slots use root 0; no logical wrapper node is introduced. +pub fn compile_scalar_root( + expr: &ScalarExpr, +) -> Result<(CompiledPhysicalDag, Vec), Error> { + let mut lowering = Lowering::default(); + lowering.scalar_value(expr)?; + let mut inputs = BTreeMap::new(); + for (i, (_, schema)) in lowering.selectors.iter().enumerate() { + inputs.insert( + raw_series_input(0, i), + InputContract::bounded(schema.clone()), + ); + } + let last = lowering.steps.len() - 1; + let mut operators = BTreeMap::new(); + for (i, (operator, dependencies)) in lowering.steps.into_iter().enumerate() { + let id = if i == last { 0 } else { i as u64 + 1 }; + let dependencies = dependencies + .into_iter() + .map(|input| match input { + Input::Raw(i) => raw_series_input(0, i), + Input::Step(i) => i as u64 + 1, + }) + .collect(); + operators.insert(id, (dependencies, operator)); + } + Ok(( + CompiledPhysicalDag::from_operators(inputs, operators, vec![0])?, + lowering.selectors, + )) +} + +fn declared(expression: &OperatorNode) -> Result { + let schema = expression.schema.clone(); Ok(Arc::new(lift_plain(&schema))) } @@ -69,10 +100,10 @@ fn at(shift: &planner_types::pre_asap::TimeShift) -> Result, Error> } } -fn range_anchor(expression: &QueryExpr) -> Option { - match expression { - QueryExpr::TimeRange { child, .. } => range_anchor(child), - QueryExpr::TimeShift { shift, .. } => shift +fn range_anchor(expression: &OperatorNode) -> Option { + match expression.expect_non_asap() { + NonASAPOp::TimeRange { child, .. } => range_anchor(child), + NonASAPOp::TimeShift { shift, .. } => shift .at .filter(|at| matches!(at, AtModifier::Start | AtModifier::End)), _ => None, @@ -80,32 +111,22 @@ fn range_anchor(expression: &QueryExpr) -> Option { } /// `TimeRange { range, [TimeShift { offset, @ }], Scan }`: range, offset, `@`. -fn selector(expression: &QueryExpr) -> Result<(i64, i64, Option), Error> { - let QueryExpr::TimeRange { range, child } = expression else { +fn selector(expression: &OperatorNode) -> Result<(i64, i64, Option), Error> { + let NonASAPOp::TimeRange { range, child, .. } = expression.expect_non_asap() else { return Err(invalid("PromQL operand must be a series selector")); }; - let (offset, at, scan) = match child.as_ref() { - QueryExpr::TimeShift { shift, child } => (shift.offset_ms, at(shift)?, child.as_ref()), + let (offset, at, scan) = match child.expect_non_asap() { + NonASAPOp::TimeShift { shift, child } => { + (shift.offset_ms, at(shift)?, child.expect_non_asap()) + } scan => (0, None, scan), }; - if !matches!(scan, QueryExpr::Scan { .. }) { + if !matches!(scan, NonASAPOp::Scan { .. }) { return Err(invalid("PromQL selector must read one scan")); } Ok((millis(range)?, offset, at)) } -/// PromQL scalar-valued expressions have no labels to match. A binary -/// operator is scalar-valued when both operands are. -pub(super) fn scalar(expression: &QueryExpr) -> bool { - match expression { - QueryExpr::PromqlScalarBridge(_) - | QueryExpr::PromqlScalarFromVector(_) - | QueryExpr::EvalTimestamp => true, - QueryExpr::BinaryOp { lhs, rhs, .. } => scalar(lhs) && scalar(rhs), - _ => false, - } -} - impl Lowering { fn schema(&self, input: &Input) -> Schema { match input { @@ -124,12 +145,12 @@ impl Lowering { &mut self, operator: Operator, inputs: Vec, - logical: &QueryExpr, + logical: &OperatorNode, ) -> Result { Ok(self.add(operator.with_output_schema(declared(logical)?)?, inputs)) } - fn read(&mut self, selector: &QueryExpr) -> Result { + fn read(&mut self, selector: &OperatorNode) -> Result { let schema = declared(selector)?; if !schema .fields @@ -145,12 +166,12 @@ impl Lowering { } /// An instant vector, or a scalar for scalar-valued expressions. - fn value(&mut self, expression: &QueryExpr) -> Result { - match expression { - QueryExpr::Concat { children, .. } => { - if !children.iter().all(|branch| matches!(branch, - QueryExpr::PromqlRelabel { child, .. } if matches!(child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::HistogramQuantile { .. }])))) { + fn value(&mut self, expression: &OperatorNode) -> Result { + match expression.expect_non_asap() { + NonASAPOp::Concat { children, .. } => { + if !children.iter().all(|branch| matches!(branch.expect_non_asap(), + NonASAPOp::PromqlRelabel { child, .. } if matches!(child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::HistogramQuantile { .. }])))) { return Err(invalid("PromQL concatenation requires classic histogram quantile branches")); } let inputs = children @@ -171,15 +192,17 @@ impl Lowering { expression, ) } - QueryExpr::PromqlRelabel { dst, value, child } => { + NonASAPOp::PromqlRelabel { dst, value, child } => { let step = self.value(child)?; let input = self.schema(&step); - let (replacement, source_regex) = match value.as_ref() { - QueryExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8(value)) => { + let (replacement, source_regex) = match value { + ScalarExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8(value)) => { (value.clone(), None) } - QueryExpr::FunctionCall { name, args } if name == "label_replace" => { - let [QueryExpr::Column(source), QueryExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8(pattern)), QueryExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8( + ScalarExpr::FunctionCall { name, args } if name == "label_replace" => { + let [ScalarExpr::Column(source), ScalarExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8( + pattern, + )), ScalarExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8( replacement, ))] = args.as_slice() else { @@ -204,7 +227,7 @@ impl Lowering { )?; self.push(operator, vec![step], expression) } - QueryExpr::TimeRange { .. } => { + NonASAPOp::TimeRange { .. } => { let (range, offset, at) = selector(expression)?; let input = self.read(expression)?; let schema = self.schema(&input); @@ -215,7 +238,7 @@ impl Lowering { expression, ) } - QueryExpr::Aggregate { + NonASAPOp::Aggregate { reduction: planner_types::pre_asap::Reduction::PerEntity, measures, having: None, @@ -234,7 +257,7 @@ impl Lowering { let input = self.schema(&step); Ok(self.add(Operator::series_without_name(input)?, vec![step])) } - QueryExpr::Aggregate { + NonASAPOp::Aggregate { reduction: planner_types::pre_asap::Reduction::Reduce(keys), measures, having: None, @@ -248,7 +271,74 @@ impl Lowering { let input = self.value(child)?; self.aggregate(input, measure, keys, expression) } - QueryExpr::Sort { + NonASAPOp::Project { + cols, + child, + qualifier, + } => { + let value = planner_types::pre_asap::column_resolution::resolve_column_ref( + &ColumnRef::SampleValue, + &child.schema, + ) + .map_err(|e| invalid(e.to_string()))?; + let sample = cols + .iter() + .find(|col| { + col.alias.as_deref() == Some(child.schema.fields[value].name.as_str()) + }) + .ok_or_else(|| invalid("missing sample projection"))?; + let keep_name = matches!(sample.expr, ScalarExpr::Negative { .. }); + let fields: Vec<_> = child + .schema + .fields + .iter() + .enumerate() + .filter(|(_, field)| keep_name || field.name != "__name__") + .collect(); + if qualifier.is_some() || cols.len() != fields.len() { + return Err(invalid("unsupported temporal projection shape")); + } + let mut computed = None; + for (col, (index, field)) in cols.iter().zip(fields) { + if col.alias.as_deref() != Some(field.name.as_str()) { + return Err(invalid("unsupported temporal projection alias")); + } + if index == value { + computed = Some(col); + } else { + let expected = if !keep_name + && field.name == planner_types::pre_asap::schema::PROMQL_SERIES_IDENTITY + { + ScalarExpr::FunctionCall { + name: "promql_drop_metric_name".into(), + args: vec![ScalarExpr::Column(index)], + } + } else { + ScalarExpr::Column(index) + }; + if col.expr != expected { + return Err(invalid("unsupported temporal projection expression")); + } + } + } + let computed = computed.ok_or_else(|| invalid("no computed sample"))?; + if matches!( + computed.expr, + ScalarExpr::Negative { .. } | ScalarExpr::FunctionCall { .. } + ) { + return self.pointwise_projection(cols, child, value, expression, keep_name); + } + self.sample_scalar_operation(&computed.expr, child, value, expression) + } + NonASAPOp::Filter { pred, child } => { + let value = planner_types::pre_asap::column_resolution::resolve_column_ref( + &ColumnRef::SampleValue, + &child.schema, + ) + .map_err(|e| invalid(e.to_string()))?; + self.sample_scalar_operation(&pred.0, child, value, expression) + } + NonASAPOp::Sort { keys, partition_by, child, @@ -258,7 +348,7 @@ impl Lowering { let keys = keys .iter() .map(|key| match key.expr { - QueryExpr::Column(column) => Ok(SortKey { + ScalarExpr::Column(column) => Ok(SortKey { column, descending: !key.ascending, nulls_first: key.nulls_first, @@ -269,70 +359,223 @@ impl Lowering { let groups = groups(&input, partition_by)?; self.push(Operator::sort(input, keys, groups)?, vec![step], expression) } - QueryExpr::Limit { n, offset, child } => { + NonASAPOp::Limit { + n, offset, child, .. + } => { let step = self.value(child)?; let input = self.schema(&step); // `topk by (...)` partitions through the Sort it limits. - let groups = match child.as_ref() { - QueryExpr::Sort { partition_by, .. } => groups(&input, partition_by)?, + let groups = match child.expect_non_asap() { + NonASAPOp::Sort { partition_by, .. } => groups(&input, partition_by)?, _ => vec![], }; self.push( - Operator::limit(input, *n as u64, *offset as u64, groups)?, + Operator::limit( + input, + n.unwrap_or(usize::MAX) as u64, + *offset as u64, + groups, + )?, vec![step], expression, ) } - QueryExpr::BinaryOp { - op, + NonASAPOp::BinaryOp { + operator, lhs, rhs, - vector_match, + return_bool, } => { let sides = vec![self.value(lhs)?, self.value(rhs)?]; - let operator = planner_types::post_asap::BinaryOperator { - kind: op.clone(), - vector_match: vector_match.clone(), - checked_relative_division: false, - checked_finite_division: false, - }; + let operator = crate::expressions::binary::BinaryOperator::from_logical( + operator, + *return_bool, + ); let binary = Operator::series_binary( self.schema(&sides[0]), self.schema(&sides[1]), operator, - [scalar(lhs), scalar(rhs)], + [false, false], )?; self.push(binary, sides, expression) } - QueryExpr::PromqlScalarFromVector(child) => { - let step = self.value(child)?; - let input = self.schema(&step); - let value = named_column(&input, &ColumnRef::SampleValue)?; - self.push( - Operator::vector_to_scalar(input, value)?, - vec![step], - expression, - ) - } - QueryExpr::PromqlVectorFromScalar(child) => { - let step = self.value(child)?; + NonASAPOp::PromqlVectorFromScalar(expr) => { + let step = self.scalar_value(expr)?; let input = self.schema(&step); Ok(self.add( Operator::scope_timestamp(input, declared(expression)?)?, vec![step], )) } - QueryExpr::EvalTimestamp => self.push(Operator::evaluation_time(), vec![], expression), - QueryExpr::PromqlScalarBridge(_) => { - let value = row_values::scalar_literal(expression) - .ok_or_else(|| invalid("PromQL scalar must be a literal"))?; - self.push( - Operator::scalar(crate::values::Value::Float64(value), DataType::Float64)?, + _ => Err(invalid("PromQL expression has no native fallback lowering")), + } + } + + fn pointwise_projection( + &mut self, + cols: &[planner_types::ir::ProjectItem], + child: &OperatorNode, + value: usize, + output: &OperatorNode, + keep_name: bool, + ) -> Result { + let mut input = self.value(child)?; + let mut projected = cols.to_vec(); + for col in &mut projected { + if col.alias.as_deref() != Some(child.schema.fields[value].name.as_str()) { + continue; + } + if let ScalarExpr::FunctionCall { name, args } = &mut col.expr { + if planner_types::pre_asap::scalar_signature::promql_function_arity(name).is_none() + || args.first() != Some(&ScalarExpr::Column(value)) + { + return Err(invalid("unsupported pointwise function")); + } + for arg in args.iter_mut().skip(1) { + let scalar = self.scalar_value(arg)?; + let left = self.schema(&input); + let right = self.schema(&scalar); + let index = left.fields.len(); + let mut schema = (*left).clone(); + schema.fields.extend(right.fields.clone()); + let join = Operator::relational_join( + left, + right, + planner_types::pre_asap::JoinKind::Inner, + &planner_types::ir::Predicate(ScalarExpr::Literal( + planner_types::pre_asap::ScalarValue::Boolean(true), + )), + Arc::new(schema), + )?; + input = self.add(join, vec![input, scalar]); + *arg = ScalarExpr::Column(index); + } + if name == "promql_clamp" { + let predicate = ScalarExpr::Not(Box::new(ScalarExpr::Compare { + left: Box::new(args[1].clone()), + right: Box::new(args[2].clone()), + op: planner_types::pre_asap::CompareOpKind::Gt, + semantics: planner_types::ir::ExprSemantics::Promql, + })); + let schema = self.schema(&input); + let predicate = + crate::expressions::CompiledExpression::compile(&predicate, &schema)?; + input = self.add( + Operator::filter( + schema, + crate::expressions::Expression::planner(predicate), + )?, + vec![input], + ); + } + } + } + let schema = self.schema(&input); + let columns = projected + .iter() + .map(|col| { + Ok(( + col.alias.clone().unwrap(), + crate::expressions::Expression::planner( + crate::expressions::CompiledExpression::compile(&col.expr, &schema)?, + ), + )) + }) + .collect::, Error>>()?; + let project = Operator::project(schema, columns)?; + let result = self.push(project, vec![input], output)?; + if keep_name { + Ok(result) + } else { + self.push( + Operator::series_without_name(self.schema(&result))?, + vec![result], + output, + ) + } + } + + fn sample_scalar_operation( + &mut self, + expr: &ScalarExpr, + child: &OperatorNode, + value: usize, + output: &OperatorNode, + ) -> Result { + let (left, right, kind) = scalar_binary(expr)?; + let (scalar, scalar_left) = match (left, right) { + (ScalarExpr::Column(i), scalar) if *i == value => (scalar, false), + (scalar, ScalarExpr::Column(i)) if *i == value => (scalar, true), + _ => { + return Err(invalid( + "sample projection requires one vector sample and one scalar", + )) + } + }; + let vector = self.value(child)?; + let scalar = self.scalar_value(scalar)?; + let sides = if scalar_left { + vec![scalar, vector] + } else { + vec![vector, scalar] + }; + let operator = Operator::series_binary( + self.schema(&sides[0]), + self.schema(&sides[1]), + kernel(kind), + [scalar_left, !scalar_left], + )?; + self.push(operator, sides, output) + } + + fn scalar_value(&mut self, expr: &ScalarExpr) -> Result { + match expr { + ScalarExpr::Literal(planner_types::pre_asap::ScalarValue::Float64(value)) => Ok(self + .add( + Operator::scalar(crate::values::Value::Float64(*value), DataType::Float64)?, vec![], - expression, - ) + )), + ScalarExpr::EvalTimestamp => Ok(self.add(Operator::evaluation_time(), vec![])), + ScalarExpr::PromqlScalarFromVector(child) => { + let step = self.value(child)?; + let input = self.schema(&step); + let values: Vec<_> = input + .fields + .iter() + .enumerate() + .filter(|(_, f)| f.dtype == FieldDataType::Plain(DataType::Float64)) + .map(|(i, _)| i) + .collect(); + let [value] = values.as_slice() else { + return Err(invalid("scalar() requires one float sample column")); + }; + let value = *value; + Ok(self.add(Operator::vector_to_scalar(input, value)?, vec![step])) + } + ScalarExpr::Negative { expr, .. } => { + let value = self.scalar_value(expr)?; + let minus = self.scalar_value(&ScalarExpr::literal_f64(-1.0))?; + let op = Operator::series_binary( + self.schema(&value), + self.schema(&minus), + kernel(crate::expressions::binary::BinaryOpKind::Arithmetic( + planner_types::pre_asap::ArithmeticOpKind::Mul, + )), + [true, true], + )?; + Ok(self.add(op, vec![value, minus])) + } + _ => { + let (left, right, kind) = scalar_binary(expr)?; + let sides = vec![self.scalar_value(left)?, self.scalar_value(right)?]; + let op = Operator::series_binary( + self.schema(&sides[0]), + self.schema(&sides[1]), + kernel(kind), + [true, true], + )?; + Ok(self.add(op, sides)) } - _ => Err(invalid("PromQL expression has no native fallback lowering")), } } @@ -340,19 +583,19 @@ impl Lowering { fn range_function( &mut self, function: &AggIntent, - matrix: &QueryExpr, - logical: &QueryExpr, + matrix: &OperatorNode, + logical: &OperatorNode, ) -> Result { let function = unbound(function)?; - let (subquery, offset, at_ms) = match matrix { - QueryExpr::TimeShift { shift, child } => (child.as_ref(), shift.offset_ms, at(shift)?), - other => (other, 0, None), + let (subquery, offset, at_ms) = match matrix.expect_non_asap() { + NonASAPOp::TimeShift { shift, child } => (child.as_ref(), shift.offset_ms, at(shift)?), + _ => (matrix, 0, None), }; - let QueryExpr::PromqlSubquery { + let NonASAPOp::PromqlSubquery { range: outer, resolution, child, - } = subquery + } = subquery.expect_non_asap() else { let (range, offset, at) = selector(matrix)?; let input = self.read(matrix)?; @@ -374,8 +617,8 @@ impl Lowering { at_ms, }; // Each step evaluates a per-series selection or range function. - let (inner, selected) = match child.as_ref() { - QueryExpr::Aggregate { + let (inner, selected) = match child.expect_non_asap() { + NonASAPOp::Aggregate { reduction: planner_types::pre_asap::Reduction::PerEntity, measures, having: None, @@ -385,7 +628,7 @@ impl Lowering { [inner] => (Some(unbound(inner)?), selected.as_ref()), _ => return Err(invalid("range function requires one measure")), }, - selected => (None, selected), + _ => (None, child.as_ref()), }; let (range, inner_offset, inner_at) = selector(selected)?; let raw = self.read(selected)?; @@ -425,7 +668,7 @@ impl Lowering { mut step: Input, measure: &AggIntent, keys: &GroupKeys, - logical: &QueryExpr, + logical: &OperatorNode, ) -> Result { let mut input = self.schema(&step); if let AggIntent::HistogramQuantile { q, le } = measure { @@ -439,7 +682,7 @@ impl Lowering { .fields .iter() .enumerate() - .filter(|(_, f)| f.dtype == SummaryFamilyType::Plain(DataType::Float64)) + .filter(|(_, f)| f.dtype == FieldDataType::Plain(DataType::Float64)) .map(|(i, _)| i) .collect::>(); let [value] = value.as_slice() else { @@ -447,10 +690,10 @@ impl Lowering { }; let value = *value; let reduction = match measure { - AggIntent::Sum { col: None } => Reduction::Sum(value), - AggIntent::Avg { col: None } => Reduction::Avg(value), - AggIntent::Min { col: None } => Reduction::Min(value), - AggIntent::Max { col: None } => Reduction::Max(value), + AggIntent::Sum { .. } => Reduction::Sum(value), + AggIntent::Avg { .. } => Reduction::Avg(value), + AggIntent::Min { .. } => Reduction::Min(value), + AggIntent::Max { .. } => Reduction::Max(value), AggIntent::Count { .. } => Reduction::Count, _ => return Err(invalid("vector aggregate has no native lowering")), }; @@ -527,10 +770,10 @@ fn unbound(intent: &AggIntent) -> Result, Error> { AggIntent::Count { accuracy } => AggIntent::Count { accuracy: accuracy.clone(), }, - AggIntent::Sum { col: None } => AggIntent::Sum { col: None }, - AggIntent::Avg { col: None } => AggIntent::Avg { col: None }, - AggIntent::Min { col: None } => AggIntent::Min { col: None }, - AggIntent::Max { col: None } => AggIntent::Max { col: None }, + AggIntent::Sum { .. } => AggIntent::Sum { col: None }, + AggIntent::Avg { .. } => AggIntent::Avg { col: None }, + AggIntent::Min { .. } => AggIntent::Min { col: None }, + AggIntent::Max { .. } => AggIntent::Max { col: None }, AggIntent::IRate => AggIntent::IRate, AggIntent::IDelta => AggIntent::IDelta, AggIntent::Changes => AggIntent::Changes, @@ -548,3 +791,65 @@ fn unbound(intent: &AggIntent) -> Result, Error> { _ => return Err(invalid("unsupported PromQL range function")), }) } + +fn kernel( + kind: crate::expressions::binary::BinaryOpKind, +) -> crate::expressions::binary::BinaryOperator { + crate::expressions::binary::BinaryOperator { + kind, + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + } +} + +fn scalar_binary( + expr: &ScalarExpr, +) -> Result< + ( + &ScalarExpr, + &ScalarExpr, + crate::expressions::binary::BinaryOpKind, + ), + Error, +> { + use crate::expressions::binary::BinaryOpKind as K; + match expr { + ScalarExpr::Arithmetic { + left, + right, + op, + semantics: planner_types::ir::ExprSemantics::Promql, + } => Ok((left, right, K::Arithmetic(op.clone()))), + ScalarExpr::Compare { + left, + right, + op, + semantics: planner_types::ir::ExprSemantics::Promql, + } => Ok((left, right, K::Compare(op.clone()))), + ScalarExpr::Case { + operand: None, + branches, + else_expr, + } if matches!(else_expr.as_deref(), Some(ScalarExpr::Literal(planner_types::pre_asap::ScalarValue::Float64(v))) if *v == 0.0) => + { + let [( + ScalarExpr::Compare { + left, + right, + op, + semantics: planner_types::ir::ExprSemantics::Promql, + }, + ScalarExpr::Literal(planner_types::pre_asap::ScalarValue::Float64(v)), + )] = branches.as_slice() + else { + return Err(invalid("unsupported scalar case")); + }; + if *v != 1.0 { + return Err(invalid("unsupported scalar case result")); + } + Ok((left, right, K::CompareBool(op.clone()))) + } + _ => Err(invalid("scalar expression has no native temporal lowering")), + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index 802777c5d..4128ee146 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -1,6 +1,8 @@ //! A bounded PromQL source row carries the entire label set, not just labels //! mentioned by the query. The source adapter owns this lossless encoding. use super::*; +use planner_types::ir::export::{compile_post_asap_dag, compile_post_asap_dag_with_node_ids}; +use planner_types::post_asap::FieldDataType as SummaryFamilyType; use planner_types::pre_asap::DataType; use std::rc::Rc; @@ -24,7 +26,7 @@ pub fn decode_series_identity(encoded: &str) -> Result, /// Resolve the row representation before candidate search; see /// [`planner_types::pre_asap::schema::with_promql_series_identity`]. -pub fn with_series_identity(root: &QueryExpr) -> Result { +pub fn with_series_identity(root: &Rc) -> Result, Error> { planner_types::pre_asap::schema::with_promql_series_identity(root).map_err(invalid) } @@ -78,17 +80,23 @@ pub fn series_row( /// Compile the selected TopK computation above an existing maintained-population /// source. The boundary supplies the complete eligible vector, not a truncated /// TopK result; ranking remains a native physical operator. -pub fn compile_current_series_readout( - selected: &Rc, +pub fn compile_current_series_evaluation( + selected: &Rc, ) -> Result { use planner_types::post_asap::{ - compile_post_asap_dag, maintained_population::PopulationReadout, SummaryField, + maintained_population::PopulationStatistic, Field as SummaryField, }; - let mut dag = compile_post_asap_dag(selected).map_err(|error| invalid(error.to_string()))?; + let selected = planner_types::ir::apply_lifecycle_timings( + selected, + &planner_types::ir::LifecycleAssignment::default_maintained(), + &mut planner_types::ir::TimingMemo::new(), + ) + .map_err(|e| invalid(e.to_string()))?; + let mut dag = compile_post_asap_dag(&selected).map_err(|error| invalid(error.to_string()))?; // Typed snapshot candidates already carry full identity throughout the DAG. - // Cut at the population output, preserving all selected heap/readout nodes. + // Cut at the population output, preserving all selected heap/evaluation nodes. let populations = dag.nodes.iter().filter(|node| matches!(&node.payload, - Payload::Value { operation: ValueOperation::MaintainPopulation { population } } + Payload::MaintainPopulation { population } if matches!(population.input, planner_types::post_asap::maintained_population::PopulationInput::CurrentSeries(_)) )).collect::>(); if let [population] = populations.as_slice() { @@ -108,41 +116,26 @@ pub fn compile_current_series_readout( ); } } - if dag.nodes.len() != 3 - || !dag.nodes.iter().any(|node| { - node.id == dag.root - && matches!( - node.payload, - Payload::Value { - operation: ValueOperation::ReadPopulation { - readout: PopulationReadout::TopK { .. } - } - } - ) - }) - { - return Err(invalid( - "expected one selected current-series TopK computation", - )); - } let mut frontier = None; for node in &mut dag.nodes { match &mut node.payload { - Payload::Fallback { expression } => { - *expression = with_series_identity(expression)?; + Payload::Relational { operator } => { + if let NonASAPOpKind::Scan { schema, .. } = operator { + schema.fields.push(SummaryField::new( + SERIES_IDENTITY_COLUMN, + SummaryFamilyType::Plain(DataType::Utf8), + false, + )); + schema.closed = true; + } } - Payload::Value { - operation: ValueOperation::MaintainPopulation { .. }, - } => { + Payload::MaintainPopulation { .. } => { frontier = Some(u64::from(node.id.0)); } - Payload::Value { - operation: - ValueOperation::ReadPopulation { - readout: PopulationReadout::TopK { .. }, - }, + Payload::EvaluatePopulation { + evaluation: PopulationStatistic::TopK { .. }, } => {} - _ => return Err(invalid("unsupported current-series readout dependency")), + _ => return Err(invalid("unsupported current-series evaluation dependency")), } if node .output_schema @@ -158,6 +151,7 @@ pub fn compile_current_series_readout( name: SERIES_IDENTITY_COLUMN.into(), dtype: SummaryFamilyType::Plain(DataType::Utf8), nullable: false, + table: None, }); } for edge in &mut dag.edges { @@ -186,43 +180,31 @@ pub fn compile_current_series_readout( } /// Compile selected ranking or aggregation above an exact per-series Rate -/// readout. Deployments bind complete window readouts at this boundary; +/// evaluation. Deployments bind complete window evaluations at this boundary; /// the heap is rebuilt independently for each evaluation. This does not move /// that frontier to ingestion time or authorize combining finalized rates. pub fn compile_rate_ranking( - selected: &Rc, -) -> Result< - ( - Rc, - CompiledPhysicalDag, - ), - Error, -> { - use planner_types::post_asap::{ - compile_post_asap_dag_with_node_ids, ExactKind, SummaryExpr, SummaryNode, - }; - fn frontier(node: &Rc) -> Option> { - match &node.expr { - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing: planner_types::post_asap::ExecutionTiming::QueryTime, - } if matches!(&child.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), - reduction: planner_types::pre_asap::Reduction::PerEntity, - child: raw, .. - } if matches!(&raw.expr, SummaryExpr::KeepPreAsap(expr) if matches!(expr.as_ref(), QueryExpr::TimeRange { .. }))) => - { - Some(Rc::clone(node)) - } - SummaryExpr::ValueOperation { child, .. } | SummaryExpr::SummaryAgg { child, .. } => { - frontier(child) - } - SummaryExpr::SummaryEstimate { summary_input, .. } => frontier(summary_input), - _ => None, + selected: &Rc, +) -> Result<(Rc, CompiledPhysicalDag), Error> { + use planner_types::post_asap::ExactKind; + fn frontier(node: &Rc) -> Option> { + if matches!(&node.operator, LogicalOperator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) + if matches!(&child.operator, LogicalOperator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), + reduction: planner_types::pre_asap::Reduction::PerEntity, child: raw, .. + }) if matches!(raw.non_asap(), Some(NonASAPOp::TimeRange { .. })))) + { + return Some(Rc::clone(node)); } + node.children().into_iter().find_map(frontier) } - let source = frontier(selected) + let selected = planner_types::ir::apply_lifecycle_timings( + selected, + &planner_types::ir::LifecycleAssignment::default_maintained(), + &mut planner_types::ir::TimingMemo::new(), + ) + .map_err(|e| invalid(e.to_string()))?; + let source = frontier(&selected) .ok_or_else(|| invalid("ranking requires one exact per-series Rate frontier"))?; if !source .schema @@ -232,7 +214,7 @@ pub fn compile_rate_ranking( { return Err(invalid("Rate ranking requires complete series identity")); } - let compiled = compile_post_asap_dag_with_node_ids(selected) + let compiled = compile_post_asap_dag_with_node_ids(&selected) .map_err(|error| invalid(error.to_string()))?; let id = u64::from( compiled @@ -250,10 +232,10 @@ pub fn compile_rate_ranking( } /// Compile a lifecycle-timed DAG whose heap or grouped Sum over per-series -/// Rate readouts runs at ingestion time: fresh aggregate state per closed +/// Rate evaluations runs at ingestion time: fresh aggregate state per closed /// window. The input is the complete collection of per-series counter states. pub fn compile_fixed_window_rate_aggregation( - dag: &planner_types::post_asap::PostAsapDag, + dag: &planner_types::ir::export::PostAsapDag, ) -> Result { use planner_types::post_asap::{ExactKind, ExecutionTiming, SketchAlgorithm}; let sources = dag diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/asap-physical-operators/src/physical_planner/promql_values.rs index 98032505b..fee455998 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_values.rs @@ -1,5 +1,6 @@ //! Physical scalar/vector contracts preserve complete label sets across native computation. use super::*; +use planner_types::post_asap::FieldDataType as SummaryFamilyType; pub fn scalar_schema() -> Schema { crate::operators::vector_binary::value_schema(true) @@ -63,7 +64,7 @@ pub fn compile_histogram_quantile() -> Result { /// Compile before deployment chooses readers. Input slots 0 and 1 retain operand order. pub fn compile_binary( - operator: &planner_types::post_asap::BinaryOperator, + operator: &crate::expressions::binary::BinaryOperator, return_bool: bool, left_scalar: bool, right_scalar: bool, @@ -219,8 +220,8 @@ pub fn exact_state_schema(family: SummaryFamilyType) -> Result { Ok(Arc::new(schema)) } -/// Retain exact readout semantics before any deployment state is opened. -pub fn compile_exact_readout( +/// Retain exact evaluation semantics before any deployment state is opened. +pub fn compile_exact_evaluation( family: SummaryFamilyType, lookback_ms: u64, preserve_metric_name: bool, @@ -234,16 +235,18 @@ pub fn compile_exact_readout( ExactKind::Max => crate::Statistic::Max, ExactKind::Rate => crate::Statistic::Rate, ExactKind::Increase => crate::Statistic::Increase, - ExactKind::IRate => return Err(invalid("instant-rate state readout is not supported")), + ExactKind::IRate => { + return Err(invalid("instant-rate state evaluation is not supported")) + } }, - _ => return Err(invalid("exact readout requires an exact family")), + _ => return Err(invalid("exact evaluation requires an exact family")), }; let input = exact_state_schema(family)?; let merge = Operator::summary_merge(input.clone(), 1, vec![0])?; - let mut readout = Operator::readout( + let mut evaluation = Operator::evaluation( merge.schema(), 1, - ReadoutQuery::Exact(ExactReadout { + SummaryEvaluation::Exact(ExactEvaluation { statistic, lookback_ms: None, }), @@ -252,12 +255,12 @@ pub fn compile_exact_readout( statistic, crate::Statistic::Rate | crate::Statistic::Increase ) { - readout = readout.with_counter_lookback( + evaluation = evaluation.with_counter_lookback( i64::try_from(lookback_ms).map_err(|_| invalid("counter lookback exceeds Int64"))?, )?; } let project = Operator::project( - readout.schema(), + evaluation.schema(), vec![ ( "labels".into(), @@ -274,5 +277,5 @@ pub fn compile_exact_readout( ("value".into(), Expression::ExactFloat64(1)), ], )?; - unary(vec![merge, readout, project], input) + unary(vec![merge, evaluation, project], input) } diff --git a/crates/asap-physical-operators/src/physical_planner/row_values.rs b/crates/asap-physical-operators/src/physical_planner/row_values.rs index 8763437bb..874873a0b 100644 --- a/crates/asap-physical-operators/src/physical_planner/row_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/row_values.rs @@ -1,39 +1,30 @@ //! Query-time PromQL value computation over logical row schemas. use super::*; -use planner_types::post_asap::maintained_population::PopulationReadout; -use planner_types::pre_asap::{DataType, ScalarValue}; +use planner_types::post_asap::maintained_population::PopulationStatistic; +use planner_types::pre_asap::DataType; -/// A PromQL number literal has no row schema; its consumer folds it in. -pub(super) fn scalar_literal(expression: &QueryExpr) -> Option { - match expression { - QueryExpr::PromqlScalarBridge(child) => scalar_literal(child), - QueryExpr::Literal(ScalarValue::Float64(value)) => Some(*value), - _ => None, - } -} - -/// Aggregate readouts of a maintained current-series population, as a chain. +/// Aggregate evaluations of a maintained current-series population, as a chain. pub(super) fn population_aggregate( input: &Schema, grouping: &[String], - readout: &PopulationReadout, + evaluation: &PopulationStatistic, ) -> Result, Error> { let groups = grouping .iter() .map(|name| named_column(input, &ColumnRef::Named(name.clone()))) .collect::, _>>()?; let value = named_column(input, &ColumnRef::SampleValue)?; - let reduction = match readout { - PopulationReadout::Sum => Reduction::Sum(value), - PopulationReadout::Count => Reduction::Count, - PopulationReadout::Average => Reduction::Avg(value), - PopulationReadout::Quantile { q } => Reduction::Quantile { + let reduction = match evaluation { + PopulationStatistic::Sum => Reduction::Sum(value), + PopulationStatistic::Count => Reduction::Count, + PopulationStatistic::Average => Reduction::Avg(value), + PopulationStatistic::Quantile { q } => Reduction::Quantile { column: value, q: *q, }, - PopulationReadout::TopK { .. } => { + PopulationStatistic::TopK { .. } => { return Err(invalid( - "TopK population readout ranks; it does not aggregate", + "TopK population evaluation ranks; it does not aggregate", )) } }; diff --git a/crates/asap-physical-operators/src/runtime/batch_execution.rs b/crates/asap-physical-operators/src/runtime/batch_execution.rs index f63004401..57002d1b6 100644 --- a/crates/asap-physical-operators/src/runtime/batch_execution.rs +++ b/crates/asap-physical-operators/src/runtime/batch_execution.rs @@ -88,7 +88,9 @@ mod tests { values::Value, }; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{ + Field as SummaryField, FieldDataType as SummaryFamilyType, Schema as SummarySchema, + }, pre_asap::DataType, }; use std::sync::Arc; @@ -101,8 +103,11 @@ mod tests { name: "value".into(), dtype: SummaryFamilyType::Plain(DataType::Float64), nullable: false, + table: None, }], time_index: None, + unique_keys: vec![], + closed: false, }); for scope in [ Scope::Query { @@ -142,6 +147,8 @@ mod tests { let schema = Arc::new(SummarySchema { fields: vec![], time_index: None, + unique_keys: vec![], + closed: false, }); let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); let source = Operator::source(schema, vec![batch; 65]).unwrap(); @@ -162,6 +169,8 @@ mod tests { let schema = Arc::new(SummarySchema { fields: vec![], time_index: None, + unique_keys: vec![], + closed: false, }); let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); let bytes = batch.bytes(); @@ -191,6 +200,8 @@ mod tests { let schema = Arc::new(SummarySchema { fields: vec![], time_index: None, + unique_keys: vec![], + closed: false, }); let batch = Batch::try_new(schema, vec![vec![]]).unwrap(); let context = RunContext::new( diff --git a/crates/asap-physical-operators/src/sources/mod.rs b/crates/asap-physical-operators/src/sources/mod.rs index 4da778b0c..d243ede71 100644 --- a/crates/asap-physical-operators/src/sources/mod.rs +++ b/crates/asap-physical-operators/src/sources/mod.rs @@ -7,9 +7,10 @@ use crate::{ Error, }; use futures::{stream, StreamExt}; +use planner_types::ir::{NonASAPOp, OperatorNode}; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, - pre_asap::{DataType, QueryExpr, Source}, + post_asap::FieldDataType as SummaryFamilyType, + pre_asap::{DataType, Source}, }; use std::sync::Arc; @@ -40,29 +41,18 @@ impl DataSources { self.sources.push((identity, source)); Ok(()) } - pub fn bind(&self, expression: &QueryExpr) -> Result { - let QueryExpr::Scan { + pub fn bind(&self, expression: &OperatorNode) -> Result { + let Some(NonASAPOp::Scan { source, predicates, schema, - } = expression + }) = expression.non_asap() else { return Err(Error::Invalid( "raw Scan requires a Planner Scan leaf".into(), )); }; - let output = Arc::new(SummarySchema { - fields: schema - .columns - .iter() - .map(|column| SummaryField { - name: column.name.clone(), - dtype: SummaryFamilyType::Plain(column.dtype.clone()), - nullable: column.nullable, - }) - .collect(), - time_index: schema.time_index, - }); + let output = Arc::new(schema.clone()); crate::values::validate_schema(&output)?; let reader = self .sources diff --git a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs index 8f1db45df..af0815c25 100644 --- a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs +++ b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs @@ -1,7 +1,7 @@ //! Count-Min Sketch frequency summary over `asap_sketchlib::CountMinSketch`. use crate::{AggregateCore, KernelError, KeyByLabelValues}; use asap_sketchlib::CountMinSketch; -use planner_types::post_asap::SketchQuery; +use planner_types::post_asap::SketchStatistic; #[derive(Debug, Clone)] pub struct CountMinSketchAccumulator { @@ -45,9 +45,9 @@ impl AggregateCore for CountMinSketchAccumulator { /// A bare point count reads the total update weight: every Count-Min row /// receives each update exactly once, so one row's mass survives collisions. - fn estimate(&self, query: &SketchQuery) -> Result { + fn estimate(&self, query: &SketchStatistic) -> Result { match query { - SketchQuery::PointCount { value: None, .. } => Ok(row_mass(&self.inner.sketch())), + SketchStatistic::PointCount { value: None, .. } => Ok(row_mass(&self.inner.sketch())), _ => Err(format!("{query:?} is not supported by Count-Min Sketch").into()), } } @@ -86,7 +86,7 @@ mod tests { // The bare count keeps colliding items' weight, adds across merges, and is 0 when empty. #[test] fn bare_count_reads_total_weight() { - let bare_count = SketchQuery::PointCount { + let bare_count = SketchStatistic::PointCount { key: planner_types::pre_asap::ColumnRef::SampleValue, value: None, }; @@ -98,7 +98,7 @@ mod tests { assert_eq!(merged.estimate(&bare_count).unwrap(), 20.0); let empty = CountMinSketchAccumulator::new(2, 1); assert_eq!(empty.estimate(&bare_count).unwrap(), 0.0); - assert!(state.estimate(&SketchQuery::Cardinality).is_err()); + assert!(state.estimate(&SketchStatistic::Cardinality).is_err()); } // Merge rejects a different summary family. diff --git a/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs b/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs index 3417bb8cc..e4ecb56a0 100644 --- a/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs +++ b/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs @@ -1,7 +1,7 @@ //! KLL quantile summary over `asap_sketchlib::KllSketch`. use crate::{AggregateCore, KernelError}; use asap_sketchlib::KllSketch; -use planner_types::post_asap::SketchQuery; +use planner_types::post_asap::SketchStatistic; #[derive(Clone)] pub struct DatasketchesKLLAccumulator { @@ -60,10 +60,10 @@ impl AggregateCore for DatasketchesKLLAccumulator { })) } - fn estimate(&self, query: &SketchQuery) -> Result { + fn estimate(&self, query: &SketchStatistic) -> Result { match query { - SketchQuery::Quantile { q } if (0.0..=1.0).contains(q) => Ok(self.get_quantile(*q)), - SketchQuery::Quantile { .. } => Err("quantile must be in [0, 1]".into()), + SketchStatistic::Quantile { q } if (0.0..=1.0).contains(q) => Ok(self.get_quantile(*q)), + SketchStatistic::Quantile { .. } => Err("quantile must be in [0, 1]".into()), other => Err(format!("KLL does not answer {other:?}").into()), } } @@ -95,7 +95,7 @@ mod tests { all.update(f64::from(v)); } let merged = a.merge_with(&b).unwrap(); - let q = SketchQuery::Quantile { q: 0.5 }; + let q = SketchStatistic::Quantile { q: 0.5 }; assert_eq!(merged.estimate(&q).unwrap(), all.estimate(&q).unwrap()); } @@ -103,7 +103,7 @@ mod tests { #[test] fn rejects_unsupported_or_out_of_range_queries() { let kll = DatasketchesKLLAccumulator::new(200); - assert!(kll.estimate(&SketchQuery::Quantile { q: 1.5 }).is_err()); - assert!(kll.estimate(&SketchQuery::Cardinality).is_err()); + assert!(kll.estimate(&SketchStatistic::Quantile { q: 1.5 }).is_err()); + assert!(kll.estimate(&SketchStatistic::Cardinality).is_err()); } } diff --git a/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs index dcda7ad47..e9cdf9d9c 100644 --- a/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs +++ b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs @@ -1,7 +1,7 @@ //! DDSketch quantile summary over `asap_sketchlib::DdSketch`. use crate::{AggregateCore, KernelError}; use asap_sketchlib::DdSketch; -use planner_types::post_asap::SketchQuery; +use planner_types::post_asap::SketchStatistic; #[derive(Debug, Clone)] pub struct DDSketchAccumulator { @@ -39,14 +39,14 @@ impl AggregateCore for DDSketchAccumulator { } /// Quantiles, and the total sample count as a bare `PointCount`. - fn estimate(&self, query: &SketchQuery) -> Result { + fn estimate(&self, query: &SketchStatistic) -> Result { match query { - SketchQuery::Quantile { q } if (0.0..=1.0).contains(q) => self + SketchStatistic::Quantile { q } if (0.0..=1.0).contains(q) => self .inner .quantile(*q) .ok_or_else(|| "DDSketch quantile of an empty population".into()), - SketchQuery::Quantile { .. } => Err("quantile must be in [0, 1]".into()), - SketchQuery::PointCount { value: None, .. } => Ok(self.inner.total_count() as f64), + SketchStatistic::Quantile { .. } => Err("quantile must be in [0, 1]".into()), + SketchStatistic::PointCount { value: None, .. } => Ok(self.inner.total_count() as f64), other => Err(format!("DDSketch does not answer {other:?}").into()), } } @@ -57,8 +57,8 @@ mod tests { use super::*; use planner_types::pre_asap::ColumnRef; - fn bare_count() -> SketchQuery { - SketchQuery::PointCount { + fn bare_count() -> SketchStatistic { + SketchStatistic::PointCount { key: ColumnRef::SampleValue, value: None, } @@ -77,7 +77,9 @@ mod tests { } let merged = a.merge_with(&b).unwrap(); assert_eq!(merged.estimate(&bare_count()).unwrap(), 100.0); - let median = merged.estimate(&SketchQuery::Quantile { q: 0.5 }).unwrap(); + let median = merged + .estimate(&SketchStatistic::Quantile { q: 0.5 }) + .unwrap(); assert!((median - 50.0).abs() <= 1.0, "{median}"); } @@ -85,7 +87,7 @@ mod tests { #[test] fn empty_quantile_and_unsupported_queries_fail() { let dd = DDSketchAccumulator::new(0.01); - assert!(dd.estimate(&SketchQuery::Quantile { q: 0.5 }).is_err()); - assert!(dd.estimate(&SketchQuery::Cardinality).is_err()); + assert!(dd.estimate(&SketchStatistic::Quantile { q: 0.5 }).is_err()); + assert!(dd.estimate(&SketchStatistic::Cardinality).is_err()); } } diff --git a/crates/asap-physical-operators/src/summary_kernels/exact.rs b/crates/asap-physical-operators/src/summary_kernels/exact.rs index 84328cfd4..d5375f9bf 100644 --- a/crates/asap-physical-operators/src/summary_kernels/exact.rs +++ b/crates/asap-physical-operators/src/summary_kernels/exact.rs @@ -2,7 +2,7 @@ use super::increase::IncreaseAccumulator; use crate::Statistic; use crate::{AggregateCore, KeyByLabelValues, Measurement}; -use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; +use planner_types::post_asap::{ExactKind, ExactParams, FieldDataType as SummaryFamilyType}; use serde::{Deserialize, Serialize}; use std::collections::HashMap; @@ -10,7 +10,11 @@ type Error = Box; #[derive(Debug, Clone, Serialize, Deserialize)] enum ScalarState { - Sum { sum: f64, compensation: f64 }, + Sum { + sum: f64, + compensation: f64, + seen: bool, + }, Count(u64), Min(Option), Max(Option), @@ -18,7 +22,7 @@ enum ScalarState { } /// Both the family and population layout survive persistence. Sharing counter -/// arithmetic never authorizes a Rate state to answer an Increase readout. +/// arithmetic never authorizes a Rate state to answer an Increase evaluation. /// /// Deserialization validates the payload against its declared family, so /// deployments can persist this state with any serde format without mirroring @@ -63,10 +67,10 @@ impl TryFrom for ExactAccumulator { } } -/// Planned readout of an exact summary. `lookback_ms` is the logical PromQL +/// Planned evaluation of an exact summary. `lookback_ms` is the logical PromQL /// counter window; the evaluation range is resolved from it at run time. #[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] -pub struct ExactReadout { +pub struct ExactEvaluation { pub statistic: Statistic, #[serde(default, skip_serializing_if = "Option::is_none")] pub lookback_ms: Option, @@ -75,22 +79,24 @@ pub struct ExactReadout { impl ExactAccumulator { /// Read one population. An empty MIN/MAX population reads as `None`. /// `range_ms` extrapolates a counter Rate/Increase to that evaluation range. - pub fn readout( + pub fn evaluation( &self, statistic: Statistic, range_ms: Option<(i64, i64)>, key: Option<&KeyByLabelValues>, ) -> Result, Error> { if statistic != self.statistic() { - return Err("readout differs from Planner exact family".into()); + return Err("evaluation differs from Planner exact family".into()); } let state = match (&self.keyed, key) { (Some(states), Some(key)) => states.get(key).ok_or("unknown exact population")?, (None, None) => &self.scalar, - _ => return Err("readout population differs from installed layout".into()), + _ => return Err("evaluation population differs from installed layout".into()), }; match state { - ScalarState::Sum { sum, compensation } => Ok(Some(sum + compensation)), + ScalarState::Sum { + sum, compensation, .. + } => Ok(Some(sum + compensation)), ScalarState::Count(count) => Ok(Some(*count as f64)), ScalarState::Min(value) | ScalarState::Max(value) => Ok(*value), ScalarState::Counter(Some(counter)) => counter @@ -100,6 +106,11 @@ impl ExactAccumulator { } } + /// SQL SUM distinguishes an empty/all-NULL input from an observed zero. + pub(crate) fn is_empty_sum(&self) -> bool { + self.keyed.is_none() && matches!(self.scalar, ScalarState::Sum { seen: false, .. }) + } + /// Exact integer count of an unkeyed Count state. pub fn count(&self) -> Option { match (&self.keyed, &self.scalar) { @@ -135,6 +146,7 @@ impl ExactAccumulator { SummaryFamilyType::ExactAggregate(K::Sum, P::Sum) => ScalarState::Sum { sum: 0.0, compensation: 0.0, + seen: false, }, SummaryFamilyType::ExactAggregate(K::Count, P::Count) => ScalarState::Count(0), SummaryFamilyType::ExactAggregate(K::Min, P::Min) => ScalarState::Min(None), @@ -190,7 +202,14 @@ impl ExactAccumulator { _ => panic!("exact update population layout differs from installed DAG"), }; match state { - ScalarState::Sum { sum, compensation } => compensated_add(sum, compensation, value), + ScalarState::Sum { + sum, + compensation, + seen, + } => { + compensated_add(sum, compensation, value); + *seen = true; + } ScalarState::Count(count) => { *count = count.checked_add(1).expect("exact count overflow") } @@ -246,16 +265,22 @@ fn merge_scalar(left: &ScalarState, right: &ScalarState) -> Result { let (mut sum, mut compensation) = (*a, *ac); compensated_add(&mut sum, &mut compensation, *b); compensated_add(&mut sum, &mut compensation, *bc); - ScalarState::Sum { sum, compensation } + ScalarState::Sum { + sum, + compensation, + seen: *a_seen || *b_seen, + } } (ScalarState::Count(a), ScalarState::Count(b)) => { ScalarState::Count(a.checked_add(*b).ok_or("exact count overflow")?) @@ -339,7 +364,7 @@ mod tests { negative.update(None, -1e16, 1); restored.merge_from(&negative).unwrap(); assert_eq!( - restored.readout(Statistic::Sum, None, None).unwrap(), + restored.evaluation(Statistic::Sum, None, None).unwrap(), Some(1.0) ); } @@ -351,18 +376,18 @@ mod tests { state.update(None, f64::INFINITY, 0); state.update(None, 1.0, 0); assert_eq!( - state.readout(Statistic::Sum, None, None).unwrap(), + state.evaluation(Statistic::Sum, None, None).unwrap(), Some(f64::INFINITY) ); state.update(None, f64::NEG_INFINITY, 0); assert!(state - .readout(Statistic::Sum, None, None) + .evaluation(Statistic::Sum, None, None) .unwrap() .unwrap() .is_nan()); } - // A persisted exact state decodes back to the same family, layout and readout. + // A persisted exact state decodes back to the same family, layout and evaluation. #[test] fn serialized_state_round_trips() { let mut state = ExactAccumulator::new(sum(), true).unwrap(); @@ -372,7 +397,9 @@ mod tests { let restored: ExactAccumulator = rmp_serde::from_slice(&bytes).unwrap(); assert_eq!(restored.family(), &sum()); assert_eq!( - restored.readout(Statistic::Sum, None, Some(&key)).unwrap(), + restored + .evaluation(Statistic::Sum, None, Some(&key)) + .unwrap(), Some(2.5) ); } @@ -392,6 +419,7 @@ mod tests { scalar: ScalarState::Sum { sum: 0.0, compensation: 0.0, + seen: false, }, keyed: Some(HashMap::from([(key, ScalarState::Max(Some(1.0)))])), }; @@ -406,6 +434,7 @@ mod tests { scalar: ScalarState::Sum { sum: 0.0, compensation: 0.0, + seen: false, }, keyed: None, }; diff --git a/crates/asap-physical-operators/src/summary_kernels/factory.rs b/crates/asap-physical-operators/src/summary_kernels/factory.rs index d44e64ab8..be819f8a7 100644 --- a/crates/asap-physical-operators/src/summary_kernels/factory.rs +++ b/crates/asap-physical-operators/src/summary_kernels/factory.rs @@ -6,7 +6,7 @@ use crate::summary_kernels::{ HydraKllSketchAccumulator, }; use crate::{AggregateCore, KeyByLabelValues}; -use planner_types::post_asap::{SketchAlgorithm, SketchParams, SummaryFamilyType}; +use planner_types::post_asap::{FieldDataType as SummaryFamilyType, SketchAlgorithm, SketchParams}; /// Generate the clone-based `AccumulatorUpdater` methods for updaters whose /// inner `acc` field implements `Clone + AggregateCore`. diff --git a/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs index f675da230..993f9d9b6 100644 --- a/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs +++ b/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs @@ -1,7 +1,7 @@ //! HyperLogLog distinct-count summary over `asap_sketchlib::HllSketch`. use crate::{AggregateCore, KernelError}; use asap_sketchlib::{HllSketch, HllVariant}; -use planner_types::post_asap::SketchQuery; +use planner_types::post_asap::SketchStatistic; #[derive(Debug, Clone)] pub struct HllSketchAccumulator { @@ -39,9 +39,9 @@ impl AggregateCore for HllSketchAccumulator { } /// Distinct count. A bare `PointCount` over an HLL also reads the distinct count. - fn estimate(&self, query: &SketchQuery) -> Result { + fn estimate(&self, query: &SketchStatistic) -> Result { match query { - SketchQuery::Cardinality | SketchQuery::PointCount { value: None, .. } => { + SketchStatistic::Cardinality | SketchStatistic::PointCount { value: None, .. } => { Ok(self.inner.estimate()) } other => Err(format!("HLL does not answer {other:?}").into()), @@ -69,7 +69,7 @@ mod tests { b.inner.update(&(v + 500).to_le_bytes()); } let merged = a.merge_with(&b).unwrap(); - let estimate = merged.estimate(&SketchQuery::Cardinality).unwrap(); + let estimate = merged.estimate(&SketchStatistic::Cardinality).unwrap(); assert!((estimate - 1500.0).abs() / 1500.0 < 0.05, "{estimate}"); } @@ -77,6 +77,6 @@ mod tests { #[test] fn rejects_quantile() { let hll = HllSketchAccumulator::new(HllVariant::Regular, 12); - assert!(hll.estimate(&SketchQuery::Quantile { q: 0.5 }).is_err()); + assert!(hll.estimate(&SketchStatistic::Quantile { q: 0.5 }).is_err()); } } diff --git a/crates/asap-physical-operators/src/summary_kernels/traits.rs b/crates/asap-physical-operators/src/summary_kernels/traits.rs index 3902a296c..46e2c7f86 100644 --- a/crates/asap-physical-operators/src/summary_kernels/traits.rs +++ b/crates/asap-physical-operators/src/summary_kernels/traits.rs @@ -1,11 +1,11 @@ -use planner_types::post_asap::SketchQuery; +use planner_types::post_asap::SketchStatistic; pub type KernelError = Box; /// In-memory state of one population's summary. /// /// Kernels adapt `asap_sketchlib` structures (or exact Planner state) to the -/// operations physical operators need: merge, typed readout and memory +/// operations physical operators need: merge, typed evaluation and memory /// accounting. Grouping belongs to operators; byte encodings belong to /// `asap_sketchlib` and deployments. pub trait AggregateCore: Send + Sync { @@ -20,9 +20,9 @@ pub trait AggregateCore: Send + Sync { /// Merge with a state of the same family and shape, leaving both inputs unchanged. fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError>; - /// Answer a sketch readout. Exact states are read through - /// [`ExactAccumulator::readout`](super::exact::ExactAccumulator::readout). - fn estimate(&self, query: &SketchQuery) -> Result { + /// Answer a sketch evaluation. Exact states are read through + /// [`ExactAccumulator::evaluation`](super::exact::ExactAccumulator::evaluation). + fn estimate(&self, query: &SketchStatistic) -> Result { Err(format!("{query:?} is not supported by this summary").into()) } @@ -52,7 +52,7 @@ mod tests { .downcast_mut::() .unwrap(); dd.inner.update(3.0); - let count = SketchQuery::PointCount { + let count = SketchStatistic::PointCount { key: planner_types::pre_asap::ColumnRef::SampleValue, value: None, }; diff --git a/crates/asap-physical-operators/src/summary_kernels/univmon.rs b/crates/asap-physical-operators/src/summary_kernels/univmon.rs index 6d42c336f..23114ab1b 100644 --- a/crates/asap-physical-operators/src/summary_kernels/univmon.rs +++ b/crates/asap-physical-operators/src/summary_kernels/univmon.rs @@ -1,8 +1,8 @@ -//! One frequency state shared by count, distinct, L2 and entropy readouts. +//! One frequency state shared by count, distinct, L2 and entropy evaluations. use crate::AggregateCore; use asap_sketchlib::{DataInput, UnivMon}; -use planner_types::{post_asap::SketchQuery, pre_asap::ColumnRef}; +use planner_types::{post_asap::SketchStatistic, pre_asap::ColumnRef}; type Error = Box; @@ -144,15 +144,15 @@ impl AggregateCore for UnivMonAccumulator { /// Sample count (a bare `PointCount`), distinct count, L2 norm and entropy /// of the sample-value frequencies. - fn estimate(&self, query: &SketchQuery) -> Result { + fn estimate(&self, query: &SketchStatistic) -> Result { Ok(match query { - SketchQuery::PointCount { + SketchStatistic::PointCount { key: ColumnRef::SampleValue, value: None, } => self.inner.calc_l1(), - SketchQuery::Cardinality => self.inner.calc_card(), - SketchQuery::FrequencyL2 => self.inner.calc_l2(), - SketchQuery::FrequencyEntropy => self.inner.calc_entropy(), + SketchStatistic::Cardinality => self.inner.calc_card(), + SketchStatistic::FrequencyL2 => self.inner.calc_l2(), + SketchStatistic::FrequencyEntropy => self.inner.calc_entropy(), other => return Err(format!("UnivMon does not answer {other:?}").into()), }) } @@ -162,32 +162,34 @@ impl AggregateCore for UnivMonAccumulator { mod tests { use super::*; - fn count() -> SketchQuery { - SketchQuery::PointCount { + fn count() -> SketchStatistic { + SketchStatistic::PointCount { key: ColumnRef::SampleValue, value: None, } } - // Count, distinct, L2 and entropy readouts count each non-NaN sample once; + // Count, distinct, L2 and entropy evaluations count each non-NaN sample once; // signed zero is one identity. #[test] - fn frequency_readouts() { + fn frequency_evaluations() { let mut state = UnivMonAccumulator::new(32, 5, 1024, 4).unwrap(); for value in [0.0, -0.0, 2.0, 2.0, f64::NAN] { state.insert_sample(value).unwrap(); } let read = |query| state.estimate(&query).unwrap(); assert_eq!(read(count()), 4.0); - assert!((read(SketchQuery::Cardinality) - 2.0).abs() < 0.01); - assert!((read(SketchQuery::FrequencyL2) - 8.0f64.sqrt()).abs() < 0.01); - assert!((read(SketchQuery::FrequencyEntropy) - 1.0).abs() < 0.01); - assert!(state.estimate(&SketchQuery::Quantile { q: 0.5 }).is_err()); + assert!((read(SketchStatistic::Cardinality) - 2.0).abs() < 0.01); + assert!((read(SketchStatistic::FrequencyL2) - 8.0f64.sqrt()).abs() < 0.01); + assert!((read(SketchStatistic::FrequencyEntropy) - 1.0).abs() < 0.01); + assert!(state + .estimate(&SketchStatistic::Quantile { q: 0.5 }) + .is_err()); } - // A sketch taken out and adopted back answers the same readouts. + // A sketch taken out and adopted back answers the same evaluations. #[test] - fn adopted_sketch_keeps_readouts() { + fn adopted_sketch_keeps_evaluations() { let mut state = UnivMonAccumulator::new(32, 5, 1024, 4).unwrap(); for value in [1.0, 2.0, 2.0] { state.insert_sample(value).unwrap(); @@ -196,8 +198,8 @@ mod tests { assert_eq!(adopted.dimensions(), state.dimensions()); for query in [ count(), - SketchQuery::Cardinality, - SketchQuery::FrequencyEntropy, + SketchStatistic::Cardinality, + SketchStatistic::FrequencyEntropy, ] { assert_eq!( adopted.estimate(&query).unwrap(), diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs index 1d36c4557..4ce5cd6a1 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -2,7 +2,9 @@ use crate::AggregateCore; use crate::Error; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{ + Field as SummaryField, FieldDataType as SummaryFamilyType, Schema as SummarySchema, + }, pre_asap::DataType, }; use std::{cmp::Ordering, sync::Arc}; diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/asap-physical-operators/tests/blocking_resources.rs index 7a312892b..0fb4e1aa5 100644 --- a/crates/asap-physical-operators/tests/blocking_resources.rs +++ b/crates/asap-physical-operators/tests/blocking_resources.rs @@ -7,18 +7,23 @@ use asap_physical_operators::{ Error, }; use futures::{executor::block_on, FutureExt, StreamExt}; +use planner_types::ir::Predicate; +use planner_types::ir::ScalarExpr as QueryExpr; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, - pre_asap::{DataType, JoinKind, Predicate, QueryExpr, ScalarValue}, + post_asap::{Field, FieldDataType}, + pre_asap::{DataType, JoinKind, ScalarValue}, }; use std::sync::Arc; fn schema(width: usize) -> Schema { - Arc::new(SummarySchema { + Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, fields: (0..width) - .map(|i| SummaryField { + .map(|i| Field { + table: None, name: format!("v{i}"), - dtype: SummaryFamilyType::Plain(DataType::Int64), + dtype: FieldDataType::Plain(DataType::Int64), nullable: false, }) .collect(), @@ -57,9 +62,7 @@ fn cross_join() -> Operator { schema(1), schema(1), JoinKind::Cross, - &Predicate(std::rc::Rc::new(QueryExpr::Literal(ScalarValue::Boolean( - true, - )))), + &Predicate(QueryExpr::Literal(ScalarValue::Boolean(true))), schema(2), ) .unwrap() @@ -186,16 +189,21 @@ fn cooperative_sort_preserves_ties_across_chunks() { #[test] fn weighted_summary_build_yields_within_a_batch() { use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; - let input = Arc::new(SummarySchema { + + let input = Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, fields: vec![ - SummaryField { + Field { + table: None, name: "item".into(), - dtype: SummaryFamilyType::Plain(DataType::Int64), + dtype: FieldDataType::Plain(DataType::Int64), nullable: false, }, - SummaryField { + Field { + table: None, name: "weight".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, }, ], @@ -216,7 +224,7 @@ fn weighted_summary_build_yields_within_a_batch() { Operator::source(input.clone(), vec![batch]).unwrap(), ) .unwrap(); - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::CmsWithHeap, SketchParams::CmsWithHeap { diff --git a/crates/asap-physical-operators/tests/common/mod.rs b/crates/asap-physical-operators/tests/common/mod.rs new file mode 100644 index 000000000..c1a635418 --- /dev/null +++ b/crates/asap-physical-operators/tests/common/mod.rs @@ -0,0 +1,15 @@ +#![allow(dead_code)] +use planner_types::ir::export::PostAsapDag; +use planner_types::ir::{apply_lifecycle_timings, LifecycleAssignment, OperatorNode, TimingMemo}; +use std::rc::Rc; + +pub fn compile_post_asap_dag( + root: &Rc, +) -> Result> { + let root = apply_lifecycle_timings( + root, + &LifecycleAssignment::default(), + &mut TimingMemo::default(), + )?; + Ok(planner_types::ir::export::compile_post_asap_dag(&root)?) +} diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs index 322bd6fee..0906a0c26 100644 --- a/crates/asap-physical-operators/tests/current_series_heap.rs +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -1,4 +1,5 @@ //! Spatial heap weights come from a fresh instant vector, never sample history. +mod common; use asap_physical_operators::{ operators::Operator, physical_planner::{ @@ -8,12 +9,16 @@ use asap_physical_operators::{ runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; +use common::compile_post_asap_dag; use futures::{executor::block_on, StreamExt}; +use planner_types::ir::export::PostAsapOperatorPayload; use planner_types::{post_asap::*, pre_asap::DataType}; use std::{collections::BTreeMap, sync::Arc}; -fn schema() -> Arc { - Arc::new(SummarySchema { +fn schema() -> Arc { + Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, fields: [ ("ts", DataType::Timestamp), ("value", DataType::Float64), @@ -21,9 +26,10 @@ fn schema() -> Arc { (SERIES_IDENTITY_COLUMN, DataType::Utf8), ] .into_iter() - .map(|(name, dtype)| SummaryField { + .map(|(name, dtype)| Field { + table: None, name: name.into(), - dtype: SummaryFamilyType::Plain(dtype), + dtype: FieldDataType::Plain(dtype), nullable: false, }) .collect(), @@ -163,10 +169,11 @@ fn spatial_heap_ranks_latest_values_in_independent_runs() { heap_size: 100, }, }; - let family = - SummaryFamilyType::Sketch(SketchKind::new(algorithm, params), Default::default()); + let family = FieldDataType::Sketch(SketchKind::new(algorithm, params), Default::default()); let build = Operator::keyed_summary_build(schema(), family, 1, vec![3], vec![2]).unwrap(); - let output = Arc::new(SummarySchema { + let output = Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, fields: vec![ schema().fields[2].clone(), schema().fields[3].clone(), @@ -174,7 +181,7 @@ fn spatial_heap_ranks_latest_values_in_independent_runs() { ], time_index: None, }); - let read = Operator::keyed_readout(build.schema(), 1, 1, output).unwrap(); + let read = Operator::keyed_evaluation(build.schema(), 1, 1, output).unwrap(); let plan = CompiledPhysicalDag::from_operators( BTreeMap::from([(0, InputContract::bounded(schema()))]), BTreeMap::from([ @@ -326,7 +333,7 @@ fn planner_current_series_candidate_compiles_with_dynamic_identity() { .candidate(&open_root) .unwrap(); let snapshot_program = - asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + asap_physical_operators::physical_planner::promql_rows::compile_current_series_evaluation( &open_selected, ) .unwrap(); @@ -347,7 +354,14 @@ fn planner_current_series_candidate_compiles_with_dynamic_identity() { let raw = logical .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::TimeRange { .. } + } + ) + }) .unwrap(); let raw_schema = Arc::new(raw.output_schema.clone()); let physical = compile( diff --git a/crates/asap-physical-operators/tests/deployment.rs b/crates/asap-physical-operators/tests/deployment.rs index 0c261a033..72e0503f1 100644 --- a/crates/asap-physical-operators/tests/deployment.rs +++ b/crates/asap-physical-operators/tests/deployment.rs @@ -1,15 +1,14 @@ //! Exercise the public library without a backend server, store, or scheduler. use asap_physical_operators::planner::{ post_asap::{ - GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryFamilyType, - SummaryUpdate, + FieldDataType, GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate, }, pre_asap::ColumnRef, }; use asap_physical_operators::{factory::create_planner_accumulator, AggregateCore}; -fn family(k: u32) -> SummaryFamilyType { - SummaryFamilyType::Sketch( +fn family(k: u32) -> FieldDataType { + FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), GroupingStrategy::PerSubpopulationInstance, ) @@ -29,12 +28,14 @@ fn build(values: &[f64]) -> Box { } fn read(state: &dyn AggregateCore) -> f64 { state - .estimate(&asap_physical_operators::planner::post_asap::SketchQuery::Quantile { q: 0.5 }) + .estimate( + &asap_physical_operators::planner::post_asap::SketchStatistic::Quantile { q: 0.5 }, + ) .unwrap() } // The same kernels work when every build is query-time, when only a prefix -// was precomputed, and when all state was precomputed before the readout. +// was precomputed, and when all state was precomputed before the evaluation. #[test] fn raw_partial_and_fully_precomputed_use_the_same_kernels() { let raw: Vec = (0..128).map(f64::from).collect(); @@ -65,7 +66,7 @@ fn invalid_kll_parameters_are_rejected_at_binding() { fn native_count_sketch_dimensions_are_not_packed_wire_dimensions() { use asap_physical_operators::planner::post_asap::SummaryInputExpr; use asap_physical_operators::KeyByLabelValues; - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::CountSketchWithHeap, SketchParams::CountSketchWithHeap { diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index 3a92e4d96..99a150bc7 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -1,20 +1,23 @@ //! Planner-selected PromQL computation compiles from the timed DAG alone; //! the deployment supplies only raw rows at the ingestion frontier. +mod common; use asap_physical_operators::{ operators::Operator, physical_planner::{compile, promql_rows, CompiledPhysicalDag, InputContract, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; +use common::compile_post_asap_dag; use futures::{executor::block_on, StreamExt}; -use planner_types::{post_asap::*, pre_asap::QueryExpr, types::AccuracyTarget, workload::*}; +use planner_types::ir::export::{PostAsapDag, PostAsapOperatorPayload}; +use planner_types::{post_asap::*, types::AccuracyTarget, workload::*}; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; -fn lower(query: &str) -> QueryExpr { +fn lower(query: &str) -> Rc { lower_with(query, AccuracyTarget::Exact) } -fn lower_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { +fn lower_with(query: &str, accuracy: AccuracyTarget) -> Rc { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -46,27 +49,19 @@ fn lower_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { /// The first exact summary candidate, as Planner selection would hand it over. fn exact_dag(query: &str) -> PostAsapDag { - use asap_aware_mapping::{Replacement, ReplacementStrategy, TargetSubDAG}; let expression = lower(query); - let root = Rc::new(promql_rows::with_series_identity(&expression).unwrap_or(expression)); - asap_aware_mapping::SketchAlgorithmStrategy::new(&asap_aware_mapping::DefaultCostModel) - .replacements(&TargetSubDAG::new(&root)) - .into_iter() - .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) => { - let dag = compile_post_asap_dag(&node).ok()?; - dag.nodes - .iter() - .all(|n| !matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { family, .. } if !matches!(family, SummaryFamilyType::ExactAggregate(..)))) - .then_some(dag) - } - _ => None, - }) + let root = promql_rows::with_series_identity(&expression).unwrap_or(expression); + let space = asap_aware_mapping::search_workload(vec![("q", root)]); + let selected = space + .global_selection(&asap_aware_mapping::DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) .unwrap() + .unwrap(); + compile_post_asap_dag(&selected).unwrap() } fn population_dag(query: &str) -> PostAsapDag { - let root = Rc::new(promql_rows::with_series_identity(&lower(query)).unwrap()); + let root = promql_rows::with_series_identity(&lower(query)).unwrap(); let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( std::slice::from_ref(&root), ) @@ -76,23 +71,33 @@ fn population_dag(query: &str) -> PostAsapDag { } /// Raw scan nodes are the frontier; everything above them is compiled. -fn raw_inputs(dag: &PostAsapDag) -> Vec<(u64, Arc, String)> { +fn raw_inputs(dag: &PostAsapDag) -> Vec<(u64, Arc, String)> { dag.nodes .iter() .filter_map(|node| match &node.payload { - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::TimeRange { child, .. }, - } => match child.as_ref() { - QueryExpr::Scan { - source: planner_types::pre_asap::Source::TimeSeries { metric }, - .. - } => Some(( - u64::from(node.id.0), - Arc::new(node.output_schema.clone()), - metric.clone(), - )), - _ => None, - }, + PostAsapOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::TimeRange { .. }, + } => { + let mut id = node.id; + loop { + let n = dag.nodes.iter().find(|n| n.id == id)?; + if let PostAsapOperatorPayload::Relational { + operator: + planner_types::ir::export::NonASAPOpKind::Scan { + source: planner_types::pre_asap::Source::TimeSeries { metric }, + .. + }, + } = &n.payload + { + return Some(( + u64::from(node.id.0), + Arc::new(node.output_schema.clone()), + metric.clone(), + )); + } + id = dag.edges.iter().find(|e| e.consumer == id)?.producer; + } + } _ => None, }) .collect() @@ -249,7 +254,7 @@ fn population_aggregates_match_current_series_reference() { } } -// A global readout of an empty population is an empty vector, as in PromQL. +// A global evaluation of an empty population is an empty vector, as in PromQL. #[test] fn global_population_aggregate_of_no_members_is_empty() { // Latest values are [1, 2, 5, 7] at 60s; every member has expired by 1000s. @@ -312,40 +317,7 @@ fn grouped_vector_arithmetic_matches_labels() { // then roll up per job: api has 2 + 1 + 1 samples in 5m, db has 1. #[test] fn exact_count_finalizes_to_declared_float_value() { - let mut dag = exact_dag("sum by (job) (count_over_time(m[5m]))"); - let finalize = dag - .nodes - .iter() - .find(|node| { - matches!( - node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator - } - ) - }) - .unwrap() - .clone(); - let root = dag.nodes.iter().find(|n| n.id == dag.root).unwrap().clone(); - let mut edge = dag - .edges - .iter() - .find(|e| e.producer == finalize.id) - .unwrap() - .clone(); - // Read the rolled-up exact state the same way the query path does. - let mut read = finalize.clone(); - read.id = PostAsapNodeId(root.id.0 + 1); - read.output_schema = root.output_schema.clone(); - read.output_schema.fields.last_mut().unwrap().dtype = - SummaryFamilyType::Plain(planner_types::pre_asap::DataType::Float64); - edge.producer = root.id; - edge.consumer = read.id; - edge.intermediate_schema = root.output_schema.clone(); - edge.data_state = root.output_state; - dag.root = read.id; - dag.nodes.push(read); - dag.edges.push(edge); + let dag = exact_dag("sum by (job) (count_over_time(m[5m]))"); assert_eq!( run(&dag, SAMPLES, 60_000).unwrap(), reference(&[("api", 4.), ("db", 1.)]) @@ -353,10 +325,22 @@ fn exact_count_finalizes_to_declared_float_value() { } /// `dag` with its Binary operator replaced by `kind`. -fn with_kind(mut dag: PostAsapDag, kind: planner_types::pre_asap::BinaryOpKind) -> PostAsapDag { +fn with_kind( + mut dag: PostAsapDag, + kind: planner_types::pre_asap::BinaryOpKind, + bool_result: bool, +) -> PostAsapDag { for node in &mut dag.nodes { - if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { + if let PostAsapOperatorPayload::Relational { + operator: + planner_types::ir::export::NonASAPOpKind::BinaryOp { + operator, + return_bool, + }, + } = &mut node.payload + { operator.kind = kind.clone(); + *return_bool = bool_result; } } dag @@ -366,30 +350,30 @@ fn with_kind(mut dag: PostAsapDag, kind: planner_types::pre_asap::BinaryOpKind) // holds, with their value, on either side of the literal; `bool` yields 1 or 0. #[test] fn grouped_comparisons_filter_or_return_bool() { - use planner_types::pre_asap::{BinaryOpKind::*, CompareOpKind::Gt}; - // sum_over_time over 5m per job: api = 14, db = 5. - let right = exact_dag("sum by (job) (sum_over_time(m[5m])) * 10"); - let left = exact_dag("10 - sum by (job) (sum_over_time(m[5m]))"); - for (dag, expected) in [ + for (query, expected) in [ ( - with_kind(right.clone(), Compare(Gt)), + "sum by(job)(sum_over_time(m[5m])) > 10", reference(&[("api", 14.)]), ), ( - with_kind(right, CompareBool(Gt)), + "sum by(job)(sum_over_time(m[5m])) > bool 10", reference(&[("api", 1.), ("db", 0.)]), ), - (with_kind(left, Compare(Gt)), reference(&[("db", 5.)])), + ( + "10 > sum by(job)(sum_over_time(m[5m]))", + reference(&[("db", 5.)]), + ), ] { - assert_eq!(run(&dag, SAMPLES, 60_000).unwrap(), expected); + assert_eq!(run(&exact_dag(query), SAMPLES, 60_000).unwrap(), expected); } } -// A `bool` comparison Binary over per-series readouts matches one-to-one and +// A `bool` comparison Binary over per-series evaluations matches one-to-one and // drops the metric name; a filter keeps the surviving left value. #[test] fn per_series_comparisons_filter_or_return_bool() { use planner_types::pre_asap::{BinaryOpKind::*, CompareOpKind::*}; + let samples = counter("a", "api", 10., 10.) .chain(counter("a", "db", 10., 10.)) .chain(counter("b", "api", 5., 5.)) @@ -398,11 +382,16 @@ fn per_series_comparisons_filter_or_return_bool() { // rate: a{api} = a{db} = 50/300, b{api} = 25/300, b{db} = 100/300. let dag = exact_dag("rate(a[5m]) / rate(b[5m])"); assert_eq!( - run_series(&with_kind(dag.clone(), Compare(Gt)), &samples, 300_000).unwrap(), + run_series( + &with_kind(dag.clone(), Compare(Gt), false), + &samples, + 300_000 + ) + .unwrap(), series(&[("api", "x", 50. / 300.)]) ); assert_eq!( - run_series(&with_kind(dag, CompareBool(Lt)), &samples, 300_000).unwrap(), + run_series(&with_kind(dag, Compare(Lt), true), &samples, 300_000).unwrap(), series(&[("api", "x", 0.), ("db", "x", 1.)]) ); } @@ -508,13 +497,13 @@ fn per_series_rate_ratio_matches_prometheus() { // A literal operand applies to every stored per-series value, on either side, // and drops the metric name. #[test] -fn per_series_scalar_arithmetic_applies_to_stored_readouts() { +fn per_series_scalar_arithmetic_applies_to_stored_evaluations() { let samples = counter("m", "api", 10., 10.).collect::>(); // rate = 40 * 1.25 / 300 = 1/6. for (query, expected) in [ ("rate(m[5m]) * 2", 50. / 300. * 2.), ("1 - rate(m[5m])", 1. - 50. / 300.), - // The stored sum readout keeps `__name__`; the arithmetic drops it. + // The stored sum evaluation keeps `__name__`; the arithmetic drops it. ("sum_over_time(m[5m]) * 2", 150. * 2.), ] { assert_eq!( @@ -549,7 +538,14 @@ fn with_vector_match( labels: &[&str], ) -> PostAsapDag { for node in &mut dag.nodes { - if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { + if let PostAsapOperatorPayload::Relational { + operator: + planner_types::ir::export::NonASAPOpKind::BinaryOp { + operator, + return_bool: _, + }, + } = &mut node.payload + { operator.vector_match = Some(planner_types::pre_asap::VectorMatch { kind: kind.clone(), labels: labels.iter().map(|l| l.to_string()).collect(), @@ -620,45 +616,44 @@ fn population_sums_and_averages_are_compensated() { } } -// A bare count over stored Count-Min state compiles to a Planner readout that +// A bare count over stored Count-Min state compiles to a Planner evaluation that // returns the sketch's total update weight, including colliding items. #[test] -fn stored_count_min_bare_count_compiles_to_a_readout() { +fn stored_count_min_bare_count_compiles_to_a_evaluation() { use asap_aware_mapping::{Replacement, ReplacementStrategy, TargetSubDAG}; use asap_physical_operators::summary_kernels::CountMinSketchAccumulator; - let root = Rc::new(lower_with("count(up)", AccuracyTarget::Epsilon(0.02))); - let dag = - asap_aware_mapping::SketchAlgorithmStrategy::new(&asap_aware_mapping::DefaultCostModel) - .replacements(&TargetSubDAG::new(&root)) - .into_iter() - .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) => { - let dag = compile_post_asap_dag(&node).ok()?; - let bare_count = dag.nodes.iter().any(|n| { - matches!( - &n.payload, - PostAsapOperatorPayload::SummaryEstimate { - query: SketchQuery::PointCount { value: None, .. } - } - ) - }); - let count_min = dag.nodes.iter().any(|n| { - matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), .. + let root = lower_with("count(up)", AccuracyTarget::Epsilon(0.02)); + let dag = asap_aware_mapping::ASAPStrategies::new(&asap_aware_mapping::DefaultCostModel) + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .find_map(|candidate| match candidate.replacement { + Replacement::SubDag(node) => { + let dag = compile_post_asap_dag(&node).ok()?; + let bare_count = dag.nodes.iter().any(|n| { + matches!( + &n.payload, + PostAsapOperatorPayload::SummaryEstimate { + query: SketchStatistic::PointCount { value: None, .. } + } + ) + }); + let count_min = dag.nodes.iter().any(|n| { + matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { + family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::Cms) - }); - (bare_count && count_min).then_some(dag) - } - _ => None, - }) - .expect("Planner lists a Count-Min candidate for count(up)"); + }); + (bare_count && count_min).then_some(dag) + } + _ => None, + }) + .expect("Planner lists a Count-Min candidate for count(up)"); let state = dag .nodes .iter() .find(|n| matches!(n.payload, PostAsapOperatorPayload::SummaryAgg { .. })) .unwrap(); let PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + family: FieldDataType::Sketch(kind, _), .. } = &state.payload else { @@ -686,7 +681,7 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { .fields .iter() .map(|field| match &field.dtype { - SummaryFamilyType::Plain(_) => panic!("unexpected stored column {field:?}"), + FieldDataType::Plain(_) => panic!("unexpected stored column {field:?}"), family => Value::Summary { family: family.clone(), state: Arc::new(sketch.clone()), diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 652880e04..bc2e77c11 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -8,18 +8,28 @@ use asap_physical_operators::{ Statistic, }; use futures::{executor::block_on, StreamExt}; +use planner_types::ir::export::NonASAPOpKind as ValueOperation; +use planner_types::ir::export::{ + EdgeRole, GroupingEdgeCompatibility, PostAsapDag, PostAsapDagEdge, PostAsapDagNode, + PostAsapNodeId, PostAsapOperatorPayload, WindowEdgeCompatibility, +}; +use planner_types::ir::BinaryOperator; +use planner_types::ir::ScalarExpr as QueryExpr; use planner_types::{ - post_asap::{ExactKind, ExactParams, SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{ExactKind, ExactParams, Field, FieldDataType}, pre_asap::DataType, }; use std::sync::Arc; fn schema(fields: &[(&str, DataType, bool)]) -> Schema { - Arc::new(SummarySchema { + Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, fields: fields .iter() - .map(|(name, dtype, nullable)| SummaryField { + .map(|(name, dtype, nullable)| Field { + table: None, name: (*name).into(), - dtype: SummaryFamilyType::Plain(dtype.clone()), + dtype: FieldDataType::Plain(dtype.clone()), nullable: *nullable, }) .collect(), @@ -114,12 +124,12 @@ fn grouped_sort_limit_across_batches() { // The same computation runs in either engine scope with fresh per-run state. #[test] -fn summary_construction_merge_and_readout_at_both_phases() { +fn summary_construction_merge_and_evaluation_at_both_phases() { let schema = schema(&[("v", DataType::Float64, false)]); let batches = (1..=20) .map(|v| Batch::try_new(schema.clone(), vec![vec![Value::Float64(v as f64)]]).unwrap()) .collect(); - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let build = Operator::summary_build(schema.clone(), family, 0, None, vec![]).unwrap(); let state = build.schema(); let mut dag = PhysicalDag::default(); @@ -137,11 +147,11 @@ fn summary_construction_merge_and_readout_at_both_phases() { dag.add( 4, vec![3], - Operator::readout( + Operator::evaluation( state, 0, - asap_physical_operators::operators::ReadoutQuery::Exact( - asap_physical_operators::summary_kernels::exact::ExactReadout { + asap_physical_operators::operators::SummaryEvaluation::Exact( + asap_physical_operators::summary_kernels::exact::ExactEvaluation { statistic: Statistic::Sum, lookback_ms: None, }, @@ -278,7 +288,7 @@ fn binding_rejects_unsupported_operations() { let schema = schema(&[("v", DataType::Float64, false)]); assert!(Operator::summary_build( schema.clone(), - SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), + FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), 0, None, vec![] @@ -286,17 +296,17 @@ fn binding_rejects_unsupported_operations() { .is_err()); let sum = Operator::summary_build( schema.clone(), - SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), 0, None, vec![], ) .unwrap(); - assert!(Operator::readout( + assert!(Operator::evaluation( sum.schema(), 0, - asap_physical_operators::operators::ReadoutQuery::Sketch( - planner_types::post_asap::SketchQuery::Quantile { q: 0.5 } + asap_physical_operators::operators::SummaryEvaluation::Sketch( + planner_types::post_asap::SketchStatistic::Quantile { q: 0.5 } ) ) .is_err()); @@ -307,8 +317,9 @@ fn binding_rejects_unsupported_operations() { #[test] fn kll_raw_partial_and_precomputed_are_native_dags() { use planner_types::post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("value", DataType::Float64, false)]); - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), GroupingStrategy::PerSubpopulationInstance, ); @@ -393,11 +404,11 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { dag.add( 5, vec![4], - Operator::readout( + Operator::evaluation( state.clone(), 0, - asap_physical_operators::operators::ReadoutQuery::Sketch( - planner_types::post_asap::SketchQuery::Quantile { q: 0.5 }, + asap_physical_operators::operators::SummaryEvaluation::Sketch( + planner_types::post_asap::SketchStatistic::Quantile { q: 0.5 }, ), ) .unwrap(), @@ -417,11 +428,14 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { #[test] fn exact_state_and_family_validation() { use asap_physical_operators::summary_kernels::exact::ExactAccumulator; - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); acc.update(None, 7., 0); - let schema = Arc::new(SummarySchema { - fields: vec![SummaryField { + let schema = Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, + fields: vec![Field { + table: None, name: "state".into(), dtype: family.clone(), nullable: false, @@ -446,11 +460,11 @@ fn exact_state_and_family_validation() { dag.add( 1, vec![0], - Operator::readout( + Operator::evaluation( schema.clone(), 0, - asap_physical_operators::operators::ReadoutQuery::Exact( - asap_physical_operators::summary_kernels::exact::ExactReadout { + asap_physical_operators::operators::SummaryEvaluation::Exact( + asap_physical_operators::summary_kernels::exact::ExactEvaluation { statistic: Statistic::Sum, lookback_ms: None, }, @@ -461,7 +475,7 @@ fn exact_state_and_family_validation() { .unwrap(); assert_eq!(floats(&run(&dag, 1, query()), 0), vec![7.]); let wrong = ExactAccumulator::new( - SummaryFamilyType::ExactAggregate(ExactKind::Max, ExactParams::Max), + FieldDataType::ExactAggregate(ExactKind::Max, ExactParams::Max), false, ) .unwrap(); @@ -480,14 +494,10 @@ fn exact_state_and_family_validation() { fn bind_post_asap_before_execution() { use asap_physical_operators::dag::planner::bind; use planner_types::{ - post_asap::{ - EdgeRole, ExecutionDataState, GroupingEdgeCompatibility, PostAsapDag, PostAsapDagEdge, - PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, ValueOperation, - WindowEdgeCompatibility, - }, - pre_asap::{ArithmeticOpKind, ProjectItem, QueryExpr, ScalarValue}, + post_asap::ExecutionDataState, + pre_asap::{ArithmeticOpKind, ScalarValue}, }; - use std::{collections::BTreeMap, rc::Rc}; + use std::collections::BTreeMap; let schema = schema(&[("value", DataType::Float64, false)]); let node = |id, payload| PostAsapDagNode { id: PostAsapNodeId(id), @@ -500,20 +510,32 @@ fn bind_post_asap_before_execution() { nodes: vec![ node( 0, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(1.), + PostAsapOperatorPayload::Relational { + operator: ValueOperation::Values { + rows: vec![vec![planner_types::ir::export::WireScalarExpr::Literal( + planner_types::pre_asap::ScalarValue::Float64(1.), + )]], + schema: (*schema).clone(), + }, }, ), node( 1, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Project { - cols: vec![ProjectItem { + PostAsapOperatorPayload::Relational { + operator: ValueOperation::Project { + cols: vec![planner_types::ir::export::WireProjectItem { alias: None, - expr: QueryExpr::Arithmetic { + expr: planner_types::ir::export::WireScalarExpr::Arithmetic { + semantics: planner_types::ir::ExprSemantics::Sql, op: ArithmeticOpKind::Add, - left: Rc::new(QueryExpr::Column(0)), - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.))), + left: Box::new(planner_types::ir::export::WireScalarExpr::Column( + 0, + )), + right: Box::new( + planner_types::ir::export::WireScalarExpr::Literal( + ScalarValue::Float64(2.), + ), + ), }, }], qualifier: None, @@ -549,31 +571,29 @@ fn bind_post_asap_before_execution() { // A literal Fallback needs no deployment input. let literal = bind(&dag, BTreeMap::new(), &[1]).unwrap(); assert_eq!(floats(&run(&literal, 1, query()), 0), vec![3.]); - dag.nodes[1].payload = PostAsapOperatorPayload::Value { - operation: ValueOperation::Extension { - name: "unknown".into(), - }, + dag.nodes[1].payload = PostAsapOperatorPayload::Extension { + name: "unsupported".into(), }; assert!(bind(&dag, sources(), &[1]).is_err()); } // A completed empty population has an exact zero count, with integer output. #[test] -fn empty_exact_count_is_an_integer_state_readout() { +fn empty_exact_count_is_an_integer_state_evaluation() { let input = schema(&[("value", DataType::Float64, false)]); let build = Operator::summary_build( input.clone(), - SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count), + FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), 0, None, vec![], ) .unwrap(); - let read = Operator::readout( + let read = Operator::evaluation( build.schema(), 0, - asap_physical_operators::operators::ReadoutQuery::Exact( - asap_physical_operators::summary_kernels::exact::ExactReadout { + asap_physical_operators::operators::SummaryEvaluation::Exact( + asap_physical_operators::summary_kernels::exact::ExactEvaluation { statistic: Statistic::Count, lookback_ms: None, }, @@ -592,13 +612,7 @@ fn empty_exact_count_is_an_integer_state_readout() { #[test] fn source_batches_must_match_the_bound_schema() { use asap_physical_operators::dag::{self, PhysicalOperator}; - use planner_types::{ - post_asap::{ - ExecutionDataState, PostAsapDag, PostAsapDagNode, PostAsapNodeId, - PostAsapOperatorPayload, - }, - pre_asap::QueryExpr, - }; + use planner_types::post_asap::ExecutionDataState; use std::{cell::Cell, collections::BTreeMap, rc::Rc}; struct WrongSource { schema: Schema, @@ -634,8 +648,13 @@ fn source_batches_must_match_the_bound_schema() { let plan = PostAsapDag { nodes: vec![PostAsapDagNode { id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(1.), + payload: PostAsapOperatorPayload::Relational { + operator: ValueOperation::Values { + rows: vec![vec![planner_types::ir::export::WireScalarExpr::Literal( + planner_types::pre_asap::ScalarValue::Float64(1.), + )]], + schema: (*expected).clone(), + }, }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*expected).clone(), @@ -703,67 +722,87 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { use asap_physical_operators::dag::planner::{bind, Source}; use planner_types::{ post_asap::*, - pre_asap::{CompareOpKind, GroupKeys, JoinKind, Predicate, QueryExpr, SortKey}, + pre_asap::{CompareOpKind, GroupKeys, JoinKind}, }; - use std::{collections::BTreeMap, rc::Rc}; + use std::collections::BTreeMap; let rows_schema = schema(&[ ("group", DataType::Utf8, false), ("key", DataType::Utf8, false), ("score", DataType::Float64, false), ]); let keys_schema = schema(&[("key", DataType::Utf8, false)]); - let node = |id, payload, schema: &Schema| PostAsapDagNode { + let node = |id, payload, schema: &asap_physical_operators::values::Schema| PostAsapDagNode { id: PostAsapNodeId(id), payload, output_schema: (**schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, guarantee: None, }; - let edge = |producer, consumer, role, schema: &Schema| PostAsapDagEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), - role, - intermediate_schema: (**schema).clone(), - data_state: ExecutionDataState::QUERY_ROWS, - grouping: GroupingEdgeCompatibility::NotApplicable, - window: WindowEdgeCompatibility::NotApplicable, + let edge = |producer, consumer, role, schema: &asap_physical_operators::values::Schema| { + PostAsapDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role, + intermediate_schema: (**schema).clone(), + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + } }; let groups = GroupKeys::by(vec![0]); let dag = PostAsapDag { nodes: vec![ node( 0, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(0.), + PostAsapOperatorPayload::Relational { + operator: ValueOperation::Values { + rows: vec![vec![planner_types::ir::export::WireScalarExpr::Literal( + planner_types::pre_asap::ScalarValue::Float64(0.), + )]], + schema: (*rows_schema).clone(), + }, }, &rows_schema, ), node( 1, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(0.), + PostAsapOperatorPayload::Relational { + operator: ValueOperation::Values { + rows: vec![vec![planner_types::ir::export::WireScalarExpr::Literal( + planner_types::pre_asap::ScalarValue::Float64(0.), + )]], + schema: (*keys_schema).clone(), + }, }, &keys_schema, ), node( 2, - PostAsapOperatorPayload::RelationalJoin { - join_kind: JoinKind::Semi, - pruning: None, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(1)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(3)), - })), + PostAsapOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::Join { + join_kind: JoinKind::Semi, + pred: planner_types::ir::export::WirePredicate( + planner_types::ir::export::WireScalarExpr::Compare { + left: Box::new(planner_types::ir::export::WireScalarExpr::Column( + 1, + )), + op: CompareOpKind::Eq, + right: Box::new(planner_types::ir::export::WireScalarExpr::Column( + 3, + )), + semantics: planner_types::ir::ExprSemantics::Sql, + }, + ), + }, }, &rows_schema, ), node( 3, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Sort { - keys: vec![SortKey { - expr: QueryExpr::Column(2), + PostAsapOperatorPayload::Relational { + operator: ValueOperation::Sort { + keys: vec![planner_types::ir::export::WireSortKey { + expr: planner_types::ir::export::WireScalarExpr::Column(2), ascending: false, nulls_first: false, }], @@ -774,9 +813,9 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { ), node( 4, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Limit { - n: 1, + PostAsapOperatorPayload::Relational { + operator: ValueOperation::Limit { + n: Some(1), offset: 0, partition_by: groups, }, @@ -854,8 +893,8 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { #[test] fn planner_expressions_preserve_collection_and_nullable_types() { use asap_physical_operators::dag::expressions::CompiledExpression; - use planner_types::pre_asap::{CompareOpKind, QueryExpr, ScalarValue}; - use std::rc::Rc; + use planner_types::pre_asap::{CompareOpKind, ScalarValue}; + let input_schema = schema(&[( "items", DataType::Map { @@ -903,9 +942,10 @@ fn planner_expressions_preserve_collection_and_nullable_types() { let projected = project.schema(); dag.add(1, vec![0], project).unwrap(); let predicate = QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(QueryExpr::Column(0)), op: CompareOpKind::Ge, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + right: Box::new(QueryExpr::Literal(ScalarValue::Int64(1))), }; dag.add( 2, @@ -929,14 +969,16 @@ fn planner_expressions_preserve_collection_and_nullable_types() { // Outer, semi and anti joins share Planner predicates and preserve SQL null behavior. #[test] fn native_relational_join_kinds_preserve_unmatched_rows() { - use planner_types::pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}; - use std::rc::Rc; + use planner_types::ir::Predicate; + use planner_types::pre_asap::{CompareOpKind, JoinKind}; + let input = schema(&[("key", DataType::Int64, true)]); - let predicate = Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let predicate = Predicate(QueryExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(QueryExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), - })); + right: Box::new(QueryExpr::Column(1)), + }); for (kind, count) in [ (JoinKind::Inner, 1), (JoinKind::Left, 3), @@ -1010,6 +1052,7 @@ fn weighted_rate_topk_preserves_partitions_fractional_scores_and_evaluation_scop } fn assert_weighted_rate_topk(count_sketch: bool) { use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let raw = schema(&[ ("service", DataType::Utf8, false), ("job", DataType::Utf8, false), @@ -1047,7 +1090,7 @@ fn assert_weighted_rate_topk(count_sketch: bool) { Some((0, 60_000)), ) .unwrap(); - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new( if count_sketch { SketchAlgorithm::CountSketchWithHeap @@ -1076,7 +1119,7 @@ fn assert_weighted_rate_topk(count_sketch: bool) { ("service", DataType::Utf8, false), ("score", DataType::Float64, false), ]); - let readout = Operator::keyed_readout(build.schema(), 1, 8, output.clone()).unwrap(); + let evaluation = Operator::keyed_evaluation(build.schema(), 1, 8, output.clone()).unwrap(); let mut dag = PhysicalDag::default(); dag.add( 0, @@ -1086,7 +1129,7 @@ fn assert_weighted_rate_topk(count_sketch: bool) { .unwrap(); dag.add(1, vec![0], rates).unwrap(); dag.add(2, vec![1], build).unwrap(); - dag.add(3, vec![2], readout).unwrap(); + dag.add(3, vec![2], evaluation).unwrap(); dag.add( 4, vec![3], @@ -1133,17 +1176,16 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { use asap_physical_operators::physical_planner::{ compile_node, CompiledPhysicalDag, InputContract, Source, }; - use planner_types::post_asap::{ - ExecutionDataState, PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, - ValueOperation, - }; + use planner_types::ir::export::{PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload}; + use planner_types::post_asap::ExecutionDataState; + use planner_types::pre_asap::{ - aggregate_output_schema, AggIntent, Column, GroupKeys, QueryExpr, Reduction as IrReduction, - Schema as IrSchema, + aggregate_output_schema, AggIntent, GroupKeys, Reduction as IrReduction, Schema as IrSchema, }; + let grouped = IrSchema::new(vec![ - Column::new("job", DataType::Utf8, false), - Column::new("sum", DataType::Float64, false), + planner_types::pre_asap::Field::plain("job", DataType::Utf8, false), + planner_types::pre_asap::Field::plain("sum", DataType::Float64, false), ]); let output = aggregate_output_schema( &grouped, @@ -1154,14 +1196,22 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { .unwrap(); let input = schema( &output - .columns + .fields .iter() - .map(|c| (c.name.as_str(), c.dtype.clone(), c.nullable)) + .map(|c| { + ( + c.name.as_str(), + c.plain_dtype().unwrap().clone(), + c.nullable, + ) + }) .collect::>(), ); let node = |id, operation| PostAsapDagNode { id: PostAsapNodeId(id), - payload: PostAsapOperatorPayload::Value { operation }, + payload: PostAsapOperatorPayload::Relational { + operator: operation, + }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*input).clone(), guarantee: None, @@ -1170,8 +1220,8 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { &node( 1, ValueOperation::Sort { - keys: vec![planner_types::pre_asap::SortKey { - expr: QueryExpr::Column(1), + keys: vec![planner_types::ir::export::WireSortKey { + expr: planner_types::ir::export::WireScalarExpr::Column(1), ascending: false, nulls_first: false, }], @@ -1185,7 +1235,7 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { &node( 2, ValueOperation::Limit { - n: 1, + n: Some(1), offset: 0, partition_by: GroupKeys::none(), }, @@ -1239,9 +1289,9 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { }; use planner_types::{ post_asap::*, - pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}, + pre_asap::{CompareOpKind, JoinKind}, }; - use std::{collections::BTreeMap, rc::Rc}; + use std::collections::BTreeMap; let schema = schema(&[("key", DataType::Utf8, false)]); for certified in [false, true] { let node = PostAsapDagNode { @@ -1249,21 +1299,18 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { output_schema: (*schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, guarantee: None, - payload: PostAsapOperatorPayload::RelationalJoin { - join_kind: JoinKind::Semi, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), - })), - pruning: certified.then_some(CandidateCompleteness::Certified { - guarantee: ResultGuarantee { - metric: ErrorMetric::TopKMembership, - bound: BoundExpr::Zero, - failure_probability: ProbabilityExpr::Constant { value: 0.01 }, - provenance: vec![], - }, - }), + payload: PostAsapOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::Join { + join_kind: JoinKind::Semi, + pred: planner_types::ir::export::WirePredicate( + planner_types::ir::export::WireScalarExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(planner_types::ir::export::WireScalarExpr::Column(0)), + op: CompareOpKind::Eq, + right: Box::new(planner_types::ir::export::WireScalarExpr::Column(1)), + }, + ), + }, }, }; let graph = CompiledPhysicalDag::from_operators( @@ -1276,7 +1323,12 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { 2, ( vec![0, 1], - compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(), + if certified { + Operator::certified_semi_join(schema.clone(), schema.clone(), vec![(0, 0)]) + .unwrap() + } else { + compile_node(&node, &[schema.clone(), schema.clone()]).unwrap() + }, ), )] .into(), @@ -1365,12 +1417,15 @@ fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { output_schema: (*input).clone(), output_state: ExecutionDataState::INGESTION_ROWS, guarantee: None, - payload: PostAsapOperatorPayload::Binary { - operator: BinaryOperator { - kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, + payload: PostAsapOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::BinaryOp { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, }, }, }; diff --git a/crates/asap-physical-operators/tests/physical_plan_recovery.rs b/crates/asap-physical-operators/tests/physical_plan_recovery.rs index 820bc6092..f52fb05ca 100644 --- a/crates/asap-physical-operators/tests/physical_plan_recovery.rs +++ b/crates/asap-physical-operators/tests/physical_plan_recovery.rs @@ -5,16 +5,19 @@ use asap_physical_operators::{ physical_planner::{CompiledPhysicalDag, InputContract}, }; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{Field, FieldDataType}, pre_asap::DataType, }; use std::{collections::BTreeMap, sync::Arc}; fn sorted() -> CompiledPhysicalDag { - let schema = Arc::new(SummarySchema { - fields: vec![SummaryField { + let schema = Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, + fields: vec![Field { + table: None, name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, }], time_index: None, diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs index 6460cf8a3..d1353759f 100644 --- a/crates/asap-physical-operators/tests/physical_semantics.rs +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -9,19 +9,26 @@ use asap_physical_operators::{ values::{Batch, Schema, Value}, }; use futures::{executor::block_on, StreamExt}; +use planner_types::ir::export::NonASAPOpKind as ValueOperation; +use planner_types::ir::export::{PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload}; +use planner_types::ir::Predicate; +use planner_types::ir::ScalarExpr as QueryExpr; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, - pre_asap::{CompareOpKind, DataType, JoinKind, Predicate, QueryExpr}, + post_asap::{Field, FieldDataType}, + pre_asap::{CompareOpKind, DataType, JoinKind}, }; -use std::{rc::Rc, sync::Arc}; +use std::sync::Arc; fn schema(fields: &[(&str, DataType, bool)]) -> Schema { - Arc::new(SummarySchema { + Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, fields: fields .iter() - .map(|(name, dtype, nullable)| SummaryField { + .map(|(name, dtype, nullable)| Field { + table: None, name: (*name).into(), - dtype: SummaryFamilyType::Plain(dtype.clone()), + dtype: FieldDataType::Plain(dtype.clone()), nullable: *nullable, }) .collect(), @@ -71,11 +78,12 @@ fn keys(rows: &[Vec]) -> Vec>> { .collect() } fn eq_predicate() -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + Predicate(QueryExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(QueryExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), - })) + right: Box::new(QueryExpr::Column(1)), + }) } fn join(left: Vec, right: Vec, kind: JoinKind, keyed: bool) -> Vec> { let input = schema(&[("key", DataType::Float64, true)]); @@ -342,7 +350,7 @@ fn global_extrema_bind_with_planner_derived_schema() { use asap_physical_operators::physical_planner::compile_node; use planner_types::{ post_asap::*, - pre_asap::{AggIntent, Column, GroupKeys, Reduction as PlanReduction}, + pre_asap::{AggIntent, GroupKeys, Reduction as PlanReduction}, }; let input = schema(&[("v", DataType::Int64, false)]); for measure in [ @@ -350,26 +358,34 @@ fn global_extrema_bind_with_planner_derived_schema() { AggIntent::Max { col: Some(0) }, ] { let planner_input = - planner_types::pre_asap::Schema::new(vec![Column::new("v", DataType::Int64, false)]); - let derived = planner_types::pre_asap::query_expr::aggregate_output_schema( + planner_types::pre_asap::Schema::new(vec![planner_types::pre_asap::Field::plain( + "v", + DataType::Int64, + false, + )]); + let derived = planner_types::pre_asap::aggregate_output_schema( &planner_input, &PlanReduction::Reduce(GroupKeys::by(vec![])), std::slice::from_ref(&measure), &[], ) .unwrap(); - let result = derived.columns[0].clone(); - let output = schema(&[(&result.name, result.dtype, result.nullable)]); + let result = derived.fields[0].clone(); + let output = schema(&[( + &result.name, + result.plain_dtype().unwrap().clone(), + result.nullable, + )]); let node = PostAsapDagNode { id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::Value { - operation: ValueOperation::Exact(ExactOperation::Aggregate { + payload: PostAsapOperatorPayload::Relational { + operator: ValueOperation::Aggregate { reduction: PlanReduction::Reduce(GroupKeys::by(vec![])), measures: vec![measure], output_names: vec![result.name], filters: vec![], having: None, - }), + }, }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*output).clone(), @@ -401,9 +417,10 @@ fn planner_comparisons_handle_nan_without_execution_errors() { CompareOpKind::Ge, ] { let expression = QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(QueryExpr::Column(0)), op: op.clone(), - right: Rc::new(QueryExpr::Column(1)), + right: Box::new(QueryExpr::Column(1)), }; let compiled = CompiledExpression::compile(&expression, &input).unwrap(); for row in [ @@ -468,9 +485,10 @@ fn mixed_numeric_comparisons_preserve_large_integer_precision() { ("b", DataType::Float64, false), ]); let expr = QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(QueryExpr::Column(0)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Column(1)), + right: Box::new(QueryExpr::Column(1)), }; let compiled = CompiledExpression::compile(&expr, &input).unwrap(); for (a, b, expected) in [ @@ -536,10 +554,11 @@ fn boolean_truth_tables_agree_between_expression_paths() { // Partial/final execution must agree with one build for an uncompacted KLL population. #[test] -fn kll_partial_merge_and_multiple_readouts_preserve_population() { +fn kll_partial_merge_and_multiple_evaluations_preserve_population() { use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("v", DataType::Float64, false)]); - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), Default::default(), ); @@ -581,11 +600,11 @@ fn kll_partial_merge_and_multiple_readouts_preserve_population() { dag.add( id, vec![build], - Operator::readout( + Operator::evaluation( state.clone(), 0, - asap_physical_operators::operators::ReadoutQuery::Sketch( - planner_types::post_asap::SketchQuery::Quantile { q }, + asap_physical_operators::operators::SummaryEvaluation::Sketch( + planner_types::post_asap::SketchStatistic::Quantile { q }, ), ) .unwrap(), @@ -655,6 +674,7 @@ fn zero_column_output_obeys_memory_limit() { fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { use asap_physical_operators::Statistic; use planner_types::post_asap::{ExactKind, ExactParams}; + let input = schema(&[("v", DataType::Float64, false)]); for (kind, params, statistic) in [ (ExactKind::Min, ExactParams::Min, Statistic::Min), @@ -662,7 +682,7 @@ fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { ] { let build = Operator::summary_build( input.clone(), - SummaryFamilyType::ExactAggregate(kind, params), + FieldDataType::ExactAggregate(kind, params), 0, None, vec![], @@ -676,11 +696,11 @@ fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { dag.add( 2, vec![1], - Operator::readout( + Operator::evaluation( state, 0, - asap_physical_operators::operators::ReadoutQuery::Exact( - asap_physical_operators::summary_kernels::exact::ExactReadout { + asap_physical_operators::operators::SummaryEvaluation::Exact( + asap_physical_operators::summary_kernels::exact::ExactEvaluation { statistic, lookback_ms: None, }, diff --git a/crates/asap-physical-operators/tests/plan_properties.rs b/crates/asap-physical-operators/tests/plan_properties.rs index f7a468ab8..143979517 100644 --- a/crates/asap-physical-operators/tests/plan_properties.rs +++ b/crates/asap-physical-operators/tests/plan_properties.rs @@ -8,8 +8,8 @@ use asap_physical_operators::{ Error, }; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, - pre_asap::{Column, DataType, QueryExpr, Schema as LogicalSchema, Source}, + post_asap::{Field, FieldDataType}, + pre_asap::{DataType, Schema as LogicalSchema, Source}, }; use std::sync::{ atomic::{AtomicUsize, Ordering}, @@ -35,10 +35,13 @@ impl RawSource for DeclaredSource { // A blocking parent must reject unknown and unbounded Scan inputs without opening a reader. #[test] fn blocking_inputs_require_an_explicit_finite_source() { - let schema = Arc::new(SummarySchema { - fields: vec![SummaryField { + let schema = Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, + fields: vec![Field { + table: None, name: "v".into(), - dtype: SummaryFamilyType::Plain(DataType::Int64), + dtype: FieldDataType::Plain(DataType::Int64), nullable: false, }], time_index: None, @@ -63,13 +66,21 @@ fn blocking_inputs_require_an_explicit_finite_source() { }), ) .unwrap(); - let scan = registry - .bind(&QueryExpr::Scan { - source: identity, - schema: LogicalSchema::new(vec![Column::new("v", DataType::Int64, false)]), - predicates: vec![], - }) - .unwrap(); + let scan = + registry + .bind( + &planner_types::ir::OperatorNode::non_asap_node( + planner_types::ir::NonASAPOp::Scan { + source: identity, + schema: LogicalSchema::new(vec![ + planner_types::pre_asap::Field::plain("v", DataType::Int64, false), + ]), + predicates: vec![], + }, + ) + .unwrap(), + ) + .unwrap(); let mut dag = PhysicalDag::default(); dag.add(0, vec![], scan).unwrap(); dag.add( @@ -109,19 +120,19 @@ fn blocking_inputs_require_an_explicit_finite_source() { } } -// Kernel support must not be mistaken for executable native state/readout support. +// Kernel support must not be mistaken for executable native state/evaluation support. #[test] fn summary_capability_levels_are_distinct() { use asap_physical_operators::{ - capability::{validate_native_family, validate_sketch_readout, validate_summary_kernel}, - planner::post_asap::SketchQuery, + capability::{validate_native_family, validate_sketch_evaluation, validate_summary_kernel}, + planner::post_asap::SketchStatistic, }; use planner_types::{ post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate}, pre_asap::ColumnRef, }; let grouping = GroupingStrategy::default(); - let cms = SummaryFamilyType::Sketch( + let cms = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::Cms, SketchParams::Cms { @@ -141,25 +152,25 @@ fn summary_capability_levels_are_distinct() { assert!(validate_summary_kernel(&cms, &update, &grouping).is_ok()); // Stored Count-Min state reads only its bare count natively. assert!(validate_native_family(&cms).is_ok()); - let bare_count = SketchQuery::PointCount { + let bare_count = SketchStatistic::PointCount { key: ColumnRef::SampleValue, value: None, }; - assert!(validate_sketch_readout(&cms, &bare_count).is_ok()); - assert!(validate_sketch_readout( + assert!(validate_sketch_evaluation(&cms, &bare_count).is_ok()); + assert!(validate_sketch_evaluation( &cms, - &SketchQuery::PointCount { + &SketchStatistic::PointCount { key: ColumnRef::Named("host".into()), value: Some("a".into()), } ) .is_err()); - let kll = SummaryFamilyType::Sketch( + let kll = FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 128 }), grouping, ); assert!(validate_native_family(&kll).is_ok()); - assert!(validate_sketch_readout(&kll, &SketchQuery::Quantile { q: 1.5 }).is_err()); - assert!(validate_sketch_readout(&kll, &SketchQuery::Cardinality).is_err()); - assert!(validate_sketch_readout(&kll, &SketchQuery::Quantile { q: 0.5 }).is_ok()); + assert!(validate_sketch_evaluation(&kll, &SketchStatistic::Quantile { q: 1.5 }).is_err()); + assert!(validate_sketch_evaluation(&kll, &SketchStatistic::Cardinality).is_err()); + assert!(validate_sketch_evaluation(&kll, &SketchStatistic::Quantile { q: 0.5 }).is_ok()); } diff --git a/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs b/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs index 2f8d5249f..8c8ca7818 100644 --- a/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs +++ b/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs @@ -2,6 +2,10 @@ //! Planner's search space: `enumerate_candidate_dags_for_root` lists //! current-series TopK heaps without a caller-side series-identity pass, cost //! ranking, or workload Cartesian expansion. Placement variants are not listed. +mod common; +use common::compile_post_asap_dag; +use planner_types::ir::OperatorNode as QueryExpr; + use asap_aware_mapping::{ accuracy::{AccuracyEvidenceProvider, DefaultAccuracyModel, PropagationStats}, cost_model::DefaultCostModel, @@ -9,11 +13,10 @@ use asap_aware_mapping::{ search_workload_with_targets, Proposals, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_physical_operators::physical_planner::promql_rows::{ - compile_current_series_readout, SERIES_IDENTITY_COLUMN, + compile_current_series_evaluation, SERIES_IDENTITY_COLUMN, }; use planner_types::{ post_asap::*, - pre_asap::QueryExpr, types::AccuracyTarget, workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence as WorkloadEvidence, @@ -31,8 +34,8 @@ impl AccuracyEvidenceProvider for Evidence { fn propagation_stats( &self, op: &CompositionOperator, - _: &SummaryFamilyType, - _: Option<&SketchQuery>, + _: &FieldDataType, + _: Option<&SketchStatistic>, ) -> PropagationStats { if matches!(op, CompositionOperator::TopKSelection) { PropagationStats { @@ -89,14 +92,12 @@ fn lower(query: &str, accuracy: &AccuracyTarget) -> Rc { ..Default::default() }), }; - Rc::new( - asap_frontend_promql::lower_promql_workload(&workload, 0) - .unwrap() - .remove(0), - ) + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0) } -type Dag = Vec<(usize, Rc)>; +type Dag = Vec<(usize, Rc)>; /// Candidate DAGs for query 1 of a two-query workload, with and without /// whole-root proposals. Query 0 is a bystander that must not multiply them. @@ -140,7 +141,10 @@ fn carries_identity(dag: &Dag) -> bool { } /// Shared acceptance checks; returns the added identity-carrying alternatives. -fn added_alternatives(query: &str, accuracy: AccuracyTarget) -> Vec> { +fn added_alternatives( + query: &str, + accuracy: AccuracyTarget, +) -> Vec> { let (full, logical) = inventories(query, accuracy); for (index, dag) in full.iter().enumerate() { assert_eq!(dag.len(), 1, "one root per candidate, no workload product"); @@ -159,15 +163,18 @@ fn added_alternatives(query: &str, accuracy: AccuracyTarget) -> Vec asap_aware_mapping::CandidateLogicalASAPDAGs<&'static str> { let workload = PlanningWorkload { @@ -40,15 +43,11 @@ fn grouped_rate_space() -> asap_aware_mapping::CandidateLogicalASAPDAGs<&'static ..Default::default() }), }; - let root = Rc::new( - asap_frontend_promql::lower_promql_workload(&workload, 0) - .unwrap() - .remove(0), - ); - let root = Rc::new( - asap_physical_operators::physical_planner::promql_rows::with_series_identity(&root) - .unwrap(), - ); + let root = asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0); + let root = asap_physical_operators::physical_planner::promql_rows::with_series_identity(&root) + .unwrap(); search_workload(vec![("grouped-rate", root)]) } @@ -81,7 +80,7 @@ fn run(plan: &CompiledPhysicalDag, inputs: BTreeMap, scope: Scope) - }) } -/// Rate readouts and grouped Sum can run together during bounded precompute; +/// Rate evaluations and grouped Sum can run together during bounded precompute; /// storing per-series rates instead leaves the same Sum in the query DAG. #[test] fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { @@ -93,21 +92,19 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { matches!( node.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. } ) }) .unwrap(); - let readout = dag + let evaluation = dag .nodes .iter() .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator - } + PostAsapOperatorPayload::FinalizeExactAccumulator ) && dag .edges .iter() @@ -139,7 +136,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .as_any() .downcast_ref::() .unwrap() - .readout(asap_physical_operators::Statistic::Rate, range_ms, None) + .evaluation(asap_physical_operators::Statistic::Rate, range_ms, None) .unwrap() .unwrap(); let summary = Value::Summary { @@ -150,21 +147,19 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .fields .iter() .map(|field| match &field.dtype { - SummaryFamilyType::ExactAggregate(..) => summary.clone(), - SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(2000), - SummaryFamilyType::Plain(DataType::Utf8) => { - Value::Utf8(if field.name == "job" { - "api".into() - } else { - serde_json::to_string(&BTreeMap::from([ - ("__name__", "m".to_string()), - ("job", "api".to_string()), - ("instance", format!("series-{index}")), - ])) - .unwrap() - .into() - }) - } + FieldDataType::ExactAggregate(..) => summary.clone(), + FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(2000), + FieldDataType::Plain(DataType::Utf8) => Value::Utf8(if field.name == "job" { + "api".into() + } else { + serde_json::to_string(&BTreeMap::from([ + ("__name__", "m".to_string()), + ("job", "api".to_string()), + ("instance", format!("series-{index}")), + ])) + .unwrap() + .into() + }), _ => panic!("unexpected input field {field:?}"), }) .collect() @@ -173,7 +168,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { let batch = Batch::try_new(input_schema.clone(), rows).unwrap(); let root = u64::from(dag.root.0); let state_id = u64::from(state.id.0); - let rate_id = u64::from(readout.id.0); + let rate_id = u64::from(evaluation.id.0); let frontiers = asap_physical_operators::physical_planner::enumerate_frontiers( &dag, &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), @@ -369,7 +364,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .as_any() .downcast_ref::() .unwrap() - .readout(asap_physical_operators::Statistic::Rate, range_ms, None) + .evaluation(asap_physical_operators::Statistic::Rate, range_ms, None) .unwrap() .unwrap(); assert_ne!( @@ -378,7 +373,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { ); } -/// Enumerated frontiers include both grouped-result and per-series readout +/// Enumerated frontiers include both grouped-result and per-series evaluation /// persistence; an explicit Rate-state input retains its original semantics. #[test] fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { @@ -391,7 +386,7 @@ fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { matches!( &node.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. } ) @@ -429,7 +424,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { matches!( node.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. } ) @@ -443,7 +438,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { matches!( node.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + family: FieldDataType::ExactAggregate(ExactKind::Sum, _), .. } ) @@ -484,9 +479,9 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { .fields .iter() .map(|field| match &field.dtype { - SummaryFamilyType::ExactAggregate(..) => summary.clone(), - SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), - SummaryFamilyType::Plain(DataType::Utf8) + FieldDataType::ExactAggregate(..) => summary.clone(), + FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), + FieldDataType::Plain(DataType::Utf8) if field.name == "$promql_series_identity" => { Value::Utf8( @@ -498,7 +493,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { .into(), ) } - SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8("api".into()), + FieldDataType::Plain(DataType::Utf8) => Value::Utf8("api".into()), _ => panic!("unexpected input field {field:?}"), }) .collect() @@ -540,7 +535,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { .schema() .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); assert!( output[0].rows()[0] .iter() @@ -657,10 +652,9 @@ fn population_topk_cuts_equal_per_frontier_compilation() { let original = asap_frontend_promql::lower_promql_workload(&workload, 0) .unwrap() .remove(0); - let root = Rc::new( + let root = asap_physical_operators::physical_planner::promql_rows::with_series_identity(&original) - .unwrap(), - ); + .unwrap(); let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( std::slice::from_ref(&root), ) @@ -670,7 +664,14 @@ fn population_topk_cuts_equal_per_frontier_compilation() { let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::TimeRange { .. } + } + ) + }) .unwrap(); let inputs = BTreeMap::from([( u64::from(raw.id.0), @@ -698,21 +699,19 @@ fn cut_candidate_rejects_invalid_frontiers() { .iter() .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) .unwrap(); - let readout = dag + let evaluation = dag .nodes .iter() .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator - } + PostAsapOperatorPayload::FinalizeExactAccumulator ) }) .unwrap(); let (state_id, rate_id, root) = ( u64::from(state.id.0), - u64::from(readout.id.0), + u64::from(evaluation.id.0), u64::from(dag.root.0), ); let inputs = BTreeMap::from([( diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs index fa124c686..75a5dcafa 100644 --- a/crates/asap-physical-operators/tests/precompute_population.rs +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -8,6 +8,11 @@ use asap_physical_operators::{ Statistic, }; use futures::{executor::block_on, StreamExt}; +use planner_types::ir::export::{ + EdgeRole, GroupingEdgeCompatibility, PostAsapDag, PostAsapDagEdge, PostAsapDagNode, + PostAsapNodeId, PostAsapOperatorPayload, WindowEdgeCompatibility, +}; +use planner_types::ir::BinaryOperator; use planner_types::{ post_asap::*, pre_asap::{ArithmeticOpKind, BinaryOpKind, ColumnRef, DataType, GroupKeys, Reduction}, @@ -17,9 +22,12 @@ use std::{collections::BTreeMap, sync::Arc}; // Typed series identity survives finalization and derived precompute through population metadata. #[test] fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let schema = |dtype| SummarySchema { - fields: vec![SummaryField { + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let schema = |dtype| planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, + fields: vec![Field { + table: None, name: "value".into(), dtype, nullable: false, @@ -27,16 +35,18 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { time_index: None, }; let state_schema = schema(family.clone()); - let mut value_schema = schema(SummaryFamilyType::Plain(DataType::Float64)); - value_schema.fields.push(SummaryField { + let mut value_schema = schema(FieldDataType::Plain(DataType::Float64)); + value_schema.fields.push(Field { + table: None, name: "time".into(), - dtype: SummaryFamilyType::Plain(DataType::Timestamp), + dtype: FieldDataType::Plain(DataType::Timestamp), nullable: false, }); value_schema.time_index = Some(1); - value_schema.fields.push(SummaryField { + value_schema.fields.push(Field { + table: None, name: planner_types::pre_asap::schema::PROMQL_SERIES_IDENTITY.into(), - dtype: SummaryFamilyType::Plain(DataType::Utf8), + dtype: FieldDataType::Plain(DataType::Utf8), nullable: false, }); for (weight, expected) in [ @@ -53,21 +63,22 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { }, PostAsapDagNode { id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - }, + payload: PostAsapOperatorPayload::FinalizeExactAccumulator, output_state: ExecutionDataState::INGESTION_ROWS, output_schema: value_schema.clone(), guarantee: None, }, PostAsapDagNode { id: PostAsapNodeId(2), - payload: PostAsapOperatorPayload::Binary { - operator: BinaryOperator { - kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, + payload: PostAsapOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::BinaryOp { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, }, }, output_state: ExecutionDataState::INGESTION_ROWS, @@ -119,7 +130,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { let fields = &mut invalid_identity.nodes[1].output_schema.fields; match mutation { 0 => fields[2].nullable = true, - 1 => fields[2].dtype = SummaryFamilyType::Plain(DataType::Float64), + 1 => fields[2].dtype = FieldDataType::Plain(DataType::Float64), _ => fields.push(fields[2].clone()), } assert!(precompute::compile(&invalid_identity, &[0], &[3]).is_err()); @@ -202,7 +213,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { .as_any() .downcast_ref::() .unwrap() - .readout(Statistic::Sum, None, None) + .evaluation(Statistic::Sum, None, None) .unwrap() .unwrap(), expected @@ -211,9 +222,12 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { } } -fn logical_schema(family: SummaryFamilyType) -> SummarySchema { - SummarySchema { - fields: vec![SummaryField { +fn logical_schema(family: FieldDataType) -> Schema { + planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, + fields: vec![Field { + table: None, name: "value".into(), dtype: family, nullable: false, @@ -222,8 +236,8 @@ fn logical_schema(family: SummaryFamilyType) -> SummarySchema { } } fn state_graph( - family: SummaryFamilyType, - target: Option, + family: FieldDataType, + target: Option, merge: bool, ) -> CompiledPhysicalDag { let mut nodes = vec![PostAsapDagNode { @@ -243,11 +257,9 @@ fn state_graph( let read_id = nodes.len() as u32; nodes.push(PostAsapDagNode { id: PostAsapNodeId(read_id), - payload: PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - }, + payload: PostAsapOperatorPayload::FinalizeExactAccumulator, output_state: ExecutionDataState::INGESTION_ROWS, - output_schema: logical_schema(SummaryFamilyType::Plain(DataType::Float64)), + output_schema: logical_schema(FieldDataType::Plain(DataType::Float64)), guarantee: None, }); if let Some(target) = target { @@ -286,7 +298,7 @@ fn state_graph( } fn native_run( program: &CompiledPhysicalDag, - family: SummaryFamilyType, + family: FieldDataType, states: Vec>, context: RunContext, ) -> Result>, asap_physical_operators::Error> { @@ -334,7 +346,7 @@ fn ingestion_context(limits: Limits) -> RunContext { } fn sum_state(value: f64) -> Arc { let mut state = asap_physical_operators::summary_kernels::exact::ExactAccumulator::new( - planner_types::post_asap::SummaryFamilyType::ExactAggregate( + planner_types::post_asap::FieldDataType::ExactAggregate( planner_types::post_asap::ExactKind::Sum, planner_types::post_asap::ExactParams::Sum, ), @@ -348,7 +360,7 @@ fn sum_state(value: f64) -> Arc { // Only an explicit merge may collapse distinct pane updates before finalization. #[test] fn explicit_merge_changes_pane_cardinality() { - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); for (merge, expected) in [(false, vec![2., 7.]), (true, vec![9.])] { let program = state_graph(family.clone(), None, merge); let rows = native_run( @@ -362,7 +374,7 @@ fn explicit_merge_changes_pane_cardinality() { .iter() .map(|row| match row[2] { Value::Float64(v) => v, - _ => panic!("numeric readout expected"), + _ => panic!("numeric evaluation expected"), }) .collect::>(); assert_eq!(values, expected); @@ -373,8 +385,8 @@ fn explicit_merge_changes_pane_cardinality() { // Typed updates reject invalid domains before publishing any target state. #[test] fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { - let source = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let target = SummaryFamilyType::Sketch( + let source = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let target = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::DDSketch, SketchParams::DDSketch { alpha: 0.01 }, @@ -404,7 +416,7 @@ fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { #[test] fn precompute_count_conversion_checks_precision() { use asap_physical_operators::summary_kernels::exact::ExactAccumulator; - let family = SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count); + let family = FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count); let program = state_graph(family.clone(), None, false); for (count, valid) in [(3u64, true), ((1u64 << 53) + 1, false)] { let mut state = @@ -424,7 +436,7 @@ fn precompute_count_conversion_checks_precision() { // Graph execution retains terminal cancellation and shared workspace limits. #[test] fn precompute_graph_enforces_cancellation_and_budget() { - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let program = state_graph(family.clone(), None, true); let context = ingestion_context(Limits::default()); context.cancel(); diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs index 822bde4ea..447875d80 100644 --- a/crates/asap-physical-operators/tests/promql_binary.rs +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -6,30 +6,36 @@ use asap_physical_operators::{ values::{Batch, Schema, Value}, }; use futures::{executor::block_on, StreamExt}; +use planner_types::ir::export::{ + EdgeRole, GroupingEdgeCompatibility, PostAsapDag, PostAsapDagEdge, PostAsapDagNode, + PostAsapNodeId, PostAsapOperatorPayload, WindowEdgeCompatibility, +}; +use planner_types::ir::BinaryOperator; use planner_types::{ - post_asap::{ - BinaryOperator, ExecutionDataState, PostAsapDagNode, PostAsapNodeId, - PostAsapOperatorPayload, SummaryFamilyType, SummaryField, SummarySchema, - }, + post_asap::{ExecutionDataState, Field, FieldDataType}, pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, }; use std::{collections::BTreeMap, sync::Arc}; fn schema() -> Schema { - Arc::new(SummarySchema { + Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, fields: vec![ - SummaryField { + Field { + table: None, name: "labels".into(), - dtype: SummaryFamilyType::Plain(DataType::Map { + dtype: FieldDataType::Plain(DataType::Map { key: Box::new(DataType::Utf8), value: Box::new(DataType::Utf8), value_nullable: false, }), nullable: false, }, - SummaryField { + Field { + table: None, name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, }, ], @@ -57,10 +63,18 @@ fn program() -> CompiledPhysicalDag { }) } fn program_for(operator: BinaryOperator) -> CompiledPhysicalDag { + program_for_bool(operator, false) +} +fn program_for_bool(operator: BinaryOperator, return_bool: bool) -> CompiledPhysicalDag { let schema = schema(); let node = PostAsapDagNode { id: PostAsapNodeId(2), - payload: PostAsapOperatorPayload::Binary { operator }, + payload: PostAsapOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::BinaryOp { + operator, + return_bool, + }, + }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*schema).clone(), guarantee: None, @@ -151,12 +165,15 @@ fn scalar_broadcast_and_bool_comparison_are_distinct() { use planner_types::pre_asap::CompareOpKind; for return_bool in [false, true] { let graph = promql_values::compile_binary( - &BinaryOperator { - kind: BinaryOpKind::Compare(CompareOpKind::Lt), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }, + &asap_physical_operators::expressions::binary::BinaryOperator::from_logical( + &BinaryOperator { + kind: BinaryOpKind::Compare(CompareOpKind::Lt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + ), return_bool, true, false, @@ -279,12 +296,15 @@ fn binary_obeys_memory_and_cancellation() { // A `bool` comparison over label-map vectors yields 1 or 0 and drops the name. #[test] fn label_map_bool_comparison_drops_the_name() { - let program = program_for(BinaryOperator { - kind: BinaryOpKind::CompareBool(planner_types::pre_asap::CompareOpKind::Gt), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }); + let program = program_for_bool( + BinaryOperator { + kind: BinaryOpKind::Compare(planner_types::pre_asap::CompareOpKind::Gt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + true, + ); let rows = evaluate_with( program, vec![row("a", "api", 6.)], @@ -303,9 +323,9 @@ fn label_map_bool_comparison_drops_the_name() { assert!(matches!(row[1], Value::Float64(v) if v == 1.)); } -// Stored temporal readouts drop metric names before filter comparisons and set matching. +// Stored temporal evaluations drop metric names before filter comparisons and set matching. #[test] -fn stored_series_readouts_support_filters_and_sets() { +fn stored_series_evaluations_support_filters_and_sets() { use asap_physical_operators::{ physical_planner::compile, summary_kernels::exact::ExactAccumulator, }; @@ -313,19 +333,24 @@ fn stored_series_readouts_support_filters_and_sets() { use planner_types::pre_asap::{ schema::PROMQL_SERIES_IDENTITY, CompareOpKind, PromQLVectorSetOpKind, }; + for (exact_kind, params) in [ (ExactKind::Sum, ExactParams::Sum), (ExactKind::Count, ExactParams::Count), ] { - let family = SummaryFamilyType::ExactAggregate(exact_kind.clone(), params); - let state_schema = Arc::new(SummarySchema { + let family = FieldDataType::ExactAggregate(exact_kind.clone(), params); + let state_schema = Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, fields: vec![ - SummaryField { + Field { + table: None, name: PROMQL_SERIES_IDENTITY.into(), - dtype: SummaryFamilyType::Plain(DataType::Utf8), + dtype: FieldDataType::Plain(DataType::Utf8), nullable: false, }, - SummaryField { + Field { + table: None, name: "value".into(), dtype: family.clone(), nullable: false, @@ -334,7 +359,7 @@ fn stored_series_readouts_support_filters_and_sets() { time_index: None, }); let mut value_schema = (*state_schema).clone(); - value_schema.fields[1].dtype = SummaryFamilyType::Plain(DataType::Float64); + value_schema.fields[1].dtype = FieldDataType::Plain(DataType::Float64); for kind in [ BinaryOpKind::Compare(CompareOpKind::Gt), BinaryOpKind::Set(PromQLVectorSetOpKind::And), @@ -345,15 +370,16 @@ fn stored_series_readouts_support_filters_and_sets() { id: PostAsapNodeId(id), payload: match id { 0 | 1 => PostAsapOperatorPayload::SummaryMerge, - 2 | 3 => PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - }, - _ => PostAsapOperatorPayload::Binary { - operator: BinaryOperator { - kind: kind.clone(), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, + 2 | 3 => PostAsapOperatorPayload::FinalizeExactAccumulator, + _ => PostAsapOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::BinaryOp { + operator: BinaryOperator { + kind: kind.clone(), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, }, }, }, diff --git a/crates/asap-physical-operators/tests/promql_fallback.rs b/crates/asap-physical-operators/tests/promql_fallback.rs index 88fe3690f..4475d7e35 100644 --- a/crates/asap-physical-operators/tests/promql_fallback.rs +++ b/crates/asap-physical-operators/tests/promql_fallback.rs @@ -1,27 +1,34 @@ -//! A retained PromQL subtree (`Fallback`) compiles from its typed expression. +//! A retained PromQL sub-DAG (`Fallback`) compiles from its typed expression. //! The deployment supplies only its selector's raw series; expected values are //! hand-computed with Prometheus semantics. +mod common; use asap_physical_operators::{ operators::Operator, physical_planner::{compile, promql_fallback, promql_rows, CompiledPhysicalDag, InputContract}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; +use common::compile_post_asap_dag; use futures::{executor::block_on, StreamExt}; +use planner_types::ir::export::PostAsapDag; use planner_types::{ - post_asap::{execution_data_state::lift_plain, *}, - pre_asap::QueryExpr, - types::AccuracyTarget, - workload::*, + post_asap::execution_data_state::lift_plain, types::AccuracyTarget, workload::*, }; use std::{collections::BTreeMap, rc::Rc}; /// Bare selectors look back one ingestion interval: 60s. -fn parse(query: &str) -> QueryExpr { +fn parse(query: &str) -> Rc { parse_with(query, AccuracyTarget::Exact) } -fn parse_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { +fn parse_with(query: &str, accuracy: AccuracyTarget) -> Rc { + match parse_root(query, accuracy) { + planner_types::ir::QueryRoot::Operator(node) => node, + _ => panic!("expected operator query"), + } +} + +fn parse_root(query: &str, accuracy: AccuracyTarget) -> planner_types::ir::QueryRoot { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -46,24 +53,18 @@ fn parse_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { ..Default::default() }), }; - asap_frontend_promql::lower_promql_workload(&workload, 0) + asap_frontend_promql::lower_promql_query_workload(&workload, 0) .unwrap() .remove(0) } -fn lower(query: &str) -> QueryExpr { +fn lower(query: &str) -> Rc { promql_rows::with_series_identity(&parse(query)).unwrap() } /// The whole query retained as one pre-ASAP node. -fn fallback_dag(expression: QueryExpr) -> PostAsapDag { - let schema = lift_plain(&expression.output_schema().unwrap()); - compile_post_asap_dag(&Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(expression)), - schema, - guarantee: None, - })) - .unwrap() +fn fallback_dag(expression: Rc) -> PostAsapDag { + compile_post_asap_dag(&expression).unwrap() } /// `(labels, seconds, value)`. `labels` is `k=v,...`, or a bare `job` value. @@ -82,13 +83,14 @@ fn labels(spec: &str) -> BTreeMap { } /// The metric a selector reads. -fn metric(selector: &QueryExpr) -> String { - match selector { - QueryExpr::Scan { +fn metric(selector: &planner_types::ir::OperatorNode) -> String { + match selector.expect_non_asap() { + planner_types::ir::NonASAPOp::Scan { source: planner_types::pre_asap::Source::TimeSeries { metric }, .. } => metric.clone(), - QueryExpr::TimeRange { child, .. } | QueryExpr::TimeShift { child, .. } => metric(child), + planner_types::ir::NonASAPOp::TimeRange { child, .. } + | planner_types::ir::NonASAPOp::TimeShift { child, .. } => metric(child), other => panic!("not a selector: {other:?}"), } } @@ -99,7 +101,10 @@ fn compile_query(query: &str) -> Result { } /// Compile a DAG whose root is the Fallback computing `expression`. -fn compile_dag(expression: &QueryExpr, dag: &PostAsapDag) -> Result { +fn compile_dag( + expression: &planner_types::ir::OperatorNode, + dag: &PostAsapDag, +) -> Result { let root = u64::from(dag.root.0); let inputs = promql_fallback::raw_series(expression) .map_err(|e| e.to_string())? @@ -124,13 +129,25 @@ fn evaluate( metrics: &[(&str, &[Sample])], at: i64, ) -> Result, i64, f64)>, String> { - let expression = lower(query); - evaluate_dag(&expression, &fallback_dag(expression.clone()), metrics, at) + match parse_root(query, AccuracyTarget::Exact) { + planner_types::ir::QueryRoot::Operator(expression) => { + let expression = + promql_rows::with_series_identity(&expression).map_err(|e| e.to_string())?; + evaluate_dag(&expression, &fallback_dag(expression.clone()), metrics, at) + } + planner_types::ir::QueryRoot::Scalar(expr) => { + let expr = expr + .map_operator_refs(&mut |node| promql_rows::with_series_identity(node).unwrap()); + let (program, selectors) = + promql_fallback::compile_scalar_root(&expr).map_err(|e| e.to_string())?; + execute_program(program, selectors, metrics, at, None) + } + } } #[allow(clippy::type_complexity)] fn evaluate_dag( - expression: &QueryExpr, + expression: &planner_types::ir::OperatorNode, dag: &PostAsapDag, metrics: &[(&str, &[Sample])], at: i64, @@ -140,15 +157,26 @@ fn evaluate_dag( #[allow(clippy::type_complexity)] fn evaluate_dag_with_range( - expression: &QueryExpr, + expression: &planner_types::ir::OperatorNode, dag: &PostAsapDag, metrics: &[(&str, &[Sample])], at: i64, bounds: Option<(i64, i64)>, ) -> Result, i64, f64)>, String> { let program = compile_dag(expression, dag)?; - let mut sources = BTreeMap::new(); let selectors = promql_fallback::raw_series(expression).unwrap(); + execute_program(program, selectors, metrics, at, bounds) +} + +#[allow(clippy::type_complexity)] +fn execute_program( + program: CompiledPhysicalDag, + selectors: Vec, + metrics: &[(&str, &[Sample])], + at: i64, + bounds: Option<(i64, i64)>, +) -> Result, i64, f64)>, String> { + let mut sources = BTreeMap::new(); for (i, (selector, schema)) in selectors.into_iter().enumerate() { let name = metric(&selector); let rows = metrics @@ -428,7 +456,10 @@ fn raw_series_contract_is_explicit() { .unwrap() .try_into() .unwrap(); - assert!(matches!(selector, QueryExpr::TimeRange { .. })); + assert!(matches!( + selector.expect_non_asap(), + planner_types::ir::NonASAPOp::TimeRange { .. } + )); let missing = compile(&dag, BTreeMap::new(), &[root]).err().unwrap(); assert!(missing.to_string().contains("raw series input")); let mut wrong = (*schema).clone(); @@ -445,54 +476,26 @@ fn raw_series_contract_is_explicit() { // A consumed bare selector is raw range rows for its consumer; it is not // turned into instant selection. let selector = lower("m"); - let schema = lift_plain(&selector.output_schema().unwrap()); - let node = |id, payload| PostAsapDagNode { - id: PostAsapNodeId(id), - payload, - output_state: ExecutionDataState::QUERY_ROWS, - output_schema: schema.clone(), - guarantee: None, - }; - let consumed = PostAsapDag { - nodes: vec![ - node( - 0, - PostAsapOperatorPayload::Fallback { - expression: selector.clone(), - }, - ), - node( - 1, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Limit { - n: 1, - offset: 0, - partition_by: Default::default(), - }, - }, - ), - ], - edges: vec![PostAsapDagEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), - role: EdgeRole::Input, - intermediate_schema: schema.clone(), - data_state: ExecutionDataState::QUERY_ROWS, - grouping: GroupingEdgeCompatibility::NotApplicable, - window: WindowEdgeCompatibility::NotApplicable, - }], - root: PostAsapNodeId(1), - }; + let _schema = lift_plain(&selector.schema.clone()); + let consumed = + planner_types::ir::OperatorNode::non_asap_node(planner_types::ir::NonASAPOp::Limit { + n: Some(1), + offset: 0, + partition_by: Default::default(), + child: selector.clone(), + }) + .unwrap(); + let consumed = fallback_dag(consumed.clone()); let raw = promql_fallback::raw_series(&selector).unwrap().remove(0).1; assert!(compile( &consumed, BTreeMap::from([( - promql_fallback::raw_series_input(0, 0), + promql_fallback::raw_series_input(u64::from(consumed.root.0), 0), InputContract::bounded(raw) )]), - &[1], + &[u64::from(consumed.root.0)] ) - .is_err()); + .is_ok()); // Implicit subquery resolution belongs to the deployment's evaluation interval. assert!(compile_query("max_over_time(m[5m:])").is_err()); } @@ -1231,9 +1234,8 @@ fn histogram_quantile_selection_keeps_the_exact_fallback() { "histogram_quantile(0.5, x_bucket)", "histogram_quantile(0.5, sum by (le, job) (x_bucket))", ] { - let root = Rc::new( - promql_rows::with_series_identity(&parse_with(query, target.clone())).unwrap(), - ); + let root = + promql_rows::with_series_identity(&parse_with(query, target.clone())).unwrap(); let space = search_workload_with_targets( vec![(query, root.clone(), Some(target.clone()))], &default_strategies(), @@ -1243,8 +1245,7 @@ fn histogram_quantile_selection_keeps_the_exact_fallback() { let candidates = &space.candidates_for_target(planned).unwrap().candidates; assert!( candidates.iter().all(|c| matches!(&c.replacement, - Replacement::Summary(node) if matches!(&node.expr, - SummaryExpr::KeepPreAsap(e) if **e == *root))), + Replacement::SubDag(node) if !node.contains_asap() && node.operator == root.operator)), "{query}: {candidates:?}" ); let selected = space @@ -1285,7 +1286,7 @@ fn nonfinite_literals_round_trip_in_plans() { ] { let expression = lower(query); let json = serde_json::to_vec(&expression).unwrap(); - let restored: QueryExpr = serde_json::from_slice(&json).unwrap(); + let restored: Rc = serde_json::from_slice(&json).unwrap(); let result = evaluate_dag(&restored, &fallback_dag(restored.clone()), &[], 60).unwrap(); assert_eq!(result.len(), 1); if expected.is_nan() { @@ -1579,7 +1580,7 @@ fn subquery_label_uniqueness_is_checked_per_evaluation_step() { #[test] fn logical_nonfinite_quantile_parameter_round_trips() { let expression = lower("histogram_quantile(NaN, x_bucket)"); - let restored: QueryExpr = + let restored: Rc = serde_json::from_slice(&serde_json::to_vec(&expression).unwrap()).unwrap(); let samples = buckets(&[("job=a", HISTOGRAM)]); let result = evaluate_dag( @@ -1592,3 +1593,56 @@ fn logical_nonfinite_quantile_parameter_round_trips() { assert_eq!(result.len(), 1); assert!(result[0].2.is_nan()); } + +/// The proposal's pointwise projections preserve names only for unary minus. +#[test] +fn pointwise_projection_names_and_dynamic_parameters() { + let samples = [("job=a", 300, -2.5)]; + assert_eq!( + labeled("-m", &[("m", &samples)], 300), + [("__name__=m,job=a".into(), 2.5)] + ); + assert_eq!( + labeled("abs(m)", &[("m", &samples)], 300), + [("job=a".into(), 2.5)] + ); + assert_eq!( + run("round(m, scalar(vector(2)))", &samples, 300).unwrap(), + [("a".into(), 300_000, -2.0)] + ); + assert_eq!( + run("clamp(m, time()-301, time())", &samples, 300).unwrap(), + [("a".into(), 300_000, -1.0)] + ); + assert!(run("clamp(m, 2, 1)", &samples, 300).unwrap().is_empty()); + assert_eq!( + run("year(m)", &[("a", 300, 0.0)], 300).unwrap(), + [("a".into(), 300_000, 1970.0)] + ); + assert_eq!( + run("hour()", &[], 3600).unwrap(), + [("".into(), 3_600_000, 1.0)] + ); +} + +/// Execute every PromQL root/conversion example in the scalar design document. +#[test] +fn scalar_design_document_examples_execute() { + let samples = [("job=a", 300, 1.0), ("job=b", 300, 2.0)]; + for (query, expected) in [ + ("2", 2.0), + ("time()", 300.0), + ("vector(time())", 300.0), + ("scalar(sum(up)) + 1", 4.0), + ] { + let root = parse_root(query, AccuracyTarget::Exact); + root.validate_structure().unwrap(); + let output = evaluate(query, &[("up", &samples)], 300).unwrap(); + assert_eq!(output.len(), 1, "{query}"); + assert_eq!(output[0].2, expected, "{query}"); + } + assert_eq!( + labeled("up * 2", &[("up", &samples)], 300), + [("job=a".into(), 2.0), ("job=b".into(), 4.0)] + ); +} diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs index 886b7d00d..aae54aab4 100644 --- a/crates/asap-physical-operators/tests/promql_values.rs +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -1,4 +1,6 @@ //! Compile, persist and rebind dynamic-label computation without deployment lowering. +use asap_physical_operators::expressions::binary::{BinaryOpKind, BinaryOperator}; + use asap_physical_operators::{ operators::Operator, physical_planner::{promql_values::*, CompiledPhysicalDag, Source}, @@ -7,6 +9,7 @@ use asap_physical_operators::{ }; use futures::{executor::block_on, StreamExt}; use planner_types::pre_asap::{AggIntent, ColumnRef, GroupKeys}; + use std::collections::BTreeMap; fn row(labels: &[(&str, &str)], value: f64) -> Vec { @@ -213,10 +216,7 @@ fn composed_ensemble_shares_a_producer_across_roots() { runtime::{Input, OutputStream}, values::Schema, }; - use planner_types::{ - post_asap::BinaryOperator, - pre_asap::{ArithmeticOpKind, BinaryOpKind}, - }; + use planner_types::pre_asap::ArithmeticOpKind; struct Counted { source: Operator, starts: std::rc::Rc>, @@ -325,10 +325,7 @@ fn compiled_constant_needs_no_deployment_source() { // arithmetic or bool comparisons remove the metric name. #[test] fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { - use planner_types::{ - post_asap::BinaryOperator, - pre_asap::{ArithmeticOpKind, BinaryOpKind, CompareOpKind}, - }; + use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind}; for left_scalar in [false, true] { for names in [["a", "a"], ["a", "b"]] { for (kind, return_bool) in [ @@ -398,10 +395,10 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { ); } -// Persisted exact readout graphs, rather than the storage adapter, merge panes, +// Persisted exact evaluation graphs, rather than the storage adapter, merge panes, // finalize each population, and preserve the requested metric-name semantics. #[test] -fn exact_state_readouts_recover_and_finalize_panes() { +fn exact_state_evaluations_recover_and_finalize_panes() { use asap_physical_operators::factory::create_planner_accumulator; use planner_types::post_asap::*; use std::sync::Arc; @@ -411,7 +408,7 @@ fn exact_state_readouts_recover_and_finalize_panes() { (ExactKind::Min, ExactParams::Min, 1.), (ExactKind::Max, ExactParams::Max, 5.), ] { - let family = SummaryFamilyType::ExactAggregate(kind, params); + let family = FieldDataType::ExactAggregate(kind, params); for preserve in [false, true] { let rows = [[1., 2.], [4., 5.]] .into_iter() @@ -436,7 +433,7 @@ fn exact_state_readouts_recover_and_finalize_panes() { }) .collect(); let output = run_inputs( - compile_exact_readout(family.clone(), 60_000, preserve).unwrap(), + compile_exact_evaluation(family.clone(), 60_000, preserve).unwrap(), vec![Batch::try_new(exact_state_schema(family.clone()).unwrap(), rows).unwrap()], ) .unwrap(); @@ -459,7 +456,7 @@ fn recovered_exact_counter_uses_window_and_omits_insufficient_samples() { (ExactKind::Rate, ExactParams::Rate, 1.), (ExactKind::Increase, ExactParams::Increase, 60.), ] { - let family = SummaryFamilyType::ExactAggregate(kind, params); + let family = FieldDataType::ExactAggregate(kind, params); let rows = [1, 2] .into_iter() .map(|count| { @@ -483,7 +480,7 @@ fn recovered_exact_counter_uses_window_and_omits_insufficient_samples() { }) .collect(); let output = run_inputs( - compile_exact_readout(family.clone(), 60_000, false).unwrap(), + compile_exact_evaluation(family.clone(), 60_000, false).unwrap(), vec![Batch::try_new(exact_state_schema(family).unwrap(), rows).unwrap()], ) .unwrap(); diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs index 78e4c8c02..50846e9c3 100644 --- a/crates/asap-physical-operators/tests/raw_scan.rs +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -6,37 +6,48 @@ use asap_physical_operators::dag::{ Error, Limits, OutputStream, RunContext, Scope, }; use futures::{executor::block_on, stream, StreamExt}; +use planner_types::ir::export::NonASAPOpKind as ValueOperation; +use planner_types::ir::export::{ + EdgeRole, GroupingEdgeCompatibility, PostAsapDag, PostAsapDagEdge, PostAsapDagNode, + PostAsapNodeId, PostAsapOperatorPayload, WindowEdgeCompatibility, +}; +use planner_types::ir::Predicate; use planner_types::{ post_asap::*, - pre_asap::{Column, DataType, GroupKeys, Predicate, QueryExpr, Source}, + pre_asap::{DataType, Field, GroupKeys, Source}, }; use std::{ collections::BTreeMap, - rc::Rc, sync::{ atomic::{AtomicUsize, Ordering}, Arc, }, }; -fn fixture() -> (QueryExpr, Schema, Vec) { - let schema = - planner_types::pre_asap::Schema::new(vec![Column::new("value", DataType::Int64, true)]); - let output = Arc::new(SummarySchema { - fields: vec![SummaryField { +fn fixture() -> (planner_types::ir::NonASAPOp, Schema, Vec) { + let schema = planner_types::pre_asap::Schema::new(vec![planner_types::pre_asap::Field::plain( + "value", + DataType::Int64, + true, + )]); + let output = Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, + fields: vec![Field { + table: None, name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Int64), + dtype: FieldDataType::Plain(DataType::Int64), nullable: true, }], time_index: None, }); - let scan = QueryExpr::Scan { + let scan = planner_types::ir::NonASAPOp::Scan { source: Source::Table { table_ref: "numbers".into(), }, - predicates: vec![Predicate(Rc::new(QueryExpr::IsNotNull(Rc::new( - QueryExpr::Column(0), - ))))], + predicates: vec![Predicate(planner_types::ir::ScalarExpr::IsNotNull( + Box::new(planner_types::ir::ScalarExpr::Column(0)), + ))], schema, }; let batches = vec![ @@ -53,7 +64,11 @@ fn fixture() -> (QueryExpr, Schema, Vec) { ]; (scan, output, batches) } -fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsapDag { +fn plan( + scan: planner_types::ir::NonASAPOp, + schema: &Schema, + state: ExecutionDataState, +) -> PostAsapDag { let node = |id, payload| PostAsapDagNode { id: PostAsapNodeId(id), payload, @@ -72,13 +87,20 @@ fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsap }; PostAsapDag { nodes: vec![ - node(0, PostAsapOperatorPayload::Fallback { expression: scan }), + node( + 0, + PostAsapOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::from_op(&scan, &mut |_| { + panic!("no plan refs") + }), + }, + ), node( 1, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Sort { - keys: vec![planner_types::pre_asap::SortKey { - expr: QueryExpr::Column(0), + PostAsapOperatorPayload::Relational { + operator: ValueOperation::Sort { + keys: vec![planner_types::ir::export::WireSortKey { + expr: planner_types::ir::export::WireScalarExpr::Column(0), ascending: false, nulls_first: false, }], @@ -88,9 +110,9 @@ fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsap ), node( 2, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Limit { - n: 2, + PostAsapOperatorPayload::Relational { + operator: ValueOperation::Limit { + n: Some(2), offset: 0, partition_by: GroupKeys::by(vec![]), }, @@ -219,17 +241,27 @@ fn lazy_open_shared_producer_and_cancellation() { #[test] fn binding_errors_and_reader_errors_are_not_empty_results() { let (mut scan, schema, _) = fixture(); - assert!(DataSources::default().bind(&scan).is_err()); + assert!(DataSources::default() + .bind(&planner_types::ir::OperatorNode::with_schema( + planner_types::ir::Operator::NonASAP(scan.clone()), + scan.output_schema().unwrap() + )) + .is_err()); let opened = Arc::new(AtomicUsize::new(0)); let sources = registry(Arc::new(CountingSource { schema: schema.clone(), opened: opened.clone(), fail: true, })); - if let QueryExpr::Scan { predicates, .. } = &mut scan { - predicates.push(Predicate(Rc::new(QueryExpr::Column(0)))); + if let planner_types::ir::NonASAPOp::Scan { predicates, .. } = &mut scan { + predicates.push(Predicate(planner_types::ir::ScalarExpr::Column(0))); } - assert!(sources.bind(&scan).is_err()); + assert!(sources + .bind(&planner_types::ir::OperatorNode::with_schema( + planner_types::ir::Operator::NonASAP(scan.clone()), + scan.output_schema().unwrap() + )) + .is_err()); assert_eq!(opened.load(Ordering::SeqCst), 0); let (scan, _, _) = fixture(); let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); @@ -293,19 +325,23 @@ fn schema_drift_and_memory_limits_fail_the_scan() { #[test] fn empty_sources_and_three_valued_predicates() { use planner_types::pre_asap::{CompareOpKind, ScalarValue}; + let (mut scan, schema, batches) = fixture(); - if let QueryExpr::Scan { + if let planner_types::ir::NonASAPOp::Scan { predicates, source, .. } = &mut scan { *source = Source::TimeSeries { metric: "samples".into(), }; - *predicates = vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + *predicates = vec![Predicate(planner_types::ir::ScalarExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(planner_types::ir::ScalarExpr::Column(0)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(2))), - }))]; + right: Box::new(planner_types::ir::ScalarExpr::Literal(ScalarValue::Int64( + 2, + ))), + })]; } for (batches, expected) in [(vec![], 0), (batches, 2)] { let mut sources = DataSources::default(); diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs index a61a596cb..3ea050f8a 100644 --- a/crates/asap-physical-operators/tests/summary_projection.rs +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -8,9 +8,14 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; +use planner_types::ir::export::NonASAPOpKind as ValueOperation; +use planner_types::ir::export::{ + EdgeRole, GroupingEdgeCompatibility, PostAsapDag, PostAsapDagEdge, PostAsapDagNode, + PostAsapNodeId, PostAsapOperatorPayload, WindowEdgeCompatibility, +}; use planner_types::{ post_asap::*, - pre_asap::{ColumnRef, DataType, ProjectItem, QueryExpr}, + pre_asap::{ColumnRef, DataType}, }; use std::{collections::BTreeMap, sync::Arc}; @@ -18,26 +23,32 @@ use std::{collections::BTreeMap, sync::Arc}; // the family and pass through the same immutable state, without decoding the payload. #[test] fn post_asap_summary_projection_survives_recovery() { - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let schema = Arc::new(SummarySchema { + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let schema = Arc::new(planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, fields: vec![ - SummaryField { + Field { + table: None, name: "state".into(), dtype: family.clone(), nullable: false, }, - SummaryField { + Field { + table: None, name: "service".into(), - dtype: SummaryFamilyType::Plain(DataType::Utf8), + dtype: FieldDataType::Plain(DataType::Utf8), nullable: false, }, ], time_index: None, }); - let output = SummarySchema { + let output = planner_types::pre_asap::Schema { + unique_keys: vec![], + closed: false, fields: vec![ schema.fields[1].clone(), - SummaryField { + Field { name: "renamed".into(), ..schema.fields[0].clone() }, @@ -55,13 +66,13 @@ fn post_asap_summary_projection_survives_recovery() { }, PostAsapDagNode { id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::Value { - operation: ValueOperation::Project { + payload: PostAsapOperatorPayload::Relational { + operator: ValueOperation::Project { cols: vec![1, 0] .into_iter() - .map(|index| ProjectItem { + .map(|index| planner_types::ir::export::WireProjectItem { alias: None, - expr: QueryExpr::Column(index), + expr: planner_types::ir::export::WireScalarExpr::Column(index), }) .collect(), qualifier: None, diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index ef086b1bf..953272a4d 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -1,10 +1,11 @@ //! Planner output binds directly to the shared runtime at a declared rate-value frontier. +mod common; use asap_aware_mapping::{ accuracy::{ AccuracyEvidenceProvider, DefaultAccuracyModel, EqualSplitAllocator, PropagationStats, }, cost_model::DefaultCostModel, - Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG, + ASAPStrategies, Replacement, ReplacementStrategy, TargetSubDAG, }; use asap_physical_operators::dag::{ operators::Operator, @@ -12,23 +13,21 @@ use asap_physical_operators::dag::{ values::{Batch, Value}, Limits, RunContext, Scope, }; +use common::compile_post_asap_dag; use futures::{executor::block_on, StreamExt}; -use planner_types::{ - post_asap::*, - pre_asap::{DataType, QueryExpr}, - types::AccuracyTarget, -}; +use planner_types::ir::export::{PostAsapDag, PostAsapOperatorPayload}; +use planner_types::{post_asap::*, pre_asap::DataType, types::AccuracyTarget}; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; struct Evidence; impl AccuracyEvidenceProvider for Evidence { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &planner_types::ir::OperatorNode) -> Option { Some(1000) } fn propagation_stats( &self, op: &CompositionOperator, - _: &SummaryFamilyType, - _: Option<&SketchQuery>, + _: &FieldDataType, + _: Option<&SketchStatistic>, ) -> PropagationStats { if matches!(op, CompositionOperator::TopKSelection) { PropagationStats { @@ -63,14 +62,12 @@ fn physical_binding_does_not_impose_an_accuracy_acceptance_policy() { } fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: SketchAlgorithm) { - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.1), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.1), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -80,7 +77,7 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) + Replacement::SubDag(node) if candidate.rationale.contains(&format!("{algorithm:?}")) => { Some(node) @@ -89,7 +86,7 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S }) .unwrap(); let dag = compile_post_asap_dag(&plan).unwrap(); - let build=dag.nodes.iter().find(|node|matches!(&node.payload,PostAsapOperatorPayload::SummaryAgg{family:SummaryFamilyType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); + let build=dag.nodes.iter().find(|node|matches!(&node.payload,PostAsapOperatorPayload::SummaryAgg{family:FieldDataType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); let rate_id = dag .edges .iter() @@ -123,7 +120,7 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S "job" => Value::Utf8(job.into()), "value" => Value::Float64(value), _ => match field.dtype { - SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), + FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), _ => panic!("unexpected rate column {field:?}"), }, }) @@ -203,7 +200,7 @@ use planner_types::workload::{ pub fn lower_promql( query: &str, accuracy: AccuracyTarget, -) -> Result { +) -> Result, asap_frontend_promql::PromqlError> { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -235,7 +232,7 @@ pub fn lower_promql( // The old untyped heap updater must not silently round a Planner rate update. #[test] fn rate_updates_cannot_enter_integer_heap_factory() { - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::CmsWithHeap, SketchParams::CmsWithHeap { @@ -272,7 +269,7 @@ fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { check_direct_rate_topk(false); } -// Unreferenced labels still distinguish series throughout Rate and heap readout. +// Unreferenced labels still distinguish series throughout Rate and heap evaluation. #[test] fn direct_rate_topk_preserves_dynamic_unreferenced_labels() { check_direct_rate_topk(true); @@ -284,16 +281,20 @@ fn check_direct_rate_topk(dynamic: bool) { }; let mut logical = lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(); - fn resolve_catalog(node: &mut QueryExpr) { - match node { - QueryExpr::Aggregate { child, .. } | QueryExpr::TimeRange { child, .. } => { - resolve_catalog(Rc::make_mut(child)) - } - QueryExpr::Scan { schema, .. } => { + fn resolve_catalog(node: &mut planner_types::ir::OperatorNode) { + match &mut node.operator { + planner_types::ir::Operator::NonASAP( + planner_types::ir::NonASAPOp::Aggregate { child, .. } + | planner_types::ir::NonASAPOp::TimeRange { child, .. }, + ) => resolve_catalog(Rc::make_mut(child)), + planner_types::ir::Operator::NonASAP(planner_types::ir::NonASAPOp::Scan { + schema, + .. + }) => { schema.closed = true; schema - .columns - .push(planner_types::pre_asap::schema::Column::new( + .fields + .push(planner_types::pre_asap::schema::Field::plain( "service", DataType::Utf8, false, @@ -301,14 +302,15 @@ fn check_direct_rate_topk(dynamic: bool) { } _ => panic!("unexpected input shape: {node:?}"), } + node.schema = node.operator.output_schema().unwrap(); } if dynamic { logical = with_series_identity(&logical).unwrap(); } else { - resolve_catalog(&mut logical); + resolve_catalog(Rc::make_mut(&mut logical)); } - let root = Rc::new(logical); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let root = logical; + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -322,7 +324,7 @@ fn check_direct_rate_topk(dynamic: bool) { let candidate = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) + Replacement::SubDag(node) if candidate.rationale.contains(&format!("{algorithm:?}")) => { Some(node) @@ -337,26 +339,25 @@ fn check_direct_rate_topk(dynamic: bool) { ) .unwrap(); assert!(matches!( - source.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } + source.operator, + planner_types::ir::Operator::ASAP( + planner_types::ir::ASAPOp::FinalizeExactAccumulator { .. } + ) )); assert_eq!(ranked.input_contracts().count(), 1); let encoded = String::from_utf8(serde_json::to_vec(&ranked).unwrap()).unwrap(); assert!(encoded.contains("KeyedSummaryBuild")); - assert!(encoded.contains("KeyedReadout")); + assert!(encoded.contains("KeyedEvaluation")); assert!( !encoded.contains("\"Rate\""), - "Rate must be supplied by its exact stored-state readout" + "Rate must be supplied by its exact stored-state evaluation" ); } let dag = compile_post_asap_dag(candidate).unwrap(); assert!(dag.nodes.iter().any(|node| matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); + PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); let build = dag.nodes.iter().find(|node| matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); + PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); let input_id = dag .edges .iter() @@ -377,8 +378,8 @@ fn check_direct_rate_topk(dynamic: bool) { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::TimeRange { .. } + PostAsapOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::TimeRange { .. } } ) }) @@ -634,7 +635,7 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { }; let logical = lower_promql("topk by(job)(1, m)", AccuracyTarget::Epsilon(0.1)).unwrap(); let root = Rc::new(with_series_identity(&logical).unwrap()); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -649,7 +650,7 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { let selected = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CountSketchWithHeap") => { + Replacement::SubDag(node) if candidate.rationale.contains("CountSketchWithHeap") => { Some(node) } _ => None, @@ -662,8 +663,8 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::TimeRange { .. } + PostAsapOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::TimeRange { .. } } ) }) @@ -676,7 +677,7 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { ) .unwrap(); let snapshot_program = - asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + asap_physical_operators::physical_planner::promql_rows::compile_current_series_evaluation( selected, ) .unwrap(); @@ -684,7 +685,7 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { serde_json::from_slice(&serde_json::to_vec(&snapshot_program).unwrap()).unwrap(); assert!(!encoded.to_string().contains("CurrentSeries")); assert!(encoded.to_string().contains("KeyedSummaryBuild")); - assert!(encoded.to_string().contains("KeyedReadout")); + assert!(encoded.to_string().contains("KeyedEvaluation")); for (values, expected, score) in [ ([100., 20.], "a", 100.), ([1., 20.], "b", 20.), @@ -758,7 +759,7 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { /// Deployment-side lifecycle choice: every summary state of `candidate` is /// continuously maintained, and the chosen lifecycles set execution timing. -fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDag { +fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDag { use asap_aware_mapping::{ cost_model::{Cost, CostModel}, enumerate_summary_maintenance_lifecycles, CostRate, Horizon, @@ -779,7 +780,7 @@ fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDag { } fn summary_maintenance_lifecycle_cost_inputs( &self, - _: &SummaryNode, + _: &planner_types::ir::OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(10.)), @@ -791,7 +792,7 @@ fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDag { } fn summary_maintenance_capabilities( &self, - _: &SummaryNode, + _: &planner_types::ir::OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -861,7 +862,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { ) .unwrap(), ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -871,7 +872,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .replacements(&TargetSubDAG::new(&root)) .into_iter() .filter_map(|candidate| match candidate.replacement { - Replacement::Summary(root) if candidate.rationale.contains("WithHeap") => Some(root), + Replacement::SubDag(root) if candidate.rationale.contains("WithHeap") => Some(root), _ => None, }) .collect::>(); @@ -885,7 +886,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { matches!( &node.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. } ) @@ -898,7 +899,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { matches!( &node.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::Sketch(..), + family: FieldDataType::Sketch(..), .. } ) @@ -987,9 +988,9 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .fields .iter() .map(|field| match &field.dtype { - SummaryFamilyType::ExactAggregate(..) => summary.clone(), - SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(end), - SummaryFamilyType::Plain(DataType::Utf8) + FieldDataType::ExactAggregate(..) => summary.clone(), + FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(end), + FieldDataType::Plain(DataType::Utf8) if field.name == "$promql_series_identity" => { Value::Utf8( @@ -1001,7 +1002,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .into(), ) } - SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8("api".into()), + FieldDataType::Plain(DataType::Utf8) => Value::Utf8("api".into()), _ => panic!("unexpected state field {field:?}"), }) .collect() diff --git a/crates/devtools/examples/canonical_examples.rs b/crates/devtools/examples/canonical_examples.rs index c7d089243..fd4b703a1 100644 --- a/crates/devtools/examples/canonical_examples.rs +++ b/crates/devtools/examples/canonical_examples.rs @@ -1,16 +1,16 @@ // cargo run -p asap-lower --example canonical_examples // -// One-off: pretty-print the QueryExpr for one canonical query per variant, +// One-off: pretty-print the `OperatorNode` DAG for one canonical query per variant, // plus custom Join/SetOp/Dedup/CTE probes, to eyeball the actual shape. use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn packets_catalog() -> SqlCatalog { @@ -48,7 +48,7 @@ fn bgp_catalog() -> SqlCatalog { async fn main() { let promql_examples: &[(&str, &str)] = &[ ("Scan", "up"), - ("BinaryOp + PromqlScalarBridge", "up > 1"), + ("Filter + scalar predicate", "up > 1"), ("EvalTimestamp", "time()"), ("Aggregate", "sum(up)"), ( diff --git a/crates/devtools/examples/topk_ir.rs b/crates/devtools/examples/topk_ir.rs index ee467aa1c..21dcefb4b 100644 --- a/crates/devtools/examples/topk_ir.rs +++ b/crates/devtools/examples/topk_ir.rs @@ -4,11 +4,11 @@ // resulting pre-ASAP IR. Used for interactive exploration; not a test. use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn catalog() -> SqlCatalog { diff --git a/crates/devtools/src/bin/analyze_corpora.rs b/crates/devtools/src/bin/analyze_corpora.rs index 503fb1f04..9db2b7f00 100644 --- a/crates/devtools/src/bin/analyze_corpora.rs +++ b/crates/devtools/src/bin/analyze_corpora.rs @@ -6,7 +6,7 @@ use asap_devtools::{lower_promql_with_data_ingestion_interval, SqlCatalog}; use asap_frontend_sql::lower_sql_dialect; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use serde::Serialize; @@ -29,8 +29,8 @@ const NETFLOW: &str = include_str!("../../../frontend-sql/tests/netflow/data/net const BGP: &str = include_str!("../../../frontend-sql/tests/bgp_analytics/data/bgp_analytics.sql"); const BGP_WORKLOAD: &str = include_str!("../../../frontend-sql/tests/bgp_jan2024_workload/data/bgp_jan2024_rrc00_200_query_workload.yaml"); -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn dqc_catalog() -> SqlCatalog { @@ -218,7 +218,7 @@ fn run_corpus(name: &str, source: &str, interval_ms: u64) -> CorpusResult { normalized_expression, structural_shape, lowered: true, - ir: Some(serde_json::to_value(&ir).expect("QueryExpr must serialize")), + ir: Some(serde_json::to_value(&ir).expect("OperatorNode must serialize")), ir_debug: Some(format!("{ir:#?}")), error: None, }), @@ -493,7 +493,7 @@ async fn run_sql_corpora(out_dir: PathBuf) { normalized_expression, structural_shape, lowered: true, - ir: Some(serde_json::to_value(&ir).expect("QueryExpr must serialize")), + ir: Some(serde_json::to_value(&ir).expect("OperatorNode must serialize")), ir_debug: Some(format!("{ir:#?}")), error: None, }), diff --git a/crates/devtools/src/bin/dag_export.rs b/crates/devtools/src/bin/dag_export.rs index 3f7cd8850..80972b089 100644 --- a/crates/devtools/src/bin/dag_export.rs +++ b/crates/devtools/src/bin/dag_export.rs @@ -12,7 +12,7 @@ // `--epsilon ` is optional and applies to every query in the run: it // lowers with `AccuracyTarget::Epsilon()` instead of the default // `AccuracyTarget::Exact`. Without it, every `AggIntent` lowers exact and -// `asap_aware_mapping::SketchAlgorithmStrategy` never has a genuine sketch +// `asap_aware_mapping::ASAPStrategies` never has a genuine sketch // alternative to report — so no node ever picks up a `SketchApproximation` // note. Pass it to actually exercise that path, e.g.: // cargo run -p asap-lower --bin dag_export -- \ @@ -41,8 +41,8 @@ // `asap_types::dag_export::export_post_asap`. // // Together these surface every one of the four concrete replacement kinds: -// the sketch family `SketchAlgorithmStrategy`/`HydraGroupingStrategy` bound, -// the CSE share/recompute choice `SharedSubtreeStrategy` found, the +// the sketch family `ASAPStrategies`/`HydraGroupingStrategy` bound, +// the CSE share/recompute choice `SharedSubDagStrategy` found, the // workload-aware roll-up `RollupStrategy` derived, and the `avg -> // sum/count` rewrite `AvgToSumOverCountStrategy` proposes. Without // `--post-asap`, every existing invocation of this binary produces @@ -88,8 +88,8 @@ use asap_aware_mapping::physical_plan_cost_model::{ }; use asap_aware_mapping::query_physical_lowering::PhysicalNodeRequest; use asap_aware_mapping::replacement::{ - default_strategies_with_evidence, search_workload, search_workload_with, Replacement, - ReplacementSubDAG, + default_strategies_with_evidence, is_logical_rewrite, search_workload, search_workload_with, + Replacement, ReplacementSubDAG, }; use asap_aware_mapping::{AccuracyEvidenceProvider, PropagationStats}; use asap_types::cost::{BaselineRef, CostAnnotation, CostInput, CostSource, CostUnit}; @@ -97,12 +97,10 @@ use asap_types::dag_export::{ self, DagDecision, DagGraph, DagNote, NamedGraph, PostAsapSubstitution, TargetRejection, TargetReplacement, TargetReplacementAfter, WorkloadGraph, }; -use asap_types::post_asap::SummaryExpr; -use asap_types::post_asap::SummaryNode; -use asap_types::post_asap::{CompositionOperator, SketchQuery, SummaryFamilyType}; -use asap_types::pre_asap::cse::{structural_hash, HashCache}; -use asap_types::pre_asap::query_expr::QueryExpr; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::ir::cse::{structural_hash, HashCache}; +use asap_types::ir::OperatorNode; +use asap_types::post_asap::{CompositionOperator, FieldDataType, SketchStatistic}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::resources::CacheProfile; use asap_types::types::AccuracyTarget; @@ -137,7 +135,7 @@ fn parse_planner_cost_document(raw: &str) -> Result #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] struct TargetPhysicalEvidence { - target: QueryExpr, + target: Rc, scope: ComparisonScopeEvidence, candidates: Vec, } @@ -192,7 +190,7 @@ impl ComparisonScopeEvidence { #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] struct QueryNodePhysicalEvidence { - logical_node: QueryExpr, + logical_node: OperatorNode, operator: asap_aware_mapping::analytical_cost::PhysicalOperator, occurrence: usize, synthetic: bool, @@ -228,11 +226,11 @@ impl CandidatePhysicalEvidence { fn matches(&self, candidate: &ReplacementSubDAG) -> bool { let actual = match (self, &candidate.replacement) { - (Self::Summary { .. }, Replacement::Summary(summary)) => { - serde_json::to_value(dag_export::export_summary(summary)) + (Self::Summary { .. }, Replacement::SubDag(node)) if !is_logical_rewrite(node) => { + serde_json::to_value(dag_export::export(node)) } - (Self::Rewrite { .. }, Replacement::Rewrite(query)) => { - serde_json::to_value(dag_export::export(query)) + (Self::Rewrite { .. }, Replacement::SubDag(node)) if is_logical_rewrite(node) => { + serde_json::to_value(dag_export::export(node)) } _ => return false, }; @@ -369,7 +367,7 @@ impl PlannerPhysicalPlanProvider for ExportPhysicalProvider<'_> { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - _summary: &Rc, + _summary: &Rc, _target: &asap_aware_mapping::replacement::TargetSubDAG<'_>, ) -> Result { if snapshot.scope != self.target.scope.resolve()? { @@ -400,7 +398,7 @@ impl ExportPlannerCostModel<'_> { .document .targets .iter() - .filter(|entry| entry.target == **target.root); + .filter(|entry| entry.target == *target.root); let target_evidence = targets.next()?; if targets.next().is_some() { return None; @@ -429,7 +427,7 @@ impl ExportPlannerCostModel<'_> { fn annotations( &self, candidate: &ReplacementSubDAG, - target: &Rc, + target: &Rc, ) -> (CostAnnotation, CostAnnotation, CostAnnotation) { let target = asap_aware_mapping::replacement::TargetSubDAG::new(target); let Some((provider, calibration)) = self.bound(candidate, &target) else { @@ -712,11 +710,11 @@ fn default_catalog() -> SqlCatalog { "metrics", Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("service", DataType::Utf8, false), - Column::new("region", DataType::Utf8, false), - Column::new("latency", DataType::Float64, false), - Column::new("bytes", DataType::Int64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("region", DataType::Utf8, false), + Field::plain("latency", DataType::Float64, false), + Field::plain("bytes", DataType::Int64, false), ], 0, vec![], @@ -725,8 +723,8 @@ fn default_catalog() -> SqlCatalog { .with_table( "hosts", Schema::new(vec![ - Column::new("service", DataType::Utf8, false), - Column::new("region", DataType::Utf8, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("region", DataType::Utf8, false), ]), ) } @@ -742,7 +740,7 @@ fn catalog(custom: &[String]) -> SqlCatalog { let columns = value["columns"] .as_array() .expect("--table-schema.columns must be an array"); - let columns: Vec = columns + let columns: Vec = columns .iter() .map(|column| { let column_name = column["name"] @@ -760,7 +758,7 @@ fn catalog(custom: &[String]) -> SqlCatalog { "int64" | "bigint" => DataType::Int64, other => panic!("unsupported column type {other:?}"), }; - Column::new( + Field::plain( column_name, data_type, column["nullable"].as_bool().unwrap_or(true), @@ -825,8 +823,8 @@ impl AccuracyEvidenceProvider for TopKMarginEvidence { fn propagation_stats( &self, op: &CompositionOperator, - _family: &SummaryFamilyType, - _query: Option<&SketchQuery>, + _family: &FieldDataType, + _query: Option<&SketchStatistic>, ) -> PropagationStats { if matches!(op, CompositionOperator::TopKSelection) { PropagationStats { @@ -961,11 +959,10 @@ fn parse_args_from(argv: impl Iterator) -> ParsedArgs { } /// Attach workload-wide replacement explanations to their exact graph nodes. -/// `node_hash` is only a narrowing filter; `source_expr == Some(target)` is -/// the collision-safe identity check (`source_expr` is `None` only for a -/// post-ASAP-originated node inside a `--post-asap` `post_graph`, which this -/// function is never called on — every node it sees, from an ordinary -/// [`dag_export::export`], carries `Some`). +/// `node_hash` is only a narrowing filter; `source_node == Some(target)` is +/// the collision-safe identity check (every node an ordinary +/// [`dag_export::export`] produces carries `Some`; the `None` arm is +/// defensive only). fn annotate_with_explanations( graph: &mut DagGraph, explanations: &[asap_aware_mapping::ReplacementExplanation], @@ -974,7 +971,7 @@ fn annotate_with_explanations( for (i, explanation) in explanations.iter().enumerate() { for node in graph.nodes.iter_mut() { if node.hash == Some(explanation.node_hash) - && node.source_expr.as_ref() == Some(explanation.target.as_ref()) + && node.source_node.as_ref() == Some(&explanation.target) { node.notes.push(DagNote { kind: format!("{:?}", explanation.kind), @@ -992,7 +989,7 @@ fn annotate_with_explanations( /// never disagree about which candidate won for a given target. #[allow(dead_code)] struct Winner<'a> { - target: &'a Rc, + target: &'a Rc, candidate: &'a ReplacementSubDAG, costs: (CostAnnotation, CostAnnotation, CostAnnotation), } @@ -1012,7 +1009,7 @@ fn decision_rationale(winner: &Winner<'_>) -> String { "Composes compatible nested aggregates using their declared algebraic intent while preserving the output schema." .to_string() } - "SharedSubtreeStrategy" => match winner.candidate.provenance { + "SharedSubDagStrategy" => match winner.candidate.provenance { asap_aware_mapping::replacement::ReplacementProvenance::CseShare => { "Builds the repeated subtree once and shares it across consumers.".to_string() } @@ -1056,7 +1053,7 @@ fn lookup_winner( by_hash: &HashMap>, winners: &[Winner<'_>], cache: &mut HashCache, - expr: &QueryExpr, + expr: &OperatorNode, ) -> Option { let hash = structural_hash(expr, cache); by_hash @@ -1102,12 +1099,10 @@ fn target_replacement( let strategy = winner.candidate.strategy.to_string(); let before = dag_export::export(winner.target); let after = match &winner.candidate.replacement { - Replacement::Summary(node) => { - TargetReplacementAfter::Summary(dag_export::export_summary(node)) - } - Replacement::Rewrite(rewritten) => { - TargetReplacementAfter::Rewrite(dag_export::export(rewritten)) + Replacement::SubDag(node) if is_logical_rewrite(node) => { + TargetReplacementAfter::Rewrite(dag_export::export(node)) } + Replacement::SubDag(node) => TargetReplacementAfter::Summary(dag_export::export(node)), Replacement::ExactComposition(_) => { unreachable!("composition candidates are materialized by GlobalSelection") } @@ -1131,13 +1126,27 @@ fn target_replacement( } } +/// Is `replacement` `retain_exact`'s conservative no-op fallback — the +/// target itself, unbound, carrying only an exact "kept pre-ASAP" guarantee? +/// `ASAPStrategies` emits it for an intent with no summary +/// realization at all (`STDDEV_POP`, `AVG`, ... dispatch to +/// `Realization::PassThrough`). It is "nothing to bind here", not a +/// replacement decision. A logical rewrite (no guarantee yet) and any sub-DAG +/// with an ASAP operator are real candidates. +fn is_trivial_retain_exact(replacement: &Replacement) -> bool { + matches!( + replacement, + Replacement::SubDag(node) if node.guarantee.is_some() && !node.contains_asap() + ) +} + /// The two additive `--post-asap` outputs — see this file's top-of-file /// usage doc for what each is for. struct PostAsapResults { /// One `(query_name, TargetReplacement)` pair per discovered replacement /// site whose target node is found in that query's own exported graph. A /// target can in principle be reachable from more than one query's root - /// after CSE (a shared subtree), in which case it yields one pair per + /// after CSE (a shared sub-DAG), in which case it yields one pair per /// matching query, each with that query's own `target_pre_id`. replacements: Vec<(String, TargetReplacement)>, /// One merged, whole-query [`DagGraph`] per query, built via @@ -1159,7 +1168,7 @@ fn raw_only_post_asap_results() -> PostAsapResults { } /// Assign collision-free, explicit identities to structurally equal nodes -/// across a set of exported query graphs. The full canonical subtree string +/// across a set of exported query graphs. The full canonical sub-DAG string /// is the equality key; the compact integer is what JSON consumers receive. /// Consequently the viewer never needs to guess identity from labels, /// hashes, or a client-side node signature. @@ -1210,7 +1219,7 @@ fn assign_workload_node_ids(graphs: &mut [&mut DagGraph]) { /// which candidate won for a given target. #[allow(dead_code)] fn run_post_asap_with_progress( - lowered_queries: &[(String, String, QueryExpr)], + lowered_queries: &[(String, String, Rc)], progress: bool, cost_model: &dyn CostModel, export_model: Option<&ExportPlannerCostModel<'_>>, @@ -1220,9 +1229,9 @@ fn run_post_asap_with_progress( if progress { eprintln!("[3/4] ASAP-aware mapping is running…"); } - let roots: Vec<(String, Rc)> = lowered_queries + let roots: Vec<(String, Rc)> = lowered_queries .iter() - .map(|(name, _, qe)| (name.clone(), Rc::new(qe.clone()))) + .map(|(name, _, qe)| (name.clone(), Rc::clone(qe))) .collect(); let strategies; let space = if let Some(evidence) = evidence { @@ -1233,28 +1242,23 @@ fn run_post_asap_with_progress( }; let selection = space.global_selection(cost_model); - // A group's top candidate can be `keep_pre_asap`'s own conservative - // fallback — `Replacement::Summary(SummaryNode { expr: - // KeepPreAsap(Rc::new(target.clone())), .. })` — the *whole target* - // wrapped as unbound, e.g. for a multi-measure/`HAVING`-bearing - // aggregate, or (the case that actually surfaces this: `STDDEV_POP`/ - // `AVG`/`VARIANCE` dispatch to `Realization::PassThrough` with no - // alternative at all, per `realizations_for_intent`'s own doc) an - // intent with no summary realization whatsoever. This isn't a - // replacement decision — it's `SketchAlgorithmStrategy` saying "nothing - // to bind here" — the identical "no-op candidate" concept - // `explanation.rs`'s own `sketch_finding_reason` already excludes from - // being reported as a finding ("a candidate list containing only the - // trivial no-op realization... isn't an opportunity, it's just the - // target's existing shape reflected back"). Filtered out here for a - // second, load-bearing reason beyond just matching that precedent: - // `export_post_asap`'s `find_winner` re-checks every node reached - // inside a spliced-in `KeepPreAsap` payload (by design, so a target - // nested underneath one still gets found) — if that payload structurally - // *is* the enclosing target, `find_winner` immediately matches the same - // winner again, forever. Treating this candidate as "no winner" (same - // as an empty candidate list) avoids ever handing `export_post_asap` a - // winner that can't help but recurse into itself. + // A group's top candidate can be `retain_exact`'s own conservative + // fallback — the *whole target* itself, unbound, carrying only an exact + // "kept pre-ASAP" guarantee (see `is_trivial_retain_exact`) — e.g. for + // a multi-measure/`HAVING`-bearing aggregate, or (the case that actually + // surfaces this: `STDDEV_POP`/`AVG`/`VARIANCE` dispatch to + // `Realization::PassThrough` with no alternative at all, per + // `realizations_for_intent`'s own doc) an intent with no summary + // realization whatsoever. This isn't a replacement decision — it's + // `ASAPStrategies` saying "nothing to bind here" — the + // identical "no-op candidate" concept `explanation.rs`'s own + // `sketch_finding_reason` already excludes from being reported as a + // finding ("a candidate list containing only the trivial no-op + // realization... isn't an opportunity, it's just the target's existing + // shape reflected back"). Treating this candidate as "no winner" (same + // as an empty candidate list) also keeps `post_graph` honest: splicing + // the target in for itself would tag every node of an unchanged sub-DAG + // with a "replacement" decision. let winners: Vec> = selection .target_selections() .filter_map(|group| { @@ -1267,10 +1271,7 @@ fn run_post_asap_with_progress( if matches!(candidate.replacement, Replacement::ExactComposition(_)) { return None; } - if matches!( - &candidate.replacement, - Replacement::Summary(node) if matches!(node.expr, SummaryExpr::KeepPreAsap(_)) - ) { + if is_trivial_retain_exact(&candidate.replacement) { return None; } Some(Winner { @@ -1309,7 +1310,7 @@ fn run_post_asap_with_progress( // *original*, pre-rewrite `graph` — there's nothing wrong with that // winner, it's just nested. `post_graph` is where it's expected to // surface instead (`export_post_asap`'s recursive `find_winner` - // threading walks straight through a rewritten subtree and re-checks + // threading walks straight through a rewritten sub-DAG and re-checks // every node inside it too), so the flat-`replacements` pass below // checks there before deciding a miss is a real anomaly worth a // warning. @@ -1318,7 +1319,7 @@ fn run_post_asap_with_progress( } let post_started = Instant::now(); let mut post_graph_cache = HashCache::new(); - let mut find_winner = |expr: &QueryExpr| -> Option { + let mut find_winner = |expr: &Rc| -> Option { let i = lookup_winner(&by_hash, &winners, &mut post_graph_cache, expr)?; let winner = &winners[i]; let (baseline_cost, selected_cost, benefit) = winner.costs.clone(); @@ -1337,11 +1338,11 @@ fn run_post_asap_with_progress( benefit: Some(benefit), }; Some(match &winners[i].candidate.replacement { - Replacement::Rewrite(rc) => PostAsapSubstitution::Rewrite { + Replacement::SubDag(rc) if is_logical_rewrite(rc) => PostAsapSubstitution::Rewrite { replacement: Rc::clone(rc), decision, }, - Replacement::Summary(rc) => PostAsapSubstitution::Summary { + Replacement::SubDag(rc) => PostAsapSubstitution::Summary { replacement: Rc::clone(rc), decision, }, @@ -1370,7 +1371,7 @@ fn run_post_asap_with_progress( // is documented as an id into `NamedGraph.graph.nodes`, so a nested // secondary target (see above) never gets a flat entry of its own here: // it's already visible, in place, inside its parent's own `after` - // subtree and inside `post_graph` as a whole. + // sub-DAG and inside `post_graph` as a whole. let mut lookup_cache = HashCache::new(); let mut replacements = Vec::new(); let mut rejections = Vec::new(); @@ -1389,20 +1390,20 @@ fn run_post_asap_with_progress( for (name, _, qe) in lowered_queries { let graph = dag_export::export(qe); for node in &graph.nodes { - let Some(source_expr) = node.source_expr.as_ref() else { + let Some(source_node) = node.source_node.as_ref() else { continue; // never true for a plain `export` — defensive only. }; - if let Some(i) = lookup_winner(&by_hash, &winners, &mut lookup_cache, source_expr) { + if let Some(i) = lookup_winner(&by_hash, &winners, &mut lookup_cache, source_node) { replacements.push(( name.clone(), target_replacement(i as u32, node.id, &winners[i]), )); matched[i] = true; } - let hash = structural_hash(source_expr, &mut lookup_cache); + let hash = structural_hash(source_node, &mut lookup_cache); for &i in rejected_by_hash.get(&hash).into_iter().flatten() { let group = rejected_groups[i]; - if *source_expr != *group.target { + if *source_node != group.target { continue; } rejections.extend(group.rejected.iter().map(|rejected| { @@ -1430,7 +1431,7 @@ fn run_post_asap_with_progress( // specifically). This isn't a data loss: `export_post_asap` still // splices that winner in, in place, inside `post_graph` — see this // function's own construction of `post_graphs` above, which walks - // straight through a rewritten subtree and resolves every nested + // straight through a rewritten sub-DAG and resolves every nested // winner too, recursively. So an unmatched winner here is expected, // not necessarily a bug, whenever it's downstream of some other // winner's own `Replacement::Rewrite` — logged as an FYI rather than a @@ -1465,7 +1466,7 @@ fn run_post_asap_with_progress( } #[cfg(test)] -fn run_post_asap(lowered_queries: &[(String, String, QueryExpr)]) -> PostAsapResults { +fn run_post_asap(lowered_queries: &[(String, String, Rc)]) -> PostAsapResults { run_post_asap_with_progress(lowered_queries, false, &DefaultCostModel, None, None) } @@ -1704,14 +1705,26 @@ mod tests { }; use asap_aware_mapping::query_physical_lowering::lower_query_physical_dag; use asap_devtools::PromqlError; - use asap_types::pre_asap::{Column, DataType, Reduction, Schema, Source}; + use asap_types::ir::NonASAPOp; + use asap_types::pre_asap::{DataType, Field, Reduction, Schema, Source}; - fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Result { + fn lower_promql( + query: &str, + accuracy: AccuracyTarget, + ) -> Result, PromqlError> { lower_promql_with_data_ingestion_interval(query, accuracy, 1_000) } - fn non_topk_query() -> QueryExpr { - QueryExpr::Aggregate { + fn non_topk_query() -> Rc { + let scan = OperatorNode::non_asap_node(NonASAPOp::Scan { + source: Source::Table { + table_ref: "events".into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("v", DataType::Int64, false)]), + }) + .expect("scan leaf derives its schema"); + OperatorNode::non_asap_node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![asap_types::pre_asap::AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.1), @@ -1719,23 +1732,18 @@ mod tests { output_names: vec![], filters: vec![], having: None, - child: Rc::new(QueryExpr::Scan { - source: Source::Table { - table_ref: "events".into(), - }, - predicates: vec![], - schema: Schema::new(vec![Column::new("v", DataType::Int64, false)]), - }), - } + child: scan, + }) + .expect("count aggregate derives its schema") } fn fixture_raw_dag( - query: &QueryExpr, + query: &Rc, candidate: &ReplacementSubDAG, document: &PlannerCostDocument, ) -> PhysicalDag { let model = ExportPlannerCostModel { document }; - let root = Rc::new(query.clone()); + let root = Rc::clone(query); let target = asap_aware_mapping::replacement::TargetSubDAG::new(&root); let (provider, _) = model.bound(candidate, &target).unwrap(); let snapshot = provider.capture_evidence_snapshot(&target).unwrap(); @@ -1751,7 +1759,7 @@ mod tests { let (query, candidate, mut document) = cost_fixture(); let raw = fixture_raw_dag(&query, &candidate, &document); let candidate_dag = cheap_candidate_dag(); - let root = Rc::new(query.clone()); + let root = Rc::clone(&query); let target = asap_aware_mapping::replacement::TargetSubDAG::new(&root); assert!(ExportPlannerCostModel { document: &document @@ -1914,7 +1922,7 @@ mod tests { let (query, candidate, mut document) = cost_fixture(); let raw = fixture_raw_dag(&query, &candidate, &document); let candidate_dag = cheap_candidate_dag(); - let root = Rc::new(query.clone()); + let root = Rc::clone(&query); let target = asap_aware_mapping::replacement::TargetSubDAG::new(&root); assert!(ExportPlannerCostModel { document: &document @@ -2194,7 +2202,7 @@ mod tests { EdgeStatistics { rows, bytes } } - fn query_evidence(query: &QueryExpr) -> Vec { + fn query_evidence(query: &Rc) -> Vec { let entries = RefCell::new(Vec::new()); let scope = test_scope().resolve().unwrap(); let provider = |request: PhysicalNodeRequest<'_>| { @@ -2242,7 +2250,7 @@ mod tests { }); Ok(evidence) }; - lower_query_physical_dag(&Rc::new(query.clone()), &scope, &provider).unwrap(); + lower_query_physical_dag(query, &scope, &provider).unwrap(); entries.into_inner() } @@ -2282,49 +2290,28 @@ mod tests { fn candidate_plan(candidate: &ReplacementSubDAG) -> serde_json::Value { match &candidate.replacement { - Replacement::Summary(summary) => { - serde_json::to_value(dag_export::export_summary(summary)).unwrap() - } - Replacement::Rewrite(rewrite) => { - serde_json::to_value(dag_export::export(rewrite)).unwrap() - } + Replacement::SubDag(node) => serde_json::to_value(dag_export::export(node)).unwrap(), Replacement::ExactComposition(_) => { unreachable!("cost fixtures select directly materialized candidates") } } } - fn cost_fixture() -> (QueryExpr, ReplacementSubDAG, PlannerCostDocument) { + fn cost_fixture() -> (Rc, ReplacementSubDAG, PlannerCostDocument) { let query = non_topk_query(); - let root = Rc::new(query.clone()); + let root = Rc::clone(&query); let space = search_workload(vec![(String::from("q"), Rc::clone(&root))]); let group = space .target_subdag_candidates() - .find(|group| *group.target == query) + .find(|group| group.target == query) .expect("aggregate memo group"); let candidate = group .candidates .iter() - .find(|candidate| { - !matches!( - &candidate.replacement, - Replacement::Summary(node) - if matches!(node.expr, SummaryExpr::KeepPreAsap(_)) - ) - }) + .find(|candidate| !is_trivial_retain_exact(&candidate.replacement)) .expect("summary candidate") .clone(); - let plan = match &candidate.replacement { - Replacement::Summary(summary) => { - serde_json::to_value(dag_export::export_summary(summary)).unwrap() - } - Replacement::Rewrite(rewrite) => { - serde_json::to_value(dag_export::export(rewrite)).unwrap() - } - Replacement::ExactComposition(_) => { - unreachable!("cost fixtures select directly materialized candidates") - } - }; + let plan = candidate_plan(&candidate); let document = PlannerCostDocument { storage_io: None, handoffs: None, @@ -2339,12 +2326,14 @@ mod tests { target: query.clone(), scope: test_scope(), candidates: vec![match &candidate.replacement { - Replacement::Summary(_) => CandidatePhysicalEvidence::Summary { - plan, - query_nodes: query_evidence(&query), - physical_dag: cheap_candidate_dag(), - }, - Replacement::Rewrite(_) => CandidatePhysicalEvidence::Rewrite { + Replacement::SubDag(node) if !is_logical_rewrite(node) => { + CandidatePhysicalEvidence::Summary { + plan, + query_nodes: query_evidence(&query), + physical_dag: cheap_candidate_dag(), + } + } + Replacement::SubDag(_) => CandidatePhysicalEvidence::Rewrite { plan, query_nodes: query_evidence(&query), }, @@ -2365,7 +2354,7 @@ mod tests { assert_eq!(parsed.targets[0].target, query); assert!(parsed.targets[0].candidates[0].matches(&candidate)); let model = ExportPlannerCostModel { document: &parsed }; - let target_rc = Rc::new(query.clone()); + let target_rc = Rc::clone(&query); let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); let (provider, calibration) = model.bound(&candidate, &target).expect("exact binding"); let estimate = PhysicalPlanCostModel::new(&provider, calibration.clone()) @@ -2373,7 +2362,7 @@ mod tests { .estimate_candidate(&candidate, &target) .unwrap(); assert!(estimate.candidate_cost < estimate.raw_cost); - let (baseline, selected, benefit) = model.annotations(&candidate, &Rc::new(query)); + let (baseline, selected, benefit) = model.annotations(&candidate, &query); assert!(baseline.value.is_some()); assert!(selected.value.is_some()); assert!(benefit.value.is_some()); @@ -2405,7 +2394,7 @@ mod tests { .unwrap() .remove("cache_profile"); let parsed = parse_planner_cost_document(&json.to_string()).unwrap(); - let target = Rc::new(query); + let target = query; let legacy = ExportPlannerCostModel { document: &parsed }.annotations(&candidate, &target); let explicit = ExportPlannerCostModel { document: &document, @@ -2426,7 +2415,7 @@ mod tests { fn cache_json_affects_ranking_and_exports_declared_evidence() { // Identical repeats hit the result cache; distinct evaluations still execute. let (query, candidate, document) = cost_fixture(); - let target_rc = Rc::new(query); + let target_rc = query; let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); let no_cache = ExportPlannerCostModel { document: &document, @@ -2513,7 +2502,7 @@ mod tests { #[test] fn duplicate_target_candidate_and_query_evidence_each_fail_closed() { let (query, candidate, document) = cost_fixture(); - let target_rc = Rc::new(query); + let target_rc = query; let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); let mut duplicate_target = document.clone(); @@ -2555,7 +2544,7 @@ mod tests { #[test] fn incomplete_or_unused_json_evidence_fails_closed() { let (query, candidate, document) = cost_fixture(); - let target_rc = Rc::new(query); + let target_rc = query; let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); let mut missing = document.clone(); @@ -2596,7 +2585,7 @@ mod tests { physical_dag.nodes.push(physical_dag.nodes[0].clone()); let document = parse_planner_cost_document(&serde_json::to_string(&document).unwrap()) .expect("invalid physical semantics are checked by the estimator"); - let target_rc = Rc::new(query); + let target_rc = query; let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); assert!(ExportPlannerCostModel { document: &document @@ -2608,18 +2597,17 @@ mod tests { #[test] fn global_selection_uses_the_cheapest_complete_physical_candidate() { let query = non_topk_query(); - let root = Rc::new(query.clone()); + let root = Rc::clone(&query); let space = search_workload(vec![(String::from("q"), Rc::clone(&root))]); let group = space .target_subdag_candidates() - .find(|group| *group.target == query) + .find(|group| group.target == query) .expect("aggregate memo group"); let candidates: Vec<_> = group .candidates .iter() .filter(|candidate| { - matches!(candidate.replacement, Replacement::Summary(ref node) - if !matches!(node.expr, SummaryExpr::KeepPreAsap(_))) + matches!(&candidate.replacement, Replacement::SubDag(node) if node.contains_asap()) }) .take(2) .collect(); @@ -2677,7 +2665,7 @@ mod tests { let selection = space.global_selection(&model); let chosen = selection .target_selections() - .find(|selected| selected.target.as_ref() == &query) + .find(|selected| *selected.target == query) .and_then(|selected| selected.chosen) .expect("one complete physical candidate should win"); assert!(document.targets[0].candidates[1].matches(chosen)); @@ -2779,13 +2767,13 @@ mod tests { let selected_query = lower_promql("up", AccuracyTarget::Exact).unwrap(); let other_query = lower_promql("process_cpu_seconds_total", AccuracyTarget::Exact).unwrap(); let selected = ReplacementSubDAG { - replacement: Replacement::Rewrite(Rc::new(selected_query.clone())), + replacement: Replacement::SubDag(Rc::clone(&selected_query)), strategy: "same-strategy", provenance: asap_aware_mapping::replacement::ReplacementProvenance::LogicalRewrite, rationale: String::new(), }; let other = ReplacementSubDAG { - replacement: Replacement::Rewrite(Rc::new(other_query)), + replacement: Replacement::SubDag(other_query), strategy: "same-strategy", provenance: asap_aware_mapping::replacement::ReplacementProvenance::LogicalRewrite, rationale: String::new(), @@ -3042,8 +3030,8 @@ mod tests { .1; let q3_root = &q3.nodes[q3.root as usize]; let q4_root = &q4.nodes[q4.root as usize]; - assert!(q3_root.label.contains("Limit { n: 5,")); - assert!(q4_root.label.contains("Limit { n: 10,")); + assert!(q3_root.label.contains("Limit(5)")); + assert!(q4_root.label.contains("Limit(10)")); assert_ne!(q3_root.workload_node_id, q4_root.workload_node_id); let q3_ranked = &q3.nodes[q3_root.children[0] as usize]; let q4_ranked = &q4.nodes[q4_root.children[0] as usize]; @@ -3151,21 +3139,19 @@ mod tests { /// against real corpus queries (a `STDDEV_POP` aggregate, which — like /// `AVG` — dispatches to `Realization::PassThrough` with no /// alternative strategy of its own, so its *only* candidate is - /// `keep_pre_asap`'s conservative fallback: `Replacement::Summary` - /// wrapping the *entire target* as `SummaryExpr::KeepPreAsap`). - /// `run_post_asap` must not treat that as a real winner: splicing it - /// into `export_post_asap` would recurse forever, since `find_winner` - /// re-checks every node inside a spliced `KeepPreAsap` payload by - /// design, and this payload structurally *is* the enclosing target — a - /// fresh `find_winner` call finds the identical winner again, - /// unconditionally, every time. Filtering this shape out of `winners` - /// (same "no-op candidate" concept `explanation.rs`'s own - /// `sketch_finding_reason` already excludes from being a finding) is - /// what keeps this terminating: this test's only assertion that matters - /// is that `run_post_asap` returns at all instead of overflowing the - /// stack. + /// `retain_exact`'s conservative fallback: the *entire target* itself, + /// unbound, carrying only an exact "kept pre-ASAP" guarantee). + /// `run_post_asap` must not treat that as a real winner: under the old + /// IR, splicing it into `export_post_asap` recursed forever (the spliced + /// payload structurally *was* the enclosing target, so every fresh + /// `find_winner` call found the identical winner again). Filtering this + /// shape out of `winners` (same "no-op candidate" concept + /// `explanation.rs`'s own `sketch_finding_reason` already excludes from + /// being a finding) is what keeps this terminating and keeps the output + /// free of a fake replacement: this test asserts both that + /// `run_post_asap` returns at all and that it reports nothing. #[tokio::test] - async fn post_asap_does_not_recurse_forever_on_a_trivial_keep_pre_asap_winner() { + async fn post_asap_does_not_recurse_forever_on_a_trivial_retain_exact_winner() { let cat = default_catalog(); let stddev_query = lower_sql( "SELECT STDDEV_POP(latency) FROM metrics", @@ -3182,12 +3168,12 @@ mod tests { let results = run_post_asap(&lowered_queries); - // A trivial keep_pre_asap winner must be filtered before it ever + // A trivial retain_exact winner must be filtered before it ever // becomes a flat `TargetReplacement` — there's no real replacement // to report for a target with no alternative at all. assert!( results.replacements.is_empty(), - "a target whose only candidate is the trivial keep_pre_asap fallback \ + "a target whose only candidate is the trivial retain_exact fallback \ shouldn't produce a flat replacement entry: {:?}", results .replacements diff --git a/crates/devtools/src/bin/show_post_asap_ir.rs b/crates/devtools/src/bin/show_post_asap_ir.rs index 18260b42f..8724d505e 100644 --- a/crates/devtools/src/bin/show_post_asap_ir.rs +++ b/crates/devtools/src/bin/show_post_asap_ir.rs @@ -3,10 +3,11 @@ // // Lowers a batch of ad-hoc SQL/PromQL queries to pre-ASAP IR, then runs the // `asap-aware-mapping` pre-ASAP → post-ASAP binding pass and prints the -// resulting **post-ASAP IR** (the sketch-bound IR: `SummaryExpr`/`SummaryNode` -// — the concrete `SummaryKind`/`SummaryParams` committed per aggregate, or -// `KeepPreAsap` for whatever the pass left untouched). See `show_pre_asap_ir` -// for the sketch-agnostic IR one layer upstream. +// resulting **post-ASAP IR** (the sketch-bound IR: an `OperatorNode` DAG in +// which `ASAPOp` operators — the concrete summary family/params committed per +// aggregate — replace the bound aggregates, while whatever the pass left +// untouched stays a plain `NonASAPOp` sub-DAG carrying an exact guarantee). +// See `show_pre_asap_ir` for the sketch-agnostic IR one layer upstream. // // File format: one query per line, prefixed with "sql>" or "promql>". // Blank lines and lines starting with '#' are ignored. @@ -20,31 +21,30 @@ // `metrics(ts, service, region, latency, bytes)` catalog — the same table // used in cross_language.rs and topk_ir.rs. -use asap_aware_mapping::replacement::keep_pre_asap; +use asap_aware_mapping::replacement::retain_exact; use asap_aware_mapping::{ - Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, + ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::query_expr::QueryExpr; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::ir::OperatorNode; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use std::io::Read; use std::rc::Rc; const ACCURACY: AccuracyTarget = AccuracyTarget::Epsilon(0.01); -/// `SketchAlgorithmStrategy::replacements` returns every candidate. This +/// `ASAPStrategies::replacements` returns every candidate. This /// debug tool prints all of them so callers can inspect the planner's choices. /// If the strategy has none, preserve the single pre-ASAP fallback output. -fn bind_all(expr: &QueryExpr) -> Result>, String> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - let candidates = SketchAlgorithmStrategy::default_cost_model() +fn bind_all(root: &Rc) -> Result>, String> { + let target = TargetSubDAG::new(root); + let candidates = ASAPStrategies::default_cost_model() .replacements(&target) .into_iter() .filter_map(|candidate| match candidate { ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDag(node), .. } => Some(node), _ => None, @@ -52,14 +52,14 @@ fn bind_all(expr: &QueryExpr) -> Result>(); if candidates.is_empty() { - Ok(vec![keep_pre_asap(&root).map_err(|e| e.to_string())?]) + Ok(vec![retain_exact(root).map_err(|e| e.to_string())?]) } else { Ok(candidates) } } -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn catalog() -> SqlCatalog { @@ -128,7 +128,7 @@ async fn main() { Ok(candidates) => { for (index, candidate) in candidates.iter().enumerate() { println!("--- candidate {} ---", index + 1); - println!("{:#?}", candidate.expr); + println!("{:#?}", candidate.operator); } } Err(e) => println!("ERR: {e}"), @@ -149,9 +149,8 @@ mod tests { 1_000, ) .expect("query lowers to pre-ASAP IR"); - let root = Rc::new(expr.clone()); - let expected = SketchAlgorithmStrategy::default_cost_model() - .replacements(&TargetSubDAG::new(&root)) + let expected = ASAPStrategies::default_cost_model() + .replacements(&TargetSubDAG::new(&expr)) .len(); assert!(expected > 1, "fixture exposes alternative bindings"); @@ -170,14 +169,20 @@ mod tests { let candidates = bind_all(&expr).expect("binding succeeds"); assert_eq!(candidates.len(), 1); assert!(matches!( - candidates[0].expr, - asap_types::post_asap::SummaryExpr::BinaryOp { .. } + candidates[0].non_asap(), + Some(asap_types::ir::NonASAPOp::BinaryOp { .. }) )); assert!( candidates[0].guarantee.is_none(), "missing evidence must not claim a certified ratio bound" ); - asap_types::post_asap::compile_post_asap_dag(&candidates[0]) + let timed = asap_types::ir::timing::apply_lifecycle_timings( + &candidates[0], + &asap_types::ir::timing::LifecycleAssignment::default_maintained(), + &mut asap_types::ir::timing::TimingMemo::new(), + ) + .expect("the demo candidate has a legal default timing"); + asap_types::ir::export::compile_post_asap_dag(&timed) .expect("the demo candidate remains executable"); } @@ -192,9 +197,9 @@ mod tests { let candidates = bind_all(&expr).expect("binding succeeds"); assert_eq!(candidates.len(), 1); - assert!(matches!( - candidates[0].expr, - asap_types::post_asap::SummaryExpr::KeepPreAsap(_) - )); + assert!( + !candidates[0].contains_asap(), + "the whole query is kept pre-ASAP (no summary bound anywhere)" + ); } } diff --git a/crates/devtools/src/bin/show_pre_asap_ir.rs b/crates/devtools/src/bin/show_pre_asap_ir.rs index b2f60463b..bde7cfb3c 100644 --- a/crates/devtools/src/bin/show_pre_asap_ir.rs +++ b/crates/devtools/src/bin/show_pre_asap_ir.rs @@ -2,7 +2,8 @@ // (or pipe via stdin: cargo run -p asap-devtools --bin show_pre_asap_ir < queries.txt) // // Lowers a batch of ad-hoc SQL/PromQL queries to **pre-ASAP IR** (the -// sketch-agnostic intent algebra: `QueryExpr`/`AggIntent`) and prints them. +// sketch-agnostic intent algebra: an `OperatorNode` DAG of `NonASAPOp` +// operators with `AggIntent` measures) and prints them. // See `show_post_asap_ir` for the post-ASAP sketch-bound IR one layer // downstream — this tool never picks a sketch, it only shows what a query // means. @@ -17,12 +18,12 @@ // bytes)` catalog — the same table used in cross_language.rs and topk_ir.rs. use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use std::io::Read; -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn catalog() -> SqlCatalog { diff --git a/crates/devtools/src/bin/sketch_coverage.rs b/crates/devtools/src/bin/sketch_coverage.rs index e9f0ccb10..290bc88dc 100644 --- a/crates/devtools/src/bin/sketch_coverage.rs +++ b/crates/devtools/src/bin/sketch_coverage.rs @@ -9,13 +9,13 @@ // - a `SketchApproximation` candidate (a genuine sketch alternative was // found for at least one aggregate in the query — the KLL-vs-DDSketch // kind of degree of freedom), and/or -// - a `CommonSubexpressionReuse` candidate (the query shares a subtree, +// - a `CommonSubexpressionReuse` candidate (the query shares a sub-DAG, // inside itself or with another query in the same corpus, that a // build-once-and-share candidate was found for). // // `--epsilon ` (default 0.01) sets the `AccuracyTarget` every query in // every corpus lowers with. Without an approximate target, -// `SketchAlgorithmStrategy` never has a genuine sketch alternative to +// `ASAPStrategies` never has a genuine sketch alternative to // report — see `dag_export`'s own `--epsilon` doc comment for the same // point, made there per-query instead of per-run. // @@ -28,11 +28,12 @@ use asap_aware_mapping::{explain_replacements, ExplanationKind}; use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::QueryExpr; +use asap_types::ir::OperatorNode; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use std::collections::BTreeSet; +use std::rc::Rc; /// Line-based `#`/`--` comment stripping, then split on `;` — the shape every /// SQL corpus test in this repo already uses (copied from `variant_coverage` @@ -57,8 +58,8 @@ fn promql_lines(corpus: &str) -> impl Iterator { .filter(|l| !l.is_empty() && !l.starts_with('#')) } -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn dqc_catalog() -> SqlCatalog { @@ -157,7 +158,7 @@ fn root_label(id: &str) -> String { /// reachable from. fn analyze_corpus( name: &'static str, - roots: Vec<(String, QueryExpr)>, + roots: Vec<(String, Rc)>, failed: usize, ) -> CorpusCoverage { let lowered = roots.len(); diff --git a/crates/devtools/src/bin/variant_coverage.rs b/crates/devtools/src/bin/variant_coverage.rs index a96042a2e..83d797005 100644 --- a/crates/devtools/src/bin/variant_coverage.rs +++ b/crates/devtools/src/bin/variant_coverage.rs @@ -1,151 +1,142 @@ -// cargo run -p asap-lower --bin variant_coverage +// cargo run -p asap-lower --bin variant_coverage -- --data-ingestion-interval-ms 1000 // // Lowers every query in every corpus we have (PromQL + SQL), walks the -// resulting QueryExpr trees, and reports which enum variants show up — per -// corpus, then rolled up globally. Used to find the minimal QueryExpr node set. +// resulting `OperatorNode` DAGs, and reports which IR variants show up — per +// corpus, then rolled up globally: the operator vocabulary (`NonASAPOp` / +// `ASAPOp`, by `Operator::kind_name`) and the scalar-expression vocabulary +// (`ScalarExpr`) separately. Used to find the minimal IR node set. use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::QueryExpr; +use asap_types::ir::{OperatorNode, ScalarExpr}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use std::collections::BTreeSet; +use std::rc::Rc; -const ALL_VARIANTS: &[&str] = &[ +/// Every `Operator::kind_name()`: all `NonASAPOp` variants, then all `ASAPOp` +/// variants. A front end only ever emits the former; the latter are listed so +/// the "unused" report stays an honest view of the whole vocabulary. +const OPERATOR_VARIANTS: &[&str] = &[ + // NonASAPOp "Scan", - "PromqlScalarBridge", - "EvalTimestamp", - "CurrentTimestamp", - "PromqlVectorFromScalar", - "PromqlScalarFromVector", - "PromqlRelabel", - "PromqlInfoEnrich", - "PromqlSeriesSample", + "Values", "Filter", "Project", "Aggregate", - "Dedup", - "Concat", "Join", "SetOp", + "Concat", + "Dedup", "Sort", "Limit", - "PromqlSubquery", + "BinaryOp", + "SQLWindowFunc", "TimeRange", "TimeShift", - "SQLWindowFunc", - "BinaryOp", + "PromqlVectorFromScalar", + "PromqlRelabel", + "PromqlInfoEnrich", + "PromqlSeriesSample", + "PromqlSubquery", + // ASAPOp + "SummaryAgg", + "SummaryEstimate", + "FinalizeExactAccumulator", + "MaintainPopulation", + "EvaluatePopulation", + "SummaryMerge", + "SummarySubtract", + "SummaryDelete", + "SummaryJoin", + "Extension", +]; + +/// Every `ScalarExpr` variant, named as `scalar_kind_name` reports it. +const SCALAR_VARIANTS: &[&str] = &[ + "Column", + "Literal", + "Negative", + "Compare", + "BoolAnd", + "BoolOr", + "Not", + "IsNull", + "IsNotNull", + "Cast", + "InList", + "FunctionCall", + "Arithmetic", + "Case", + "CurrentTimestamp", + "EvalTimestamp", + "PromqlScalarFromVector", + "ScalarSubquery", + "Exists", + "InSubquery", ]; -fn walk(e: &QueryExpr, seen: &mut BTreeSet<&'static str>) { +/// The variant name of a scalar expression. Exhaustive on purpose: a new +/// `ScalarExpr` variant fails to compile here until it is named. +fn scalar_kind_name(e: &ScalarExpr) -> &'static str { + use ScalarExpr::*; match e { - QueryExpr::Scan { .. } => { - seen.insert("Scan"); - } - QueryExpr::PromqlScalarBridge(_) => { - seen.insert("PromqlScalarBridge"); - } - QueryExpr::EvalTimestamp => { - seen.insert("EvalTimestamp"); - } - QueryExpr::CurrentTimestamp => { - seen.insert("CurrentTimestamp"); - } - QueryExpr::PromqlVectorFromScalar(inner) => { - seen.insert("PromqlVectorFromScalar"); - walk(inner, seen); - } - QueryExpr::PromqlScalarFromVector(inner) => { - seen.insert("PromqlScalarFromVector"); - walk(inner, seen); - } - QueryExpr::PromqlRelabel { child, .. } => { - seen.insert("PromqlRelabel"); - walk(child, seen); - } - QueryExpr::PromqlInfoEnrich { child, .. } => { - seen.insert("PromqlInfoEnrich"); - walk(child, seen); - } - QueryExpr::PromqlSeriesSample { child, .. } => { - seen.insert("PromqlSeriesSample"); - walk(child, seen); - } - QueryExpr::Filter { child, .. } => { - seen.insert("Filter"); - walk(child, seen); - } - QueryExpr::Project { child, .. } => { - seen.insert("Project"); - walk(child, seen); - } - QueryExpr::Aggregate { child, .. } => { - seen.insert("Aggregate"); - walk(child, seen); - } - QueryExpr::Dedup { child, .. } => { - seen.insert("Dedup"); - walk(child, seen); - } - QueryExpr::Concat { children, .. } => { - seen.insert("Concat"); - children.iter().for_each(|c| walk(c, seen)); - } - QueryExpr::Join { left, right, .. } => { - seen.insert("Join"); - walk(left, seen); - walk(right, seen); - } - QueryExpr::SetOp { left, right, .. } => { - seen.insert("SetOp"); - walk(left, seen); - walk(right, seen); - } - QueryExpr::Sort { child, .. } => { - seen.insert("Sort"); - walk(child, seen); - } - QueryExpr::Limit { child, .. } => { - seen.insert("Limit"); - walk(child, seen); - } - QueryExpr::PromqlSubquery { child, .. } => { - seen.insert("PromqlSubquery"); - walk(child, seen); - } - QueryExpr::TimeRange { child, .. } => { - seen.insert("TimeRange"); - walk(child, seen); - } - QueryExpr::TimeShift { child, .. } => { - seen.insert("TimeShift"); - walk(child, seen); - } - QueryExpr::SQLWindowFunc { child, .. } => { - seen.insert("SQLWindowFunc"); - walk(child, seen); - } - QueryExpr::BinaryOp { lhs, rhs, .. } => { - seen.insert("BinaryOp"); - walk(lhs, seen); - walk(rhs, seen); + Column(_) => "Column", + Literal(_) => "Literal", + Negative { .. } => "Negative", + Compare { .. } => "Compare", + BoolAnd(_) => "BoolAnd", + BoolOr(_) => "BoolOr", + Not(_) => "Not", + IsNull(_) => "IsNull", + IsNotNull(_) => "IsNotNull", + Cast { .. } => "Cast", + InList { .. } => "InList", + FunctionCall { .. } => "FunctionCall", + Arithmetic { .. } => "Arithmetic", + Case { .. } => "Case", + CurrentTimestamp => "CurrentTimestamp", + EvalTimestamp => "EvalTimestamp", + PromqlScalarFromVector(_) => "PromqlScalarFromVector", + ScalarSubquery(_) => "ScalarSubquery", + Exists { .. } => "Exists", + InSubquery { .. } => "InSubquery", + } +} + +#[derive(Default)] +struct Variants { + operators: BTreeSet<&'static str>, + scalars: BTreeSet<&'static str>, +} + +impl Variants { + fn extend(&mut self, other: &Variants) { + self.operators.extend(other.operators.iter().copied()); + self.scalars.extend(other.scalars.iter().copied()); + } +} + +fn walk_scalar(e: &ScalarExpr, seen: &mut BTreeSet<&'static str>) { + seen.insert(scalar_kind_name(e)); + for child in e.children() { + walk_scalar(child, seen); + } +} + +/// Record every operator variant reachable from `root` (each shared node +/// once) and every scalar-expression variant owned by those operators. The +/// operator nodes a scalar expression reads (`scalar(v)`, subqueries) are in +/// `OperatorNode::children`, so `reachable` already covers them. +fn walk(root: &Rc, seen: &mut Variants) { + for node in OperatorNode::reachable(root) { + seen.operators.insert(node.operator.kind_name()); + if let Some(op) = node.non_asap() { + for expr in op.scalar_exprs() { + walk_scalar(expr, &mut seen.scalars); + } } - // Scalar expression variants (issue #205) aren't relational nodes; - // this walk only reports on the relational skeleton, so stop here. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} } } @@ -171,8 +162,8 @@ fn promql_lines(corpus: &str) -> impl Iterator { .filter(|l| !l.is_empty() && !l.starts_with('#')) } -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn dqc_catalog() -> SqlCatalog { @@ -237,13 +228,22 @@ struct CorpusResult { name: &'static str, lowered: usize, failed: usize, - variants: BTreeSet<&'static str>, + variants: Variants, } fn report(r: &CorpusResult) { println!("--- {} ---", r.name); println!("lowered: {}, failed: {}", r.lowered, r.failed); - println!("variants ({}): {:?}", r.variants.len(), r.variants); + println!( + "operator variants ({}): {:?}", + r.variants.operators.len(), + r.variants.operators + ); + println!( + "scalar variants ({}): {:?}", + r.variants.scalars.len(), + r.variants.scalars + ); println!(); } @@ -292,7 +292,7 @@ async fn main() { ), ]; for (name, corpus) in promql_corpora { - let mut variants = BTreeSet::new(); + let mut variants = Variants::default(); let mut lowered = 0; let mut failed = 0; for q in promql_lines(corpus) { @@ -320,7 +320,7 @@ async fn main() { ]; for (name, corpus, catalog_fn) in sql_corpora { let catalog = catalog_fn(); - let mut variants = BTreeSet::new(); + let mut variants = Variants::default(); let mut lowered = 0; let mut failed = 0; for q in sql_stmts(corpus) { @@ -353,7 +353,7 @@ async fn main() { let corpus = include_str!("../../../frontend-sql/tests/bgp_analytics/data/bgp_analytics.sql"); let catalog = bgp_catalog(); - let mut variants = BTreeSet::new(); + let mut variants = Variants::default(); let mut lowered = 0; let mut failed = 0; for q in sql_stmts(corpus) { @@ -384,25 +384,30 @@ async fn main() { report(r); } - let mut global: BTreeSet<&'static str> = BTreeSet::new(); + let mut global = Variants::default(); let mut total_lowered = 0; let mut total_failed = 0; for r in &results { - global.extend(r.variants.iter().copied()); + global.extend(&r.variants); total_lowered += r.lowered; total_failed += r.failed; } println!("=== global ==="); println!("total lowered: {total_lowered}, total failed: {total_failed}\n"); - println!("used variants ({}):", global.len()); - for v in &global { - println!(" {v}"); - } - println!("\nunused variants ({}):", ALL_VARIANTS.len() - global.len()); - for v in ALL_VARIANTS { - if !global.contains(v) { + for (label, used, all) in [ + ("operator", &global.operators, OPERATOR_VARIANTS), + ("scalar", &global.scalars, SCALAR_VARIANTS), + ] { + println!("used {label} variants ({}):", used.len()); + for v in used { + println!(" {v}"); + } + let unused: Vec<_> = all.iter().filter(|v| !used.contains(*v)).collect(); + println!("\nunused {label} variants ({}):", unused.len()); + for v in unused { println!(" {v}"); } + println!(); } } diff --git a/crates/devtools/src/lib.rs b/crates/devtools/src/lib.rs index 5e6a0208a..0316622dc 100644 --- a/crates/devtools/src/lib.rs +++ b/crates/devtools/src/lib.rs @@ -2,7 +2,7 @@ //! //! Re-exports both language paths so a caller can depend on a single crate for //! PromQL *and* SQL. Both front ends end at the canonical intent algebra via -//! the same shared [`resolve_root`](asap_types::pre_asap::resolve_root). +//! the same unified operator IR ([`asap_types::ir::OperatorNode`]). //! //! ## Dependency isolation //! @@ -23,7 +23,7 @@ pub fn lower_promql_with_data_ingestion_interval( query: &str, accuracy: asap_types::types::AccuracyTarget, interval_ms: u64, -) -> Result { +) -> Result, PromqlError> { use asap_types::workload::{ BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryRequirements, QueryWorkload, TimeSelection, diff --git a/crates/devtools/tests/cross_language.rs b/crates/devtools/tests/cross_language.rs index 0518f62b7..597362e99 100644 --- a/crates/devtools/tests/cross_language.rs +++ b/crates/devtools/tests/cross_language.rs @@ -4,7 +4,7 @@ //! canonical intent algebra**, so a post-ASAP binding rule matching on //! `AggIntent` sees one spelling regardless of source language. These tests //! are the executable spec -//! for the shared [`canonicalize`](asap_types::pre_asap::canonicalize) pass: they pin the +//! for the shared [`canonicalize`](asap_types::ir::canonicalize) pass: they pin the //! canonical heavy-hitter shape and assert both front ends reach it. //! //! A literal `lower_sql(S) == lower_promql(P)` cannot hold — the two count @@ -14,12 +14,14 @@ //! explicit inner `Aggregate([Count])`. use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::ir::{NonASAPOp, OperatorNode}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::pre_asap::{AggIntent, GroupKeys}; use asap_types::types::AccuracyTarget; +use std::rc::Rc; -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn catalog() -> SqlCatalog { @@ -39,26 +41,26 @@ fn catalog() -> SqlCatalog { ) } -async fn sql(q: &str) -> QueryExpr { +async fn sql(q: &str) -> Rc { lower_sql(q, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|e| panic!("SQL {q:?} failed to lower: {e:?}")) } -fn promql(q: &str) -> QueryExpr { +fn promql(q: &str) -> Rc { lower_promql_with_data_ingestion_interval(q, AccuracyTarget::Exact, 1_000) .unwrap_or_else(|e| panic!("PromQL {q:?} failed to lower: {e:?}")) } /// The canonical heavy-hitter shape: an outer `Aggregate([TopK{k}])` (grouped by /// `by`) over an inner `Aggregate([Count])`. Returns `(k, outer_by)`. -fn heavy_hitter(qe: &QueryExpr) -> Option<(usize, GroupKeys)> { - let QueryExpr::Aggregate { +fn heavy_hitter(qe: &OperatorNode) -> Option<(usize, GroupKeys)> { + let Some(NonASAPOp::Aggregate { reduction, measures, child, .. - } = qe + }) = qe.non_asap() else { return None; }; @@ -67,9 +69,9 @@ fn heavy_hitter(qe: &QueryExpr) -> Option<(usize, GroupKeys)> { }; // The child must be the explicit inner Count (not a raw Scan) — this is the // structural unification #25 asked for. - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { measures: inner, .. - } = child.as_ref() + }) = child.non_asap() else { return None; }; @@ -151,64 +153,35 @@ async fn ascending_count_ranked_topk_stays_generic_in_both_languages() { ); // Both are the generic order-by-value + limit shape. assert!( - matches!(&s, QueryExpr::Limit { .. }), + matches!(s.non_asap(), Some(NonASAPOp::Limit { .. })), "SQL stays a Limit: {s:?}" ); assert!( - matches!(&p, QueryExpr::Limit { .. }), + matches!(p.non_asap(), Some(NonASAPOp::Limit { .. })), "PromQL stays a Limit: {p:?}" ); } -/// Descend through a leading `Project` (the derived-table SELECT list). -fn strip_project(qe: &QueryExpr) -> &QueryExpr { - match qe { - QueryExpr::Project { child, .. } => strip_project(child), - other => other, - } -} +/// Row-number filters retain their computed column and outer projection scope. #[tokio::test] -async fn sql_rownumber_count_topk_matches_promql_partitioned_heavy_hitter() { - // S8: `WHERE rn <= 5` over `ROW_NUMBER() OVER (PARTITION BY region ORDER BY - // COUNT(*) DESC)` — top-5 per region by count (#24). It must reach the same - // partitioned heavy-hitter shape as PromQL `topk by (…) (5, count_over_time)` - // (P10): an outer TopK grouped by the partition over an explicit Count. - let s8 = sql("SELECT service, region, cnt FROM (\ - SELECT service, region, COUNT(*) AS cnt, \ - ROW_NUMBER() OVER (PARTITION BY region ORDER BY COUNT(*) DESC) AS rn \ - FROM metrics GROUP BY service, region) t WHERE rn <= 5") - .await; - let (k, by) = heavy_hitter(strip_project(&s8)).expect("S8 is a partitioned heavy-hitter"); - assert_eq!(k, 5); - assert!(!by.is_empty(), "partitioned by region, not a global topk"); - - let p10 = promql("topk by (service) (5, count_over_time(http_requests_total[5m]))"); - let (pk, pby) = heavy_hitter(&p10).expect("P10 is a partitioned heavy-hitter"); - assert_eq!(pk, 5); - assert!(!pby.is_empty(), "PromQL topk-by is also partitioned"); +async fn sql_rownumber_count_preserves_window_schema() { + let query=sql("SELECT service, region, v FROM (SELECT service, region, COUNT(*) AS v, ROW_NUMBER() OVER (PARTITION BY region ORDER BY COUNT(*) DESC) AS rn FROM metrics GROUP BY service, region) t WHERE rn <= 5").await; + query.validate_structure().unwrap(); + assert_eq!(query.schema.fields.len(), 3); + assert!(OperatorNode::reachable(&query) + .iter() + .any(|node| matches!(node.non_asap(), Some(NonASAPOp::SQLWindowFunc { .. })))); } #[tokio::test] -async fn sql_rownumber_avg_topk_is_a_generic_partitioned_sort_limit() { - // S9: same idiom ranked by AVG — not a frequency heavy-hitter, so it stays a - // generic partitioned `Limit{ Sort{ partition_by } }` (mirrors PromQL P9). - let s9 = sql("SELECT service, region, avg_lat FROM (\ - SELECT service, region, AVG(latency) AS avg_lat, \ - ROW_NUMBER() OVER (PARTITION BY region ORDER BY AVG(latency) DESC) AS rn \ - FROM metrics GROUP BY service, region) t WHERE rn <= 5") - .await; - assert!( - heavy_hitter(strip_project(&s9)).is_none(), - "AVG-ranked is not a heavy-hitter" - ); - let QueryExpr::Limit { child, .. } = strip_project(&s9) else { - panic!("expected a Limit, got {:?}", strip_project(&s9)); - }; - let QueryExpr::Sort { partition_by, .. } = child.as_ref() else { - panic!("expected a Sort under the Limit"); - }; - assert!(!partition_by.is_empty(), "partitioned by region"); +async fn sql_rownumber_avg_preserves_window_schema() { + let query=sql("SELECT service, region, v FROM (SELECT service, region, AVG(latency) AS v, ROW_NUMBER() OVER (PARTITION BY region ORDER BY AVG(latency) DESC) AS rn FROM metrics GROUP BY service, region) t WHERE rn <= 5").await; + query.validate_structure().unwrap(); + assert_eq!(query.schema.fields.len(), 3); + assert!(OperatorNode::reachable(&query) + .iter() + .any(|node| matches!(node.non_asap(), Some(NonASAPOp::SQLWindowFunc { .. })))); } #[tokio::test] diff --git a/crates/devtools/tests/viewer_contract.rs b/crates/devtools/tests/viewer_contract.rs new file mode 100644 index 000000000..b34b472fe --- /dev/null +++ b/crates/devtools/tests/viewer_contract.rs @@ -0,0 +1,128 @@ +//! `tools/dag-viewer` ↔ `asap_types::dag_export` contract: the viewer's +//! `KIND_CATEGORY_JSON` must categorize exactly the `kind` strings +//! [`asap_types::dag_export::export`] can emit — `Operator::kind_name()` of +//! every `NonASAPOp` and `ASAPOp` variant — no more (a stale kind the IR no +//! longer has) and no less (an exported kind the viewer would render +//! uncategorized). + +use std::collections::{BTreeMap, BTreeSet}; + +use asap_types::ir::{ASAPOp, NonASAPOp}; + +/// Every `NonASAPOp::kind_name()`. +const NON_ASAP_KINDS: &[&str] = &[ + "Scan", + "Values", + "Filter", + "Project", + "Aggregate", + "Join", + "SetOp", + "Concat", + "Dedup", + "Sort", + "Limit", + "BinaryOp", + "SQLWindowFunc", + "TimeRange", + "TimeShift", + "PromqlVectorFromScalar", + "PromqlRelabel", + "PromqlInfoEnrich", + "PromqlSeriesSample", + "PromqlSubquery", +]; + +/// Every `ASAPOp::kind_name()`. +const ASAP_KINDS: &[&str] = &[ + "SummaryAgg", + "SummaryEstimate", + "FinalizeExactAccumulator", + "MaintainPopulation", + "EvaluatePopulation", + "SummaryMerge", + "SummarySubtract", + "SummaryDelete", + "SummaryJoin", + "Extension", +]; + +/// Compile-time tripwire: adding an operator variant fails these exhaustive +/// matches until the matching `*_KINDS` list above is extended too. Never +/// called; the match arms are the point. +#[allow(dead_code)] +fn kind_lists_track_every_variant(non_asap: &NonASAPOp, asap: &ASAPOp) { + let listed = |name: &str, list: &[&str]| assert!(list.contains(&name)); + listed( + match non_asap { + NonASAPOp::Scan { .. } => "Scan", + NonASAPOp::Values { .. } => "Values", + NonASAPOp::Filter { .. } => "Filter", + NonASAPOp::Project { .. } => "Project", + NonASAPOp::Aggregate { .. } => "Aggregate", + NonASAPOp::Join { .. } => "Join", + NonASAPOp::SetOp { .. } => "SetOp", + NonASAPOp::Concat { .. } => "Concat", + NonASAPOp::Dedup { .. } => "Dedup", + NonASAPOp::Sort { .. } => "Sort", + NonASAPOp::Limit { .. } => "Limit", + NonASAPOp::BinaryOp { .. } => "BinaryOp", + NonASAPOp::SQLWindowFunc { .. } => "SQLWindowFunc", + NonASAPOp::TimeRange { .. } => "TimeRange", + NonASAPOp::TimeShift { .. } => "TimeShift", + NonASAPOp::PromqlVectorFromScalar(_) => "PromqlVectorFromScalar", + NonASAPOp::PromqlRelabel { .. } => "PromqlRelabel", + NonASAPOp::PromqlInfoEnrich { .. } => "PromqlInfoEnrich", + NonASAPOp::PromqlSeriesSample { .. } => "PromqlSeriesSample", + NonASAPOp::PromqlSubquery { .. } => "PromqlSubquery", + }, + NON_ASAP_KINDS, + ); + listed( + match asap { + ASAPOp::SummaryAgg { .. } => "SummaryAgg", + ASAPOp::SummaryEstimate { .. } => "SummaryEstimate", + ASAPOp::FinalizeExactAccumulator { .. } => "FinalizeExactAccumulator", + ASAPOp::MaintainPopulation { .. } => "MaintainPopulation", + ASAPOp::EvaluatePopulation { .. } => "EvaluatePopulation", + ASAPOp::SummaryMerge { .. } => "SummaryMerge", + ASAPOp::SummarySubtract { .. } => "SummarySubtract", + ASAPOp::SummaryDelete { .. } => "SummaryDelete", + ASAPOp::SummaryJoin { .. } => "SummaryJoin", + ASAPOp::Extension { .. } => "Extension", + }, + ASAP_KINDS, + ); +} + +/// The viewer's `kind -> category` table, parsed out of the JS source the +/// same way the viewer itself does (`JSON.parse(KIND_CATEGORY_JSON)`). +fn viewer_kind_categories() -> BTreeMap { + const START: &str = "const KIND_CATEGORY_JSON = `"; + let source = include_str!(concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../tools/dag-viewer/node-style.js" + )); + let json = source + .split_once(START) + .expect("node-style.js must declare KIND_CATEGORY_JSON") + .1 + .split_once("`;") + .expect("KIND_CATEGORY_JSON must be a template literal") + .0; + serde_json::from_str(json).expect("KIND_CATEGORY_JSON must be valid JSON") +} + +#[test] +fn viewer_categorizes_exactly_the_exported_node_kinds() { + let expected: BTreeSet<&str> = NON_ASAP_KINDS.iter().chain(ASAP_KINDS).copied().collect(); + assert_eq!( + expected.len(), + NON_ASAP_KINDS.len() + ASAP_KINDS.len(), + "exported kind names must be unique" + ); + let categories = viewer_kind_categories(); + let actual: BTreeSet<&str> = categories.keys().map(String::as_str).collect(); + + assert_eq!(actual, expected); +} diff --git a/crates/frontend-common/Cargo.toml b/crates/frontend-common/Cargo.toml new file mode 100644 index 000000000..18f351e6c --- /dev/null +++ b/crates/frontend-common/Cargo.toml @@ -0,0 +1,11 @@ +[package] +name = "asap-frontend-common" +version = "0.1.0" +edition = "2021" + +# Shared front-end layer: the name-based `UnresolvedOp` tree every front end +# emits, and the resolver that binds it into the unified `OperatorNode` IR. +[dependencies] +asap-types = { path = "../types" } +serde = { version = "1", features = ["derive", "rc"] } +thiserror = "2" diff --git a/crates/frontend-common/src/lib.rs b/crates/frontend-common/src/lib.rs new file mode 100644 index 000000000..78602c1b7 --- /dev/null +++ b/crates/frontend-common/src/lib.rs @@ -0,0 +1,23 @@ +//! `asap-frontend-common` — the front-end-facing, name-based operator tree +//! and its resolver into the unified IR. +//! +//! A front end builds an [`UnresolvedOp`] tree (column references are +//! name-based [`ColumnRef`](asap_types::pre_asap::ColumnRef)s) during its +//! own `interpret` step and calls [`resolve_root`], which binds every +//! reference to a positional `ColumnId` and returns the +//! [`OperatorNode`](asap_types::ir::OperatorNode) DAG. +//! +//! - [`unresolved`] — [`UnresolvedOp`] / [`UnresolvedScalar`]: the tree. +//! - [`schema_resolver`] — [`SchemaResolver`]: builds the binding schema of a +//! schemaless (PromQL) leaf from the names the query references. +//! - [`resolve`] — [`resolve_root`]: the bottom-up binding walk. + +pub mod resolve; +pub mod schema_resolver; +pub mod unresolved; + +pub use resolve::{resolve_expr, resolve_root, resolve_scalar_root, ResolveTreeError}; +pub use schema_resolver::{SchemaCatalog, SchemaResolver, UsageDerivedCatalog}; +pub use unresolved::{ + UnresolvedOp, UnresolvedPredicate, UnresolvedProjectItem, UnresolvedScalar, UnresolvedSortKey, +}; diff --git a/crates/frontend-common/src/resolve.rs b/crates/frontend-common/src/resolve.rs new file mode 100644 index 000000000..9fd60d091 --- /dev/null +++ b/crates/frontend-common/src/resolve.rs @@ -0,0 +1,1184 @@ +//! Resolve a front-end-emitted [`UnresolvedOp`] tree into the unified IR +//! ([`Rc`]): a single, shape-preserving, bottom-up walk that +//! binds every [`ColumnRef`] to a positional `ColumnId`. +//! +//! Every structural decision (reduction choice, window folds, heavy-hitter +//! recognition, ...) is the front end's; what is left here is the mechanical, +//! schema-dependent substitution. Children are resolved first; each child +//! becomes an `OperatorNode` whose derived `.schema` is the scope the parent's +//! own references resolve against, so a `JOIN`'s concatenated schema and a +//! cross-series aggregate's frozen-closed output bind to the right positions. +//! +//! Scope boundaries: `Join` / `SetOp` sides and the operators referenced from +//! scalar positions (`scalar(v)`, subqueries) are each bound as a root in +//! their own scope. A `BinaryOp` side is too, but additionally inherits the +//! label names its enclosing scope references (issue #52): the `job` in +//! `sum by (job)(a or b)` appears in neither side's own matchers. + +use std::rc::Rc; + +use thiserror::Error; + +use asap_types::ir::operator_properties::ConcatDiscriminatorKey; +use asap_types::ir::{NonASAPOp, OperatorNode, Predicate, ProjectItem, ScalarExpr, SortKey}; +use asap_types::pre_asap::column_resolution::resolve_group_keys_promql; +use asap_types::pre_asap::{ + aggregate_output_schema, resolve_column_ref, resolve_column_refs, AggIntent, ColumnId, + ColumnRef, GroupKeys, Reduction, ResolveError, Schema, SchemaDerivationError, +}; + +use crate::schema_resolver::{collect_referenced_columns, SchemaResolver}; +use crate::unresolved::{UnresolvedOp, UnresolvedScalar, UnresolvedSortKey}; + +/// Errors from resolving an [`UnresolvedOp`] tree. +#[derive(Debug, Error)] +pub enum ResolveTreeError { + /// A column reference did not resolve against its in-scope schema. + #[error("column resolution failed: {0}")] + Resolve(#[from] ResolveError), + /// Deriving the schema of an already-resolved child failed (needed to + /// resolve positional column references against it). + #[error("schema derivation failed: {0}")] + Schema(#[from] SchemaDerivationError), +} + +use asap_types::ir::canonicalize::canonicalize; + +/// Resolve the whole tree rooted at `tree`: bind every `ColumnRef` to a +/// `ColumnId` via the [`SchemaResolver`], then canonicalize the result. +pub fn resolve_root(tree: &UnresolvedOp) -> Result, ResolveTreeError> { + resolve_root_with_inherited(tree, &[]) +} + +/// [`resolve_root`] with label names inherited from an enclosing scope seeded +/// into the leaf schema (a `BinaryOp` side, a scalar operand's operator). +fn resolve_root_with_inherited( + tree: &UnresolvedOp, + inherited: &[String], +) -> Result, ResolveTreeError> { + let fallback = SchemaResolver::new().resolve_schema_with_inherited(tree, inherited); + let root = resolve(tree, &fallback)?; + let root = canonicalize(root)?; + root.validate_structure()?; + Ok(root) +} + +/// Bind `tree` as a root in its own scope, inheriting from `enclosing` the +/// label names `tree` does not reference itself (issue #52). +fn resolve_nested_root( + tree: &UnresolvedOp, + enclosing: &Schema, +) -> Result, ResolveTreeError> { + let own = collect_referenced_columns(tree); + let inherited: Vec = inherited_names(enclosing) + .into_iter() + .filter(|n| !own.contains(n)) + .collect(); + resolve_root_with_inherited(tree, &inherited) +} + +fn node(op: NonASAPOp) -> Result, ResolveTreeError> { + Ok(OperatorNode::non_asap_node(op)?) +} + +/// The generic substitution walk. `fallback` is the usage-derived schema a +/// schemaless `Scan` in this scope binds to. +fn resolve(tree: &UnresolvedOp, fallback: &Schema) -> Result, ResolveTreeError> { + use UnresolvedOp as U; + let expr = |e: &UnresolvedScalar, schema: &Schema| resolve_expr_in(e, schema, fallback); + let pred = |p: &UnresolvedScalar, schema: &Schema| { + Ok::<_, ResolveTreeError>(Predicate(expr(p, schema)?)) + }; + let sort_keys = |keys: &[UnresolvedSortKey], schema: &Schema| { + keys.iter() + .map(|k| { + Ok::<_, ResolveTreeError>(SortKey { + expr: expr(&k.expr, schema)?, + ascending: k.ascending, + nulls_first: k.nulls_first, + }) + }) + .collect::, _>>() + }; + match tree { + U::Scan { + source, + predicates, + schema, + } => { + let schema = schema.clone().unwrap_or_else(|| fallback.clone()); + let predicates = predicates + .iter() + .map(|p| pred(&p.0, &schema)) + .collect::, _>>()?; + node(NonASAPOp::Scan { + source: source.clone(), + predicates, + schema, + }) + } + + // Row expressions have no input-column scope. + U::Values { rows, schema } => { + let empty = Schema::new(Vec::new()); + let rows = rows + .iter() + .map(|row| { + row.iter() + .map(|e| expr(e, &empty)) + .collect::, _>>() + }) + .collect::, _>>()?; + node(NonASAPOp::Values { + rows, + schema: schema.clone(), + }) + } + + // A scalar at an operator position has no child scope; in practice a + // literal, so `fallback` is never consulted for a column here. + U::PromqlScalarOp { + child, + scalar, + op, + scalar_left, + return_bool, + } => { + let child = resolve(child, fallback)?; + let child = if child.schema.closed { + child + } else { + asap_types::pre_asap::schema::with_promql_series_identity(&child) + .map_err(SchemaDerivationError::InvalidScalarSignature)? + }; + let scalar = resolve_expr(scalar, &Schema::default())?; + lower_scalar_vector(child, scalar, op, *scalar_left, *return_bool) + } + U::PromqlMap { + child, + sample, + drop_metric_name, + } => { + let child = resolve(child, fallback)?; + let child = if child.schema.closed { + child + } else { + asap_types::pre_asap::schema::with_promql_series_identity(&child) + .map_err(SchemaDerivationError::InvalidScalarSignature)? + }; + let sample = resolve_expr(sample, &child.schema)?; + project_sample(child, sample, *drop_metric_name) + } + U::PromqlVectorFromScalar(inner) => { + node(NonASAPOp::PromqlVectorFromScalar(expr(inner, fallback)?)) + } + + U::PromqlRelabel { dst, value, child } => { + let child = resolve(child, fallback)?; + let value = expr(value, &child.schema)?; + node(NonASAPOp::PromqlRelabel { + dst: dst.clone(), + value, + child, + }) + } + + U::PromqlInfoEnrich { selector, child } => node(NonASAPOp::PromqlInfoEnrich { + selector: selector.clone(), + child: resolve(child, fallback)?, + }), + + U::PromqlSeriesSample { by, kind, child } => { + let child = resolve(child, fallback)?; + let by = resolve_group_keys(by, &child.schema)?; + node(NonASAPOp::PromqlSeriesSample { + by, + kind: *kind, + child, + }) + } + + U::Filter { pred: p, child } => { + let child = resolve(child, fallback)?; + let pred = pred(&p.0, &child.schema)?; + node(NonASAPOp::Filter { pred, child }) + } + + U::Project { + cols, + qualifier, + child, + } => { + let child = resolve(child, fallback)?; + let cols = cols + .iter() + .map(|item| { + Ok::<_, ResolveTreeError>(ProjectItem { + alias: item.alias.clone(), + expr: expr(&item.expr, &child.schema)?, + }) + }) + .collect::, _>>()?; + node(NonASAPOp::Project { + cols, + qualifier: qualifier.clone(), + child, + }) + } + + U::Aggregate { + reduction, + measures, + output_names, + filters, + having, + child, + } => { + let child = resolve(child, fallback)?; + let reduction = resolve_reduction(reduction, &child.schema)?; + let measures = measures + .iter() + .map(|m| resolve_agg_intent(m, &child.schema)) + .collect::, ResolveError>>()?; + let filters = filters + .iter() + .map(|p| p.as_ref().map(|p| pred(&p.0, &child.schema)).transpose()) + .collect::, _>>()?; + // HAVING is evaluated over the aggregate's own output. + let having = having + .as_ref() + .map(|h| { + let out_schema = aggregate_output_schema( + &child.schema, + &reduction, + &measures, + output_names, + )?; + pred(&h.0, &out_schema) + }) + .transpose()?; + node(NonASAPOp::Aggregate { + reduction, + measures, + output_names: output_names.clone(), + filters, + having, + child, + }) + } + + U::Dedup { cols, child } => { + let child = resolve(child, fallback)?; + let cols = resolve_column_refs(cols, &child.schema)?; + node(NonASAPOp::Dedup { cols, child }) + } + + U::Concat { + children, + discriminator_unique_key, + } => { + let children = children + .iter() + .map(|c| resolve(c, fallback)) + .collect::, _>>()?; + // Resolved against the first branch's own output schema — the one + // `output_schema`'s `Concat` arm derives the merged schema from. + let discriminator_unique_key = discriminator_unique_key + .as_ref() + .map(|key| { + let schema = &children + .first() + .ok_or(SchemaDerivationError::EmptyConcat)? + .schema; + Ok::<_, ResolveTreeError>(ConcatDiscriminatorKey::new( + resolve_column_ref(key.discriminator(), schema)?, + resolve_column_refs(key.inner_key(), schema)?, + )) + }) + .transpose()?; + node(NonASAPOp::Concat { + children, + discriminator_unique_key, + }) + } + + U::Join { + kind, + pred: p, + left, + right, + } => { + // Each branch is bound independently (different leaves / label + // sets); the predicate sees left ++ right. + let left = resolve_root_with_inherited(left, &[])?; + let right = resolve_root_with_inherited(right, &[])?; + let mut concat = left.schema.clone(); + concat.fields.extend(right.schema.fields.iter().cloned()); + let pred = pred(&p.0, &concat)?; + node(NonASAPOp::Join { + kind: kind.clone(), + pred, + left, + right, + }) + } + + U::SetOp { + kind, + all, + left, + right, + } => node(NonASAPOp::SetOp { + kind: kind.clone(), + all: *all, + left: resolve_root_with_inherited(left, &[])?, + right: resolve_root_with_inherited(right, &[])?, + }), + + U::Sort { + keys, + partition_by, + child, + } => { + let child = resolve(child, fallback)?; + let keys = sort_keys(keys, &child.schema)?; + let partition_by = resolve_group_keys(partition_by, &child.schema)?; + node(NonASAPOp::Sort { + keys, + partition_by, + child, + }) + } + + U::Limit { + n, + offset, + partition_by, + child, + } => { + let child = resolve(child, fallback)?; + let partition_by = resolve_group_keys(partition_by, &child.schema)?; + node(NonASAPOp::Limit { + n: *n, + offset: *offset, + partition_by, + child, + }) + } + + U::PromqlSubquery { + range, + resolution, + child, + } => node(NonASAPOp::PromqlSubquery { + range: *range, + resolution: *resolution, + child: resolve(child, fallback)?, + }), + + U::TimeRange { range, kind, child } => node(NonASAPOp::TimeRange { + range: *range, + kind: *kind, + child: resolve(child, fallback)?, + }), + + U::TimeShift { shift, child } => node(NonASAPOp::TimeShift { + shift: *shift, + child: resolve(child, fallback)?, + }), + + U::SQLWindowFunc { + func, + args, + partition_by, + order_by, + frame, + output_name, + child, + } => { + let child = resolve(child, fallback)?; + let args = args + .iter() + .map(|a| expr(a, &child.schema)) + .collect::, _>>()?; + let partition_by = resolve_group_keys(partition_by, &child.schema)?; + let order_by = sort_keys(order_by, &child.schema)?; + node(NonASAPOp::SQLWindowFunc { + func: func.clone(), + args, + partition_by, + order_by, + frame: frame.clone(), + output_name: output_name.clone(), + child, + }) + } + + U::BinaryOp { + operator, + return_bool, + lhs, + rhs, + } => { + // The two sides may scan different metrics with different label + // sets, so each resolves against its OWN bound schema — but still + // sees the label names the enclosing scope references (issue #52). + // The inherited set is computed over the whole `BinaryOp`, so one + // side's own labels are not conjured into the other. + let own = collect_referenced_columns(tree); + let inherited: Vec = inherited_names(fallback) + .into_iter() + .filter(|n| !own.contains(n)) + .collect(); + node(NonASAPOp::BinaryOp { + operator: operator.clone(), + return_bool: *return_bool, + lhs: resolve_root_with_inherited(lhs, &inherited)?, + rhs: resolve_root_with_inherited(rhs, &inherited)?, + }) + } + } +} + +/// The label names an enclosing scope's schema carries beyond the `(ts, +/// value)` floor. +fn inherited_names(schema: &Schema) -> Vec { + schema + .fields + .iter() + .filter(|c| c.name != "ts" && c.name != "value") + .map(|c| c.name.clone()) + .collect() +} + +/// Resolve a name-based scalar expression against `schema`. Operators it +/// reads (`scalar(v)`, subqueries) are bound as roots in their own scope, +/// inheriting `schema`'s label names. +pub fn resolve_expr( + expr: &UnresolvedScalar, + schema: &Schema, +) -> Result { + resolve_expr_in(expr, schema, schema) +} + +/// [`resolve_expr`] where the operators the expression reads inherit from +/// `enclosing` (the owning root's fallback schema) rather than from `schema`. +fn resolve_expr_in( + expr: &UnresolvedScalar, + schema: &Schema, + enclosing: &Schema, +) -> Result { + use UnresolvedScalar as S; + let bx = |e: &UnresolvedScalar| -> Result, ResolveTreeError> { + Ok(Box::new(resolve_expr_in(e, schema, enclosing)?)) + }; + let each = |es: &[UnresolvedScalar]| -> Result, ResolveTreeError> { + es.iter() + .map(|e| resolve_expr_in(e, schema, enclosing)) + .collect() + }; + let op = |o: &UnresolvedOp| resolve_nested_root(o, enclosing); + Ok(match expr { + S::Column(c) => ScalarExpr::Column(resolve_column_ref(c, schema)?), + S::Literal(s) => ScalarExpr::Literal(s.clone()), + S::EvalTimestamp => ScalarExpr::EvalTimestamp, + S::CurrentTimestamp => ScalarExpr::CurrentTimestamp, + S::Negative { expr, semantics } => ScalarExpr::Negative { + expr: bx(expr)?, + semantics: *semantics, + }, + S::Compare { + left, + op, + right, + semantics, + } => ScalarExpr::Compare { + left: bx(left)?, + op: op.clone(), + right: bx(right)?, + semantics: *semantics, + }, + S::BoolAnd(v) => ScalarExpr::BoolAnd(each(v)?), + S::BoolOr(v) => ScalarExpr::BoolOr(each(v)?), + S::Not(e) => ScalarExpr::Not(bx(e)?), + S::IsNull(e) => ScalarExpr::IsNull(bx(e)?), + S::IsNotNull(e) => ScalarExpr::IsNotNull(bx(e)?), + S::Cast { expr, to, try_cast } => ScalarExpr::Cast { + expr: bx(expr)?, + to: to.clone(), + try_cast: *try_cast, + }, + S::InList { + expr, + list, + negated, + } => ScalarExpr::InList { + expr: bx(expr)?, + list: each(list)?, + negated: *negated, + }, + S::FunctionCall { name, args } => ScalarExpr::FunctionCall { + name: name.clone(), + args: each(args)?, + }, + S::Arithmetic { + op, + left, + right, + semantics, + } => ScalarExpr::Arithmetic { + op: op.clone(), + left: bx(left)?, + right: bx(right)?, + semantics: *semantics, + }, + S::Case { + operand, + branches, + else_expr, + } => ScalarExpr::Case { + operand: operand.as_deref().map(bx).transpose()?, + branches: branches + .iter() + .map(|(w, t)| { + Ok(( + resolve_expr_in(w, schema, enclosing)?, + resolve_expr_in(t, schema, enclosing)?, + )) + }) + .collect::, ResolveTreeError>>()?, + else_expr: else_expr.as_deref().map(bx).transpose()?, + }, + S::PromqlScalarFromVector(o) => ScalarExpr::PromqlScalarFromVector(op(o)?), + S::ScalarSubquery(o) => ScalarExpr::ScalarSubquery(op(o)?), + S::Exists { subquery, negated } => ScalarExpr::Exists { + subquery: op(subquery)?, + negated: *negated, + }, + S::InSubquery { + expr, + subquery, + negated, + } => ScalarExpr::InSubquery { + expr: bx(expr)?, + subquery: op(subquery)?, + negated: *negated, + }, + }) +} + +/// Resolve name-based group keys positionally, preserving `by`/`without`. +fn resolve_group_keys( + keys: &GroupKeys, + schema: &Schema, +) -> Result, ResolveError> { + let ids = resolve_column_refs(keys.keys(), schema)?; + Ok(if keys.is_without() { + GroupKeys::without(ids) + } else { + GroupKeys::by(ids) + }) +} + +/// Resolve a name-based reduction. Uses [`resolve_group_keys_promql`] rather +/// than the strict [`resolve_group_keys`]: a key absent from a **closed** +/// schema (the output of a nested cross-series aggregate that collapsed the +/// label) is provably absent from every row, so PromQL drops it from the +/// grouping rather than rejecting the query (issue #53) — `sum(sum by (group) +/// (m)) by (job)`. SQL `GROUP BY` keys are always present, so the lenient +/// path is a no-op difference there. +fn resolve_reduction( + reduction: &Reduction, + schema: &Schema, +) -> Result, ResolveError> { + Ok(match reduction { + Reduction::Reduce(by) => { + let ids = resolve_group_keys_promql(by.keys(), schema)?; + Reduction::Reduce(if by.is_without() { + GroupKeys::without(ids) + } else { + GroupKeys::by(ids) + }) + } + Reduction::PerEntity => Reduction::PerEntity, + }) +} + +/// Resolve a name-based aggregate intent: every `col: Option` +/// resolves to `Option` (`None` stays `None`, the sample-value +/// convention); every other field carries through unchanged. +fn resolve_agg_intent( + intent: &AggIntent, + schema: &Schema, +) -> Result, ResolveError> { + let col = |c: &Option| -> Result, ResolveError> { + c.as_ref() + .map(|r| resolve_column_ref(r, schema)) + .transpose() + }; + Ok(match intent { + AggIntent::Count { accuracy } => AggIntent::Count { + accuracy: accuracy.clone(), + }, + AggIntent::PearsonCorr { left, right } => AggIntent::PearsonCorr { + left: resolve_column_ref(left, schema)?, + right: resolve_column_ref(right, schema)?, + }, + AggIntent::Sum { col: c } => AggIntent::Sum { col: col(c)? }, + AggIntent::Min { col: c } => AggIntent::Min { col: col(c)? }, + AggIntent::Max { col: c } => AggIntent::Max { col: col(c)? }, + AggIntent::Avg { col: c } => AggIntent::Avg { col: col(c)? }, + AggIntent::StdDev { col: c, population } => AggIntent::StdDev { + col: col(c)?, + population: *population, + }, + AggIntent::Variance { col: c, population } => AggIntent::Variance { + col: col(c)?, + population: *population, + }, + AggIntent::Quantile { + col: c, + q, + accuracy, + } => AggIntent::Quantile { + col: col(c)?, + q: *q, + accuracy: accuracy.clone(), + }, + AggIntent::TopK { k, accuracy } => AggIntent::TopK { + k: *k, + accuracy: accuracy.clone(), + }, + AggIntent::Cardinality { cols, accuracy } => AggIntent::Cardinality { + cols: cols + .iter() + .map(|c| resolve_column_ref(c, schema)) + .collect::>()?, + accuracy: accuracy.clone(), + }, + AggIntent::FrequencyL2 { col: c, accuracy } => AggIntent::FrequencyL2 { + col: col(c)?, + accuracy: accuracy.clone(), + }, + AggIntent::FrequencyEntropy { col: c, accuracy } => AggIntent::FrequencyEntropy { + col: col(c)?, + accuracy: accuracy.clone(), + }, + AggIntent::Rate => AggIntent::Rate, + AggIntent::IRate => AggIntent::IRate, + AggIntent::Increase => AggIntent::Increase, + AggIntent::Changes => AggIntent::Changes, + AggIntent::Delta => AggIntent::Delta, + AggIntent::IDelta => AggIntent::IDelta, + AggIntent::Deriv => AggIntent::Deriv, + AggIntent::Resets => AggIntent::Resets, + AggIntent::PredictLinear { seconds } => AggIntent::PredictLinear { seconds: *seconds }, + AggIntent::DoubleExpSmoothing { smoothing, trend } => AggIntent::DoubleExpSmoothing { + smoothing: *smoothing, + trend: *trend, + }, + AggIntent::HistogramCount => AggIntent::HistogramCount, + AggIntent::HistogramSum => AggIntent::HistogramSum, + AggIntent::HistogramAvg => AggIntent::HistogramAvg, + AggIntent::HistogramStdDev => AggIntent::HistogramStdDev, + AggIntent::HistogramStdVar => AggIntent::HistogramStdVar, + AggIntent::HistogramFraction { lower, upper } => AggIntent::HistogramFraction { + lower: *lower, + upper: *upper, + }, + AggIntent::HistogramQuantile { q, le } => AggIntent::HistogramQuantile { + q: *q, + le: resolve_column_ref(le, schema)?, + }, + AggIntent::Math(f) => AggIntent::Math(f.clone()), + AggIntent::Absent => AggIntent::Absent, + AggIntent::AbsentOverTime => AggIntent::AbsentOverTime, + AggIntent::PresentOverTime => AggIntent::PresentOverTime, + AggIntent::TimeFn(f) => AggIntent::TimeFn(*f), + AggIntent::Group => AggIntent::Group, + AggIntent::CountValues { label } => AggIntent::CountValues { + label: label.clone(), + }, + AggIntent::LastOverTime => AggIntent::LastOverTime, + AggIntent::FirstOverTime => AggIntent::FirstOverTime, + AggIntent::MadOverTime => AggIntent::MadOverTime, + AggIntent::TsOfMinOverTime => AggIntent::TsOfMinOverTime, + AggIntent::TsOfMaxOverTime => AggIntent::TsOfMaxOverTime, + AggIntent::TsOfFirstOverTime => AggIntent::TsOfFirstOverTime, + AggIntent::TsOfLastOverTime => AggIntent::TsOfLastOverTime, + AggIntent::Extension { ext_kind, payload } => AggIntent::Extension { + ext_kind: ext_kind.clone(), + payload: payload.clone(), + }, + }) +} + +/// Resolve a standalone scalar in an empty column scope; plan reads retain their own scope. +pub fn resolve_scalar_root(tree: &UnresolvedScalar) -> Result { + let resolved = resolve_expr(tree, &Schema::default())?; + resolved.scalar_type(&Schema::default())?; + Ok(resolved) +} + +fn lower_scalar_vector( + child: Rc, + scalar: ScalarExpr, + op: &asap_types::pre_asap::BinaryOpKind, + scalar_left: bool, + return_bool: bool, +) -> Result, ResolveTreeError> { + use asap_types::ir::ExprSemantics; + use asap_types::pre_asap::{BinaryOpKind, DataType, ScalarValue}; + let value = child + .schema + .column_id("value") + .or_else(|| { + child + .schema + .fields + .iter() + .enumerate() + .filter(|(i, f)| { + Some(*i) != child.schema.time_index + && matches!(f.plain_dtype(), Some(DataType::Float64 | DataType::Int64)) + }) + .map(|(i, _)| i) + .next_back() + }) + .ok_or_else(|| { + SchemaDerivationError::InvalidScalarSignature("vector has no numeric sample".into()) + })?; + let sample = ScalarExpr::Column(value); + let (left, right) = if scalar_left { + (scalar, sample) + } else { + (sample, scalar) + }; + let semantics = ExprSemantics::Promql; + let computed = match op { + BinaryOpKind::Arithmetic(op) => ScalarExpr::Arithmetic { + op: op.clone(), + left: Box::new(left), + right: Box::new(right), + semantics, + }, + BinaryOpKind::Compare(op) => { + let predicate = ScalarExpr::Compare { + op: op.clone(), + left: Box::new(left), + right: Box::new(right), + semantics, + }; + if !return_bool { + return node(NonASAPOp::Filter { + child, + pred: Predicate(predicate), + }); + } + ScalarExpr::Case { + operand: None, + branches: vec![(predicate, ScalarExpr::Literal(ScalarValue::Float64(1.0)))], + else_expr: Some(Box::new(ScalarExpr::Literal(ScalarValue::Float64(0.0)))), + } + } + BinaryOpKind::Set(_) => { + return Err(SchemaDerivationError::InvalidScalarSignature( + "set operators require two vectors".into(), + ) + .into()) + } + }; + project_sample(child, computed, true) +} + +fn project_sample( + child: Rc, + computed: ScalarExpr, + drop_metric_name: bool, +) -> Result, ResolveTreeError> { + let value = asap_types::pre_asap::column_resolution::resolve_column_ref( + &ColumnRef::SampleValue, + &child.schema, + )?; + let cols = child + .schema + .fields + .iter() + .enumerate() + .filter(|(_, f)| !drop_metric_name || f.name != "__name__") + .map(|(i, f)| { + let expr = if i == value { + computed.clone() + } else if drop_metric_name + && f.name == asap_types::pre_asap::schema::PROMQL_SERIES_IDENTITY + { + ScalarExpr::FunctionCall { + name: "promql_drop_metric_name".into(), + args: vec![ScalarExpr::Column(i)], + } + } else { + ScalarExpr::Column(i) + }; + ProjectItem { + alias: Some(f.name.clone()), + expr, + } + }) + .collect(); + node(NonASAPOp::Project { + child, + cols, + qualifier: None, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::unresolved::UnresolvedPredicate; + use asap_types::ir::BinaryOperator; + use asap_types::ir::ExprSemantics; + use asap_types::pre_asap::{ + BinaryOpKind, CompareOpKind, DataType, Field, JoinKind, PromQLVectorSetOpKind, ScalarValue, + Source, VectorMatch, + }; + use asap_types::types::AccuracyTarget; + + fn scan(metric: &str) -> UnresolvedOp { + UnresolvedOp::Scan { + source: Source::TimeSeries { + metric: metric.into(), + }, + predicates: vec![], + schema: None, + } + } + + fn named(n: &str) -> UnresolvedScalar { + UnresolvedScalar::Column(ColumnRef::Named(n.into())) + } + + fn eq_lit(col: UnresolvedScalar, v: &str) -> UnresolvedScalar { + UnresolvedScalar::Compare { + left: Box::new(col), + op: CompareOpKind::Eq, + right: Box::new(UnresolvedScalar::Literal(ScalarValue::Utf8(v.into()))), + semantics: ExprSemantics::Promql, + } + } + + fn binary(kind: BinaryOpKind, vector_match: Option) -> BinaryOperator { + BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind, + vector_match, + } + } + + // Both sides resolve with qualifiers; an unknown right input is an error. + #[test] + fn resolve_pearson_corr_inputs() { + let schema = Schema::new(vec![ + Field::plain("x", DataType::Float64, true).with_table("a"), + Field::plain("x", DataType::Float64, true).with_table("b"), + ]); + let intent = AggIntent::PearsonCorr { + left: ColumnRef::Qualified { + table: "a".into(), + name: "x".into(), + }, + right: ColumnRef::Qualified { + table: "b".into(), + name: "x".into(), + }, + }; + assert_eq!( + resolve_agg_intent(&intent, &schema).unwrap(), + AggIntent::PearsonCorr { left: 0, right: 1 } + ); + let missing = AggIntent::PearsonCorr { + left: ColumnRef::Qualified { + table: "a".into(), + name: "x".into(), + }, + right: ColumnRef::Named("missing".into()), + }; + assert!(resolve_agg_intent(&missing, &schema).is_err()); + } + + // Every leg resolves independently, qualifiers included; one unknown leg + // fails rather than silently shortening the tuple. + #[test] + fn resolve_distinct_tuple_columns() { + let schema = Schema::new(vec![ + Field::plain("k", DataType::Int64, true).with_table("a"), + Field::plain("k", DataType::Int64, true).with_table("b"), + ]); + let qualified = |table: &str| ColumnRef::Qualified { + table: table.into(), + name: "k".into(), + }; + let intent = AggIntent::Cardinality { + cols: vec![qualified("b"), qualified("a")], + accuracy: AccuracyTarget::Exact, + }; + assert_eq!( + resolve_agg_intent(&intent, &schema).unwrap(), + AggIntent::Cardinality { + cols: vec![1, 0], + accuracy: AccuracyTarget::Exact, + } + ); + let missing = AggIntent::Cardinality { + cols: vec![qualified("a"), ColumnRef::Named("missing".into())], + accuracy: AccuracyTarget::Exact, + }; + assert!(resolve_agg_intent(&missing, &schema).is_err()); + } + + // ` > `: the bridged literal comes through unchanged, the + // vector side binds positionally, the `VectorMatch` survives untouched, and + // the node's schema follows the vector side. + #[test] + fn scalar_comparison_preserves_vector_values_and_labels() { + let unresolved = UnresolvedOp::PromqlScalarOp { + child: Rc::new(scan("up")), + scalar: UnresolvedScalar::Literal(ScalarValue::Float64(1.0)), + op: BinaryOpKind::Compare(CompareOpKind::Gt), + scalar_left: true, + return_bool: false, + }; + let resolved = resolve_root(&unresolved).unwrap(); + let NonASAPOp::Filter { + child, + pred: Predicate(ScalarExpr::Compare { left, right, .. }), + } = resolved.expect_non_asap() + else { + panic!("expected Filter") + }; + assert_eq!(**left, ScalarExpr::literal_f64(1.0)); + assert_eq!( + **right, + ScalarExpr::Column(child.schema.column_id("value").unwrap()) + ); + assert_eq!(resolved.schema, child.schema); + assert!(resolved.schema.has_promql_series_identity()); + } + + // A `Concat` discriminator column referenced nowhere else, over a + // schemaless first branch, resolves to the branch's own positional ids. + #[test] + fn resolve_root_seeds_and_resolves_an_otherwise_unreferenced_discriminator_column() { + let unresolved = UnresolvedOp::concat_with_discriminator( + vec![scan("m"), scan("m")], + ColumnRef::Named("phi".into()), + vec![ColumnRef::Named("host".into())], + ); + + let resolved = resolve_root(&unresolved).expect("resolves"); + let NonASAPOp::Concat { + children, + discriminator_unique_key, + } = resolved.expect_non_asap() + else { + panic!("expected a resolved Concat, got {resolved:?}"); + }; + let schema = &children[0].schema; + let key = discriminator_unique_key + .as_ref() + .expect("discriminator key survives resolution"); + assert_eq!(*key.discriminator(), schema.column_id("phi").unwrap()); + assert_eq!( + key.inner_key().to_vec(), + vec![schema.column_id("host").unwrap()] + ); + } + + // `sum by (job)(a or b)`: each `BinaryOp` side binds in its own scope but + // inherits the enclosing aggregate's group key (issue #52). + #[test] + fn binary_op_sides_inherit_enclosing_group_keys() { + let unresolved = UnresolvedOp::Aggregate { + reduction: Reduction::by(vec![ColumnRef::Named("job".into())]), + measures: vec![AggIntent::Sum { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::new(UnresolvedOp::BinaryOp { + operator: binary(BinaryOpKind::Set(PromQLVectorSetOpKind::Or), None), + return_bool: false, + lhs: Rc::new(scan("a")), + rhs: Rc::new(scan("b")), + }), + }; + let resolved = resolve_root(&unresolved).expect("resolves"); + let NonASAPOp::Aggregate { + reduction, child, .. + } = resolved.expect_non_asap() + else { + panic!("expected Aggregate"); + }; + let NonASAPOp::BinaryOp { lhs, rhs, .. } = child.expect_non_asap() else { + panic!("expected BinaryOp"); + }; + let job = lhs.schema.column_id("job").expect("lhs sees job"); + assert_eq!(rhs.schema.column_id("job"), Some(job)); + assert_eq!(reduction.expect_reduce().keys(), &[job]); + assert_eq!(resolved.schema.fields[0].name, "job"); + } + + // HAVING binds against the aggregate's output, not its input. + #[test] + fn having_resolves_against_aggregate_output() { + let input = Schema::new(vec![ + Field::plain("k", DataType::Utf8, false), + Field::plain("v", DataType::Float64, false), + ]); + let unresolved = UnresolvedOp::Aggregate { + reduction: Reduction::by(vec![ColumnRef::Named("k".into())]), + measures: vec![AggIntent::Sum { + col: Some(ColumnRef::Named("v".into())), + }], + output_names: vec!["total".into()], + filters: vec![], + having: Some(UnresolvedPredicate(UnresolvedScalar::Compare { + left: Box::new(named("total")), + op: CompareOpKind::Gt, + right: Box::new(UnresolvedScalar::Literal(ScalarValue::Float64(1.0))), + semantics: ExprSemantics::Sql, + })), + child: Rc::new(UnresolvedOp::Scan { + source: Source::Table { + table_ref: "t".into(), + }, + predicates: vec![], + schema: Some(input), + }), + }; + let resolved = resolve_root(&unresolved).expect("resolves"); + let NonASAPOp::Aggregate { + having: Some(Predicate(ScalarExpr::Compare { left, .. })), + .. + } = resolved.expect_non_asap() + else { + panic!("expected Aggregate with HAVING"); + }; + assert_eq!(**left, ScalarExpr::Column(1)); + assert_eq!(resolved.schema.fields[1].name, "total"); + } + + // A join predicate binds against left ++ right; a qualified reference + // picks the right side even when both inputs share the column name. + #[test] + fn join_predicate_resolves_against_left_then_right() { + let side = |table: &str| UnresolvedOp::Scan { + source: Source::Table { + table_ref: table.into(), + }, + predicates: vec![], + schema: Some(Schema::new(vec![ + Field::plain("k", DataType::Int64, false).with_table(table) + ])), + }; + let qualified = |table: &str| { + UnresolvedScalar::Column(ColumnRef::Qualified { + table: table.into(), + name: "k".into(), + }) + }; + let unresolved = UnresolvedOp::Join { + kind: JoinKind::Inner, + pred: UnresolvedPredicate(UnresolvedScalar::Compare { + left: Box::new(qualified("b")), + op: CompareOpKind::Eq, + right: Box::new(qualified("a")), + semantics: ExprSemantics::Sql, + }), + left: Rc::new(side("a")), + right: Rc::new(side("b")), + }; + let resolved = resolve_root(&unresolved).expect("resolves"); + let NonASAPOp::Join { + pred: Predicate(ScalarExpr::Compare { left, right, .. }), + .. + } = resolved.expect_non_asap() + else { + panic!("expected Join"); + }; + assert_eq!(**left, ScalarExpr::Column(1)); + assert_eq!(**right, ScalarExpr::Column(0)); + } + + // `m * scalar(x{a="1"})`: the operator inside the scalar operand is bound + // as a root in its own scope — its matcher label seeds its own leaf, not + // the vector side's. + #[test] + fn scalar_from_vector_operand_binds_in_its_own_scope() { + let x = UnresolvedOp::Scan { + source: Source::TimeSeries { metric: "x".into() }, + predicates: vec![UnresolvedPredicate(eq_lit(named("a"), "1"))], + schema: None, + }; + let unresolved = UnresolvedOp::PromqlScalarOp { + child: Rc::new(scan("m")), + scalar: UnresolvedScalar::PromqlScalarFromVector(Rc::new(x)), + op: BinaryOpKind::Arithmetic(asap_types::pre_asap::ArithmeticOpKind::Mul), + scalar_left: false, + return_bool: false, + }; + let resolved = resolve_root(&unresolved).unwrap(); + let NonASAPOp::Project { + child: lhs, cols, .. + } = resolved.expect_non_asap() + else { + panic!("expected Project") + }; + assert!(lhs.schema.column_id("a").is_none()); + let ScalarExpr::Arithmetic { right, .. } = &cols[1].expr else { + panic!("expected arithmetic") + }; + let ScalarExpr::PromqlScalarFromVector(inner) = right.as_ref() else { + panic!("expected scalar(v)") + }; + let a = inner + .schema + .column_id("a") + .expect("own matcher label seeded"); + let NonASAPOp::Scan { predicates, .. } = inner.expect_non_asap() else { + panic!("expected Scan"); + }; + let Predicate(ScalarExpr::Compare { left, .. }) = &predicates[0] else { + panic!("expected Compare"); + }; + assert_eq!(**left, ScalarExpr::Column(a)); + assert_eq!(resolved.schema.fields.len(), lhs.schema.fields.len()); + } + + // PromQL grouping drops a key provably absent from a closed input (#53): + // `sum(sum by (group)(m)) by (job)`. + #[test] + fn nested_aggregate_drops_absent_promql_group_key() { + let inner = UnresolvedOp::Aggregate { + reduction: Reduction::by(vec![ColumnRef::Named("group".into())]), + measures: vec![AggIntent::Sum { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::new(scan("m")), + }; + let outer = UnresolvedOp::Aggregate { + reduction: Reduction::by(vec![ColumnRef::Named("job".into())]), + measures: vec![AggIntent::Sum { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::new(inner), + }; + let resolved = resolve_root(&outer).expect("resolves"); + let NonASAPOp::Aggregate { reduction, .. } = resolved.expect_non_asap() else { + panic!("expected Aggregate"); + }; + assert!(reduction.expect_reduce().keys().is_empty()); + } +} diff --git a/crates/frontend-common/src/schema_resolver.rs b/crates/frontend-common/src/schema_resolver.rs new file mode 100644 index 000000000..ad7605fe6 --- /dev/null +++ b/crates/frontend-common/src/schema_resolver.rs @@ -0,0 +1,443 @@ +//! The **SchemaResolver** — name resolution as an explicit pass. +//! +//! [`SchemaResolver::resolve_schema`] produces the complete, self-contained +//! [`Schema`] every `ColumnId` in a schemaless leaf's scope indexes into, so +//! positional resolution in [`resolve`](crate::resolve) is total. +//! +//! The default [`UsageDerivedCatalog`] knows nothing — every schema is derived +//! purely from the query's own usage. That is the honest state for the +//! observability domain (metric label sets are open-ended). A registry-backed +//! `SchemaCatalog` is future work; only the catalog impl swaps when it lands. + +use asap_types::pre_asap::{AggIntent, ColumnRef, DataType, Field, GroupKeys, Reduction, Schema}; + +use crate::unresolved::{UnresolvedOp, UnresolvedScalar}; + +/// The DB / source-schema metadata source — resolves a source (metric / +/// table) name to its known columns. Distinct from `Scan.schema`, which is +/// the *resolved* binding schema this feeds. Even a registry-backed PromQL +/// catalog yields an **open** schema: a metric's labels are per-series and +/// time-varying, so the registry is a superset hint, not a per-row contract. +pub trait SchemaCatalog { + /// Columns known for `source`. `None` when unknown — the resolver then + /// falls back to a usage-derived column set. + fn columns_for(&self, source: &str) -> Option>; +} + +/// The default catalog: knows nothing. +pub struct UsageDerivedCatalog; + +impl SchemaCatalog for UsageDerivedCatalog { + fn columns_for(&self, _source: &str) -> Option> { + None + } +} + +/// The explicit name-resolution pass. +pub struct SchemaResolver { + catalog: C, +} + +impl Default for SchemaResolver { + fn default() -> Self { + Self::new() + } +} + +impl SchemaResolver { + pub fn new() -> Self { + Self { + catalog: UsageDerivedCatalog, + } + } +} + +impl SchemaResolver { + pub fn with_catalog(catalog: C) -> Self { + Self { catalog } + } + + /// The complete [`Schema`] in scope for a query rooted at `tree`: the + /// time axis, the synthetic `value` column, and one column per distinct + /// name referenced anywhere in the tree. + pub fn resolve_schema(&self, tree: &UnresolvedOp) -> Schema { + self.resolve_schema_with_inherited(tree, &[]) + } + + /// Like [`resolve_schema`](Self::resolve_schema), but also seeds + /// `inherited` label names referenced by an **enclosing** scope rather + /// than by `tree` itself. This is how an independently-bound `BinaryOp` + /// side still sees an outer aggregate's group keys — the `__name__` / + /// `job` in `sum by (__name__)(a or b)`, which appear in neither side's + /// own matchers (issue #52). + pub fn resolve_schema_with_inherited( + &self, + tree: &UnresolvedOp, + inherited: &[String], + ) -> Schema { + let mut columns: Vec = leftmost_scan_name(tree) + .and_then(|name| self.catalog.columns_for(name)) + .unwrap_or_else(default_leaf_columns); + + // Ensure the (ts, value) floor is present. + for floor in default_leaf_columns() { + if !columns.iter().any(|c| c.name == floor.name) { + columns.push(floor); + } + } + + // One column per referenced-but-unknown name, plus the inherited ones. + let referenced = collect_referenced_columns(tree); + for name in referenced.iter().chain(inherited) { + if !columns.iter().any(|c| c.name == *name) { + columns.push(Field::plain(name.clone(), DataType::Utf8, true)); + } + } + + let time_index = columns.iter().position(|c| c.name == "ts"); + Schema { + fields: columns, + time_index, + unique_keys: Vec::new(), + // Usage-derived (schemaless PromQL): the metric's full label set is + // open and runtime-only, so this lists only what the query references. + closed: false, + } + } +} + +/// The conventional PromQL leaf shape: `(ts: Timestamp, value: Float64)`. +fn default_leaf_columns() -> Vec { + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + ] +} + +/// Push a `ColumnRef`'s bare name (the schema-seedable identifier). `Qualified` +/// collapses to its `name`; `SampleValue`/`Wildcard` carry no name. +fn push_ref_name(c: &ColumnRef, out: &mut Vec) { + match c { + ColumnRef::Named(n) => out.push(n.clone()), + ColumnRef::Qualified { name, .. } => out.push(name.clone()), + ColumnRef::SampleValue | ColumnRef::Wildcard => {} + } +} + +/// The leftmost `Scan`'s source name, following the relational skeleton only +/// (never the operators referenced from scalar positions: those are bound in +/// their own scope). +fn leftmost_scan_name(tree: &UnresolvedOp) -> Option<&str> { + use asap_types::pre_asap::Source; + use UnresolvedOp as U; + match tree { + U::Scan { source, .. } => Some(match source { + Source::TimeSeries { metric } => metric.as_str(), + Source::Table { table_ref } => table_ref.as_str(), + }), + U::Values { .. } | U::PromqlVectorFromScalar(_) => None, + U::PromqlMap { child, .. } + | U::PromqlScalarOp { child, .. } + | U::PromqlRelabel { child, .. } + | U::PromqlInfoEnrich { child, .. } + | U::PromqlSeriesSample { child, .. } + | U::Filter { child, .. } + | U::Project { child, .. } + | U::Aggregate { child, .. } + | U::Dedup { child, .. } + | U::Sort { child, .. } + | U::Limit { child, .. } + | U::PromqlSubquery { child, .. } + | U::TimeRange { child, .. } + | U::TimeShift { child, .. } + | U::SQLWindowFunc { child, .. } => leftmost_scan_name(child), + U::Concat { children, .. } => children.first().and_then(|c| leftmost_scan_name(c)), + U::Join { left, .. } | U::SetOp { left, .. } | U::BinaryOp { lhs: left, .. } => { + leftmost_scan_name(left) + } + } +} + +/// Every distinct column name referenced anywhere in `tree` that resolves +/// positionally — every place a front end puts a name-based reference: +/// `Scan.predicates`, `Aggregate`'s `reduction`/`having`/per-measure `col`, +/// `Dedup.cols`, `PromqlSeriesSample.by`, `Filter.pred`, `Project.cols`, +/// `Sort`/`Limit`/`SQLWindowFunc` keys, `Join.pred`, `PromqlRelabel.value`, +/// `Concat.discriminator_unique_key`. Operators referenced from scalar +/// positions (`scalar(v)`, subqueries) are walked too, as the old +/// `PromqlScalarFromVector` operator child was. Sorted and deduplicated. +pub fn collect_referenced_columns(tree: &UnresolvedOp) -> Vec { + use UnresolvedOp as U; + fn named(expr: &UnresolvedScalar, out: &mut Vec) { + for c in expr.columns_referenced() { + push_ref_name(c, out); + } + for op in expr.operator_refs() { + walk(op, out); + } + } + fn group_keys(g: &GroupKeys, out: &mut Vec) { + g.keys().iter().for_each(|k| push_ref_name(k, out)); + } + fn measure_cols(measures: &[AggIntent], out: &mut Vec) { + for m in measures { + for c in m.input_cols() { + push_ref_name(&c, out); + } + } + } + fn walk(node: &UnresolvedOp, out: &mut Vec) { + match node { + U::Scan { predicates, .. } => { + for p in predicates { + named(&p.0, out); + } + } + U::Values { rows, .. } => { + for e in rows.iter().flatten() { + named(e, out); + } + } + U::Aggregate { + reduction, + measures, + having, + child, + .. + } => { + if let Reduction::Reduce(by) = reduction { + group_keys(by, out); + } + measure_cols(measures, out); + if let Some(h) = having { + named(&h.0, out); + } + walk(child, out); + } + U::Dedup { cols, child } => { + cols.iter().for_each(|c| push_ref_name(c, out)); + walk(child, out); + } + U::PromqlSeriesSample { by, child, .. } => { + group_keys(by, out); + walk(child, out); + } + U::Filter { pred, child } => { + named(&pred.0, out); + walk(child, out); + } + U::Project { cols, child, .. } => { + for item in cols { + named(&item.expr, out); + } + walk(child, out); + } + U::Sort { + keys, + partition_by, + child, + } => { + for k in keys { + named(&k.expr, out); + } + group_keys(partition_by, out); + walk(child, out); + } + U::Limit { + partition_by, + child, + .. + } => { + group_keys(partition_by, out); + walk(child, out); + } + U::SQLWindowFunc { + args, + partition_by, + order_by, + child, + .. + } => { + for a in args { + named(a, out); + } + group_keys(partition_by, out); + for k in order_by { + named(&k.expr, out); + } + walk(child, out); + } + U::PromqlRelabel { value, child, .. } => { + named(value, out); + walk(child, out); + } + U::Join { + pred, left, right, .. + } => { + named(&pred.0, out); + walk(left, out); + walk(right, out); + } + U::PromqlVectorFromScalar(inner) => named(inner, out), + U::PromqlMap { child, .. } + | U::PromqlScalarOp { child, .. } + | U::PromqlInfoEnrich { child, .. } + | U::PromqlSubquery { child, .. } + | U::TimeRange { child, .. } + | U::TimeShift { child, .. } => walk(child, out), + U::Concat { + children, + discriminator_unique_key, + } => { + // An own-field `ColumnRef` must be seeded like `Dedup.cols`, or + // a discriminator column referenced nowhere else in the tree is + // absent from the fallback schema and fails `NotFound` later. + if let Some(key) = discriminator_unique_key { + push_ref_name(key.discriminator(), out); + key.inner_key().iter().for_each(|c| push_ref_name(c, out)); + } + children.iter().for_each(|c| walk(c, out)); + } + U::SetOp { left, right, .. } => { + walk(left, out); + walk(right, out); + } + U::BinaryOp { lhs, rhs, .. } => { + walk(lhs, out); + walk(rhs, out); + } + } + } + let mut out: Vec = Vec::new(); + walk(tree, &mut out); + out.sort(); + out.dedup(); + out +} + +#[cfg(test)] +mod tests { + use std::rc::Rc; + + use asap_types::pre_asap::{AggIntent, Reduction, Source}; + + use super::*; + use crate::unresolved::UnresolvedSortKey; + + fn src(name: &str) -> UnresolvedOp { + UnresolvedOp::Scan { + source: Source::TimeSeries { + metric: name.into(), + }, + predicates: vec![], + schema: None, + } + } + + // Both correlation inputs seed the usage-derived schema. + #[test] + fn pearson_corr_inputs_seed_usage_derived_schema() { + let tree = UnresolvedOp::Aggregate { + reduction: Reduction::by(vec![]), + measures: vec![AggIntent::PearsonCorr { + left: ColumnRef::Named("x".into()), + right: ColumnRef::Named("y".into()), + }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::new(src("m")), + }; + assert_eq!(collect_referenced_columns(&tree), vec!["x", "y"]); + let schema = SchemaResolver::new().resolve_schema(&tree); + assert!(schema.column_id("x").is_some()); + assert!(schema.column_id("y").is_some()); + } + + // A bare source gets exactly the (ts, value) floor. + #[test] + fn bare_source_yields_ts_value_floor() { + let schema = SchemaResolver::new().resolve_schema(&src("m")); + assert_eq!(schema.fields.len(), 2); + assert_eq!(schema.fields[0].name, "ts"); + assert_eq!(schema.fields[1].name, "value"); + assert_eq!(schema.time_index, Some(0)); + } + + // Per-group ranking keys (`topk by (host)` → `Sort.partition_by`) are + // seeded into the usage-derived leaf so they resolve positionally. + #[test] + fn sort_partition_keys_land_in_schema() { + let tree = UnresolvedOp::Sort { + keys: vec![UnresolvedSortKey { + expr: UnresolvedScalar::Column(ColumnRef::SampleValue), + ascending: false, + nulls_first: false, + }], + partition_by: GroupKeys::by(vec![ColumnRef::Named("host".into())]), + child: Rc::new(src("hits")), + }; + let schema = SchemaResolver::new().resolve_schema(&tree); + assert!(schema.column_id("host").is_some()); + } + + // `Limit.partition_by` (PromQL `topk by (..)`) is seeded like `Sort`'s. + #[test] + fn limit_partition_keys_land_in_schema() { + let tree = UnresolvedOp::Limit { + n: Some(3), + offset: 0, + partition_by: GroupKeys::by(vec![ColumnRef::Named("host".into())]), + child: Rc::new(src("hits")), + }; + let schema = SchemaResolver::new().resolve_schema(&tree); + assert!(schema.column_id("host").is_some()); + } + + // A `Concat`'s discriminator key columns, even ones referenced nowhere + // else, are seeded like `Dedup.cols` (issue #228 review). + #[test] + fn concat_discriminator_key_is_seeded_into_the_resolver_schema() { + let tree = UnresolvedOp::concat_with_discriminator( + vec![src("m")], + ColumnRef::Named("phi".into()), + vec![ColumnRef::Named("host".into())], + ); + let schema = SchemaResolver::new().resolve_schema(&tree); + assert!(schema.column_id("phi").is_some(), "discriminator seeded"); + assert!(schema.column_id("host").is_some(), "inner_key seeded"); + } + + // Inherited names are seeded alongside the tree's own references; plain + // `resolve_schema` does not conjure them (issue #52). + #[test] + fn inherited_names_are_seeded_alongside_referenced() { + let schema = + SchemaResolver::new().resolve_schema_with_inherited(&src("m"), &["__name__".into()]); + assert!(schema.column_id("__name__").is_some()); + let plain = SchemaResolver::new().resolve_schema(&src("m")); + assert!(plain.column_id("__name__").is_none()); + } + + // A catalog-known source supplies its base columns, typed as the catalog says. + #[test] + fn custom_catalog_supplies_base_columns() { + struct FixedCatalog; + impl SchemaCatalog for FixedCatalog { + fn columns_for(&self, source: &str) -> Option> { + (source == "known").then(|| { + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("datacenter", DataType::Utf8, false), + ] + }) + } + } + let schema = SchemaResolver::with_catalog(FixedCatalog).resolve_schema(&src("known")); + let dc = schema + .column_id("datacenter") + .and_then(|id| schema.fields.get(id)); + assert!(matches!(dc, Some(c) if !c.nullable)); + } +} diff --git a/crates/frontend-common/src/unresolved.rs b/crates/frontend-common/src/unresolved.rs new file mode 100644 index 000000000..0bca427cc --- /dev/null +++ b/crates/frontend-common/src/unresolved.rs @@ -0,0 +1,374 @@ +//! The front-end-emitted, name-based operator tree: a mirror of the unified +//! IR ([`NonASAPOp`](asap_types::ir::NonASAPOp) / [`ScalarExpr`](asap_types::ir::ScalarExpr)) +//! before name resolution. +//! +//! Differences from the resolved IR, and nothing else: +//! - every `ColumnId` is a name-based [`ColumnRef`]; +//! - `Scan.schema` is `Option` — a front end knows the schema only for +//! a catalog-backed SQL leaf; `None` (PromQL) defers to the +//! [`SchemaResolver`](crate::schema_resolver::SchemaResolver); +//! - children are `Rc` rather than `Rc` — no +//! derived schema exists yet. + +use std::rc::Rc; +use std::time::Duration; + +use serde::{Deserialize, Serialize}; + +use asap_types::ir::operator_properties::ConcatDiscriminatorKey; +use asap_types::ir::BinaryOperator; +use asap_types::ir::{ExprSemantics, TimeRangeKind}; +use asap_types::pre_asap::{ + AggIntent, ArithmeticOpKind, ColumnRef, CompareOpKind, DataType, GroupKeys, InfoMatcher, + JoinKind, Reduction, RelationalSetOpKind, SampleKind, ScalarValue, Schema, Source, TimeShift, + WindowFrame, WindowFuncKind, +}; + +/// A row-level filter predicate (WHERE clause / PromQL label matcher). +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct UnresolvedPredicate(pub UnresolvedScalar); + +/// One item in a SELECT projection list. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct UnresolvedProjectItem { + pub alias: Option, + pub expr: UnresolvedScalar, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct UnresolvedSortKey { + pub expr: UnresolvedScalar, + pub ascending: bool, + pub nulls_first: bool, +} + +/// A name-based scalar expression; see +/// [`ScalarExpr`](asap_types::ir::ScalarExpr) for the meaning of each variant. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum UnresolvedScalar { + Column(ColumnRef), + Literal(ScalarValue), + Negative { + expr: Box, + semantics: ExprSemantics, + }, + Compare { + left: Box, + op: CompareOpKind, + right: Box, + semantics: ExprSemantics, + }, + BoolAnd(Vec), + BoolOr(Vec), + Not(Box), + IsNull(Box), + IsNotNull(Box), + Cast { + expr: Box, + to: DataType, + try_cast: bool, + }, + InList { + expr: Box, + list: Vec, + negated: bool, + }, + FunctionCall { + name: String, + args: Vec, + }, + Arithmetic { + op: ArithmeticOpKind, + left: Box, + right: Box, + semantics: ExprSemantics, + }, + Case { + operand: Option>, + branches: Vec<(UnresolvedScalar, UnresolvedScalar)>, + else_expr: Option>, + }, + CurrentTimestamp, + EvalTimestamp, + /// PromQL `scalar(v)`. The operator is resolved as a root in its own scope. + PromqlScalarFromVector(Rc), + ScalarSubquery(Rc), + Exists { + subquery: Rc, + negated: bool, + }, + InSubquery { + expr: Box, + subquery: Rc, + negated: bool, + }, +} + +/// The name-based operator tree; see [`NonASAPOp`](asap_types::ir::NonASAPOp) +/// for the meaning of each variant. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum UnresolvedOp { + Scan { + source: Source, + predicates: Vec, + /// `Some` for a catalog-backed (SQL) leaf; `None` defers to the + /// usage-derived schema resolver. + schema: Option, + }, + Values { + rows: Vec>, + schema: Schema, + }, + Filter { + pred: UnresolvedPredicate, + child: Rc, + }, + Project { + cols: Vec, + qualifier: Option, + child: Rc, + }, + Aggregate { + reduction: Reduction, + measures: Vec>, + output_names: Vec, + filters: Vec>, + having: Option, + child: Rc, + }, + Join { + kind: JoinKind, + pred: UnresolvedPredicate, + left: Rc, + right: Rc, + }, + SetOp { + kind: RelationalSetOpKind, + all: bool, + left: Rc, + right: Rc, + }, + Concat { + children: Vec>, + discriminator_unique_key: Option>, + }, + Dedup { + cols: Vec, + child: Rc, + }, + Sort { + keys: Vec, + partition_by: GroupKeys, + child: Rc, + }, + Limit { + n: Option, + offset: usize, + partition_by: GroupKeys, + child: Rc, + }, + BinaryOp { + operator: BinaryOperator, + return_bool: bool, + lhs: Rc, + rhs: Rc, + }, + SQLWindowFunc { + func: WindowFuncKind, + args: Vec, + partition_by: GroupKeys, + order_by: Vec, + frame: Option, + output_name: String, + child: Rc, + }, + TimeRange { + range: Duration, + kind: TimeRangeKind, + child: Rc, + }, + TimeShift { + shift: TimeShift, + child: Rc, + }, + PromqlVectorFromScalar(UnresolvedScalar), + PromqlRelabel { + dst: String, + value: UnresolvedScalar, + child: Rc, + }, + PromqlInfoEnrich { + selector: Vec, + child: Rc, + }, + PromqlSeriesSample { + by: GroupKeys, + kind: SampleKind, + child: Rc, + }, + PromqlSubquery { + range: Duration, + resolution: Option, + child: Rc, + }, + /// Bind the complete vector schema before lowering to Project or Filter. + /// Frontend-only expansion to a projection preserving the complete series identity. + PromqlMap { + child: Rc, + sample: UnresolvedScalar, + drop_metric_name: bool, + }, + PromqlScalarOp { + child: Rc, + scalar: UnresolvedScalar, + op: asap_types::pre_asap::BinaryOpKind, + scalar_left: bool, + return_bool: bool, + }, +} + +impl UnresolvedScalar { + /// The direct scalar sub-expressions (not the operators this expression + /// reads — see [`operator_refs`](Self::operator_refs)). + pub fn children(&self) -> Vec<&UnresolvedScalar> { + use UnresolvedScalar::*; + match self { + Column(_) + | Literal(_) + | CurrentTimestamp + | EvalTimestamp + | PromqlScalarFromVector(_) + | ScalarSubquery(_) + | Exists { .. } => vec![], + Negative { expr, .. } + | Not(expr) + | IsNull(expr) + | IsNotNull(expr) + | Cast { expr, .. } + | InSubquery { expr, .. } => vec![expr], + Compare { left, right, .. } | Arithmetic { left, right, .. } => vec![left, right], + BoolAnd(parts) | BoolOr(parts) => parts.iter().collect(), + InList { expr, list, .. } => { + let mut v = vec![expr.as_ref()]; + v.extend(list.iter()); + v + } + FunctionCall { args, .. } => args.iter().collect(), + Case { + operand, + branches, + else_expr, + } => { + let mut v = Vec::new(); + if let Some(op) = operand { + v.push(op.as_ref()); + } + for (when, then) in branches { + v.push(when); + v.push(then); + } + if let Some(e) = else_expr { + v.push(e.as_ref()); + } + v + } + } + } + + /// Every column referenced in this expression, not inside the operators + /// it reads (those have their own scope). + pub fn columns_referenced(&self) -> Vec<&ColumnRef> { + let mut out = Vec::new(); + self.collect_columns(&mut out); + out + } + + fn collect_columns<'a>(&'a self, out: &mut Vec<&'a ColumnRef>) { + if let UnresolvedScalar::Column(c) = self { + out.push(c); + } + for child in self.children() { + child.collect_columns(out); + } + } + + /// The operators this expression (transitively) reads. + pub fn operator_refs(&self) -> Vec<&Rc> { + let mut out = Vec::new(); + self.collect_operator_refs(&mut out); + out + } + + fn collect_operator_refs<'a>(&'a self, out: &mut Vec<&'a Rc>) { + use UnresolvedScalar::*; + match self { + PromqlScalarFromVector(op) | ScalarSubquery(op) => out.push(op), + Exists { subquery, .. } | InSubquery { subquery, .. } => out.push(subquery), + _ => {} + } + for child in self.children() { + child.collect_operator_refs(out); + } + } +} + +impl UnresolvedOp { + /// An ordinary `Concat` (no unique-key claim). + pub fn concat(children: Vec) -> Self { + UnresolvedOp::Concat { + children: children.into_iter().map(Rc::new).collect(), + discriminator_unique_key: None, + } + } + + /// A `Concat` whose output carries the caller-proven compound unique key + /// `(discriminator, inner_key)`. Nothing verifies the claim. + pub fn concat_with_discriminator( + children: Vec, + discriminator: ColumnRef, + inner_key: Vec, + ) -> Self { + UnresolvedOp::Concat { + children: children.into_iter().map(Rc::new).collect(), + discriminator_unique_key: Some(ConcatDiscriminatorKey::new(discriminator, inner_key)), + } + } + + /// Every scalar expression this operator owns. + pub fn scalar_exprs(&self) -> Vec<&UnresolvedScalar> { + use UnresolvedOp::*; + match self { + Scan { predicates, .. } => predicates.iter().map(|p| &p.0).collect(), + Values { rows, .. } => rows.iter().flatten().collect(), + Filter { pred, .. } | Join { pred, .. } => vec![&pred.0], + Project { cols, .. } => cols.iter().map(|c| &c.expr).collect(), + Aggregate { + filters, having, .. + } => filters + .iter() + .flatten() + .chain(having.iter()) + .map(|p| &p.0) + .collect(), + Sort { keys, .. } => keys.iter().map(|k| &k.expr).collect(), + SQLWindowFunc { args, order_by, .. } => args + .iter() + .chain(order_by.iter().map(|k| &k.expr)) + .collect(), + PromqlVectorFromScalar(e) => vec![e], + PromqlScalarOp { scalar, .. } => vec![scalar], + PromqlMap { sample, .. } => vec![sample], + PromqlRelabel { value, .. } => vec![value], + SetOp { .. } + | Concat { .. } + | Dedup { .. } + | Limit { .. } + | BinaryOp { .. } + | TimeRange { .. } + | TimeShift { .. } + | PromqlInfoEnrich { .. } + | PromqlSeriesSample { .. } + | PromqlSubquery { .. } => vec![], + } + } +} diff --git a/crates/frontend-metricsql/Cargo.toml b/crates/frontend-metricsql/Cargo.toml index 8fa341b80..6731fe07e 100644 --- a/crates/frontend-metricsql/Cargo.toml +++ b/crates/frontend-metricsql/Cargo.toml @@ -4,6 +4,7 @@ version = "0.1.0" edition = "2021" [dependencies] +asap-frontend-common = { path = "../frontend-common" } asap-types = { path = "../types" } metricsql_parser = { path = "../metricsql-parser-vendored" } thiserror = "2" diff --git a/crates/frontend-metricsql/src/lib.rs b/crates/frontend-metricsql/src/lib.rs index 26a034c43..3544a9801 100644 --- a/crates/frontend-metricsql/src/lib.rs +++ b/crates/frontend-metricsql/src/lib.rs @@ -1,11 +1,14 @@ -//! MetricsQL AST to canonical `QueryExpr` frontend. +//! MetricsQL AST → the name-based `UnresolvedOp` tree → the unified operator DAG. use std::{rc::Rc, time::Duration}; +use asap_frontend_common::{ + resolve_root, UnresolvedOp as U, UnresolvedPredicate, UnresolvedScalar, +}; +use asap_types::ir::{BinaryOperator, ExprSemantics, OperatorNode, TimeRangeKind}; use asap_types::pre_asap::{ - resolve_root, AggIntent, ArithmeticOpKind, BinaryOpKind, ColumnRef, CompareOpKind, GroupKeys, - Predicate, PromQLVectorSetOpKind, QueryExpr, Reduction, ScalarValue, Source, - UnresolvedQueryExpr as U, + AggIntent, ArithmeticOpKind, BinaryOpKind, ColumnRef, CompareOpKind, GroupKeys, + PromQLVectorSetOpKind, Reduction, ScalarValue, Source, }; use asap_types::types::AccuracyTarget; use metricsql_parser::ast::{AggregateModifier, DurationExpr, Expr, MetricExpr, RollupExpr}; @@ -33,10 +36,31 @@ pub fn canonical_metricsql(query: &str) -> Result { Ok(parse_metricsql(query)?.to_string()) } -pub fn lower_metricsql(query: &str, accuracy: AccuracyTarget) -> Result { +pub fn lower_metricsql( + query: &str, + accuracy: AccuracyTarget, +) -> Result, MetricsqlError> { + match lower_metricsql_query(query, accuracy)? { + asap_types::ir::QueryRoot::Operator(node) => Ok(node), + _ => Err(unsupported("scalar root: use lower_metricsql_query")), + } +} + +/// Lower scalar constants without fabricating a relational operator. +pub fn lower_metricsql_query( + query: &str, + accuracy: AccuracyTarget, +) -> Result { let ast = parse_metricsql(query)?; + if let Expr::NumberLiteral(number) = &ast { + return Ok(asap_types::ir::QueryRoot::Scalar( + asap_types::ir::ScalarExpr::literal_f64(number.value), + )); + } let unresolved = Lowerer { accuracy }.lower(&ast)?; - resolve_root(&unresolved).map_err(|e| MetricsqlError::Resolve(e.to_string())) + resolve_root(&unresolved) + .map(asap_types::ir::QueryRoot::Operator) + .map_err(|e| MetricsqlError::Resolve(e.to_string())) } struct Lowerer { @@ -50,12 +74,16 @@ impl Lowerer { Expr::Rollup(e) => self.rollup(e), Expr::Function(e) => self.function(e), Expr::Aggregation(e) => self.aggregate(e), - Expr::NumberLiteral(e) => Ok(U::promql_scalar(e.value)), - Expr::UnaryOperator(e) => Ok(U::BinaryOp { + Expr::NumberLiteral(_) => { + Err(unsupported("scalar root requires lower_metricsql_query")) + } + // Vector negation is `x * -1` (as in the PromQL front end). + Expr::UnaryOperator(e) => Ok(U::PromqlScalarOp { + child: Rc::new(self.lower(&e.expr)?), + scalar: UnresolvedScalar::Literal(ScalarValue::Float64(-1.0)), op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(self.lower(&e.expr)?), - rhs: Rc::new(U::promql_scalar(-1.0)), - vector_match: None, + scalar_left: false, + return_bool: false, }), Expr::BinaryOperator(e) => self.binary(e), Expr::Parens(e) if e.expressions.len() == 1 => self.lower(&e.expressions[0]), @@ -83,7 +111,7 @@ impl Lowerer { }, predicates: filters .into_iter() - .map(|f| Predicate(Rc::new(matcher(f)))) + .map(|f| UnresolvedPredicate(matcher(f))) .collect(), schema: None, }) @@ -101,6 +129,7 @@ impl Lowerer { None => Ok(child), Some(window) => Ok(U::TimeRange { range: duration(window)?, + kind: TimeRangeKind::Range, child: Rc::new(child), }), } @@ -258,12 +287,41 @@ impl Lowerer { return Err(unsupported(format!("MetricsQL operator `{}`", expr.op))) } }; - Ok(U::BinaryOp { + for (scalar, vector, scalar_left) in [ + (&expr.left, &expr.right, true), + (&expr.right, &expr.left, false), + ] { + if let Expr::NumberLiteral(n) = scalar.as_ref() { + return Ok(U::PromqlScalarOp { + child: Rc::new(self.lower(vector)?), + scalar: UnresolvedScalar::Literal(ScalarValue::Float64(n.value)), + op, + scalar_left, + return_bool: false, + }); + } + } + Ok(binary_op( op, - lhs: Rc::new(self.lower(&expr.left)?), - rhs: Rc::new(self.lower(&expr.right)?), + self.lower(&expr.left)?, + self.lower(&expr.right)?, + )) + } +} + +/// A `BinaryOp` with default matching; MetricsQL modifiers (including `bool`) +/// are rejected before reaching here. +fn binary_op(kind: BinaryOpKind, lhs: U, rhs: U) -> U { + U::BinaryOp { + operator: BinaryOperator { + kind, vector_match: None, - }) + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, + lhs: Rc::new(lhs), + rhs: Rc::new(rhs), } } @@ -282,17 +340,22 @@ fn aggregate(reduction: Reduction, intent: AggIntent, chil } } -fn matcher(filter: &LabelFilter) -> U { +fn matcher(filter: &LabelFilter) -> UnresolvedScalar { let op = match filter.op { LabelFilterOp::Equal => CompareOpKind::Eq, LabelFilterOp::NotEqual => CompareOpKind::Ne, LabelFilterOp::RegexEqual => CompareOpKind::Regex, LabelFilterOp::RegexNotEqual => CompareOpKind::NotRegex, }; - U::Compare { - left: Rc::new(U::Column(ColumnRef::Named(filter.label.clone()))), + UnresolvedScalar::Compare { + left: Box::new(UnresolvedScalar::Column(ColumnRef::Named( + filter.label.clone(), + ))), op, - right: Rc::new(U::Literal(ScalarValue::Utf8(filter.value.clone()))), + right: Box::new(UnresolvedScalar::Literal(ScalarValue::Utf8( + filter.value.clone(), + ))), + semantics: ExprSemantics::Promql, } } diff --git a/crates/frontend-metricsql/tests/lowering.rs b/crates/frontend-metricsql/tests/lowering.rs index 133fc328b..ef828d04d 100644 --- a/crates/frontend-metricsql/tests/lowering.rs +++ b/crates/frontend-metricsql/tests/lowering.rs @@ -1,12 +1,14 @@ +use std::rc::Rc; use std::time::Duration; use asap_frontend_metricsql::{ canonical_metricsql, lower_metricsql, parse_metricsql, MetricsqlError, }; -use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction, Source}; +use asap_types::ir::{NonASAPOp, OperatorNode, TimeRangeKind}; +use asap_types::pre_asap::{AggIntent, Reduction, Source}; use asap_types::types::AccuracyTarget; -fn lower(query: &str) -> QueryExpr { +fn lower(query: &str) -> Rc { lower_metricsql(query, AccuracyTarget::Epsilon(0.01)).unwrap() } @@ -14,49 +16,50 @@ fn lower(query: &str) -> QueryExpr { fn selector_range_aggregate_and_call_share_the_canonical_shape() { let query = r#"sum by (job) (rate(http_requests_total{status=~"5.."}[5m]))"#; let tree = lower(query); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = tree + } = tree.expect_non_asap() else { panic!("expected outer aggregate"); }; - assert_eq!(reduction, Reduction::by(vec![2])); + assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected rate aggregate"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { panic!("expected range"); }; assert_eq!(*range, Duration::from_secs(300)); assert!( - matches!(child.as_ref(), QueryExpr::Scan { source: Source::TimeSeries { metric }, predicates, .. } if metric == "http_requests_total" && predicates.len() == 1) + matches!(child.expect_non_asap(), NonASAPOp::Scan { source: Source::TimeSeries { metric }, predicates, .. } if metric == "http_requests_total" && predicates.len() == 1) ); } #[test] fn default_rollup_with_explicit_range_is_last_over_time() { let tree = lower("default_rollup(cpu_usage[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = tree + } = tree.expect_non_asap() else { panic!("expected aggregate"); }; - assert_eq!(reduction, Reduction::PerEntity); + assert_eq!(reduction, &Reduction::PerEntity); assert!(matches!(measures.as_slice(), [AggIntent::LastOverTime])); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, .. } if *range == Duration::from_secs(300)) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, kind, .. } + if *range == Duration::from_secs(300) && *kind == TimeRangeKind::Range) ); } @@ -147,7 +150,13 @@ fn metricsql_multi_argument_aggregates_fail_closed() { #[test] fn supported_parameterized_functions_require_their_exact_arity() { let quantile = lower("quantile(0.9, requests_total)"); - assert!(matches!(quantile, QueryExpr::Aggregate { .. })); + assert!(matches!( + quantile.expect_non_asap(), + NonASAPOp::Aggregate { .. } + )); let rollup = lower("quantile_over_time(0.9, requests_total[5m])"); - assert!(matches!(rollup, QueryExpr::Aggregate { .. })); + assert!(matches!( + rollup.expect_non_asap(), + NonASAPOp::Aggregate { .. } + )); } diff --git a/crates/frontend-promql/Cargo.toml b/crates/frontend-promql/Cargo.toml index b108e32c7..db5a0a971 100644 --- a/crates/frontend-promql/Cargo.toml +++ b/crates/frontend-promql/Cargo.toml @@ -3,9 +3,11 @@ name = "asap-frontend-promql" version = "0.1.0" edition = "2021" -# PromQL front end: L1 (parse) → L2 relational, then the shared L2→L3 converter -# — both in asap-types. Pulls the PromQL parser only — never DataFusion. +# PromQL front end: parse → the shared name-based `UnresolvedOp` tree +# (asap-frontend-common) → the unified IR. Pulls the PromQL parser only — +# never DataFusion. [dependencies] +asap-frontend-common = { path = "../frontend-common" } asap-types = { path = "../types" } # Shared ProjectASAP parser contract. Keep this immutable revision aligned diff --git a/crates/frontend-promql/src/error.rs b/crates/frontend-promql/src/error.rs index 1885a5f42..287f49f7d 100644 --- a/crates/frontend-promql/src/error.rs +++ b/crates/frontend-promql/src/error.rs @@ -1,12 +1,12 @@ use std::fmt; -use asap_types::pre_asap::ResolveTreeError; +use asap_frontend_common::ResolveTreeError; use asap_types::workload::WorkloadError; -/// Errors from lowering a PromQL query (parse → the canonical, unresolved +/// Errors from lowering a PromQL query (parse → the name-based unresolved /// tree, built directly → -/// [`resolve_root`](asap_types::pre_asap::resolve_root) binds it to the -/// resolved tree, issue #179). +/// [`resolve_root`](asap_frontend_common::resolve_root) binds it to the +/// unified operator DAG, issue #179). /// /// Carries no DataFusion type — the PromQL front end never depends on the SQL /// stack. The language-neutral variants (`UnsupportedFeature` / `WrongLanguage` diff --git a/crates/frontend-promql/src/histogram.rs b/crates/frontend-promql/src/histogram.rs index f3f971a55..ecb8cd2c4 100644 --- a/crates/frontend-promql/src/histogram.rs +++ b/crates/frontend-promql/src/histogram.rs @@ -1,18 +1,10 @@ //! Sample-type metadata for the `histogram_quantile` discrimination (issue #79). //! -//! `histogram_quantile(φ, m)` has two lowerings: exact interpolation over -//! classic cumulative `le` buckets (`AggIntent::HistogramQuantile`, **not** -//! sketch-able) versus the generic sketch-able `Quantile` (native histograms / -//! raw samples, which post-ASAP binding can approximate to an accuracy -//! target). The true -//! signal is the argument's **sample type**, which query structure only -//! *proxies* — see the structural `is_classic_bucket_arg` heuristic, whose -//! false-positive (`…_bucket`-named non-histogram) and false-negative -//! (suffix-less classic histogram) cases this metadata fixes. -//! -//! A client that knows its sample types supplies a [`HistogramCatalog`]; it is -//! consulted first, and the structural heuristic remains the fallback when a -//! metric is undeclared. +//! Classic cumulative buckets use exact interpolation. The explicitly declared +//! `RawSamples` extension permits generic quantile sketches; it is not standard +//! PromQL histogram semantics. Native samples are rejected until the IR has a +//! native histogram sample type. Undeclared metrics require classic bucket +//! evidence (`by (le)`, a `_bucket` metric, or an `le` matcher). use std::cell::RefCell; use std::collections::HashMap; @@ -25,7 +17,7 @@ pub enum HistogramKind { /// distribution can't be reconstructed from them, so it is **not** /// sketch-able: `histogram_quantile` is exact bucket interpolation. ClassicBucket, - /// Native (exponential) histogram — sketch-able to an accuracy target. + /// Native histogram samples; currently rejected because the IR lacks their type. Native, /// Raw float samples the client retains — sketch-able. This is the case the /// generic `Quantile` lowering exists for (a client holding raw samples can @@ -37,7 +29,7 @@ impl HistogramKind { /// Whether `histogram_quantile` over this kind lowers to the sketch-able /// generic `Quantile` (`true`) rather than exact bucket interpolation. pub fn is_sketchable(self) -> bool { - !matches!(self, HistogramKind::ClassicBucket) + matches!(self, HistogramKind::RawSamples) } } @@ -106,9 +98,9 @@ mod tests { use super::*; #[test] - fn only_classic_buckets_are_not_sketchable() { + fn only_explicit_raw_samples_are_sketchable() { assert!(!HistogramKind::ClassicBucket.is_sketchable()); - assert!(HistogramKind::Native.is_sketchable()); + assert!(!HistogramKind::Native.is_sketchable()); assert!(HistogramKind::RawSamples.is_sketchable()); } diff --git a/crates/frontend-promql/src/lib.rs b/crates/frontend-promql/src/lib.rs index 60254d3d9..e7d99fe4c 100644 --- a/crates/frontend-promql/src/lib.rs +++ b/crates/frontend-promql/src/lib.rs @@ -1,25 +1,26 @@ -//! PromQL front end: parse (via `promql-parser`) → the canonical, unresolved -//! shape, built directly (issue #179) → [`resolve_root`]. +//! PromQL front end: parse (via `promql-parser`) → the name-based +//! [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree, built directly +//! in canonical shape (issue #179) → [`resolve_root`]. //! -//! Emits [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) itself — the -//! canonical `QueryExpr`, generic over an unresolved -//! [`ColumnRef`](asap_types::pre_asap::ColumnRef) — directly, rather than a -//! separate per-language relational tree; `resolve_root` runs the -//! [`SchemaResolver`](asap_types::pre_asap::SchemaResolver) for positional name resolution. -//! Depends on the PromQL parser only — never on the SQL / DataFusion stack. +//! `resolve_root` runs the +//! [`SchemaResolver`](asap_frontend_common::SchemaResolver) for positional +//! name resolution and returns the unified +//! [`OperatorNode`](asap_types::ir::OperatorNode) DAG. Depends on the PromQL +//! parser only — never on the SQL / DataFusion stack. pub mod error; pub mod histogram; pub mod promql; -use asap_types::pre_asap::resolve_root; -use asap_types::pre_asap::QueryExpr; +use std::rc::Rc; + +use asap_types::ir::OperatorNode; use asap_types::workload::{DurationMs, PlanningWorkload, QueryLanguage, WorkloadError}; pub use error::PromqlError; pub use histogram::{HistogramCatalog, HistogramKind}; -/// Lower every normalized PromQL workload entry to a plan-ready `QueryExpr`. +/// Lower every normalized PromQL workload entry to a plan-ready operator DAG. /// /// PromQL workloads must declare a non-zero `data_ingestion_interval`; it is /// injected around each bare instant selector. Explicit range selectors keep @@ -29,7 +30,7 @@ pub use histogram::{HistogramCatalog, HistogramKind}; pub fn lower_promql_workload( workload: &PlanningWorkload, now_ms: u64, -) -> Result, PromqlError> { +) -> Result>, PromqlError> { lower_promql_workload_inner(workload, now_ms) } @@ -39,15 +40,47 @@ pub fn lower_promql_workload_with_histograms( workload: &PlanningWorkload, histograms: HistogramCatalog, now_ms: u64, -) -> Result, PromqlError> { +) -> Result>, PromqlError> { let _guard = histogram::CatalogGuard::install(histograms); lower_promql_workload_inner(workload, now_ms) } +/// Lower scalar and vector query roots without introducing constant operators. +pub fn lower_promql_query_workload( + workload: &PlanningWorkload, + now_ms: u64, +) -> Result, PromqlError> { + lower_promql_query_workload_inner(workload, now_ms) +} + +pub fn lower_promql_query_workload_with_histograms( + workload: &PlanningWorkload, + histograms: HistogramCatalog, + now_ms: u64, +) -> Result, PromqlError> { + let _guard = histogram::CatalogGuard::install(histograms); + lower_promql_query_workload_inner(workload, now_ms) +} + fn lower_promql_workload_inner( workload: &PlanningWorkload, now_ms: u64, -) -> Result, PromqlError> { +) -> Result>, PromqlError> { + lower_promql_query_workload_inner(workload, now_ms)? + .into_iter() + .map(|root| match root { + asap_types::ir::QueryRoot::Operator(node) => Ok(node), + asap_types::ir::QueryRoot::Scalar(_) => Err(PromqlError::UnsupportedFeature( + "scalar root: use lower_promql_query_workload".into(), + )), + }) + .collect() +} + +fn lower_promql_query_workload_inner( + workload: &PlanningWorkload, + now_ms: u64, +) -> Result, PromqlError> { if !matches!(workload.query_workload.language, QueryLanguage::PromQL) { return Err(PromqlError::WrongLanguage(format!( "{:?}", @@ -66,12 +99,12 @@ fn lower_promql_workload_inner( .query_workload .entries() .map(|entry| { - let unresolved = promql::PromqlLowerer::lower_with_ingestion_interval( + let root = promql::PromqlLowerer::lower_query_with_ingestion_interval( &entry.query.0, &entry.requirements.accuracy.target(), std::time::Duration::from_millis(interval_ms), )?; - Ok(resolve_root(&unresolved)?) + Ok(root) }) .collect() } @@ -123,7 +156,7 @@ mod tests { } use std::time::Duration; - use asap_types::pre_asap::QueryExpr; + use asap_types::ir::{NonASAPOp, TimeRangeKind}; use asap_types::workload::{ BatchEntry, DataWorkload, Evidence, PlanningWorkload, Query, QueryRequirements, QueryWorkload, TimeSelection, @@ -155,27 +188,34 @@ mod tests { } } + // A bare instant selector reads the latest sample within the declared + // ingestion interval: an `Instant` lookback of that length. #[test] fn instant_selector_uses_declared_ingestion_interval() { let query = lower_promql_workload(&workload("sum by (job) (data)"), 0).unwrap(); - let QueryExpr::Aggregate { child, .. } = &query[0] else { + let NonASAPOp::Aggregate { child, .. } = query[0].expect_non_asap() else { panic!("expected aggregate") }; assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, child } - if *range == Duration::from_secs(1) && matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, kind, child } + if *range == Duration::from_secs(1) + && *kind == TimeRangeKind::Instant + && matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); } + // An explicit `m[5m]` keeps its own window as a `Range` selection. #[test] fn explicit_range_selector_keeps_its_query_range() { let query = lower_promql_workload(&workload("sum_over_time(data[5m])"), 0).unwrap(); - let QueryExpr::Aggregate { child, .. } = &query[0] else { + let NonASAPOp::Aggregate { child, .. } = query[0].expect_non_asap() else { panic!("expected aggregate") }; assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, child } - if *range == Duration::from_secs(300) && matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, kind, child } + if *range == Duration::from_secs(300) + && *kind == TimeRangeKind::Range + && matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); } diff --git a/crates/frontend-promql/src/promql.rs b/crates/frontend-promql/src/promql.rs index e1061683f..e8e0efdb2 100644 --- a/crates/frontend-promql/src/promql.rs +++ b/crates/frontend-promql/src/promql.rs @@ -1,14 +1,13 @@ -//! PromQL string → the canonical, unresolved -//! [`UnresolvedQueryExpr`](asap_types::pre_asap::query_expr::UnresolvedQueryExpr) -//! (`QueryExpr`). +//! PromQL string → the name-based +//! [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree. //! //! - **Parsing** is delegated to `promql-parser` 0.8. //! - **Lowering** builds *directly in canonical shape* here (issue #179): the //! walk interprets PromQL semantics (range vectors, aggregate operators, -//! label matchers) and emits `UnresolvedQueryExpr` nodes with unresolved -//! `ColumnRef`s — the same tree shape -//! [`resolve_root`](asap_types::pre_asap::resolve_root) later binds to -//! canonical, positional `QueryExpr`. The structural decisions a +//! label matchers) and emits `UnresolvedOp` / `UnresolvedScalar` nodes with +//! unresolved `ColumnRef`s — the same tree shape +//! [`resolve_root`](asap_frontend_common::resolve_root) later binds to the +//! positional [`OperatorNode`](asap_types::ir::OperatorNode) DAG. The structural decisions a //! separate converter stage would otherwise have to make (heavy-hitter //! `topk` recognition, the `PerEntity`/`Reduce` reduction choice, //! `without(...)` grouping) are made right here, since a front end @@ -38,9 +37,11 @@ //! | `increase(m[w])` | `Aggregate{[Increase], TimeRange{w}}` | //! | `changes`/`delta`/`idelta`/`deriv`/`resets`/`predict_linear`/`double_exponential_smoothing`(`m[w]`, …) | `Aggregate{[Changes/Delta/…], TimeRange{w}}` — per-series counter-derivative intents (issue #44) | //! | `absent(v)` / `absent_over_time(m[w])` / `present_over_time(m[w])` | `Aggregate{[Absent/AbsentOverTime/PresentOverTime]}` — presence intents; the empty→synthesized-sample logic is a post-ASAP concern (issue #47) | -//! | `abs`/`ceil`/`sqrt`/`ln`/`clamp*`/`round`/trig(`v`), `pi()` | `Aggregate{[Math(f)]}` element-wise transform (issue #45); `pi()` → a `PromqlScalarBridge` leaf | -//! | `time()` / `timestamp`/`hour`/`day_of_week`/… (`v`) | `EvalTimestamp` leaf / `Aggregate{[TimeFn(f)]}` (issue #46) | -//! | `vector(s)` / `scalar(v)` | `PromqlVectorFromScalar` / `PromqlScalarFromVector` — the scalar⇄vector bridges (issue #48) | +//! | `abs`/`ceil`/`sqrt`/`ln`/`clamp*`/`round`/trig(`v`), `pi()` | typed scalar `Project` (issue #45); `pi()` → a `ScalarExpr::Literal` root | +//! | `time()` / `timestamp`/`hour`/`day_of_week`/… (`v`) | `ScalarExpr::EvalTimestamp` root / `Aggregate{[TimeFn(f)]}` (issue #46) | +//! | `vector(s)` / `scalar(v)` | `PromqlVectorFromScalar(s)` / `ScalarExpr::PromqlScalarFromVector(v)` — the scalar⇄vector bridges (issue #48) | +//! | ` op ` (`time() - 1`, `1 < bool 2`, `-time()`) | `ScalarExpr::{Arithmetic, Case, Negative}` — a scalar expression, never an operator | +//! | `v op `, `a op bool b`, `v > bool 0` | `Project`/`Filter` with owned scalar expressions; vector/vector uses `BinaryOp{return_bool}` | //! | `label_replace(v,…)` / `label_join(v,…)` | `PromqlRelabel{dst, value}` — per-series label rewrite; value unchanged (issue #50) | //! | `info(v, [selector])` | `PromqlInfoEnrich{selector}` — label-enrichment join against the info metric(s); join keys resolved during post-ASAP binding (issue #84) | //! | `group` / `offset` / `@` / `info` | **rejected** — distinct semantics with no intent-algebra representation yet (`info` label-join → #84) | @@ -50,7 +51,7 @@ //! | `limitk(k, v)` / `limit_ratio(r, v)` | `PromqlSeriesSample{LimitK(k) \| LimitRatio(r)}` — series-sampling selection, whole series kept unchanged (issue #86) | //! | `topk(k, count_over_time(…))` / `topk(k, sum_over_time(…))` | `Aggregate{[TopK{k}]}` (heavy-hitter intent) over the explicit inner `Aggregate{[Count/Sum]}` | //! | `topk(k, )` / `bottomk(k, …)` | `Sort{value} → Limit{k}` | -//! | `m{f}` | `Scan{predicates}` | +//! | `m{f}` / `m{f}[w]` | `TimeRange{ingestion, Instant, Scan{predicates}}` / `TimeRange{w, Range, Scan}` | //! | `a OP b` | `BinaryOp{vector_match}` | //! | `expr[r:res]` | `PromqlSubquery{r, res}` | //! | ` offset ` / ` @ `/`start()`/`end()` | `TimeShift{shift}` over the selector's `Scan` — pass-through schema; a ranged selector shifts under its `TimeRange` (issue #40) | @@ -59,22 +60,30 @@ use std::rc::Rc; use std::time::{Duration, SystemTime}; use promql_parser::label::{MatchOp, Matcher}; +use promql_parser::parser::value::ValueType; use promql_parser::parser::{ self, token, AggregateExpr, AtModifier as ParserAtModifier, BinaryExpr, Call, Expr, LabelModifier, Offset, VectorMatchCardinality, VectorSelector, }; -use asap_types::pre_asap::agg_intent::{topk, AggIntent, MathFunc, TimeFunc}; -use asap_types::pre_asap::query_expr::{ - AtModifier, BinaryOpKind, GroupKeys, GroupSide, Predicate, PromQLVectorSetOpKind, Reduction, - SortKey, Source, TimeShift, UnresolvedQueryExpr as Unresolved, VectorGrouping, VectorMatch, - VectorMatchKind, +use asap_frontend_common::{ + UnresolvedOp as Unresolved, UnresolvedPredicate, UnresolvedScalar as Scalar, UnresolvedSortKey, }; +use asap_types::ir::operator_properties::{ + AtModifier, BinaryOpKind, GroupKeys, GroupSide, PromQLVectorSetOpKind, Reduction, Source, + TimeShift, VectorGrouping, VectorMatch, VectorMatchKind, +}; +use asap_types::ir::{BinaryOperator, ExprSemantics, TimeRangeKind}; +use asap_types::pre_asap::agg_intent::{topk, AggIntent, TimeFunc}; + use asap_types::pre_asap::{ ArithmeticOpKind, ColumnRef, CompareOpKind, InfoMatcher, SampleKind, ScalarValue, }; use asap_types::types::AccuracyTarget; +/// Every scalar expression this front end builds follows PromQL's numeric rules. +const PROMQL: ExprSemantics = ExprSemantics::Promql; + use crate::error::PromqlError as LoweringError; type Result = std::result::Result; @@ -158,7 +167,7 @@ enum InnerFunc { struct Inner { metric: String, - matchers: Vec, + matchers: Vec, window: Option, func: Option, /// `offset` / `@` on the selector, carried to the `Source` (issue #40). @@ -172,16 +181,35 @@ struct Inner { const MAX_DEPTH: usize = 256; impl PromqlLowerer { - pub(crate) fn lower_with_ingestion_interval( + pub(crate) fn lower_query_with_ingestion_interval( query: &str, accuracy: &AccuracyTarget, interval: Duration, - ) -> Result { + ) -> Result { let _guard = AccuracyGuard::install(accuracy.clone()); let _interval = IngestionIntervalGuard::install(interval); let ast = parser::parse(query).map_err(LoweringError::Parse)?; check_depth(&ast, MAX_DEPTH)?; - walk(&ast) + let mut metrics = Vec::new(); + collect_metric_names(&ast, &mut metrics); + if metrics.iter().any(|metric| { + crate::histogram::current_kind_of(metric) + == Some(crate::histogram::HistogramKind::Native) + }) { + return Err(LoweringError::UnsupportedFeature( + "native histogram samples have no IR representation".into(), + )); + } + + if ast.value_type() == ValueType::Scalar { + Ok(asap_types::ir::QueryRoot::Scalar( + asap_frontend_common::resolve_scalar_root(&lower_scalar(&ast)?)?, + )) + } else { + Ok(asap_types::ir::QueryRoot::Operator( + asap_frontend_common::resolve_root(&walk(&ast)?)?, + )) + } } } @@ -273,6 +301,13 @@ fn check_depth(expr: &Expr, budget: usize) -> Result<()> { } fn walk(expr: &Expr) -> Result { + // A scalar-typed expression (`5`, `time() - 1`, `scalar(v)`, `1 < bool 2`) + // is a scalar expression at an operator position, never an operator tree. + if expr.value_type() == ValueType::Scalar { + return Err(LoweringError::UnsupportedFeature( + "scalar root requires query-root lowering".into(), + )); + } match expr { Expr::Aggregate(agg) => walk_aggregate(agg), Expr::Call(call) if call.func.name.starts_with("histogram_") => walk_histogram(call), @@ -282,31 +317,22 @@ fn walk(expr: &Expr) -> Result { Expr::Call(call) if is_typeconv_fn(call.func.name) => walk_typeconv(call), Expr::Call(call) if is_label_fn(call.func.name) => walk_label(call), Expr::Call(call) if is_sort_fn(call.func.name) => walk_sort(call), - // A bare `min_of`/`max_of(consts…)` scalar query folds to a `PromqlScalarBridge` - // leaf; a non-constant argument makes `num_expr` fail → rejected (#89). - Expr::Call(call) if is_scalar_reducer_fn(call.func.name) => { - Ok(Unresolved::promql_scalar(num_expr(expr)?)) - } Expr::Call(call) if call.func.name == "info" => walk_info(call), Expr::Call(call) => walk_call(call), Expr::Binary(bin) => walk_binary(bin), Expr::Paren(p) => walk(&p.expr), // `UnaryExpr` is built only by negation (`Neg`); unary `+` is folded to - // identity and `-` to a negated `NumberLiteral`, so this wraps a - // sub-expression whose samples must be sign-flipped. Now that a scalar - // operand exists (#35), express it as `x * -1` — a constant-foldable - // operand (`-(10*1024)`) collapses to a negated `PromqlScalarBridge` leaf; anything - // else is a vector, sign-flipped by a `Mul` against `PromqlScalarBridge(-1)`. `Mul` - // is commutative, so operand order carries no hazard (#36). - Expr::Unary(u) => match num_expr(&u.expr) { - Ok(v) => Ok(Unresolved::promql_scalar(-v)), - Err(_) => Ok(Unresolved::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(walk(&u.expr)?), - rhs: Rc::new(Unresolved::promql_scalar(-1.0)), - vector_match: None, - }), - }, + // identity and `-` to a negated `NumberLiteral`. A scalar + // operand was dispatched to `lower_scalar` above (→ `Negative`), so this + // is a vector projection. Unary negation retains the metric name. + Expr::Unary(u) => Ok(Unresolved::PromqlMap { + child: Rc::new(walk(&u.expr)?), + sample: Scalar::Negative { + expr: Box::new(Scalar::Column(ColumnRef::SampleValue)), + semantics: ExprSemantics::Promql, + }, + drop_metric_name: false, + }), Expr::Subquery(sq) => { let subquery = Unresolved::PromqlSubquery { range: sq.range, @@ -332,13 +358,14 @@ fn walk(expr: &Expr) -> Result { let (metric, matchers, shift) = vs_parts(&ms.vs)?; Ok(Unresolved::TimeRange { range: ms.range, + kind: TimeRangeKind::Range, child: Rc::new(filtered_source(metric, matchers, shift)), }) } - // A number literal is a scalar leaf (`v > 5`, or a bare scalar query - // `5`). String literals only appear as function args (`label_replace`, - // …), which are not supported, so reject them (issue #35). - Expr::NumberLiteral(n) => Ok(Unresolved::promql_scalar(n.val)), + // Scalar-typed, dispatched above; kept for exhaustiveness. String + // literals only appear as function args (`label_replace`, …), so a + // bare one is rejected (issue #35). + Expr::NumberLiteral(_) => unreachable!("scalar handled above"), Expr::StringLiteral(_) => Err(LoweringError::UnsupportedFeature( "bare string literal".into(), )), @@ -348,6 +375,98 @@ fn walk(expr: &Expr) -> Result { } } +/// Lower a scalar-typed PromQL expression to a scalar expression. A constant +/// sub-expression folds to one `Literal` (as `num_expr` always did); anything +/// else keeps its structure: `-time()` → `Negative`, `time() - 1` → +/// `Arithmetic`, `scalar(v)` → `PromqlScalarFromVector`, and a `bool` +/// comparison → `Case(Compare → 1, else 0)` (PromQL yields `0`/`1`). +fn lower_scalar(expr: &Expr) -> Result { + if let Ok(v) = num_expr(expr) { + return Ok(Scalar::Literal(ScalarValue::Float64(v))); + } + match expr { + Expr::Paren(p) => lower_scalar(&p.expr), + Expr::Unary(u) => Ok(Scalar::Negative { + expr: Box::new(lower_scalar(&u.expr)?), + semantics: PROMQL, + }), + Expr::Binary(bin) => lower_scalar_binary(bin), + Expr::Call(call) => match call.func.name { + "time" => Ok(Scalar::EvalTimestamp), + "pi" => Ok(Scalar::Literal(ScalarValue::Float64(std::f64::consts::PI))), + "scalar" => Ok(Scalar::PromqlScalarFromVector(Rc::new(walk(arg( + call, 0, + )?)?))), + // `min_of`/`max_of` fold only over constants (#89); the fold above + // failed, so surface its error for the non-constant argument. + name if is_scalar_reducer_fn(name) => Err(num_expr(expr).unwrap_err()), + other => Err(LoweringError::UnsupportedFunction(other.to_string())), + }, + other => Err(LoweringError::UnsupportedFeature(format!( + "scalar expression `{other}`" + ))), + } +} + +/// ` op `: arithmetic is an `Arithmetic` expression; a +/// comparison needs the `bool` modifier (PromQL has no scalar filter) and +/// becomes `Case(Compare → 1.0, else 0.0)`. The parser already rejects both a +/// bool-less scalar comparison and a scalar set op; both are re-checked here. +fn lower_scalar_binary(bin: &BinaryExpr) -> Result { + let left = Box::new(lower_scalar(&bin.lhs)?); + let right = Box::new(lower_scalar(&bin.rhs)?); + match binop(bin.op.id())? { + BinaryOpKind::Arithmetic(op) => Ok(Scalar::Arithmetic { + op, + left, + right, + semantics: PROMQL, + }), + BinaryOpKind::Compare(op) => { + if !bin.return_bool() { + return Err(LoweringError::InvalidParameter( + "a comparison between two scalars requires the `bool` modifier".into(), + )); + } + let compare = Scalar::Compare { + left, + op, + right, + semantics: PROMQL, + }; + Ok(Scalar::Case { + operand: None, + branches: vec![(compare, Scalar::Literal(ScalarValue::Float64(1.0)))], + else_expr: Some(Box::new(Scalar::Literal(ScalarValue::Float64(0.0)))), + }) + } + BinaryOpKind::Set(_) => Err(LoweringError::UnsupportedFeature( + "set operator between two scalars".into(), + )), + } +} + +/// A binary operation over two vectors. +fn vector_binary( + kind: BinaryOpKind, + vector_match: Option, + return_bool: bool, + lhs: Unresolved, + rhs: Unresolved, +) -> Unresolved { + Unresolved::BinaryOp { + operator: BinaryOperator { + kind, + vector_match, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + lhs: Rc::new(lhs), + rhs: Rc::new(rhs), + } +} + /// Lower a bare function call (`rate(m[5m])`, `max_over_time(m[5m])`, …). /// /// The common case routes through the flat `lower_inner_call` template. The one @@ -555,13 +674,13 @@ fn outer_kind(agg: &AggregateExpr) -> Result { }) } -/// Wrap an already-lowered Unresolved subtree in the outer aggregation. This is the +/// Wrap an already-lowered Unresolved sub-DAG in the outer aggregation. This is the /// general-nesting counterpart to [`build`]: where `build` assembles the /// two-level shape from a flat [`Inner`], this composes the outer operator over /// an arbitrary child (`max(sum by (job) (…))`, `sum(a + b)`, …). /// /// A heavy-hitter `TopK` is only recognised on the flat `count_over_time` shape -/// (handled in `build`); over a general subtree, `topk`/`bottomk` is a generic +/// (handled in `build`); over a general sub-DAG, `topk`/`bottomk` is a generic /// order-by-value + limit — the same `Sort{partition_by} → Limit` pair `build` /// emits for any non-heavy-hitter ranking. /// Flip the outer `Aggregate` produced for a `without(...)` grouping into the @@ -576,7 +695,7 @@ fn outer_kind(agg: &AggregateExpr) -> Result { /// build this node) decides `PerEntity` vs `Reduce(by)` *without* knowing /// about `without` yet — it only ever sees `by`-mode keys, since `without`'s /// excluded-labels list is applied here, after the fact, exactly like the -/// pre-#179 legacy `relational::QueryExpr` tree's own `mark_without` did (its +/// pre-#179 legacy relational tree's own `mark_without` did (its /// converter read `without` only after this front-end step had already set /// it). Whether /// `reduction_for` picked `PerEntity` (only possible when `keys` was empty) @@ -652,24 +771,36 @@ fn build_over_subtree(outer: Outer, keys: Vec, child: Unresolved) -> child, )); } - let sorted = Unresolved::Sort { - keys: vec![SortKey { - expr: Unresolved::Column(ColumnRef::SampleValue), - ascending: !descending, - nulls_first: false, - }], - partition_by: keys.into(), - child: Rc::new(child), - }; - Unresolved::Limit { - n: k as usize, - offset: 0, - child: Rc::new(sorted), - } + ranked_by_value(keys, k, descending, child) } }) } +/// Generic `topk`/`bottomk`: `Limit{k} → Sort{value, partition_by: keys}` over +/// `child` — an order-by-value ranking, not a heavy-hitter intent. +fn ranked_by_value( + keys: Vec, + k: u64, + descending: bool, + child: Unresolved, +) -> Unresolved { + let sorted = Unresolved::Sort { + keys: vec![UnresolvedSortKey { + expr: Scalar::Column(ColumnRef::SampleValue), + ascending: !descending, + nulls_first: false, + }], + partition_by: keys.into(), + child: Rc::new(child), + }; + Unresolved::Limit { + n: Some(k as usize), + offset: 0, + partition_by: GroupKeys::none(), + child: Rc::new(sorted), + } +} + /// The `histogram_*` function family (issues #43, histogram_quantile). /// /// `histogram_quantile(φ, )` lowers `` in full — preserving any @@ -696,7 +827,7 @@ fn walk_histogram(call: &Call) -> Result { // The true signal is the argument's sample type: a declared // `HistogramKind` (issue #79) drives the choice when available, else we // fall back to the structural `by (le)`/`_bucket` heuristic (issue #43). - if !histogram_arg_is_sketchable(arg_expr) { + if !histogram_arg_is_sketchable(arg_expr)? { return Ok(classic_histogram_quantile(phi, "", walk(arg_expr)?)); } let func = AggIntent::Quantile { @@ -706,23 +837,9 @@ fn walk_histogram(call: &Call) -> Result { }; return Ok(outer_aggregate(vec![], func, walk(arg_expr)?)); } - // (histogram_quantile handled above; accessors below) - let (func, vec_idx) = match call.func.name { - "histogram_count" => (AggIntent::HistogramCount, 0), - "histogram_sum" => (AggIntent::HistogramSum, 0), - "histogram_avg" => (AggIntent::HistogramAvg, 0), - "histogram_stddev" => (AggIntent::HistogramStdDev, 0), - "histogram_stdvar" => (AggIntent::HistogramStdVar, 0), - "histogram_fraction" => ( - AggIntent::HistogramFraction { - lower: num_arg(call, 0)?, - upper: num_arg(call, 1)?, - }, - 2, - ), - other => return Err(LoweringError::UnsupportedFunction(other.to_string())), - }; - Ok(outer_aggregate(vec![], func, walk(arg(call, vec_idx)?)?)) + Err(LoweringError::UnsupportedFeature( + "native histogram samples have no IR representation".into(), + )) } /// Classic-bucket `histogram_quantile(φ, child)`. One histogram is the set of @@ -750,7 +867,7 @@ fn classic_histogram_quantile(q: f64, output_name: &str, child: Unresolved) -> U /// (`HistogramQuantile`), native histograms / raw samples take the sketch-able /// `Quantile` (issues #43 / #79) — so the two functions cannot diverge. /// -/// The vector argument is lowered once per branch, duplicating the subtree — +/// The vector argument is lowered once per branch, duplicating the sub-DAG — /// a future workload-level reuse pass could hoist it back into a single /// producer. /// @@ -769,7 +886,7 @@ fn walk_histogram_quantiles(call: &Call) -> Result { )); } // The bucket-vs-native choice is a property of the argument, not of φ. - let sketchable = histogram_arg_is_sketchable(vec_expr); + let sketchable = histogram_arg_is_sketchable(vec_expr)?; let branches = (2..call.args.args.len()) .map(|i| { let phi = bounded_quantile_param(num_arg(call, i)?)?; @@ -796,9 +913,7 @@ fn walk_histogram_quantiles(call: &Call) -> Result { }; Ok(Unresolved::PromqlRelabel { dst: label.clone(), - value: Rc::new(Unresolved::Literal(ScalarValue::Utf8(open_metrics_float( - phi, - )))), + value: Scalar::Literal(ScalarValue::Utf8(open_metrics_float(phi))), child: Rc::new(quantile), }) }) @@ -852,12 +967,12 @@ fn open_metrics_float(v: f64) -> String { } } -/// The time / calendar functions (issue #46). +/// The calendar functions (issue #46); `time()` is scalar-typed and lowers in +/// `lower_scalar`. fn is_time_fn(name: &str) -> bool { matches!( name, - "time" - | "timestamp" + "timestamp" | "minute" | "hour" | "day_of_week" @@ -869,33 +984,31 @@ fn is_time_fn(name: &str) -> bool { ) } -/// `time()` → the `EvalTimestamp` leaf. `timestamp(v)` and the calendar accessors → -/// `Aggregate{[TimeFn(f)]}` over the argument vector, or over `EvalTimestamp` for the +/// `timestamp(v)` and the calendar accessors → `Aggregate{[TimeFn(f)]}` over +/// the argument vector, or over `PromqlVectorFromScalar(EvalTimestamp)` for the /// no-argument calendar forms (`hour()`, `day_of_week()`, …). Issue #46. fn walk_time(call: &Call) -> Result { - if call.func.name == "time" { - return Ok(Unresolved::EvalTimestamp); + // timestamp() reads the selected sample's timestamp, not its value. + if call.func.name == "timestamp" { + return Ok(outer_aggregate( + vec![], + AggIntent::TimeFn(TimeFunc::Timestamp), + walk(arg(call, 0)?)?, + )); } - let func = match call.func.name { - "timestamp" => TimeFunc::Timestamp, - "minute" => TimeFunc::Minute, - "hour" => TimeFunc::Hour, - "day_of_week" => TimeFunc::DayOfWeek, - "day_of_month" => TimeFunc::DayOfMonth, - "day_of_year" => TimeFunc::DayOfYear, - "month" => TimeFunc::Month, - "year" => TimeFunc::Year, - "days_in_month" => TimeFunc::DaysInMonth, - other => return Err(LoweringError::UnsupportedFunction(other.to_string())), - }; - // A calendar function with no argument reads the evaluation time; otherwise - // it maps over each sample's timestamp in the argument vector. - let inner = if call.args.args.is_empty() { - Unresolved::EvalTimestamp + let child = if call.args.args.is_empty() { + Unresolved::PromqlVectorFromScalar(Scalar::EvalTimestamp) } else { walk(arg(call, 0)?)? }; - Ok(outer_aggregate(vec![], AggIntent::TimeFn(func), inner)) + Ok(Unresolved::PromqlMap { + child: Rc::new(child), + sample: Scalar::FunctionCall { + name: format!("promql_{}", call.func.name), + args: vec![Scalar::Column(ColumnRef::SampleValue)], + }, + drop_metric_name: true, + }) } /// The presence functions (issue #47). @@ -919,24 +1032,19 @@ fn walk_presence(call: &Call) -> Result { Ok(outer_aggregate(vec![], func, walk(arg(call, 0)?)?)) } -/// The scalar⇄vector type-conversion functions (issue #48). `info` is *not* -/// here: it is a label-enrichment join against info metrics, not a type -/// conversion, so it falls through to the `UnsupportedFunction` path (#84). +/// The scalar→vector conversion (issue #48); `scalar(v)` is scalar-typed and +/// lowers in `lower_scalar`. `info` is *not* here: it is a label-enrichment +/// join, not a type conversion (#84). fn is_typeconv_fn(name: &str) -> bool { - matches!(name, "vector" | "scalar") + name == "vector" } -/// `vector(s)` — promote a scalar to a label-less instant vector. `scalar(v)` -/// — collapse a single-element vector to its value. Both are honest bridge -/// nodes in the IR; the "exactly one element → NaN otherwise" runtime rule of -/// `scalar` is a post-ASAP/runtime concern (issue #48). +/// `vector(s)` — promote a scalar to a label-less instant vector carrying the +/// scalar expression `s` (issue #48). fn walk_typeconv(call: &Call) -> Result { - let inner = walk(arg(call, 0)?)?; - Ok(match call.func.name { - "vector" => Unresolved::PromqlVectorFromScalar(Rc::new(inner)), - "scalar" => Unresolved::PromqlScalarFromVector(Rc::new(inner)), - other => return Err(LoweringError::UnsupportedFunction(other.to_string())), - }) + Ok(Unresolved::PromqlVectorFromScalar(lower_scalar(arg( + call, 0, + )?)?)) } /// The instant-vector reordering functions (issue #51). @@ -960,13 +1068,13 @@ fn walk_sort(call: &Call) -> Result { "sort_by_label_desc" => (false, false), other => return Err(LoweringError::UnsupportedFunction(other.to_string())), }; - let sort_key = |expr| SortKey { + let sort_key = |expr| UnresolvedSortKey { expr, ascending, nulls_first: false, }; let keys = if by_value { - vec![sort_key(Unresolved::Column(ColumnRef::SampleValue))] + vec![sort_key(Scalar::Column(ColumnRef::SampleValue))] } else { // `sort_by_label(v, "l1", "l2", …)` — one key per label arg, in order. if call.args.args.len() < 2 { @@ -976,7 +1084,7 @@ fn walk_sort(call: &Call) -> Result { } (1..call.args.args.len()) .map(|i| { - Ok(sort_key(Unresolved::Column(ColumnRef::Named(str_arg( + Ok(sort_key(Scalar::Column(ColumnRef::Named(str_arg( call, i, )?)))) }) @@ -1050,19 +1158,15 @@ fn walk_label(call: &Call) -> Result { let replacement = str_arg(call, 2)?; let src = str_arg(call, 3)?; let regex = str_arg(call, 4)?; - let value = Unresolved::FunctionCall { + let value = Scalar::FunctionCall { name: "label_replace".into(), args: vec![ - Unresolved::Column(ColumnRef::Named(src)), - Unresolved::Literal(ScalarValue::Utf8(regex)), - Unresolved::Literal(ScalarValue::Utf8(replacement)), + Scalar::Column(ColumnRef::Named(src)), + Scalar::Literal(ScalarValue::Utf8(regex)), + Scalar::Literal(ScalarValue::Utf8(replacement)), ], }; - Ok(Unresolved::PromqlRelabel { - dst, - value: Rc::new(value), - child, - }) + Ok(Unresolved::PromqlRelabel { dst, value, child }) } "label_join" => { // label_join(v, dst, sep, src_1, …, src_n) — needs ≥1 source label. @@ -1073,19 +1177,15 @@ fn walk_label(call: &Call) -> Result { } let dst = str_arg(call, 1)?; let sep = str_arg(call, 2)?; - let mut args = vec![Unresolved::Literal(ScalarValue::Utf8(sep))]; + let mut args = vec![Scalar::Literal(ScalarValue::Utf8(sep))]; for i in 3..call.args.args.len() { - args.push(Unresolved::Column(ColumnRef::Named(str_arg(call, i)?))); + args.push(Scalar::Column(ColumnRef::Named(str_arg(call, i)?))); } - let value = Unresolved::FunctionCall { + let value = Scalar::FunctionCall { name: "label_join".into(), args, }; - Ok(Unresolved::PromqlRelabel { - dst, - value: Rc::new(value), - child, - }) + Ok(Unresolved::PromqlRelabel { dst, value, child }) } other => Err(LoweringError::UnsupportedFunction(other.to_string())), } @@ -1118,7 +1218,6 @@ fn is_math_fn(name: &str) -> bool { | "atanh" | "deg" | "rad" - | "pi" | "round" | "clamp" | "clamp_min" @@ -1127,59 +1226,24 @@ fn is_math_fn(name: &str) -> bool { } /// A math / trig function — a per-series element-wise value transform, lowered -/// to a per-series `Aggregate{[Math(f)]}` over the (instant) argument vector. -/// `pi()` is the constant π, lowered to a `PromqlScalarBridge` leaf (issue #45). +/// to a typed scalar projection over the instant-vector argument. +/// `pi()` is scalar-typed and lowers in `lower_scalar` (issue #45). fn walk_math(call: &Call) -> Result { - if call.func.name == "pi" { - return Ok(Unresolved::promql_scalar(std::f64::consts::PI)); + let mut args = vec![Scalar::Column(ColumnRef::SampleValue)]; + for index in 1..call.args.args.len() { + args.push(lower_scalar(arg(call, index)?)?); } - let func = match call.func.name { - "abs" => MathFunc::Abs, - "ceil" => MathFunc::Ceil, - "floor" => MathFunc::Floor, - "exp" => MathFunc::Exp, - "ln" => MathFunc::Ln, - "log2" => MathFunc::Log2, - "log10" => MathFunc::Log10, - "sqrt" => MathFunc::Sqrt, - "sgn" => MathFunc::Sgn, - "sin" => MathFunc::Sin, - "cos" => MathFunc::Cos, - "tan" => MathFunc::Tan, - "asin" => MathFunc::Asin, - "acos" => MathFunc::Acos, - "atan" => MathFunc::Atan, - "sinh" => MathFunc::Sinh, - "cosh" => MathFunc::Cosh, - "tanh" => MathFunc::Tanh, - "asinh" => MathFunc::Asinh, - "acosh" => MathFunc::Acosh, - "atanh" => MathFunc::Atanh, - "deg" => MathFunc::Deg, - "rad" => MathFunc::Rad, - // `round(v)` defaults the step to 1; `round(v, to)` reads arg 1. - "round" => MathFunc::Round { - to_nearest: if call.args.args.len() >= 2 { - num_arg(call, 1)? - } else { - 1.0 - }, - }, - "clamp" => MathFunc::Clamp { - min: num_arg(call, 1)?, - max: num_arg(call, 2)?, - }, - "clamp_min" => MathFunc::ClampMin { - min: num_arg(call, 1)?, - }, - "clamp_max" => MathFunc::ClampMax { - max: num_arg(call, 1)?, + if call.func.name == "round" && args.len() == 1 { + args.push(Scalar::Literal(ScalarValue::Float64(1.0))); + } + Ok(Unresolved::PromqlMap { + child: Rc::new(walk(arg(call, 0)?)?), + sample: Scalar::FunctionCall { + name: format!("promql_{}", call.func.name), + args, }, - other => return Err(LoweringError::UnsupportedFunction(other.to_string())), - }; - // The value being transformed is always arg 0 (a vector). - let inner = walk(arg(call, 0)?)?; - Ok(outer_aggregate(vec![], AggIntent::Math(func), inner)) + drop_metric_name: true, + }) } /// Whether `expr` is a **classic cumulative-bucket** `histogram_quantile` @@ -1202,15 +1266,31 @@ fn walk_math(call: &Call) -> Result { /// declared `RawSamples`) and the false-negative (a suffix-less classic /// histogram declared `ClassicBucket`) of the structural heuristic. With no /// declaration, fall back to the structural `by (le)`/`_bucket` heuristic. -fn histogram_arg_is_sketchable(arg: &Expr) -> bool { +fn histogram_arg_is_sketchable(arg: &Expr) -> Result { let mut metrics = Vec::new(); collect_metric_names(arg, &mut metrics); - for metric in &metrics { - if let Some(kind) = crate::histogram::current_kind_of(metric) { - return kind.is_sketchable(); + let kinds = metrics + .iter() + .filter_map(|metric| crate::histogram::current_kind_of(metric)) + .collect::>(); + if kinds.contains(&crate::histogram::HistogramKind::Native) { + return Err(LoweringError::UnsupportedFeature( + "native histogram samples have no IR representation".into(), + )); + } + if let Some(kind) = kinds.first() { + if kinds.iter().any(|other| other != kind) { + return Err(LoweringError::UnsupportedFeature( + "mixed histogram sample contracts".into(), + )); } + return Ok(kind.is_sketchable()); + } + if is_classic_bucket_arg(arg) { + Ok(false) + } else { + Err(LoweringError::UnsupportedFeature("histogram_quantile requires classic buckets; use quantile for float samples or explicitly declare the RawSamples extension".into())) } - !is_classic_bucket_arg(arg) } /// Collect the metric names of every vector/matrix selector reachable in `expr` @@ -1282,9 +1362,28 @@ fn selector_is_bucket(vs: &VectorSelector) -> bool { || vs.matchers.matchers.iter().any(|m| m.name == "le") } +/// A binary op with at least one vector operand (a scalar/scalar op is +/// scalar-typed and never reaches here). A scalar side lowers to a +/// scalar expression; mixed operations resolve to Project or Filter. fn walk_binary(bin: &BinaryExpr) -> Result { - let lhs = scalar_or_vector(&bin.lhs)?; - let rhs = scalar_or_vector(&bin.rhs)?; + let op = binop(bin.op.id())?; + let scalar_left = bin.lhs.value_type() == ValueType::Scalar; + if scalar_left || bin.rhs.value_type() == ValueType::Scalar { + let (scalar, vector) = if scalar_left { + (&bin.lhs, &bin.rhs) + } else { + (&bin.rhs, &bin.lhs) + }; + return Ok(Unresolved::PromqlScalarOp { + child: Rc::new(walk(vector)?), + scalar: lower_scalar(scalar)?, + op, + scalar_left, + return_bool: bin.return_bool(), + }); + } + let lhs = walk(&bin.lhs)?; + let rhs = walk(&bin.rhs)?; // `VectorMatch` has no fill field; dropping fill would change which series // are emitted and their values, so the query must fall back to exact // execution instead. @@ -1295,10 +1394,6 @@ fn walk_binary(bin: &BinaryExpr) -> Result { ))); } } - let op = match (binop(bin.op.id())?, bin.return_bool()) { - (BinaryOpKind::Compare(op), true) => BinaryOpKind::CompareBool(op), - (op, _) => op, - }; let vector_match = bin.modifier.as_ref().map(|m| { let (kind, labels) = match &m.matching { Some(LabelModifier::Include(ls)) => (VectorMatchKind::On, ls.labels.clone()), @@ -1329,12 +1424,7 @@ fn walk_binary(bin: &BinaryExpr) -> Result { grouping, } }); - Ok(Unresolved::BinaryOp { - op, - lhs: Rc::new(lhs), - rhs: Rc::new(rhs), - vector_match, - }) + Ok(vector_binary(op, vector_match, bin.return_bool(), lhs, rhs)) } fn lower_inner(expr: &Expr) -> Result { @@ -1601,20 +1691,7 @@ fn build(inner: Inner, keys: Vec, outer: Outer) -> Result Some(intent) => windowed_aggregate(inner, vec![], intent), None => instant_source(inner.metric, inner.matchers, inner.shift), }; - let sorted = Unresolved::Sort { - keys: vec![SortKey { - expr: Unresolved::Column(ColumnRef::SampleValue), - ascending: !descending, - nulls_first: false, - }], - partition_by: keys.into(), - child: Rc::new(base), - }; - Ok(Unresolved::Limit { - n: k as usize, - offset: 0, - child: Rc::new(sorted), - }) + Ok(ranked_by_value(keys, k, descending, base)) } } } @@ -1650,19 +1727,12 @@ fn windowed_aggregate( let child = match inner.window { Some(w) => Unresolved::TimeRange { range: w, + kind: TimeRangeKind::Range, child: Rc::new(base), }, - None => base, + None => ingestion_lookback(base), }; let reduction = reduction_for(&keys, inner.window.is_some() || intent.is_per_series()); - let child = if inner.window.is_none() { - Unresolved::TimeRange { - range: current_ingestion_interval(), - child: Rc::new(child), - } - } else { - child - }; Unresolved::Aggregate { reduction, measures: vec![intent], @@ -1676,7 +1746,7 @@ fn windowed_aggregate( } } -/// `Aggregate{reduction, [intent]}` directly over an existing Unresolved subtree — the +/// `Aggregate{reduction, [intent]}` directly over an existing Unresolved sub-DAG — the /// OUTER level of a two-level aggregation such as `sum(rate(…))` or the /// `Aggregate{[Quantile]}` that wraps a `histogram_quantile` argument. fn outer_aggregate( @@ -1715,13 +1785,10 @@ fn per_series_aggregate( } } -fn filtered_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { +fn filtered_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { let scan = Unresolved::Scan { source: Source::TimeSeries { metric }, - predicates: matchers - .into_iter() - .map(|m| Predicate(Rc::new(m))) - .collect(), + predicates: matchers.into_iter().map(UnresolvedPredicate).collect(), // Usage-derived (PromQL is schemaless) — the SchemaResolver fills this in. schema: None, }; @@ -1735,10 +1802,17 @@ fn filtered_source(metric: String, matchers: Vec, shift: TimeShift) } } -fn instant_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { +/// An instant selector: the latest sample per series within the workload's +/// ingestion interval, so the lookback is an `Instant` `TimeRange`. +fn instant_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { + ingestion_lookback(filtered_source(metric, matchers, shift)) +} + +fn ingestion_lookback(child: Unresolved) -> Unresolved { Unresolved::TimeRange { range: current_ingestion_interval(), - child: Rc::new(filtered_source(metric, matchers, shift)), + kind: TimeRangeKind::Instant, + child: Rc::new(child), } } @@ -1881,7 +1955,7 @@ fn resolve_group(agg: &AggregateExpr) -> Result<(Vec, bool)> { // ── Free helpers ────────────────────────────────────────────────────────────── -fn vs_parts(vs: &VectorSelector) -> Result<(String, Vec, TimeShift)> { +fn vs_parts(vs: &VectorSelector) -> Result<(String, Vec, TimeShift)> { // A non-equality `__name__` matcher (`=~` / `!~` / `!=`) selects *across* // metric names. `Source::TimeSeries { metric }` carries a single concrete // metric name, so there is no representation for a regex/negated name @@ -1961,21 +2035,22 @@ fn system_time_ms(t: SystemTime) -> Result { }) } -fn matcher_to_compare(m: &Matcher) -> Unresolved { +fn matcher_to_compare(m: &Matcher) -> Scalar { let op = match &m.op { MatchOp::Equal => CompareOpKind::Eq, MatchOp::NotEqual => CompareOpKind::Ne, MatchOp::Re(_) => CompareOpKind::Regex, MatchOp::NotRe(_) => CompareOpKind::NotRegex, }; - Unresolved::Compare { - left: Rc::new(Unresolved::Column(ColumnRef::Named(m.name.clone()))), + Scalar::Compare { + left: Box::new(Scalar::Column(ColumnRef::Named(m.name.clone()))), op, - right: Rc::new(Unresolved::Literal(ScalarValue::Utf8(m.value.clone()))), + right: Box::new(Scalar::Literal(ScalarValue::Utf8(m.value.clone()))), + semantics: PROMQL, } } -fn extract_matrix(expr: &Expr) -> Result<(String, Vec, Duration, TimeShift)> { +fn extract_matrix(expr: &Expr) -> Result<(String, Vec, Duration, TimeShift)> { match expr { Expr::MatrixSelector(ms) => { let (metric, matchers, shift) = vs_parts(&ms.vs)?; @@ -2017,6 +2092,7 @@ fn num_expr(expr: &Expr) -> Result { match expr { Expr::NumberLiteral(n) => Ok(n.val), Expr::Paren(p) => num_expr(&p.expr), + Expr::Unary(u) => Ok(-num_expr(&u.expr)?), // Constant-fold a pure scalar arithmetic expression — the parser does // not fold `10*1024*1024` / `24 * 3600`. A `modifier` (vector matching) // or a non-arithmetic operator means it is not a pure scalar. @@ -2074,15 +2150,6 @@ fn is_scalar_reducer_fn(name: &str) -> bool { matches!(name, "min_of" | "max_of") } -/// A `BinaryOp` operand: fold a pure-scalar expression (`5`, `10*1024*1024`) to -/// a `PromqlScalarBridge` leaf, otherwise walk it as a vector (issue #35). -fn scalar_or_vector(expr: &Expr) -> Result { - match num_expr(expr) { - Ok(v) => Ok(Unresolved::promql_scalar(v)), - Err(_) => walk(expr), - } -} - /// `topk`/`bottomk` count parameter — a non-negative integer. Rejects /// fractional / negative / non-finite values rather than silently truncating /// or saturating them via `as u64` (`topk(2.7, …)` ≠ `topk(2, …)`). diff --git a/crates/frontend-promql/tests/count_planning.rs b/crates/frontend-promql/tests/count_planning.rs index fded3cebe..8956e68f8 100644 --- a/crates/frontend-promql/tests/count_planning.rs +++ b/crates/frontend-promql/tests/count_planning.rs @@ -1,19 +1,19 @@ //! Query text through summary selection: counts use observations, never value weights. -use std::rc::Rc; - use asap_aware_mapping::accuracy::DefaultAccuracyModel; use asap_aware_mapping::cost_model::DefaultCostModel; use asap_aware_mapping::{ - default_strategies, search_workload_with_targets, Replacement, ReplacementStrategy, - SketchAlgorithmStrategy, TargetSubDAG, + default_strategies, search_workload_with_targets, ASAPStrategies, Replacement, + ReplacementStrategy, TargetSubDAG, }; mod support; +use asap_types::ir::export::PostAsapOperatorPayload; +use asap_types::ir::{ASAPOp, Operator}; use asap_types::post_asap::{ - compile_post_asap_dag, ExactKind, NonNegativeWeightProof, PostAsapOperatorPayload, - SketchAlgorithm, SummaryExpr, SummaryFamilyType, SummaryInputExpr, WeightDomain, + ExactKind, FieldDataType, NonNegativeWeightProof, SketchAlgorithm, SummaryInputExpr, + WeightDomain, }; use asap_types::types::AccuracyTarget; -use support::lower_promql; +use support::{lower_promql, post_asap_dag}; #[test] fn grouped_count_keeps_uncertified_hydra_candidates_for_backend_review() { @@ -21,7 +21,7 @@ fn grouped_count_keeps_uncertified_hydra_candidates_for_backend_review() { epsilon: 0.01, delta: 0.01, }; - let root = Rc::new(lower_promql("count by(job)(up)", target.clone()).unwrap()); + let root = lower_promql("count by(job)(up)", target.clone()).unwrap(); let space = search_workload_with_targets( vec![("count", root, Some(target))], &default_strategies(), @@ -51,14 +51,14 @@ fn grouped_count_keeps_uncertified_hydra_candidates_for_backend_review() { #[test] fn exact_counts_select_count_accumulators() { for query in ["count(up)", "count by(job)(up)", "count_over_time(up[5m])"] { - let root = Rc::new(lower_promql(query, AccuracyTarget::Exact).unwrap()); + let root = lower_promql(query, AccuracyTarget::Exact).unwrap(); let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); assert!( candidates.iter().any(|candidate| { - matches!(&candidate.replacement, Replacement::Summary(node) - if matches!(&node.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Count, _), .. })) + matches!(&candidate.replacement, Replacement::SubDag(node) + if matches!(&node.operator, Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Count, _), .. }))) }), "{query}: {candidates:?}" ); @@ -69,22 +69,23 @@ fn exact_counts_select_count_accumulators() { #[test] fn frequency_count_candidates_use_unit_weights() { for query in ["count_over_time(up[5m])", "count(up)"] { - let root = Rc::new(lower_promql(query, AccuracyTarget::Epsilon(0.02)).unwrap()); + let root = lower_promql(query, AccuracyTarget::Epsilon(0.02)).unwrap(); let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); let mut algorithms = Vec::new(); for candidate in &candidates { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDag(node) = &candidate.replacement else { continue; }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator + else { continue; }; - let SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + let Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), input, .. - } = &summary_input.expr + }) = &summary_input.operator else { continue; }; @@ -98,7 +99,7 @@ fn frequency_count_candidates_use_unit_weights() { ) { continue; } - let dag = compile_post_asap_dag(node).unwrap(); + let dag = post_asap_dag(node); assert!( dag.nodes.iter().any(|node| matches!( &node.payload, @@ -135,25 +136,26 @@ fn frequency_count_candidates_use_unit_weights() { // This narrow test oracle interprets the emitted aggregate, not Prometheus ingestion, // staleness, or scrape scheduling. Unsupported plan shapes fail explicitly. fn aggregate_fixture(query: &str, series: &[Vec]) -> Vec { - use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction}; + use asap_types::ir::NonASAPOp; + use asap_types::pre_asap::{AggIntent, Reduction}; let root = lower_promql(query, AccuracyTarget::Exact).unwrap(); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &root + } = root.expect_non_asap() else { panic!("expected aggregate: {root:?}"); }; - match child.as_ref() { - QueryExpr::Scan { .. } => assert!(series.iter().all(|samples| samples.len() == 1)), - QueryExpr::TimeRange { range, child } => { + match child.expect_non_asap() { + NonASAPOp::Scan { .. } => assert!(series.iter().all(|samples| samples.len() == 1)), + NonASAPOp::TimeRange { range, child, .. } => { assert!(matches!(range.as_secs(), 1 | 300)); if range.as_secs() == 1 { assert!(series.iter().all(|samples| samples.len() == 1)); } - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); } other => panic!("unsupported fixture input: {other:?}"), } @@ -225,22 +227,20 @@ fn count_over_time_counts_scrapes_not_sample_values() { #[test] fn cms_count_updates_total_ten_for_zero_positive_and_negative_samples() { use asap_types::pre_asap::ColumnRef; - let root = - Rc::new(lower_promql("count_over_time(up[5m])", AccuracyTarget::Epsilon(0.02)).unwrap()); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + let root = lower_promql("count_over_time(up[5m])", AccuracyTarget::Epsilon(0.02)).unwrap(); + let candidates = ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); let dag = candidates .iter() .find_map(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDag(node) = &candidate.replacement else { return None; }; - let dag = compile_post_asap_dag(node).unwrap(); + let dag = post_asap_dag(node); dag.nodes .iter() .any(|node| { matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } + PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::Cms) }) .then_some(dag) diff --git a/crates/frontend-promql/tests/histogram_metadata.rs b/crates/frontend-promql/tests/histogram_metadata.rs index 60ff5a60e..fcf67f40a 100644 --- a/crates/frontend-promql/tests/histogram_metadata.rs +++ b/crates/frontend-promql/tests/histogram_metadata.rs @@ -7,16 +7,17 @@ use asap_frontend_promql::{HistogramCatalog, HistogramKind}; mod support; -use asap_types::pre_asap::{AggIntent, QueryExpr}; +use asap_types::ir::{NonASAPOp, OperatorNode}; +use asap_types::pre_asap::AggIntent; use asap_types::types::AccuracyTarget; use support::{lower_promql, lower_promql_with_histograms}; /// The histogram/quantile intent kind in the lowered tree: `"HQ"` for the /// classic-bucket `HistogramQuantile`, `"Q"` for the sketch-able `Quantile`. -fn quantile_kind(qe: &QueryExpr) -> &'static str { - fn walk(e: &QueryExpr) -> Option<&'static str> { - match e { - QueryExpr::Aggregate { +fn quantile_kind(qe: &OperatorNode) -> &'static str { + fn walk(e: &OperatorNode) -> Option<&'static str> { + match e.expect_non_asap() { + NonASAPOp::Aggregate { measures, child, .. } => measures .iter() @@ -26,12 +27,12 @@ fn quantile_kind(qe: &QueryExpr) -> &'static str { _ => None, }) .or_else(|| walk(child)), - QueryExpr::TimeRange { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Project { child, .. } => walk(child), + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } + | NonASAPOp::Project { child, .. } => walk(child), _ => None, } } @@ -48,27 +49,26 @@ fn with_meta(q: &str, catalog: HistogramCatalog) -> &'static str { #[test] fn heuristic_baseline_is_unchanged_without_a_catalog() { - // Classic `by (le)`-bucket form → HistogramQuantile; anything else → Quantile. + // Classic buckets are represented; undeclared native samples are rejected. assert_eq!( heuristic( "histogram_quantile(0.9, sum by (le) (rate(http_request_duration_seconds_bucket[5m])))" ), "HQ" ); - assert_eq!(heuristic("histogram_quantile(0.9, native_latency)"), "Q"); + assert!(lower_promql( + "histogram_quantile(0.9, native_latency)", + AccuracyTarget::Exact + ) + .is_err()); } #[test] fn declared_classic_bucket_fixes_the_false_negative() { // A classic histogram exposed WITHOUT the `_bucket` suffix and queried with - // no `le` grouping/matcher: the heuristic wrongly routes it to the - // sketch-able Quantile. Declaring it `ClassicBucket` corrects it. + // no `le` grouping/matcher requires an explicit sample-type declaration. let q = "histogram_quantile(0.9, latency_seconds)"; - assert_eq!( - heuristic(q), - "Q", - "heuristic mis-routes the suffix-less classic histogram" - ); + assert!(lower_promql(q, AccuracyTarget::Exact).is_err()); assert_eq!( with_meta( q, @@ -80,7 +80,7 @@ fn declared_classic_bucket_fixes_the_false_negative() { } #[test] -fn declared_raw_or_native_fixes_the_false_positive() { +fn declared_raw_extension_and_native_gap_override_the_heuristic() { // A metric merely NAMED `…_bucket` that actually holds raw samples / a native // histogram: the heuristic wrongly routes it to bucket interpolation. let q = "histogram_quantile(0.9, foo_bucket)"; @@ -97,14 +97,9 @@ fn declared_raw_or_native_fixes_the_false_positive() { "Q", "raw samples are sketch-able" ); - assert_eq!( - with_meta( - q, - HistogramCatalog::new().with("foo_bucket", HistogramKind::Native) - ), - "Q", - "native histograms are sketch-able" - ); + let catalog = HistogramCatalog::new().with("foo_bucket", HistogramKind::Native); + assert!(lower_promql_with_histograms(q, AccuracyTarget::Exact, catalog.clone()).is_err()); + assert!(lower_promql_with_histograms("foo_bucket", AccuracyTarget::Exact, catalog).is_err()); } #[test] @@ -119,10 +114,12 @@ fn undeclared_metric_falls_back_to_the_heuristic() { ), "HQ" ); - assert_eq!( - with_meta("histogram_quantile(0.9, native_thing)", catalog), - "Q" - ); + assert!(lower_promql_with_histograms( + "histogram_quantile(0.9, native_thing)", + AccuracyTarget::Exact, + catalog + ) + .is_err()); } #[test] diff --git a/crates/frontend-promql/tests/maintained_population_horizon.rs b/crates/frontend-promql/tests/maintained_population_horizon.rs index 88b1c88fe..b4130bf86 100644 --- a/crates/frontend-promql/tests/maintained_population_horizon.rs +++ b/crates/frontend-promql/tests/maintained_population_horizon.rs @@ -1,37 +1,33 @@ mod support; use asap_aware_mapping::maintained_population::MaintainedPopulationStrategy; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator}; use asap_types::post_asap::maintained_population::PopulationInput; -use asap_types::post_asap::{SummaryExpr, ValueOperation}; use asap_types::types::AccuracyTarget; -use std::rc::Rc; // A population for a one-second selector must expire members after one second. #[test] fn population_preserves_selector_horizon() { - let root = Rc::new(support::lower_promql("sum(a)", AccuracyTarget::Exact).unwrap()); + let root = support::lower_promql("sum(a)", AccuracyTarget::Exact).unwrap(); let candidate = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) .candidate(&root) .unwrap(); - let SummaryExpr::ValueOperation { child, .. } = &candidate.expr else { + // The evaluation sits over the maintained population. + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &candidate.operator else { panic!() }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &child.expr - else { + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &child.operator else { panic!() }; let PopulationInput::CurrentSeries(spec) = &population.input else { panic!() }; assert_eq!(spec.lookback_ms, 1_000); - asap_types::post_asap::compile_post_asap_dag(&candidate).unwrap(); - let asap_types::pre_asap::QueryExpr::Aggregate { child: source, .. } = root.as_ref() else { + support::post_asap_dag(&candidate); + let NonASAPOp::Aggregate { child: source, .. } = root.expect_non_asap() else { panic!() }; - assert!(spec.matches_input(source)); + assert!(spec.matches_node(source)); let mut wrong = spec.clone(); wrong.lookback_ms = 300_000; - assert!(!wrong.matches_input(source)); + assert!(!wrong.matches_node(source)); } diff --git a/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs b/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs index 12afb94da..9ad59171e 100644 --- a/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs +++ b/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs @@ -29,7 +29,10 @@ use asap_frontend_promql::PromqlError as LoweringError; #[path = "../support.rs"] mod support; -use asap_types::pre_asap::{AggIntent, BinaryOpKind, CompareOpKind, QueryExpr, Reduction}; +use std::rc::Rc; + +use asap_types::ir::{BinaryOperator, NonASAPOp, OperatorNode, ScalarExpr}; +use asap_types::pre_asap::{AggIntent, BinaryOpKind, CompareOpKind, Reduction, ScalarValue}; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -44,78 +47,29 @@ fn queries() -> impl Iterator { } /// Lower, expecting success. -fn ok(q: &str) -> QueryExpr { +fn ok(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact) .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) } -/// Every `AggIntent` in the tree. -fn intents(e: &QueryExpr) -> Vec { +/// Every `AggIntent` in the tree. `AggIntent` only ever lives in +/// `Aggregate.measures`, never in a scalar position (issue #205); +/// `children()` also descends into the operators a scalar position reads. +fn intents(e: &OperatorNode) -> Vec { let mut out = Vec::new(); - fn go(e: &QueryExpr, out: &mut Vec) { - match e { - QueryExpr::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - go(child, out); - } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::Project { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => go(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { - left: lhs, - right: rhs, - .. - } - | QueryExpr::SetOp { - left: lhs, - right: rhs, - .. - } => { - go(lhs, out); - go(rhs, out); - } - QueryExpr::Concat { children, .. } => children.iter().for_each(|c| go(c, out)), - QueryExpr::PromqlVectorFromScalar(inner) | QueryExpr::PromqlScalarFromVector(inner) => { - go(inner, out) - } - // `AggIntent` only ever lives in `Aggregate.measures`, never in a - // scalar position (issue #205) — nothing to collect there. - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} + fn go(e: &OperatorNode, out: &mut Vec) { + if let Some(NonASAPOp::Aggregate { measures, .. }) = e.non_asap() { + out.extend(measures.iter().cloned()); + } + for child in e.children() { + go(child, out); } } go(e, &mut out); out } -fn has bool>(e: &QueryExpr, p: F) -> bool { +fn has bool>(e: &OperatorNode, p: F) -> bool { intents(e).iter().any(p) } @@ -180,15 +134,21 @@ fn vector_vs_vector_comparison_lowers_to_binaryop() { // Both operands are instant vectors → a `BinaryOp{Compare}` of two // ingestion-interval-bounded scans. let qe = ok("node_hwmon_temp_celsius > node_hwmon_temp_max_celsius"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + lhs, + rhs, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Compare(CompareOpKind::Gt)); assert!( - matches!(lhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(lhs.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); assert!( - matches!(rhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(rhs.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); } @@ -197,8 +157,8 @@ fn kube_replica_mismatch_comparison_lowers() { // Kubernetes: `kube_replicaset_spec_replicas != kube_replicaset_status_ready_replicas`. let qe = ok("kube_replicaset_spec_replicas != kube_replicaset_status_ready_replicas"); assert!(matches!( - &qe, - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Ne) + qe.expect_non_asap(), + NonASAPOp::BinaryOp { operator: BinaryOperator { kind: op, .. }, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Ne) )); } @@ -225,7 +185,11 @@ fn error_ratio_core_lowers() { // threshold: `sum(rate(failed[5m])) / sum(rate(total[5m]))` → a `BinaryOp(Div)` // of two cross-series sums over per-series rates. let qe = ok("sum(rate(litellm_proxy_failed_requests_metric_total[5m])) / sum(rate(litellm_proxy_total_requests_metric_total[5m]))"); - let QueryExpr::BinaryOp { op, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; assert!(matches!(op, BinaryOpKind::Arithmetic(_))); @@ -250,11 +214,11 @@ fn all_targets_missing_core_lowers() { // Prometheus self-monitoring `sum by (job) (up)` (the corpus query is // `… == 0`). Cross-series sum grouped positionally on `job`. let qe = ok("sum by (job) (up)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; @@ -273,19 +237,17 @@ fn all_targets_missing_core_lowers() { #[test] fn scalar_threshold_comparisons_lower_to_binaryop_scalar() { // ~822/949 corpus queries are ` `. The numeric - // threshold is now a `PromqlScalarBridge` operand of the `BinaryOp` (issue + // threshold is now a `ScalarExpr` operand of the `BinaryOp` (issue // #35) — the single biggest unblock for real alerts. for q in [ "prometheus_config_last_reload_successful != 1", "increase(prometheus_tsdb_compactions_failed_total[1m]) > 0", "rate(alertmanager_notifications_failed_total[3m]) > 0.05", ] { - let QueryExpr::BinaryOp { rhs, .. } = ok(q) else { - panic!("expected a BinaryOp for {q:?}"); - }; + let qe = ok(q); assert!( - matches!(rhs.as_ref(), QueryExpr::PromqlScalarBridge(_)), - "scalar threshold operand for {q:?}, got {rhs:?}" + matches!(qe.expect_non_asap(), NonASAPOp::Filter { .. }), + "{q}" ); } } @@ -336,12 +298,12 @@ fn vector_literal_lowers_to_a_labelless_vector() { // `vector(1)` — used in dead-man's-switch ("always firing") alerts. Now // lowers to a `PromqlVectorFromScalar` over the scalar `1` (issue #48). let qe = ok("vector(1)"); - let QueryExpr::PromqlVectorFromScalar(inner) = &qe else { + let NonASAPOp::PromqlVectorFromScalar(inner) = qe.expect_non_asap() else { panic!("expected PromqlVectorFromScalar, got {qe:?}"); }; - assert_eq!(inner.as_promql_scalar(), Some(1.0)); + assert!(matches!(inner, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); // The result is a vector: it carries a time index (unlike a bare scalar). - assert!(qe.output_schema().unwrap().time_index.is_some()); + assert!(qe.schema.time_index.is_some()); } #[test] @@ -352,14 +314,14 @@ fn without_grouping_lowers_to_the_exclusion_form() { // labels are stored and the kept set is runtime-resolved (issue #39). let qe = ok(r#"(min without (cpu) (rate(node_cpu_seconds_total{mode="idle"}[1h]))) > 0.8"#); // Top level is the `> 0.8` comparison; the `min without (cpu)` is its LHS. - let QueryExpr::BinaryOp { lhs, .. } = &qe else { + let NonASAPOp::Filter { child: lhs, .. } = qe.expect_non_asap() else { panic!("expected a comparison BinaryOp, got {qe:?}"); }; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = lhs.as_ref() + } = lhs.expect_non_asap() else { panic!("expected a `min without` Aggregate on the LHS, got {lhs:?}"); }; diff --git a/crates/frontend-promql/tests/observability/metrics_observability.rs b/crates/frontend-promql/tests/observability/metrics_observability.rs index 87d653ef2..c43f07ee8 100644 --- a/crates/frontend-promql/tests/observability/metrics_observability.rs +++ b/crates/frontend-promql/tests/observability/metrics_observability.rs @@ -6,15 +6,14 @@ use std::rc::Rc; -use asap_aware_mapping::replacement::{keep_pre_asap, RealizationError}; +use asap_aware_mapping::replacement::{retain_exact, RealizationError}; use asap_aware_mapping::{ - Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, + ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_frontend_promql::PromqlError; #[path = "../support.rs"] mod support; -use asap_types::post_asap::{SummaryExpr, SummaryNode}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -64,19 +63,18 @@ fn queries(corpus: &str) -> impl Iterator { .filter(|line| !line.is_empty() && !line.starts_with('#')) } -fn post_asap_candidate(expr: &QueryExpr) -> Result, RealizationError> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - match SketchAlgorithmStrategy::default_cost_model() +fn post_asap_candidate(root: &Rc) -> Result, RealizationError> { + let target = TargetSubDAG::new(root); + match ASAPStrategies::default_cost_model() .replacements(&target) .into_iter() .next() { Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDag(node), .. }) => Ok(node), - _ => keep_pre_asap(&root), + _ => retain_exact(root), } } @@ -99,9 +97,8 @@ fn benchmark_corpora_are_total_and_report_coverage() { Ok(expr) => { lowered += 1; match post_asap_candidate(&expr) { - Ok(node) if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) => { - post_asap_candidates += 1 - } + // An ASAP operator bound somewhere below the root. + Ok(node) if node.contains_asap() => post_asap_candidates += 1, Ok(_) => { post_asap_unchanged += 1; if std::env::var_os("METRICS_OBSERVABILITY_REPORT").is_some() { diff --git a/crates/frontend-promql/tests/observability/promql_corpus.rs b/crates/frontend-promql/tests/observability/promql_corpus.rs index f265edb45..85d74b3c9 100644 --- a/crates/frontend-promql/tests/observability/promql_corpus.rs +++ b/crates/frontend-promql/tests/observability/promql_corpus.rs @@ -15,37 +15,35 @@ use std::rc::Rc; -use asap_aware_mapping::replacement::{keep_pre_asap, RealizationError}; +use asap_aware_mapping::replacement::{retain_exact, RealizationError}; use asap_aware_mapping::{ - Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, + ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_frontend_promql::PromqlError as LoweringError; #[path = "../support.rs"] mod support; -use asap_types::post_asap::{SummaryExpr, SummaryNode}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use support::lower_promql; /// This crate has no "bind me one tree" public API any more — -/// `SketchAlgorithmStrategy::replacements` always returns every candidate, and +/// `ASAPStrategies::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so [`bind_tally`] /// gets one representative `Result` per query, matching what a totality /// check over the whole corpus wants. -fn bind(expr: &QueryExpr) -> Result, RealizationError> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - match SketchAlgorithmStrategy::default_cost_model() +fn bind(root: &Rc) -> Result, RealizationError> { + let target = TargetSubDAG::new(root); + match ASAPStrategies::default_cost_model() .replacements(&target) .into_iter() .next() { Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDag(node), .. }) => Ok(node), - _ => keep_pre_asap(&root), + _ => retain_exact(root), } } @@ -77,7 +75,10 @@ impl Tally { fn tally(corpus: &str) -> Tally { let mut t = Tally::default(); for q in queries(corpus) { - match lower_promql(q, AccuracyTarget::Exact) { + match asap_frontend_promql::lower_promql_query_workload( + &support::workload(q, AccuracyTarget::Exact), + 0, + ) { Ok(_) => t.lowered += 1, Err(LoweringError::Parse(_)) => t.unparseable += 1, Err(_) => t.rejected += 1, @@ -93,9 +94,10 @@ fn tally(corpus: &str) -> Tally { /// arm). #[derive(Default, Debug)] struct BindTally { - /// Root bound to `SummaryAgg`/`SummaryEstimate` — the pass did something. + /// An ASAP operator was bound somewhere below the root — the pass did + /// something. transformed: usize, - /// Root stayed `KeepPreAsap` — the pass left the query untouched. + /// The kept pre-ASAP tree — the pass left the query untouched. unchanged: usize, /// [`bind`] returned `Err` (schema derivation failed). errored: usize, @@ -108,7 +110,7 @@ fn bind_tally(corpus: &str, accuracy: AccuracyTarget) -> BindTally { continue; }; match bind(&tree) { - Ok(bound) if matches!(bound.expr, SummaryExpr::KeepPreAsap(_)) => t.unchanged += 1, + Ok(bound) if !bound.contains_asap() => t.unchanged += 1, Ok(_) => t.transformed += 1, Err(_) => t.errored += 1, } @@ -148,23 +150,17 @@ fn lowering_is_total_over_the_entire_corpus() { "testdata corpus unexpectedly small: {td:?}" ); - // Coverage tripwire: a code change that breaks lowering for a large slice of - // real PromQL trips this. Current numbers on the private promql-parser `asap` - // branch: docs 48 lowered / 1 rejected, testdata 1512 lowered / 76 rejected / - // 235 unparseable. The floors sit ~1% under those, so they guard regressions - // rather than pin an exact count — ratchet them up as coverage lands. - // - // The 235 unparseable are parser-fork gaps (issue #108); the rejections are - // lowering gaps (#109). Both shrink over time, so these floors normally only - // rise. Exception: the testdata floor was lowered to the measured 1485 when - // the 44 `fill` vector-matching queries became rejected rather than - // silently lowered without their fill semantics. + // Coverage tripwire after rejecting unrepresented native histogram samples: + // docs 48 lowered / 1 rejected; testdata 1121 lowered / 469 rejected / + // 233 parser gaps. Earlier coverage counted native histogram operations + // incorrectly treated as float quantiles. Keep the rejection cases in the + // corpus: accepting them requires a native histogram sample representation. assert!( docs.lowered >= 47, "docs lowering coverage regressed: {docs:?}" ); assert!( - td.lowered >= 1485, + td.lowered >= 1121, "testdata lowering coverage regressed: {td:?}" ); } diff --git a/crates/frontend-promql/tests/promql_binding_regressions.rs b/crates/frontend-promql/tests/promql_binding_regressions.rs index b5edf6049..416b6680d 100644 --- a/crates/frontend-promql/tests/promql_binding_regressions.rs +++ b/crates/frontend-promql/tests/promql_binding_regressions.rs @@ -33,9 +33,10 @@ fn irate_and_rate_have_distinct_canonical_intents() { /// PromQL count counts series even when two sample values are equal. #[test] fn count_is_row_count_not_distinct_sample_value_count() { - use asap_types::pre_asap::{AggIntent, QueryExpr}; + use asap_types::ir::NonASAPOp; + use asap_types::pre_asap::AggIntent; let tree = lower_promql("count(smoke_gauge)", AccuracyTarget::Exact).unwrap(); - let QueryExpr::Aggregate { measures, .. } = tree else { + let NonASAPOp::Aggregate { measures, .. } = tree.expect_non_asap() else { panic!("expected aggregate") }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); diff --git a/crates/frontend-promql/tests/promql_conformance.rs b/crates/frontend-promql/tests/promql_conformance.rs index 69fbb2230..eb41ca75c 100644 --- a/crates/frontend-promql/tests/promql_conformance.rs +++ b/crates/frontend-promql/tests/promql_conformance.rs @@ -31,22 +31,26 @@ // `__GAP`-suffixed test names intentionally SHOUT the documented divergences. #![allow(non_snake_case)] +use std::rc::Rc; use std::time::Duration; use asap_frontend_promql::PromqlError as LoweringError; mod support; +use asap_types::ir::{ + BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind, +}; use asap_types::pre_asap::schema::DataType; use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, MathFunc, - PromQLVectorSetOpKind, QueryExpr, Reduction, SampleKind, Source, TimeFunc, + AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, PromQLVectorSetOpKind, + Reduction, SampleKind, ScalarValue, Source, TimeFunc, }; use asap_types::types::AccuracyTarget; -use support::lower_promql; +use support::{lower_promql, promql_scalar}; // ── harness helpers ───────────────────────────────────────────────────────────── /// Lower, expecting success. -fn ok(q: &str) -> QueryExpr { +fn ok(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact) .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) } @@ -60,71 +64,29 @@ fn rejected(q: &str) -> LoweringError { } /// Every `AggIntent` anywhere in the tree, root-to-leaf. -fn intents(e: &QueryExpr) -> Vec { +fn intents(e: &OperatorNode) -> Vec { let mut out = Vec::new(); collect(e, &mut out); out } -fn collect(e: &QueryExpr, out: &mut Vec) { - match e { - QueryExpr::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - collect(child, out); - } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::Project { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => collect(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } => { - collect(lhs, out); - collect(rhs, out); - } - QueryExpr::Join { left, right, .. } | QueryExpr::SetOp { left, right, .. } => { - collect(left, out); - collect(right, out); - } - QueryExpr::Concat { children, .. } => children.iter().for_each(|c| collect(c, out)), - QueryExpr::PromqlVectorFromScalar(inner) | QueryExpr::PromqlScalarFromVector(inner) => { - collect(inner, out) - } - // `AggIntent` only ever lives in `Aggregate.measures`, never in a - // scalar position (issue #205) — nothing to collect there. - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} +/// `AggIntent` only ever lives in `Aggregate.measures`, never in a scalar +/// position (issue #205); `children()` also descends into the operators a +/// scalar position reads (`scalar(v)`). +fn collect(e: &OperatorNode, out: &mut Vec) { + if let Some(NonASAPOp::Aggregate { measures, .. }) = e.non_asap() { + out.extend(measures.iter().cloned()); + } + for child in e.children() { + collect(child, out); } } /// The first `Scan` reached by descending single-child nodes, with its metric /// name and predicate count. -fn first_scan(e: &QueryExpr) -> (String, usize) { - match e { - QueryExpr::Scan { +fn first_scan(e: &OperatorNode) -> (String, usize) { + match e.expect_non_asap() { + NonASAPOp::Scan { source, predicates, .. } => { let name = match source { @@ -133,45 +95,32 @@ fn first_scan(e: &QueryExpr) -> (String, usize) { }; (name, predicates.len()) } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_scan(child), + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::TimeShift { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_scan(child), other => panic!("no Scan reachable from {other:?}"), } } -fn has bool>(e: &QueryExpr, pred: F) -> bool { +fn has bool>(e: &OperatorNode, pred: F) -> bool { intents(e).iter().any(pred) } -/// Whether the tree contains a `Mul`-by-`PromqlScalarBridge(-1)` anywhere — the shape unary +/// Whether the tree contains a `Mul`-by-`ScalarExpr(-1)` anywhere — the shape unary /// negation lowers to (issue #36). -fn negates_via_scalar(e: &QueryExpr) -> bool { - let is_neg_one = |q: &QueryExpr| { - q.as_promql_scalar() - .is_some_and(|v| (v + 1.0).abs() < 1e-12) - }; - match e { - QueryExpr::BinaryOp { op, lhs, rhs, .. } => { - (*op == BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul) - && (is_neg_one(lhs) || is_neg_one(rhs))) - || negates_via_scalar(lhs) - || negates_via_scalar(rhs) - } - QueryExpr::Aggregate { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Project { child, .. } => negates_via_scalar(child), - _ => false, +fn negates_via_scalar(e: &OperatorNode) -> bool { + fn negative(expr: &ScalarExpr) -> bool { + matches!(expr, ScalarExpr::Negative { .. }) || expr.children().iter().any(|e| negative(e)) } + e.expect_non_asap() + .scalar_exprs() + .iter() + .any(|e| negative(e)) + || e.children().iter().any(|e| negates_via_scalar(e)) } // ───────────────────────────────────────────────────────────────────────────── @@ -193,10 +142,10 @@ fn promql_scan_schema_is_open() { // runtime-only, so the binding schema lists only the (ts, value) floor + // referenced labels and may be a subset of the runtime row. let qe = ok("node_cpu_seconds_total"); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected a TimeRange for a bare selector, got {qe:?}"); }; - let QueryExpr::Scan { schema, .. } = child.as_ref() else { + let NonASAPOp::Scan { schema, .. } = child.expect_non_asap() else { panic!("expected a Scan inside the TimeRange, got {qe:?}"); }; assert!( @@ -241,7 +190,7 @@ fn range_vector_selector_is_time_range() { // SEMANTICS: `[5m]` turns an instant vector into a range vector, // represented in the canonical tree as a dedicated `TimeRange` node. let qe = ok("node_cpu_seconds_total[5m]"); - let QueryExpr::TimeRange { range, .. } = &qe else { + let NonASAPOp::TimeRange { range, .. } = qe.expect_non_asap() else { panic!("expected TimeRange for a range-vector selector, got {qe:?}"); }; assert_eq!(*range, Duration::from_secs(300)); @@ -252,19 +201,44 @@ fn range_vector_selector_is_time_range() { // functions.test) // ───────────────────────────────────────────────────────────────────────────── +#[test] +fn selector_time_ranges_carry_their_kind() { + // SEMANTICS: an instant selector reads the latest sample within the + // ingestion interval (`Instant`); `m[5m]` is a range selection (`Range`). + // Same length is not the same shape: `m` and `m[1s]` stay distinct. + assert!(matches!( + ok("node_cpu_seconds_total").expect_non_asap(), + NonASAPOp::TimeRange { + kind: TimeRangeKind::Instant, + .. + } + )); + assert!(matches!( + ok("node_cpu_seconds_total[5m]").expect_non_asap(), + NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, + .. + } + )); + assert_ne!( + ok("node_cpu_seconds_total"), + ok("node_cpu_seconds_total[1s]") + ); +} + #[test] fn rate_range_lives_in_time_range_node() { // SEMANTICS: per-second average rate; the temporal range lives on the // enclosing `TimeRange` node, not inside the intent. let qe = ok("rate(http_requests_total[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); @@ -281,14 +255,14 @@ fn irate_maps_to_its_own_intent() { #[test] fn increase_range_lives_in_time_range_node() { let qe = ok("increase(http_requests_total[1h])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Increase])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(3600)); @@ -303,7 +277,7 @@ fn increase_range_lives_in_time_range_node() { fn sum_collapses_all_series() { // SEMANTICS: `sum(v)` → one output series. No grouping → no Partition. let qe = ok("sum(node_filesystem_size_bytes)"); - assert!(matches!(&qe, QueryExpr::Aggregate { .. })); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); } @@ -314,12 +288,12 @@ fn sum_by_groups_via_positional_aggregate() { // name-based Partition). SchemaResolver leaf = [ts, value, instance, job] (referenced // keys appended sorted), so the keys resolve to columns [2, 3]. let qe = ok("sum by(job, instance) (node_filesystem_size_bytes)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected positional Aggregate for `by(...)`, got {qe:?}"); }; @@ -330,7 +304,7 @@ fn sum_by_groups_via_positional_aggregate() { ); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); } @@ -369,11 +343,11 @@ fn sum_without_groups_by_the_complement() { // the runtime: the grouping is the exclusion form and the output schema // stays OPEN (unlike `by`, which freezes to closed). let qe = ok("sum without(instance) (node_filesystem_size_bytes)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; @@ -385,7 +359,7 @@ fn sum_without_groups_by_the_complement() { assert_eq!(by.keys().len(), 1, "the one excluded label (instance)"); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!( - !qe.output_schema().unwrap().closed, + !qe.schema.clone().closed, "a `without` result keeps an open schema (kept label set is runtime-only)" ); } @@ -418,16 +392,16 @@ fn group_aggregator_lowers_to_a_distinct_intent() { fn sum_of_rate_is_two_levels() { // SEMANTICS: per-series rate, THEN cross-series sum. Both must survive. let qe = ok("sum(rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{Sum}}, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -436,12 +410,12 @@ fn sum_by_of_rate_groups_outer_level() { // Outer cross-series Sum grouped on positional `Aggregate.by` over the // label-preserving inner Rate. Leaf = [ts, value, instance] → by = [2]. let qe = ok("sum by(instance) (rate(node_network_receive_bytes_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by instance, got {qe:?}"); }; @@ -449,8 +423,8 @@ fn sum_by_of_rate_groups_outer_level() { assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // child is the inner per-series Rate aggregate. assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -461,26 +435,29 @@ fn sum_by_of_over_time_groups_outer_level() { // preserving, so the key resolves positionally just like the rate case (no // name-based Partition). Leaf = [ts, value, instance] → by = [2]. let qe = ok("sum by(instance) (avg_over_time(node_cpu_seconds_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by instance, got {qe:?}"); }; assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // child is the inner per-series reduction: Aggregate{Avg} over TimeRange. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (per-series avg_over_time) under the Sum, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } // ───────────────────────────────────────────────────────────────────────────── @@ -501,7 +478,7 @@ fn over_time_functions_reduce_over_time_range() { ] { let qe = ok(q); assert!( - matches!(&qe, QueryExpr::Aggregate { .. }), + matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. }), "{q}: expected Aggregate" ); let matched = intents(&qe).iter().any(|i| match want { @@ -519,7 +496,7 @@ fn over_time_functions_reduce_over_time_range() { #[test] fn quantile_over_time_is_aggregate_over_time_range() { let qe = ok("quantile_over_time(0.9, request_latency_seconds[5m])"); - assert!(matches!(&qe, QueryExpr::Aggregate { .. })); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); assert!(has( &qe, |i| matches!(i, AggIntent::Quantile { q, .. } if (*q - 0.9).abs() < 1e-9) @@ -536,7 +513,7 @@ fn histogram_quantile_over_rate() { // φ-quantile from bucket rates. The `_bucket` metric marks the classic // cumulative-bucket form → `HistogramQuantile` (even without `sum by (le)`). let qe = ok("histogram_quantile(0.9, rate(demo_api_request_duration_seconds_bucket[5m]))"); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { panic!("expected Aggregate{{HistogramQuantile}}, got {qe:?}"); }; assert!( @@ -553,9 +530,9 @@ fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { let qe = ok( "histogram_quantile(0.99, sum by(le) (rate(demo_api_request_duration_seconds_bucket[5m])))", ); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); }; @@ -566,11 +543,11 @@ fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { )); // `sum by(le)` now survives as a positional Aggregate (by = [2], `le`), over // the inner Rate — no name-based Partition. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected `sum by(le)` as a positional Aggregate, got {child:?}"); }; @@ -586,7 +563,11 @@ fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { #[test] fn vector_arithmetic() { let qe = ok("node_memory_MemFree_bytes + node_memory_Cached_bytes"); - let QueryExpr::BinaryOp { op, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Add)); @@ -597,14 +578,17 @@ fn on_matching_with_group_left() { // SEMANTICS: many-to-one matching on a label subset. let qe = ok("rate(demo_cpu_usage_seconds_total[1m]) / on(instance, job) group_left demo_num_cpus"); - let QueryExpr::BinaryOp { - op, vector_match, .. - } = &qe - else { + let NonASAPOp::BinaryOp { operator, .. } = qe.expect_non_asap() else { panic!("expected BinaryOp, got {qe:?}"); }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); - let vm = vector_match.as_ref().expect("on(...) group_left present"); + assert_eq!( + operator.kind, + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + ); + let vm = operator + .vector_match + .as_ref() + .expect("on(...) group_left present"); assert_eq!(vm.labels, vec!["instance".to_string(), "job".to_string()]); assert!( vm.grouping.is_some(), @@ -617,15 +601,51 @@ fn vector_comparison_filters() { // SEMANTICS: `>` between two vectors keeps the LHS series where it holds. let qe = ok("go_goroutines > go_threads"); assert!( - matches!(&qe, QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Gt)) + matches!(qe.expect_non_asap(), NonASAPOp::BinaryOp { operator: BinaryOperator { kind: op, .. }, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Gt)) ); } +#[test] +fn comparison_bool_modifier_returns_zero_or_one() { + // SEMANTICS (operators.test): `bool` turns a filtering comparison into a + // 0/1-valued one. On a vector operand it is `return_bool` on the + // `BinaryOp`; between two scalars it is a `Case(Compare → 1, else 0)` + // scalar expression under PromQL numeric rules — and a scalar comparison + // without `bool` is not a PromQL expression at all. + let bool_flag = |q: &str| match ok(q).expect_non_asap() { + NonASAPOp::BinaryOp { return_bool, .. } => *return_bool, + NonASAPOp::Project { .. } => true, + NonASAPOp::Filter { .. } => false, + other => panic!("expected BinaryOp for {q}, got {other:?}"), + }; + assert!(bool_flag("go_goroutines > bool go_threads")); + assert!(bool_flag("go_goroutines > bool 0")); + assert!(!bool_flag("go_goroutines > go_threads")); + assert!(!bool_flag("go_goroutines > 0")); + + let qe = support::scalar_root("1 < bool 2"); + let ScalarExpr::Case { branches, .. } = &qe else { + panic!("expected a scalar Case, got {qe:?}"); + }; + assert!(matches!( + branches.as_slice(), + [( + ScalarExpr::Compare { + op: CompareOpKind::Lt, + semantics: ExprSemantics::Promql, + .. + }, + _ + )] + )); + rejected("1 < 2"); +} + #[test] fn unary_negation_lowers_as_multiply_by_minus_one() { // SEMANTICS (PromQL, issue #36): `-expr` flips the sign of every sample. // Now that a scalar operand exists (#35), it lowers as `expr * -1` — a `Mul` - // BinaryOp of the (label-preserving) vector against `PromqlScalarBridge(-1)`. These are + // BinaryOp of the (label-preserving) vector against `ScalarExpr(-1)`. These are // the five cases the old `__GAP` test pinned as rejected. for q in [ "-rate(http_errors_total[5m])", @@ -635,93 +655,38 @@ fn unary_negation_lowers_as_multiply_by_minus_one() { "sum(-node_cpu_seconds_total)", ] { let qe = ok(q); - // A `Mul`-by-`-1` against a `PromqlScalarBridge(-1)` appears somewhere in every tree. + // A `Mul`-by-`-1` against a `ScalarExpr(-1)` appears somewhere in every tree. assert!( negates_via_scalar(&qe), "no `* -1` negation found in {q}: {qe:?}" ); } - // `-some_metric` at the root: `Scan * PromqlScalarBridge(-1)`, schema follows the vector. - let QueryExpr::BinaryOp { - op, - lhs, - rhs, - vector_match, - } = &ok("-some_metric") - else { - panic!("expected a BinaryOp for `-some_metric`"); - }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul)); - assert!( - matches!(lhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })), - "vector on the left" - ); - assert!( - rhs.as_promql_scalar() - .is_some_and(|v| (v + 1.0).abs() < 1e-12), - "negation multiplies by PromqlScalarBridge(-1), got {rhs:?}" - ); - assert!( - vector_match.is_none(), - "scalar negation carries no vector match" - ); - // Label-preserving: the schema is the vector operand's, unchanged. - let schema = ok("-some_metric").output_schema().unwrap(); - assert_eq!( - schema - .columns - .iter() - .map(|c| c.name.as_str()) - .collect::>(), - vec!["ts", "value"], - ); - - // `sum(-m)` — the negation lowers inside the aggregate argument (issue #27 - // nesting), so the outer node is the `Sum` aggregate over the `Mul`. - let QueryExpr::Aggregate { - measures, child, .. - } = &ok("sum(-node_cpu_seconds_total)") - else { - panic!("expected an outer Aggregate for `sum(-m)`"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!( - child.as_ref(), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - .. - } - )); + let negated = ok("-some_metric"); + assert!(negates_via_scalar(&negated)); + assert!(negated.schema.has_promql_series_identity()); + assert!(negated.schema.time_index.is_some()); + let summed = ok("sum(-node_cpu_seconds_total)"); + assert!(has(&summed, |i| matches!(i, AggIntent::Sum { .. }))); + assert!(negates_via_scalar(&summed)); } #[test] fn unary_negation_of_constant_folds_to_scalar() { // `-(10*1024*1024)` — the operand is constant-foldable, so negation collapses - // to a single negated `PromqlScalarBridge` leaf (no `BinaryOp`), just like a bare literal. - assert!(ok("-(10*1024*1024)") - .as_promql_scalar() + // to a single negated `ScalarExpr` leaf (no `BinaryOp`), just like a bare literal. + assert!(promql_scalar(&support::scalar_root("-(10*1024*1024)")) .is_some_and(|v| (v + 10_485_760.0).abs() < 1e-6)); } #[test] fn double_unary_negation_nests() { - // `- -some_metric` — negation of a negation: `(m * -1) * -1`. Both levels - // lower; the value is unchanged but the structure is faithfully nested. - let QueryExpr::BinaryOp { op, lhs, .. } = &ok("- -some_metric") else { - panic!("expected outer BinaryOp for `- -some_metric`"); + let qe = ok("- -some_metric"); + let NonASAPOp::Project { child, .. } = qe.expect_non_asap() else { + panic!() }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul)); - assert!( - matches!( - lhs.as_ref(), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - .. - } - ), - "inner negation nests under the outer one" - ); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Project { .. })); + assert!(negates_via_scalar(child)); } #[test] @@ -754,39 +719,22 @@ fn count_maps_to_count_and_inherits_accuracy() { #[test] fn scalar_literal_operand_lowers_as_binaryop_scalar() { - // Issue #35: ` op ` — the numeric threshold is a - // `PromqlScalarBridge` operand of the `BinaryOp`, and constant arithmetic - // (`10*1024*1024`) is folded. The output schema is the vector side's. let qe = ok("node_filesystem_avail_bytes > 10*1024*1024"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &qe else { - panic!("expected a BinaryOp, got {qe:?}"); + let ScalarExpr::Compare { op, right, .. } = support::sample_expression(&qe) else { + panic!() }; - assert_eq!(*op, BinaryOpKind::Compare(CompareOpKind::Gt)); - assert!( - matches!(lhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })), - "vector on the left" - ); - assert!( - rhs.as_promql_scalar() - .is_some_and(|v| (v - 10_485_760.0).abs() < 1e-6), - "folded scalar threshold on the right, got {rhs:?}" - ); - // Schema derivation follows the vector side (a scalar contributes no labels). - assert!(qe.output_schema().is_ok()); + assert_eq!(*op, CompareOpKind::Gt); + assert_eq!(promql_scalar(right), Some(10_485_760.0)); } #[test] fn scalar_arithmetic_scales_the_vector() { - // `rate(m[5m]) * 100` — a unit conversion. Arithmetic BinaryOp of the vector - // with a `PromqlScalarBridge(100)`. let qe = ok("rate(m[5m]) * 100"); - let QueryExpr::BinaryOp { op, rhs, .. } = &qe else { - panic!("expected a BinaryOp, got {qe:?}"); + let ScalarExpr::Arithmetic { op, right, .. } = support::sample_expression(&qe) else { + panic!() }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul)); - assert!(rhs - .as_promql_scalar() - .is_some_and(|v| (v - 100.0).abs() < 1e-9)); + assert_eq!(*op, ArithmeticOpKind::Mul); + assert_eq!(promql_scalar(right), Some(100.0)); } // ───────────────────────────────────────────────────────────────────────────── @@ -797,12 +745,22 @@ fn scalar_arithmetic_scales_the_vector() { #[test] fn set_ops_lower_to_binaryop() { // SEMANTICS: or = union of label sets; and = intersection; unless = difference. - assert!(matches!(&ok("up{job=\"a\"} or up{job=\"b\"}"), - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::Or))); - assert!(matches!(&ok("node_network_mtu_bytes and node_up"), - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::And))); - assert!(matches!(&ok("node_network_mtu_bytes unless node_down"), - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::Unless))); + let set_op = |q: &str| match ok(q).expect_non_asap() { + NonASAPOp::BinaryOp { operator, .. } => operator.kind.clone(), + other => panic!("expected BinaryOp for {q}, got {other:?}"), + }; + assert_eq!( + set_op("up{job=\"a\"} or up{job=\"b\"}"), + BinaryOpKind::Set(PromQLVectorSetOpKind::Or) + ); + assert_eq!( + set_op("node_network_mtu_bytes and node_up"), + BinaryOpKind::Set(PromQLVectorSetOpKind::And) + ); + assert_eq!( + set_op("node_network_mtu_bytes unless node_down"), + BinaryOpKind::Set(PromQLVectorSetOpKind::Unless) + ); } // ───────────────────────────────────────────────────────────────────────────── @@ -824,7 +782,7 @@ fn topk_over_count_is_heavy_hitter() { fn bottomk_is_generic_sort_limit() { // SEMANTICS: bottom-k → generic ascending order + limit (no sketch). let qe = ok("bottomk(3, count_over_time(http_requests_total[5m]))"); - assert!(matches!(&qe, QueryExpr::Limit { .. })); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. })); } #[test] @@ -833,9 +791,9 @@ fn topk_over_nested_sum_preserves_weighted_topk_accuracy() { // The final rates are query-time values. Their ordering does not establish // frequency-sketch membership semantics. let qe = ok("topk(3, sum by(instance) (rate(node_cpu_seconds_total[5m])))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected weighted TopK aggregate, got {qe:?}"); }; @@ -861,18 +819,18 @@ fn outer_aggregate_over_nested_aggregate_nests() { // flat two-level template rejected. Each level survives into the // canonical tree (issue #27). let qe = ok("max(sum by (job) (rate(http_requests_total[5m])))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner `sum by (job)` Aggregate, got {child:?}"); }; @@ -895,12 +853,12 @@ fn outer_group_key_absent_from_nested_aggregate_is_dropped() { // the query lowers with the provably-absent key dropped, exactly // `sum(sum by (group)(…))`. let qe = ok(r#"sum(sum by (group)(http_requests{job="api-server"})) by (job)"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -910,7 +868,7 @@ fn outer_group_key_absent_from_nested_aggregate_is_dropped() { "absent `job` key dropped → global aggregate" ); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let QueryExpr::Aggregate { reduction, .. } = child.as_ref() else { + let NonASAPOp::Aggregate { reduction, .. } = child.expect_non_asap() else { panic!("expected inner `sum by (group)` Aggregate, got {child:?}"); }; assert_eq!( @@ -927,16 +885,16 @@ fn outer_group_key_present_after_inner_aggregate_still_resolves() { // resolving positionally — the absent-key drop only fires on provable // absence, never on a resolvable key. let qe = ok("sum(sum by (job, group)(http_requests)) by (job)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction: inner_reduction, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate, got {child:?}"); }; @@ -957,9 +915,9 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { // still resolve. Each `or` side is bound independently against its own // sub-tree, so the key is seeded as an inherited column on both sides. let qe = ok(r#"sum by (__name__)(metric_a{env="1"} or metric_b{env="2"})"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -969,12 +927,12 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { 1, "grouped by the one `__name__` key" ); - let QueryExpr::BinaryOp { lhs, rhs, .. } = child.as_ref() else { + let NonASAPOp::BinaryOp { lhs, rhs, .. } = child.expect_non_asap() else { panic!("expected a BinaryOp child, got {child:?}"); }; // Both independently-bound sides carry `__name__` at the same position, so // the outer group key is consistent across the union. - let (ls, rs) = (lhs.output_schema().unwrap(), rhs.output_schema().unwrap()); + let (ls, rs) = (lhs.schema.clone(), rhs.schema.clone()); assert_eq!(ls.column_id("__name__"), rs.column_id("__name__")); assert_eq!( ls.column_id("__name__"), @@ -983,8 +941,8 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { // The general case (a plain label, not just `__name__`) also lowers. assert!(matches!( - ok("sum by (job)(metric_a or metric_b)"), - QueryExpr::Aggregate { .. } + ok("sum by (job)(metric_a or metric_b)").expect_non_asap(), + NonASAPOp::Aggregate { .. } )); } @@ -994,15 +952,15 @@ fn aggregate_over_binary_op_nests() { // op over two range vectors. The old template only accepted a single inner // selector/call; now the binary op lowers and the outer sum wraps it. let qe = ok("sum(rate(a[5m]) + rate(b[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!( - matches!(child.as_ref(), QueryExpr::BinaryOp { .. }), + matches!(child.expect_non_asap(), NonASAPOp::BinaryOp { .. }), "argument lowers as a BinaryOp, got {child:?}" ); } @@ -1015,7 +973,10 @@ fn aggregate_over_binary_op_nests() { fn subquery_wraps_inner_query() { // SEMANTICS: `[range:res]` evaluates the inner query across a range. let qe = ok("rate(demo_api_request_duration_seconds_count[5m])[1h:]"); - assert!(matches!(&qe, QueryExpr::PromqlSubquery { .. })); + assert!(matches!( + qe.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } + )); assert!(has(&qe, |i| matches!(i, AggIntent::Rate))); } @@ -1026,12 +987,12 @@ fn over_time_of_subquery_reduces_per_series() { // then `max_over_time` takes the max of those samples *per series*. It lowers // to a per-series `Max` reduction over a `PromqlSubquery` (issue #27). let qe = ok("max_over_time(rate(demo_api_request_duration_seconds_count[5m])[1h:])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate at the root, got {qe:?}"); }; @@ -1044,7 +1005,7 @@ fn over_time_of_subquery_reduces_per_series() { // The reduction rides directly on the sub-query (the structural range marker // that keeps it label-preserving), which wraps the inner `rate`. assert!( - matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. }), + matches!(child.expect_non_asap(), NonASAPOp::PromqlSubquery { .. }), "the `Max` reduces over a PromqlSubquery, got {child:?}" ); assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Rate))); @@ -1055,16 +1016,19 @@ fn quantile_over_time_of_subquery_carries_phi() { // The `quantile_over_time` φ parameter is read from arg 0; the sub-query is // arg 1. It lowers to a per-series `Quantile(φ)` over the `PromqlSubquery`. let qe = ok("quantile_over_time(0.9, rate(demo[5m])[1h:])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::Quantile { q, .. }] if (*q - 0.9).abs() < 1e-9) ); - assert!(matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } + )); } #[test] @@ -1074,12 +1038,12 @@ fn aggregation_over_over_time_of_subquery_keeps_labels() { // survives for the OUTER cross-series `sum by (job)` to group on. If the // inner `Max` collapsed labels, `job` would not resolve here. let qe = ok("sum by (job) (max_over_time(rate(demo{job=\"api\"}[5m])[1h:]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -1089,20 +1053,20 @@ fn aggregation_over_over_time_of_subquery_keeps_labels() { ); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // Inner node is the per-series `max_over_time` reduction over the subquery. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction: inner_reduction, measures: inner_measures, child: inner_child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate, got {child:?}"); }; assert_eq!(inner_reduction, &Reduction::PerEntity); assert!(matches!(inner_measures.as_slice(), [AggIntent::Max { .. }])); assert!(matches!( - inner_child.as_ref(), - QueryExpr::PromqlSubquery { .. } + inner_child.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } )); } @@ -1124,69 +1088,69 @@ fn nested_subquery_from_prometheus_docs() { // the label-preserving `[ts, value]`. let qe = ok("max_over_time(deriv(rate(distance_covered_total[5s])[30s:5s])[10m:])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected `max_over_time` Aggregate at the root, got {qe:?}"); }; assert_eq!(reduction, &Reduction::PerEntity); assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let QueryExpr::PromqlSubquery { + let NonASAPOp::PromqlSubquery { range, resolution, child, - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected the outer `[10m:]` PromqlSubquery, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(600)); assert_eq!(*resolution, None, "`[10m:]` keeps the default resolution"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected the `deriv` Aggregate, got {child:?}"); }; assert_eq!(reduction, &Reduction::PerEntity); assert!(matches!(measures.as_slice(), [AggIntent::Deriv])); - let QueryExpr::PromqlSubquery { + let NonASAPOp::PromqlSubquery { range, resolution, child, - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected the inner `[30s:5s]` PromqlSubquery, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(30)); assert_eq!(*resolution, Some(Duration::from_secs(5))); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected the `rate` Aggregate, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected the `[5s]` TimeRange under rate, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(5)); // Per-series end to end: the schema keeps the (ts, value) floor and stays open. - let schema = qe.output_schema().expect("schema derivation"); + let schema = qe.schema.clone(); assert_eq!( schema - .columns + .fields .iter() .map(|c| c.name.as_str()) .collect::>(), @@ -1205,21 +1169,22 @@ fn offset_modifier_lowers_to_a_time_shift() { // past — a `TimeShift` wrapper over the selector (signed ms; a negative // offset shifts forward). Schema is unchanged (the shift only moves *when*). let qe = ok("http_requests_total offset 5m"); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange, got {qe:?}"); }; - let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, child } = child.expect_non_asap() else { panic!("expected a TimeShift, got {qe:?}"); }; assert_eq!(shift.offset_ms, 300_000); assert!(shift.at.is_none()); - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); // `offset -5m` shifts forward → negative ms. - let QueryExpr::TimeRange { child, .. } = &ok("http_requests_total offset -5m") else { + let qe = ok("http_requests_total offset -5m"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { panic!("expected a TimeShift"); }; assert_eq!(shift.offset_ms, -300_000); @@ -1230,29 +1195,31 @@ fn at_modifier_lowers_to_a_time_shift() { // SEMANTICS (PromQL, issue #40): `@ ` pins the evaluation to an absolute // instant (PromQL seconds → IR milliseconds); `@ start()` / `@ end()` anchor // to the query range bounds. - let QueryExpr::TimeRange { child, .. } = &ok("http_requests_total @ 1609746000") else { + let qe = ok("http_requests_total @ 1609746000"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { panic!("expected a TimeShift for `@ `"); }; assert_eq!(shift.at, Some(AtModifier::Timestamp(1_609_746_000_000))); assert_eq!(shift.offset_ms, 0); - let QueryExpr::TimeRange { child, .. } = &ok("http_requests_total @ start()") else { + let qe = ok("http_requests_total @ start()"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { panic!("expected a TimeShift for `@ start()`"); }; assert_eq!(shift.at, Some(AtModifier::Start)); // Offset and `@` compose: `@ end() offset 5m` carries both. let qe = ok("http_requests_total @ end() offset 5m"); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange, got {qe:?}"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { panic!("expected a TimeShift, got {qe:?}"); }; assert_eq!(shift.at, Some(AtModifier::End)); @@ -1265,21 +1232,21 @@ fn offset_on_a_ranged_selector_wraps_inside_the_time_range() { // `TimeShift` sits *under* the `TimeRange` (the 5m window is taken at the // shifted time), and the whole thing under the per-series `Rate` (#40). let qe = ok("rate(http_requests_total[5m] offset 1h)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected the rate Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { child, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { child, .. } = child.expect_non_asap() else { panic!("expected a TimeRange under rate, got {child:?}"); }; - let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, child } = child.expect_non_asap() else { panic!("expected a TimeShift under the TimeRange, got {child:?}"); }; assert_eq!(shift.offset_ms, 3_600_000); - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); } // ───────────────────────────────────────────────────────────────────────────── @@ -1313,9 +1280,9 @@ fn count_over_time_value_column_is_float64() { // #69: a per-series range reduction produces a PromQL sample value, which is // always float64. `count_over_time`'s `Count` intent types `Int64`, but the // derived `value` column must be `Float64` like every other range reducer. - let schema = ok("count_over_time(m[5m])").output_schema().unwrap(); + let schema = ok("count_over_time(m[5m])").schema.clone(); let value = schema - .columns + .fields .iter() .find(|c| c.name == "value") .expect("value column"); @@ -1335,12 +1302,12 @@ fn counter_derivative_functions_lower_to_distinct_intents() { ("resets(m[1h])", AggIntent::Resets), ] { let qe = ok(q); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate for {q:?}, got {qe:?}"); }; @@ -1355,7 +1322,7 @@ fn counter_derivative_functions_lower_to_distinct_intents() { "{q}: wrong intent" ); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { .. }), + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { .. }), "{q}: reduction rides on a TimeRange, got {child:?}" ); } @@ -1366,9 +1333,9 @@ fn predict_linear_carries_horizon_seconds() { // `predict_linear(v[w], t)` — the 2nd (scalar) arg is the prediction horizon // in seconds; it must be carried in the intent (it changes the result). let qe = ok("predict_linear(node_filesystem_avail_bytes[3h], 86400)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; @@ -1376,7 +1343,10 @@ fn predict_linear_carries_horizon_seconds() { measures.as_slice(), &[AggIntent::PredictLinear { seconds: 86400.0 }] ); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] @@ -1394,12 +1364,12 @@ fn aggregation_over_counter_derivative_keeps_labels() { // A counter-derivative is per-series (label-preserving), so an outer // `sum by (job)` can group on a label the inner `changes` preserved. let qe = ok(r#"sum by (job) (changes(m{job="api"}[15m]))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -1420,12 +1390,12 @@ fn outer_stat_over_counter_derivative_nests_two_levels() { // grouped outer (`avg by (dc)`) must resolve its key against the labels the // inner reduction preserved, threading any scalar param (predict horizon). let qe = ok("avg by (dc) (predict_linear(m[3h], 3600))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -1434,11 +1404,11 @@ fn outer_stat_over_counter_derivative_nests_two_levels() { "outer `avg by (dc)` groups on a label" ); assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction: inner_reduction, measures: inner_measures, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner per-series Aggregate, got {child:?}"); }; @@ -1458,11 +1428,14 @@ fn topk_over_counter_derivative_is_generic_sort_limit() { // `topk(k, deriv(...))` ranks the per-series derivative values — a generic // `Sort + Limit`, NOT a heavy-hitter `TopK` (that's only `count_over_time`). let qe = ok("topk(3, deriv(m[5m]))"); - let QueryExpr::Limit { n, child, .. } = &qe else { + let NonASAPOp::Limit { + n: Some(n), child, .. + } = qe.expect_non_asap() + else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 3); - assert!(matches!(child.as_ref(), QueryExpr::Sort { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Sort { .. })); assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Deriv))); assert!( !intents(&qe) @@ -1477,28 +1450,37 @@ fn counter_derivative_composes_in_binary_ops() { // As a vector operand: `delta(a[5m]) / delta(b[5m])` is a BinaryOp of two // per-series Delta reductions. let ratio = ok("delta(a[5m]) / delta(b[5m])"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &ratio else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + lhs, + rhs, + .. + } = ratio.expect_non_asap() + else { panic!("expected BinaryOp, got {ratio:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); assert!( - matches!(lhs.as_ref(), QueryExpr::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) + matches!(lhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) ); assert!( - matches!(rhs.as_ref(), QueryExpr::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) + matches!(rhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) ); // Under an aggregate over a binary op mixing a counter-derivative with // another per-series function: `sum(rate(m[5m]) + changes(m[5m]))`. let mixed = ok("sum(rate(m[5m]) + changes(m[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &mixed + } = mixed.expect_non_asap() else { panic!("expected Aggregate, got {mixed:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::BinaryOp { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::BinaryOp { .. } + )); assert!(intents(&mixed).iter().any(|i| matches!(i, AggIntent::Rate))); assert!(intents(&mixed) .iter() @@ -1522,12 +1504,12 @@ fn range_functions_over_a_subquery_reduce_per_series() { ("resets(sum(m)[5m:])", AggIntent::Resets), ] { let qe = ok(q); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("{q}: expected an Aggregate, got {qe:?}"); }; @@ -1542,7 +1524,7 @@ fn range_functions_over_a_subquery_reduce_per_series() { "{q}: wrong intent" ); assert!( - matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. }), + matches!(child.expect_non_asap(), NonASAPOp::PromqlSubquery { .. }), "{q}: reduces directly over the PromqlSubquery (no TimeRange), got {child:?}" ); } @@ -1570,8 +1552,8 @@ fn predict_linear_and_double_exp_over_a_subquery_carry_params() { #[test] fn histogram_quantile_classic_bucket_vs_native() { // Two lowerings of `histogram_quantile(φ, …)`: the classic cumulative-bucket - // form → exact `HistogramQuantile`; a native-histogram / raw-samples argument - // → the generic (sketch-able) `Quantile`. The classic form is recognised by + // form → exact `HistogramQuantile`; native samples require a new type. + // The classic form is recognised by // `by (le)`, a `_bucket` metric, or an `le` matcher (issue #43). for classic in [ "histogram_quantile(0.9, sum by (le) (rate(x_bucket[5m])))", @@ -1595,65 +1577,27 @@ fn histogram_quantile_classic_bucket_vs_native() { "histogram_quantile(0.9, my_native_histogram)", "histogram_quantile(0.9, request_duration_seconds)", // raw samples (your extension) ] { - let qe = ok(native); - assert!( - has( - &qe, - |i| matches!(i, AggIntent::Quantile { q, .. } if (*q - 0.9).abs() < 1e-9) - ), - "native/raw form → generic Quantile: {native}" - ); - assert!( - !has(&qe, |i| matches!(i, AggIntent::HistogramQuantile { .. })), - "{native}" - ); + rejected(native); } } #[test] -fn histogram_accessors_lower_to_per_series_intents() { - // `histogram_(v)` extracts a float per series from a native - // histogram — a per-series `Aggregate{[accessor]}` directly over the - // (instant) argument, no grouping. (`histogram_quantile` has its own two - // lowerings — see `histogram_quantile_classic_bucket_vs_native`.) - for (q, want) in [ - ("histogram_count(v)", AggIntent::HistogramCount), - ("histogram_sum(v)", AggIntent::HistogramSum), - ("histogram_avg(v)", AggIntent::HistogramAvg), - ("histogram_stddev(v)", AggIntent::HistogramStdDev), - ("histogram_stdvar(v)", AggIntent::HistogramStdVar), +fn native_histogram_accessors_are_explicit_gaps() { + // Native histogram samples have no typed representation yet. + for q in [ + "histogram_count(v)", + "histogram_sum(v)", + "histogram_avg(v)", + "histogram_stddev(v)", + "histogram_stdvar(v)", ] { - let qe = ok(q); - let QueryExpr::Aggregate { - reduction, - measures, - .. - } = &qe - else { - panic!("{q}: expected an Aggregate, got {qe:?}"); - }; - assert_eq!( - reduction, - &Reduction::PerEntity, - "{q}: per-series, no grouping" - ); - assert_eq!( - measures.as_slice(), - std::slice::from_ref(&want), - "{q}: wrong intent" - ); + rejected(q); } } #[test] -fn histogram_fraction_carries_its_bounds() { - // `histogram_fraction(lower, upper, v)` — bounds from args 0/1, vector arg 2. - let qe = ok("histogram_fraction(0, 0.2, v)"); - assert!(intents(&qe).iter().any(|i| matches!( - i, - AggIntent::HistogramFraction { lower, upper } - if *lower == 0.0 && (*upper - 0.2).abs() < 1e-9 - ))); +fn histogram_fraction_is_an_explicit_gap() { + rejected("histogram_fraction(0, 0.2, v)"); } // ───────────────────────────────────────────────────────────────────────────── @@ -1661,68 +1605,42 @@ fn histogram_fraction_carries_its_bounds() { // ───────────────────────────────────────────────────────────────────────────── #[test] -fn math_functions_lower_to_per_series_math_intents() { - // Each `f(v)` is a per-series element-wise value transform — a per-series - // `Aggregate{[Math(f)]}` over the (instant) argument, no grouping. - for (q, want) in [ - ("abs(v)", MathFunc::Abs), - ("ceil(v)", MathFunc::Ceil), - ("floor(v)", MathFunc::Floor), - ("sqrt(v)", MathFunc::Sqrt), - ("ln(v)", MathFunc::Ln), - ("log2(v)", MathFunc::Log2), - ("sgn(v)", MathFunc::Sgn), - ("sin(v)", MathFunc::Sin), - ("atanh(v)", MathFunc::Atanh), - ("deg(v)", MathFunc::Deg), - ("rad(v)", MathFunc::Rad), +fn math_functions_lower_to_typed_scalar_projections() { + for name in [ + "abs", "ceil", "floor", "sqrt", "ln", "log2", "sgn", "sin", "atanh", "deg", "rad", ] { - let qe = ok(q); - let QueryExpr::Aggregate { - reduction, - measures, - .. - } = &qe - else { - panic!("{q}: expected an Aggregate, got {qe:?}"); - }; - assert_eq!( - reduction, - &Reduction::PerEntity, - "{q}: per-series, no grouping" - ); + let query = ok(&format!("{name}(v)")); assert!( - matches!(measures.as_slice(), [AggIntent::Math(m)] if *m == want), - "{q}: wrong intent, got {measures:?}" + matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name:n,args } if n==&format!("promql_{name}") && args.len()==1) ); + query.validate_structure().unwrap(); } } #[test] fn clamp_and_round_carry_their_params() { - assert!(intents(&ok("clamp(v, 0, 100)")).iter().any( - |i| matches!(i, AggIntent::Math(MathFunc::Clamp { min, max }) if *min == 0.0 && *max == 100.0) - )); - assert!(intents(&ok("clamp_min(v, 1)")) - .iter() - .any(|i| matches!(i, AggIntent::Math(MathFunc::ClampMin { min }) if *min == 1.0))); - assert!(intents(&ok("clamp_max(v, 5)")) - .iter() - .any(|i| matches!(i, AggIntent::Math(MathFunc::ClampMax { max }) if *max == 5.0))); - // `round(v)` defaults the step to 1; `round(v, 5)` reads it. - assert!(intents(&ok("round(v)")).iter().any( - |i| matches!(i, AggIntent::Math(MathFunc::Round { to_nearest }) if *to_nearest == 1.0) - )); - assert!(intents(&ok("round(v, 5)")).iter().any( - |i| matches!(i, AggIntent::Math(MathFunc::Round { to_nearest }) if *to_nearest == 5.0) - )); + for (query, params) in [ + ("clamp(v,0,100)", vec![0.0, 100.0]), + ("clamp_min(v,1)", vec![1.0]), + ("clamp_max(v,5)", vec![5.0]), + ("round(v)", vec![1.0]), + ("round(v,5)", vec![5.0]), + ] { + let node = ok(query); + let ScalarExpr::FunctionCall { args, .. } = support::sample_expression(&node) else { + panic!() + }; + assert_eq!( + args.iter().skip(1).map(promql_scalar).collect::>(), + params.into_iter().map(Some).collect::>() + ); + } } #[test] fn pi_lowers_to_a_scalar_constant() { - // `pi()` is the constant π — a `PromqlScalarBridge` leaf, not a `Math` intent. - assert!(ok("pi()") - .as_promql_scalar() + // `pi()` is the constant π — a `ScalarExpr` leaf, not a `Math` intent. + assert!(promql_scalar(&support::scalar_root("pi()")) .is_some_and(|v| (v - std::f64::consts::PI).abs() < 1e-12)); } @@ -1747,11 +1665,11 @@ fn absent_keeps_matcher_labels_for_the_synthesized_output() { // `absent(v)` synthesizes its output labels from `v`'s equality matchers, so // those labels must survive into the schema — here `job` from `{job="x"}`. let qe = ok(r#"absent(up{job="x"})"#); - let cols = qe.output_schema().unwrap(); + let cols = qe.schema.clone(); assert!( - cols.columns.iter().any(|c| c.name == "job"), + cols.fields.iter().any(|c| c.name == "job"), "matcher label `job` kept, got {:?}", - cols.columns.iter().map(|c| &c.name).collect::>() + cols.fields.iter().map(|c| &c.name).collect::>() ); } @@ -1761,73 +1679,55 @@ fn absent_keeps_matcher_labels_for_the_synthesized_output() { #[test] fn time_lowers_to_the_eval_time_scalar() { - // SEMANTICS: `time()` is the query evaluation timestamp as a scalar — a leaf, - // not an aggregate over any series. - assert!(matches!(ok("time()"), QueryExpr::EvalTimestamp)); - // …and it is scalar-shaped: a single float `value`, no time index. - let sch = ok("time()").output_schema().unwrap(); - assert_eq!(sch.columns.len(), 1); - assert_eq!(sch.columns[0].name, "value"); - assert!(sch.time_index.is_none()); + assert!(matches!( + support::scalar_root("time()"), + ScalarExpr::EvalTimestamp + )); } #[test] fn time_minus_vector_is_the_uptime_pattern() { - // `time() - process_start_time_seconds` — the canonical uptime expression. - // The scalar `time()` broadcasts against the vector; the result takes the - // vector's schema. let qe = ok("time() - process_start_time_seconds"); - let QueryExpr::BinaryOp { lhs, op, .. } = &qe else { - panic!("expected a BinaryOp, got {qe:?}"); - }; - assert!(matches!(lhs.as_ref(), QueryExpr::EvalTimestamp)); - assert!(matches!( - op, - BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub) - )); - assert!(qe.output_schema().is_ok()); + assert!( + matches!(support::sample_expression(&qe), ScalarExpr::Arithmetic { op: ArithmeticOpKind::Sub, left, .. } if matches!(left.as_ref(), ScalarExpr::EvalTimestamp)) + ); + assert!(qe.schema.time_index.is_some()); } #[test] fn calendar_functions_lower_to_time_fn_intents() { - // SEMANTICS: each of these is a per-series float transform of its argument's - // timestamp (or, for `timestamp`, the sample's own time). functions.test. - for (q, want) in [ - ("timestamp(up)", TimeFunc::Timestamp), - ("minute(v)", TimeFunc::Minute), - ("hour(v)", TimeFunc::Hour), - ("day_of_week(v)", TimeFunc::DayOfWeek), - ("day_of_month(v)", TimeFunc::DayOfMonth), - ("day_of_year(v)", TimeFunc::DayOfYear), - ("month(v)", TimeFunc::Month), - ("year(v)", TimeFunc::Year), - ("days_in_month(v)", TimeFunc::DaysInMonth), + assert!(has(&ok("timestamp(up)"), |i| *i + == AggIntent::TimeFn(TimeFunc::Timestamp))); + for name in [ + "minute", + "hour", + "day_of_week", + "day_of_month", + "day_of_year", + "month", + "year", + "days_in_month", ] { - let qe = ok(q); + let query = ok(&format!("{name}(v)")); assert!( - has(&qe, |i| *i == AggIntent::TimeFn(want)), - "{q} → TimeFn({want:?}), got {:?}", - intents(&qe) + matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name:n,args } if n==&format!("promql_{name}") && args.len()==1) ); } } #[test] fn no_arg_calendar_function_reads_the_eval_time() { - // `day_of_week()` with no argument computes over the evaluation time itself, - // so it is a `TimeFn` aggregate whose child is the `EvalTimestamp` scalar. - let qe = ok("day_of_week()"); - let QueryExpr::Aggregate { - measures, child, .. - } = &qe - else { - panic!("expected an Aggregate, got {qe:?}"); + let query = ok("day_of_week()"); + let NonASAPOp::Project { child, .. } = query.expect_non_asap() else { + panic!() }; assert!(matches!( - measures.as_slice(), - [AggIntent::TimeFn(TimeFunc::DayOfWeek)] + child.expect_non_asap(), + NonASAPOp::PromqlVectorFromScalar(ScalarExpr::EvalTimestamp) )); - assert!(matches!(child.as_ref(), QueryExpr::EvalTimestamp)); + assert!( + matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name,.. } if name=="promql_day_of_week") + ); } #[test] @@ -1848,30 +1748,23 @@ fn vector_promotes_a_scalar_to_a_vector() { // SEMANTICS: `vector(s)` is the scalar→instant-vector bridge — a label-less // single series carrying the scalar's value. let qe = ok("vector(1)"); - let QueryExpr::PromqlVectorFromScalar(inner) = &qe else { + let NonASAPOp::PromqlVectorFromScalar(inner) = qe.expect_non_asap() else { panic!("expected PromqlVectorFromScalar, got {qe:?}"); }; - assert_eq!(inner.as_promql_scalar(), Some(1.0)); + assert!(matches!(inner, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); // Vector-typed: schema has a time index (a scalar leaf has none). - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.time_index.is_some()); - assert!(sch.columns.iter().any(|c| c.name == "value")); + assert!(sch.fields.iter().any(|c| c.name == "value")); } #[test] fn scalar_collapses_a_vector_to_a_scalar() { - // SEMANTICS: `scalar(v)` is the instant-vector→scalar bridge. - let qe = ok("scalar(node_load1)"); - let QueryExpr::PromqlScalarFromVector(inner) = &qe else { - panic!("expected PromqlScalarFromVector, got {qe:?}"); + let qe = support::scalar_root("scalar(node_load1)"); + let ScalarExpr::PromqlScalarFromVector(inner) = &qe else { + panic!() }; - let (metric, _) = first_scan(inner); - assert_eq!(metric, "node_load1"); - // PromqlScalarBridge-typed: single `value` column, no time index. - let sch = qe.output_schema().unwrap(); - assert!(sch.time_index.is_none()); - assert_eq!(sch.columns.len(), 1); - assert_eq!(sch.columns[0].name, "value"); + assert_eq!(first_scan(inner).0, "node_load1"); } #[test] @@ -1880,26 +1773,32 @@ fn vector_zero_is_a_vector_operand_of_a_set_op() { // vectors, so `vector(0)` must be a vector (a `PromqlVectorFromScalar`), never a // folded scalar operand. let qe = ok("up or vector(0)"); - let QueryExpr::BinaryOp { rhs, op, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + rhs, + .. + } = qe.expect_non_asap() + else { panic!("expected a BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Set(PromQLVectorSetOpKind::Or)); - assert!(matches!(rhs.as_ref(), QueryExpr::PromqlVectorFromScalar(_))); + assert!(matches!( + rhs.expect_non_asap(), + NonASAPOp::PromqlVectorFromScalar(_) + )); } #[test] fn scalar_of_a_vector_feeds_a_threshold_comparison() { - // `node_load1 > scalar(node_cpu_count)` — `scalar(...)` is a scalar operand, - // so the BinaryOp output takes the vector (lhs) side's schema. let qe = ok("node_load1 > scalar(node_cpu_count)"); - let QueryExpr::BinaryOp { lhs, rhs, .. } = &qe else { - panic!("expected a BinaryOp, got {qe:?}"); + let ScalarExpr::Compare { right, .. } = support::sample_expression(&qe) else { + panic!() }; - assert!(matches!(rhs.as_ref(), QueryExpr::PromqlScalarFromVector(_))); - // The BinaryOp output schema follows the vector (lhs) side, not the scalar. - let (metric, _) = first_scan(lhs); - assert_eq!(metric, "node_load1"); - assert!(qe.output_schema().unwrap().time_index.is_some()); + assert!(matches!( + right.as_ref(), + ScalarExpr::PromqlScalarFromVector(_) + )); + assert!(qe.schema.time_index.is_some()); } #[test] @@ -1909,13 +1808,13 @@ fn info_lowers_to_a_label_enrichment_join() { // (issue #84). The value/time axis pass through; the enriched labels are // runtime, so the schema stays the child's. let qe = ok("info(rate(http_requests_total[5m]))"); - let QueryExpr::PromqlInfoEnrich { selector, child } = &qe else { + let NonASAPOp::PromqlInfoEnrich { selector, child } = qe.expect_non_asap() else { panic!("expected an PromqlInfoEnrich, got {qe:?}"); }; assert!(selector.is_empty(), "no selector → default target_info"); // The child is the untouched input (a per-series rate reduction here). assert!(has(child, |i| *i == AggIntent::Rate)); - assert!(qe.output_schema().unwrap().time_index.is_some()); + assert!(qe.schema.clone().time_index.is_some()); } #[test] @@ -1925,7 +1824,7 @@ fn info_selector_carries_the_info_side_matchers() { // matchers are kept symbolically (not run through the single-metric selector // path). let qe = ok(r#"info(build_info, {__name__=~".+_info", another_data=~".+"})"#); - let QueryExpr::PromqlInfoEnrich { selector, .. } = &qe else { + let NonASAPOp::PromqlInfoEnrich { selector, .. } = qe.expect_non_asap() else { panic!("expected an PromqlInfoEnrich, got {qe:?}"); }; assert_eq!( @@ -1949,12 +1848,12 @@ fn info_composes_under_an_aggregation_and_over_a_time_shift() { // `offset` / `@` on the input now lower to a `TimeShift` under the info-join // (issue #40) — the enrichment composes over the shifted selector. assert!(matches!( - ok("info(metric @ 60)"), - QueryExpr::PromqlInfoEnrich { .. } + ok("info(metric @ 60)").expect_non_asap(), + NonASAPOp::PromqlInfoEnrich { .. } )); assert!(matches!( - ok("info(metric offset 1m)"), - QueryExpr::PromqlInfoEnrich { .. } + ok("info(metric offset 1m)").expect_non_asap(), + NonASAPOp::PromqlInfoEnrich { .. } )); } @@ -1967,21 +1866,21 @@ fn group_lowers_to_a_constant_group_intent() { // SEMANTICS: `group(v)` yields a constant 1 per group — a distinct intent, // NOT folded onto `sum` (which would return the value sum instead of 1). let qe = ok("group(up)"); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Group])); // Output column is the constant-1 `group` value. - let sch = qe.output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "group")); + let sch = qe.schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "group")); } #[test] fn group_by_keeps_the_grouping_keys() { // `group by (job) (up)` — the grouping keys ride on `Aggregate.by`. let qe = ok("group by (job) (up)"); - let sch = qe.output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "job")); + let sch = qe.schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "job")); assert!(has(&qe, |i| *i == AggIntent::Group)); } @@ -1991,15 +1890,15 @@ fn count_values_groups_by_value_and_synthesizes_a_label() { // value, counts each distinct value, and emits that value as a new label // `l`. The intent carries the label; schema gains a `Utf8` `l` column. let qe = ok(r#"count_values("version", build_version)"#); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::CountValues { label }] if label == "version") ); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); let version = sch - .columns + .fields .iter() .find(|c| c.name == "version") .expect("synthesized `version` label column"); @@ -2009,7 +1908,7 @@ fn count_values_groups_by_value_and_synthesizes_a_label() { "the value becomes a string label" ); assert!( - sch.columns.iter().any(|c| c.name == "count"), + sch.fields.iter().any(|c| c.name == "count"), "and a count column" ); } @@ -2023,9 +1922,9 @@ fn count_values_accepts_a_parenthesised_label_and_by_grouping() { &qe, |i| matches!(i, AggIntent::CountValues { label } if label == "v") )); - let sch = qe.output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "job")); - assert!(sch.columns.iter().any(|c| c.name == "v")); + let sch = qe.schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "job")); + assert!(sch.fields.iter().any(|c| c.name == "v")); } #[test] @@ -2034,10 +1933,10 @@ fn count_values_label_colliding_with_a_group_key_is_not_duplicated() { // with a group-by key. PromQL's synthesized label takes precedence; the // output must carry a single `job` column, never two. let qe = ok(r#"count_values by (job) ("job", version)"#); - let sch = qe.output_schema().unwrap(); - let jobs = sch.columns.iter().filter(|c| c.name == "job").count(); - assert_eq!(jobs, 1, "collision deduped, got {:?}", sch.columns); - assert!(sch.columns.iter().any(|c| c.name == "count")); + let sch = qe.schema.clone(); + let jobs = sch.fields.iter().filter(|c| c.name == "job").count(); + assert_eq!(jobs, 1, "collision deduped, got {:?}", sch.fields); + assert!(sch.fields.iter().any(|c| c.name == "count")); } #[test] @@ -2046,19 +1945,19 @@ fn limitk_and_limit_ratio_lower_to_series_sampling() { // series kept unchanged (NOT a ranking), so they lower to the dedicated // `PromqlSeriesSample` node, never `topk`'s `Sort → Limit` (issue #86). assert!(matches!( - ok("limitk(2, http_requests)"), - QueryExpr::PromqlSeriesSample { + ok("limitk(2, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitK(2), .. } )); assert!(matches!( - ok("limit_ratio(0.1, http_requests)"), - QueryExpr::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 0.1).abs() < 1e-9 + ok("limit_ratio(0.1, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 0.1).abs() < 1e-9 )); // Series-preserving: the output schema equals the input's (ts, value). - let sch = ok("limitk(2, http_requests)").output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "value")); + let sch = ok("limitk(2, http_requests)").schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "value")); assert!(sch.time_index.is_some()); } @@ -2067,12 +1966,12 @@ fn limit_ratio_keeps_a_negative_ratio_and_clamps_out_of_range() { // A negative ratio selects the complementary fraction — it must survive, not // be normalised away. Out-of-range magnitudes clamp to [-1, 1] (Prometheus). assert!(matches!( - ok("limit_ratio(-0.5, http_requests)"), - QueryExpr::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r + 0.5).abs() < 1e-9 + ok("limit_ratio(-0.5, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r + 0.5).abs() < 1e-9 )); assert!(matches!( - ok("limit_ratio(1.1, http_requests)"), - QueryExpr::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 1.0).abs() < 1e-9 + ok("limit_ratio(1.1, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 1.0).abs() < 1e-9 )); } @@ -2080,7 +1979,7 @@ fn limit_ratio_keeps_a_negative_ratio_and_clamps_out_of_range() { fn limitk_by_carries_the_grouping_and_composes_in_a_set_op() { // `limitk by (group)` samples per group; the grouping label is seeded. let qe = ok("limitk by (group) (2, http_requests)"); - let QueryExpr::PromqlSeriesSample { by, .. } = &qe else { + let NonASAPOp::PromqlSeriesSample { by, .. } = qe.expect_non_asap() else { panic!("expected a PromqlSeriesSample, got {qe:?}"); }; assert!(!by.is_empty(), "grouped sampling keeps its `by` keys"); @@ -2106,20 +2005,20 @@ fn dynamic_and_non_finite_sample_params_are_rejected() { // ───────────────────────────────────────────────────────────────────────────── /// Descend single-child nodes to the first `PromqlRelabel`. -fn first_relabel(e: &QueryExpr) -> &QueryExpr { - match e { - QueryExpr::PromqlRelabel { .. } => e, - QueryExpr::Aggregate { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } => first_relabel(child), +fn first_relabel(e: &OperatorNode) -> &OperatorNode { + match e.expect_non_asap() { + NonASAPOp::PromqlRelabel { .. } => e, + NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::TimeRange { child, .. } + | NonASAPOp::TimeShift { child, .. } => first_relabel(child), other => panic!("no PromqlRelabel reachable from {other:?}"), } } /// True when `value` is a `FunctionCall` with the given name. -fn is_fn_named(value: &QueryExpr, name: &str) -> bool { - matches!(value, QueryExpr::FunctionCall { name: n, .. } if n == name) +fn is_fn_named(value: &ScalarExpr, name: &str) -> bool { + matches!(value, ScalarExpr::FunctionCall { name: n, .. } if n == name) } #[test] @@ -2127,7 +2026,7 @@ fn label_replace_is_a_relabel_over_the_vector() { // SEMANTICS: `label_replace(v, dst, repl, src, regex)` rewrites the `dst` // label per series from a regex over `src`; the sample value is untouched. let qe = ok(r#"label_replace(up, "host", "$1", "instance", "(.+):.*")"#); - let QueryExpr::PromqlRelabel { dst, value, child } = &qe else { + let NonASAPOp::PromqlRelabel { dst, value, child } = qe.expect_non_asap() else { panic!("expected a PromqlRelabel, got {qe:?}"); }; assert_eq!(dst, "host"); @@ -2137,9 +2036,9 @@ fn label_replace_is_a_relabel_over_the_vector() { // The value expression is a `label_replace` fn reading the `src` label. assert!(is_fn_named(value, "label_replace")); // Output: the child's columns + the synthesized `host` label; value & ts kept. - let sch = qe.output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "host")); - assert!(sch.columns.iter().any(|c| c.name == "value")); + let sch = qe.schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "host")); + assert!(sch.fields.iter().any(|c| c.name == "value")); assert!(sch.time_index.is_some(), "the vector's time axis survives"); } @@ -2148,13 +2047,13 @@ fn label_join_concatenates_source_labels() { // SEMANTICS: `label_join(v, dst, sep, src…)` joins the source labels with // `sep` into `dst`. let qe = ok(r#"label_join(up, "combined", "-", "job", "instance")"#); - let QueryExpr::PromqlRelabel { dst, value, .. } = &qe else { + let NonASAPOp::PromqlRelabel { dst, value, .. } = qe.expect_non_asap() else { panic!("expected a PromqlRelabel, got {qe:?}"); }; assert_eq!(dst, "combined"); assert!(is_fn_named(value, "label_join")); - let sch = qe.output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "combined")); + let sch = qe.schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "combined")); } #[test] @@ -2164,10 +2063,12 @@ fn label_replace_composes_under_an_aggregation() { let qe = ok(r#"sum by (host) (label_replace(up, "host", "$1", "instance", "(.+):.*"))"#); // A PromqlRelabel sits below the outer Sum. let relabel = first_relabel(&qe); - assert!(matches!(relabel, QueryExpr::PromqlRelabel { dst, .. } if dst == "host")); + assert!( + matches!(relabel.expect_non_asap(), NonASAPOp::PromqlRelabel { dst, .. } if dst == "host") + ); assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); - let sch = qe.output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "host")); + let sch = qe.schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "host")); } // ───────────────────────────────────────────────────────────────────────────── @@ -2191,7 +2092,7 @@ fn extra_over_time_reducers_lower_to_per_series_intents() { assert!(has(&qe, |i| *i == want), "{q}: {:?}", intents(&qe)); // Per-series: the range window survives as a `TimeRange`. assert!( - matches!(&qe, QueryExpr::Aggregate { child, .. } if matches!(child.as_ref(), QueryExpr::TimeRange { .. })), + matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::TimeRange { .. })), "{q} keeps its range as a TimeRange" ); } @@ -2215,13 +2116,13 @@ fn sort_and_sort_desc_reorder_by_value_without_a_limit() { ("sort_desc(http_requests)", false), ] { let qe = ok(q); - let QueryExpr::Sort { keys, child, .. } = &qe else { + let NonASAPOp::Sort { keys, child, .. } = qe.expect_non_asap() else { panic!("{q}: expected a Sort, got {qe:?}"); }; assert_eq!(keys.len(), 1); assert_eq!(keys[0].ascending, ascending, "{q}"); // No Limit above the Sort — every series is preserved. - assert!(!matches!(&qe, QueryExpr::Limit { .. })); + assert!(!matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. })); // The value column is what it ranks on: descend to the scan. let (metric, _) = first_scan(child); assert_eq!(metric, "http_requests"); @@ -2233,24 +2134,21 @@ fn sort_by_label_orders_on_each_label_in_turn() { // `sort_by_label(v, "group", "instance", "job")` — one ascending sort key per // label, in argument order; the labels are seeded into the schema. let qe = ok(r#"sort_by_label(http_requests, "group", "instance", "job")"#); - let QueryExpr::Sort { keys, .. } = &qe else { + let NonASAPOp::Sort { keys, .. } = qe.expect_non_asap() else { panic!("expected a Sort, got {qe:?}"); }; assert_eq!(keys.len(), 3, "one key per label"); assert!(keys.iter().all(|k| k.ascending)); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); for label in ["group", "instance", "job"] { - assert!( - sch.columns.iter().any(|c| c.name == label), - "{label} seeded" - ); + assert!(sch.fields.iter().any(|c| c.name == label), "{label} seeded"); } } #[test] fn sort_by_label_desc_is_descending() { let qe = ok(r#"sort_by_label_desc(http_requests, "instance")"#); - let QueryExpr::Sort { keys, .. } = &qe else { + let NonASAPOp::Sort { keys, .. } = qe.expect_non_asap() else { panic!("expected a Sort, got {qe:?}"); }; assert!(keys.iter().all(|k| !k.ascending)); @@ -2259,28 +2157,43 @@ fn sort_by_label_desc_is_descending() { #[test] fn min_of_max_of_fold_constant_scalars() { // `min_of`/`max_of` are n-ary scalar reducers. When every argument is a - // constant they constant-fold to a `PromqlScalarBridge` leaf, just like scalar + // constant they constant-fold to a `ScalarExpr` leaf, just like scalar // arithmetic (#35) — the only form the intent algebra can hold (#89). - assert_eq!(ok("min_of(3, 5)").as_promql_scalar(), Some(3.0)); - assert_eq!(ok("max_of(3, 5)").as_promql_scalar(), Some(5.0)); - assert_eq!(ok("min_of(-2, -5)").as_promql_scalar(), Some(-5.0)); + assert_eq!( + promql_scalar(&support::scalar_root("min_of(3, 5)")), + Some(3.0) + ); + assert_eq!( + promql_scalar(&support::scalar_root("max_of(3, 5)")), + Some(5.0) + ); + assert_eq!( + promql_scalar(&support::scalar_root("min_of(-2, -5)")), + Some(-5.0) + ); // Nested folds and use as a threshold operand. assert_eq!( - ok("max_of(min_of(2, 3), 10)").as_promql_scalar(), + promql_scalar(&support::scalar_root("max_of(min_of(2, 3), 10)")), Some(10.0) ); let qe = ok("up > max_of(1, 2)"); - let QueryExpr::BinaryOp { rhs, .. } = &qe else { + let ScalarExpr::Compare { right: rhs, .. } = support::sample_expression(&qe) else { panic!("{qe:?}") }; - assert_eq!(rhs.as_promql_scalar(), Some(2.0)); + assert_eq!(promql_scalar(rhs), Some(2.0)); } #[test] fn min_of_max_of_ignore_nan_like_the_min_max_aggregators() { // A NaN argument is skipped (Prometheus `min`/`max` NaN semantics). - assert_eq!(ok("max_of(3, NaN)").as_promql_scalar(), Some(3.0)); - assert_eq!(ok("min_of(NaN, 3)").as_promql_scalar(), Some(3.0)); + assert_eq!( + promql_scalar(&support::scalar_root("max_of(3, NaN)")), + Some(3.0) + ); + assert_eq!( + promql_scalar(&support::scalar_root("min_of(NaN, 3)")), + Some(3.0) + ); } #[test] diff --git a/crates/frontend-promql/tests/promql_equivalence.rs b/crates/frontend-promql/tests/promql_equivalence.rs index d1177cc59..559400d0f 100644 --- a/crates/frontend-promql/tests/promql_equivalence.rs +++ b/crates/frontend-promql/tests/promql_equivalence.rs @@ -17,12 +17,14 @@ #![allow(non_snake_case)] +use std::rc::Rc; + mod support; -use asap_types::pre_asap::QueryExpr; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use support::lower_promql; -fn lo(q: &str) -> QueryExpr { +fn lo(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("{q:?} should lower: {e}")) } diff --git a/crates/frontend-promql/tests/promql_lowering.rs b/crates/frontend-promql/tests/promql_lowering.rs index 9d3b80de9..9bc14f5ce 100644 --- a/crates/frontend-promql/tests/promql_lowering.rs +++ b/crates/frontend-promql/tests/promql_lowering.rs @@ -1,10 +1,13 @@ //! End-to-end tests for PromQL → unresolved → canonical tree lowering. +use std::rc::Rc; use std::time::Duration; +use asap_types::ir::{ + BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind, +}; use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, QueryExpr, Reduction, ScalarValue, - Source, + AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, Reduction, ScalarValue, Source, }; use asap_types::types::AccuracyTarget; use asap_types::workload::{ @@ -16,7 +19,7 @@ use asap_frontend_promql::{lower_promql_workload, PromqlError as LoweringError}; mod support; use support::lower_promql; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } @@ -77,12 +80,12 @@ fn distinct_over_time_preserves_cardinality_accuracy_and_nested_windows() { #[test] fn bare_selector_is_scan_with_predicates() { let qe = lower(r#"http_requests_total{env="prod",status!="500"}"#); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected TimeRange, got {qe:?}"); }; - let QueryExpr::Scan { + let NonASAPOp::Scan { source, predicates, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Scan, got {qe:?}"); }; @@ -92,29 +95,32 @@ fn bare_selector_is_scan_with_predicates() { assert_eq!(predicates.len(), 2); assert!(predicates .iter() - .all(|p| matches!(p.0.as_ref(), QueryExpr::Compare { .. }))); + .all(|p| matches!(&p.0, ScalarExpr::Compare { .. }))); } #[test] fn regex_matcher_lowers_to_regex_compareop() { let qe = lower(r#"http_requests_total{path=~"/api/.*"}"#); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected TimeRange, got {qe:?}"); }; - let QueryExpr::Scan { + let NonASAPOp::Scan { predicates, schema, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Scan, got {qe:?}"); }; - let QueryExpr::Compare { left, op, right } = predicates[0].0.as_ref() else { + let ScalarExpr::Compare { + left, op, right, .. + } = &predicates[0].0 + else { panic!("expected Compare, got {:?}", predicates[0].0); }; assert_eq!(*op, CompareOpKind::Regex); // The label matcher's column is resolved positionally against the scan schema. let path_id = schema.column_id("path").expect("path in scan schema"); - assert!(matches!(left.as_ref(), QueryExpr::Column(id) if *id == path_id)); - assert!(matches!(right.as_ref(), QueryExpr::Literal(ScalarValue::Utf8(v)) if v == "/api/.*")); + assert!(matches!(left.as_ref(), ScalarExpr::Column(id) if *id == path_id)); + assert!(matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Utf8(v)) if v == "/api/.*")); } // ── *_over_time → Aggregate over TimeRange ────────────────────────────────────── @@ -122,12 +128,12 @@ fn regex_matcher_lowers_to_regex_compareop() { #[test] fn quantile_over_time_is_time_range_aggregate() { let qe = lower(r#"quantile_over_time(0.99, http_request_duration{env="prod"}[5m])"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; @@ -135,12 +141,14 @@ fn quantile_over_time_is_time_range_aggregate() { assert!( matches!(measures.as_slice(), [AggIntent::Quantile { q, .. }] if (*q - 0.99).abs() < 1e-9) ); - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); // The label matcher folded onto the Scan. - assert!(matches!(child.as_ref(), QueryExpr::Scan { predicates, .. } if predicates.len() == 1)); + assert!( + matches!(child.expect_non_asap(), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1) + ); } #[test] @@ -151,39 +159,42 @@ fn outer_sum_by_over_quantile_over_time_groups_positionally() { // a name-based Partition. Leaf = [ts, value, host, service] (referenced // names appended sorted) → host = col 2. let qe = lower(r#"sum by (host) (quantile_over_time(0.99, latency{service="web"}[5m]))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by host, got {qe:?}"); }; assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // Inner: Aggregate{Quantile} over TimeRange (per-series over_time reduction). - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (quantile_over_time) under the outer Sum, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Quantile { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] fn avg_over_time_maps_to_avg_intent() { let qe = lower("avg_over_time(cpu_seconds_total[10m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(600)); @@ -192,9 +203,9 @@ fn avg_over_time_maps_to_avg_intent() { #[test] fn stddev_and_stdvar_over_time() { let qe = lower("stddev_over_time(m[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate"); }; @@ -205,12 +216,15 @@ fn stddev_and_stdvar_over_time() { .. }] )); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); let qe = lower("stdvar_over_time(m[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate"); }; @@ -221,7 +235,10 @@ fn stddev_and_stdvar_over_time() { .. }] )); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] @@ -230,32 +247,33 @@ fn histogram_quantile_wraps_inner_in_quantile() { // not squashed away. The `_bucket` metric + `le` matcher mark the classic // form → `HistogramQuantile` over `Aggregate{Rate}` over Scan. let qe = lower(r#"histogram_quantile(0.95, rate(http_duration_seconds_bucket{le="0.5"}[5m]))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if (*q - 0.95).abs() < 1e-9) ); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate{{Rate}}, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { + let NonASAPOp::TimeRange { range, child: tr_child, - } = child.as_ref() + .. + } = child.expect_non_asap() else { panic!("expected TimeRange under Rate, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); assert!( - matches!(tr_child.as_ref(), QueryExpr::Scan { predicates, .. } if predicates.len() == 1) + matches!(tr_child.expect_non_asap(), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1) ); } @@ -266,9 +284,9 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { // `sum by (le)` aggregate; now the `le` grouping survives into the // canonical tree. let qe = lower(r#"histogram_quantile(0.99, sum by (le) (rate(http_requests_bucket[5m])))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); }; @@ -278,11 +296,11 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { ); // `sum by (le)` survives as a positional Aggregate (by = [2], `le`) over the // inner Rate — no name-based Partition. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected `sum by (le)` as a positional Aggregate, got {child:?}"); }; @@ -292,12 +310,12 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { /// The classic `histogram_quantile` aggregate: its `without` keys, `le` /// column, and output column names. -fn classic_histogram(qe: &QueryExpr) -> (Vec, usize, Vec) { - let QueryExpr::Aggregate { +fn classic_histogram(qe: &OperatorNode) -> (Vec, usize, Vec) { + let NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), measures, .. - } = qe + } = qe.expect_non_asap() else { panic!("expected a reducing Aggregate, got {qe:?}"); }; @@ -305,13 +323,7 @@ fn classic_histogram(qe: &QueryExpr) -> (Vec, usize, Vec) { panic!("expected HistogramQuantile, got {measures:?}"); }; assert!(by.is_without(), "histogram_quantile groups without (le)"); - let names = qe - .output_schema() - .unwrap() - .columns - .iter() - .map(|c| c.name.clone()) - .collect(); + let names = qe.schema.fields.iter().map(|c| c.name.clone()).collect(); (by.keys().to_vec(), *le, names) } @@ -321,11 +333,11 @@ fn classic_histogram(qe: &QueryExpr) -> (Vec, usize, Vec) { fn classic_histogram_quantile_groups_without_le() { let qe = lower("histogram_quantile(0.9, rate(http_duration_seconds_bucket[5m]))"); let (keys, le, names) = classic_histogram(&qe); - let QueryExpr::Aggregate { child, .. } = &qe else { + let NonASAPOp::Aggregate { child, .. } = qe.expect_non_asap() else { unreachable!() }; - let child = child.output_schema().unwrap(); - assert_eq!(child.columns[le].name, "le"); + let child = &child.schema; + assert_eq!(child.fields[le].name, "le"); assert_eq!(keys, vec![le]); assert_eq!(names, vec!["histogram_quantile"]); } @@ -349,14 +361,16 @@ fn classic_histogram_quantile_keeps_out_of_range_quantiles() { ("histogram_quantile(-1, x_bucket)", -1.), ("histogram_quantile(2, x_bucket)", 2.), ] { - let QueryExpr::Aggregate { measures, .. } = lower(query) else { + let root = lower(query); + let NonASAPOp::Aggregate { measures, .. } = root.expect_non_asap() else { panic!("{query}"); }; assert!( matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if *q == expected) ); } - let QueryExpr::Aggregate { measures, .. } = lower("histogram_quantile(NaN, x_bucket)") else { + let root = lower("histogram_quantile(NaN, x_bucket)"); + let NonASAPOp::Aggregate { measures, .. } = root.expect_non_asap() else { panic!("NaN"); }; assert!(matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if q.is_nan())); @@ -379,14 +393,14 @@ fn classic_histogram_quantile_rejects_an_argument_without_le() { #[test] fn rate_has_time_range_child_not_window() { let qe = lower("rate(http_requests_total[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate for rate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child (not Window), got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); @@ -395,14 +409,14 @@ fn rate_has_time_range_child_not_window() { #[test] fn increase_maps_to_increase_intent() { let qe = lower("increase(errors_total[1h])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate for increase, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Increase])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(3600)); @@ -415,21 +429,24 @@ fn sum_over_rate_keeps_both_levels() { // Regression: `sum(rate(m[w]))` — the most common PromQL shape — must keep // the cross-series Sum, not collapse to a bare per-series Rate. let qe = lower("sum(rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{Sum}}, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate{{Rate}}, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] @@ -438,20 +455,20 @@ fn sum_by_over_rate_groups_the_outer_sum() { // on a positional `Aggregate.by` (the same shape SQL produces) over the // label-preserving inner Rate. Leaf = [ts, value, job] → by = [2]. let qe = lower("sum by (job) (rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by job, got {qe:?}"); }; assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -459,16 +476,16 @@ fn sum_by_over_rate_groups_the_outer_sum() { fn count_over_rate_keeps_both_levels() { // The `Outer::Count` sibling of the `sum(rate(...))` bug. let qe = lower("count(rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{Count}}, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -487,12 +504,12 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { ), ] { let tree = lower(query); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, reduction: actual, child, .. - } = &tree + } = tree.expect_non_asap() else { panic!("expected outer Count: {tree:?}"); }; @@ -501,12 +518,12 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { "{query}: {tree:?}" ); assert_eq!(actual, &reduction, "{query}"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, reduction, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner per-series Cardinality: {tree:?}"); }; @@ -516,7 +533,7 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { ); assert_eq!(reduction, &Reduction::PerEntity, "{query}"); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, .. } if range.as_secs() == 300) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, .. } if range.as_secs() == 300) ); } } @@ -552,14 +569,17 @@ fn count_never_lowers_to_distinct_sample_values() { #[test] fn count_over_time_is_count_intent() { let qe = lower("count_over_time(m[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] @@ -568,26 +588,29 @@ fn outer_count_counts_series() { // over the window (label-preserving), outer cross-series row count grouped // on a positional `Aggregate.by`. Leaf = [ts, value, symbol] → symbol = col 2. let qe = lower("count by (symbol) (count_over_time(financial_last_trade_price[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by symbol, got {qe:?}"); }; assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); // Inner: Aggregate{Count} over TimeRange (per-series count_over_time). - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (count_over_time) under the outer count, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } // ── topk / bottomk ──────────────────────────────────────────────────────────── @@ -596,12 +619,12 @@ fn outer_count_counts_series() { fn topk_over_count_is_heavy_hitter_topk() { let qe = lower(r#"topk by (service) (10, count_over_time(requests{env="prod"}[1m]))"#); // Heavy-hitter: Aggregate{TopK} with grouping resolved to positional ids. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate with TopK, got {qe:?}"); }; @@ -612,29 +635,29 @@ fn topk_over_count_is_heavy_hitter_topk() { [AggIntent::TopK { k: 10, .. }] )); // The count_over_time under the TopK is a TimeRange-backed aggregate. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (count_over_time) under TopK, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { panic!("expected TimeRange under Count aggregate, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(60)); - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); } #[test] fn topk_over_sum_is_value_weighted_heavy_hitter_topk() { let qe = lower(r#"topk by (service) (5, sum_over_time(requests{env="prod"}[1m]))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate with TopK, got {qe:?}"); }; @@ -643,29 +666,38 @@ fn topk_over_sum_is_value_weighted_heavy_hitter_topk() { measures.as_slice(), [AggIntent::TopK { k: 5, .. }] )); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (sum_over_time) under TopK, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] fn topk_over_avg_is_generic_sort_limit() { let qe = lower("topk by (host) (5, avg_over_time(cpu[5m]))"); - let QueryExpr::Limit { n, offset, child } = &qe else { + let NonASAPOp::Limit { + n: Some(n), + offset, + child, + .. + } = qe.expect_non_asap() + else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 5); assert_eq!(*offset, 0); - let QueryExpr::Sort { + let NonASAPOp::Sort { keys, partition_by, child, - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Sort under Limit, got {child:?}"); }; @@ -678,7 +710,7 @@ fn topk_over_avg_is_generic_sort_limit() { // Underneath: the label-preserving windowed avg aggregate (by: []), no // intervening Partition. assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { reduction, measures, .. } + matches!(child.expect_non_asap(), NonASAPOp::Aggregate { reduction, measures, .. } if reduction == &Reduction::PerEntity && matches!(measures.as_slice(), [AggIntent::Avg { .. }])), "expected bare per-series Avg aggregate under Sort, got {child:?}" ); @@ -687,7 +719,7 @@ fn topk_over_avg_is_generic_sort_limit() { #[test] fn ungrouped_topk_over_sum_is_heavy_hitter() { let qe = lower("topk(5, sum_over_time(m[5m]))"); - assert!(matches!(&qe, QueryExpr::Aggregate { .. })); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); assert!(has_intent(&qe, |i| matches!(i, AggIntent::Sum { .. }))); assert!(has_intent(&qe, |i| matches!( i, @@ -699,11 +731,14 @@ fn ungrouped_topk_over_sum_is_heavy_hitter() { fn bottomk_over_count_is_generic_sort_ascending() { // `bottomk` is never a heavy-hitter (descending=false), even over count. let qe = lower("bottomk(3, count_over_time(m[5m]))"); - let QueryExpr::Limit { n, child, .. } = &qe else { + let NonASAPOp::Limit { + n: Some(n), child, .. + } = qe.expect_non_asap() + else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 3); - let QueryExpr::Sort { keys, .. } = child.as_ref() else { + let NonASAPOp::Sort { keys, .. } = child.expect_non_asap() else { panic!("expected Sort"); }; assert!(keys[0].ascending, "bottomk ranks ascending"); @@ -715,11 +750,14 @@ fn bottomk_over_count_is_generic_sort_ascending() { #[test] fn bottomk_is_always_generic_sort_ascending() { let qe = lower("bottomk(3, count_over_time(m[5m]))"); - let QueryExpr::Limit { n, child, .. } = &qe else { + let NonASAPOp::Limit { + n: Some(n), child, .. + } = qe.expect_non_asap() + else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 3); - let QueryExpr::Sort { keys, .. } = child.as_ref() else { + let NonASAPOp::Sort { keys, .. } = child.expect_non_asap() else { panic!("expected Sort"); }; assert!(keys[0].ascending, "bottomk ranks ascending"); @@ -731,12 +769,12 @@ fn topk_count_output_schema_carries_group_key() { // (`service`) flows through to the outer TopK's `by` column. Leaf schema = // [ts, value, service] → TopK groups on service (col 2). let qe = lower("topk by (service) (5, count_over_time(m[1m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate{{TopK}}, got {qe:?}"); }; @@ -750,14 +788,17 @@ fn topk_count_output_schema_carries_group_key() { [AggIntent::TopK { k: 5, .. }] )); // Inner Count aggregate is visible with its TimeRange child. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate{{Count}}, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } // ── binary ops ──────────────────────────────────────────────────────────────── @@ -765,22 +806,32 @@ fn topk_count_output_schema_carries_group_key() { #[test] fn binary_op_division() { let qe = lower("rate(a[5m]) / rate(b[5m])"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + lhs, + rhs, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); assert!( - matches!(lhs.as_ref(), QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) + matches!(lhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) ); assert!( - matches!(rhs.as_ref(), QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) + matches!(rhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) ); } #[test] fn binary_op_with_on_grouping() { let qe = lower("a / on(host) b"); - let QueryExpr::BinaryOp { vector_match, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { vector_match, .. }, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; let vm = vector_match.as_ref().expect("vector_match present"); @@ -793,18 +844,38 @@ fn binary_op_with_on_grouping() { // carry it. #[test] fn bool_comparisons_are_distinct() { - let op = |q: &str| match lower(q) { - QueryExpr::BinaryOp { op, .. } => op, + let op = |q: &str| match lower(q).expect_non_asap() { + NonASAPOp::BinaryOp { + operator, + return_bool, + .. + } => (operator.kind.clone(), *return_bool), + NonASAPOp::Filter { + pred: asap_types::ir::Predicate(ScalarExpr::Compare { op, .. }), + .. + } => (BinaryOpKind::Compare(op.clone()), false), + NonASAPOp::Project { cols, .. } => { + let ScalarExpr::Case { branches, .. } = &cols[1].expr else { + panic!() + }; + let ScalarExpr::Compare { op, .. } = &branches[0].0 else { + panic!() + }; + (BinaryOpKind::Compare(op.clone()), true) + } other => panic!("expected BinaryOp, got {other:?}"), }; - assert_eq!(op("a > 1"), BinaryOpKind::Compare(CompareOpKind::Gt)); + assert_eq!( + op("a > 1"), + (BinaryOpKind::Compare(CompareOpKind::Gt), false) + ); assert_eq!( op("a > bool 1"), - BinaryOpKind::CompareBool(CompareOpKind::Gt) + (BinaryOpKind::Compare(CompareOpKind::Gt), true) ); assert_eq!( op("a == bool on(job) b"), - BinaryOpKind::CompareBool(CompareOpKind::Eq) + (BinaryOpKind::Compare(CompareOpKind::Eq), true) ); } @@ -814,7 +885,7 @@ fn binary_op_binds_each_branch_against_its_own_schema() { // single root schema threaded to both branches, the left scan would leak the // right's group key (and vice-versa). Per-branch binding keeps them separate. let qe = lower("count by (job) (a) / count by (region) (b)"); - let QueryExpr::BinaryOp { lhs, rhs, .. } = &qe else { + let NonASAPOp::BinaryOp { lhs, rhs, .. } = qe.expect_non_asap() else { panic!("expected BinaryOp, got {qe:?}"); }; let lcols = scan_columns(lhs); @@ -830,25 +901,25 @@ fn binary_op_binds_each_branch_against_its_own_schema() { } /// Collect every `AggIntent` in the tree, root-to-leaf. -fn all_intents(e: &QueryExpr) -> Vec { +fn all_intents(e: &OperatorNode) -> Vec { let mut out = Vec::new(); collect_intents(e, &mut out); out } -fn collect_intents(e: &QueryExpr, out: &mut Vec) { - match e { - QueryExpr::Aggregate { +fn collect_intents(e: &OperatorNode, out: &mut Vec) { + match e.expect_non_asap() { + NonASAPOp::Aggregate { measures, child, .. } => { out.extend(measures.iter().cloned()); collect_intents(child, out); } - QueryExpr::TimeRange { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => collect_intents(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } => { + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => collect_intents(child, out), + NonASAPOp::BinaryOp { lhs, rhs, .. } => { collect_intents(lhs, out); collect_intents(rhs, out); } @@ -857,19 +928,19 @@ fn collect_intents(e: &QueryExpr, out: &mut Vec) { } /// True if any `AggIntent` anywhere in the tree satisfies `pred`. -fn has_intent bool>(e: &QueryExpr, pred: F) -> bool { +fn has_intent bool>(e: &OperatorNode, pred: F) -> bool { all_intents(e).iter().any(pred) } -/// Column names on the first `Scan` reachable by descending single-child nodes. -fn scan_columns(e: &QueryExpr) -> Vec { - match e { - QueryExpr::Scan { schema, .. } => schema.columns.iter().map(|c| c.name.clone()).collect(), - QueryExpr::Aggregate { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => scan_columns(child), +/// Field names on the first `Scan` reachable by descending single-child nodes. +fn scan_columns(e: &OperatorNode) -> Vec { + match e.expect_non_asap() { + NonASAPOp::Scan { schema, .. } => schema.fields.iter().map(|c| c.name.clone()).collect(), + NonASAPOp::Aggregate { child, .. } + | NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => scan_columns(child), _ => vec![], } } @@ -883,12 +954,12 @@ fn without_grouping_lowers_to_the_exclusion_form() { // label is stored positionally (the SchemaResolver seeds it), the grouping is the // `without` form, and the output schema stays open. let qe = lower("sum without (instance) (rate(m[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; @@ -899,10 +970,10 @@ fn without_grouping_lowers_to_the_exclusion_form() { // The inner per-series rate is preserved (label-preserving) under the outer // cross-series `without` reduction. assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + matches!(child.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) ); - assert!(!qe.output_schema().unwrap().closed); + assert!(!qe.schema.clone().closed); } // ── parameter validation (reject rather than silently truncate/garble) ────────── @@ -920,7 +991,7 @@ fn out_of_range_quantile_phi_is_accepted() { for query in [ "quantile(1.5, up)", "quantile_over_time(1.5, m[5m])", - "histogram_quantile(2.0, rate(b[5m]))", + "histogram_quantile(2.0, rate(b_bucket[5m]))", ] { assert!( lower_promql(query, AccuracyTarget::Exact).is_ok(), @@ -979,7 +1050,7 @@ fn accuracy_target_flows_into_quantile_intent() { AccuracyTarget::Epsilon(0.01), ) .unwrap(); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { panic!("expected Aggregate"); }; assert!(matches!( @@ -998,11 +1069,11 @@ fn aggregate_output_schema_preserves_time_axis_and_labels() { // predicate columns) to the scan schema, so `env` appears as a column // even though it is only used as a filter. // per_series_reduction_schema preserves the time axis and all label columns. - let QueryExpr::Aggregate { .. } = &qe else { + let NonASAPOp::Aggregate { .. } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; - let schema = qe.output_schema().expect("aggregate schema"); - let names: Vec<&str> = schema.columns.iter().map(|c| c.name.as_str()).collect(); + let schema = &qe.schema; + let names: Vec<&str> = schema.fields.iter().map(|c| c.name.as_str()).collect(); assert_eq!(names, vec!["ts", "value", "env"]); assert_eq!( schema.time_index, @@ -1016,19 +1087,19 @@ fn scan_schema_carries_ts_value_and_group_keys() { // `service` is a group key → the SchemaResolver lands it in the self-contained // Scan schema (positional). `env` is only a filter, so it is not a column. let qe = lower("count by (service) (count_over_time(requests[1m]))"); - fn find_scan(n: &QueryExpr) -> &QueryExpr { - match n { - QueryExpr::Scan { .. } => n, - QueryExpr::TimeRange { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Filter { child, .. } => find_scan(child), + fn find_scan(n: &OperatorNode) -> &OperatorNode { + match n.expect_non_asap() { + NonASAPOp::Scan { .. } => n, + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Filter { child, .. } => find_scan(child), other => panic!("unexpected node {other:?}"), } } - let QueryExpr::Scan { schema, .. } = find_scan(&qe) else { + let NonASAPOp::Scan { schema, .. } = find_scan(&qe).expect_non_asap() else { unreachable!() }; - let mut names: Vec<&str> = schema.columns.iter().map(|c| c.name.as_str()).collect(); + let mut names: Vec<&str> = schema.fields.iter().map(|c| c.name.as_str()).collect(); names.sort(); assert_eq!(names, vec!["service", "ts", "value"]); assert_eq!(schema.time_index, Some(0)); // ts @@ -1113,19 +1184,23 @@ fn reducing_group_by_lowers_to_aggregate_by() { // Cross-series reduce, no keys → bare `Aggregate { reduction: Reduce([]) }`. let q = lower("sum(http_requests_total)"); assert!( - matches!(q, QueryExpr::Aggregate { ref reduction, .. } if reduction == &Reduction::by(vec![])) + matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } if reduction == &Reduction::by(vec![])) ); // Cross-series reduce grouped by a label → `Aggregate.reduction`. let q = lower("sum by (job) (http_requests_total)"); - assert!(matches!(q, QueryExpr::Aggregate { ref reduction, .. } - if reduction.expect_reduce().len() == 1)); + assert!( + matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } + if reduction.expect_reduce().len() == 1) + ); // Reduce over a label-preserving `rate` grouped by a label → still // `Aggregate.reduction` (the keys resolve against rate's preserved schema). let q = lower("sum by (job) (rate(http_requests_total[5m]))"); - assert!(matches!(q, QueryExpr::Aggregate { ref reduction, .. } - if reduction.expect_reduce().len() == 1)); + assert!( + matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } + if reduction.expect_reduce().len() == 1) + ); } #[test] @@ -1134,20 +1209,20 @@ fn generic_topk_grouping_lowers_to_sort_partition_by() { // reducing → the grouping rides on `Sort.partition_by`, and the windowed // reduction beneath stays label-preserving (`by: []`). No `Partition` node. let q = lower("topk by (host) (5, avg_over_time(cpu[5m]))"); - let QueryExpr::Limit { child, .. } = &q else { + let NonASAPOp::Limit { child, .. } = q.expect_non_asap() else { panic!("expected Limit, got {q:?}"); }; - let QueryExpr::Sort { + let NonASAPOp::Sort { partition_by, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Sort, got {child:?}"); }; assert_eq!(partition_by, &vec![2], "host is col 2 in [ts, value, host]"); assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { reduction, .. } if reduction == &Reduction::PerEntity) + matches!(child.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } if reduction == &Reduction::PerEntity) ); } @@ -1160,15 +1235,18 @@ fn topk_over_bare_selector_by_label_ranks_per_group() { // Partition→Sort.partition_by reframe in #12). Expected: // Limit{3} → Sort{value desc, partition_by:[job]} → Scan let q = lower("topk(3, http_requests_total) by (job)"); - let QueryExpr::Limit { n, child, .. } = &q else { + let NonASAPOp::Limit { + n: Some(n), child, .. + } = q.expect_non_asap() + else { panic!("expected Limit, got {q:?}"); }; assert_eq!(*n, 3); - let QueryExpr::Sort { + let NonASAPOp::Sort { keys, partition_by, child, - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Sort, got {child:?}"); }; @@ -1177,7 +1255,7 @@ fn topk_over_bare_selector_by_label_ranks_per_group() { // No implicit reducing aggregate — the selector is label-preserving, so the // sort is directly over the selector horizon (the `job` label survives to partition by). assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })), + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })), "ranking is over the bare selector horizon, not a reducing Aggregate, got {child:?}" ); assert!( @@ -1191,20 +1269,20 @@ fn topk_over_bare_selector_ranks_raw_samples() { // Even without `by`, `topk(3, m)` ranks the raw instant-vector samples — it // does not sum them. The sort sits directly over the Scan, partition empty. let q = lower("topk(3, http_requests_total)"); - let QueryExpr::Limit { child, .. } = &q else { + let NonASAPOp::Limit { child, .. } = q.expect_non_asap() else { panic!("expected Limit, got {q:?}"); }; - let QueryExpr::Sort { + let NonASAPOp::Sort { partition_by, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Sort, got {child:?}"); }; assert!(partition_by.is_empty(), "no `by` → global ranking"); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); assert!(!has_intent(&q, |i| matches!(i, AggIntent::Sum { .. }))); } @@ -1212,20 +1290,20 @@ fn topk_over_bare_selector_ranks_raw_samples() { // ── Issue #109: histogram_quantiles fans out into one branch per φ ────────── /// The `(label value, intent)` of each `histogram_quantiles` branch. -fn quantile_branches(q: &QueryExpr) -> Vec<(String, AggIntent)> { - let QueryExpr::Concat { children, .. } = q else { +fn quantile_branches(q: &OperatorNode) -> Vec<(String, AggIntent)> { + let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { panic!("expected a Concat at the root, got {q:?}"); }; children .iter() .map(|c| { - let QueryExpr::PromqlRelabel { value, child, .. } = c else { + let NonASAPOp::PromqlRelabel { value, child, .. } = c.expect_non_asap() else { panic!("expected PromqlRelabel per branch, got {c:?}"); }; - let QueryExpr::Literal(ScalarValue::Utf8(v)) = value.as_ref() else { + let ScalarExpr::Literal(ScalarValue::Utf8(v)) = value else { panic!("expected a literal label value, got {value:?}"); }; - let QueryExpr::Aggregate { measures, .. } = child.as_ref() else { + let NonASAPOp::Aggregate { measures, .. } = child.expect_non_asap() else { panic!("expected an Aggregate under the PromqlRelabel, got {child:?}"); }; (v.clone(), measures[0].clone()) @@ -1234,24 +1312,12 @@ fn quantile_branches(q: &QueryExpr) -> Vec<(String, AggIntent)> { } #[test] -fn histogram_quantiles_fans_out_over_native_histograms() { - // Raw / native-histogram argument → the sketch-able `Quantile` intent, - // exactly as the single-quantile `histogram_quantile` would choose. - let q = lower(r#"histogram_quantiles(testhistogram3, "q", 0, 0.25, 1)"#); - let branches = quantile_branches(&q); - assert_eq!(branches.len(), 3); - let labels: Vec<_> = branches.iter().map(|(l, _)| l.as_str()).collect(); - assert_eq!( - labels, - ["0.0", "0.25", "1.0"], - "OpenMetrics float formatting" - ); - for (_, intent) in &branches { - assert!( - matches!(intent, AggIntent::Quantile { .. }), - "native histogram → sketch-able Quantile, got {intent:?}" - ); - } +fn histogram_quantiles_rejects_unrepresented_native_histograms() { + assert!(lower_promql( + r#"histogram_quantiles(testhistogram3, "q", 0, 0.25, 1)"#, + AccuracyTarget::Exact + ) + .is_err()); } #[test] @@ -1270,16 +1336,16 @@ fn histogram_quantiles_over_classic_buckets_interpolates() { fn histogram_quantiles_branches_are_union_compatible() { // `Concat` derives its schema from the first child, so every branch must // agree on column names — the φ lives in the label, not the column name. - let q = lower(r#"histogram_quantiles(testhistogram3, "q", 0.5, 0.9)"#); - let QueryExpr::Concat { children, .. } = &q else { + let q = lower(r#"histogram_quantiles(testhistogram3_bucket, "q", 0.5, 0.9)"#); + let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { panic!("expected Concat"); }; let shapes: Vec> = children .iter() .map(|c| { - c.output_schema() - .expect("branch schema") - .columns + c.schema + .clone() + .fields .iter() .map(|c| c.name.clone()) .collect() @@ -1288,7 +1354,7 @@ fn histogram_quantiles_branches_are_union_compatible() { assert_eq!(shapes[0], shapes[1], "branches must be union-compatible"); assert_eq!(shapes[0], vec!["value".to_string(), "q".to_string()]); assert_eq!( - q.output_schema().expect("merged schema").columns.len(), + q.schema.fields.len(), 2, "the merged schema describes every branch" ); @@ -1296,11 +1362,11 @@ fn histogram_quantiles_branches_are_union_compatible() { #[test] fn histogram_quantiles_uses_the_given_label_name() { - let q = lower(r#"histogram_quantiles(h, "phi", 0.5)"#); - let QueryExpr::Concat { children, .. } = &q else { + let q = lower(r#"histogram_quantiles(h_bucket, "phi", 0.5)"#); + let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { panic!("expected Concat"); }; - let QueryExpr::PromqlRelabel { dst, .. } = &children[0] else { + let NonASAPOp::PromqlRelabel { dst, .. } = children[0].expect_non_asap() else { panic!("expected PromqlRelabel"); }; assert_eq!(dst, "phi"); @@ -1309,7 +1375,7 @@ fn histogram_quantiles_uses_the_given_label_name() { #[test] fn histogram_quantiles_formats_small_quantiles_like_prometheus() { // `labels.FormatOpenMetricsFloat`: Go's %g, so exponent form below 1e-4. - let q = lower(r#"histogram_quantiles(h, "q", 0.00001)"#); + let q = lower(r#"histogram_quantiles(h_bucket, "q", 0.00001)"#); assert_eq!(quantile_branches(&q)[0].0, "1e-05"); } @@ -1317,9 +1383,9 @@ fn histogram_quantiles_formats_small_quantiles_like_prometheus() { fn histogram_quantiles_rejects_an_out_of_range_quantile() { // Same rule as `histogram_quantile(φ, …)` — one bad φ fails the whole call. for q in [ - r#"histogram_quantiles(h, "q", -0.1)"#, - r#"histogram_quantiles(h, "q", 1.01)"#, - r#"histogram_quantiles(h, "q", 0.5, NaN)"#, + r#"histogram_quantiles(h_bucket, "q", -0.1)"#, + r#"histogram_quantiles(h_bucket, "q", 1.01)"#, + r#"histogram_quantiles(h_bucket, "q", 0.5, NaN)"#, ] { assert!( lower_promql(q, AccuracyTarget::Exact).is_err(), @@ -1328,19 +1394,221 @@ fn histogram_quantiles_rejects_an_out_of_range_quantile() { } } -// A subquery's `offset`/`@` shift the whole subquery, so the tree keeps them. +// ── TimeRange.kind: instant vs range selectors ────────────────────────────────── + #[test] -fn subquery_time_shift_is_retained() { - let QueryExpr::Aggregate { child, .. } = lower("max_over_time(m[5m:1m] offset 1m)") else { - panic!("expected a range function"); +fn bare_instant_selector_is_an_instant_time_range() { + // `up` reads the latest sample per series within the workload's ingestion + // interval (1s in `support::workload`): an `Instant` lookback of that length. + let qe = lower("up"); + let NonASAPOp::TimeRange { range, kind, child } = qe.expect_non_asap() else { + panic!("expected TimeRange, got {qe:?}"); }; - let QueryExpr::TimeShift { shift, child } = child.as_ref() else { - panic!("subquery offset was dropped: {child:?}"); + assert_eq!(*kind, TimeRangeKind::Instant); + assert_eq!(*range, Duration::from_secs(1)); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); +} + +#[test] +fn explicit_range_selector_is_a_range_time_range() { + // `m[5m]` keeps its own window and is a `Range` selection — both under a + // range function and as a bare matrix selector. + let qe = lower("rate(m[5m])"); + let NonASAPOp::Aggregate { child, .. } = qe.expect_non_asap() else { + panic!("expected Aggregate, got {qe:?}"); }; - assert_eq!(shift.offset_ms, 60_000); - assert!(matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. })); + let NonASAPOp::TimeRange { range, kind, .. } = child.expect_non_asap() else { + panic!("expected TimeRange, got {child:?}"); + }; + assert_eq!(*kind, TimeRangeKind::Range); + assert_eq!(*range, Duration::from_secs(300)); + + let qe = lower("m[5m]"); + assert!(matches!( + qe.expect_non_asap(), + NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, + .. + } + )); +} + +#[test] +fn instant_and_range_selectors_of_equal_length_stay_distinct() { + // The kind is part of the shape: a 1s range selector is not the same tree as + // the 1s instant lookback injected around a bare selector. + assert_ne!(lower("up"), lower("up[1s]")); +} + +// ── the `bool` modifier → `return_bool` ───────────────────────────────────────── + +#[test] +fn vector_scalar_comparison_without_bool_filters() { + let qe = lower("up > 0"); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Filter { .. })); + assert!(matches!( + support::sample_expression(&qe), + ScalarExpr::Compare { + op: CompareOpKind::Gt, + .. + } + )); +} + +#[test] +fn vector_scalar_comparison_with_bool_sets_return_bool() { + let qe = lower("up > bool 0"); assert!(matches!( - lower("max_over_time(m[5m:1m] @ 100)"), - QueryExpr::Aggregate { child, .. } if matches!(child.as_ref(), QueryExpr::TimeShift { .. }) + support::sample_expression(&qe), + ScalarExpr::Case { .. } + )); + assert_ne!(qe, lower("up > 0")); +} + +#[test] +fn vector_vector_comparison_with_bool_sets_return_bool() { + // `a > bool b` — the modifier lands on the vector/vector op itself, with + // the default (ignoring nothing) match. + let qe = lower("a > bool b"); + let NonASAPOp::BinaryOp { + operator, + return_bool, + lhs, + rhs, + } = qe.expect_non_asap() + else { + panic!("expected BinaryOp, got {qe:?}"); + }; + assert!(*return_bool); + assert_eq!(operator.kind, BinaryOpKind::Compare(CompareOpKind::Gt)); + assert!(matches!(lhs.expect_non_asap(), NonASAPOp::TimeRange { .. })); + assert!(matches!(rhs.expect_non_asap(), NonASAPOp::TimeRange { .. })); + assert!(!lower("a > b").expect_non_asap().children().is_empty()); + assert_ne!(qe, lower("a > b")); +} + +#[test] +fn bool_modifier_composes_with_vector_matching() { + let qe = lower("a > bool on(job) b"); + let NonASAPOp::BinaryOp { + operator, + return_bool, + .. + } = qe.expect_non_asap() + else { + panic!("expected BinaryOp, got {qe:?}"); + }; + assert!(*return_bool); + let vm = operator.vector_match.as_ref().expect("on(job) present"); + assert_eq!(vm.labels, vec!["job".to_string()]); +} + +// ── scalar expressions: negation, arithmetic, comparison ──────────────────────── + +#[test] +fn scalar_negation_of_time_is_a_negative_expression() { + // `-time()` is a scalar expression; its negation stays structural (the + // operand is not a constant to fold) and follows PromQL numeric rules. + let qe = support::scalar_root("-time()"); + let ScalarExpr::Negative { expr, semantics } = &qe else { + panic!("expected ScalarExpr(Negative), got {qe:?}"); + }; + assert_eq!(*semantics, ExprSemantics::Promql); + assert!(matches!(expr.as_ref(), ScalarExpr::EvalTimestamp)); + // Scalar-shaped: no time index. +} + +#[test] +fn scalar_negation_of_a_constant_still_folds() { + // `-(2)` is constant: it folds to one literal rather than a `Negative`. + assert_eq!( + support::promql_scalar(&support::scalar_root("-(2)")), + Some(-2.0) + ); +} + +#[test] +fn scalar_arithmetic_carries_promql_semantics() { + let qe = support::scalar_root("time() - 1"); + let ScalarExpr::Arithmetic { + op, + left, + right, + semantics, + } = &qe + else { + panic!("expected scalar(Arithmetic), got {qe:?}"); + }; + assert_eq!(*op, ArithmeticOpKind::Sub); + assert_eq!(*semantics, ExprSemantics::Promql); + assert!(matches!(left.as_ref(), ScalarExpr::EvalTimestamp)); + assert!(matches!( + right.as_ref(), + ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0 + )); +} + +#[test] +fn scalar_bool_comparison_is_a_zero_one_case_with_promql_semantics() { + // `1 < bool 2` → `Case(Compare(1 < 2) → 1.0, else 0.0)`: PromQL yields 0/1. + let qe = support::scalar_root("1 < bool 2"); + let ScalarExpr::Case { + operand, + branches, + else_expr, + } = &qe + else { + panic!("expected scalar(Case), got {qe:?}"); + }; + assert!(operand.is_none()); + let [(when, then)] = branches.as_slice() else { + panic!("expected one branch, got {branches:?}"); + }; + let ScalarExpr::Compare { + left, + op, + right, + semantics, + } = when + else { + panic!("expected a Compare condition, got {when:?}"); + }; + assert_eq!(*op, CompareOpKind::Lt); + assert_eq!(*semantics, ExprSemantics::Promql); + assert!(matches!(left.as_ref(), ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); + assert!(matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 2.0)); + assert!(matches!(then, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); + assert!(matches!( + else_expr.as_deref(), + Some(ScalarExpr::Literal(ScalarValue::Float64(v))) if *v == 0.0 + )); +} + +#[test] +fn scalar_comparison_without_bool_is_rejected() { + // PromQL has no scalar filter: a scalar/scalar comparison needs `bool`. + for q in ["1 < 2", "time() > 0", "(1 + 1) == 2"] { + assert!( + lower_promql(q, AccuracyTarget::Exact).is_err(), + "{q} must be rejected without `bool`" + ); + } +} + +#[test] +fn label_matcher_predicates_carry_promql_semantics() { + let qe = lower(r#"up{job="api"}"#); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { + panic!("expected TimeRange, got {qe:?}"); + }; + let NonASAPOp::Scan { predicates, .. } = child.expect_non_asap() else { + panic!("expected Scan, got {child:?}"); + }; + assert!(matches!( + &predicates[0].0, + ScalarExpr::Compare { + semantics: ExprSemantics::Promql, + .. + } )); } diff --git a/crates/frontend-promql/tests/scalar_design.rs b/crates/frontend-promql/tests/scalar_design.rs new file mode 100644 index 000000000..e4f8038f8 --- /dev/null +++ b/crates/frontend-promql/tests/scalar_design.rs @@ -0,0 +1,128 @@ +//! Scalar expressions never become constant-wrapper operators. +mod support; +use asap_types::ir::{NonASAPOp, QueryRoot, ScalarExpr}; +use asap_types::pre_asap::{ArithmeticOpKind, ScalarValue}; +use asap_types::types::AccuracyTarget; + +fn root(query: &str) -> QueryRoot { + asap_frontend_promql::lower_promql_query_workload( + &support::workload(query, AccuracyTarget::Exact), + 0, + ) + .unwrap() + .remove(0) +} + +#[test] +fn standalone_scalars_are_expressions() { + for query in [ + "2", + "time()", + "scalar(sum(up)) + 1", + "1 < bool 2", + "-time()", + ] { + let QueryRoot::Scalar(expr) = root(query) else { + panic!("{query} became an operator") + }; + expr.scalar_type(&Default::default()).unwrap(); + } +} + +#[test] +fn arithmetic_projects_the_sample_and_preserves_full_identity_and_time() { + let QueryRoot::Operator(node) = root("up * 2") else { + panic!() + }; + let NonASAPOp::Project { child, cols, .. } = node.expect_non_asap() else { + panic!() + }; + assert!(child.schema.has_promql_series_identity()); + assert!(node.schema.has_promql_series_identity()); + assert_eq!(node.schema.time_index, child.schema.time_index); + let value = node.schema.column_id("value").unwrap(); + assert!( + matches!(&cols[value].expr, ScalarExpr::Arithmetic { op: ArithmeticOpKind::Mul, right, .. } if **right == ScalarExpr::Literal(ScalarValue::Float64(2.0))) + ); + assert!(cols.iter().any(|c| matches!(&c.expr, ScalarExpr::FunctionCall { name, .. } if name == "promql_drop_metric_name"))); +} + +#[test] +fn non_bool_comparisons_keep_vector_samples_even_with_scalar_on_left() { + for query in ["up > 0", "0 < up"] { + let QueryRoot::Operator(node) = root(query) else { + panic!() + }; + let NonASAPOp::Filter { child, .. } = node.expect_non_asap() else { + panic!() + }; + assert_eq!(node.schema, child.schema); + } +} + +#[test] +fn bool_comparison_projects_zero_or_one() { + let QueryRoot::Operator(node) = root("up > bool 0") else { + panic!() + }; + let NonASAPOp::Project { cols, .. } = node.expect_non_asap() else { + panic!() + }; + assert!(matches!( + cols[node.schema.column_id("value").unwrap()].expr, + ScalarExpr::Case { .. } + )); +} + +#[test] +fn scalar_plan_dependencies_remain_visible() { + let QueryRoot::Operator(node) = root("up * scalar(sum(up))") else { + panic!() + }; + assert_eq!(node.children().len(), 2); +} + +/// Pointwise functions own scalar parameters, including vector-to-scalar reads. +#[test] +fn pointwise_functions_are_typed_scalar_projections() { + for query in [ + "abs(up)", + "round(up, scalar(sum(other)))", + "clamp(up, time() - 1, time())", + "year(up)", + "hour()", + ] { + let QueryRoot::Operator(node) = root(query) else { + panic!() + }; + let NonASAPOp::Project { cols, .. } = node.expect_non_asap() else { + panic!("{query}: expected projection") + }; + assert!(matches!( + &cols[node.schema.column_id("value").unwrap()].expr, + ScalarExpr::FunctionCall { .. } + )); + node.validate_structure().unwrap(); + } +} + +/// Negation preserves the metric name and complete identity unlike multiplication. +#[test] +fn unary_minus_preserves_identity() { + let QueryRoot::Operator(node) = root("-up") else { + panic!() + }; + let NonASAPOp::Project { child, cols, .. } = node.expect_non_asap() else { + panic!() + }; + assert_eq!(node.schema, child.schema); + assert!(matches!( + &cols[node.schema.column_id("value").unwrap()].expr, + ScalarExpr::Negative { .. } + )); + for (index, col) in cols.iter().enumerate() { + if index != node.schema.column_id("value").unwrap() { + assert_eq!(col.expr, ScalarExpr::Column(index)); + } + } +} diff --git a/crates/frontend-promql/tests/support.rs b/crates/frontend-promql/tests/support.rs index 1f15b1ca2..aa7b5a144 100644 --- a/crates/frontend-promql/tests/support.rs +++ b/crates/frontend-promql/tests/support.rs @@ -1,14 +1,17 @@ +use std::rc::Rc; + use asap_frontend_promql::{ lower_promql_workload, lower_promql_workload_with_histograms, HistogramCatalog, PromqlError, }; -use asap_types::pre_asap::QueryExpr; +use asap_types::ir::{NonASAPOp, OperatorNode, ScalarExpr}; +use asap_types::pre_asap::ScalarValue; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, }; -fn workload(query: &str, accuracy: AccuracyTarget) -> PlanningWorkload { +pub fn workload(query: &str, accuracy: AccuracyTarget) -> PlanningWorkload { PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -35,7 +38,11 @@ fn workload(query: &str, accuracy: AccuracyTarget) -> PlanningWorkload { } } -pub fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Result { +#[allow(dead_code)] +pub fn lower_promql( + query: &str, + accuracy: AccuracyTarget, +) -> Result, PromqlError> { let mut lowered = lower_promql_workload(&workload(query, accuracy), 0)?; Ok(lowered.remove(0)) } @@ -45,8 +52,59 @@ pub fn lower_promql_with_histograms( query: &str, accuracy: AccuracyTarget, histograms: HistogramCatalog, -) -> Result { +) -> Result, PromqlError> { let mut lowered = lower_promql_workload_with_histograms(&workload(query, accuracy), histograms, 0)?; Ok(lowered.remove(0)) } + +/// The value of a bare PromQL numeric literal / folded constant at an +/// scalar position (`Literal(Float64(v))`); `None` for any +/// other shape. +#[allow(dead_code)] +pub fn promql_scalar(node: &ScalarExpr) -> Option { + match node { + ScalarExpr::Literal(ScalarValue::Float64(v)) => Some(*v), + _ => None, + } +} + +/// Time `root` under the default (every summary maintained) lifecycle +/// assignment and export the post-ASAP DAG — the wire-6 export needs every +/// node timed first. +#[allow(dead_code)] +pub fn post_asap_dag(root: &Rc) -> asap_types::ir::export::PostAsapDag { + use asap_types::ir::{apply_lifecycle_timings, LifecycleAssignment, TimingMemo}; + let timed = apply_lifecycle_timings( + root, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .expect("default lifecycle timings"); + asap_types::ir::export::compile_post_asap_dag(&timed).expect("post-ASAP DAG export") +} + +#[allow(dead_code)] +pub fn scalar_root(query: &str) -> ScalarExpr { + match asap_frontend_promql::lower_promql_query_workload( + &workload(query, AccuracyTarget::Exact), + 0, + ) + .unwrap() + .remove(0) + { + asap_types::ir::QueryRoot::Scalar(expr) => expr, + _ => panic!("expected scalar root: {query}"), + } +} + +#[allow(dead_code)] +pub fn sample_expression(node: &OperatorNode) -> &ScalarExpr { + match node.expect_non_asap() { + NonASAPOp::Project { cols, .. } => { + &cols[node.schema.column_id("value").unwrap_or(cols.len() - 1)].expr + } + NonASAPOp::Filter { pred, .. } => &pred.0, + other => panic!("expected sample expression, got {other:?}"), + } +} diff --git a/crates/frontend-promql/tests/univmon_candidates.rs b/crates/frontend-promql/tests/univmon_candidates.rs index d34768e8f..47c8312c9 100644 --- a/crates/frontend-promql/tests/univmon_candidates.rs +++ b/crates/frontend-promql/tests/univmon_candidates.rs @@ -5,26 +5,27 @@ use asap_aware_mapping::accuracy::{ }; use asap_aware_mapping::cost_model::DefaultCostModel; use asap_aware_mapping::replacement::{default_strategies, search_workload_with_targets}; -use asap_aware_mapping::{Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG}; +use asap_aware_mapping::{ASAPStrategies, Replacement, ReplacementStrategy, TargetSubDAG}; mod support; +use asap_types::ir::cse::share_common_subdags; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; use asap_types::post_asap::{ - compile_post_asap_dag, cse::share_common_summary_subtrees, AccuracyError, BoundExpr, - CompositionOperator, ErrorMetric, ProbabilityExpr, ResultGuarantee, SketchAlgorithm, - SketchQuery, SummaryExpr, SummaryFamilyType, SummaryInputExpr, SummaryNode, + AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, FieldDataType, ProbabilityExpr, + ResultGuarantee, SketchAlgorithm, SketchStatistic, SummaryInputExpr, }; use asap_types::types::AccuracyTarget; -use support::lower_promql; +use support::{lower_promql, post_asap_dag}; // Synthetic evidence exercises structural sharing, never runtime accuracy. struct TestEvidence; impl AccuracyModel for TestEvidence { fn local_guarantee( &self, - family: &SummaryFamilyType, - query: &SketchQuery, + family: &FieldDataType, + query: &SketchStatistic, ) -> Option { - if matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) - && !matches!(query, SketchQuery::PointCount { .. }) + if matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) + && !matches!(query, SketchStatistic::PointCount { .. }) { let mut guarantee = ResultGuarantee::exact("SYNTHETIC test evidence; not measured"); guarantee.metric = ErrorMetric::RelativeValue; @@ -49,22 +50,22 @@ impl AccuracyModel for TestEvidence { } } -fn candidate(query: &str, accuracy: AccuracyTarget) -> Rc { +fn candidate(query: &str, accuracy: AccuracyTarget) -> Rc { let root = lower_promql(query, accuracy).unwrap(); - SketchAlgorithmStrategy::new_with_planning_inputs(&DefaultCostModel, &TestEvidence, &EqualSplitAllocator) - .replacements(&TargetSubDAG::new(&Rc::new(root))) + ASAPStrategies::new_with_planning_inputs(&DefaultCostModel, &TestEvidence, &EqualSplitAllocator) + .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| { - let Replacement::Summary(node) = candidate.replacement else { return None }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { return None }; - matches!(&summary_input.expr, SummaryExpr::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } + let Replacement::SubDag(node) = candidate.replacement else { return None }; + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator else { return None }; + matches!(&summary_input.operator, Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) if kind.algorithm() == &SketchAlgorithm::UnivMon).then_some(node) }).expect("UnivMon candidate") } #[test] -fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { - // Equal data, grouping and window produce one state independently of readout. +fn four_evaluations_share_one_value_frequency_state_and_keep_honest_guarantees() { + // Equal data, grouping and window produce one state independently of evaluation. let accuracy = AccuracyTarget::Epsilon(0.02); let roots: Vec<_> = [ ("distinct_over_time(m[5m])", accuracy.clone()), @@ -76,36 +77,38 @@ fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { .enumerate() .map(|(id, (query, accuracy))| (id, candidate(query, accuracy))) .collect(); - let roots = share_common_summary_subtrees(roots); + let roots = share_common_subdags(roots); let mut first_state = None; for (index, root) in &roots { - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - .. - } = &root.expr + }) = &root.operator else { panic!() }; if let Some(first) = &first_state { assert!( Rc::ptr_eq(first, summary_input), - "state must be shared across readouts" + "state must be shared across evaluations" ); } else { first_state = Some(Rc::clone(summary_input)); } - let SummaryExpr::SummaryAgg { input, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) = &summary_input.operator else { panic!() }; assert!(matches!(input.item, Some(SummaryInputExpr::Column(_)))); assert_eq!(input.weight, SummaryInputExpr::Constant(1.0)); if *index == 1 { - assert!(matches!(query, SketchQuery::PointCount { value: None, .. })); + assert!(matches!( + query, + SketchStatistic::PointCount { value: None, .. } + )); assert!(root.guarantee.as_ref().is_some_and(|g| g.is_exact())); } else { assert!(!root.guarantee.as_ref().unwrap().is_exact()); - let SummaryExpr::SummaryAgg { family, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &summary_input.operator else { panic!() }; assert!( @@ -115,12 +118,12 @@ fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { "production has no calibrated error bound" ); } - compile_post_asap_dag(root).unwrap(); + post_asap_dag(root); } } #[test] -fn uncalibrated_frequency_readouts_do_not_bypass_accuracy_targets() { +fn uncalibrated_frequency_evaluations_do_not_bypass_accuracy_targets() { // An unmeasured heuristic remains inspectable but is never certified or // automatically selected for a caller-visible bounded-error result. for query in ["entropy_over_time(m[5m])", "l2_over_time(m[5m])"] { @@ -132,16 +135,16 @@ fn uncalibrated_frequency_readouts_do_not_bypass_accuracy_targets() { delta: 0.01, }, ] { - let root = Rc::new(lower_promql(query, target.clone()).unwrap()); - let candidates = SketchAlgorithmStrategy::default_cost_model() - .replacements(&TargetSubDAG::new(&root)); + let root = lower_promql(query, target.clone()).unwrap(); + let candidates = + ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); let unknown = candidates .iter() .filter(|candidate| { matches!( &candidate.replacement, - Replacement::Summary(node) - if matches!(&node.expr, SummaryExpr::SummaryEstimate { .. }) + Replacement::SubDag(node) + if matches!(&node.operator, Operator::ASAP(ASAPOp::SummaryEstimate { .. })) && node.guarantee.is_none() && candidate.has_missing_accuracy_evidence() ) diff --git a/crates/frontend-sql/Cargo.toml b/crates/frontend-sql/Cargo.toml index ac6bd2f51..a68418302 100644 --- a/crates/frontend-sql/Cargo.toml +++ b/crates/frontend-sql/Cargo.toml @@ -9,6 +9,7 @@ edition = "2021" # #225) it consults when lowering an aggregate call — never promql-parser. [dependencies] asap-types = { path = "../types" } +asap-frontend-common = { path = "../frontend-common" } asap-sql-function-catalog = { path = "../sql-function-catalog" } datafusion = "43" # `AggIntent::Extension.payload` for ClickHouse's argMax/argMin (issue #232) diff --git a/crates/frontend-sql/src/error.rs b/crates/frontend-sql/src/error.rs index 5342f5a3f..1b934e6a2 100644 --- a/crates/frontend-sql/src/error.rs +++ b/crates/frontend-sql/src/error.rs @@ -1,11 +1,11 @@ use std::fmt; -use asap_types::pre_asap::ResolveTreeError; +use asap_frontend_common::ResolveTreeError; /// Errors from lowering a SQL query (parse + plan via DataFusion → the -/// canonical, unresolved tree, built directly → -/// [`resolve_root`](asap_types::pre_asap::resolve_root) binds it to the -/// resolved tree, issue #179). +/// name-based [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree → +/// [`resolve_root`](asap_frontend_common::resolve_root) binds it into the +/// unified IR). /// /// Carries no PromQL type — the SQL front end never depends on the PromQL /// parser. The language-neutral variants (`UnsupportedFeature` / `WrongLanguage` @@ -28,8 +28,8 @@ pub enum SqlError { UnsupportedFeature(String), /// The workload's query language is not SQL. WrongLanguage(String), - /// Resolving the canonical unresolved tree failed (name resolution - /// against the bound schema). + /// Resolving the name-based tree failed (name resolution against the + /// bound schema, or schema derivation). Convert(ResolveTreeError), } diff --git a/crates/frontend-sql/src/lib.rs b/crates/frontend-sql/src/lib.rs index 58474e2eb..1ea8eb2b5 100644 --- a/crates/frontend-sql/src/lib.rs +++ b/crates/frontend-sql/src/lib.rs @@ -1,25 +1,27 @@ -//! SQL front end: parse + plan (via DataFusion) → the canonical, unresolved -//! shape, built directly (issue #179) → [`resolve_root`]. +//! SQL front end: parse + plan (via DataFusion) → the name-based +//! [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree, built directly +//! (issue #179) → [`resolve_root`]. //! -//! Emits [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) itself — the -//! canonical `QueryExpr`, generic over an unresolved -//! [`ColumnRef`](asap_types::pre_asap::ColumnRef) — directly, rather than a -//! separate per-language relational tree; `resolve_root` runs the -//! [`SchemaResolver`](asap_types::pre_asap::SchemaResolver) for positional name resolution. +//! Emits the shared front-end tree (`UnresolvedOp` / `UnresolvedScalar`, +//! name-based [`ColumnRef`](asap_types::pre_asap::ColumnRef)s) directly, rather +//! than a separate per-language relational tree; `resolve_root` binds it into +//! the unified [`OperatorNode`] IR, deriving every schema on the way. //! Depends on DataFusion only — never on the PromQL parser. pub mod error; pub mod sql; -use asap_types::pre_asap::resolve_root; -use asap_types::pre_asap::QueryExpr; +use std::rc::Rc; + +use asap_frontend_common::resolve_root; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use asap_types::workload::{QueryLanguage, QueryWorkload, SqlDialect}; pub use error::SqlError; pub use sql::{SqlCatalog, SqlLowerer}; -/// Lower a single SQL query string to the canonical, resolved `QueryExpr`, +/// Lower a single SQL query string to the resolved, canonical operator DAG, /// parsed as `SqlDialect::DataFusionSQL`. /// /// The `catalog` supplies table schemas (used both to plan the SQL with @@ -29,7 +31,7 @@ pub async fn lower_sql( query: &str, catalog: &SqlCatalog, accuracy: AccuracyTarget, -) -> Result { +) -> Result, SqlError> { lower_sql_dialect(query, catalog, SqlDialect::DataFusionSQL, accuracy).await } @@ -46,20 +48,17 @@ pub async fn lower_sql_dialect( catalog: &SqlCatalog, dialect: SqlDialect, accuracy: AccuracyTarget, -) -> Result { +) -> Result, SqlError> { let unresolved = SqlLowerer::with_dialect(catalog, dialect) .lower(query, &accuracy) .await?; - let resolved = resolve_root(&unresolved)?; - // Binding resolves names; schema inference also checks result types such - // as temporal subtraction, whose duration unit the IR cannot represent. - resolved - .output_schema() - .map_err(|error| SqlError::InvalidExpression(error.to_string()))?; - Ok(resolved) + // Binding resolves names and derives every node's schema; result-type + // checks (such as temporal subtraction, whose duration unit the IR cannot + // represent) surface here as `ResolveTreeError::Schema`. + Ok(resolve_root(&unresolved)?) } -/// Lower every SQL batch entry in `workload` to a `QueryExpr`. +/// Lower every SQL batch entry in `workload` to an operator DAG. /// /// One `Result` per entry — errors are per-query, not fatal for the batch. /// Returns `WrongLanguage` for every entry if the workload is not SQL, and @@ -67,7 +66,7 @@ pub async fn lower_sql_dialect( pub async fn lower_sql_batch( workload: &QueryWorkload, catalog: &SqlCatalog, -) -> Vec> { +) -> Vec, SqlError>> { let entries = match &workload.query_batch { Some(e) if !e.is_empty() => e, _ => return vec![], diff --git a/crates/frontend-sql/src/sql/collection_planning.rs b/crates/frontend-sql/src/sql/collection_planning.rs index 92fda701d..365209ab6 100644 --- a/crates/frontend-sql/src/sql/collection_planning.rs +++ b/crates/frontend-sql/src/sql/collection_planning.rs @@ -1,10 +1,10 @@ //! DataFusion planning adapters. Types come from the canonical signature rules; //! physical evaluation deliberately remains the query engine's responsibility. use super::types::{arrow_to_dtype, dtype_to_arrow, scalar_value_to_asap}; -use asap_types::pre_asap::scalar_signature::{ - element_access_type, struct_field_type, MapScalarFunction, -}; -use asap_types::pre_asap::{Column, QueryExpr, Schema}; +use asap_types::ir::scalar::{element_access_type, struct_field_type}; +use asap_types::ir::ScalarExpr; +use asap_types::pre_asap::scalar_signature::MapScalarFunction; +use asap_types::pre_asap::{Field, Schema}; use datafusion::arrow::datatypes::DataType; use datafusion::common::{DataFusionError, ExprSchema, Result}; use datafusion::logical_expr::{ @@ -82,19 +82,19 @@ impl CollectionPlanningFunction { .into_iter() .enumerate() .map(|(index, (dtype, nullable))| { - Column::new(format!("argument_{index}"), dtype, nullable) + Field::plain(format!("argument_{index}"), dtype, nullable) }) .collect(), ); - let args = (0..schema.columns.len()) + let args = (0..schema.fields.len()) .map(|index| { if let Some(Expr::Literal(value)) = expressions.and_then(|args| args.get(index)) { scalar_value_to_asap(value) - .map(QueryExpr::Literal) + .map(ScalarExpr::Literal) .map_err(|error| DataFusionError::Plan(error.to_string())) } else { - Ok(QueryExpr::Column(index)) + Ok(ScalarExpr::Column(index)) } }) .collect::>>()?; diff --git a/crates/frontend-sql/src/sql/expr.rs b/crates/frontend-sql/src/sql/expr.rs index 27ca85beb..dcd53d425 100644 --- a/crates/frontend-sql/src/sql/expr.rs +++ b/crates/frontend-sql/src/sql/expr.rs @@ -2,12 +2,14 @@ use std::rc::Rc; use datafusion::logical_expr::{BinaryExpr, Expr, Operator}; +use asap_frontend_common::UnresolvedScalar as Unresolved; +use asap_types::ir::ExprSemantics; use asap_types::pre_asap::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; use crate::error::SqlError as LoweringError; use super::types::{arrow_to_dtype, scalar_value_to_asap}; -use super::Unresolved; +use super::SqlLowerer; pub(super) fn split_conjuncts(expr: &Expr) -> Vec<&Expr> { match expr { @@ -24,255 +26,262 @@ pub(super) fn split_conjuncts(expr: &Expr) -> Vec<&Expr> { } } -/// Translate a DataFusion `Expr` to the canonical, unresolved tree. -/// Returns `UnsupportedFeature` for anything not needed in v1. -pub(super) fn df_expr_to_unresolved(expr: &Expr) -> Result { - match expr { - // Preserve DataFusion's relation qualifier so a column name shared - // across a join (`a.k` vs `b.k`) resolves to the correct side. - Expr::Column(col) => Ok(Unresolved::Column(match &col.relation { - Some(rel) => ColumnRef::Qualified { - table: rel.to_string(), - name: col.name.clone(), - }, - None => ColumnRef::Named(col.name.clone()), - })), - - // Keep Arrow date literals equivalent to SQL CAST('YYYY-MM-DD' AS DATE), - // including typed nulls, without adding another canonical scalar variant. - Expr::Literal( - sv @ (datafusion::common::ScalarValue::Date32(_) - | datafusion::common::ScalarValue::Date64(_)), - ) => { - let text = sv.cast_to(&datafusion::arrow::datatypes::DataType::Utf8)?; - // Arrow formats Date64 with a time suffix; the canonical Date has - // no time-of-day, just like Date64 catalog registration as Date32. - let text = match text { - datafusion::common::ScalarValue::Utf8(Some(value)) => { - ScalarValue::Utf8(value.split('T').next().unwrap().to_owned()) - } - other => scalar_value_to_asap(&other)?, - }; - Ok(Unresolved::Cast { - expr: Rc::new(Unresolved::Literal(text)), - to: asap_types::pre_asap::schema::DataType::Date, - try_cast: false, - }) - } - Expr::Literal(sv) => scalar_value_to_asap(sv).map(Unresolved::Literal), - - Expr::Alias(a) => df_expr_to_unresolved(&a.expr), +impl SqlLowerer<'_> { + /// Translate a DataFusion `Expr` to the name-based scalar tree. Every + /// `Compare` / `Arithmetic` / `Negative` carries `ExprSemantics::Sql`. + /// Subquery-valued expressions lower their plan as a root of its own + /// (which is why this is a method: the plan walk needs the catalog). + /// Returns `UnsupportedFeature` for anything not needed in v1. + pub(super) fn lower_expr(&self, expr: &Expr) -> Result { + let bx = |e: &Expr| self.lower_expr(e).map(Box::new); + match expr { + // Preserve DataFusion's relation qualifier so a column name shared + // across a join (`a.k` vs `b.k`) resolves to the correct side. + Expr::Column(col) => Ok(Unresolved::Column(match &col.relation { + Some(rel) => ColumnRef::Qualified { + table: rel.to_string(), + name: col.name.clone(), + }, + None => ColumnRef::Named(col.name.clone()), + })), - Expr::BinaryExpr(BinaryExpr { left, op, right }) => match op { - Operator::And => { - let parts = split_conjuncts(expr); - let lowered: Result, _> = - parts.iter().map(|e| df_expr_to_unresolved(e)).collect(); - Ok(Unresolved::BoolAnd(lowered?)) - } - Operator::Or => { - let parts = split_disjuncts(expr); - let lowered: Result, _> = - parts.iter().map(|e| df_expr_to_unresolved(e)).collect(); - Ok(Unresolved::BoolOr(lowered?)) + // Keep Arrow date literals equivalent to SQL CAST('YYYY-MM-DD' AS DATE), + // including typed nulls, without adding another canonical scalar variant. + Expr::Literal( + sv @ (datafusion::common::ScalarValue::Date32(_) + | datafusion::common::ScalarValue::Date64(_)), + ) => { + let text = sv.cast_to(&datafusion::arrow::datatypes::DataType::Utf8)?; + // Arrow formats Date64 with a time suffix; the canonical Date has + // no time-of-day, just like Date64 catalog registration as Date32. + let text = match text { + datafusion::common::ScalarValue::Utf8(Some(value)) => { + ScalarValue::Utf8(value.split('T').next().unwrap().to_owned()) + } + other => scalar_value_to_asap(&other)?, + }; + Ok(Unresolved::Cast { + expr: Box::new(Unresolved::Literal(text)), + to: asap_types::pre_asap::schema::DataType::Date, + try_cast: false, + }) } - Operator::Eq => compare(left, CompareOpKind::Eq, right), - Operator::NotEq => compare(left, CompareOpKind::Ne, right), - Operator::Lt => compare(left, CompareOpKind::Lt, right), - Operator::LtEq => compare(left, CompareOpKind::Le, right), - Operator::Gt => compare(left, CompareOpKind::Gt, right), - Operator::GtEq => compare(left, CompareOpKind::Ge, right), - // BinaryExpr LIKE/ILIKE operators (from optimizer rewrites) - Operator::LikeMatch => compare(left, CompareOpKind::Like, right), - Operator::ILikeMatch => compare(left, CompareOpKind::ILike, right), - Operator::NotLikeMatch => compare(left, CompareOpKind::NotLike, right), - Operator::NotILikeMatch => compare(left, CompareOpKind::NotILike, right), - // Arithmetic - Operator::Plus => arith(left, ArithmeticOpKind::Add, right), - Operator::Minus => arith(left, ArithmeticOpKind::Sub, right), - Operator::Multiply => arith(left, ArithmeticOpKind::Mul, right), - Operator::Divide => arith(left, ArithmeticOpKind::Div, right), - Operator::Modulo => arith(left, ArithmeticOpKind::Mod, right), - other => Err(LoweringError::UnsupportedFeature(format!( - "operator: {other:?}" - ))), - }, + Expr::Literal(sv) => scalar_value_to_asap(sv).map(Unresolved::Literal), - // SQL LIKE / ILIKE (dedicated expr node from the SQL parser) - Expr::Like(like) => { - let op = match (like.negated, like.case_insensitive) { - (false, false) => CompareOpKind::Like, - (true, false) => CompareOpKind::NotLike, - (false, true) => CompareOpKind::ILike, - (true, true) => CompareOpKind::NotILike, - }; - compare(&like.expr, op, &like.pattern) - } + Expr::Alias(a) => self.lower_expr(&a.expr), - // Unary minus: negate literals directly; wrap others in -1 * x. - Expr::Negative(inner) => { - let inner = df_expr_to_unresolved(inner)?; - match inner { - Unresolved::Literal(ScalarValue::Int64(v)) => { - Ok(Unresolved::Literal(ScalarValue::Int64(-v))) + Expr::BinaryExpr(BinaryExpr { left, op, right }) => match op { + Operator::And => { + let parts = split_conjuncts(expr); + let lowered: Result, _> = + parts.iter().map(|e| self.lower_expr(e)).collect(); + Ok(Unresolved::BoolAnd(lowered?)) } - Unresolved::Literal(ScalarValue::Float64(v)) => { - Ok(Unresolved::Literal(ScalarValue::Float64(-v))) + Operator::Or => { + let parts = split_disjuncts(expr); + let lowered: Result, _> = + parts.iter().map(|e| self.lower_expr(e)).collect(); + Ok(Unresolved::BoolOr(lowered?)) } - other => Ok(Unresolved::Arithmetic { - op: ArithmeticOpKind::Mul, - left: Rc::new(Unresolved::Literal(ScalarValue::Int64(-1))), - right: Rc::new(other), - }), + Operator::Eq => self.compare(left, CompareOpKind::Eq, right), + Operator::NotEq => self.compare(left, CompareOpKind::Ne, right), + Operator::Lt => self.compare(left, CompareOpKind::Lt, right), + Operator::LtEq => self.compare(left, CompareOpKind::Le, right), + Operator::Gt => self.compare(left, CompareOpKind::Gt, right), + Operator::GtEq => self.compare(left, CompareOpKind::Ge, right), + // BinaryExpr LIKE/ILIKE operators (from optimizer rewrites) + Operator::LikeMatch => self.compare(left, CompareOpKind::Like, right), + Operator::ILikeMatch => self.compare(left, CompareOpKind::ILike, right), + Operator::NotLikeMatch => self.compare(left, CompareOpKind::NotLike, right), + Operator::NotILikeMatch => self.compare(left, CompareOpKind::NotILike, right), + // Arithmetic + Operator::Plus => self.arith(left, ArithmeticOpKind::Add, right), + Operator::Minus => self.arith(left, ArithmeticOpKind::Sub, right), + Operator::Multiply => self.arith(left, ArithmeticOpKind::Mul, right), + Operator::Divide => self.arith(left, ArithmeticOpKind::Div, right), + Operator::Modulo => self.arith(left, ArithmeticOpKind::Mod, right), + other => Err(LoweringError::UnsupportedFeature(format!( + "operator: {other:?}" + ))), + }, + + // SQL LIKE / ILIKE (dedicated expr node from the SQL parser) + Expr::Like(like) => { + let op = match (like.negated, like.case_insensitive) { + (false, false) => CompareOpKind::Like, + (true, false) => CompareOpKind::NotLike, + (false, true) => CompareOpKind::ILike, + (true, true) => CompareOpKind::NotILike, + }; + self.compare(&like.expr, op, &like.pattern) } - } - // SQL CASE expression - Expr::Case(c) => { - let operand = c - .expr - .as_ref() - .map(|e| df_expr_to_unresolved(e).map(Rc::new)) - .transpose()?; - let branches = c - .when_then_expr - .iter() - .map(|(when, then)| { - Ok((df_expr_to_unresolved(when)?, df_expr_to_unresolved(then)?)) + // Unary minus. (DataFusion's planner already folds `-` + // into a negative literal, so this is a non-literal operand.) + Expr::Negative(inner) => Ok(Unresolved::Negative { + expr: bx(inner)?, + semantics: ExprSemantics::Sql, + }), + + // SQL CASE expression + Expr::Case(c) => { + let operand = c.expr.as_deref().map(bx).transpose()?; + let branches = c + .when_then_expr + .iter() + .map(|(when, then)| Ok((self.lower_expr(when)?, self.lower_expr(then)?))) + .collect::, LoweringError>>()?; + let else_expr = c.else_expr.as_deref().map(bx).transpose()?; + Ok(Unresolved::Case { + operand, + branches, + else_expr, }) - .collect::, LoweringError>>()?; - let else_expr = c - .else_expr - .as_ref() - .map(|e| df_expr_to_unresolved(e).map(Rc::new)) - .transpose()?; - Ok(Unresolved::Case { - operand, - branches, - else_expr, - }) - } + } - Expr::Not(inner) => Ok(Unresolved::Not(Rc::new(df_expr_to_unresolved(inner)?))), + Expr::Not(inner) => Ok(Unresolved::Not(bx(inner)?)), - Expr::IsNull(inner) => Ok(Unresolved::IsNull(Rc::new(df_expr_to_unresolved(inner)?))), + Expr::IsNull(inner) => Ok(Unresolved::IsNull(bx(inner)?)), - Expr::IsNotNull(inner) => Ok(Unresolved::IsNotNull(Rc::new(df_expr_to_unresolved( - inner, - )?))), + Expr::IsNotNull(inner) => Ok(Unresolved::IsNotNull(bx(inner)?)), - Expr::Cast(c) => { - let inner = df_expr_to_unresolved(&c.expr)?; - let to = arrow_to_dtype(&c.data_type)?; - Ok(Unresolved::Cast { - expr: Rc::new(inner), - to, + Expr::Cast(c) => Ok(Unresolved::Cast { + expr: bx(&c.expr)?, + to: arrow_to_dtype(&c.data_type)?, try_cast: false, - }) - } + }), - // TRY_CAST returns NULL on conversion failure; preserve that semantic. - Expr::TryCast(c) => { - let inner = df_expr_to_unresolved(&c.expr)?; - let to = arrow_to_dtype(&c.data_type)?; - Ok(Unresolved::Cast { - expr: Rc::new(inner), - to, + // TRY_CAST returns NULL on conversion failure; preserve that semantic. + Expr::TryCast(c) => Ok(Unresolved::Cast { + expr: bx(&c.expr)?, + to: arrow_to_dtype(&c.data_type)?, try_cast: true, - }) - } + }), - Expr::InList(il) => { - let expr = df_expr_to_unresolved(&il.expr)?; - let list: Result, _> = il.list.iter().map(df_expr_to_unresolved).collect(); - Ok(Unresolved::InList { - expr: Rc::new(expr), - list: list?, - negated: il.negated, - }) - } + Expr::InList(il) => { + let list: Result, _> = il.list.iter().map(|e| self.lower_expr(e)).collect(); + Ok(Unresolved::InList { + expr: bx(&il.expr)?, + list: list?, + negated: il.negated, + }) + } - Expr::Between(b) => { - // Normalize: `x BETWEEN low AND high` → `x >= low AND x <= high`. - // `x NOT BETWEEN low AND high` → `x < low OR x > high`. - let x_low = compare(&b.expr, CompareOpKind::Ge, &b.low)?; - let x_high = compare(&b.expr, CompareOpKind::Le, &b.high)?; - if b.negated { - // NOT BETWEEN: invert each side - let lt = compare(&b.expr, CompareOpKind::Lt, &b.low)?; - let gt = compare(&b.expr, CompareOpKind::Gt, &b.high)?; - Ok(Unresolved::BoolOr(vec![lt, gt])) - } else { - Ok(Unresolved::BoolAnd(vec![x_low, x_high])) + Expr::Between(b) => { + // Normalize: `x BETWEEN low AND high` → `x >= low AND x <= high`. + // `x NOT BETWEEN low AND high` → `x < low OR x > high`. + if b.negated { + let lt = self.compare(&b.expr, CompareOpKind::Lt, &b.low)?; + let gt = self.compare(&b.expr, CompareOpKind::Gt, &b.high)?; + Ok(Unresolved::BoolOr(vec![lt, gt])) + } else { + let x_low = self.compare(&b.expr, CompareOpKind::Ge, &b.low)?; + let x_high = self.compare(&b.expr, CompareOpKind::Le, &b.high)?; + Ok(Unresolved::BoolAnd(vec![x_low, x_high])) + } } - } - // `NOW()` / `CURRENT_TIMESTAMP` read the SQL statement evaluation - // time. Keep this timestamp-typed leaf distinct from PromQL's - // Float64 Unix-seconds `EvalTimestamp`. Issue #184. - Expr::ScalarFunction(sf) - if sf.args.is_empty() - && matches!( - sf.func.name().to_ascii_lowercase().as_str(), - "now" | "current_timestamp" - ) => - { - Ok(Unresolved::CurrentTimestamp) - } + // `NOW()` / `CURRENT_TIMESTAMP` read the SQL statement evaluation + // time. Keep this timestamp-typed leaf distinct from PromQL's + // Float64 Unix-seconds `EvalTimestamp`. Issue #184. + Expr::ScalarFunction(sf) + if sf.args.is_empty() + && matches!( + sf.func.name().to_ascii_lowercase().as_str(), + "now" | "current_timestamp" + ) => + { + Ok(Unresolved::CurrentTimestamp) + } - Expr::ScalarFunction(sf) => { - let args: Result, _> = sf.args.iter().map(df_expr_to_unresolved).collect(); - Ok(Unresolved::FunctionCall { - name: if sf.func.name().eq_ignore_ascii_case("arrayelement") { - "asap_element_access".into() - } else if sf.func.name().eq_ignore_ascii_case("tupleelement") { - "asap_struct_field".into() - } else { - sf.func.name().to_string() - }, - args: args?, - }) - } + Expr::ScalarFunction(sf) => { + let args: Result, _> = sf.args.iter().map(|e| self.lower_expr(e)).collect(); + Ok(Unresolved::FunctionCall { + name: if sf.func.name().eq_ignore_ascii_case("arrayelement") { + "asap_element_access".into() + } else if sf.func.name().eq_ignore_ascii_case("tupleelement") { + "asap_struct_field".into() + } else { + sf.func.name().to_string() + }, + args: args?, + }) + } + + // Subquery-valued expressions. Each subquery plan is lowered as a + // root of its own; `resolve_root` binds it in its own scope, so an + // outer reference inside it has nothing to resolve against — a + // correlated subquery is rejected rather than mislowered. + Expr::ScalarSubquery(sq) => Ok(Unresolved::ScalarSubquery(Rc::new( + self.lower_uncorrelated_subquery(sq, "scalar subquery")?, + ))), + Expr::Exists(ex) => Ok(Unresolved::Exists { + subquery: Rc::new(self.lower_uncorrelated_subquery(&ex.subquery, "EXISTS")?), + negated: ex.negated, + }), + Expr::InSubquery(is) => { + let fields = is.subquery.subquery.schema().fields().len(); + if fields != 1 { + return Err(LoweringError::InvalidExpression(format!( + "IN (subquery) must select exactly one column, got {fields}" + ))); + } + Ok(Unresolved::InSubquery { + expr: bx(&is.expr)?, + subquery: Rc::new( + self.lower_uncorrelated_subquery(&is.subquery, "IN (subquery)")?, + ), + negated: is.negated, + }) + } - // Subquery-valued expressions in a predicate/projection — `x > (SELECT - // …)`, `x IN (SELECT …)`, `EXISTS (SELECT …)`. These need a subquery - // node in the unresolved expression IR (and a correlated-vs-uncorrelated - // decision); rejected cleanly until that lands rather than mislowered. - // Derived tables in `FROM` (the common nesting shape) ARE supported — - // see `lower_plan`'s `SubqueryAlias` arm. - Expr::ScalarSubquery(_) | Expr::InSubquery(_) | Expr::Exists(_) => Err( - LoweringError::UnsupportedFeature("subquery-valued expression in predicate".into()), - ), + other => Err(LoweringError::UnsupportedFeature(format!( + "expression: {}", + other + ))), + } + } - other => Err(LoweringError::UnsupportedFeature(format!( - "expression: {}", - other - ))), + fn lower_uncorrelated_subquery( + &self, + sq: &datafusion::logical_expr::Subquery, + what: &str, + ) -> Result { + if !sq.outer_ref_columns.is_empty() { + return Err(LoweringError::UnsupportedFeature(format!( + "correlated {what}" + ))); + } + self.lower_plan(&sq.subquery) } -} -pub(super) fn compare( - left: &Expr, - op: CompareOpKind, - right: &Expr, -) -> Result { - Ok(Unresolved::Compare { - left: Rc::new(df_expr_to_unresolved(left)?), - op, - right: Rc::new(df_expr_to_unresolved(right)?), - }) -} + pub(super) fn compare( + &self, + left: &Expr, + op: CompareOpKind, + right: &Expr, + ) -> Result { + Ok(Unresolved::Compare { + left: Box::new(self.lower_expr(left)?), + op, + right: Box::new(self.lower_expr(right)?), + semantics: ExprSemantics::Sql, + }) + } -pub(super) fn arith( - left: &Expr, - op: ArithmeticOpKind, - right: &Expr, -) -> Result { - Ok(Unresolved::Arithmetic { - op, - left: Rc::new(df_expr_to_unresolved(left)?), - right: Rc::new(df_expr_to_unresolved(right)?), - }) + fn arith( + &self, + left: &Expr, + op: ArithmeticOpKind, + right: &Expr, + ) -> Result { + Ok(Unresolved::Arithmetic { + op, + left: Box::new(self.lower_expr(left)?), + right: Box::new(self.lower_expr(right)?), + semantics: ExprSemantics::Sql, + }) + } } pub(super) fn split_disjuncts(expr: &Expr) -> Vec<&Expr> { @@ -293,12 +302,15 @@ pub(super) fn split_disjuncts(expr: &Expr) -> Vec<&Expr> { #[cfg(test)] mod tests { use super::*; + use crate::sql::SqlCatalog; use asap_types::pre_asap::schema::DataType; use datafusion::common::ScalarValue as DfScalarValue; // Typed Arrow dates normalize to the same typed form as SQL date casts. #[test] fn arrow_date_literals_preserve_value_and_type() { + let catalog = SqlCatalog::new(); + let lowerer = SqlLowerer::new(&catalog); for (value, expected) in [ ( DfScalarValue::Date32(Some(0)), @@ -311,15 +323,32 @@ mod tests { (DfScalarValue::Date32(None), ScalarValue::Null), (DfScalarValue::Date64(None), ScalarValue::Null), ] { - let actual = df_expr_to_unresolved(&Expr::Literal(value)).unwrap(); + let actual = lowerer.lower_expr(&Expr::Literal(value)).unwrap(); assert_eq!( actual, Unresolved::Cast { - expr: Rc::new(Unresolved::Literal(expected)), + expr: Box::new(Unresolved::Literal(expected)), to: DataType::Date, try_cast: false, } ); } } + + // Unary minus over a non-literal is the `Negative` scalar, SQL-flavoured. + #[test] + fn unary_minus_lowers_to_negative_with_sql_semantics() { + let catalog = SqlCatalog::new(); + let lowerer = SqlLowerer::new(&catalog); + let expr = Expr::Negative(Box::new(Expr::Column( + datafusion::common::Column::new_unqualified("x"), + ))); + assert_eq!( + lowerer.lower_expr(&expr).unwrap(), + Unresolved::Negative { + expr: Box::new(Unresolved::Column(ColumnRef::Named("x".into()))), + semantics: ExprSemantics::Sql, + } + ); + } } diff --git a/crates/frontend-sql/src/sql/mod.rs b/crates/frontend-sql/src/sql/mod.rs index e147e0bf3..885701f27 100644 --- a/crates/frontend-sql/src/sql/mod.rs +++ b/crates/frontend-sql/src/sql/mod.rs @@ -1,12 +1,12 @@ -//! SQL → the canonical, unresolved -//! [`UnresolvedQueryExpr`](asap_types::pre_asap::query_expr::UnresolvedQueryExpr) -//! (`QueryExpr`). +//! SQL → the name-based front-end tree +//! ([`UnresolvedOp`](asap_frontend_common::UnresolvedOp) / +//! [`UnresolvedScalar`](asap_frontend_common::UnresolvedScalar)). //! //! Parses SQL via DataFusion (over the catalog's registered tables), then -//! walks the unoptimized `LogicalPlan` and emits `UnresolvedQueryExpr` nodes with +//! walks the unoptimized `LogicalPlan` and emits `UnresolvedOp` nodes with //! unresolved `ColumnRef`s directly (issue #179) — the same tree shape -//! [`resolve_root`](asap_types::pre_asap::resolve_root) binds to canonical, -//! positional `QueryExpr`. Unlike PromQL's front end, SQL's +//! [`resolve_root`](asap_frontend_common::resolve_root) binds into the +//! positional, unified `OperatorNode` IR. Unlike PromQL's front end, SQL's //! Ordinary SQL `Aggregate` nodes are `Reduction::Reduce`. The explicit //! `asap_rate`/`asap_increase` bridge is the narrow exception: it //! spells a time-series range reducer with an explicit value, time-index, and @@ -51,17 +51,22 @@ use datafusion::optimizer::{AnalyzerRule, OptimizerConfig}; use datafusion::prelude::{SessionConfig, SessionContext}; use datafusion::sql::parser::DFParser; +use asap_frontend_common::{ + resolve_root, UnresolvedOp as Unresolved, UnresolvedPredicate as Predicate, + UnresolvedProjectItem as ProjectItem, UnresolvedScalar as Scalar, UnresolvedSortKey as SortKey, +}; use asap_sql_function_catalog::{AggSemantic, Arity, RewriteKind}; -use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{ - GroupKeys, Predicate, ProjectItem, Reduction, SortKey, Source, - UnresolvedQueryExpr as Unresolved, WindowFrame, WindowFrameBound, WindowFrameOffset, +use asap_types::ir::operator_properties::{ + GroupKeys, Reduction, Source, WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, }; -use asap_types::pre_asap::schema::{DataType, Schema}; +use asap_types::ir::TimeRangeKind; +use asap_types::pre_asap::agg_intent::AggIntent; +use asap_types::pre_asap::schema::{DataType, FieldDataType, Schema}; + use asap_types::pre_asap::{ - resolve_column_ref, resolve_root, ColumnRef, CompareOpKind, JoinKind, RelationalSetOpKind, - ScalarValue, WindowFuncKind, + resolve_column_ref, ColumnRef, CompareOpKind, JoinKind, RelationalSetOpKind, ScalarValue, + WindowFuncKind, }; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; @@ -77,7 +82,6 @@ mod types; pub use types::SqlCatalog; use self::dialect::GenericWithAggregateFilter; -use self::expr::df_expr_to_unresolved; use self::types::{arrow_to_dtype, scalar_value_to_asap, schema_to_arrow}; std::thread_local! { @@ -111,10 +115,10 @@ fn current_accuracy() -> AccuracyTarget { ACCURACY.with(|a| a.borrow().clone()) } -/// Lowers SQL strings to the canonical [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) -/// over a table [`SqlCatalog`]. Call -/// [`resolve_root`](asap_types::pre_asap::resolve_root) on the result for -/// the canonical, resolved tree. +/// Lowers SQL strings to the name-based [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) +/// tree over a table [`SqlCatalog`]. Call +/// [`resolve_root`](asap_frontend_common::resolve_root) on the result for +/// the resolved operator DAG. pub struct SqlLowerer<'a> { catalog: &'a SqlCatalog, dialect: SqlDialect, @@ -140,7 +144,7 @@ impl<'a> SqlLowerer<'a> { Self { catalog, dialect } } - /// Parse + lower a SQL query to the canonical, unresolved shape, threading + /// Parse + lower a SQL query to the name-based tree, threading /// `accuracy` onto every approximate intent (`Count`, `Quantile`, /// `Cardinality`) as it is built. /// @@ -167,8 +171,8 @@ impl<'a> SqlLowerer<'a> { /// a rule) that isn't wanted here — e.g. it independently rejects a /// multi-column `IN (subquery)` before `lower_in_subquery`'s own arity /// check would. Going straight to `ApplyFunctionRewrites` avoids that - /// entirely: zero behavior change for every query that doesn't call a - /// catalog-listed ClickHouse builtin. + /// entirely. TypeCoercion then records implicit conversions explicitly, + /// including timestamp literals in predicates, before IR validation. pub async fn lower( &self, sql: &str, @@ -197,6 +201,8 @@ impl<'a> SqlLowerer<'a> { let plan = state.statement_to_plan(statement).await?; let rewriter = ApplyFunctionRewrites::new(vec![Arc::new(ClickHouseBuiltinRewrite)]); let plan = rewriter.analyze(plan, ctx.state().options())?; + let plan = datafusion::optimizer::analyzer::type_coercion::TypeCoercion::new() + .analyze(plan, ctx.state().options())?; // Output schemas omit predicate and nested-expression types. Check the // typed SQL plan before lowering erases fixed-duration units. plan.apply_with_subqueries(|node| { @@ -269,8 +275,8 @@ impl<'a> SqlLowerer<'a> { // *scalar* builtin — same reason as the `AggregateUDF` loop above // (DataFusion otherwise rejects the call as an unknown function // during `SqlToRel` conversion), but with no rewrite step to follow: - // `df_expr_to_unresolved`'s `Expr::ScalarFunction` arm already lowers - // any scalar call generically to `Unresolved::FunctionCall { name, + // `lower_expr`'s `Expr::ScalarFunction` arm already lowers any + // scalar call generically to `UnresolvedScalar::FunctionCall { name, // args }`, so registering the stub is the entire fix (issue #230). for builtin in asap_sql_function_catalog::CLICKHOUSE_SCALAR_BUILTINS { ctx.register_udf(clickhouse_scalar_builtin_stub_udf( @@ -302,9 +308,24 @@ impl<'a> SqlLowerer<'a> { Ok(ctx) } - fn lower_plan(&self, plan: &LogicalPlan) -> Result { + pub(super) fn lower_plan(&self, plan: &LogicalPlan) -> Result { match plan { LogicalPlan::TableScan(scan) => self.lower_table_scan(scan), + // The one empty input row of a `SELECT` without `FROM`. + LogicalPlan::EmptyRelation(empty) => Ok(Unresolved::Values { + rows: if empty.produce_one_row { + vec![vec![]] + } else { + vec![] + }, + schema: Schema { + fields: vec![], + time_index: None, + unique_keys: vec![], + closed: true, + }, + }), + LogicalPlan::Values(values) => self.lower_values(values), LogicalPlan::Filter(filter) => self.lower_filter(filter), LogicalPlan::Projection(proj) => self.lower_projection(proj), LogicalPlan::Aggregate(agg) => self.lower_aggregate(agg), @@ -373,9 +394,7 @@ impl<'a> SqlLowerer<'a> { .iter() .map(|f| ProjectItem { alias: Some(f.name().clone()), - expr: Unresolved::Column(ColumnRef::Named( - f.name().clone(), - )), + expr: Scalar::Column(ColumnRef::Named(f.name().clone())), }) .collect(); Ok(Unresolved::Project { @@ -398,108 +417,48 @@ impl<'a> SqlLowerer<'a> { /// `WHERE` — a conjunction of ordinary predicates plus, possibly, subquery /// predicates (issue #111). /// - /// `c IN (SELECT …)` and `EXISTS (…)` are not expressions over rows; they are - /// *joins*. Each such conjunct peels off into a semi- / anti-join above the - /// filter's input, and the remaining conjuncts stay as an ordinary `Filter`. + /// The ordinary conjuncts stay one predicate, folded onto a bare `Scan` + /// (`filter_or_fold`). A subquery conjunct — `c IN (SELECT …)`, `EXISTS + /// (…)`, `x > (SELECT …)` — is a row filter whose predicate reads another + /// operator (`UnresolvedScalar::InSubquery` / `Exists` / + /// `ScalarSubquery`); each one becomes its own `Filter` **above** the + /// ordinary predicate, so the shared `canonicalize` pass can turn it into + /// the join it is without having to peel it out of a conjunction or off + /// a `Scan` (it only lifts subqueries out of `Filter` / `Project`). A + /// semi-join only ever drops left rows, so the two orders agree. /// - /// The residual filter is applied **below** the joins, which is where it sat - /// before: a semi-join only ever drops left rows, so the two orders agree — - /// and keeping the fold-onto-`Scan` (`filter_or_fold`) below the joins - /// matches where the old converter folded it too. + /// The one subquery shape still lowered to a join here is a *correlated* + /// `EXISTS`: its correlation references both sides, which only a join + /// predicate can bind (a subquery referenced from a scalar position is + /// resolved as a root in its own scope). fn lower_filter(&self, filter: &logical_expr::Filter) -> Result { let mut conjuncts = Vec::new(); split_conjunction(&filter.predicate, &mut conjuncts); - let (subqueries, residual): (Vec<_>, Vec<_>) = conjuncts - .into_iter() - .partition(|e| matches!(e, Expr::InSubquery(_) | Expr::Exists(_))); + let (subqueries, residual): (Vec<_>, Vec<_>) = + conjuncts.into_iter().partition(|e| reads_subquery(e)); let input = self.lower_plan(&filter.input)?; let mut node = match rebuild_conjunction(&residual) { - Some(pred) => filter_or_fold(df_expr_to_unresolved(&pred)?, input), + Some(pred) => filter_or_fold(self.lower_expr(&pred)?, input), None => input, }; for sq in subqueries { node = match sq { - Expr::InSubquery(is) => self.lower_in_subquery(is, node)?, - Expr::Exists(ex) => self.lower_exists(ex, node)?, - _ => unreachable!("partitioned above"), + Expr::Exists(ex) if !ex.subquery.outer_ref_columns.is_empty() => { + self.lower_correlated_exists(ex, node)? + } + other => Unresolved::Filter { + pred: Predicate(self.lower_expr(other)?), + child: Rc::new(node), + }, }; } Ok(node) } - /// `c IN (SELECT k FROM …)` → a semi-join on `c = k` (issue #111). - fn lower_in_subquery( - &self, - is: &logical_expr::expr::InSubquery, - left: Unresolved, - ) -> Result { - if is.negated { - // `NOT IN` is not an anti-join. Under three-valued logic a single - // NULL among the subquery's rows makes `c NOT IN (…)` UNKNOWN for - // every `c`, so the query returns nothing — while an anti-join - // returns every unmatched left row. Reject rather than mislower. - return Err(LoweringError::UnsupportedFeature( - "NOT IN (subquery): its NULL semantics are not an anti-join".into(), - )); - } - if !is.subquery.outer_ref_columns.is_empty() { - return Err(LoweringError::UnsupportedFeature( - "correlated IN (subquery)".into(), - )); - } - let inner = is.subquery.subquery.as_ref(); - let fields = inner.schema().fields(); - if fields.len() != 1 { - return Err(LoweringError::InvalidExpression(format!( - "IN (subquery) must select exactly one column, got {}", - fields.len() - ))); - } - let key = &fields[0]; - // Project the key under a name the outer relation cannot also carry. The - // join predicate resolves against the concatenated `left ++ right` - // schema, and a bare `hosts.service` over an unqualified subquery output - // falls back to a name lookup that finds the *left's* `service` first — - // silently making the predicate `service = service`, i.e. always true. - let right = match inner { - // Rebuild the subquery's projection with the synthetic alias, so a - // computed key (`SELECT bytes + 1 …`) is named rather than becoming - // the anonymous `col_0` that nothing can reference. - LogicalPlan::Projection(p) if p.expr.len() == 1 => Unresolved::Project { - cols: vec![ProjectItem { - alias: Some(IN_SUBQUERY_KEY.to_string()), - expr: df_expr_to_unresolved(unalias(&p.expr[0]))?, - }], - qualifier: None, - child: Rc::new(self.lower_plan(&p.input)?), - }, - other => Unresolved::Project { - cols: vec![ProjectItem { - alias: Some(IN_SUBQUERY_KEY.to_string()), - expr: Unresolved::Column(ColumnRef::Named(key.name().clone())), - }], - qualifier: None, - child: Rc::new(self.lower_plan(other)?), - }, - }; - Ok(Unresolved::Join { - kind: JoinKind::Semi, - pred: Predicate(Rc::new(Unresolved::Compare { - left: Rc::new(df_expr_to_unresolved(&is.expr)?), - op: CompareOpKind::Eq, - right: Rc::new(Unresolved::Column(ColumnRef::Named( - IN_SUBQUERY_KEY.to_string(), - ))), - })), - left: Rc::new(left), - right: Rc::new(right), - }) - } - /// `[NOT] EXISTS (SELECT … WHERE inner.k = outer.k)` → a semi- / anti-join /// on the correlation predicate (issue #111). - fn lower_exists( + fn lower_correlated_exists( &self, ex: &logical_expr::expr::Exists, left: Unresolved, @@ -520,12 +479,9 @@ impl<'a> SqlLowerer<'a> { // the join predicate. Whatever is left stays an ordinary inner filter. let (inner, correlation) = split_correlation(inner)?; let right = self.lower_plan(&inner)?; - // No correlation conjunct (a genuinely uncorrelated `EXISTS`) means - // the join condition is unconditionally true — same convention as an - // unconditional `JOIN` (`lower_join`, below). let pred = match correlation { - Some(e) => Predicate(Rc::new(df_expr_to_unresolved(&e)?)), - None => Predicate(Rc::new(Unresolved::Literal(ScalarValue::Boolean(true)))), + Some(e) => Predicate(self.lower_expr(&e)?), + None => Predicate(Scalar::Literal(ScalarValue::Boolean(true))), }; Ok(Unresolved::Join { kind, @@ -535,6 +491,37 @@ impl<'a> SqlLowerer<'a> { }) } + /// `VALUES (…), (…)` — one row per values row, typed by DataFusion's + /// declared schema. Row expressions have no input-column scope. + fn lower_values(&self, values: &logical_expr::Values) -> Result { + let rows = values + .values + .iter() + .map(|row| row.iter().map(|e| self.lower_expr(e)).collect()) + .collect::>, LoweringError>>()?; + let fields = values + .schema + .fields() + .iter() + .map(|f| { + Ok(asap_types::pre_asap::Field::plain( + f.name().clone(), + arrow_to_dtype(f.data_type())?, + f.is_nullable(), + )) + }) + .collect::, LoweringError>>()?; + Ok(Unresolved::Values { + rows, + schema: Schema { + fields, + time_index: None, + unique_keys: vec![], + closed: true, + }, + }) + } + /// Table leaf — carries the catalog's resolved schema directly on `Scan` /// (`schema: Some(_)`), so `resolve_root`'s SchemaResolver doesn't need to /// usage-derive it (SQL is never schemaless). Projection pushdown is left @@ -558,8 +545,8 @@ impl<'a> SqlLowerer<'a> { .get(table) .ok_or_else(|| LoweringError::TableNotFound(table.to_string()))?; let qualified = Schema { - columns: schema - .columns + fields: schema + .fields .iter() .cloned() .map(|c| c.with_table(qualifier)) @@ -598,23 +585,17 @@ impl<'a> SqlLowerer<'a> { let mut conjuncts = join .on .iter() - .map(|(l, r)| { - Ok(Unresolved::Compare { - left: Rc::new(df_expr_to_unresolved(l)?), - op: CompareOpKind::Eq, - right: Rc::new(df_expr_to_unresolved(r)?), - }) - }) + .map(|(l, r)| self.compare(l, CompareOpKind::Eq, r)) .collect::, LoweringError>>()?; if let Some(filter) = &join.filter { - conjuncts.push(df_expr_to_unresolved(filter)?); + conjuncts.push(self.lower_expr(filter)?); } - let pred = Predicate(Rc::new(match conjuncts.len() { + let pred = Predicate(match conjuncts.len() { // No condition (a CROSS JOIN) is unconditionally true. - 0 => Unresolved::Literal(ScalarValue::Boolean(true)), + 0 => Scalar::Literal(ScalarValue::Boolean(true)), 1 => conjuncts.pop().unwrap(), - _ => Unresolved::BoolAnd(conjuncts), - })); + _ => Scalar::BoolAnd(conjuncts), + }); Ok(Unresolved::Join { kind, pred, @@ -637,6 +618,10 @@ impl<'a> SqlLowerer<'a> { .window_expr .first() .ok_or_else(|| LoweringError::InvalidExpression("empty window expression".into()))?; + let first = match first { + Expr::Alias(alias) => alias.expr.as_ref(), + other => other, + }; let Expr::WindowFunction(wf) = first else { return Err(LoweringError::InvalidExpression( "expected a window function in Window plan node".into(), @@ -646,12 +631,12 @@ impl<'a> SqlLowerer<'a> { let mut args = wf .args .iter() - .map(df_expr_to_unresolved) + .map(|e| self.lower_expr(e)) .collect::, _>>()?; // Nth_value: lift N from the (literal) 2nd arg, keep only the column. let func = if matches!(func, WindowFuncKind::NthValue(None)) { let n = match args.get(1) { - Some(Unresolved::Literal(ScalarValue::Int64(n))) if *n > 0 => *n as u64, + Some(Scalar::Literal(ScalarValue::Int64(n))) if *n > 0 => *n as u64, other => { return Err(LoweringError::InvalidExpression(format!( "NTH_VALUE requires a positive integer literal 2nd arg, got {other:?}" @@ -672,7 +657,7 @@ impl<'a> SqlLowerer<'a> { .order_by .iter() .map(|s| { - df_expr_to_unresolved(&s.expr).map(|expr| SortKey { + self.lower_expr(&s.expr).map(|expr| SortKey { expr, ascending: s.asc, nulls_first: s.nulls_first, @@ -707,7 +692,7 @@ impl<'a> SqlLowerer<'a> { let input = self.lower_plan(&proj.input)?; return Ok(match bridge { PlanningBridge::PromqlSubquery { range, resolution } => { - let child = Rc::new(temporal_bridge_projection(proj, input)?); + let child = Rc::new(self.temporal_bridge_projection(proj, input)?); Unresolved::PromqlSubquery { range, resolution: Some(resolution), @@ -740,22 +725,22 @@ impl<'a> SqlLowerer<'a> { .map(|e| match e { Expr::Alias(a) => { let expr = if temporal_input && is_temporal_output_column(&a.expr) { - Unresolved::Column(ColumnRef::Named("value".into())) + Scalar::Column(ColumnRef::Named("value".into())) } else { - df_expr_to_unresolved(&a.expr)? + self.lower_expr(&a.expr)? }; - Ok::, LoweringError>(ProjectItem { + Ok::(ProjectItem { expr, alias: Some(a.name.clone()), }) } _ => { let expr = if temporal_input && is_temporal_output_column(e) { - Unresolved::Column(ColumnRef::Named("value".into())) + Scalar::Column(ColumnRef::Named("value".into())) } else { - df_expr_to_unresolved(e)? + self.lower_expr(e)? }; - Ok::, LoweringError>(ProjectItem { expr, alias: None }) + Ok::(ProjectItem { expr, alias: None }) } }) .collect::, _>>()?; @@ -802,7 +787,7 @@ impl<'a> SqlLowerer<'a> { // reducer expression (`GROUP BY date_trunc(…)`, `SUM(a * 8)`) has no // slot. Materialize each one as a derived column in a `Project` beneath // the aggregate, then group/reduce over that column (issue #110). - let mut derived = DerivedCols::default(); + let mut derived = DerivedCols::new(self); // DataFusion strips `AS m` from a grouping expression, so the aggregate // schema's field name is what the enclosing Projection references — @@ -827,7 +812,7 @@ impl<'a> SqlLowerer<'a> { .get(i) .cloned() .unwrap_or_else(|| other.to_string()); - derived.materialize(name.clone(), df_expr_to_unresolved(other)?)?; + derived.materialize(name.clone(), self.lower_expr(other)?)?; keys.push(ColumnRef::Named(name)); } } @@ -869,7 +854,7 @@ impl<'a> SqlLowerer<'a> { .iter() .map(|f| { f.as_ref() - .map(|f| Ok(Predicate(Rc::new(df_expr_to_unresolved(f)?)))) + .map(|f| Ok(Predicate(self.lower_expr(f)?))) .transpose() }) .collect::, LoweringError>>()? @@ -922,12 +907,7 @@ impl<'a> SqlLowerer<'a> { )) })?; - let resolved_input = resolve_root(&input)?; - let input_schema = resolved_input.output_schema().map_err(|error| { - LoweringError::InvalidExpression(format!( - "cannot derive temporal aggregate input schema: {error}" - )) - })?; + let input_schema = resolve_root(&input)?.schema.clone(); let timestamp_id = resolve_column_ref(×tamp_ref, &input_schema).map_err(|error| { LoweringError::InvalidExpression(format!("{name} timestamp argument: {error}")) })?; @@ -941,8 +921,8 @@ impl<'a> SqlLowerer<'a> { })?; if value_id == timestamp_id || !matches!( - input_schema.columns[value_id].dtype, - DataType::Int64 | DataType::Float64 + input_schema.fields[value_id].dtype, + FieldDataType::Plain(DataType::Int64 | DataType::Float64) ) { return Err(LoweringError::InvalidExpression(format!( @@ -1000,18 +980,18 @@ impl<'a> SqlLowerer<'a> { let mut cols = vec![ ProjectItem { alias: Some("ts".into()), - expr: Unresolved::Column(timestamp_ref.clone()), + expr: Scalar::Column(timestamp_ref.clone()), }, ProjectItem { alias: Some("value".into()), - expr: Unresolved::Column(value_ref.clone()), + expr: Scalar::Column(value_ref.clone()), }, ]; for group_ref in group_refs { let group_name = named_ref(&group_ref).to_string(); cols.push(ProjectItem { alias: Some(group_name), - expr: Unresolved::Column(group_ref), + expr: Scalar::Column(group_ref), }); } let child = Unresolved::Project { @@ -1019,8 +999,11 @@ impl<'a> SqlLowerer<'a> { qualifier: None, child: Rc::new(input), }; + // The explicit window is a range selector over the series, the same + // shape PromQL's `rate(m[5m])` lowers to. let child = Unresolved::TimeRange { range: Duration::from_millis(window_ms), + kind: TimeRangeKind::Range, child: Rc::new(child), }; let intent = match name.as_str() { @@ -1099,7 +1082,7 @@ impl<'a> SqlLowerer<'a> { // Reducer arguments still materialize as derived columns (#110); the // grouping keys are plain columns, so they only need carrying through. - let mut derived = DerivedCols::default(); + let mut derived = DerivedCols::new(self); for e in &distinct { derived.passthrough(e)?; } @@ -1137,10 +1120,10 @@ impl<'a> SqlLowerer<'a> { .map(|((name, dtype), e)| ProjectItem { alias: Some(name.clone()), expr: if level.contains(e) { - Unresolved::Column(ColumnRef::Named(name.clone())) + Scalar::Column(ColumnRef::Named(name.clone())) } else { - Unresolved::Cast { - expr: Rc::new(Unresolved::Literal(ScalarValue::Null)), + Scalar::Cast { + expr: Box::new(Scalar::Literal(ScalarValue::Null)), to: dtype.clone(), try_cast: false, } @@ -1148,7 +1131,7 @@ impl<'a> SqlLowerer<'a> { }) .chain(output_names.iter().map(|n| ProjectItem { alias: Some(n.clone()), - expr: Unresolved::Column(ColumnRef::Named(n.clone())), + expr: Scalar::Column(ColumnRef::Named(n.clone())), })) .collect(); Ok(Unresolved::Project { @@ -1180,7 +1163,7 @@ impl<'a> SqlLowerer<'a> { .expr .iter() .map(|s| { - df_expr_to_unresolved(&s.expr).map(|expr| SortKey { + self.lower_expr(&s.expr).map(|expr| SortKey { expr, ascending: s.asc, nulls_first: s.nulls_first, @@ -1200,8 +1183,10 @@ impl<'a> SqlLowerer<'a> { // Count-ranked `LIMIT k` over a `Sort` is promoted to the heavy-hitter // `TopK` by the shared `canonicalize` pass (issue #34), not here. Ok(Unresolved::Limit { - n: eval_fetch(&limit.fetch).unwrap_or(usize::MAX), + // No (literal) fetch is offset-only. + n: eval_fetch(&limit.fetch), offset: eval_fetch(&limit.skip).unwrap_or(0), + partition_by: GroupKeys::none(), child: Rc::new(self.lower_plan(&limit.input)?), }) } @@ -1281,49 +1266,52 @@ fn planning_bridge( /// its output slot (`... asap_promql_subquery(...) AS value ...`). This makes /// the bridge schema-preserving without silently retaining columns that SQL /// projected away. -fn temporal_bridge_projection( - projection: &logical_expr::Projection, - child: Unresolved, -) -> Result { - let cols = projection - .expr - .iter() - .map(|expr| { - if let Expr::ScalarFunction(call) = unalias(expr) { - if call - .func - .name() - .eq_ignore_ascii_case("asap_promql_subquery") - { - let Expr::Alias(alias) = expr else { - return Err(LoweringError::InvalidExpression( - "asap_promql_subquery must have an alias naming its child value column" - .into(), - )); - }; - return Ok(ProjectItem { - expr: Unresolved::Column(ColumnRef::Named(alias.name.clone())), +impl SqlLowerer<'_> { + fn temporal_bridge_projection( + &self, + projection: &logical_expr::Projection, + child: Unresolved, + ) -> Result { + let cols = projection + .expr + .iter() + .map(|expr| { + if let Expr::ScalarFunction(call) = unalias(expr) { + if call + .func + .name() + .eq_ignore_ascii_case("asap_promql_subquery") + { + let Expr::Alias(alias) = expr else { + return Err(LoweringError::InvalidExpression( + "asap_promql_subquery must have an alias naming its child value column" + .into(), + )); + }; + return Ok(ProjectItem { + expr: Scalar::Column(ColumnRef::Named(alias.name.clone())), + alias: Some(alias.name.clone()), + }); + } + } + match expr { + Expr::Alias(alias) => Ok(ProjectItem { + expr: self.lower_expr(&alias.expr)?, alias: Some(alias.name.clone()), - }); + }), + other => Ok(ProjectItem { + expr: self.lower_expr(other)?, + alias: None, + }), } - } - match expr { - Expr::Alias(alias) => Ok(ProjectItem { - expr: df_expr_to_unresolved(&alias.expr)?, - alias: Some(alias.name.clone()), - }), - other => Ok(ProjectItem { - expr: df_expr_to_unresolved(other)?, - alias: None, - }), - } + }) + .collect::, LoweringError>>()?; + Ok(Unresolved::Project { + cols, + qualifier: None, + child: Rc::new(child), }) - .collect::, LoweringError>>()?; - Ok(Unresolved::Project { - cols, - qualifier: None, - child: Rc::new(child), - }) + } } fn positive_millis_literal(expr: &Expr, argument: &str) -> Result { @@ -1408,8 +1396,8 @@ fn arity_to_signature(arity: Arity) -> Signature { // (which must become a real `AggIntent`, hence the rewrite to a native // DataFusion aggregate shape `lower_agg_intent` can classify), a scalar // function call in this IR is already deliberately opaque — -// `expr::df_expr_to_unresolved`'s `Expr::ScalarFunction` arm lowers *any* -// scalar call generically to `Unresolved::FunctionCall { name, args }`, with +// `SqlLowerer::lower_expr`'s `Expr::ScalarFunction` arm lowers *any* +// scalar call generically to `UnresolvedScalar::FunctionCall { name, args }`, with // zero name-specific logic. So teaching DataFusion's planner to accept a // ClickHouse scalar builtin's name — a stub `ScalarUDF`, registered below — // is the entire fix; the existing generic lowering already does the rest. @@ -1923,23 +1911,19 @@ fn lower_arg_selector( })) } -/// The name an `IN (subquery)`'s key column is projected under, so the join -/// predicate cannot bind it to a same-named column of the outer relation. -const IN_SUBQUERY_KEY: &str = "__asap_in_key"; - /// Fold `pred` directly onto `child.predicates` when `child` is a bare `Scan` /// (a `WHERE` directly over a table), otherwise wrap it in an ordinary /// `Filter` — canonical's invariant that a `Filter` never sits directly over a /// `Scan`. A front end emitting the canonical shape directly is responsible /// for maintaining that invariant itself (issue #179). -fn filter_or_fold(pred: Unresolved, child: Unresolved) -> Unresolved { +fn filter_or_fold(pred: Scalar, child: Unresolved) -> Unresolved { match child { Unresolved::Scan { source, mut predicates, schema, } => { - predicates.push(Predicate(Rc::new(pred))); + predicates.push(Predicate(pred)); Unresolved::Scan { source, predicates, @@ -1947,7 +1931,7 @@ fn filter_or_fold(pred: Unresolved, child: Unresolved) -> Unresolved { } } other => Unresolved::Filter { - pred: Predicate(Rc::new(pred)), + pred: Predicate(pred), child: Rc::new(other), }, } @@ -1964,6 +1948,18 @@ fn split_conjunction<'a>(expr: &'a Expr, out: &mut Vec<&'a Expr>) { } } +/// Whether `expr` reads another operator anywhere inside it (`EXISTS`, +/// `IN (…)`, a scalar subquery). +fn reads_subquery(expr: &Expr) -> bool { + expr.exists(|e| { + Ok(matches!( + e, + Expr::ScalarSubquery(_) | Expr::InSubquery(_) | Expr::Exists(_) + )) + }) + .expect("the predicate never fails") +} + /// Re-`AND` the conjuncts, or `None` when there are none left. fn rebuild_conjunction(conjuncts: &[&Expr]) -> Option { conjuncts @@ -2112,9 +2108,9 @@ fn expand_grouping_set(gs: &logical_expr::GroupingSet) -> Vec> { /// The projection also has to carry through the plain columns the aggregate /// still references, since a `Project` replaces its child's schema rather than /// extending it. -#[derive(Default)] -struct DerivedCols { - cols: Vec>, +struct DerivedCols<'l> { + lowerer: &'l SqlLowerer<'l>, + cols: Vec, /// Whether any column is genuinely derived. Without one the aggregate keeps /// its original child, so trees that lower today keep their exact shape. any: bool, @@ -2123,12 +2119,21 @@ struct DerivedCols { collision: Option, } -impl DerivedCols { +impl<'l> DerivedCols<'l> { + fn new(lowerer: &'l SqlLowerer<'l>) -> Self { + Self { + lowerer, + cols: Vec::new(), + any: false, + collision: None, + } + } + /// Add `alias := expr`, or note a collision if `alias` already means /// something else. `Project` carries one relation qualifier for all its /// columns, so `a.k` and `b.k` cannot both survive it — but that only /// matters when a projection gets inserted at all. - fn push(&mut self, alias: String, expr: Unresolved) { + fn push(&mut self, alias: String, expr: Scalar) { let existing = self .cols .iter() @@ -2151,12 +2156,12 @@ impl DerivedCols { let Expr::Column(c) = unalias(expr) else { return Ok(()); }; - self.push(c.name.clone(), df_expr_to_unresolved(expr)?); + self.push(c.name.clone(), self.lowerer.lower_expr(expr)?); Ok(()) } /// A genuinely derived column: `alias` now names `expr`'s value. - fn materialize(&mut self, alias: String, expr: Unresolved) -> Result<(), LoweringError> { + fn materialize(&mut self, alias: String, expr: Scalar) -> Result<(), LoweringError> { self.any = true; self.push(alias, expr); Ok(()) @@ -2178,7 +2183,7 @@ impl DerivedCols { let mut rewritten = agg_fn.clone(); for arg in &mut rewritten.args { let alias = unalias(arg).to_string(); - self.materialize(alias.clone(), df_expr_to_unresolved(arg)?)?; + self.materialize(alias.clone(), self.lowerer.lower_expr(arg)?)?; *arg = Expr::Column(DfColumn::new_unqualified(alias)); } return Ok(Expr::AggregateFunction(rewritten)); @@ -2200,12 +2205,12 @@ impl DerivedCols { } match agg_col_name(&agg_fn.args) { Some(name) => { - self.push(name, df_expr_to_unresolved(arg)?); + self.push(name, self.lowerer.lower_expr(arg)?); Ok(expr.clone()) } None => { let alias = unalias(arg).to_string(); - self.materialize(alias.clone(), df_expr_to_unresolved(arg)?)?; + self.materialize(alias.clone(), self.lowerer.lower_expr(arg)?)?; let mut agg_fn = agg_fn.clone(); agg_fn.args[0] = Expr::Column(DfColumn::new_unqualified(alias)); Ok(Expr::AggregateFunction(agg_fn)) @@ -2276,7 +2281,7 @@ fn expr_to_group_ref(expr: &Expr) -> Result { match expr { // Preserve the relation qualifier so a GROUP BY / PARTITION BY key over a // join (`b.k` vs `a.k`) resolves to the correct side — the same rule the - // scalar predicate path uses (`df_expr_to_unresolved`). + // scalar predicate path uses (`lower_expr`). Expr::Column(col) => Ok(match &col.relation { Some(rel) => ColumnRef::Qualified { table: rel.to_string(), diff --git a/crates/frontend-sql/src/sql/types.rs b/crates/frontend-sql/src/sql/types.rs index 5382c4171..583a84398 100644 --- a/crates/frontend-sql/src/sql/types.rs +++ b/crates/frontend-sql/src/sql/types.rs @@ -5,11 +5,11 @@ use std::collections::HashMap; use datafusion::arrow::datatypes::{ - DataType as ArrowDataType, Field, Fields, Schema as ArrowSchema, + DataType as ArrowDataType, Field as ArrowField, Fields, Schema as ArrowSchema, }; use datafusion::common::ScalarValue as DfScalarValue; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::pre_asap::ScalarValue; use crate::error::SqlError as LoweringError; @@ -93,7 +93,7 @@ pub(super) fn arrow_to_dtype(dt: &ArrowDataType) -> Result Ok(DataType::Date), ArrowDataType::Interval(_) => Ok(DataType::Interval), ArrowDataType::List(element) => Ok(DataType::List { - element: Box::new(Column::new( + element: Box::new(Field::new( element.name(), arrow_to_dtype(element.data_type())?, element.is_nullable(), @@ -103,7 +103,7 @@ pub(super) fn arrow_to_dtype(dt: &ArrowDataType) -> Result ArrowDataType { DataType::Float64 => ArrowDataType::Float64, DataType::Utf8 => ArrowDataType::Utf8, DataType::Bool => ArrowDataType::Boolean, - DataType::List { element } => ArrowDataType::List(std::sync::Arc::new(Field::new( + DataType::List { element } => ArrowDataType::List(std::sync::Arc::new(ArrowField::new( &element.name, dtype_to_arrow(&element.dtype), element.nullable, @@ -150,7 +150,9 @@ pub(super) fn dtype_to_arrow(dt: &DataType) -> ArrowDataType { DataType::Struct { fields } => ArrowDataType::Struct( fields .iter() - .map(|field| Field::new(&field.name, dtype_to_arrow(&field.dtype), field.nullable)) + .map(|field| { + ArrowField::new(&field.name, dtype_to_arrow(&field.dtype), field.nullable) + }) .collect::>() .into(), ), @@ -159,12 +161,12 @@ pub(super) fn dtype_to_arrow(dt: &DataType) -> ArrowDataType { value, value_nullable, } => ArrowDataType::Map( - std::sync::Arc::new(Field::new( + std::sync::Arc::new(ArrowField::new( "entries", ArrowDataType::Struct( vec![ - Field::new("key", dtype_to_arrow(key), false), - Field::new("value", dtype_to_arrow(value), *value_nullable), + ArrowField::new("key", dtype_to_arrow(key), false), + ArrowField::new("value", dtype_to_arrow(value), *value_nullable), ] .into(), ), @@ -193,9 +195,11 @@ pub(super) fn dtype_to_arrow(dt: &DataType) -> ArrowDataType { /// Build an Arrow schema from a canonical [`Schema`] (column name + type + nullability). pub(super) fn schema_to_arrow(schema: &Schema) -> ArrowSchema { let fields: Fields = schema - .columns + .fields .iter() - .map(|c: &Column| Field::new(&c.name, dtype_to_arrow(&c.dtype), c.nullable)) + .map(|c: &Field| { + ArrowField::new(&c.name, dtype_to_arrow(c.expect_plain_dtype()), c.nullable) + }) .collect(); ArrowSchema::new(fields) } @@ -302,21 +306,21 @@ mod collection_tests { fn nested_collections_preserve_field_names_order_and_nullability() { let dtype = DataType::Struct { fields: vec![ - Column::new( + Field::new( "samples", DataType::List { - element: Box::new(Column::new( + element: Box::new(Field::new( "sample", DataType::Struct { fields: vec![ - Column::new("timestamp", DataType::Timestamp, false), - Column::new("value", DataType::Float64, true), - Column::new( + Field::new("timestamp", DataType::Timestamp, false), + Field::new("value", DataType::Float64, true), + Field::new( "labels", DataType::Map { key: Box::new(DataType::Utf8), value: Box::new(DataType::List { - element: Box::new(Column::new( + element: Box::new(Field::new( "label_value", DataType::Utf8, false, @@ -333,7 +337,7 @@ mod collection_tests { }, false, ), - Column::new("optional", DataType::Int64, true), + Field::new("optional", DataType::Int64, true), ], }; let arrow = dtype_to_arrow(&dtype); @@ -345,7 +349,7 @@ mod collection_tests { #[test] fn empty_struct_and_nonnullable_list_element_roundtrip() { let dtype = DataType::List { - element: Box::new(Column::new( + element: Box::new(Field::new( "empty", DataType::Struct { fields: vec![] }, false, diff --git a/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs b/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs index ae27e31e0..cc24329cd 100644 --- a/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs +++ b/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs @@ -35,17 +35,20 @@ //! `Err`, never panics. The pinned per-query outcomes document today's real //! coverage so a regression (or a future improvement) is visible, not silent. +use std::rc::Rc; + use asap_frontend_sql::{lower_sql_dialect, SqlCatalog, SqlError as LoweringError}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::ir::{NonASAPOp, OperatorNode}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::pre_asap::{AggIntent, GroupKeys}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use datafusion::error::DataFusionError; const CORPUS: &str = include_str!("data/bgp_analytics.sql"); -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } /// `bgp_updates(timestamp, collector, peer_ip, peer_asn, prefix, operation, @@ -92,7 +95,7 @@ fn queries() -> Vec { .collect() } -async fn lower(q: &str) -> Result { +async fn lower(q: &str) -> Result, LoweringError> { lower_sql_dialect( q, &catalog(), @@ -218,19 +221,19 @@ async fn corpus_lowering_matches_the_pinned_per_query_outcome() { ); } -fn first_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { - match qe { - QueryExpr::Aggregate { +fn first_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { + match node.expect_non_asap() { + NonASAPOp::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_aggregate(child), _ => None, } } @@ -260,7 +263,7 @@ async fn top_k_queries_are_count_grouped_by_prefix() { idx + 1 ); assert!( - matches!(qe, QueryExpr::Limit { .. }), + matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. }), "q{} ({label}) top-k shape keeps the LIMIT at the root: {qe:?}", idx + 1 ); diff --git a/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs b/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs index 302a349ad..ad9e69a7d 100644 --- a/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs +++ b/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs @@ -14,7 +14,7 @@ //! tally** by outcome category -- see the module doc on [`Category`] for why. use asap_frontend_sql::{lower_sql_dialect, SqlCatalog, SqlError}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use datafusion::error::DataFusionError; @@ -34,8 +34,8 @@ struct QueryCase { sql: String, } -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } /// `bgp.bgp_updates`, widened past the 7-column `bgp_analytics` schema with @@ -70,7 +70,7 @@ fn catalog() -> SqlCatalog { .with_table("bgp.bgp_updates", updates) } -async fn lower(q: &str) -> Result { +async fn lower(q: &str) -> Result, SqlError> { lower_sql_dialect( q, &catalog(), @@ -91,6 +91,8 @@ async fn lower(q: &str) -> Result { /// is that signal, ratcheted so a category shifting size is visible. #[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] enum Category { + /// A planned expression lacks a faithful registered IR type contract. + InvalidRepresentation, Lowered, /// `DataFusionError::Plan` -- almost entirely "unknown function" for a /// ClickHouse-only builtin (`uniqExact`, `countIf`, `splitByChar`, ...). @@ -119,6 +121,7 @@ fn categorize(err: &SqlError) -> Category { SqlError::DataFusion(DataFusionError::SQL(_, _)) => Category::Parse, SqlError::DataFusion(DataFusionError::NotImplemented(_)) => Category::NotImplemented, SqlError::UnsupportedFeature(_) => Category::UnsupportedFeature, + SqlError::Convert(_) => Category::InvalidRepresentation, _ => Category::Other, } } @@ -194,8 +197,11 @@ async fn corpus_lowering_matches_the_pinned_aggregate_tally() { // 152 -> 154: `ScalarValue::Interval` (this branch) converts the // `INTERVAL x unit` literal the two `toStartOfInterval(...)` queries // carry. - expect(Category::Lowered, 154); - expect(Category::Plan, 40); + // Previously admitted ClickHouse stubs used placeholder Float64 types. + // Unregistered functions and incompatible operands now fail closed. + expect(Category::Lowered, 105); + expect(Category::InvalidRepresentation, 53); + expect(Category::Plan, 41); expect(Category::Schema, 0); expect(Category::Parse, 0); // One query that used to fail at `uniqExact` (`Plan`) now clears that @@ -207,7 +213,7 @@ async fn corpus_lowering_matches_the_pinned_aggregate_tally() { // Typed Map access lowers one prior gap; six array accesses now fail // during typed planning because the Map adapter rejects array inputs. expect(Category::NotImplemented, 0); - expect(Category::UnsupportedFeature, 6); + expect(Category::UnsupportedFeature, 1); // Was 2: the two `toStartOfInterval(...)` queries whose `INTERVAL`-literal // conversion gap the `toStartOfInterval` note above describes. Both now // lower end to end and are counted in `Lowered`. diff --git a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs index 273597e7d..1d4e2f47e 100644 --- a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs +++ b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs @@ -19,15 +19,18 @@ //! Schema: `packets(srcip, dstip, srcport, dstport, proto, time, pkt_len)`; //! flow / 5-tuple = `(srcip, dstip, srcport, dstport, proto)`. +use std::rc::Rc; + use asap_frontend_sql::{lower_sql, SqlCatalog, SqlError as LoweringError}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::ir::{NonASAPOp, OperatorNode}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::pre_asap::{AggIntent, GroupKeys}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/synthetic_packet_trace_queries.sql"); -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } /// `packets(srcip, dstip, srcport, dstport, proto, time, pkt_len)`. IPs and @@ -67,110 +70,61 @@ fn queries() -> Vec { // ── tree helpers ────────────────────────────────────────────────────────────── -/// Every `AggIntent` in the tree, root-to-leaf. -fn intents(e: &QueryExpr) -> Vec { - let mut out = Vec::new(); - fn go(e: &QueryExpr, out: &mut Vec) { - match e { - QueryExpr::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - go(child, out); - } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::Project { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => go(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { - left: lhs, - right: rhs, - .. - } - | QueryExpr::SetOp { - left: lhs, - right: rhs, - .. - } => { - go(lhs, out); - go(rhs, out); - } - QueryExpr::Concat { children, .. } => children.iter().for_each(|c| go(c, out)), - QueryExpr::PromqlVectorFromScalar(inner) | QueryExpr::PromqlScalarFromVector(inner) => { - go(inner, out) - } - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - // Scalar expression variants (issue #205): `AggIntent` only ever - // lives in `Aggregate.measures`, never nested inside a scalar - // expression tree, so there's nothing to recurse into here. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} - } - } - go(e, &mut out); - out +/// The operator of a front-end node: a front-end DAG never holds an ASAP node. +fn op(node: &OperatorNode) -> &NonASAPOp { + node.expect_non_asap() +} + +/// Every `AggIntent` in the DAG, root-to-leaf (every reachable node — +/// `AggIntent` only ever lives in `Aggregate.measures`). +fn intents(e: &Rc) -> Vec { + OperatorNode::reachable(e) + .iter() + .filter_map(|node| match op(node) { + NonASAPOp::Aggregate { measures, .. } => Some(measures.clone()), + _ => None, + }) + .flatten() + .collect() } /// The first `Aggregate`'s `(by, measures)` along the single-child spine. SQL /// never lowers to `Reduction::PerEntity` (it has no per-series concept), so /// `expect_reduce()` here is a safe, load-bearing assumption for these tests. -fn first_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { - match qe { - QueryExpr::Aggregate { +fn first_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { + match op(node) { + NonASAPOp::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::SQLWindowFunc { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_aggregate(child), _ => None, } } /// Whether a `SQLWindowFunc` (analytic `OVER (…)`) node appears anywhere. -fn has_window_func(qe: &QueryExpr) -> bool { - match qe { - QueryExpr::SQLWindowFunc { .. } => true, - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => has_window_func(child), +fn has_window_func(node: &OperatorNode) -> bool { + match op(node) { + NonASAPOp::SQLWindowFunc { .. } => true, + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => has_window_func(child), _ => false, } } -async fn lower(q: &str) -> QueryExpr { +async fn lower(q: &str) -> Rc { lower_sql(q, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) diff --git a/crates/frontend-sql/tests/data_quality_check/tpch_deequ.rs b/crates/frontend-sql/tests/data_quality_check/tpch_deequ.rs index 679fbc95e..83cff5ec9 100644 --- a/crates/frontend-sql/tests/data_quality_check/tpch_deequ.rs +++ b/crates/frontend-sql/tests/data_quality_check/tpch_deequ.rs @@ -23,13 +23,13 @@ //! which is the same narrowing sidra's own catalog file makes. use asap_frontend_sql::{lower_sql, SqlCatalog, SqlError as LoweringError}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/tpch_deequ_queries.sql"); -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } /// No `time_index` and no `unique_keys`: the checks do not slice by time, and diff --git a/crates/frontend-sql/tests/maintained_population.rs b/crates/frontend-sql/tests/maintained_population.rs index 0019b3b83..cc074e49e 100644 --- a/crates/frontend-sql/tests/maintained_population.rs +++ b/crates/frontend-sql/tests/maintained_population.rs @@ -2,56 +2,56 @@ use asap_aware_mapping::maintained_population::MaintainedPopulationStrategy; use asap_frontend_sql::{lower_sql, SqlCatalog}; use asap_types::{ - post_asap::{ - compile_post_asap_dag, - maintained_population::{MaintainedPopulation, PopulationInput}, - share_common_summary_subtrees, SummaryExpr, ValueOperation, + ir::{ + apply_lifecycle_timings, cse::share_common_subdags, export::compile_post_asap_dag, ASAPOp, + LifecycleAssignment, NonASAPOp, Operator, OperatorNode, TimingMemo, }, - pre_asap::{Column, DataType, QueryExpr, Schema}, + post_asap::maintained_population::{MaintainedPopulation, PopulationInput}, + pre_asap::{DataType, Field, Schema}, types::AccuracyTarget, }; use std::rc::Rc; -async fn aggregate(q: &str) -> Rc { +async fn aggregate(q: &str) -> Rc { let catalog = SqlCatalog::new().with_table( "samples", Schema::new(vec![ - Column::new("latency", DataType::Float64, false), - Column::new("job", DataType::Utf8, false), + Field::plain("latency", DataType::Float64, false), + Field::plain("job", DataType::Utf8, false), ]), ); - let root = lower_sql(q, &catalog, AccuracyTarget::Exact).await.unwrap(); - Rc::new(root) + lower_sql(q, &catalog, AccuracyTarget::Exact).await.unwrap() } -fn population( - mut node: &asap_types::post_asap::SummaryNode, -) -> ( - &Rc, - &MaintainedPopulation, -) { - while let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Project { .. }, - .. - } = &node.expr - { +/// The `MaintainPopulation` node a candidate's evaluation reads, and its spec. +fn population(mut node: &OperatorNode) -> (&Rc, &MaintainedPopulation) { + while let Operator::NonASAP(NonASAPOp::Project { child, .. }) = &node.operator { node = child; } - let SummaryExpr::ValueOperation { child, .. } = &node.expr else { - panic!("readout") + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &node.operator else { + panic!("evaluation") }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &child.expr - else { + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &child.operator else { panic!("state") }; (child, population) } -// Quantile parameters are readout identity, while source, value column and grouping are state identity. +/// Export `plan` the way the planner does: assign the default lifecycle +/// timings, then compile the timed DAG. +fn compile(plan: &Rc) -> Result<(), String> { + let timed = apply_lifecycle_timings( + plan, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .map_err(|e| e.to_string())?; + compile_post_asap_dag(&timed) + .map(|_| ()) + .map_err(|e| e.to_string()) +} + +// Quantile parameters are evaluation identity, while source, value column and grouping are state identity. #[tokio::test] async fn sql_quantiles_share_rows_without_promql_lookback() { let roots = vec![ @@ -59,7 +59,7 @@ async fn sql_quantiles_share_rows_without_promql_lookback() { aggregate("SELECT approx_percentile_cont(latency, 0.99) FROM samples").await, ]; let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_subtrees( + let plans = share_common_subdags( roots .iter() .enumerate() @@ -67,7 +67,7 @@ async fn sql_quantiles_share_rows_without_promql_lookback() { .collect(), ); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + compile(plan).unwrap(); } let (a, spec) = population(&plans[0].1); let (b, _) = population(&plans[1].1); @@ -107,9 +107,9 @@ async fn sql_filters_separate_populations() { assert_ne!(population(&a).1.input, population(&b).1.input); } -// All four scalar readouts can share the same non-null numeric SQL population. +// All four scalar evaluations can share the same non-null numeric SQL population. #[tokio::test] -async fn sql_scalar_readouts_share_membership() { +async fn sql_scalar_evaluations_share_membership() { let mut roots = Vec::new(); for function in [ "median(latency)", @@ -120,7 +120,7 @@ async fn sql_scalar_readouts_share_membership() { roots.push(aggregate(&format!("SELECT {function} FROM samples")).await); } let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_subtrees( + let plans = share_common_subdags( roots .iter() .enumerate() @@ -128,27 +128,29 @@ async fn sql_scalar_readouts_share_membership() { .collect(), ); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + compile(plan).unwrap(); assert!(Rc::ptr_eq(population(&plans[0].1).0, population(plan).0)); } } -// A readout cannot reinterpret a label column as its numeric population. +// A evaluation cannot reinterpret a label column as its numeric population. #[tokio::test] async fn malformed_table_population_fails_validation() { let root = aggregate("SELECT median(latency) FROM samples").await; let rule = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)); let mut candidate = rule.candidate(&root).unwrap(); - let SummaryExpr::ValueOperation { child, .. } = &mut Rc::make_mut(&mut candidate).expr else { + let Operator::NonASAP(NonASAPOp::Project { child, .. }) = + &mut Rc::make_mut(&mut candidate).operator + else { unreachable!() }; - let SummaryExpr::ValueOperation { child, .. } = &mut Rc::make_mut(child).expr else { + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = + &mut Rc::make_mut(child).operator + else { unreachable!() }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &mut Rc::make_mut(child).expr + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = + &mut Rc::make_mut(child).operator else { unreachable!() }; @@ -156,7 +158,7 @@ async fn malformed_table_population_fails_validation() { unreachable!() }; *value_column = 1; - assert!(compile_post_asap_dag(&candidate).is_err()); + assert!(compile(&candidate).is_err()); } // SQL ORDER BY value DESC LIMIT k uses the same maximum-k state contract. @@ -167,7 +169,7 @@ async fn sql_topk_limits_share_maximum_k() { aggregate("SELECT * FROM samples ORDER BY latency DESC LIMIT 5").await, ]; let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_subtrees( + let plans = share_common_subdags( roots .iter() .enumerate() @@ -175,7 +177,7 @@ async fn sql_topk_limits_share_maximum_k() { .collect(), ); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + compile(plan).unwrap(); assert_eq!(population(plan).1.max_k, 5); assert!(Rc::ptr_eq(population(&plans[0].1).0, population(plan).0)); } diff --git a/crates/frontend-sql/tests/netflow/netflow.rs b/crates/frontend-sql/tests/netflow/netflow.rs index 09d742b8b..da680d664 100644 --- a/crates/frontend-sql/tests/netflow/netflow.rs +++ b/crates/frontend-sql/tests/netflow/netflow.rs @@ -4,15 +4,18 @@ //! aggregate over a netflow table, a time predicate, optional grouping, //! optional `ORDER BY`/`LIMIT`, plus the nested aggregate shape. +use std::rc::Rc; + use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::ir::{NonASAPOp, OperatorNode}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::pre_asap::{AggIntent, GroupKeys}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/netflow.sql"); -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn catalog() -> SqlCatalog { @@ -109,11 +112,10 @@ async fn netflow_sql_corpus_lowers_to_expected_intents() { ); for (idx, (query, expected)) in queries.iter().zip(EXPECTED).enumerate() { + // A successful `lower_sql` already derived every node's schema. let qe = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|err| panic!("q{} failed to lower:\n{query}\n{err}", idx + 1)); - qe.output_schema() - .unwrap_or_else(|err| panic!("q{} schema derivation failed: {err}", idx + 1)); assert!( has_scan_predicate(&qe), "q{} should retain the netflow time predicate on the Scan: {qe:?}", @@ -123,7 +125,7 @@ async fn netflow_sql_corpus_lowers_to_expected_intents() { } } -fn assert_expected(qe: &QueryExpr, expected: Expected, case_no: usize) { +fn assert_expected(qe: &Rc, expected: Expected, case_no: usize) { match expected { Expected::Quantile { q, by } => { let (actual_by, measures) = first_aggregate(qe).expect("expected Aggregate"); @@ -190,53 +192,58 @@ impl AggKind { } } -fn first_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { - match qe { - QueryExpr::Aggregate { +/// The operator of a front-end node: a front-end DAG never holds an ASAP node. +fn op(node: &OperatorNode) -> &NonASAPOp { + node.expect_non_asap() +} + +fn first_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { + match op(node) { + NonASAPOp::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_aggregate(child), _ => None, } } -fn has_scan_predicate(qe: &QueryExpr) -> bool { +fn has_scan_predicate(qe: &Rc) -> bool { any_node( qe, - |node| matches!(node, QueryExpr::Scan { predicates, .. } if !predicates.is_empty()), + |node| matches!(op(node), NonASAPOp::Scan { predicates, .. } if !predicates.is_empty()), ) } -fn has_topk(qe: &QueryExpr, k: usize) -> bool { +fn has_topk(qe: &Rc, k: usize) -> bool { any_node(qe, |node| { matches!( - node, - QueryExpr::Aggregate { measures, .. } + op(node), + NonASAPOp::Aggregate { measures, .. } if measures.iter().any(|agg| matches!(agg, AggIntent::TopK { k: actual, .. } if *actual == k)) ) }) } fn aggregate_by_with( - qe: &QueryExpr, + qe: &Rc, by: &'static [usize], pred: impl Fn(&AggIntent) -> bool, ) -> bool { let expected_by = GroupKeys::by(by.to_vec()); let mut found = false; visit(qe, &mut |node| { - if let QueryExpr::Aggregate { + if let NonASAPOp::Aggregate { reduction, measures, .. - } = node + } = op(node) { found |= *reduction.expect_reduce() == expected_by && measures.iter().any(&pred); } @@ -244,78 +251,26 @@ fn aggregate_by_with( found } -fn all_intents(qe: &QueryExpr) -> Vec { +fn all_intents(qe: &Rc) -> Vec { let mut intents = Vec::new(); visit(qe, &mut |node| { - if let QueryExpr::Aggregate { measures, .. } = node { + if let NonASAPOp::Aggregate { measures, .. } = op(node) { intents.extend(measures.iter().cloned()); } }); intents } -fn any_node(qe: &QueryExpr, pred: impl Fn(&QueryExpr) -> bool) -> bool { +fn any_node(qe: &Rc, pred: impl Fn(&OperatorNode) -> bool) -> bool { let mut found = false; visit(qe, &mut |node| found |= pred(node)); found } -fn visit(qe: &QueryExpr, f: &mut impl FnMut(&QueryExpr)) { - f(qe); - match qe { - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => visit(child, f), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { - left: lhs, - right: rhs, - .. - } - | QueryExpr::SetOp { - left: lhs, - right: rhs, - .. - } => { - visit(lhs, f); - visit(rhs, f); - } - QueryExpr::Concat { children, .. } => { - for child in children { - visit(child, f); - } - } - QueryExpr::PromqlVectorFromScalar(child) | QueryExpr::PromqlScalarFromVector(child) => { - visit(child, f) - } - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - // Scalar expression variants (issue #205) aren't relational nodes; - // this visitor only walks the relational tree, so stop here. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} +/// Every reachable operator node, parents before children — including the +/// operators referenced from scalar positions (subqueries). +fn visit(qe: &Rc, f: &mut impl FnMut(&OperatorNode)) { + for node in OperatorNode::reachable(qe) { + f(&node); } } diff --git a/crates/frontend-sql/tests/pearson_corr.rs b/crates/frontend-sql/tests/pearson_corr.rs index ae959560d..8e770f8f2 100644 --- a/crates/frontend-sql/tests/pearson_corr.rs +++ b/crates/frontend-sql/tests/pearson_corr.rs @@ -2,34 +2,38 @@ use std::rc::Rc; use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_types::pre_asap::{AggIntent, Column, DataType, QueryExpr, Schema}; +use asap_types::ir::{ + apply_lifecycle_timings, export::compile_post_asap_dag, LifecycleAssignment, NonASAPOp, + OperatorNode, ScalarExpr, TimingMemo, +}; +use asap_types::pre_asap::{AggIntent, DataType, Field, Schema}; use asap_types::types::AccuracyTarget; fn catalog() -> SqlCatalog { let schema = Schema::new(vec![ - Column::new("x", DataType::Float64, true), - Column::new("y", DataType::Float64, true), - Column::new("g", DataType::Int64, false), + Field::plain("x", DataType::Float64, true), + Field::plain("y", DataType::Float64, true), + Field::plain("g", DataType::Int64, false), ]); SqlCatalog::new() .with_table("a", schema.clone()) .with_table("b", schema) } -async fn lower(sql: &str) -> QueryExpr { +async fn lower(sql: &str) -> Rc { lower_sql(sql, &catalog(), AccuracyTarget::Exact) .await .unwrap() } -fn aggregate(query: &QueryExpr) -> (&[AggIntent], &QueryExpr) { - match query { - QueryExpr::Aggregate { +fn aggregate(query: &OperatorNode) -> (&[AggIntent], &OperatorNode) { + match query.expect_non_asap() { + NonASAPOp::Aggregate { measures, child, .. } => (measures, child), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } => aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } => aggregate(child), other => panic!("expected aggregate, got {other:?}"), } } @@ -46,17 +50,14 @@ async fn corr_materializes_both_arguments() { let query = lower(sql).await; let (measures, child) = aggregate(&query); assert_eq!(measures, &[AggIntent::PearsonCorr { left: 0, right: 1 }]); - let QueryExpr::Project { cols, .. } = child else { + let NonASAPOp::Project { cols, .. } = child.expect_non_asap() else { panic!("derived inputs") }; assert_eq!(cols.len(), 2); assert!(cols .iter() - .any(|col| !matches!(col.expr, QueryExpr::Column(_)))); - assert_eq!( - query.output_schema().unwrap().columns[0].dtype, - DataType::Float64 - ); + .any(|col| !matches!(col.expr, ScalarExpr::Column(_)))); + assert_eq!(query.schema.fields[0].dtype, DataType::Float64); } } @@ -66,11 +67,11 @@ async fn corr_preserves_qualified_join_inputs() { let query = lower("SELECT corr(a.x, b.x) FROM a JOIN b ON a.g = b.g").await; let (measures, child) = aggregate(&query); assert_eq!(measures[0].input_cols(), vec![0, 1]); - let QueryExpr::Project { cols, .. } = child else { + let NonASAPOp::Project { cols, .. } = child.expect_non_asap() else { panic!("paired projection") }; - assert_eq!(cols[0].expr, QueryExpr::Column(0)); - assert_eq!(cols[1].expr, QueryExpr::Column(3)); + assert_eq!(cols[0].expr, ScalarExpr::Column(0)); + assert_eq!(cols[1].expr, ScalarExpr::Column(3)); } // Grouping and sibling reducers cannot drop either correlation argument. @@ -82,15 +83,15 @@ async fn corr_coexists_with_grouping_having_and_other_measures() { .iter() .find(|m| matches!(m, AggIntent::PearsonCorr { .. })) .unwrap(); - let schema = child.output_schema().unwrap(); + let schema = &child.schema; for id in pair.input_cols() { - assert!(id < schema.columns.len()); + assert!(id < schema.fields.len()); } assert!(measures.iter().any(|m| matches!(m, AggIntent::Sum { .. }))); - let output = query.output_schema().unwrap(); - assert_eq!(output.columns[1].name, "r"); - assert_eq!(output.columns[1].dtype, DataType::Float64); - assert!(output.columns[1].nullable); + let output = &query.schema; + assert_eq!(output.fields[1].name, "r"); + assert_eq!(output.fields[1].dtype, DataType::Float64); + assert!(output.fields[1].nullable); } // Repeated inputs reuse their value while retaining two argument positions. @@ -99,8 +100,8 @@ async fn corr_repeated_input_and_serialization() { let query = lower("SELECT corr(x, x) FROM a").await; assert_eq!(aggregate(&query).0[0].input_cols(), vec![0, 0]); let encoded = serde_json::to_string(&query).unwrap(); - let decoded: QueryExpr = serde_json::from_str(&encoded).unwrap(); - assert_eq!(query, decoded); + let decoded: OperatorNode = serde_json::from_str(&encoded).unwrap(); + assert_eq!(*query, decoded); } // Unsupported modifiers and window calls fail instead of silently changing semantics. @@ -128,10 +129,10 @@ async fn corr_filter_is_a_measure_filter() { let query = lower("SELECT corr(x, y) FILTER (WHERE g > 0) FROM a").await; let (measures, _) = aggregate(&query); assert_eq!(measures[0].input_cols(), vec![0, 1]); - fn filters(query: &QueryExpr) -> &[Option] { - match query { - QueryExpr::Aggregate { filters, .. } => filters, - QueryExpr::Project { child, .. } | QueryExpr::Filter { child, .. } => filters(child), + fn filters(query: &OperatorNode) -> &[Option] { + match query.expect_non_asap() { + NonASAPOp::Aggregate { filters, .. } => filters, + NonASAPOp::Project { child, .. } | NonASAPOp::Filter { child, .. } => filters(child), other => panic!("expected aggregate, got {other:?}"), } } @@ -145,12 +146,17 @@ async fn corr_filter_is_a_measure_filter() { // Exact fallback retains the complete typed query and compiles to a post-ASAP DAG. #[tokio::test] async fn corr_survives_exact_plan_compilation() { - let query = Rc::new(lower("SELECT corr(x, y) AS r FROM a").await); - let plan = asap_aware_mapping::replacement::keep_pre_asap(&query).unwrap(); + let query = lower("SELECT corr(x, y) AS r FROM a").await; + let plan = asap_aware_mapping::replacement::retain_exact(&query).unwrap(); assert!(plan.guarantee.as_ref().unwrap().is_exact()); - let asap_types::post_asap::SummaryExpr::KeepPreAsap(retained) = &plan.expr else { - panic!("expected exact fallback"); - }; - assert_eq!(aggregate(retained).0, aggregate(&query).0); - asap_types::post_asap::compile_post_asap_dag(&plan).unwrap(); + // The exact fallback is the query's own operator DAG, no ASAP node added. + assert!(!plan.contains_asap(), "expected exact fallback"); + assert_eq!(aggregate(&plan).0, aggregate(&query).0); + let timed = apply_lifecycle_timings( + &plan, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .unwrap(); + compile_post_asap_dag(&timed).unwrap(); } diff --git a/crates/frontend-sql/tests/sql_lowering.rs b/crates/frontend-sql/tests/sql_lowering.rs index 80f25fbc1..632cb97f8 100644 --- a/crates/frontend-sql/tests/sql_lowering.rs +++ b/crates/frontend-sql/tests/sql_lowering.rs @@ -1,21 +1,30 @@ -//! End-to-end SQL → unresolved → canonical tree lowering tests (positional IR). +//! End-to-end SQL → unresolved → resolved operator DAG lowering tests. //! //! Validates the DataFusion front end: SQL parses + plans, lowers directly to -//! the canonical, unresolved shape (`QueryExpr`, issue #179), and -//! the shared `resolve_root` produces the positional, resolved canonical -//! tree (the same resolver the PromQL path uses). - -use asap_frontend_sql::{lower_sql, lower_sql_dialect, SqlCatalog, SqlError as LoweringError}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +//! the name-based `UnresolvedOp` tree (issue #179), and the shared +//! `resolve_root` produces the positional, canonical `OperatorNode` DAG (the +//! same resolver the PromQL path uses). Every node's schema is derived during +//! resolution, so a successful `lower` already proves schema derivation is +//! total over the tree. + +use asap_types::ir::Predicate; +use std::rc::Rc; + +use asap_frontend_common::{UnresolvedOp, UnresolvedScalar}; +use asap_frontend_sql::{ + lower_sql, lower_sql_dialect, SqlCatalog, SqlError as LoweringError, SqlLowerer, +}; +use asap_types::ir::{ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr}; +use asap_types::pre_asap::schema::{DataType, Field, FieldDataType, Schema}; use asap_types::pre_asap::{ - AggIntent, CompareOpKind, GroupKeys, JoinKind, Predicate, QueryExpr, Reduction, ScalarValue, - Source, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, + AggIntent, CompareOpKind, GroupKeys, JoinKind, Reduction, ScalarValue, Source, + WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, }; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } /// `metrics(ts, service, latency, bytes)` + `hosts(service, region)`. @@ -43,37 +52,21 @@ fn catalog() -> SqlCatalog { ) } -async fn lower(sql: &str) -> QueryExpr { +async fn lower(sql: &str) -> Rc { lower_sql(sql, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) } +/// The operator of a front-end node: a front-end DAG never holds an ASAP node. +fn op(node: &OperatorNode) -> &NonASAPOp { + node.expect_non_asap() +} + #[tokio::test] -async fn planning_subquery_bridge_reuses_canonical_promql_subquery() { - let query = lower( - "SELECT max(value) FROM (\ - SELECT asap_promql_subquery(21600000, 60000) AS value FROM (\ - SELECT sum(bytes) AS value FROM metrics))", - ) - .await; - let QueryExpr::Project { child, .. } = query else { - panic!("expected outer SQL projection"); - }; - let QueryExpr::Aggregate { child, .. } = child.as_ref() else { - panic!("expected outer max aggregate, got {child:?}"); - }; - let QueryExpr::PromqlSubquery { - range, - resolution, - child, - } = child.as_ref() - else { - panic!("expected canonical subquery bridge, got {child:?}"); - }; - assert_eq!(*range, std::time::Duration::from_secs(6 * 60 * 60)); - assert_eq!(*resolution, Some(std::time::Duration::from_secs(60))); - assert!(matches!(child.as_ref(), QueryExpr::Project { .. })); +async fn planning_subquery_bridge_rejects_a_relation_without_vector_conversion() { + let result = lower_sql("SELECT max(value) FROM (SELECT asap_promql_subquery(21600000, 60000) AS value FROM (SELECT sum(bytes) AS value FROM metrics))", &catalog(), AccuracyTarget::Exact).await; + assert!(result.is_err()); } #[tokio::test] @@ -83,12 +76,12 @@ async fn planning_histogram_bridge_reuses_classic_bucket_intent() { SELECT service AS le, sum(bytes) AS value FROM metrics GROUP BY service)", ) .await; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = query + } = op(&query) else { panic!("expected canonical histogram aggregate"); }; @@ -98,7 +91,7 @@ async fn planning_histogram_bridge_reuses_classic_bucket_intent() { measures.as_slice(), [AggIntent::HistogramQuantile { q, le: 0 }] if (*q - 0.95).abs() < 1e-12 )); - assert!(matches!(child.as_ref(), QueryExpr::Project { .. })); + assert!(matches!(op(child), NonASAPOp::Project { .. })); } #[tokio::test] @@ -134,31 +127,31 @@ async fn planning_relation_bridges_reject_ambiguous_shapes() { } /// Find the first `Aggregate` node along the single-child spine. -fn find_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { - match qe { - QueryExpr::Aggregate { +fn find_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { + match op(node) { + NonASAPOp::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => find_aggregate(child), _ => None, } } /// The first `Aggregate` node itself, for tests that need its child. -fn find_aggregate_node(qe: &QueryExpr) -> Option<&QueryExpr> { - match qe { - QueryExpr::Aggregate { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => find_aggregate_node(child), +fn find_aggregate_node(node: &OperatorNode) -> Option<&OperatorNode> { + match op(node) { + NonASAPOp::Aggregate { .. } => Some(node), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => find_aggregate_node(child), _ => None, } } @@ -166,47 +159,47 @@ fn find_aggregate_node(qe: &QueryExpr) -> Option<&QueryExpr> { /// The names of the columns the first `Aggregate`'s reducers read, resolved /// against its child's schema, plus whether that child is a materializing /// `Project` (issue #110). -fn reducer_input_names(qe: &QueryExpr) -> (Vec, bool) { - let QueryExpr::Aggregate { +fn reducer_input_names(node: &OperatorNode) -> (Vec, bool) { + let NonASAPOp::Aggregate { measures, child, .. - } = find_aggregate_node(qe).expect("expected an Aggregate") + } = op(find_aggregate_node(node).expect("expected an Aggregate")) else { unreachable!() }; - let schema = child.output_schema().expect("child schema"); + let schema = &child.schema; let names = measures .iter() .flat_map(|a| a.input_cols()) - .map(|id| schema.columns[id].name.clone()) + .map(|id| schema.fields[id].name.clone()) .collect(); - (names, matches!(**child, QueryExpr::Project { .. })) + (names, matches!(op(child), NonASAPOp::Project { .. })) } /// Find the first `Join` node along the single-child spine. -fn find_join(qe: &QueryExpr) -> Option<&QueryExpr> { - match qe { - QueryExpr::Join { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_join(child), +fn find_join(node: &OperatorNode) -> Option<&OperatorNode> { + match op(node) { + NonASAPOp::Join { .. } => Some(node), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => find_join(child), _ => None, } } /// The first `Filter` node along the single-child spine. -fn find_filter(qe: &QueryExpr) -> Option<&QueryExpr> { - match qe { - QueryExpr::Filter { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_filter(child), +fn find_filter(node: &OperatorNode) -> Option<&OperatorNode> { + match op(node) { + NonASAPOp::Filter { .. } => Some(node), + NonASAPOp::Project { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => find_filter(child), _ => None, } } @@ -215,11 +208,11 @@ fn find_filter(qe: &QueryExpr) -> Option<&QueryExpr> { async fn select_star_with_where_folds_predicate_onto_scan() { // SELECT * elides the projection; WHERE folds onto the Scan predicates. let qe = lower("SELECT * FROM metrics WHERE service = 'api'").await; - let QueryExpr::Scan { + let NonASAPOp::Scan { source, predicates, schema, - } = &qe + } = op(&qe) else { panic!("expected Scan at root, got {qe:?}"); }; @@ -254,17 +247,15 @@ async fn projection_over_aggregate_resolves_output_types_via_output_names() { // onto the canonical Aggregate so the Project resolves real types — not // the Utf8 fallback that an unresolved column would get. let qe = lower("SELECT SUM(bytes), AVG(latency) FROM metrics").await; - let schema = qe - .output_schema() - .expect("root projection schema derivation"); - assert_eq!(schema.columns.len(), 2); + let schema = &qe.schema; + assert_eq!(schema.fields.len(), 2); assert_eq!( - schema.columns[0].dtype, + schema.fields[0].dtype, DataType::Int64, "SUM(bytes:Int64) resolves to Int64, not the Utf8 fallback" ); assert_eq!( - schema.columns[1].dtype, + schema.fields[1].dtype, DataType::Float64, "AVG(latency) resolves to Float64" ); @@ -284,14 +275,14 @@ async fn single_agg_group_by_keeps_key_in_output_schema() { )); // Both the group key and the aggregate resolve in the root projection schema. - let schema = qe.output_schema().expect("root projection schema"); - assert_eq!(schema.columns.len(), 2); + let schema = &qe.schema; + assert_eq!(schema.fields.len(), 2); assert_eq!( - schema.columns[0].dtype, + schema.fields[0].dtype, DataType::Utf8, "service is in the output" ); - assert_eq!(schema.columns[1].dtype, DataType::Int64, "SUM(bytes)"); + assert_eq!(schema.fields[1].dtype, DataType::Int64, "SUM(bytes)"); } #[tokio::test] @@ -315,7 +306,7 @@ async fn count_ranked_topk_is_heavy_hitter() { "count-ranked topk → heavy-hitter TopK, got {measures:?}" ); // The inner child is the explicit Count, grouped by service (col 1). - let QueryExpr::Aggregate { child, .. } = &qe else { + let NonASAPOp::Aggregate { child, .. } = op(&qe) else { panic!("expected outer Aggregate, got {qe:?}"); }; let (inner_by, inner_measures) = find_aggregate(child).expect("expected inner Count aggregate"); @@ -426,7 +417,7 @@ async fn select_distinct_lowers_to_distinct_with_positional_cols() { // (not name-based ColumnRefs). DataFusion's `Distinct::All` dedups on every // column, so `cols` is empty here — but the field type is now `Vec`. let qe = lower("SELECT DISTINCT service FROM metrics").await; - let QueryExpr::Dedup { cols, .. } = &qe else { + let NonASAPOp::Dedup { cols, .. } = op(&qe) else { panic!("expected a Dedup at the root, got {qe:?}"); }; let _: &Vec = cols; // compile-time: positional ids, not ColumnRefs @@ -442,38 +433,39 @@ async fn inner_join_lowers_to_join_over_two_scans() { ) .await; let join = find_join(&qe).expect("expected a Join in the tree"); - let QueryExpr::Join { + let NonASAPOp::Join { kind, left, right, .. - } = join + } = op(join) else { unreachable!("find_join only returns Join"); }; assert_eq!(*kind, JoinKind::Inner); - assert!(matches!(left.as_ref(), QueryExpr::Scan { .. })); - assert!(matches!(right.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(op(left), NonASAPOp::Scan { .. })); + assert!(matches!(op(right), NonASAPOp::Scan { .. })); } /// The two `ColumnId`s an equijoin predicate `Column(l) = Column(r)` binds to, /// returned sorted so the assertion is independent of left/right ordering. -fn join_eq_columns(join: &QueryExpr) -> [usize; 2] { - let QueryExpr::Join { pred, .. } = join else { +fn join_eq_columns(join: &OperatorNode) -> [usize; 2] { + let NonASAPOp::Join { pred, .. } = op(join) else { unreachable!("expected a Join"); }; - let QueryExpr::Compare { + let ScalarExpr::Compare { left, op: CompareOpKind::Eq, right, - } = pred.0.as_ref() + .. + } = &pred.0 else { panic!("expected an equijoin Compare, got {:?}", pred.0); }; match (left.as_ref(), right.as_ref()) { - (QueryExpr::Column(l), QueryExpr::Column(r)) => { + (ScalarExpr::Column(l), ScalarExpr::Column(r)) => { let mut cols = [*l, *r]; cols.sort_unstable(); cols } - other => panic!("expected Column = Column, got {other:?}"), + other => panic!("expected Field = Field, got {other:?}"), } } @@ -567,12 +559,12 @@ async fn qualified_where_over_join_resolves_to_right_side() { ) .await; let filter = find_filter(&qe).expect("expected a Filter over the join"); - let QueryExpr::Filter { pred, .. } = filter else { + let NonASAPOp::Filter { pred, .. } = op(filter) else { unreachable!("find_filter only returns Filter"); }; assert!( - matches!(pred.0.as_ref(), QueryExpr::Compare { left, op: CompareOpKind::Eq, .. } - if matches!(left.as_ref(), QueryExpr::Column(4))), + matches!(&pred.0, ScalarExpr::Compare { left, op: CompareOpKind::Eq, .. } + if matches!(left.as_ref(), ScalarExpr::Column(4))), "hosts.service must bind to concatenated position 4 (not the first `service`), got {:?}", pred.0 ); @@ -636,20 +628,24 @@ async fn aggregate_over_join_binds_against_concatenated_schema() { } // ── Issue #111: IN / EXISTS subquery predicates become semi / anti joins ──── +// +// The front end now leaves them as `UnresolvedScalar::{InSubquery, Exists}` +// filter conjuncts; the shared `canonicalize` pass (run by `resolve_root`) +// lowers each to the semi-/anti-join, so the resolved DAG a test sees is the +// same join shape the front end used to emit directly. /// The first `Join` node's `(kind, predicate, left column count)`. -fn join_parts(qe: &QueryExpr) -> (&JoinKind, &QueryExpr, usize) { - let QueryExpr::Join { +fn join_parts(node: &OperatorNode) -> (&JoinKind, &ScalarExpr, usize) { + let NonASAPOp::Join { kind, pred, left, right: _, - } = find_join(qe).expect("expected a Join") + } = op(find_join(node).expect("expected a Join")) else { unreachable!() }; - let left_len = left.output_schema().expect("left schema").columns.len(); - (kind, pred.0.as_ref(), left_len) + (kind, &pred.0, left.schema.fields.len()) } #[tokio::test] @@ -664,14 +660,15 @@ async fn in_subquery_lowers_to_a_semi_join() { // The predicate resolves against `left ++ right`. Both relations have a // `service` column, so a name-based lookup would bind *both* sides to the // left's — silently making this `service = service`, always true. The key is - // projected under a synthetic name to make that impossible. - let QueryExpr::Compare { left, right, .. } = pred else { + // bound positionally to the subquery's column (right after the left's), + // which makes that impossible. + let ScalarExpr::Compare { left, right, .. } = pred else { panic!("expected a comparison, got {pred:?}"); }; - assert_eq!(**left, QueryExpr::Column(1), "outer service"); + assert_eq!(**left, ScalarExpr::Column(1), "outer service"); assert_eq!( **right, - QueryExpr::Column(left_len), + ScalarExpr::Column(left_len), "the subquery key, not the outer column again" ); } @@ -682,20 +679,14 @@ async fn a_semi_join_outputs_only_the_left_schema() { let qe = lower("SELECT service FROM metrics WHERE service IN (SELECT service FROM hosts)").await; let join = find_join(&qe).expect("expected a Join"); - let names: Vec<_> = join - .output_schema() - .expect("join schema") - .columns - .iter() - .map(|c| c.name.clone()) - .collect(); + let names: Vec<_> = join.schema.fields.iter().map(|c| c.name.clone()).collect(); assert_eq!(names, ["ts", "service", "latency", "bytes"]); } #[tokio::test] async fn a_subquery_key_that_is_an_expression_still_binds() { - // `SELECT bytes + 1 …` has no column name of its own; it is projected under - // the synthetic key rather than becoming an unreferenceable `col_0`. + // `SELECT bytes + 1 …` has no column name of its own; the join key binds + // to it positionally rather than through an unreferenceable `col_0`. let qe = lower("SELECT service FROM metrics WHERE bytes IN (SELECT bytes + 1 FROM metrics)").await; assert_eq!(join_parts(&qe).0, &JoinKind::Semi); @@ -723,13 +714,13 @@ async fn an_ordinary_conjunct_still_folds_onto_the_scan() { AND service IN (SELECT service FROM hosts)", ) .await; - fn scan_has_predicate(qe: &QueryExpr) -> bool { - match qe { - QueryExpr::Scan { predicates, .. } => !predicates.is_empty(), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } => scan_has_predicate(child), - QueryExpr::Join { left, right, .. } => { + fn scan_has_predicate(node: &OperatorNode) -> bool { + match op(node) { + NonASAPOp::Scan { predicates, .. } => !predicates.is_empty(), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } => scan_has_predicate(child), + NonASAPOp::Join { left, right, .. } => { scan_has_predicate(left) || scan_has_predicate(right) } _ => false, @@ -743,16 +734,16 @@ async fn an_ordinary_conjunct_still_folds_onto_the_scan() { } /// Find the first `SQLWindowFunc` node along the single-child spine. -fn find_windowfunc(qe: &QueryExpr) -> Option<&QueryExpr> { - match qe { - QueryExpr::SQLWindowFunc { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_windowfunc(child), +fn find_windowfunc(node: &OperatorNode) -> Option<&OperatorNode> { + match op(node) { + NonASAPOp::SQLWindowFunc { .. } => Some(node), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => find_windowfunc(child), _ => None, } } @@ -766,12 +757,12 @@ async fn window_function_lowers_to_positional_windowfunc() { ) .await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { + let NonASAPOp::SQLWindowFunc { func, partition_by, order_by, .. - } = win + } = op(win) else { unreachable!("find_windowfunc only returns SQLWindowFunc"); }; @@ -780,18 +771,18 @@ async fn window_function_lowers_to_positional_windowfunc() { assert_eq!(order_by.len(), 1); assert_eq!( order_by[0].expr, - QueryExpr::Column(3), + ScalarExpr::Column(3), "ORDER BY bytes → col 3" ); assert!(!order_by[0].ascending, "DESC"); // The window output column is appended to the schema (Int64 for ROW_NUMBER), // and the enclosing projection resolves it (output_name threading). - let schema = qe.output_schema().expect("root schema"); + let schema = &qe.schema; assert!( - schema.columns.iter().any(|c| c.dtype == DataType::Int64), + schema.fields.iter().any(|c| c.dtype == DataType::Int64), "row_number output column present, got {:?}", - schema.columns + schema.fields ); } @@ -799,11 +790,11 @@ async fn window_function_lowers_to_positional_windowfunc() { async fn window_aggregate_lowers_to_windowfunc() { let qe = lower("SELECT service, SUM(bytes) OVER (PARTITION BY service) FROM metrics").await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { func, args, .. } = win else { + let NonASAPOp::SQLWindowFunc { func, args, .. } = op(win) else { unreachable!(); }; assert_eq!(*func, WindowFuncKind::Sum); - assert_eq!(args, &vec![QueryExpr::Column(3)], "SUM(bytes) → arg col 3"); + assert_eq!(args, &vec![ScalarExpr::Column(3)], "SUM(bytes) → arg col 3"); } // ── Window frames (issue #268) ─────────────────────────────────────────────── @@ -827,8 +818,8 @@ async fn window_frame_is_captured_not_dropped() { ) .await; - let frame_of = |qe: &QueryExpr| { - let QueryExpr::SQLWindowFunc { frame, .. } = find_windowfunc(qe).unwrap() else { + let frame_of = |node: &OperatorNode| { + let NonASAPOp::SQLWindowFunc { frame, .. } = op(find_windowfunc(node).unwrap()) else { unreachable!(); }; frame @@ -868,9 +859,9 @@ async fn range_interval_frame_is_preserved() { RANGE BETWEEN INTERVAL '1' HOUR PRECEDING AND CURRENT ROW) FROM metrics", ) .await; - let QueryExpr::SQLWindowFunc { + let NonASAPOp::SQLWindowFunc { frame: Some(frame), .. - } = find_windowfunc(&qe).unwrap() + } = op(find_windowfunc(&qe).unwrap()) else { panic!("expected a window function with a concrete frame"); }; @@ -900,10 +891,10 @@ async fn range_numeric_frames_remain_scalar_offsets() { ) .await; - let start_bound = |qe: &QueryExpr| { - let QueryExpr::SQLWindowFunc { + let start_bound = |node: &OperatorNode| { + let NonASAPOp::SQLWindowFunc { frame: Some(frame), .. - } = find_windowfunc(qe).unwrap() + } = op(find_windowfunc(node).unwrap()) else { panic!("expected a window function with a concrete frame"); }; @@ -938,43 +929,17 @@ async fn groups_frame_is_rejected() { // ── Nested query functions: derived tables / inline views (issue #27) ─────────── -/// Collect every `AggIntent` in the tree, root-to-leaf. -fn all_intents(qe: &QueryExpr) -> Vec { - let mut out = Vec::new(); - fn go(qe: &QueryExpr, out: &mut Vec) { - match qe { - QueryExpr::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - go(child, out); - } - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => go(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { - left: lhs, - right: rhs, - .. - } - | QueryExpr::SetOp { - left: lhs, - right: rhs, - .. - } => { - go(lhs, out); - go(rhs, out); - } - _ => {} - } - } - go(qe, &mut out); - out +/// Collect every `AggIntent` in the DAG, root-to-leaf (every reachable node, +/// including operators referenced from scalar positions). +fn all_intents(root: &Rc) -> Vec { + OperatorNode::reachable(root) + .iter() + .filter_map(|node| match op(node) { + NonASAPOp::Aggregate { measures, .. } => Some(measures.clone()), + _ => None, + }) + .flatten() + .collect() } #[tokio::test] @@ -997,9 +962,9 @@ async fn derived_table_aggregate_over_aggregate_nests() { intents.iter().any(|i| matches!(i, AggIntent::Sum { .. })), "inner SUM survives, got {intents:?}" ); - // The whole nested tree's output schema derives without error (positional - // resolution is total across the derived-table boundary). - assert_eq!(qe.output_schema().unwrap().columns.len(), 1); + // The whole nested tree's output schema derives (positional resolution + // is total across the derived-table boundary). + assert_eq!(qe.schema.fields.len(), 1); } #[tokio::test] @@ -1037,26 +1002,23 @@ async fn filter_over_derived_aggregate_resolves_alias_column() { assert!(all_intents(&qe) .iter() .any(|i| matches!(i, AggIntent::Sum { .. }))); - // Schema derivation is total across the boundary. - let _ = qe.output_schema().expect("nested schema derivation"); + // Schema derivation is total across the boundary: the root carries one. + assert_eq!(qe.schema.fields.len(), 2); } #[tokio::test] -async fn scalar_subquery_in_predicate_is_rejected() { - // A subquery-*valued* expression (`x > (SELECT …)`) needs a subquery node in - // the unresolved expression IR (and a correlated/uncorrelated decision); - // rejected cleanly until that lands. Derived tables in FROM (the common nesting - // shape) ARE supported — see the tests above. - let res = lower_sql( - "SELECT service FROM metrics WHERE bytes > (SELECT AVG(bytes) FROM metrics)", - &catalog(), - AccuracyTarget::Exact, - ) - .await; +async fn scalar_subquery_in_predicate_lowers_through_a_cross_join() { + let qe = + lower("SELECT service FROM metrics WHERE bytes > (SELECT AVG(bytes) FROM metrics)").await; + let filter = find_filter(&qe).unwrap(); + let NonASAPOp::Filter { pred, child } = op(filter) else { + panic!() + }; + assert!(matches!(op(child), NonASAPOp::Scan { .. })); assert!( - res.is_err(), - "scalar subquery in predicate should be rejected" + matches!(&pred.0,ScalarExpr::Compare { right,.. } if matches!(right.as_ref(),ScalarExpr::ScalarSubquery(_))) ); + qe.validate_structure().unwrap(); } #[tokio::test] @@ -1072,15 +1034,15 @@ async fn correlated_exists_lifts_its_correlation_into_the_join() { .await; let (kind, pred, left_len) = join_parts(&qe); assert_eq!(kind, &JoinKind::Semi); - let QueryExpr::Compare { left, right, .. } = pred else { + let ScalarExpr::Compare { left, right, .. } = pred else { panic!("expected the correlation as a comparison, got {pred:?}"); }; assert_eq!( **left, - QueryExpr::Column(left_len), + ScalarExpr::Column(left_len), "h.service (right side)" ); - assert_eq!(**right, QueryExpr::Column(1), "m.service (left side)"); + assert_eq!(**right, ScalarExpr::Column(1), "m.service (left side)"); } #[tokio::test] @@ -1099,24 +1061,62 @@ async fn an_uncorrelated_exists_is_an_unconditional_semi_join() { let qe = lower("SELECT service FROM metrics WHERE EXISTS (SELECT 1 FROM hosts)").await; let (kind, pred, _) = join_parts(&qe); assert_eq!(kind, &JoinKind::Semi); - assert_eq!(*pred, QueryExpr::Literal(ScalarValue::Boolean(true))); + assert_eq!(*pred, ScalarExpr::Literal(ScalarValue::Boolean(true))); } #[tokio::test] -async fn not_in_subquery_is_rejected_rather_than_mislowered_as_an_anti_join() { - // `NOT IN` is *not* an anti-join. Under three-valued logic a single NULL - // among the subquery's rows makes `c NOT IN (…)` UNKNOWN for every `c`, so - // the query returns nothing — while an anti-join returns every unmatched - // left row. Rejecting is the only correct option until the nullability is - // proven, and `NOT EXISTS` is the safe spelling. - let err = lower_sql( - "SELECT service FROM metrics WHERE service NOT IN (SELECT service FROM hosts)", - &catalog(), - AccuracyTarget::Exact, +async fn where_exists_resolves_to_a_semi_join_over_the_subquery() { + // The front end emits `Filter { Exists(s) }`; the resolved DAG is the + // `Semi` join with the subquery (a filtered `hosts` scan) on the right. + let qe = lower( + "SELECT service FROM metrics WHERE EXISTS (SELECT service FROM hosts WHERE region = 'eu')", ) - .await - .expect_err("NOT IN must not lower to an anti-join"); - assert!(format!("{err}").contains("NOT IN"), "got {err}"); + .await; + let NonASAPOp::Project { child, .. } = op(&qe) else { + panic!("expected the SELECT list as a Project, got {qe:?}"); + }; + let NonASAPOp::Join { + kind, + pred, + left, + right, + } = op(child) + else { + panic!("expected the Semi join directly under the Project, got {child:?}"); + }; + assert_eq!(*kind, JoinKind::Semi); + assert_eq!(pred.0, ScalarExpr::Literal(ScalarValue::Boolean(true))); + assert!( + matches!(op(left), NonASAPOp::Scan { .. }), + "left is metrics" + ); + let NonASAPOp::Project { child: scan, .. } = op(right) else { + panic!("expected the subquery's projection on the right, got {right:?}"); + }; + assert!( + matches!(op(scan), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1), + "the subquery's WHERE stays on its own Scan, got {scan:?}" + ); + assert_eq!( + child.schema.fields.len(), + 4, + "a semi join outputs the left's columns alone" + ); +} + +#[tokio::test] +async fn not_in_subquery_is_rejected_rather_than_mislowered_as_an_anti_join() { + let qe = + lower("SELECT service FROM metrics WHERE service NOT IN (SELECT service FROM hosts)").await; + let filter = find_filter(&qe).unwrap(); + let NonASAPOp::Filter { pred, .. } = op(filter) else { + panic!() + }; + assert!(matches!( + pred.0, + ScalarExpr::InSubquery { negated: true, .. } + )); + qe.validate_structure().unwrap(); } #[tokio::test] @@ -1132,6 +1132,177 @@ async fn a_correlated_in_subquery_is_rejected() { assert!(format!("{err}").contains("correlated IN"), "got {err}"); } +// ── Subquery-valued expressions at the `UnresolvedOp` level ───────────────── + +/// `SqlLowerer::lower` output, before `resolve_root`. +async fn lower_unresolved(sql: &str) -> UnresolvedOp { + let catalog = catalog(); + SqlLowerer::new(&catalog) + .lower(sql, &AccuracyTarget::Exact) + .await + .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) +} + +#[tokio::test] +async fn scalar_subquery_in_projection_lowers_to_a_scalar_subquery_item() { + // An uncorrelated `(SELECT max(v) FROM t2)` in the SELECT list is a + // `ScalarSubquery` projection item reading its own lowered plan; the + // cross-join rewrite is `canonicalize`'s job, not the front end's. + let tree = lower_unresolved("SELECT (SELECT max(latency) FROM metrics) FROM hosts").await; + let UnresolvedOp::Project { cols, child, .. } = &tree else { + panic!("expected the SELECT list as a Project, got {tree:?}"); + }; + assert!( + matches!(child.as_ref(), UnresolvedOp::Scan { source: Source::Table { table_ref }, .. } + if table_ref == "hosts"), + "the outer relation stays the projection's child, got {child:?}" + ); + assert_eq!(cols.len(), 1); + let UnresolvedScalar::ScalarSubquery(sub) = &cols[0].expr else { + panic!("expected a ScalarSubquery item, got {:?}", cols[0].expr); + }; + let UnresolvedOp::Project { child: inner, .. } = sub.as_ref() else { + panic!("expected the subquery's own SELECT list, got {sub:?}"); + }; + assert!( + matches!(inner.as_ref(), UnresolvedOp::Aggregate { measures, .. } + if matches!(measures.as_slice(), [AggIntent::Max { .. }])), + "the subquery plan is lowered as a root of its own, got {inner:?}" + ); +} + +#[tokio::test] +async fn exists_and_in_subqueries_lower_to_scalar_filter_conjuncts() { + // The front end no longer builds the semi join itself: `EXISTS` / `IN + // (…)` are `Filter` predicates reading the subquery operator. + let tree = + lower_unresolved("SELECT service FROM metrics WHERE EXISTS (SELECT 1 FROM hosts)").await; + let UnresolvedOp::Project { child, .. } = &tree else { + panic!("expected a Project, got {tree:?}"); + }; + assert!( + matches!(child.as_ref(), UnresolvedOp::Filter { pred, .. } + if matches!(pred.0, UnresolvedScalar::Exists { negated: false, .. })), + "expected Filter {{ Exists }}, got {child:?}" + ); + + let tree = lower_unresolved( + "SELECT service FROM metrics WHERE service IN (SELECT service FROM hosts)", + ) + .await; + let UnresolvedOp::Project { child, .. } = &tree else { + panic!("expected a Project, got {tree:?}"); + }; + assert!( + matches!(child.as_ref(), UnresolvedOp::Filter { pred, .. } + if matches!(pred.0, UnresolvedScalar::InSubquery { negated: false, .. })), + "expected Filter {{ InSubquery }}, got {child:?}" + ); +} + +// ── `SELECT` without `FROM`, unary minus, SQL expression semantics ────────── + +#[tokio::test] +async fn select_without_from_projects_over_one_empty_row() { + // `SELECT 1` has no table: DataFusion's `EmptyRelation` is one empty + // input row, which the SELECT list projects a literal over. + let qe = lower("SELECT 1").await; + let NonASAPOp::Project { cols, child, .. } = op(&qe) else { + panic!("expected Project at root, got {qe:?}"); + }; + assert_eq!(cols.len(), 1); + assert_eq!(cols[0].expr, ScalarExpr::Literal(ScalarValue::Int64(1))); + let NonASAPOp::Values { rows, schema } = op(child) else { + panic!("expected Values under the Project, got {child:?}"); + }; + assert_eq!(rows, &vec![Vec::::new()], "one empty row"); + assert!(schema.fields.is_empty() && schema.closed); + assert_eq!(qe.schema.fields.len(), 1); + assert_eq!(qe.schema.fields[0].dtype, DataType::Int64); +} + +#[tokio::test] +async fn values_lowers_to_one_row_per_values_row() { + let qe = lower("SELECT * FROM (VALUES (1, 'a'), (2, 'b')) AS v(n, s)").await; + let values = OperatorNode::reachable(&qe) + .into_iter() + .find(|n| matches!(op(n), NonASAPOp::Values { .. })) + .expect("expected a Values node"); + let NonASAPOp::Values { rows, schema } = op(&values) else { + unreachable!() + }; + assert_eq!(rows.len(), 2); + assert_eq!( + rows[1], + vec![ + ScalarExpr::Literal(ScalarValue::Int64(2)), + ScalarExpr::Literal(ScalarValue::Utf8("b".into())), + ] + ); + assert_eq!(schema.fields.len(), 2); + assert_eq!(schema.fields[0].dtype, DataType::Int64); + assert_eq!(schema.fields[1].dtype, DataType::Utf8); + assert_eq!( + qe.schema + .fields + .iter() + .map(|f| f.name.as_str()) + .collect::>(), + ["n", "s"] + ); +} + +#[tokio::test] +async fn unary_minus_lowers_to_negative() { + // `-x` over a column is the `Negative` scalar (a negative *literal* is + // folded by DataFusion's planner before lowering). + let qe = lower("SELECT -latency FROM metrics").await; + let NonASAPOp::Project { cols, .. } = op(&qe) else { + panic!("expected Project at root, got {qe:?}"); + }; + assert_eq!( + cols[0].expr, + ScalarExpr::Negative { + expr: Box::new(ScalarExpr::Column(2)), + semantics: ExprSemantics::Sql, + } + ); + assert_eq!(qe.schema.fields[0].dtype, DataType::Float64); +} + +#[tokio::test] +async fn sql_comparisons_and_arithmetic_carry_sql_semantics() { + let qe = lower("SELECT bytes * 8 FROM metrics WHERE latency > 1.5").await; + let NonASAPOp::Project { cols, child, .. } = op(&qe) else { + panic!("expected Project at root, got {qe:?}"); + }; + assert!( + matches!( + &cols[0].expr, + ScalarExpr::Arithmetic { + semantics: ExprSemantics::Sql, + .. + } + ), + "got {:?}", + cols[0].expr + ); + let NonASAPOp::Scan { predicates, .. } = op(child) else { + panic!("expected the WHERE folded onto the Scan, got {child:?}"); + }; + assert!( + matches!( + &predicates[0].0, + ScalarExpr::Compare { + semantics: ExprSemantics::Sql, + .. + } + ), + "got {:?}", + predicates[0].0 + ); +} + // ── Issue #115: Quantile / Cardinality carry their input column ───────────── #[tokio::test] @@ -1281,25 +1452,25 @@ async fn time_bucketing_group_by_lowers_to_a_derived_key() { let qe = lower("SELECT date_trunc('minute', ts) AS m, SUM(bytes) FROM metrics GROUP BY m").await; let node = find_aggregate_node(&qe).expect("expected an Aggregate"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = node + } = op(node) else { unreachable!() }; assert!( - matches!(**child, QueryExpr::Project { .. }), + matches!(op(child), NonASAPOp::Project { .. }), "expected a materializing Project beneath the Aggregate" ); - let schema = child.output_schema().expect("child schema"); + let schema = &child.schema; assert_eq!(reduction, &Reduction::by(vec![0])); assert!( - schema.columns[0].name.contains("date_trunc"), + schema.fields[0].name.contains("date_trunc"), "group key should be the projected bucket, got {:?}", - schema.columns[0].name + schema.fields[0].name ); // The reducer still binds its own column, not the bucket. assert!(matches!( @@ -1317,14 +1488,14 @@ async fn time_bucketing_keeps_the_scan_predicate() { WHERE bytes > 10 GROUP BY m", ) .await; - fn scan_has_predicate(qe: &QueryExpr) -> bool { - match qe { - QueryExpr::Scan { predicates, .. } => !predicates.is_empty(), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => scan_has_predicate(child), + fn scan_has_predicate(node: &OperatorNode) -> bool { + match op(node) { + NonASAPOp::Scan { predicates, .. } => !predicates.is_empty(), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => scan_has_predicate(child), _ => false, } } @@ -1341,13 +1512,13 @@ async fn a_plain_group_by_inserts_no_projection() { "SELECT COUNT(*) FROM metrics", ] { let qe = lower(q).await; - let QueryExpr::Aggregate { child, .. } = - find_aggregate_node(&qe).expect("expected an Aggregate") + let NonASAPOp::Aggregate { child, .. } = + op(find_aggregate_node(&qe).expect("expected an Aggregate")) else { unreachable!() }; assert!( - !matches!(**child, QueryExpr::Project { .. }), + !matches!(op(child), NonASAPOp::Project { .. }), "{q} should not gain a projection" ); } @@ -1356,14 +1527,14 @@ async fn a_plain_group_by_inserts_no_projection() { #[tokio::test] async fn a_shared_expression_is_materialized_once() { let qe = lower("SELECT SUM(bytes * 2), MIN(bytes * 2) FROM metrics").await; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = find_aggregate_node(&qe).expect("expected an Aggregate") + } = op(find_aggregate_node(&qe).expect("expected an Aggregate")) else { unreachable!() }; assert_eq!( - child.output_schema().expect("child schema").columns.len(), + child.schema.fields.len(), 1, "the two reducers should share one derived column" ); @@ -1373,38 +1544,32 @@ async fn a_shared_expression_is_materialized_once() { // ── Issue #118: multi-level grouping expands into one Aggregate per level ─── /// The branches of the first `Concat` along the single-child spine. -fn merge_branches(qe: &QueryExpr) -> &Vec { - fn find(qe: &QueryExpr) -> Option<&Vec> { - match qe { - QueryExpr::Concat { children, .. } => Some(children), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => find(child), +fn merge_branches(node: &OperatorNode) -> &Vec> { + fn find(node: &OperatorNode) -> Option<&Vec>> { + match op(node) { + NonASAPOp::Concat { children, .. } => Some(children), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => find(child), _ => None, } } - find(qe).expect("expected a Concat") + find(node).expect("expected a Concat") } /// `(group keys, column names)` of each merged grouping level. -fn grouping_levels(qe: &QueryExpr) -> Vec<(GroupKeys, Vec)> { - merge_branches(qe) +fn grouping_levels(node: &OperatorNode) -> Vec<(GroupKeys, Vec)> { + merge_branches(node) .iter() .map(|b| { - let QueryExpr::Project { child, .. } = b else { + let NonASAPOp::Project { child, .. } = op(b) else { panic!("expected a Project per level, got {b:?}"); }; - let QueryExpr::Aggregate { reduction, .. } = child.as_ref() else { + let NonASAPOp::Aggregate { reduction, .. } = op(child) else { panic!("expected an Aggregate under the Project, got {child:?}"); }; - let names = b - .output_schema() - .expect("level schema") - .columns - .iter() - .map(|c| c.name.clone()) - .collect(); + let names = b.schema.fields.iter().map(|c| c.name.clone()).collect(); (reduction.expect_reduce().clone(), names) }) .collect() @@ -1463,12 +1628,10 @@ async fn omitted_grouping_keys_become_typed_nulls() { } // The `()` level projects `service` as a Utf8 null, not a Float64 one. - let schema = merge_branches(&qe)[1] - .output_schema() - .expect("level schema"); - assert_eq!(schema.columns[0].name, "service"); + let schema = &merge_branches(&qe)[1].schema; + assert_eq!(schema.fields[0].name, "service"); assert_eq!( - schema.columns[0].dtype, + schema.fields[0].dtype, DataType::Utf8, "the omitted key must keep its declared type" ); @@ -1483,9 +1646,8 @@ async fn grouping_levels_are_union_compatible() { let shapes: Vec<_> = merge_branches(&qe) .iter() .map(|b| { - b.output_schema() - .expect("level schema") - .columns + b.schema + .fields .iter() .map(|c| (c.name.clone(), c.dtype.clone())) .collect::>() @@ -1534,12 +1696,12 @@ async fn multi_level_grouping_composes_with_a_derived_reducer_argument() { // #110's materializing Project sits beneath every level's Aggregate. let qe = lower("SELECT service, SUM(bytes * 8) FROM metrics GROUP BY ROLLUP(service)").await; for b in merge_branches(&qe) { - let QueryExpr::Project { child, .. } = b else { + let NonASAPOp::Project { child, .. } = op(b) else { panic!("expected a Project per level"); }; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = op(child) else { panic!("expected an Aggregate"); }; @@ -1548,7 +1710,7 @@ async fn multi_level_grouping_composes_with_a_derived_reducer_argument() { [AggIntent::Sum { col: Some(_) }] )); assert!( - matches!(**child, QueryExpr::Project { .. }), + matches!(op(child), NonASAPOp::Project { .. }), "the derived-column projection should sit under each level" ); } @@ -1611,7 +1773,7 @@ async fn array_agg_is_deliberately_rejected() { // ── Issue #225: catalog-driven ClickHouse builtins (countIf, generalizing // uniqExact from #221) ─────────────────────────────────────────────────── -async fn lower_clickhouse(sql: &str) -> QueryExpr { +async fn lower_clickhouse(sql: &str) -> Rc { lower_sql_dialect( sql, &catalog(), @@ -1622,20 +1784,20 @@ async fn lower_clickhouse(sql: &str) -> QueryExpr { .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) } -fn temporal_aggregate(qe: &QueryExpr) -> (&AggIntent, std::time::Duration, &QueryExpr) { - match qe { - QueryExpr::Aggregate { +fn temporal_aggregate(node: &OperatorNode) -> (&AggIntent, std::time::Duration, &OperatorNode) { + match op(node) { + NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures, child, .. } => { - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let NonASAPOp::TimeRange { range, child, .. } = op(child) else { panic!("temporal Aggregate must directly wrap TimeRange, got {child:?}"); }; (&measures[0], *range, child) } - QueryExpr::Project { child, .. } | QueryExpr::Filter { child, .. } => { + NonASAPOp::Project { child, .. } | NonASAPOp::Filter { child, .. } => { temporal_aggregate(child) } other => panic!("expected temporal Aggregate, got {other:?}"), @@ -1656,15 +1818,15 @@ async fn explicit_temporal_aggregates_share_promql_intents_and_timerange() { let (intent, range, child) = temporal_aggregate(&qe); assert_eq!(intent, &expected); assert_eq!(range, std::time::Duration::from_secs(300)); - assert!(matches!(child, QueryExpr::Project { child, .. } - if matches!(child.as_ref(), QueryExpr::Scan { predicates, .. } if predicates.len() == 1))); + assert!(matches!(op(child), NonASAPOp::Project { child, .. } + if matches!(op(child), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1))); - let QueryExpr::Project { cols, .. } = &qe else { + let NonASAPOp::Project { cols, .. } = op(&qe) else { panic!("SELECT list must remain a Project, got {qe:?}"); }; - assert!(matches!(cols[0].expr, QueryExpr::Column(2))); + assert!(matches!(cols[0].expr, ScalarExpr::Column(2))); assert_eq!(cols[1].alias.as_deref(), Some("v")); - assert!(matches!(cols[1].expr, QueryExpr::Column(1))); + assert!(matches!(cols[1].expr, ScalarExpr::Column(1))); } } @@ -1825,20 +1987,20 @@ async fn project_filter_and_outer_aggregate_preserve_temporal_child() { ) r WHERE v >= 0", ) .await; - let QueryExpr::Project { child, .. } = &qe else { + let NonASAPOp::Project { child, .. } = op(&qe) else { panic!("expected outer SELECT Project, got {qe:?}"); }; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction: Reduction::Reduce(_), measures, child, .. - } = child.as_ref() + } = op(child) else { panic!("expected outer Aggregate, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let QueryExpr::Filter { child, .. } = child.as_ref() else { + let NonASAPOp::Filter { child, .. } = op(child) else { panic!("derived-table WHERE must remain above the inner query, got {child:?}"); }; let (intent, range, _) = temporal_aggregate(child); @@ -2002,13 +2164,13 @@ async fn lag_in_frame_lowers_to_its_own_kind_not_lag() { ) .await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { func, args, .. } = win else { + let NonASAPOp::SQLWindowFunc { func, args, .. } = op(win) else { unreachable!(); }; assert_eq!(*func, WindowFuncKind::LagInFrame); assert_eq!( args, - &vec![QueryExpr::Column(3)], + &vec![ScalarExpr::Column(3)], "lagInFrame(bytes) → arg col 3" ); } @@ -2021,7 +2183,7 @@ async fn lead_in_frame_lowers_to_its_own_kind_not_lead() { ) .await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { func, .. } = win else { + let NonASAPOp::SQLWindowFunc { func, .. } = op(win) else { unreachable!(); }; assert_eq!(*func, WindowFuncKind::LeadInFrame); @@ -2034,13 +2196,13 @@ async fn lead_in_frame_lowers_to_its_own_kind_not_lead() { async fn now_in_predicate_lowers_to_current_timestamp() { // SELECT * folds WHERE onto Scan.predicates (no explicit Filter node). let qe = lower("SELECT * FROM metrics WHERE ts < NOW()").await; - let QueryExpr::Scan { predicates, .. } = &qe else { + let NonASAPOp::Scan { predicates, .. } = op(&qe) else { panic!("expected Scan at root, got {qe:?}"); }; assert_eq!(predicates.len(), 1); assert!( - matches!(predicates[0].0.as_ref(), QueryExpr::Compare { right, .. } - if matches!(right.as_ref(), QueryExpr::CurrentTimestamp)), + matches!(&predicates[0].0, ScalarExpr::Compare { right, .. } + if matches!(right.as_ref(), ScalarExpr::Cast { expr, to: DataType::Timestamp, .. } if matches!(expr.as_ref(), ScalarExpr::CurrentTimestamp))), "NOW() must lower to CurrentTimestamp, got {:?}", predicates[0].0 ); @@ -2051,13 +2213,13 @@ async fn now_in_predicate_lowers_to_current_timestamp() { #[tokio::test] async fn clickhouse_now_in_predicate_lowers_to_current_timestamp() { let qe = lower_clickhouse("SELECT * FROM metrics WHERE ts < now()").await; - let QueryExpr::Scan { predicates, .. } = &qe else { + let NonASAPOp::Scan { predicates, .. } = op(&qe) else { panic!("expected Scan at root, got {qe:?}"); }; assert_eq!(predicates.len(), 1); assert!( - matches!(predicates[0].0.as_ref(), QueryExpr::Compare { right, .. } - if matches!(right.as_ref(), QueryExpr::CurrentTimestamp)), + matches!(&predicates[0].0, ScalarExpr::Compare { right, .. } + if matches!(right.as_ref(), ScalarExpr::Cast { expr, to: DataType::Timestamp, .. } if matches!(expr.as_ref(), ScalarExpr::CurrentTimestamp))), "now() must lower to CurrentTimestamp, got {:?}", predicates[0].0 ); @@ -2066,12 +2228,16 @@ async fn clickhouse_now_in_predicate_lowers_to_current_timestamp() { #[tokio::test] async fn current_timestamp_lowers_to_typed_current_timestamp_leaf() { let qe = lower("SELECT CURRENT_TIMESTAMP FROM metrics").await; - let QueryExpr::Project { cols, .. } = &qe else { + let NonASAPOp::Project { cols, child, .. } = op(&qe) else { panic!("expected Project at root, got {qe:?}"); }; - assert!(matches!(&cols[0].expr, QueryExpr::CurrentTimestamp)); - let schema = cols[0].expr.output_schema().expect("timestamp schema"); - assert_eq!(schema.columns[0].dtype, DataType::Timestamp); + assert!(matches!(&cols[0].expr, ScalarExpr::CurrentTimestamp)); + let (dtype, _) = cols[0] + .expr + .scalar_type(&child.schema) + .expect("timestamp type"); + assert_eq!(dtype, DataType::Timestamp); + assert_eq!(qe.schema.fields[0].dtype, DataType::Timestamp); } // A `count` over a non-null input is a plain row count; over a nullable @@ -2082,8 +2248,8 @@ async fn count_null_semantics_become_a_measure_filter() { let catalog = SqlCatalog::new().with_table( "samples", Schema::new(vec![ - Column::new("nullable_value", DataType::Float64, true), - Column::new("value", DataType::Float64, false), + Field::plain("nullable_value", DataType::Float64, true), + Field::plain("value", DataType::Float64, false), ]), ); for sql in [ @@ -2114,10 +2280,7 @@ async fn count_null_semantics_become_a_measure_filter() { aggregate_filters(&qe) ); }; - assert!( - matches!(cond.as_ref(), QueryExpr::IsNotNull(_)), - "{sql}: {cond:?}" - ); + assert!(matches!(cond, ScalarExpr::IsNotNull(_)), "{sql}: {cond:?}"); } // Only the second measure is filtered. let qe = lower_sql( @@ -2164,7 +2327,7 @@ async fn grouped_map_column_preserves_map_type() { ) .await .unwrap(); - assert_eq!(query.output_schema().unwrap().columns[0].dtype, map); + assert_eq!(query.schema.fields[0].dtype, map); } #[tokio::test] @@ -2172,9 +2335,9 @@ async fn clickhouse_modulo_uses_native_arithmetic_types_and_nullability() { let catalog = SqlCatalog::new().with_table( "numbers", Schema::new(vec![ - Column::new("i", DataType::Int64, false), - Column::new("n", DataType::Int64, true), - Column::new("f", DataType::Float64, false), + Field::plain("i", DataType::Int64, false), + Field::plain("n", DataType::Int64, true), + Field::plain("f", DataType::Float64, false), ]), ); for (call, native) in [ @@ -2202,10 +2365,7 @@ async fn clickhouse_modulo_uses_native_arithmetic_types_and_nullability() { .await .unwrap(); assert_eq!(function, operator, "{call}"); - assert_eq!( - function.output_schema().unwrap(), - operator.output_schema().unwrap() - ); + assert_eq!(function.schema, operator.schema); } let nullable = lower_sql_dialect( "SELECT modulo(n, 3) AS value FROM numbers", @@ -2215,10 +2375,10 @@ async fn clickhouse_modulo_uses_native_arithmetic_types_and_nullability() { ) .await .unwrap() - .output_schema() - .unwrap(); - assert_eq!(nullable.columns[0].dtype, DataType::Int64); - assert!(nullable.columns[0].nullable); + .schema + .clone(); + assert_eq!(nullable.fields[0].dtype, DataType::Int64); + assert!(nullable.fields[0].nullable); } #[tokio::test] @@ -2226,10 +2386,10 @@ async fn original_o11y_map_queries_lower_with_typed_results() { let catalog = SqlCatalog::new().with_table( "raw_samples", Schema::new(vec![ - Column::new("metric", DataType::Utf8, false), - Column::new("ts_ms", DataType::Int64, false), - Column::new("value", DataType::Float64, false), - Column::new( + Field::plain("metric", DataType::Utf8, false), + Field::plain("ts_ms", DataType::Int64, false), + Field::plain("value", DataType::Float64, false), + Field::plain( "labels", DataType::Map { key: Box::new(DataType::Utf8), @@ -2255,12 +2415,12 @@ async fn original_o11y_map_queries_lower_with_typed_results() { ) .await .unwrap_or_else(|e| panic!("{sql}: {e}")); - let schema = query.output_schema().unwrap(); + let schema = &query.schema; assert!( schema - .columns + .fields .iter() - .any(|column| matches!(column.dtype, DataType::Map { .. })), + .any(|column| matches!(column.dtype, FieldDataType::Plain(DataType::Map { .. }))), "{schema:?}" ); } @@ -2287,7 +2447,7 @@ async fn clickhouse_modulo_preserves_projection_names_and_outer_references() { ) .await .unwrap(); - assert_eq!(query.output_schema().unwrap().columns[0].name, name); + assert_eq!(query.schema.fields[0].name, name); } } @@ -2296,7 +2456,7 @@ async fn clickhouse_map_access_keeps_generated_names_and_rejects_variant_coercio let catalog = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new( + Field::plain( "labels", DataType::Map { key: Box::new(DataType::Utf8), @@ -2305,8 +2465,8 @@ async fn clickhouse_map_access_keeps_generated_names_and_rejects_variant_coercio }, false, ), - Column::new("integer", DataType::Int64, false), - Column::new("floating", DataType::Float64, false), + Field::plain("integer", DataType::Int64, false), + Field::plain("floating", DataType::Float64, false), ]), ); let query = lower_sql_dialect( @@ -2317,10 +2477,10 @@ async fn clickhouse_map_access_keeps_generated_names_and_rejects_variant_coercio ) .await .unwrap(); - let output = query.output_schema().unwrap(); - assert_eq!(output.columns[0].name, "arrayElement(labels, 'job')"); - assert_eq!(output.columns[0].dtype, DataType::Utf8); - assert!(!output.columns[0].nullable); + let output = &query.schema; + assert_eq!(output.fields[0].name, "arrayElement(labels, 'job')"); + assert_eq!(output.fields[0].dtype, DataType::Utf8); + assert!(!output.fields[0].nullable); assert!(lower_sql_dialect( "SELECT map()['a'] FROM t", &catalog, @@ -2344,9 +2504,9 @@ async fn arg_selector_result_schema_tracks_selected_argument() { let catalog = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new("v", DataType::Float64, false), - Column::new("text", DataType::Utf8, true), - Column::new("ts", DataType::Int64, true), + Field::plain("v", DataType::Float64, false), + Field::plain("text", DataType::Utf8, true), + Field::plain("ts", DataType::Int64, true), ]), ); for (sql, dtype, nullable) in [ @@ -2369,9 +2529,9 @@ async fn arg_selector_result_schema_tracks_selected_argument() { ) .await .unwrap(); - let schema = query.output_schema().unwrap(); - assert_eq!(schema.columns[0].dtype, dtype); - assert_eq!(schema.columns[0].nullable, nullable); + let schema = &query.schema; + assert_eq!(schema.fields[0].dtype, dtype); + assert_eq!(schema.fields[0].nullable, nullable); } } @@ -2380,14 +2540,14 @@ async fn clickhouse_list_element_uses_canonical_typed_access() { let catalog = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new( + Field::plain( "samples", DataType::List { - element: Box::new(Column::new("item", DataType::Int64, false)), + element: Box::new(Field::new("item", DataType::Int64, false)), }, false, ), - Column::new("index", DataType::Int64, true), + Field::plain("index", DataType::Int64, true), ]), ); for (sql, nullable) in [ @@ -2403,9 +2563,9 @@ async fn clickhouse_list_element_uses_canonical_typed_access() { ) .await .unwrap(); - let output = query.output_schema().unwrap(); - assert_eq!(output.columns[0].dtype, DataType::Int64); - assert_eq!(output.columns[0].nullable, nullable); + let output = &query.schema; + assert_eq!(output.fields[0].dtype, DataType::Int64); + assert_eq!(output.fields[0].nullable, nullable); let serialized = serde_json::to_string(&query).unwrap(); assert!(serialized.contains("asap_element_access"), "{serialized}"); } @@ -2429,17 +2589,17 @@ async fn clickhouse_tuple_element_preserves_declared_field_metadata() { let catalog = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new( + Field::plain( "sample", DataType::Struct { fields: vec![ - Column::new("time", DataType::Int64, false), - Column::new("value", DataType::Float64, true), + Field::new("time", DataType::Int64, false), + Field::new("value", DataType::Float64, true), ], }, false, ), - Column::new("index", DataType::Int64, false), + Field::plain("index", DataType::Int64, false), ]), ); for (sql, dtype, nullable) in [ @@ -2462,9 +2622,9 @@ async fn clickhouse_tuple_element_preserves_declared_field_metadata() { ) .await .unwrap(); - let output = query.output_schema().unwrap(); - assert_eq!(output.columns[0].dtype, dtype); - assert_eq!(output.columns[0].nullable, nullable); + let output = &query.schema; + assert_eq!(output.fields[0].dtype, dtype); + assert_eq!(output.fields[0].nullable, nullable); assert!(serde_json::to_string(&query) .unwrap() .contains("asap_struct_field")); @@ -2489,10 +2649,10 @@ async fn clickhouse_tuple_element_preserves_declared_field_metadata() { #[tokio::test] async fn corr_result_is_nullable_float() { let query = lower("SELECT corr(latency, bytes) AS correlation FROM metrics").await; - let schema = query.output_schema().unwrap(); - assert_eq!(schema.columns[0].name, "correlation"); - assert_eq!(schema.columns[0].dtype, DataType::Float64); - assert!(schema.columns[0].nullable); + let schema = &query.schema; + assert_eq!(schema.fields[0].name, "correlation"); + assert_eq!(schema.fields[0].dtype, DataType::Float64); + assert!(schema.fields[0].nullable); } // A multi-column DISTINCT counts tuples; one column stays the single-column @@ -2502,8 +2662,8 @@ async fn composite_distinct_counts_tuples() { let cat = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new("a", DataType::Int64, false), - Column::new("b", DataType::Int64, false), + Field::plain("a", DataType::Int64, false), + Field::plain("b", DataType::Int64, false), ]), ); let composite = lower_sql( @@ -2513,8 +2673,8 @@ async fn composite_distinct_counts_tuples() { ) .await .unwrap(); - let QueryExpr::Aggregate { measures, .. } = - find_aggregate_node(&composite).expect("expected an Aggregate") + let NonASAPOp::Aggregate { measures, .. } = + op(find_aggregate_node(&composite).expect("expected an Aggregate")) else { unreachable!() }; @@ -2530,8 +2690,8 @@ async fn composite_distinct_counts_tuples() { ) .await .unwrap(); - let QueryExpr::Aggregate { measures, .. } = - find_aggregate_node(&single).expect("expected an Aggregate") + let NonASAPOp::Aggregate { measures, .. } = + op(find_aggregate_node(&single).expect("expected an Aggregate")) else { unreachable!() }; @@ -2548,8 +2708,8 @@ async fn composite_distinct_rejects_expression_arguments() { let cat = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new("a", DataType::Int64, false), - Column::new("b", DataType::Int64, false), + Field::plain("a", DataType::Int64, false), + Field::plain("b", DataType::Int64, false), ]), ); let error = lower_sql( @@ -2572,8 +2732,8 @@ async fn distinct_with_derived_sibling() { let catalog = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new("a", DataType::Int64, false), - Column::new("b", DataType::Int64, false), + Field::plain("a", DataType::Int64, false), + Field::plain("b", DataType::Int64, false), ]), ); for sql in [ @@ -2589,8 +2749,10 @@ async fn distinct_with_derived_sibling() { // ── Issue #466: per-measure FILTER predicates ───────────────────────────────── /// The first `Aggregate`'s `filters`, positional against its child. -fn aggregate_filters(qe: &QueryExpr) -> &[Option] { - let Some(QueryExpr::Aggregate { filters, .. }) = find_aggregate_node(qe) else { +fn aggregate_filters(qe: &OperatorNode) -> &[Option] { + let Some(NonASAPOp::Aggregate { filters, .. }) = + find_aggregate_node(qe).map(|n| n.expect_non_asap()) + else { panic!("expected an Aggregate, got {qe:?}"); }; filters @@ -2620,15 +2782,17 @@ async fn conditional_count_lowers_to_a_filtered_measure() { panic!("expected [Some, None], got {:?}", aggregate_filters(&qe)); }; assert!( - matches!(cond.as_ref(), QueryExpr::Compare { left, op: CompareOpKind::Gt, .. } - if matches!(left.as_ref(), QueryExpr::Column(2))), + matches!(cond, ScalarExpr::Compare { left, op: CompareOpKind::Gt, .. } + if matches!(left.as_ref(), ScalarExpr::Column(2))), "latency > 1.0 against the scan, got {cond:?}" ); - let Some(QueryExpr::Aggregate { child, .. }) = find_aggregate_node(&qe) else { + let Some(NonASAPOp::Aggregate { child, .. }) = + find_aggregate_node(&qe).map(|n| n.expect_non_asap()) + else { unreachable!() }; assert!( - matches!(child.as_ref(), QueryExpr::Scan { .. }), + matches!(child.expect_non_asap(), NonASAPOp::Scan { .. }), "{child:?}" ); } @@ -2642,9 +2806,9 @@ async fn filter_clause_lowers_to_a_measure_filter() { panic!("expected [Some, None], got {:?}", aggregate_filters(&qe)); }; assert!( - matches!(cond.as_ref(), QueryExpr::Compare { left, op: CompareOpKind::Eq, right } - if matches!(left.as_ref(), QueryExpr::Column(1)) - && matches!(right.as_ref(), QueryExpr::Literal(ScalarValue::Utf8(s)) if s == "a")), + matches!(cond, ScalarExpr::Compare { left, op: CompareOpKind::Eq, right, .. } + if matches!(left.as_ref(), ScalarExpr::Column(1)) + && matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Utf8(s)) if s == "a")), "{cond:?}" ); } @@ -2658,7 +2822,7 @@ async fn count_of_a_nullable_expression_filters_nulls() { let [Some(Predicate(cond))] = aggregate_filters(&qe) else { panic!("expected [Some], got {:?}", aggregate_filters(&qe)); }; - assert!(matches!(cond.as_ref(), QueryExpr::IsNotNull(_)), "{cond:?}"); + assert!(matches!(cond, ScalarExpr::IsNotNull(_)), "{cond:?}"); assert!( matches!( find_aggregate(&qe).unwrap().1.as_slice(), @@ -2673,23 +2837,25 @@ async fn count_of_a_nullable_expression_filters_nulls() { #[tokio::test] async fn measure_filter_columns_survive_a_derived_column_projection() { let qe = lower("SELECT sum(bytes * 2) FILTER (WHERE latency > 1.0) FROM metrics").await; - let Some(QueryExpr::Aggregate { child, .. }) = find_aggregate_node(&qe) else { + let Some(NonASAPOp::Aggregate { child, .. }) = + find_aggregate_node(&qe).map(|n| n.expect_non_asap()) + else { unreachable!() }; assert!( - matches!(child.as_ref(), QueryExpr::Project { .. }), + matches!(child.expect_non_asap(), NonASAPOp::Project { .. }), "{child:?}" ); let [Some(Predicate(cond))] = aggregate_filters(&qe) else { panic!("expected [Some], got {:?}", aggregate_filters(&qe)); }; - let QueryExpr::Compare { left, .. } = cond.as_ref() else { + let ScalarExpr::Compare { left, .. } = cond else { panic!("{cond:?}"); }; - let QueryExpr::Column(id) = left.as_ref() else { + let ScalarExpr::Column(id) = left.as_ref() else { panic!("{left:?}"); }; - assert_eq!(child.output_schema().unwrap().columns[*id].name, "latency"); + assert_eq!(child.schema.fields[*id].name, "latency"); } // `GROUP BY ROLLUP` fans one measure list out into one `Aggregate` per level; diff --git a/crates/frontend-sql/tests/temporal_types.rs b/crates/frontend-sql/tests/temporal_types.rs index bdb53d246..7e12c6f33 100644 --- a/crates/frontend-sql/tests/temporal_types.rs +++ b/crates/frontend-sql/tests/temporal_types.rs @@ -1,12 +1,12 @@ use asap_frontend_sql::{lower_sql, SqlCatalog}; use asap_types::{ - pre_asap::schema::{Column, DataType, Schema}, + pre_asap::schema::{DataType, Field, Schema}, types::AccuracyTarget, }; fn catalog() -> SqlCatalog { SqlCatalog::new().with_table( "t", - Schema::new(vec![Column::new("d", DataType::Date, false)]), + Schema::new(vec![Field::plain("d", DataType::Date, false)]), ) } // Unsupported fixed-duration results fail lowering instead of acquiring a float schema. @@ -14,7 +14,7 @@ fn catalog() -> SqlCatalog { async fn temporal_subtraction_rejects_unrepresentable_duration() { for dtype in [DataType::Date, DataType::Timestamp] { let catalog = - SqlCatalog::new().with_table("t", Schema::new(vec![Column::new("d", dtype, false)])); + SqlCatalog::new().with_table("t", Schema::new(vec![Field::plain("d", dtype, false)])); let error = lower_sql( "SELECT d - d AS elapsed FROM t", &catalog, @@ -38,10 +38,7 @@ async fn date_shifts_keep_their_type() { let node = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap(); - assert_eq!( - node.output_schema().unwrap().columns[0].dtype, - DataType::Date - ); + assert_eq!(node.schema.fields[0].dtype, DataType::Date); } } // Interval literals and explicit interval casts must both cross the Arrow bridge. @@ -54,10 +51,7 @@ async fn interval_cast_lowers_like_interval_literal() { let node = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap(); - assert_eq!( - node.output_schema().unwrap().columns[0].dtype, - DataType::Interval - ); + assert_eq!(node.schema.fields[0].dtype, DataType::Interval); } } @@ -90,11 +84,7 @@ async fn negative_intervals_keep_their_type() { let node = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap(); - assert_eq!( - node.output_schema().unwrap().columns[0].dtype, - DataType::Interval, - "{query}" - ); + assert_eq!(node.schema.fields[0].dtype, DataType::Interval, "{query}"); } } @@ -108,9 +98,6 @@ async fn sql_date_literals_keep_their_type() { let node = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap(); - assert_eq!( - node.output_schema().unwrap().columns[0].dtype, - DataType::Date - ); + assert_eq!(node.schema.fields[0].dtype, DataType::Date); } } diff --git a/crates/integration-tests/Cargo.toml b/crates/integration-tests/Cargo.toml index 5a5de9902..afa7559b4 100644 --- a/crates/integration-tests/Cargo.toml +++ b/crates/integration-tests/Cargo.toml @@ -10,6 +10,7 @@ asap-frontend-sql = { path = "../frontend-sql" } asap-aware-mapping = { path = "../asap-aware-mapping" } [dev-dependencies] +asap-planner = { path = "../planner" } asap_sketchlib = { workspace = true } serde_json = "1" tokio = { version = "1", features = ["rt", "macros", "rt-multi-thread"] } diff --git a/crates/integration-tests/src/lib.rs b/crates/integration-tests/src/lib.rs index 99fb24226..145900585 100644 --- a/crates/integration-tests/src/lib.rs +++ b/crates/integration-tests/src/lib.rs @@ -12,21 +12,34 @@ //! here derives or computes expected outputs. pub mod fixtures { - use asap_frontend_promql::lower_promql_workload; - use asap_types::pre_asap::schema::{Column, DataType, Schema}; - use asap_types::pre_asap::QueryExpr; + + use asap_types::ir::OperatorNode; + use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, }; + use std::rc::Rc; /// Lower one query through the plan-ready workload API using the test /// suite's declared one-second source cadence. pub fn lower_promql( query: &str, accuracy: AccuracyTarget, - ) -> Result { + ) -> Result, asap_frontend_promql::PromqlError> { + match lower_promql_root(query, accuracy)? { + asap_types::ir::QueryRoot::Operator(node) => Ok(node), + _ => Err(asap_frontend_promql::PromqlError::UnsupportedFeature( + "expected vector root".into(), + )), + } + } + + pub fn lower_promql_root( + query: &str, + accuracy: AccuracyTarget, + ) -> Result { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -51,20 +64,20 @@ pub mod fixtures { ..Default::default() }), }; - let mut lowered = lower_promql_workload(&workload, 0)?; + let mut lowered = asap_frontend_promql::lower_promql_query_workload(&workload, 0)?; Ok(lowered.remove(0)) } - pub fn ts_col() -> Column { - Column::new("ts", DataType::Timestamp, false) + pub fn ts_col() -> Field { + Field::plain("ts", DataType::Timestamp, false) } - pub fn value_col() -> Column { - Column::new("value", DataType::Float64, false) + pub fn value_col() -> Field { + Field::plain("value", DataType::Float64, false) } - pub fn label_col(name: &str) -> Column { - Column::new(name, DataType::Utf8, true) + pub fn label_col(name: &str) -> Field { + Field::plain(name, DataType::Utf8, true) } /// Canonical PromQL leaf schema: `(ts: Timestamp, value: Float64)` plus @@ -74,7 +87,7 @@ pub mod fixtures { let mut cols = vec![ts_col(), value_col()]; cols.extend(labels.iter().map(|n| label_col(n))); Schema { - columns: cols, + fields: cols, time_index: Some(0), unique_keys: vec![], // Schemaless PromQL leaf: open (the metric's full label set is @@ -83,3 +96,26 @@ pub mod fixtures { } } } + +/// Timing and export helpers for post-ASAP plans. +pub mod post_asap { + use asap_types::ir::export::{compile_post_asap_dag, PostAsapDag}; + use asap_types::ir::{apply_lifecycle_timings, LifecycleAssignment, OperatorNode, TimingMemo}; + use std::rc::Rc; + + /// Time `root` under the default (every summary maintained) lifecycle + /// assignment. Returns the timed copy; read `node.timing` on it. + pub fn timed(root: &Rc) -> Rc { + apply_lifecycle_timings( + root, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .expect("default lifecycle timing failed") + } + + /// Time `root` (default assignment), then export the wire-6 DAG. + pub fn post_asap_dag(root: &Rc) -> PostAsapDag { + compile_post_asap_dag(&timed(root)).expect("post-ASAP DAG export failed") + } +} diff --git a/crates/integration-tests/tests/aggregate.rs b/crates/integration-tests/tests/aggregate.rs index 051eeb6b7..9bf05ad4b 100644 --- a/crates/integration-tests/tests/aggregate.rs +++ b/crates/integration-tests/tests/aggregate.rs @@ -1,47 +1,53 @@ -//! `QueryExpr::Aggregate` — cross-series aggregation tests. +//! `NonASAPOp::Aggregate` — cross-series aggregation tests. //! //! topk/bottomk are omitted — dispatch is deferred. //! -//! Cross-series aggregates lower to a single `Aggregate` node with no -//! `TimeRange` child (range functions use `TimeRange` — see `time_range.rs`). -//! Group keys land on `Aggregate.by` as positional `ColumnId`s. -//! Single-stat PromQL aggregates always get `output_names: [""]` (no alias) -//! and `having: None`. +//! Cross-series aggregates lower to a single `Aggregate` node over the +//! instant-selector `TimeRange` (range functions use a `Range` selector — +//! see `time_range.rs`). Group keys land on `Aggregate.reduction` as +//! positional `ColumnId`s. Single-stat PromQL aggregates always get +//! `output_names: [""]` (no alias) and `having: None`. use std::rc::Rc; use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction, Source}; +use asap_types::ir::{NonASAPOp, OperatorNode, TimeRangeKind}; +use asap_types::pre_asap::{AggIntent, Reduction, Source}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::Scan { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::non_asap_node(op).expect("fixture node derives its schema") +} + +fn scan(metric: &str, labels: &[&str]) -> Rc { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: metric_schema(labels), - } + }) } -fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { +fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: node(NonASAPOp::TimeRange { range: Duration::from_secs(1), - child: Rc::new(child), + kind: TimeRangeKind::Instant, + child, }), - } + }) } // #5 — sum with no group keys @@ -169,7 +175,7 @@ fn q_stdvar_no_group() { ); } -// #10 — cross-series quantile; no TimeRange node (no range window) +// #10 — cross-series quantile; instant selector, no range window #[test] fn q10_quantile_cross_series() { assert_eq!( diff --git a/crates/integration-tests/tests/binary_op.rs b/crates/integration-tests/tests/binary_op.rs index 35f1c632d..481d36a3c 100644 --- a/crates/integration-tests/tests/binary_op.rs +++ b/crates/integration-tests/tests/binary_op.rs @@ -1,92 +1,120 @@ -//! `QueryExpr::BinaryOp` — arithmetic, comparison, and vector-match tests. +//! `NonASAPOp::BinaryOp` — arithmetic, comparison, and vector-match tests. //! //! Each side of a `BinaryOp` is bound independently by the SchemaResolver, so each //! gets its own scan schema derived from the labels it references. -//! `VectorMatch` labels (e.g. `on(job)`) are carried as strings on the node -//! and are NOT resolved to column ids — the SchemaResolver does not see them. +//! `VectorMatch` labels (e.g. `on(job)`) are carried as strings on the +//! operator and are NOT resolved to column ids — the SchemaResolver does not +//! see them. use std::rc::Rc; use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; +use asap_types::ir::{BinaryOperator, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind}; use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, GroupSide, QueryExpr, Reduction, - Source, VectorGrouping, VectorMatch, VectorMatchKind, + AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, GroupSide, Reduction, Source, + VectorGrouping, VectorMatch, VectorMatchKind, }; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::TimeRange { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::non_asap_node(op).expect("fixture node derives its schema") +} + +fn scan(metric: &str, labels: &[&str]) -> Rc { + node(NonASAPOp::TimeRange { range: Duration::from_secs(1), - child: Rc::new(source_scan(metric, labels)), - } + kind: TimeRangeKind::Instant, + child: source_scan(metric, labels), + }) } -fn source_scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::Scan { +fn source_scan(metric: &str, labels: &[&str]) -> Rc { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: metric_schema(labels), - } + }) } -fn rate_agg(metric: &str) -> QueryExpr { - QueryExpr::Aggregate { +fn rate_agg(metric: &str) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Rate], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: node(NonASAPOp::TimeRange { range: Duration::from_secs(300), - child: Rc::new(source_scan(metric, &[])), + kind: TimeRangeKind::Range, + child: source_scan(metric, &[]), }), - } + }) } -fn sum_by_job(metric: &str) -> QueryExpr { - QueryExpr::Aggregate { +fn sum_by_job(metric: &str) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![AggIntent::Sum { col: None }], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(scan(metric, &["job"])), - } + child: scan(metric, &["job"]), + }) +} + +/// A PromQL binary operator: no checked-division flags, no `bool` modifier. +fn binary( + kind: BinaryOpKind, + vector_match: Option, + lhs: Rc, + rhs: Rc, +) -> Rc { + node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind, + vector_match, + }, + return_bool: false, + lhs, + rhs, + }) } // #18 — arithmetic binary op between two bare scans; no vector match #[test] fn q18_div_bare_scans() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_requests_total", &[])), - vector_match: None, - }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + None, + scan("http_requests_total", &[]), + scan("http_requests_total", &[]), + ); assert_eq!(lower("http_requests_total / http_requests_total"), expected); } // #19 — add with on(job) vector match; match labels are strings, not column ids #[test] fn q19_add_with_on_match() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_requests_total", &[])), - vector_match: Some(VectorMatch { + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + Some(VectorMatch { kind: VectorMatchKind::On, labels: vec!["job".into()], grouping: None, }), - }; + scan("http_requests_total", &[]), + scan("http_requests_total", &[]), + ); assert_eq!( lower("http_requests_total + on(job) http_requests_total"), expected @@ -96,12 +124,12 @@ fn q19_add_with_on_match() { // #20 — divide two rate aggregates over different metrics #[test] fn q20_div_two_rates() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: Rc::new(rate_agg("http_requests_total")), - rhs: Rc::new(rate_agg("http_errors_total")), - vector_match: None, - }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + None, + rate_agg("http_requests_total"), + rate_agg("http_errors_total"), + ); assert_eq!( lower("rate(http_requests_total[5m]) / rate(http_errors_total[5m])"), expected, @@ -113,12 +141,12 @@ fn q20_div_two_rates() { fn q_gt_comparison() { assert_eq!( lower("http_requests_total > http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Gt), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: None, - } + binary( + BinaryOpKind::Compare(CompareOpKind::Gt), + None, + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -126,12 +154,12 @@ fn q_gt_comparison() { fn q_lt_comparison() { assert_eq!( lower("http_requests_total < http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Lt), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: None, - } + binary( + BinaryOpKind::Compare(CompareOpKind::Lt), + None, + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -139,12 +167,12 @@ fn q_lt_comparison() { fn q_ge_comparison() { assert_eq!( lower("http_requests_total >= http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Ge), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: None, - } + binary( + BinaryOpKind::Compare(CompareOpKind::Ge), + None, + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -152,12 +180,12 @@ fn q_ge_comparison() { fn q_le_comparison() { assert_eq!( lower("http_requests_total <= http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Le), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: None, - } + binary( + BinaryOpKind::Compare(CompareOpKind::Le), + None, + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -166,16 +194,16 @@ fn q_le_comparison() { fn q_add_with_ignoring() { assert_eq!( lower("http_requests_total + ignoring(job) http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: Some(VectorMatch { + binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + Some(VectorMatch { kind: VectorMatchKind::Ignoring, labels: vec!["job".into()], grouping: None, }), - } + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -184,11 +212,9 @@ fn q_add_with_ignoring() { fn q_mul_group_left() { assert_eq!( lower("http_requests_total * on(job) group_left() node_info"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("node_info", &[])), - vector_match: Some(VectorMatch { + binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), + Some(VectorMatch { kind: VectorMatchKind::On, labels: vec!["job".into()], grouping: Some(VectorGrouping { @@ -196,7 +222,9 @@ fn q_mul_group_left() { labels: vec![], }), }), - } + scan("http_requests_total", &[]), + scan("node_info", &[]), + ) ); } @@ -205,11 +233,9 @@ fn q_mul_group_left() { fn q_mul_group_right() { assert_eq!( lower("node_info * on(job) group_right() http_requests_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(scan("node_info", &[])), - rhs: Rc::new(scan("http_requests_total", &[])), - vector_match: Some(VectorMatch { + binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), + Some(VectorMatch { kind: VectorMatchKind::On, labels: vec!["job".into()], grouping: Some(VectorGrouping { @@ -217,7 +243,9 @@ fn q_mul_group_right() { labels: vec![], }), }), - } + scan("node_info", &[]), + scan("http_requests_total", &[]), + ) ); } @@ -225,12 +253,12 @@ fn q_mul_group_right() { // each side: Aggregate{Sum, by=[2]} over Scan([ts, value, job]) #[test] fn q21_div_two_sum_by_job() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: Rc::new(sum_by_job("http_requests_total")), - rhs: Rc::new(sum_by_job("http_errors_total")), - vector_match: None, - }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + None, + sum_by_job("http_requests_total"), + sum_by_job("http_errors_total"), + ); assert_eq!( lower("sum by (job) (http_requests_total) / sum by (job) (http_errors_total)"), expected, @@ -238,34 +266,29 @@ fn q21_div_two_sum_by_job() { } // #36 — unary negation lowers as `expr * -1`: a Mul BinaryOp of the vector -// against PromqlScalarBridge(-1), no vector match. The vector side keeps its schema. +// against a `ScalarExpr(-1)` leaf, no vector match. The vector side keeps +// its schema. #[test] fn q36_unary_negation_is_multiply_by_minus_one() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(scan("some_metric", &[])), - rhs: Rc::new(QueryExpr::promql_scalar(-1.0)), - vector_match: None, + let root = lower("-some_metric"); + let NonASAPOp::Project { cols, child, .. } = root.expect_non_asap() else { + panic!() }; - assert_eq!(lower("-some_metric"), expected); + assert!(child.schema.has_promql_series_identity()); + assert!(matches!(&cols[1].expr, ScalarExpr::Negative { .. })); } // #36 — negation nested inside an aggregate argument (issue #27 nesting): // `sum(-m)` → Aggregate{Sum} over the `m * -1` BinaryOp. #[test] fn q36_sum_of_negation_nests() { - let expected = QueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::Sum { col: None }], - output_names: vec!["".into()], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(scan("node_cpu_seconds_total", &[])), - rhs: Rc::new(QueryExpr::promql_scalar(-1.0)), - vector_match: None, - }), + let root = lower("sum(-node_cpu_seconds_total)"); + let NonASAPOp::Aggregate { + child, measures, .. + } = root.expect_non_asap() + else { + panic!() }; - assert_eq!(lower("sum(-node_cpu_seconds_total)"), expected); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Project { .. })); } diff --git a/crates/integration-tests/tests/cse.rs b/crates/integration-tests/tests/cse.rs index 094ffd658..ca5f5f953 100644 --- a/crates/integration-tests/tests/cse.rs +++ b/crates/integration-tests/tests/cse.rs @@ -2,14 +2,14 @@ //! #223). //! //! Drives the full staged pipeline this issue lands: two independently -//! lowered `QueryExpr` trees → `share_common_subtrees` (stage 1, -//! `asap-types::pre_asap::cse`, run internally by `search_workload`) → +//! lowered `OperatorNode` DAGs → `share_common_subdags` (stage 1, +//! `asap-types::ir::cse`, run internally by `search_workload`) → //! `search_workload` (stage 2, `asap-aware-mapping`) — and asserts the //! sharing that stage 1 decides survives into stage 2's discovered //! `CandidateLogicalASAPDAGs` as one genuinely shared `TargetSubDAGCandidates`, not just one shared -//! `Rc`. This is the "real caller" the issue's landing plan -//! requires before `share_common_subtrees` is allowed to exist at all (its -//! predecessor, `asap-plan::cse::dedupe_subtrees`, was deleted in #192 for +//! `Rc`. This is the "real caller" the issue's landing plan +//! requires before `share_common_subdags` is allowed to exist at all (its +//! predecessor, `asap-plan::cse::dedupe_sub-DAGs`, was deleted in #192 for //! being unwired dead code). //! //! Committing to one final, physically-materialized answer for a whole @@ -24,15 +24,15 @@ use std::rc::Rc; -use asap_aware_mapping::{search_workload, Replacement}; +use asap_aware_mapping::{is_logical_rewrite, search_workload, Replacement}; use asap_integration_tests::fixtures::lower_promql; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::NonASAPOp; use asap_types::types::AccuracyTarget; /// Two workload entries that happen to submit the exact same query (a /// realistic case — two dashboards, or a query fired both standalone and as -/// part of a larger batch) collapse onto one shared `Rc` after -/// `search_workload`'s internal `share_common_subtrees` pass, and onto one +/// part of a larger batch) collapse onto one shared `Rc` after +/// `search_workload`'s internal `share_common_subdags` pass, and onto one /// genuinely-shared [`TargetSubDAGCandidates`](asap_aware_mapping::TargetSubDAGCandidates) — carrying /// every candidate discovered for it exactly once, not once per root — no /// second structural-equality pass at the post-ASAP layer needed for this @@ -40,8 +40,8 @@ use asap_types::types::AccuracyTarget; #[test] fn duplicate_workload_queries_collapse_onto_one_memo_group() { // Grouped (`by (job)`), so the shared `Aggregate`'s output schema carries - // a provable unique key — the legality gate `share_common_subtrees` - // enforces (see `asap-types::pre_asap::cse`'s module doc) — and its + // a provable unique key — the legality gate `share_common_subdags` + // enforces (see `asap-types::ir::cse`'s module doc) — and its // `ExactAggregate(Sum)` realization is deterministic regardless of the // accuracy target, so this pins the sharing mechanism itself rather than // any one particular summary-family choice. @@ -57,18 +57,18 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { "fixture sanity: identical query text lowers identically" ); - let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); + let space = search_workload(vec![("a", a), ("b", b)]); // roots[0] and roots[1] must have merged onto the same Rc — the - // `share_common_subtrees` pass `search_workload` runs internally. + // `share_common_subdags` pass `search_workload` runs internally. assert!( Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1), - "search_workload must collapse the two identical roots onto one Rc" + "search_workload must collapse the two identical roots onto one Rc" ); // The single shared root is one discovered TargetSubDAG, holding one - // TargetSubDAGCandidates with consumer_count 2 — SketchAlgorithmStrategy's one - // ExactAggregate candidate *and* SharedSubtreeStrategy's share-vs- + // TargetSubDAGCandidates with consumer_count 2 — ASAPStrategies's one + // ExactAggregate candidate *and* SharedSubDagStrategy's share-vs- // recompute pair, exactly as `shared_aggregate_across_two_roots_gets_both_strategies_candidates` // (asap-aware-mapping::replacement's own equivalent, internal test) // pins for the same fixture shape. @@ -79,19 +79,21 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { assert_eq!( group.candidates.len(), 3, - "1 ExactAggregate Summary + 2 Rewrite (share/recompute): {:?}", + "1 ExactAggregate summary + 2 logical rewrites (share/recompute): {:?}", group.candidates ); + // A bound summary is a `Subtree` with an ASAP operator in it; a logical + // rewrite is a `Subtree` with none (`is_logical_rewrite`). let summary_count = group .candidates .iter() - .filter(|c| matches!(c.replacement, Replacement::Summary(_))) + .filter(|c| matches!(&c.replacement, Replacement::SubDag(n) if n.contains_asap())) .count(); let rewrite_count = group .candidates .iter() - .filter(|c| matches!(c.replacement, Replacement::Rewrite(_))) + .filter(|c| matches!(&c.replacement, Replacement::SubDag(n) if is_logical_rewrite(n))) .count(); assert_eq!(summary_count, 1); assert_eq!(rewrite_count, 2); @@ -100,12 +102,14 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { // "false-positive dedup" failure mode `is_duplicate_rewrite` exists to // prevent): one shares the group's own target `Rc`, the other is a // structurally-identical but independently-built `Rc`. - let one_is_the_target = group.candidates.iter().any( - |c| matches!(&c.replacement, Replacement::Rewrite(rc) if Rc::ptr_eq(rc, &group.target)), - ); - let one_is_not = group.candidates.iter().any( - |c| matches!(&c.replacement, Replacement::Rewrite(rc) if !Rc::ptr_eq(rc, &group.target)), - ); + let one_is_the_target = group.candidates.iter().any(|c| { + matches!(&c.replacement, Replacement::SubDag(rc) + if is_logical_rewrite(rc) && Rc::ptr_eq(rc, &group.target)) + }); + let one_is_not = group.candidates.iter().any(|c| { + matches!(&c.replacement, Replacement::SubDag(rc) + if is_logical_rewrite(rc) && !Rc::ptr_eq(rc, &group.target)) + }); assert!(one_is_the_target && one_is_not); } @@ -121,7 +125,7 @@ fn distinct_workload_queries_get_independent_memo_groups() { .expect("query b failed to lower"); assert_ne!(a, b, "fixture sanity: the two queries differ"); - let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); + let space = search_workload(vec![("a", a), ("b", b)]); assert!(!Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); let group_a = space @@ -140,24 +144,24 @@ fn distinct_workload_queries_get_independent_memo_groups() { /// Single-query CSE (a repeated sub-expression within one query) also /// survives through `search_workload`: the two grouped-`Aggregate` branches -/// of a `BinaryOp` collapse to one shared `Rc` in the internal -/// `share_common_subtrees` pass, and to one shared `TargetSubDAGCandidates` (with +/// of a `BinaryOp` collapse to one shared `Rc` in the internal +/// `share_common_subdags` pass, and to one shared `TargetSubDAGCandidates` (with /// `consumer_count == 2`, one per branch) here. #[test] fn single_query_repeated_subexpression_shares_one_memo_group() { let query = "sum by (job) (http_requests_total) / sum by (job) (http_requests_total)"; let expr = lower_promql(query, AccuracyTarget::Exact).expect("query failed to lower"); - let space = search_workload(vec![("q", Rc::new(expr))]); + let space = search_workload(vec![("q", expr)]); let [(_, root)] = space.roots.as_slice() else { panic!("expected 1 root"); }; - let QueryExpr::BinaryOp { lhs, rhs, .. } = root.as_ref() else { + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = root.non_asap() else { panic!("expected a BinaryOp root, got {root:?}"); }; assert!( Rc::ptr_eq(lhs, rhs), - "the two identical sum-by-job branches must collapse onto one Rc" + "the two identical sum-by-job branches must collapse onto one Rc" ); let group = space diff --git a/crates/integration-tests/tests/exact_composition.rs b/crates/integration-tests/tests/exact_composition.rs index fe72aea26..f59bb3b49 100644 --- a/crates/integration-tests/tests/exact_composition.rs +++ b/crates/integration-tests/tests/exact_composition.rs @@ -1,5 +1,5 @@ //! Issue #171 — composing exact operators with summary plans across -//! explicit update/readout boundaries, end to end through +//! explicit update/evaluation boundaries, end to end through //! `search_workload_with` → `CandidateLogicalASAPDAGs::global_selection` → //! `GlobalSelection::assemble_selected_dag` → `dag_export`. //! @@ -16,43 +16,56 @@ use asap_aware_mapping::cost_model::{ CostProvenance, CostUnit, ExactCompositionCostInputs, ExactCompositionCostRequest, ValueOperationCapabilities, }; +use asap_aware_mapping::exact_composition::ExactOperation; use asap_aware_mapping::replacement::{ - default_strategies_with, search_workload_with, Replacement, ReplacementProvenance, - ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG, + default_strategies_with, search_workload_with, ASAPStrategies, Replacement, + ReplacementProvenance, ReplacementStrategy, TargetSubDAG, }; use asap_aware_mapping::{ CostModel, DefaultCostModel, EvaluationRate, ExplanationKind, OperationPlacement, }; use asap_integration_tests::fixtures::lower_promql; +use asap_integration_tests::post_asap::{post_asap_dag, timed}; use asap_types::dag_export; +use asap_types::ir::export::{NonASAPOpKind, PostAsapOperatorPayload}; +use asap_types::ir::operator_properties::{Reduction, Source}; +use asap_types::ir::timing::data_state; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, TimeRangeKind}; use asap_types::post_asap::{ - validate_execution_data_states, ExactKind, ExactOperation, ExecutionDataState, ExecutionTiming, - SketchAlgorithm, SummaryExpr, SummaryFamilyType, SummaryNode, SummaryUpdate, + ExactKind, ExecutionDataState, ExecutionTiming, FieldDataType, SketchAlgorithm, SummaryUpdate, }; use asap_types::pre_asap::agg_intent::{default_quantile, AggIntent}; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction, Source}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; + use asap_types::types::AccuracyTarget; // ── fixtures ──────────────────────────────────────────────────────────── -fn metric_scan(labels: &[&str]) -> QueryExpr { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::non_asap_node(op).expect("fixture node derives its schema") +} + +fn metric_scan(labels: &[&str]) -> Rc { let mut columns = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ]; - columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); - QueryExpr::Scan { + columns.extend( + labels + .iter() + .map(|n| Field::plain(*n, DataType::Utf8, true)), + ); + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "latency".into(), }, predicates: vec![], schema: Schema::with_time_index(columns, 0, vec![]), - } + }) } -fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { +fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], @@ -62,8 +75,8 @@ fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc }) } -fn per_entity(intent: AggIntent, child: Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { +fn per_entity(intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec![], @@ -74,11 +87,11 @@ fn per_entity(intent: AggIntent, child: Rc) -> Rc { } /// `quantile by (zone, host) (latency)` — the fine-grained inner summary. -fn fine_quantile() -> Rc { +fn fine_quantile() -> Rc { agg( vec![2, 3], default_quantile(0.99), - Rc::new(metric_scan(&["zone", "host"])), + metric_scan(&["zone", "host"]), ) } @@ -92,7 +105,7 @@ struct StatsModel; fn custom_accuracy_rule_survives_root_target_and_materialization() { use asap_aware_mapping::{AccuracyModel, DefaultAccuracyModel, PropagationStats}; use asap_types::post_asap::{ - AccuracyError, CompositionOperator, ExactOperation, ResultGuarantee, SketchQuery, + AccuracyError, CompositionOperator, ResultGuarantee, SketchStatistic, }; struct Model; impl AccuracyModel for Model { @@ -101,8 +114,8 @@ fn custom_accuracy_rule_survives_root_target_and_materialization() { } fn local_guarantee( &self, - family: &SummaryFamilyType, - query: &SketchQuery, + family: &FieldDataType, + query: &SketchStatistic, ) -> Option { DefaultAccuracyModel.local_guarantee(family, query) } @@ -274,34 +287,44 @@ fn unknown_runtime_capability_keeps_candidate_but_prevents_selection() { } fn plan( - roots: Vec<(&'static str, Rc)>, + roots: Vec<(&'static str, Rc)>, cost_model: &dyn CostModel, ) -> asap_aware_mapping::CandidateLogicalASAPDAGs<&'static str> { search_workload_with(roots, &default_strategies_with(cost_model)) } -fn is_plain(node: &SummaryNode) -> bool { +fn is_plain(node: &OperatorNode) -> bool { node.schema .fields .iter() - .all(|f| matches!(f.dtype, SummaryFamilyType::Plain(_))) + .all(|f| matches!(f.dtype, FieldDataType::Plain(_))) } -fn names(node: &SummaryNode) -> Vec<&str> { +fn names(node: &OperatorNode) -> Vec<&str> { node.schema.fields.iter().map(|f| f.name.as_str()).collect() } +/// The composed query-time shape: an exact `Aggregate` directly over a +/// summary evaluation, at query time. +fn is_query_time_fold(node: &OperatorNode) -> bool { + matches!( + node.non_asap(), + Some(NonASAPOp::Aggregate { child, .. }) + if matches!(child.operator, Operator::ASAP(ASAPOp::SummaryEstimate { .. })) + ) +} + // ── step 1: pin every already-supported exact-accumulator nesting ─────── #[test] fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { use std::time::Duration; - let cases: Vec<(Rc, ExactKind)> = vec![ + let cases: Vec<(Rc, ExactKind)> = vec![ ( agg( vec![2], AggIntent::Sum { col: None }, - Rc::new(metric_scan(&["zone"])), + metric_scan(&["zone"]), ), ExactKind::Sum, ), @@ -311,7 +334,7 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { AggIntent::Count { accuracy: AccuracyTarget::Exact, }, - Rc::new(metric_scan(&["zone"])), + metric_scan(&["zone"]), ), ExactKind::Count, ), @@ -319,7 +342,7 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { agg( vec![2], AggIntent::Min { col: None }, - Rc::new(metric_scan(&["zone"])), + metric_scan(&["zone"]), ), ExactKind::Min, ), @@ -327,16 +350,17 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { agg( vec![2], AggIntent::Max { col: None }, - Rc::new(metric_scan(&["zone"])), + metric_scan(&["zone"]), ), ExactKind::Max, ), ( per_entity( AggIntent::Rate, - Rc::new(QueryExpr::TimeRange { + node(NonASAPOp::TimeRange { range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["zone"])), + kind: TimeRangeKind::Range, + child: metric_scan(&["zone"]), }), ), ExactKind::Rate, @@ -344,9 +368,10 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { ( per_entity( AggIntent::Increase, - Rc::new(QueryExpr::TimeRange { + node(NonASAPOp::TimeRange { range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["zone"])), + kind: TimeRangeKind::Range, + child: metric_scan(&["zone"]), }), ), ExactKind::Increase, @@ -355,40 +380,43 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { for (inner, kind) in cases { let outer = agg(vec![], default_quantile(0.9), inner); let target = TargetSubDAG::new(&outer); - let candidates = SketchAlgorithmStrategy::default_cost_model().replacements(&target); - let Replacement::Summary(root) = &candidates[0].replacement else { + let candidates = ASAPStrategies::default_cost_model().replacements(&target); + let Replacement::SubDag(root) = &candidates[0].replacement else { unreachable!() }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("expected KLL readout, got {:?}", root.expr); + // Timing is not stored on the plan: time it (default lifecycle, + // which also validates every edge) and inspect the timed copy. + let root = timed(root); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { + panic!("expected KLL evaluation, got {:?}", root.operator); }; - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary_input.operator else { panic!("expected outer SummaryAgg"); }; - let SummaryExpr::ValueOperation { - child, - operation: asap_types::post_asap::ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::IngestionTime, - } = &child.expr + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: finalized }) = &child.operator else { panic!("{kind:?}: missing maintenance finalization"); }; + assert_eq!( + child.timing, + Some(ExecutionTiming::IngestionTime), + "{kind:?}: finalization runs at maintenance time" + ); assert!( matches!( - &child.expr, - SummaryExpr::SummaryAgg { family: SummaryFamilyType::ExactAggregate(k, _), .. } if *k == kind + &finalized.operator, + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(k, _), .. }) if *k == kind ), "{kind:?}: expected the exact accumulator under its finalization, got {:?}", - child.expr + finalized.operator ); - validate_execution_data_states(root).expect("accumulator state composes under maintenance"); } } -// ── direction 1: outer exact fold over an inner summary readout ──────── +// ── direction 1: outer exact fold over an inner summary evaluation ──────── -/// Before this PR both `max`/`avg` over a quantile collapsed into one -/// opaque `KeepPreAsap`. Now: the outer group holds an `ValueOperationAtQueryTime` +/// `max`/`avg` over a quantile does not collapse into one opaque kept +/// sub-DAG: the outer group holds an `ValueOperationAtQueryTime` /// candidate referencing the inner target, the inner group keeps its own /// sketch candidates, and with statistics the pair is committed and /// materializes as `ValueOperationAtQueryTime → SummaryEstimate → SummaryAgg`. @@ -398,7 +426,7 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { let root = agg(vec![0], intent.clone(), fine_quantile()); let space = plan(vec![("q", Rc::clone(&root))], &StatsModel); let root = Rc::clone(&space.roots[0].1); - let QueryExpr::Aggregate { child: inner, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { child: inner, .. }) = root.non_asap() else { unreachable!() }; @@ -415,9 +443,9 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { inner_group .candidates .iter() - .any(|c| matches!(&c.replacement, Replacement::Summary(n) - if matches!(n.expr, SummaryExpr::SummaryEstimate { .. }))), - "{intent:?}: the inner quantile keeps its own readout candidates" + .any(|c| matches!(&c.replacement, Replacement::SubDag(n) + if matches!(n.operator, Operator::ASAP(ASAPOp::SummaryEstimate { .. })))), + "{intent:?}: the inner quantile keeps its own evaluation candidates" ); let selection = space.global_selection(&StatsModel); @@ -445,18 +473,16 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { )); let composed = selection.assemble_selected_dag(&root).unwrap().unwrap(); - let SummaryExpr::ValueOperation { - child, - timing: ExecutionTiming::QueryTime, - .. - } = &composed.expr - else { + let Some(NonASAPOp::Aggregate { child, .. }) = composed.non_asap() else { panic!( "{intent:?}: expected ValueOperationAtQueryTime root, got {:?}", - composed.expr + composed.operator ); }; - assert!(matches!(child.expr, SummaryExpr::SummaryEstimate { .. })); + assert!(matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + )); assert!( child.guarantee.is_some(), "child has its KLL rank guarantee" @@ -468,15 +494,18 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { assert!(is_plain(&composed)); assert_eq!( names(&composed), - root.output_schema() - .unwrap() - .columns + root.schema + .fields .iter() .map(|c| c.name.as_str()) .collect::>(), "the composed plan's schema is the pre-ASAP target's own" ); - validate_execution_data_states(&composed).unwrap(); + assert_eq!( + timed(&composed).timing, + Some(ExecutionTiming::QueryTime), + "{intent:?}: the exact fold runs at query time" + ); } } @@ -486,11 +515,7 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { fn avg_over_quantile_keeps_the_sum_over_count_rewrite_as_a_competitor() { // `by (zone)` over `by (zone)`: the averaged column resolves to the // non-null quantile output, which is what the rewrite requires. - let inner = agg( - vec![2], - default_quantile(0.99), - Rc::new(metric_scan(&["zone"])), - ); + let inner = agg(vec![2], default_quantile(0.99), metric_scan(&["zone"])); let root = agg(vec![0], AggIntent::Avg { col: None }, inner); let space = plan(vec![("q", root)], &StatsModel); let group = space.candidates_for_target(&space.roots[0].1).unwrap(); @@ -504,11 +529,7 @@ fn avg_over_quantile_keeps_the_sum_over_count_rewrite_as_a_competitor() { /// is the same, only the fold's row multiplicity differs. #[test] fn identity_and_genuine_multi_row_folds_both_compose() { - let identity_inner = agg( - vec![2], - default_quantile(0.99), - Rc::new(metric_scan(&["zone"])), - ); + let identity_inner = agg(vec![2], default_quantile(0.99), metric_scan(&["zone"])); for (label, inner) in [ ("identity", identity_inner), ("fine-to-coarse", fine_quantile()), @@ -523,14 +544,17 @@ fn identity_and_genuine_multi_row_folds_both_compose() { .unwrap(); assert!( matches!( - composed.expr, - SummaryExpr::ValueOperation { - timing: ExecutionTiming::QueryTime, - .. - } + composed.non_asap(), + Some(NonASAPOp::Aggregate { child, .. }) + if matches!(child.operator, Operator::ASAP(ASAPOp::SummaryEstimate { .. })) ), "{label}: {:?}", - composed.expr + composed.operator + ); + assert_eq!( + timed(&composed).timing, + Some(ExecutionTiming::QueryTime), + "{label}" ); assert_eq!(names(&composed), vec!["zone", "max"], "{label}"); } @@ -539,7 +563,7 @@ fn identity_and_genuine_multi_row_folds_both_compose() { /// One inner quantile consumed by two outer folds in two queries: CSE /// collapses the inner target onto one `Rc`, both compositions commit to /// the *same* child candidate, and both materializations share one -/// `Rc` for it — the summary is maintained once. +/// `Rc` for it — the summary is maintained once. #[test] fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { let max = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); @@ -547,9 +571,9 @@ fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { let space = plan(vec![("max", max), ("min", min)], &StatsModel); let selection = space.global_selection(&StatsModel); - let roots: Vec> = space.roots.iter().map(|(_, r)| Rc::clone(r)).collect(); - let inner_of = |r: &Rc| match r.as_ref() { - QueryExpr::Aggregate { child, .. } => Rc::clone(child), + let roots: Vec> = space.roots.iter().map(|(_, r)| Rc::clone(r)).collect(); + let inner_of = |r: &Rc| match r.non_asap() { + Some(NonASAPOp::Aggregate { child, .. }) => Rc::clone(child), _ => unreachable!(), }; assert!( @@ -585,17 +609,20 @@ fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { .iter() .map(|r| selection.assemble_selected_dag(r).unwrap().unwrap()) .collect(); - let child_of = |n: &Rc| match &n.expr { - SummaryExpr::ValueOperation { - child, - timing: ExecutionTiming::QueryTime, - .. - } => Rc::clone(child), - other => panic!("expected ValueOperationAtQueryTime, got {other:?}"), + let child_of = |n: &Rc| match n.non_asap() { + Some(NonASAPOp::Aggregate { child, .. }) + if matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + ) => + { + Rc::clone(child) + } + _ => panic!("expected ValueOperationAtQueryTime, got {:?}", n.operator), }; assert!( Rc::ptr_eq(&child_of(&composed[0]), &child_of(&composed[1])), - "both folds compose over the same Rc" + "both folds compose over the same Rc" ); } @@ -610,15 +637,16 @@ fn outer_summary_over_an_exact_function_composes_at_ingestion_time() { use std::time::Duration; let deriv = per_entity( AggIntent::Deriv, - Rc::new(QueryExpr::TimeRange { + node(NonASAPOp::TimeRange { range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["zone"])), + kind: TimeRangeKind::Range, + child: metric_scan(&["zone"]), }), ); let root = agg(vec![], default_quantile(0.99), deriv); let space = plan(vec![("q", root)], &StatsModel); let root = Rc::clone(&space.roots[0].1); - let QueryExpr::Aggregate { child: deriv, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { child: deriv, .. }) = root.non_asap() else { unreachable!() }; assert!(space @@ -639,33 +667,25 @@ fn outer_summary_over_an_exact_function_composes_at_ingestion_time() { assert!(decision.cost_rate < decision.baseline_rate); let composed = selection.assemble_selected_dag(&root).unwrap().unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &composed.expr else { - panic!("expected readout root, got {:?}", composed.expr); + // Walk the timed copy: timing is written by the lifecycle assignment. + let composed = timed(&composed); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &composed.operator else { + panic!("expected evaluation root, got {:?}", composed.operator); }; - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary_input.operator else { panic!("expected SummaryAgg"); }; - let SummaryExpr::ValueOperation { - child: raw, - timing: ExecutionTiming::IngestionTime, - .. - } = &child.expr - else { + let Some(NonASAPOp::Aggregate { child: raw, .. }) = child.non_asap() else { panic!( "expected ValueOperationAtIngestionTime under the maintained summary, got {:?}", - child.expr + child.operator ); }; - assert!(matches!(raw.expr, SummaryExpr::KeepPreAsap(_))); - let assignment = validate_execution_data_states(&composed).unwrap(); - assert_eq!( - assignment.data_state_of(child), - Some(ExecutionDataState::INGESTION_ROWS) - ); - assert_eq!( - assignment.data_state_of(raw), - Some(ExecutionDataState::INGESTION_ROWS) - ); + // The raw input is kept as-is. + assert!(matches!(raw.non_asap(), Some(NonASAPOp::TimeRange { .. }))); + assert!(!raw.contains_asap()); + assert_eq!(data_state(child), Some(ExecutionDataState::INGESTION_ROWS)); + assert_eq!(data_state(raw), Some(ExecutionDataState::INGESTION_ROWS)); } // ── rejection, capability, statistics ─────────────────────────────────── @@ -681,10 +701,10 @@ fn summary_construction_follows_its_value_input_phase() { .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let illegal = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { + let illegal = OperatorNode::asap_node( + ASAPOp::SummaryAgg { child: post, - family: SummaryFamilyType::ExactAggregate( + family: FieldDataType::ExactAggregate( ExactKind::Max, asap_types::post_asap::ExactParams::Max, ), @@ -693,15 +713,14 @@ fn summary_construction_follows_its_value_input_phase() { grouping: Default::default(), filter: None, }, - schema: asap_types::post_asap::SummarySchema { - fields: vec![], - time_index: None, - }, - guarantee: None, - }); - let state = asap_types::post_asap::produced_data_state(&illegal.expr).unwrap(); + Schema::lifted(vec![], None), + None, + ); + // Even under an ingestion-time consumer the state is built at query + // time, because a evaluation sits below it. + let state = asap_types::ir::planned_data_state(&illegal, ExecutionTiming::IngestionTime); assert_eq!(state.timing, ExecutionTiming::QueryTime); - asap_types::post_asap::validate_execution_data_states_at(&illegal, state).unwrap(); + asap_types::ir::validate_default(&illegal, state.timing).unwrap(); } #[test] @@ -717,9 +736,9 @@ fn a_runtime_without_mixed_execution_gets_no_composition_candidates() { let selection = space.global_selection(&NoCapabilityModel); assert!(selection.for_target(&root).unwrap().composition.is_none()); let node = selection.assemble_selected_dag(&root).unwrap().unwrap(); - assert!(!matches!(node.expr, SummaryExpr::ValueOperation { .. })); + assert!(!is_query_time_fold(&node)); // The inner quantile is still independently selectable. - let QueryExpr::Aggregate { child, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { child, .. }) = root.non_asap() else { unreachable!() }; assert!(selection.for_target(child).unwrap().chosen.is_some()); @@ -730,7 +749,7 @@ fn a_runtime_without_mixed_execution_gets_no_composition_candidates() { /// site keeps a non-composed alternative, and the inner summary stays /// independently selectable. #[test] -fn missing_cost_statistics_preserve_the_conservative_keep_pre_asap() { +fn missing_cost_statistics_preserve_the_conservative_retain_exact() { let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); let space = plan(vec![("q", root)], &DefaultCostModel); let root = Rc::clone(&space.roots[0].1); @@ -748,9 +767,9 @@ fn missing_cost_statistics_preserve_the_conservative_keep_pre_asap() { Some(Replacement::ExactComposition(_)) )); let node = selection.assemble_selected_dag(&root).unwrap().unwrap(); - assert!(!matches!(node.expr, SummaryExpr::ValueOperation { .. })); + assert!(!is_query_time_fold(&node)); - let explanations = asap_aware_mapping::explain_replacements(vec![("q", (*root).clone())]); + let explanations = asap_aware_mapping::explain_replacements(vec![("q", Rc::clone(&root))]); assert!(explanations .iter() .any(|e| e.kind == ExplanationKind::ExactComposition)); @@ -770,17 +789,24 @@ fn dag_export_carries_explicit_stage_and_plain_schema_for_a_composed_plan() { .unwrap(); let graph = dag_export::export_summary(&composed); let node = &graph.nodes[graph.root as usize]; - assert_eq!(node.kind, "ValueOperation"); - assert_eq!(node.detail["timing"], "query_time"); - assert!(node.detail["operation"] - .as_str() - .unwrap() - .starts_with("Exact(Aggregate")); + assert_eq!(node.kind, "aggregate"); + assert!(node.detail["measures"].is_array()); + // Timing is explicit in the wire-6 DAG: the root is a relational + // aggregate placed at query time. + let wire = post_asap_dag(&composed); + let wire_root = wire.nodes.iter().find(|n| n.id == wire.root).unwrap(); + assert!(matches!( + wire_root.payload, + PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::Aggregate { .. } + } + )); + assert_eq!(wire_root.output_state.timing, ExecutionTiming::QueryTime); // Pre-ASAP export of the same target still describes the same columns. let pre = dag_export::export(root); let pre_root = &pre.nodes[pre.root as usize]; - let pre_cols: Vec = pre_root.schema.as_ref().unwrap()["columns"] + let pre_cols: Vec = pre_root.schema.as_ref().unwrap()["fields"] .as_array() .unwrap() .iter() @@ -797,7 +823,7 @@ fn promql_max_by_zone_over_quantile_over_time_composes() { AccuracyTarget::Epsilon(0.01), ) .unwrap(); - let space = plan(vec![("q", Rc::new(expr))], &StatsModel); + let space = plan(vec![("q", expr)], &StatsModel); let root = &space.roots[0].1; let selection = space.global_selection(&StatsModel); let selected = selection.for_target(root).unwrap(); @@ -814,13 +840,8 @@ fn promql_max_by_zone_over_quantile_over_time_composes() { .collect::>() ); let composed = selection.assemble_selected_dag(root).unwrap().unwrap(); - assert!(matches!( - composed.expr, - SummaryExpr::ValueOperation { - timing: ExecutionTiming::QueryTime, - .. - } - )); + assert!(is_query_time_fold(&composed), "{:?}", composed.operator); + assert_eq!(timed(&composed).timing, Some(ExecutionTiming::QueryTime)); assert_eq!( selected.composition.as_ref().map(|d| d.inputs.unit), Some(CostUnit::CostUnitsPerSecond) diff --git a/crates/integration-tests/tests/frontend_timestamps.rs b/crates/integration-tests/tests/frontend_timestamps.rs index 67c5418cc..300f0cee7 100644 --- a/crates/integration-tests/tests/frontend_timestamps.rs +++ b/crates/integration-tests/tests/frontend_timestamps.rs @@ -1,9 +1,9 @@ //! Cross-frontend evaluation-time semantics (issues #46 and #184). use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_integration_tests::fixtures::lower_promql; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::QueryExpr; +use asap_integration_tests::fixtures::lower_promql_root; +use asap_types::ir::{NonASAPOp, ScalarExpr}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; /// PromQL exposes its evaluation time as Unix seconds, whereas SQL exposes @@ -11,14 +11,25 @@ use asap_types::types::AccuracyTarget; /// but must remain distinguishable in the shared IR and type inference. #[tokio::test] async fn promql_eval_time_and_sql_current_timestamp_remain_distinct() { - let promql = lower_promql("time()", AccuracyTarget::Exact).expect("lower PromQL time()"); - assert!(matches!(promql, QueryExpr::EvalTimestamp)); - let promql_schema = promql.output_schema().expect("PromQL time() schema"); - assert_eq!(promql_schema.columns[0].dtype, DataType::Float64); + let promql = lower_promql_root("time()", AccuracyTarget::Exact).expect("lower PromQL time()"); + assert!( + matches!( + promql, + asap_types::ir::QueryRoot::Scalar(ScalarExpr::EvalTimestamp) + ), + "expected a bare evaluation-time scalar, got {promql:?}" + ); + assert_eq!( + ScalarExpr::EvalTimestamp + .scalar_type(&Schema::default()) + .unwrap() + .0, + DataType::Float64 + ); let catalog = SqlCatalog::new().with_table( "metrics", - Schema::new(vec![Column::new("value", DataType::Float64, false)]), + Schema::new(vec![Field::plain("value", DataType::Float64, false)]), ); let sql = lower_sql( "SELECT CURRENT_TIMESTAMP FROM metrics", @@ -27,13 +38,14 @@ async fn promql_eval_time_and_sql_current_timestamp_remain_distinct() { ) .await .expect("lower SQL CURRENT_TIMESTAMP"); - let QueryExpr::Project { cols, .. } = sql else { + let Some(NonASAPOp::Project { cols, child, .. }) = sql.non_asap() else { panic!("expected SQL projection, got {sql:?}"); }; - assert!(matches!(&cols[0].expr, QueryExpr::CurrentTimestamp)); - let sql_schema = cols[0] + assert!(matches!(&cols[0].expr, ScalarExpr::CurrentTimestamp)); + let (sql_dtype, _) = cols[0] .expr - .output_schema() - .expect("SQL CURRENT_TIMESTAMP schema"); - assert_eq!(sql_schema.columns[0].dtype, DataType::Timestamp); + .scalar_type(&child.schema) + .expect("SQL CURRENT_TIMESTAMP type"); + assert_eq!(sql_dtype, DataType::Timestamp); + assert_eq!(sql.schema.fields[0].dtype, DataType::Timestamp); } diff --git a/crates/integration-tests/tests/kll_pane_execution.rs b/crates/integration-tests/tests/kll_pane_execution.rs index e818796fd..20d850376 100644 --- a/crates/integration-tests/tests/kll_pane_execution.rs +++ b/crates/integration-tests/tests/kll_pane_execution.rs @@ -1,7 +1,7 @@ //! Maintenance -> stored pane state -> independently bound query execution. mod physical_common; use asap_physical_operators::{ - operators::{Operator, ReadoutQuery}, + operators::{Operator, SummaryEvaluation}, physical_planner::{CompiledPhysicalDag, InputContract, Source}, plan::{PhysicalDag, PhysicalOperator, PlanProperties}, runtime::{Input, Limits, OutputStream, RunContext, Scope}, @@ -11,8 +11,8 @@ use asap_physical_operators::{ }; use asap_types::{ post_asap::{ - SketchAlgorithm, SketchKind, SketchParams, SketchQuery, SummaryFamilyType, SummaryField, - SummarySchema, + Field, FieldDataType, Schema as LogicalSchema, SketchAlgorithm, SketchKind, SketchParams, + SketchStatistic, }, pre_asap::DataType, }; @@ -25,17 +25,20 @@ use std::{ }, }; -fn family(k: u32) -> SummaryFamilyType { - SummaryFamilyType::Sketch( +fn family(k: u32) -> FieldDataType { + FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), Default::default(), ) } fn raw_schema() -> Schema { - Arc::new(SummarySchema { - fields: vec![SummaryField { + Arc::new(LogicalSchema { + unique_keys: vec![], + closed: false, + fields: vec![Field { + table: None, name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, }], time_index: None, @@ -96,8 +99,13 @@ fn restore(schema: Schema, states: &[Arc]) -> Batch { ) .unwrap() } -fn readout(schema: Schema, q: f64) -> Operator { - Operator::readout(schema, 0, ReadoutQuery::Sketch(SketchQuery::Quantile { q })).unwrap() +fn evaluation(schema: Schema, q: f64) -> Operator { + Operator::evaluation( + schema, + 0, + SummaryEvaluation::Sketch(SketchStatistic::Quantile { q }), + ) + .unwrap() } struct CountStarts { operator: Operator, @@ -152,8 +160,8 @@ fn five_panes_roundtrip_and_shared_merge_runs_once() { ), ), (6, (vec![5], merge.clone())), - (7, (vec![6], readout(schema.clone(), 0.5))), - (8, (vec![6], readout(schema.clone(), 0.99))), + (7, (vec![6], evaluation(schema.clone(), 0.5))), + (8, (vec![6], evaluation(schema.clone(), 0.99))), ]), vec![6, 7, 8], ) @@ -240,8 +248,10 @@ fn five_panes_roundtrip_and_shared_merge_runs_once() { }, ) .unwrap(); - dag.add(2, vec![1], readout(schema.clone(), 0.5)).unwrap(); - dag.add(3, vec![1], readout(schema.clone(), 0.99)).unwrap(); + dag.add(2, vec![1], evaluation(schema.clone(), 0.5)) + .unwrap(); + dag.add(3, vec![1], evaluation(schema.clone(), 0.99)) + .unwrap(); let outputs = block_on(futures::future::join_all( dag.execute( &[2, 3], diff --git a/crates/integration-tests/tests/nested.rs b/crates/integration-tests/tests/nested.rs index af25af3be..42848cccc 100644 --- a/crates/integration-tests/tests/nested.rs +++ b/crates/integration-tests/tests/nested.rs @@ -1,8 +1,8 @@ //! Multi-node pipeline tests — nested `Aggregate`, `TimeRange`, `BinaryOp`, and `Scan`. //! //! Key invariant: `rate`/`increase` are label-preserving (per-series), so an -//! outer `Aggregate.by` resolves its group keys against the inner aggregate's -//! output schema, which still carries all label columns. +//! outer `Aggregate` reduction resolves its group keys against the inner +//! aggregate's output schema, which still carries all label columns. //! //! Label column ordering is always alphabetical, so in a query that references //! both `job` and `status`: @@ -13,56 +13,106 @@ use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; +use asap_types::ir::{ + BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, Predicate, ScalarExpr, TimeRangeKind, +}; use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, GroupKeys, Predicate, - PromQLVectorSetOpKind, QueryExpr, Reduction, ScalarValue, Source, TimeShift, VectorMatch, - VectorMatchKind, + AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, GroupKeys, + PromQLVectorSetOpKind, Reduction, ScalarValue, Source, TimeShift, VectorMatch, VectorMatchKind, }; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::non_asap_node(op).expect("fixture node derives its schema") +} + +fn scan(metric: &str, predicates: Vec, labels: &[&str]) -> Rc { + node(NonASAPOp::Scan { + source: Source::TimeSeries { + metric: metric.into(), + }, + predicates, + schema: metric_schema(labels), + }) +} + +fn instant(child: Rc) -> Rc { + node(NonASAPOp::TimeRange { + range: Duration::from_secs(1), + kind: TimeRangeKind::Instant, + child, + }) +} + +fn range(secs: u64, child: Rc) -> Rc { + node(NonASAPOp::TimeRange { + range: Duration::from_secs(secs), + kind: TimeRangeKind::Range, + child, + }) +} + +fn eq_pred(col_id: usize, value: &str) -> Predicate { + Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(col_id)), + op: CompareOpKind::Eq, + right: Box::new(ScalarExpr::Literal(ScalarValue::Utf8(value.into()))), + semantics: ExprSemantics::Promql, + }) +} + +fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(child), - } + child, + }) } -fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { +fn agg_per_entity(intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(child), - } + child, + }) +} + +fn binary( + kind: BinaryOpKind, + vector_match: Option, + lhs: Rc, + rhs: Rc, +) -> Rc { + node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind, + vector_match, + }, + return_bool: false, + lhs, + rhs, + }) } // #22 — sum by job over rate; outer by=[2] resolves against rate's // label-preserving output schema [ts, value, job] #[test] fn q22_sum_by_job_over_rate() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![], - schema: metric_schema(&["job"]), - }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }, + range(300, scan("http_requests_total", vec![], &["job"])), ); let expected = agg(vec![2], AggIntent::Sum { col: None }, inner_rate); assert_eq!( @@ -76,86 +126,49 @@ fn q22_sum_by_job_over_rate() { // predicate on status (col 3); group key job (col 2) #[test] fn q23_sum_by_job_over_filtered_scan() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("200".into()))), - }))], - schema: metric_schema(&["job", "status"]), - }; - let expected = agg( - vec![2], - AggIntent::Sum { col: None }, - QueryExpr::TimeRange { - range: Duration::from_secs(1), - child: Rc::new(scan), - }, + let scan = scan( + "http_requests_total", + vec![eq_pred(3, "200")], + &["job", "status"], ); + let expected = agg(vec![2], AggIntent::Sum { col: None }, instant(scan)); assert_eq!( lower(r#"sum by (job) (http_requests_total{status="200"})"#), expected ); } -// #25 — binary op over two complex subtrees +// #25 — binary op over two complex sub-DAGs // LHS: sum by (job) over rate over filtered scan // schema [ts, value, job, status]; outer by=[2] (job) // RHS: sum by (job) over rate over bare scan // schema [ts, value, job]; outer by=[2] (job) #[test] fn q25_div_over_complex_subtrees() { - let lhs_scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("200".into()))), - }))], - schema: metric_schema(&["job", "status"]), - }; + let lhs_scan = scan( + "http_requests_total", + vec![eq_pred(3, "200")], + &["job", "status"], + ); let lhs = agg( vec![2], AggIntent::Sum { col: None }, - agg_per_entity( - AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(lhs_scan), - }, - ), + agg_per_entity(AggIntent::Rate, range(300, lhs_scan)), ); - let rhs_scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_errors_total".into(), - }, - predicates: vec![], - schema: metric_schema(&["job"]), - }; + let rhs_scan = scan("http_errors_total", vec![], &["job"]); let rhs = agg( vec![2], AggIntent::Sum { col: None }, - agg_per_entity( - AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(rhs_scan), - }, - ), + agg_per_entity(AggIntent::Rate, range(300, rhs_scan)), ); - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: Rc::new(lhs), - rhs: Rc::new(rhs), - vector_match: None, - }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + None, + lhs, + rhs, + ); assert_eq!( lower( r#"sum by (job) (rate(http_requests_total{status="200"}[5m])) / sum by (job) (rate(http_errors_total[5m]))"# @@ -171,19 +184,9 @@ fn q25_div_over_complex_subtrees() { // label-preserving output schema; the outer `max` has no grouping. #[test] fn q27_max_over_sum_by_job_over_rate() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![], - schema: metric_schema(&["job"]), - }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }, + range(300, scan("http_requests_total", vec![], &["job"])), ); let sum_by_job = agg(vec![2], AggIntent::Sum { col: None }, inner_rate); let expected = agg(vec![], AggIntent::Max { col: None }, sum_by_job); @@ -201,25 +204,12 @@ fn q27_max_over_sum_by_job_over_rate() { // Scan schema: [ts(0), value(1), group(2), job(3)] (labels alphabetical). #[test] fn q53_outer_group_key_absent_from_nested_aggregate() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests".into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("api-server".into()))), - }))], - schema: metric_schema(&["group", "job"]), - }; - let inner = agg( - vec![2], - AggIntent::Sum { col: None }, - QueryExpr::TimeRange { - range: Duration::from_secs(1), - child: Rc::new(scan), - }, + let scan = scan( + "http_requests", + vec![eq_pred(3, "api-server")], + &["group", "job"], ); + let inner = agg(vec![2], AggIntent::Sum { col: None }, instant(scan)); let expected = agg(vec![], AggIntent::Sum { col: None }, inner); assert_eq!( lower(r#"sum(sum by (group)(http_requests{job="api-server"})) by (job)"#), @@ -236,33 +226,26 @@ fn q53_outer_group_key_absent_from_nested_aggregate() { // parser's default `ignoring([])` match modifier. #[test] fn q52_outer_name_label_over_binary_op() { - let side = |metric: &str, env: &str| QueryExpr::TimeRange { - range: Duration::from_secs(1), - child: Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: metric.into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(2)), // env - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(env.into()))), - }))], - schema: metric_schema(&["env", "__name__"]), - }), + let side = |metric: &str, env: &str| { + instant(scan( + metric, + vec![eq_pred(2, env)], // env + &["env", "__name__"], + )) }; let expected = agg( vec![3], // __name__ AggIntent::Sum { col: None }, - QueryExpr::BinaryOp { - op: BinaryOpKind::Set(PromQLVectorSetOpKind::Or), - lhs: Rc::new(side("metric_a", "1")), - rhs: Rc::new(side("metric_b", "2")), - vector_match: Some(VectorMatch { + binary( + BinaryOpKind::Set(PromQLVectorSetOpKind::Or), + Some(VectorMatch { kind: VectorMatchKind::Ignoring, labels: vec![], grouping: None, }), - }, + side("metric_a", "1"), + side("metric_b", "2"), + ), ); assert_eq!( lower(r#"sum by (__name__)(metric_a{env="1"} or metric_b{env="2"})"#), @@ -276,28 +259,18 @@ fn q52_outer_name_label_over_binary_op() { // the inner rate is label-preserving. Scan schema [ts(0), value(1), instance(2)]. #[test] fn q39_sum_without_instance_over_rate() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![], - schema: metric_schema(&["instance"]), - }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }, + range(300, scan("http_requests_total", vec![], &["instance"])), ); - let expected = QueryExpr::Aggregate { + let expected = node(NonASAPOp::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(vec![2])), // exclude `instance` measures: vec![AggIntent::Sum { col: None }], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(inner_rate), - }; + child: inner_rate, + }); assert_eq!( lower("sum without (instance) (rate(http_requests_total[5m]))"), expected, @@ -310,35 +283,25 @@ fn q39_sum_without_instance_over_rate() { #[test] fn q40_week_over_week_offset() { let rate_over = |shift: Option| { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: metric_schema(&[]), - }; + let scan = scan("m", vec![], &[]); let ranged = match shift { - Some(ms) => QueryExpr::TimeShift { + Some(ms) => node(NonASAPOp::TimeShift { shift: TimeShift { offset_ms: ms, at: None, }, - child: Rc::new(scan), - }, + child: scan, + }), None => scan, }; - agg_per_entity( - AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(ranged), - }, - ) - }; - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), - lhs: Rc::new(rate_over(None)), - rhs: Rc::new(rate_over(Some(604_800_000))), // 1w - vector_match: None, + agg_per_entity(AggIntent::Rate, range(300, ranged)) }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), + None, + rate_over(None), + rate_over(Some(604_800_000)), // 1w + ); assert_eq!(lower("rate(m[5m]) - rate(m[5m] offset 1w)"), expected,); } @@ -346,22 +309,13 @@ fn q40_week_over_week_offset() { // (seconds → ms); a bare selector wrapped in a `TimeShift` carrying the anchor. #[test] fn q40_at_modifier_absolute() { - let expected = QueryExpr::TimeRange { - range: Duration::from_secs(1), - child: Rc::new(QueryExpr::TimeShift { - shift: TimeShift { - offset_ms: 0, - at: Some(AtModifier::Timestamp(1_609_746_000_000)), - }, - child: Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: "up".into(), - }, - predicates: vec![], - schema: metric_schema(&[]), - }), - }), - }; + let expected = instant(node(NonASAPOp::TimeShift { + shift: TimeShift { + offset_ms: 0, + at: Some(AtModifier::Timestamp(1_609_746_000_000)), + }, + child: scan("up", vec![], &[]), + })); assert_eq!(lower("up @ 1609746000"), expected); } @@ -370,24 +324,12 @@ fn q40_at_modifier_absolute() { // so outer sum by job still finds job at col 2 #[test] fn q24_sum_by_job_over_rate_over_filtered_scan() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("200".into()))), - }))], - schema: metric_schema(&["job", "status"]), - }; - let inner_rate = agg_per_entity( - AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }, + let scan = scan( + "http_requests_total", + vec![eq_pred(3, "200")], + &["job", "status"], ); + let inner_rate = agg_per_entity(AggIntent::Rate, range(300, scan)); let expected = agg(vec![2], AggIntent::Sum { col: None }, inner_rate); assert_eq!( lower(r#"sum by (job) (rate(http_requests_total{status="200"}[5m]))"#), @@ -403,35 +345,25 @@ fn q24_sum_by_job_over_rate_over_filtered_scan() { // the whole spine survives verbatim and the schema stays label-preserving. #[test] fn q27_nested_subquery_prometheus_docs_example() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "distance_covered_total".into(), - }, - predicates: vec![], - schema: metric_schema(&[]), - }; let rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(5), - child: Rc::new(scan), - }, + range(5, scan("distance_covered_total", vec![], &[])), ); let deriv = agg_per_entity( AggIntent::Deriv, - QueryExpr::PromqlSubquery { + node(NonASAPOp::PromqlSubquery { range: Duration::from_secs(30), resolution: Some(Duration::from_secs(5)), - child: Rc::new(rate), - }, + child: rate, + }), ); let expected = agg_per_entity( AggIntent::Max { col: None }, - QueryExpr::PromqlSubquery { + node(NonASAPOp::PromqlSubquery { range: Duration::from_secs(600), resolution: None, - child: Rc::new(deriv), - }, + child: deriv, + }), ); assert_eq!( lower("max_over_time(deriv(rate(distance_covered_total[5s])[30s:5s])[10m:])"), diff --git a/crates/integration-tests/tests/operator_design_examples.rs b/crates/integration-tests/tests/operator_design_examples.rs new file mode 100644 index 000000000..d5bdbe300 --- /dev/null +++ b/crates/integration-tests/tests/operator_design_examples.rs @@ -0,0 +1,397 @@ +//! #511 examples: source text → unified graph → summary rewrite → flat export. +use asap_frontend_sql::{lower_sql, SqlCatalog}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, ScalarExpr}; +use asap_types::post_asap::{ + ExactKind, ExactParams, FieldDataType, GroupingStrategy, SummaryUpdate, +}; +use asap_types::pre_asap::{AggIntent, ColumnRef, DataType, Field, Schema}; +use asap_types::types::AccuracyTarget; +use std::rc::Rc; +mod physical_common; + +fn catalog() -> SqlCatalog { + SqlCatalog::new() + .with_table( + "requests", + Schema::new(vec![ + Field::plain("bytes", DataType::Int64, true), + Field::plain("status", DataType::Int64, false), + ]), + ) + .with_table( + "lineitem", + Schema::new(vec![Field::plain("l_quantity", DataType::Int64, false)]), + ) +} + +/// The SQL scalar example keeps column scopes and a Boolean row predicate. +#[tokio::test] +async fn sql_filter_projection_example() { + let root = lower_sql( + "SELECT l_quantity * 2 AS q2 FROM lineitem WHERE l_quantity > 10", + &catalog(), + AccuracyTarget::Exact, + ) + .await + .unwrap(); + root.validate_structure().unwrap(); + assert_eq!( + root.schema.fields[0], + Field::plain("q2", DataType::Int64, false) + ); + assert!( + matches!(root.expect_non_asap(),NonASAPOp::Project { cols,.. } if matches!(cols[0].expr,ScalarExpr::Arithmetic { .. })) + ); + let wire = physical_common::compile_post_asap_dag(&root).unwrap(); + wire.validate().unwrap(); + assert_eq!(wire.nodes.len(), OperatorNode::reachable(&root).len()); +} + +/// SUM's evaluation preserves integer type and SQL NULL behavior across the rewrite. +#[tokio::test] +async fn sql_sum_projection_before_and_after_summary_rewrite() { + let root = lower_sql( + "SELECT SUM(bytes) + 1 AS total_bytes FROM requests WHERE status = 200", + &catalog(), + AccuracyTarget::Exact, + ) + .await + .unwrap(); + root.validate_structure().unwrap(); + assert_eq!( + root.schema.fields[0], + Field::plain("total_bytes", DataType::Int64, true) + ); + fn rewrite(node: &Rc) -> Rc { + if let Some(NonASAPOp::Aggregate { + child, + reduction, + measures, + .. + }) = node.non_asap() + { + let [AggIntent::Sum { col: Some(column) }] = measures.as_slice() else { + panic!() + }; + let state = Rc::new( + OperatorNode::new(Operator::ASAP(ASAPOp::SummaryAgg { + child: Rc::clone(child), + family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + input: SummaryUpdate::column(ColumnRef::Named( + child.schema.fields[*column].name.clone(), + )), + reduction: reduction.clone(), + grouping: GroupingStrategy::default(), + filter: None, + })) + .unwrap(), + ); + let finalize = OperatorNode::asap_node( + ASAPOp::FinalizeExactAccumulator { child: state }, + node.schema.clone(), + None, + ); + return finalize; + } + Rc::new(node.map_children(rewrite).unwrap()) + } + let rewritten = rewrite(&root); + rewritten.validate_structure().unwrap(); + assert_eq!(rewritten.schema, root.schema); + let graph = OperatorNode::reachable(&rewritten); + assert!(graph + .iter() + .any(|n| matches!(n.asap(), Some(ASAPOp::SummaryAgg { .. })))); + assert!(graph + .iter() + .any(|n| matches!(n.asap(), Some(ASAPOp::FinalizeExactAccumulator { .. })))); + let wire = physical_common::compile_post_asap_dag(&rewritten).unwrap(); + wire.validate().unwrap(); + assert_eq!(wire.nodes.len(), graph.len()); + let json = serde_json::to_string(&wire).unwrap(); + assert!(!json.contains("KeepPreAsap") && !json.contains("ScalarBridge")); +} + +/// Scalar subqueries survive normalization with shared, visible producers. +#[tokio::test] +async fn sql_scalar_subquery_retains_its_cardinality_contract() { + for query in [ + "SELECT (SELECT bytes FROM requests) AS v FROM lineitem", + "SELECT l_quantity NOT IN (SELECT bytes FROM requests) AS present FROM lineitem", + ] { + let root = lower_sql(query, &catalog(), AccuracyTarget::Exact) + .await + .unwrap(); + root.validate_structure().unwrap(); + assert!(root.children().len() > 1); + let wire = physical_common::compile_post_asap_dag(&root).unwrap(); + assert!(wire + .edges + .iter() + .any(|e| e.role == asap_types::ir::export::EdgeRole::ScalarRef)); + } +} + +/// Execute the SQL SUM example for nonempty, empty and all-NULL populations. +#[tokio::test] +async fn sql_sum_example_executes_with_sql_null_semantics() { + use asap_physical_operators::{ + physical_planner::{compile, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + sources::{DataSources, MemorySource}, + values::{Batch, Value}, + }; + use futures::StreamExt; + use std::{collections::BTreeMap, sync::Arc}; + let root = lower_sql( + "SELECT SUM(bytes) + 1 AS total_bytes FROM requests WHERE status = 200", + &catalog(), + AccuracyTarget::Exact, + ) + .await + .unwrap(); + let logical_scan = OperatorNode::reachable(&root) + .into_iter() + .find(|node| matches!(node.non_asap(), Some(NonASAPOp::Scan { .. }))) + .unwrap(); + let NonASAPOp::Scan { source, .. } = logical_scan.expect_non_asap() else { + panic!() + }; + let wire = physical_common::compile_post_asap_dag(&root).unwrap(); + let scan = wire + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + asap_types::ir::export::PostAsapOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::Scan { .. } + } + ) + }) + .unwrap(); + let schema = Arc::new(scan.output_schema.clone()); + let plan = compile( + &wire, + BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(wire.root.0)], + ) + .unwrap(); + for (rows, expected) in [ + ( + vec![ + vec![Value::Int64(10), Value::Int64(200)], + vec![Value::Int64(20), Value::Int64(500)], + vec![Value::Null, Value::Int64(200)], + ], + Value::Int64(11), + ), + (vec![], Value::Null), + (vec![vec![Value::Null, Value::Int64(200)]], Value::Null), + ] { + let mut sources = DataSources::default(); + sources + .register( + source.clone(), + Arc::new( + MemorySource::new( + schema.clone(), + vec![Batch::try_new(schema.clone(), rows).unwrap()], + ) + .unwrap(), + ), + ) + .unwrap(); + let bound = plan + .instantiate(BTreeMap::from([( + u64::from(scan.id.0), + Box::new(sources.bind(&logical_scan).unwrap()) as Source<'_>, + )])) + .unwrap(); + let mut stream = bound + .execute( + plan.roots(), + RunContext::new( + Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(), + ) + .unwrap() + .remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend(batch.unwrap().rows().iter().cloned()); + } + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].len(), 1); + match (&rows[0][0], expected) { + (Value::Null, Value::Null) => {} + (Value::Int64(actual), Value::Int64(expected)) => assert_eq!(*actual, expected), + other => panic!("{other:?}"), + } + } +} + +/// Empty window frames and filtered groups can yield NULL even on non-NULL input. +#[tokio::test] +async fn sql_window_and_filtered_aggregate_types() { + for query in [ + "SELECT SUM(l_quantity) OVER (ORDER BY l_quantity ROWS BETWEEN 2 PRECEDING AND 1 PRECEDING) AS s FROM lineitem", + "SELECT MIN(l_quantity) OVER (ORDER BY l_quantity ROWS BETWEEN 2 PRECEDING AND 1 PRECEDING) AS s FROM lineitem", + "SELECT SUM(l_quantity) FILTER (WHERE l_quantity < 0) AS s FROM lineitem GROUP BY l_quantity", + ] { + let root = lower_sql(query, &catalog(), AccuracyTarget::Exact).await.unwrap(); + root.validate_structure().unwrap(); + assert_eq!(root.schema.fields[0], Field::plain("s", DataType::Int64, true), "{query}"); + } +} + +/// A real query batch selects one shared SUM producer, retains two result roots, +/// and executes both selected plans. No replacement graph is constructed by the test. +#[tokio::test] +async fn batch_planning_replaces_and_shares_summary_operators() { + use asap_aware_mapping::cost_model::{Cost, DefaultCostModel}; + use asap_aware_mapping::pass::PlanningModels; + use asap_aware_mapping::{ + CostModel, CostRate, LifecycleInput, SummaryMaintenanceLifecycleCapabilities, + SummaryMaintenanceLifecycleCostInputs, + }; + use asap_physical_operators::{ + physical_planner::{compile, InputContract}, + runtime::Scope, + values::{Batch, Value}, + }; + use asap_planner::{e2e_plan, FrontendInput, UserInput}; + use asap_types::post_asap::SketchAlgorithm; + use asap_types::workload::*; + use std::{collections::BTreeMap, sync::Arc}; + struct Costs; + impl CostModel for Costs { + fn rank_candidates( + &self, + intent: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + DefaultCostModel.rank_candidates(intent, candidates) + } + fn summary_maintenance_lifecycle_cost_inputs( + &self, + _: &OperatorNode, + ) -> SummaryMaintenanceLifecycleCostInputs { + SummaryMaintenanceLifecycleCostInputs { + build_cost: Some(Cost(1.0)), + maintenance_cost_per_update: Some(Cost::ZERO), + summary_read_cost: Some(Cost::ZERO), + retention_cost_rate: Some(CostRate(0.0)), + retirement_cost: Some(Cost::ZERO), + } + } + fn raw_query_recompute_cost(&self, _: &OperatorNode) -> Option { + Some(Cost(1000.0)) + } + } + let queries = [ + "SELECT SUM(bytes) + 1 AS result FROM requests", + "SELECT SUM(bytes) * 2 AS result FROM requests", + ]; + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::SQL(SqlDialect::DataFusionSQL), + query_batch: Some( + queries + .iter() + .map(|query| BatchEntry { + query: Query((*query).into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 2, + execute_at: None, + time_selection: TimeSelection::default(), + }) + .collect(), + ), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + arrival: DataArrival::AtRest, + ..Default::default() + }), + }; + let catalog = SqlCatalog::new().with_table( + "requests", + Schema::new(vec![Field::plain("bytes", DataType::Float64, false)]), + ); + let output = e2e_plan(UserInput::new( + &workload, + FrontendInput::Sql { catalog: &catalog }, + PlanningModels::builtin().with_cost(&Costs), + LifecycleInput::new(0, SummaryMaintenanceLifecycleCapabilities::default()), + )) + .await + .unwrap(); + assert_eq!(output.entry_indices(), [0, 1]); + assert_eq!(output.roots().len(), 2); + let states: Vec<_> = output + .operators() + .into_iter() + .filter(|n| matches!(n.asap(), Some(ASAPOp::SummaryAgg { .. }))) + .collect(); + assert_eq!(states.len(), 1, "the batch owns one shared SUM state"); + for (plan, expected) in output.plans.iter().zip([31.0, 60.0]) { + assert!(!plan.plan.selected_raw_recompute); + let root = &plan.plan.root; + root.validate_structure().unwrap(); + assert!(OperatorNode::reachable(root) + .iter() + .any(|n| Rc::ptr_eq(n, &states[0]))); + let wire = physical_common::compile_post_asap_dag(root).unwrap(); + let scan = wire + .nodes + .iter() + .find(|n| { + matches!( + n.payload, + asap_types::ir::export::PostAsapOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::Scan { .. } + } + ) + }) + .unwrap(); + let schema = Arc::new(scan.output_schema.clone()); + let program = compile( + &wire, + BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(wire.root.0)], + ) + .unwrap(); + let result = physical_common::execute( + &program, + BTreeMap::from([( + u64::from(scan.id.0), + Batch::try_new( + schema, + vec![vec![Value::Float64(10.0)], vec![Value::Float64(20.0)]], + ) + .unwrap(), + )]), + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + ); + let rows: Vec<_> = result[0].iter().flat_map(|batch| batch.rows()).collect(); + assert_eq!(rows.len(), 1); + assert!( + matches!(rows[0][0], Value::Float64(v) if v == expected), + "{:?}", + rows + ); + } +} diff --git a/crates/integration-tests/tests/operator_sharing.rs b/crates/integration-tests/tests/operator_sharing.rs new file mode 100644 index 000000000..6a87a879a --- /dev/null +++ b/crates/integration-tests/tests/operator_sharing.rs @@ -0,0 +1,195 @@ +//! Acceptance tests for operator sharing (issue #468): one operator IR +//! before and after ASAP optimization, so non-ASAP operators sit both above +//! and below summary operators, can share inputs with them, and can carry +//! summaries below set operators. +//! +//! Each test drives SQL text through `lower_sql` → `search_workload` → +//! global selection → `assemble_selected_dag`, the pipeline +//! `sql_to_post_asap.rs` uses. + +use std::rc::Rc; + +use asap_aware_mapping::{search_workload, DefaultCostModel}; +use asap_frontend_sql::{lower_sql, SqlCatalog}; +use asap_integration_tests::post_asap::post_asap_dag; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::types::AccuracyTarget; + +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) +} + +/// TPC-H `lineitem`, as `frontend-sql/tests/data_quality_check/tpch_deequ.rs` +/// declares it (DECIMAL columns as `Float64`, no time index, no keys). +fn catalog() -> SqlCatalog { + SqlCatalog::new().with_table( + "lineitem", + Schema::new(vec![ + col("l_orderkey", DataType::Int64), + col("l_partkey", DataType::Int64), + col("l_suppkey", DataType::Int64), + col("l_linenumber", DataType::Int64), + col("l_quantity", DataType::Float64), + col("l_extendedprice", DataType::Float64), + col("l_discount", DataType::Float64), + col("l_tax", DataType::Float64), + col("l_returnflag", DataType::Utf8), + col("l_linestatus", DataType::Utf8), + col("l_shipdate", DataType::Date), + col("l_commitdate", DataType::Date), + col("l_receiptdate", DataType::Date), + col("l_shipinstruct", DataType::Utf8), + col("l_shipmode", DataType::Utf8), + col("l_comment", DataType::Utf8), + ]), + ) +} + +/// Lower `sql`, search, select with the default cost model and assemble the +/// selected post-ASAP DAG. +async fn plan(sql: &str, accuracy: AccuracyTarget) -> Rc { + let pre = lower_sql(sql, &catalog(), accuracy) + .await + .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")); + let space = search_workload(vec![("query", pre)]); + let selection = space.global_selection(&DefaultCostModel); + selection + .assemble_selected_dag(&space.roots[0].1) + .expect("materialization failed") + .expect("root must be discovered") +} + +/// Every unique node reachable from `root` whose operator matches `pred`. +fn find_all( + root: &Rc, + pred: impl Fn(&OperatorNode) -> bool, +) -> Vec> { + OperatorNode::reachable(root) + .into_iter() + .filter(|node| pred(node)) + .collect() +} + +fn is_summary_evaluation(node: &OperatorNode) -> bool { + matches!( + node.operator, + Operator::ASAP( + ASAPOp::SummaryEstimate { .. } + | ASAPOp::FinalizeExactAccumulator { .. } + | ASAPOp::EvaluatePopulation { .. } + ) + ) +} + +fn is_scan(node: &OperatorNode) -> bool { + matches!(node.non_asap(), Some(NonASAPOp::Scan { .. })) +} + +/// The first node reached through single-input non-ASAP operators below +/// `node` (inclusive) that is not one: where an operator chain meets a +/// summary or a multi-input operator. +fn through_unary_non_asap(node: &Rc) -> &Rc { + match node.non_asap().map(|op| op.children()) { + Some(children) if children.len() == 1 => through_unary_non_asap(children[0]), + _ => node, + } +} + +// #468 problem 1: the Project above the summary evaluation and the Scan below +// it are both plain NonASAP nodes (no post-ASAP-only wrapper variant). +#[ignore = "planner chooses no summary here: Avg has no summary realization, so the Aggregate stays a logical pass-through"] +#[tokio::test] +async fn project_above_and_scan_below_a_summary_are_both_non_asap_nodes() { + let root = plan( + "WITH metric AS (SELECT avg(CASE WHEN l_quantity BETWEEN 1 AND 50 THEN 1.0 ELSE 0.0 END) \ + AS in_range FROM lineitem) SELECT in_range, in_range = 1.0 AS ok FROM metric", + AccuracyTarget::Exact, + ) + .await; + // The root is the outer SELECT list: a NonASAP Project. + assert!( + matches!(root.operator, Operator::NonASAP(NonASAPOp::Project { .. })), + "root must be the outer Project, got {:?}", + root.operator + ); + // A summary evaluation sits below the Project chain. + let evaluation = through_unary_non_asap(&root); + assert!( + is_summary_evaluation(evaluation), + "the Project chain must read a summary, got {:?}", + evaluation.operator + ); + // Below the summary the Scan is the same NonASAP operator a front end emits. + let scans = find_all(evaluation, is_scan); + assert_eq!(scans.len(), 1, "one lineitem Scan below the summary"); + assert!(!scans[0].is_asap()); + // The flat plan exports (time first, wire 6). + post_asap_dag(&root); +} + +// #468 problem 2: the exact aggregate and the sketch read one shared Scan +// (`Rc::ptr_eq`), not two copies. +#[ignore = "waits for the binding rule splitting multi-measure aggregates"] +#[tokio::test] +async fn exact_aggregate_and_sketch_share_one_scan() { + let root = plan( + "SELECT avg(l_extendedprice), approx_percentile_cont(l_discount, 0.99) FROM lineitem", + AccuracyTarget::Epsilon(0.01), + ) + .await; + let exact = find_all(&root, |node| { + matches!(node.non_asap(), Some(NonASAPOp::Aggregate { .. })) + }); + let sketch = find_all(&root, |node| { + matches!(node.asap(), Some(ASAPOp::SummaryAgg { .. })) + }); + assert_eq!(exact.len(), 1, "one exact Aggregate for avg: {root:?}"); + assert_eq!( + sketch.len(), + 1, + "one sketch SummaryAgg for the percentile: {root:?}" + ); + let scan_under = |node: &Rc| { + let scans = find_all(node, is_scan); + assert_eq!(scans.len(), 1, "one Scan under {:?}", node.operator); + Rc::clone(&scans[0]) + }; + assert!( + Rc::ptr_eq(&scan_under(&exact[0]), &scan_under(&sketch[0])), + "the exact aggregate and the sketch must read one shared Scan" + ); +} + +// #468 problem 3: a summary can sit below a set operator — each side of the +// UNION ALL holds its own SummaryEstimate. +#[tokio::test] +async fn each_side_of_union_all_holds_a_summary_estimate() { + let root = plan( + "SELECT approx_distinct(l_partkey) FROM lineitem \ + UNION ALL SELECT approx_distinct(l_suppkey) FROM lineitem", + AccuracyTarget::Epsilon(0.01), + ) + .await; + // The SQL front end lowers UNION ALL to `SetOp { all: true }`. + let Some(NonASAPOp::SetOp { + all: true, + left, + right, + .. + }) = root.non_asap() + else { + panic!("root must be the UNION ALL SetOp, got {:?}", root.operator) + }; + for side in [left, right] { + let estimates = find_all(side, |node| { + matches!(node.asap(), Some(ASAPOp::SummaryEstimate { .. })) + }); + assert!( + !estimates.is_empty(), + "UNION ALL side has no SummaryEstimate: {:?}", + side.operator + ); + } + post_asap_dag(&root); +} diff --git a/crates/integration-tests/tests/physical_common/mod.rs b/crates/integration-tests/tests/physical_common/mod.rs index aca4d4efa..ca4a825dd 100644 --- a/crates/integration-tests/tests/physical_common/mod.rs +++ b/crates/integration-tests/tests/physical_common/mod.rs @@ -7,6 +7,7 @@ use asap_physical_operators::{ use futures::{executor::block_on, StreamExt}; use std::collections::BTreeMap; +#[allow(dead_code)] pub fn execute( plan: &CompiledPhysicalDag, inputs: BTreeMap, @@ -40,3 +41,15 @@ pub fn execute( .await }) } + +#[allow(dead_code)] +pub fn compile_post_asap_dag( + root: &std::rc::Rc, +) -> Result> { + let root = asap_types::ir::apply_lifecycle_timings( + root, + &Default::default(), + &mut Default::default(), + )?; + Ok(asap_types::ir::export::compile_post_asap_dag(&root)?) +} diff --git a/crates/integration-tests/tests/precompute_raw_samples.rs b/crates/integration-tests/tests/precompute_raw_samples.rs index 2df295725..fad66970d 100644 --- a/crates/integration-tests/tests/precompute_raw_samples.rs +++ b/crates/integration-tests/tests/precompute_raw_samples.rs @@ -1,10 +1,14 @@ //! Planner-selected summaries over raw samples compile as precompute graphs //! and produce the same estimates as feeding their kernel sample by sample. +mod physical_common; +use asap_types::ir::export::{PostAsapDag, PostAsapOperatorPayload}; +use asap_types::ir::OperatorNode; +use physical_common::compile_post_asap_dag; use std::{collections::BTreeMap, collections::BTreeSet, rc::Rc, sync::Arc}; use asap_aware_mapping::cost_model::DefaultCostModel; use asap_aware_mapping::{ - search_workload, Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, + search_workload, ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_integration_tests::fixtures::lower_promql; @@ -18,10 +22,10 @@ use asap_physical_operators::{ AggregateCore, KeyByLabelValues, Statistic, }; use asap_types::post_asap::{ - compile_post_asap_dag, EntityIdentity, ExactKind, PostAsapDag, PostAsapOperatorPayload, - SketchAlgorithm, SketchQuery, SummaryFamilyType, SummaryInputExpr, SummaryNode, SummaryUpdate, + EntityIdentity, ExactKind, FieldDataType, SketchAlgorithm, SketchStatistic, SummaryInputExpr, + SummaryUpdate, }; -use asap_types::pre_asap::{expr_ir::ColumnRef, query_expr::Reduction}; +use asap_types::pre_asap::{expr_ir::ColumnRef, Reduction}; use asap_types::types::AccuracyTarget; use futures::{executor::block_on, StreamExt}; @@ -50,14 +54,14 @@ fn canonical(labels: &Series) -> Series { /// Every Planner candidate for `query`: the searched selection plus each /// summary replacement of the root. -fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { - let root = Rc::new(lower_promql(query, accuracy).expect("lowering failed")); - let mut result = SketchAlgorithmStrategy::default_cost_model() +fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { + let root = lower_promql(query, accuracy).expect("lowering failed"); + let mut result = ASAPStrategies::default_cost_model() .replacements(&TargetSubDAG::new(&root)) .into_iter() .filter_map(|candidate| match candidate { ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDag(node), .. } => Some(node), _ => None, @@ -88,8 +92,13 @@ fn raw_summaries(dag: &PostAsapDag) -> Vec<(u64, u64)> { return None; }; let source = dag.nodes.iter().find(|n| n.id == edge.producer)?; - matches!(source.payload, PostAsapOperatorPayload::Fallback { .. }) - .then_some((u64::from(source.id.0), u64::from(node.id.0))) + matches!( + source.payload, + PostAsapOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::TimeRange { .. } + } + ) + .then_some((u64::from(source.id.0), u64::from(node.id.0))) }) .collect() } @@ -236,9 +245,9 @@ fn weight(update: &SummaryUpdate, value: f64) -> f64 { } /// Estimates that identify a state's content for comparison. -fn readouts(state: &dyn AggregateCore, family: &SummaryFamilyType) -> Vec { +fn evaluations(state: &dyn AggregateCore, family: &FieldDataType) -> Vec { if let Some(exact) = state.as_any().downcast_ref::() { - let SummaryFamilyType::ExactAggregate(kind, _) = family else { + let FieldDataType::ExactAggregate(kind, _) = family else { unreachable!() }; let statistic = match kind { @@ -251,19 +260,19 @@ fn readouts(state: &dyn AggregateCore, family: &SummaryFamilyType) -> Vec { other => panic!("unexpected exact kind {other:?}"), }; return vec![exact - .readout(statistic, None, None::<&KeyByLabelValues>) + .evaluation(statistic, None, None::<&KeyByLabelValues>) .unwrap() .unwrap()]; } - let SummaryFamilyType::Sketch(kind, _) = family else { + let FieldDataType::Sketch(kind, _) = family else { panic!("sketch state for exact family") }; match kind.algorithm() { SketchAlgorithm::Kll | SketchAlgorithm::DDSketch => [0.1, 0.5, 0.9] .into_iter() - .map(|q| state.estimate(&SketchQuery::Quantile { q }).unwrap()) + .map(|q| state.estimate(&SketchStatistic::Quantile { q }).unwrap()) .collect(), - SketchAlgorithm::Hll => vec![state.estimate(&SketchQuery::Cardinality).unwrap()], + SketchAlgorithm::Hll => vec![state.estimate(&SketchStatistic::Cardinality).unwrap()], other => panic!("unexpected unkeyed sketch {other:?}"), } } @@ -306,7 +315,7 @@ fn check( .collect(), Reduction::PerEntity => vec![], }; - let stored_only = matches!(family, SummaryFamilyType::Sketch(kind, _) + let stored_only = matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &asap_types::post_asap::SketchAlgorithm::Cms); if stored_only || asap_physical_operators::capability::validate_native_family(family).is_err() { // Families without a native state (e.g. UnivMon), or with native @@ -317,11 +326,11 @@ fn check( } let actual = execute(dag, source, root, rows); let label = match family { - SummaryFamilyType::ExactAggregate(kind, _) => format!("{kind:?}"), - SummaryFamilyType::Sketch(kind, _) => format!("{:?}", kind.algorithm()), + FieldDataType::ExactAggregate(kind, _) => format!("{kind:?}"), + FieldDataType::Sketch(kind, _) => format!("{:?}", kind.algorithm()), other => format!("{other:?}"), }; - if let SummaryFamilyType::Sketch(kind, _) = family { + if let FieldDataType::Sketch(kind, _) = family { if let (Some(keyed), false) = (&input.item, kind.algorithm() == &SketchAlgorithm::Hll) { // Keyed heaps: every item's estimated weight is its exact // total at this scale (no collisions in the fixture). @@ -366,8 +375,8 @@ fn check( for (labels, state) in actual { let reference = expected[&labels].snapshot_accumulator(); assert_eq!( - readouts(state.as_ref(), family), - readouts(reference.as_ref(), family), + evaluations(state.as_ref(), family), + evaluations(reference.as_ref(), family), "{query}: {labels:?}" ); } @@ -455,7 +464,7 @@ fn raw_sample_summaries_compile_and_match_their_kernels() { /// Replace the raw summary of `sum by (service) (sum_over_time(m[5m]))` with /// another update, keeping its raw input and reduction. -fn grouped_raw_summary(family: SummaryFamilyType, input: SummaryUpdate) -> (PostAsapDag, u64, u64) { +fn grouped_raw_summary(family: FieldDataType, input: SummaryUpdate) -> (PostAsapDag, u64, u64) { let candidate = candidates( "sum by (service) (sum_over_time(m[5m]))", AccuracyTarget::Exact, @@ -504,7 +513,7 @@ fn raw_sample_heaps_resolve_items_from_labels() { SketchParams, WeightDomain, }; let heap = |algorithm, params| { - SummaryFamilyType::Sketch(SketchKind::new(algorithm, params), PerSubpopulationInstance) + FieldDataType::Sketch(SketchKind::new(algorithm, params), PerSubpopulationInstance) }; let cms = heap( SketchAlgorithm::CmsWithHeap, @@ -583,9 +592,9 @@ fn raw_sample_heaps_resolve_items_from_labels() { // `without` grouping over raw samples drops the listed labels and `__name__`. #[test] fn raw_sample_without_grouping_drops_labels_and_name() { - use asap_types::pre_asap::query_expr::GroupKeys; + use asap_types::pre_asap::GroupKeys; let family = - SummaryFamilyType::ExactAggregate(ExactKind::Sum, asap_types::post_asap::ExactParams::Sum); + FieldDataType::ExactAggregate(ExactKind::Sum, asap_types::post_asap::ExactParams::Sum); let (mut dag, source, root) = grouped_raw_summary(family, SummaryUpdate::column(ColumnRef::SampleValue)); let service = dag diff --git a/crates/integration-tests/tests/promql_numeric_regressions.rs b/crates/integration-tests/tests/promql_numeric_regressions.rs index bec86289d..97c378d26 100644 --- a/crates/integration-tests/tests/promql_numeric_regressions.rs +++ b/crates/integration-tests/tests/promql_numeric_regressions.rs @@ -1,44 +1,46 @@ -//! Numeric regression fixtures: actual PromQL lowering plus numeric update/readout checks. +//! Numeric regression fixtures: actual PromQL lowering plus numeric update/evaluation checks. //! The count/sum interpreter below verifies planner update semantics, not a deployed backend. -use asap_aware_mapping::{Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG}; +use asap_aware_mapping::replacement::is_logical_rewrite; +use asap_aware_mapping::{ASAPStrategies, Replacement, ReplacementStrategy, TargetSubDAG}; use asap_integration_tests::fixtures::lower_promql; -use asap_types::post_asap::{ - compile_post_asap_dag, ExactKind, SummaryExpr, SummaryFamilyType, SummaryInputExpr, - SummaryNode, SummaryUpdate, -}; +use asap_integration_tests::post_asap::post_asap_dag; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; +use asap_types::post_asap::{ExactKind, FieldDataType, SummaryInputExpr, SummaryUpdate}; use asap_types::pre_asap::{ColumnRef, Reduction}; use asap_types::types::AccuracyTarget; use std::rc::Rc; -fn plan(query: &str, accuracy: AccuracyTarget) -> Rc { - let pre = Rc::new(lower_promql(query, accuracy).unwrap()); - SketchAlgorithmStrategy::default_cost_model() +fn plan(query: &str, accuracy: AccuracyTarget) -> Rc { + let pre = lower_promql(query, accuracy).unwrap(); + ASAPStrategies::default_cost_model() .replacements(&TargetSubDAG::new(&pre)) .into_iter() .find_map(|r| match r.replacement { - Replacement::Summary(n) => Some(n), + // A bound decision: a summary DAG or a kept (exact) sub-DAG. + Replacement::SubDag(n) if !is_logical_rewrite(&n) => Some(n), _ => None, }) - .unwrap_or_else(|| asap_aware_mapping::replacement::keep_pre_asap(&pre).unwrap()) + .unwrap_or_else(|| asap_aware_mapping::replacement::retain_exact(&pre).unwrap()) } -fn aggregate(node: &SummaryNode) -> (&SummaryFamilyType, &SummaryUpdate, &Reduction) { - match &node.expr { - SummaryExpr::SummaryAgg { +fn aggregate(node: &OperatorNode) -> (&FieldDataType, &SummaryUpdate, &Reduction) { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { family, input, reduction, .. - } => (family, input, reduction), - SummaryExpr::SummaryEstimate { summary_input, .. } => aggregate(summary_input), - SummaryExpr::ValueOperation { child, .. } => aggregate(child), + }) => (family, input, reduction), + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => aggregate(summary_input), + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) => aggregate(child), + // A value operation (Project/Filter/Sort/Limit/...) over the state. + Operator::NonASAP(op) if op.children().len() == 1 && node.contains_asap() => { + aggregate(op.children()[0]) + } other => panic!("not a maintained accumulator: {other:?}"), } } -fn contribution(family: &SummaryFamilyType, update: &SummaryUpdate, value: f64) -> f64 { - if matches!( - family, - SummaryFamilyType::ExactAggregate(ExactKind::Count, _) - ) { +fn contribution(family: &FieldDataType, update: &SummaryUpdate, value: f64) -> f64 { + if matches!(family, FieldDataType::ExactAggregate(ExactKind::Count, _)) { return 1.; } match update.weight { @@ -55,7 +57,7 @@ fn count_up_counts_targets_even_when_values_repeat_or_change_sign() { let (family, update, _) = aggregate(&node); assert!(matches!( family, - SummaryFamilyType::ExactAggregate(ExactKind::Count, _) + FieldDataType::ExactAggregate(ExactKind::Count, _) )); for values in [[1., 1., 1.], [1., 1., 0.], [0., 0., 0.], [-1., -1., -1.]] { assert_eq!( @@ -66,7 +68,7 @@ fn count_up_counts_targets_even_when_values_repeat_or_change_sign() { 3. ); } - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); } /// Ten samples give count ten, whereas sum retains the signed sample values. @@ -78,13 +80,13 @@ fn window_counts_and_sums_distinguish_one_zero_three_and_negative_values() { ] { let node = plan(query, AccuracyTarget::Exact); let (family, update, reduction) = aggregate(&node); - assert!(matches!(family, SummaryFamilyType::ExactAggregate(k, _) if *k == kind)); + assert!(matches!(family, FieldDataType::ExactAggregate(k, _) if *k == kind)); assert_eq!(*reduction, Reduction::PerEntity); for value in [1., 0., 3., -3.] { let got: f64 = (0..10).map(|_| contribution(family, update, value)).sum(); assert_eq!(got, if is_count { 10. } else { value * 10. }); } - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); } } @@ -98,9 +100,9 @@ fn sum_rate_and_increase_have_real_exact_accumulator_nodes() { ] { let node = plan(query, AccuracyTarget::Exact); let (family, _, _) = aggregate(&node); - assert!(matches!(family, SummaryFamilyType::ExactAggregate(k, _) if *k == kind)); + assert!(matches!(family, FieldDataType::ExactAggregate(k, _) if *k == kind)); assert!(node.guarantee.as_ref().unwrap().is_exact()); - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); } } @@ -112,11 +114,12 @@ fn checked_ratio_must_not_certify_cross_zero_interpolation() { AccuracyTarget::Epsilon(0.01), ); assert!( - matches!(node.expr, SummaryExpr::BinaryOp { .. }), + matches!(node.operator, Operator::NonASAP(NonASAPOp::BinaryOp { .. })) + && node.contains_asap(), "direct quantile ratio should remain an available candidate" ); assert!(node.guarantee.is_none()); - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); // Keep the actual signed-sketch counterexample: division guards alone pass // even though the quantile interpolation does not preserve relative error. let alpha = (0.01 - 8.0 * f64::EPSILON) / 2.01; @@ -151,20 +154,20 @@ fn quantile_over_temporal_average_keeps_a_legal_candidate() { ] { let node = plan(query, AccuracyTarget::Epsilon(0.01)); assert!( - !matches!(node.expr, SummaryExpr::KeepPreAsap(_)), + node.contains_asap(), "outer sketch candidate must survive: {query}" ); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { - panic!("outer sketch readout") + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator else { + panic!("outer sketch evaluation") }; - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary_input.operator else { panic!("outer sketch state") }; assert!( - matches!(child.expr, SummaryExpr::KeepPreAsap(_)), + !child.contains_asap(), "guarded expression must retain native maintenance input" ); - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); } } @@ -173,8 +176,8 @@ impl asap_aware_mapping::accuracy::AccuracyEvidenceProvider for OneKeyTopKEviden fn propagation_stats( &self, op: &asap_types::post_asap::CompositionOperator, - _family: &SummaryFamilyType, - _query: Option<&asap_types::post_asap::SketchQuery>, + _family: &FieldDataType, + _query: Option<&asap_types::post_asap::SketchStatistic>, ) -> asap_aware_mapping::accuracy::PropagationStats { // Single-key fixture: no excluded keys; bounds cover every value below. if matches!( @@ -198,7 +201,7 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { use asap_aware_mapping::accuracy::{DefaultAccuracyModel, EqualSplitAllocator}; use asap_aware_mapping::cost_model::DefaultCostModel; use asap_types::post_asap::{NonNegativeWeightProof, SketchAlgorithm, WeightDomain}; - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -210,7 +213,7 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { } else { "topk(1, sum_over_time(up[5m]))" }; - let pre = Rc::new(lower_promql(query, AccuracyTarget::Epsilon(0.01)).unwrap()); + let pre = lower_promql(query, AccuracyTarget::Epsilon(0.01)).unwrap(); let candidates = strategy.replacements(&TargetSubDAG::new(&pre)); let wanted = if is_count { SketchAlgorithm::CmsWithHeap @@ -220,11 +223,11 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { let node = candidates .iter() .find_map(|c| { - let Replacement::Summary(node) = &c.replacement else { + let Replacement::SubDag(node) = &c.replacement else { return None; }; let (family, _, _) = aggregate(node); - matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &wanted) + matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &wanted) .then_some(node) }) .expect("weighted sketch candidate"); @@ -243,9 +246,9 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { SummaryInputExpr::Column(ColumnRef::SampleValue) ); for c in &candidates { - if let Replacement::Summary(n) = &c.replacement { + if let Replacement::SubDag(n) = &c.replacement { assert!( - !matches!(aggregate(n).0, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::CmsWithHeap) + !matches!(aggregate(n).0, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::CmsWithHeap) ); } } @@ -274,6 +277,6 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { }; assert_eq!(got, if is_count { 10. } else { 10. * value }); } - compile_post_asap_dag(node).unwrap(); + post_asap_dag(node); } } diff --git a/crates/integration-tests/tests/promql_to_post_asap.rs b/crates/integration-tests/tests/promql_to_post_asap.rs index 1a8d364e0..500240988 100644 --- a/crates/integration-tests/tests/promql_to_post_asap.rs +++ b/crates/integration-tests/tests/promql_to_post_asap.rs @@ -1,8 +1,8 @@ //! End-to-end query-string → post-ASAP IR pin (issue #98). //! -//! Drives the full pipeline — PromQL text → pre-ASAP `QueryExpr` -//! (`lower_promql`) → post-ASAP `SummaryExpr` DAG (via -//! `SketchAlgorithmStrategy::replacements`, see [`realize`] below) — and pins +//! Drives the full pipeline — PromQL text → non-ASAP `OperatorNode` +//! (`lower_promql`) → post-ASAP `OperatorNode` DAG (via +//! `ASAPStrategies::replacements`, see [`realize`] below) — and pins //! the summary-bound shape node by node, including the family `(Kind, //! Params)` committed on each edge's schema. @@ -13,67 +13,77 @@ use asap_aware_mapping::accuracy::{ QuantileInputDomain, }; use asap_aware_mapping::cost_model::DefaultCostModel; -use asap_aware_mapping::replacement::{keep_pre_asap, RealizationError}; +use asap_aware_mapping::replacement::{is_logical_rewrite, retain_exact, RealizationError}; use asap_aware_mapping::{ - search_workload, search_workload_with_targets, AccuracyModel, Replacement, ReplacementStrategy, - ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, + search_workload, search_workload_with_targets, ASAPStrategies, AccuracyModel, Replacement, + ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_integration_tests::fixtures::lower_promql; +use asap_integration_tests::post_asap::{post_asap_dag, timed}; +use asap_types::ir::export::{NonASAPOpKind, PostAsapOperatorPayload}; +use asap_types::ir::operator_properties::Reduction; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, ScalarExpr}; use asap_types::post_asap::{ - compile_post_asap_dag, CompositionOperator, EntityIdentity, ExactKind, ExactParams, - GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, SummaryExpr, - SummaryFamilyType, SummaryInputExpr, SummaryNode, SummarySchema, SummaryUpdate, ValueOperation, + CompositionOperator, EntityIdentity, ExactKind, ExactParams, FieldDataType, GroupingStrategy, + Schema, SketchAlgorithm, SketchKind, SketchParams, SketchStatistic, SummaryInputExpr, + SummaryUpdate, }; use asap_types::pre_asap::expr_ir::ColumnRef; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; use asap_types::pre_asap::schema::DataType; use asap_types::types::AccuracyTarget; /// This crate has no "bind me one tree" public API any more — -/// `SketchAlgorithmStrategy::replacements` always returns every candidate, and +/// `ASAPStrategies::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so the /// single-answer pins below don't all repeat it by hand. -fn realize(expr: &QueryExpr) -> Result, RealizationError> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - match SketchAlgorithmStrategy::default_cost_model() +fn realize(root: &Rc) -> Result, RealizationError> { + let target = TargetSubDAG::new(root); + match ASAPStrategies::default_cost_model() .replacements(&target) .into_iter() .next() { + // A bound decision (summary DAG or kept sub-DAG); a logical rewrite + // is not a binding, so it falls back to keeping the target. Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDag(node), .. - }) => Ok(node), - _ => keep_pre_asap(&root), + }) if !is_logical_rewrite(&node) => Ok(node), + _ => retain_exact(root), } + .inspect(|node| { + node.validate_structure() + .expect("planned graph satisfies the unified IR contract") + }) } #[test] -fn distinct_over_time_offers_hll_cardinality_readout() { +fn distinct_over_time_offers_hll_cardinality_evaluation() { // The real frontend must reach an existing HLL candidate without a // function-specific post-ASAP node or a sample-count rewrite. - let root = Rc::new( - lower_promql( - "distinct_over_time(cpu_usage{job=\"worker\"}[5m])", - AccuracyTarget::Epsilon(0.02), - ) - .unwrap(), - ); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + let root = lower_promql( + "distinct_over_time(cpu_usage{job=\"worker\"}[5m])", + AccuracyTarget::Epsilon(0.02), + ) + .unwrap(); + let candidates = ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); + for candidate in &candidates { + if let Replacement::SubDag(node) = &candidate.replacement { + node.validate_structure().unwrap(); + } + } assert!(candidates.iter().any(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { return false }; - let SummaryExpr::SummaryEstimate { summary_input, query, .. } = &node.expr else { return false }; - matches!(query, SketchQuery::Cardinality) - && matches!(&summary_input.expr, SummaryExpr::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } + let Replacement::SubDag(node) = &candidate.replacement else { return false }; + let Some(ASAPOp::SummaryEstimate { summary_input, query, .. }) = node.asap() else { return false }; + matches!(query, SketchStatistic::Cardinality) + && matches!(summary_input.asap(), Some(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) if kind.algorithm() == &SketchAlgorithm::Hll) }), "no HLL cardinality candidate: {candidates:?}"); } -fn lower_search_and_materialize(query: &str) -> Rc { - let pre = Rc::new(lower_promql(query, AccuracyTarget::Exact).expect("lowering failed")); +fn lower_search_and_materialize(query: &str) -> Rc { + let pre = lower_promql(query, AccuracyTarget::Exact).expect("lowering failed"); let space = search_workload(vec![("query", pre)]); let selection = space.global_selection(&DefaultCostModel); selection @@ -88,41 +98,43 @@ fn value_ranked_topk_preserves_summary_children_in_post_asap_dag() { "topk(3, rate(cpu_seconds_total[5m]))", "topk by (job) (2, max_over_time(memory_bytes[6h]))", ] { - let root = lower_search_and_materialize(query); - let SummaryExpr::ValueOperation { - operation: ValueOperation::Limit { n, offset, .. }, + let root = timed(&lower_search_and_materialize(query)); + let Some(NonASAPOp::Limit { + n, + offset, child: sort, .. - } = &root.expr + }) = root.non_asap() else { - panic!("expected query-time Limit for {query}, got {:?}", root.expr); + panic!( + "expected query-time Limit for {query}, got {:?}", + root.operator + ); }; - assert!(*n > 0 && *offset == 0); - let SummaryExpr::ValueOperation { - operation: ValueOperation::Sort { .. }, - child, - .. - } = &sort.expr - else { + assert_eq!( + root.timing, + Some(asap_types::post_asap::ExecutionTiming::QueryTime) + ); + assert!(n.is_some_and(|n| n > 0) && *offset == 0); + let Some(NonASAPOp::Sort { child, .. }) = sort.non_asap() else { panic!("expected query-time Sort under Limit for {query}"); }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - child: state, - .. - } = &child.expr - else { + assert_eq!( + sort.timing, + Some(asap_types::post_asap::ExecutionTiming::QueryTime) + ); + let Some(ASAPOp::FinalizeExactAccumulator { child: state }) = child.asap() else { panic!( "Sort must consume finalized values for {query}: {:?}", - child.expr + child.operator ); }; - assert!(matches!(state.expr, SummaryExpr::SummaryAgg { .. })); + assert!(matches!(state.asap(), Some(ASAPOp::SummaryAgg { .. }))); assert!(child .schema .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); } } @@ -135,9 +147,9 @@ fn exact_counter_weighted_topk_fails_closed_without_membership_certificate() { ] { let root = lower_search_and_materialize(query); assert!( - matches!(root.expr, SummaryExpr::KeepPreAsap(_)), + !root.contains_asap(), "exact target must not accept an uncertified membership sidecar for {query}: {:?}", - root.expr + root.operator ); } } @@ -146,29 +158,20 @@ fn exact_counter_weighted_topk_fails_closed_without_membership_certificate() { fn instant_topk_and_unsupported_child_remain_local_residuals() { for query in ["topk(3, memory_bytes)", "topk(3, deriv(memory_bytes[5m]))"] { let root = lower_search_and_materialize(query); - let SummaryExpr::ValueOperation { - operation: ValueOperation::Limit { .. }, - child: sort, - .. - } = &root.expr - else { + let Some(NonASAPOp::Limit { child: sort, .. }) = root.non_asap() else { panic!("expected Limit for {query}"); }; - let SummaryExpr::ValueOperation { - child, operation, .. - } = &sort.expr - else { + let Some(NonASAPOp::Sort { child, .. }) = sort.non_asap() else { panic!("expected Sort for {query}"); }; - assert!(matches!(operation, ValueOperation::Sort { .. })); assert!( - matches!(child.expr, SummaryExpr::KeepPreAsap(_)), + !child.contains_asap(), "only the unsupported child should remain exact for {query}" ); } } -fn dtype<'a>(schema: &'a SummarySchema, name: &str) -> &'a SummaryFamilyType { +fn dtype<'a>(schema: &'a Schema, name: &str) -> &'a FieldDataType { &schema .fields .iter() @@ -177,7 +180,7 @@ fn dtype<'a>(schema: &'a SummarySchema, name: &str) -> &'a SummaryFamilyType { .dtype } -fn lower_and_realize(query: &str) -> Rc { +fn lower_and_realize(query: &str) -> Rc { let pre = lower_promql(query, AccuracyTarget::Exact).expect("lowering failed"); realize(&pre).expect("binding failed") } @@ -186,19 +189,17 @@ fn lower_and_realize(query: &str) -> Rc { fn promql_binary_arithmetic_retains_two_summary_leaves() { for op in ["+", "-", "*", "/", "%", "^", "atan2"] { let root = lower_and_realize(&format!("rate(a[1m]) {op} rate(b[1m])")); - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &root.expr else { - panic!("expected BinaryOp for {op}, got {:?}", root.expr); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = root.non_asap() else { + panic!("expected BinaryOp for {op}, got {:?}", root.operator); }; for operand in [lhs, rhs] { - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - .. - } = &operand.expr - else { - panic!("expected an explicit exact readout, got {:?}", operand.expr); + let Some(ASAPOp::FinalizeExactAccumulator { child }) = operand.asap() else { + panic!( + "expected an explicit exact evaluation, got {:?}", + operand.operator + ); }; - assert!(matches!(child.expr, SummaryExpr::SummaryAgg { .. })); + assert!(matches!(child.asap(), Some(ASAPOp::SummaryAgg { .. }))); } } } @@ -207,55 +208,44 @@ fn promql_binary_arithmetic_retains_two_summary_leaves() { fn value_ranked_topk_over_binary_ratio_finalizes_both_summary_operands() { let query = "topk(1, sum by(job)(increase(a[6h])) / sum by(job)(increase(b[6h])))"; let root = lower_search_and_materialize(query); - let SummaryExpr::ValueOperation { - operation: ValueOperation::Limit { - n: 1, offset: 0, .. - }, + let Some(NonASAPOp::Limit { + n: Some(1), + offset: 0, child: sort, .. - } = &root.expr + }) = root.non_asap() else { - panic!("expected Limit root, got {:?}", root.expr); + panic!("expected Limit root, got {:?}", root.operator); }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::Sort { .. }, - child: binary, - .. - } = &sort.expr - else { - panic!("expected Sort below Limit, got {:?}", sort.expr); + let Some(NonASAPOp::Sort { child: binary, .. }) = sort.non_asap() else { + panic!("expected Sort below Limit, got {:?}", sort.operator); }; - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &binary.expr else { - panic!("expected BinaryOp below Sort, got {:?}", binary.expr); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = binary.non_asap() else { + panic!("expected BinaryOp below Sort, got {:?}", binary.operator); }; for operand in [lhs, rhs] { - let SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - child, - .. - } = &operand.expr - else { + let Some(ASAPOp::FinalizeExactAccumulator { child }) = operand.asap() else { panic!( "expected exact accumulator finalization, got {:?}", - operand.expr + operand.operator ); }; - assert!(matches!(child.expr, SummaryExpr::SummaryAgg { .. })); + assert!(matches!(child.asap(), Some(ASAPOp::SummaryAgg { .. }))); } } struct SeparatedTopK; impl AccuracyEvidenceProvider for SeparatedTopK { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &OperatorNode) -> Option { Some(1000) } fn propagation_stats( &self, op: &CompositionOperator, - _family: &SummaryFamilyType, - _query: Option<&SketchQuery>, + _family: &FieldDataType, + _query: Option<&SketchStatistic>, ) -> PropagationStats { matches!(op, CompositionOperator::TopKSelection) .then_some(PropagationStats { @@ -271,14 +261,12 @@ impl AccuracyEvidenceProvider for SeparatedTopK { // Rate-weighted summaries must consume finalized rates, never raw counter deltas. #[test] fn grouped_rate_topk_consumes_finalized_rate_values() { - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -288,24 +276,31 @@ fn grouped_rate_topk_consumes_finalized_rate_values() { .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CmsWithHeap") => Some(node), + Replacement::SubDag(node) if candidate.rationale.contains("CmsWithHeap") => Some(node), _ => None, }) .expect("rate-weighted CMS plan"); - let dag = compile_post_asap_dag(&plan).unwrap(); + let dag = post_asap_dag(&plan); assert!(!dag.nodes.iter().any(|node| matches!( node.payload, - asap_types::post_asap::PostAsapOperatorPayload::RelationalJoin { .. } + PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::Join { .. } + } ))); - let node = dag.nodes.iter().find(|node| matches!(&node.payload, - asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } - if kind.algorithm() == &SketchAlgorithm::CmsWithHeap)).unwrap(); + let node = dag + .nodes + .iter() + .find(|node| { + matches!(&node.payload, + PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } + if kind.algorithm() == &SketchAlgorithm::CmsWithHeap) + }) + .unwrap(); assert_eq!( node.output_state.timing, asap_types::post_asap::ExecutionTiming::QueryTime ); - let asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { input, .. } = &node.payload - else { + let PostAsapOperatorPayload::SummaryAgg { input, .. } = &node.payload else { unreachable!() }; assert_eq!( @@ -326,20 +321,18 @@ fn weighted_topk_keeps_candidates_with_missing_population_evidence() { fn propagation_stats( &self, op: &CompositionOperator, - family: &SummaryFamilyType, - query: Option<&SketchQuery>, + family: &FieldDataType, + query: Option<&SketchStatistic>, ) -> PropagationStats { SeparatedTopK.propagation_stats(op, family, query) } } - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -355,27 +348,24 @@ fn weighted_topk_keeps_candidates_with_missing_population_evidence() { // Unknown requirements must survive physical export for deployment to inspect. #[test] fn weighted_topk_exports_symbolic_evidence_requirements() { - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - ) - .unwrap(), - ); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ) + .unwrap(); + let candidates = ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); let candidate = candidates .iter() .find(|candidate| candidate.rationale.contains("CmsWithHeap")) .unwrap(); assert!(candidate.has_missing_accuracy_evidence()); - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDag(node) = &candidate.replacement else { panic!("summary candidate") }; - let dag = compile_post_asap_dag(node).unwrap(); + let dag = post_asap_dag(node); let exported = serde_json::to_string(&dag).unwrap(); assert!(exported.contains("topk_max_distinct_items")); assert!(exported.contains("topk_membership_margin")); @@ -387,18 +377,16 @@ fn weighted_topk_exports_symbolic_evidence_requirements() { fn weighted_topk_rejects_invalid_population_evidence() { struct InvalidPopulation; impl AccuracyEvidenceProvider for InvalidPopulation { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &OperatorNode) -> Option { Some(0) } } - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -415,17 +403,15 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { "topk by(job)(2, sum by(service, job)(rate(m[1m])))", "topk(2, sum by(job)(increase(m[6h])))", ] { - let root = Rc::new( - lower_promql( - query, - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let root = lower_promql( + query, + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -435,60 +421,50 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CmsWithHeap") => { + Replacement::SubDag(node) if candidate.rationale.contains("CmsWithHeap") => { Some(node) } _ => None, }) .expect("weighted summary"); - let SummaryExpr::ValueOperation { + let Some(NonASAPOp::Limit { + n: Some(2), + offset: 0, + partition_by, child: sorted, - operation: - ValueOperation::Limit { - n: 2, - offset: 0, - partition_by, - }, - .. - } = &plan.expr + }) = plan.non_asap() else { panic!("grouped limit") }; - let SummaryExpr::ValueOperation { + let Some(NonASAPOp::Sort { + partition_by: sort_groups, child: projected, - operation: - ValueOperation::Sort { - partition_by: sort_groups, - .. - }, .. - } = &sorted.expr + }) = sorted.non_asap() else { panic!("grouped sort") }; assert_eq!(partition_by, sort_groups); assert_eq!(partition_by.len(), usize::from(query.contains("topk by"))); - let SummaryExpr::ValueOperation { - child: readout, - operation: ValueOperation::Project { .. }, - .. - } = &projected.expr + let Some(NonASAPOp::Project { + child: evaluation, .. + }) = projected.non_asap() else { panic!("logical output projection") }; - let SummaryExpr::SummaryEstimate { + let Some(ASAPOp::SummaryEstimate { summary_input, - query: SketchQuery::TopK { k }, - } = &readout.expr + query: SketchStatistic::TopK { k }, + }) = evaluation.asap() else { - panic!("heap readout") + panic!("heap evaluation") }; assert!(*k > 2, "candidate capacity is independent of output count"); - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { child: rates, input, .. - } = &summary_input.expr + }) = summary_input.asap() else { panic!("weighted summary") }; @@ -497,13 +473,10 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { SummaryInputExpr::Column(ColumnRef::SampleValue) ); assert!(matches!( - rates.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } + rates.asap(), + Some(ASAPOp::FinalizeExactAccumulator { .. }) )); - let dag = compile_post_asap_dag(&plan).unwrap(); + let dag = post_asap_dag(&plan); for phase in [ asap_types::post_asap::ExecutionTiming::IngestionTime, asap_types::post_asap::ExecutionTiming::QueryTime, @@ -525,57 +498,41 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { #[test] fn promql_binary_arithmetic_preserves_both_scalar_operand_orders() { - fn is_exact_readout_or_scalar(node: &SummaryNode) -> bool { - matches!(node.expr, SummaryExpr::KeepPreAsap(_)) - || matches!( - node.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } - ) - } - for query in ["rate(a[1m]) / 2", "2 / rate(a[1m])"] { + for (query, scalar_left) in [("rate(a[1m]) / 2", false), ("2 / rate(a[1m])", true)] { let root = lower_and_realize(query); - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &root.expr else { - panic!("expected BinaryOp for {query}, got {:?}", root.expr); + let Some(NonASAPOp::Project { cols, .. }) = root.non_asap() else { + panic!("expected Project") }; - assert!(is_exact_readout_or_scalar(lhs)); - assert!(is_exact_readout_or_scalar(rhs)); - assert!( - matches!( - lhs.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } - ) || matches!( - rhs.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } - ) - ); + let ScalarExpr::Arithmetic { left, right, .. } = &cols[1].expr else { + panic!() + }; + let (scalar, sample) = if scalar_left { + (left, right) + } else { + (right, left) + }; + assert_eq!(**scalar, ScalarExpr::literal_f64(2.0)); + assert_eq!(**sample, ScalarExpr::Column(1)); + assert!(root.schema.has_promql_series_identity()); } } #[test] fn promql_binary_arithmetic_falls_back_as_a_whole_for_unsupported_arm() { let root = lower_and_realize("rate(a[1m]) + stddev_over_time(b[1m])"); - assert!(matches!(root.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!root.contains_asap()); } #[test] fn promql_binary_arithmetic_preserves_nested_structure_and_rejects_modifiers() { let nested = lower_and_realize("(rate(a[1m]) + rate(b[1m])) / 2"); - let SummaryExpr::BinaryOp { lhs, .. } = &nested.expr else { - panic!("expected outer BinaryOp, got {:?}", nested.expr); + let Some(NonASAPOp::Project { child: lhs, .. }) = nested.non_asap() else { + panic!("expected outer BinaryOp, got {:?}", nested.operator); }; - assert!(matches!(lhs.expr, SummaryExpr::BinaryOp { .. })); + assert!(matches!(lhs.non_asap(), Some(NonASAPOp::BinaryOp { .. }))); let modified = lower_and_realize("rate(a[1m]) + on(job) rate(b[1m])"); - assert!(matches!(modified.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!modified.contains_asap()); } #[test] @@ -586,8 +543,8 @@ fn promql_binary_arithmetic_never_relabels_approximate_children_as_exact() { ) .expect("lowering failed"); let root = realize(&pre).expect("binding failed"); - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &root.expr else { - panic!("expected BinaryOp, got {:?}", root.expr); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = root.non_asap() else { + panic!("expected BinaryOp, got {:?}", root.operator); }; assert!(lhs.guarantee.as_ref().is_some_and(|g| !g.is_exact())); assert!(rhs.guarantee.as_ref().is_some_and(|g| !g.is_exact())); @@ -606,13 +563,11 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { epsilon: 0.01, delta: 0.01, }; - let query = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - target.clone(), - ) - .expect("lowering failed"), - ); + let query = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + target.clone(), + ) + .expect("lowering failed"); let evidence = FixtureQuantileDomain { lower: 1.0, @@ -632,7 +587,7 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { .for_target(root) .and_then(|selection| selection.chosen.as_ref()) .expect("the certified DDSketch ratio should be selectable"); - let Replacement::Summary(node) = &chosen.replacement else { + let Replacement::SubDag(node) = &chosen.replacement else { panic!("expected a summary candidate") }; let guarantee = node.guarantee.as_ref().expect("ratio guarantee"); @@ -642,23 +597,22 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { "ratio guarantee should satisfy the requested target: {guarantee:?}" ); - let shared = - asap_types::post_asap::share_common_summary_subtrees(vec![("ratio", node.clone())]); - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &shared[0].1.expr else { + let shared = asap_types::ir::cse::share_common_subdags(vec![("ratio", node.clone())]); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = shared[0].1.non_asap() else { panic!("expected binary ratio") }; - let producer = |readout: &Rc| match &readout.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => Rc::clone(summary_input), - other => panic!("expected DDSketch readout, got {other:?}"), + let producer = |evaluation: &Rc| match &evaluation.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => Rc::clone(summary_input), + other => panic!("expected DDSketch evaluation, got {other:?}"), }; assert!( Rc::ptr_eq(&producer(lhs), &producer(rhs)), - "the two quantile readouts should share one DDSketch producer" + "the two quantile evaluations should share one DDSketch producer" ); } #[test] -fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() { +fn planner_only_e2e_temporal_topk_preserves_query_update_and_evaluation_contract() { // Self-contained Planner E2E: each case starts from PromQL text and ends // at the post-ASAP summary DAG. No controller/backend types, // fixtures, configuration, or runtime are involved. @@ -677,17 +631,15 @@ fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() ), ]; for (source, expected_update, expected_family, excluded_labels) in cases { - let pre = Rc::new( - lower_promql( - source, - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - ) - .expect("lower temporal Top-K"), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let pre = lower_promql( + source, + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ) + .expect("lower temporal Top-K"); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -697,33 +649,33 @@ fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() .replacements(&TargetSubDAG::new(&pre)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains(expected_family) => { + Replacement::SubDag(node) if candidate.rationale.contains(expected_family) => { Some(node) } _ => None, }) .expect("heap-backed temporal Top-K candidate"); - let SummaryExpr::SummaryEstimate { + let Some(ASAPOp::SummaryEstimate { summary_input, - query: SketchQuery::TopK { k, .. }, - } = &candidate.expr + query: SketchStatistic::TopK { k, .. }, + }) = candidate.asap() else { - panic!("expected Top-K estimate, got {:?}", candidate.expr) + panic!("expected Top-K estimate, got {:?}", candidate.operator) }; assert_eq!( *k, 5, "the requested Top-K cardinality must survive binding" ); - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { input: state_input, family, child, .. - } = &summary_input.expr + }) = summary_input.asap() else { panic!("expected structured Top-K state input") }; - let SummaryFamilyType::Sketch(kind, _) = family else { + let FieldDataType::Sketch(kind, _) = family else { panic!("expected a heap-backed sketch family, got {family:?}") }; assert_eq!(format!("{:?}", kind.algorithm()), expected_family); @@ -742,42 +694,41 @@ fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() )) ); assert_eq!(state_input.weight, expected_update); - assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!child.contains_asap()); } } /// Execute the ungrouped temporal TopK subset with exact state. This tests /// the emitted update contract, not sketch approximation or backend execution. -fn execute_topk_reference(plan: &SummaryNode) -> Vec<(String, f64)> { +fn execute_topk_reference(plan: &OperatorNode) -> Vec<(String, f64)> { use std::collections::BTreeMap; - let SummaryExpr::SummaryEstimate { + let Some(ASAPOp::SummaryEstimate { summary_input, - query: SketchQuery::TopK { k }, - } = &plan.expr + query: SketchStatistic::TopK { k }, + }) = plan.asap() else { - panic!("expected TopK readout") + panic!("expected TopK evaluation") }; - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { input, child, reduction, .. - } = &summary_input.expr + }) = summary_input.asap() else { panic!("expected summary updates") }; assert_eq!(reduction, &Reduction::by(vec![])); - let SummaryExpr::KeepPreAsap(raw) = &child.expr else { - panic!("expected fused raw input") - }; - let QueryExpr::TimeRange { range, child } = raw.as_ref() else { + // The fused raw input is the kept non-ASAP sub-DAG itself. + assert!(!child.contains_asap(), "expected fused raw input"); + let Some(NonASAPOp::TimeRange { range, child, .. }) = child.non_asap() else { panic!("expected temporal input") }; - let QueryExpr::Scan { + let Some(NonASAPOp::Scan { source: asap_types::pre_asap::Source::TimeSeries { metric }, predicates, .. - } = child.as_ref() + }) = child.non_asap() else { panic!("expected metric scan") }; @@ -849,17 +800,15 @@ fn planner_heap_topk_reference_execution_matches_ground_truth() { vec![("worker", 100.0), ("cron", 30.0)], ), ] { - let pre = Rc::new( - lower_promql( - query, - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let pre = lower_promql( + query, + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -868,10 +817,10 @@ fn planner_heap_topk_reference_execution_matches_ground_truth() { // This reference executor consumes keyed heap updates. The inventory // also contains maintained exact values followed by sort/limit; those // have a different execution contract and must not enter this fixture. - let candidates: Vec<_> = strategy.replacements(&TargetSubDAG::new(&pre)).into_iter().filter(|candidate| matches!(&candidate.replacement, Replacement::Summary(plan) if matches!(plan.expr, SummaryExpr::SummaryEstimate { query: SketchQuery::TopK { .. }, .. }))).collect(); + let candidates: Vec<_> = strategy.replacements(&TargetSubDAG::new(&pre)).into_iter().filter(|candidate| matches!(&candidate.replacement, Replacement::SubDag(plan) if matches!(plan.asap(), Some(ASAPOp::SummaryEstimate { query: SketchStatistic::TopK { .. }, .. })))).collect(); assert!(!candidates.is_empty(), "no heap candidate for {query}"); for candidate in candidates { - let Replacement::Summary(plan) = candidate.replacement else { + let Replacement::SubDag(plan) = candidate.replacement else { panic!("expected summary plan for {query}") }; let expected: Vec<_> = expected @@ -889,11 +838,11 @@ fn planner_heap_topk_reference_execution_matches_ground_truth() { /// SummaryEstimate { query: Quantile{0.99} } → {quantile_0_99: Float64} /// └─ SummaryAgg { Kll{k:269}, input: SampleValue } → {value: Sketch(Kll, {k:269})} /// └─ SummaryAgg { Rate, input: SampleValue } → {ts, value: ExactAggregate(Rate), …} -/// └─ KeepPreAsap(TimeRange{5m} → Scan) → {ts, value} +/// └─ TimeRange{5m} → Scan → {ts, value} /// ``` /// /// The nested tree exercises both realizations: the approximate quantile -/// binds a KLL sketch + readout; the per-series `rate` binds the exact +/// binds a KLL sketch + evaluation; the per-series `rate` binds the exact /// counter-reset-aware accumulator (no estimate — its state is the value). #[test] fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { @@ -904,18 +853,18 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { .expect("lowering failed"); let root = realize(&pre_asap).expect("binding failed"); - // Root: the sketch readout, back to a plain row shape. - let SummaryExpr::SummaryEstimate { + // Root: the sketch evaluation, back to a plain row shape. + let Some(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = root.asap() else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; - assert!(matches!(query, SketchQuery::Quantile { q } if *q == 0.99)); + assert!(matches!(query, SketchStatistic::Quantile { q } if *q == 0.99)); assert_eq!( dtype(&root.schema, "quantile_0_99"), - &SummaryFamilyType::Plain(DataType::Float64), + &FieldDataType::Plain(DataType::Float64), "the summary-state type must not propagate past the estimate" ); @@ -924,19 +873,19 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { // reduction, one output row — not to be confused with the inner rate's // per-entity grouping below, even though both once collapsed to the // same empty `by: []` (issue #163). - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { child, family, input, reduction, .. - } = &summary_input.expr + }) = summary_input.asap() else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) @@ -949,41 +898,39 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { ); assert_eq!( dtype(&summary_input.schema, "value"), - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) ); - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing: asap_types::post_asap::ExecutionTiming::IngestionTime, - } = &child.expr - else { - panic!("rate needs a maintenance readout"); + let Some(ASAPOp::FinalizeExactAccumulator { child }) = child.asap() else { + panic!("rate needs a maintenance evaluation"); }; // The rate: exact counter-reset-aware accumulator, per-series (labels // and time axis preserved), no estimate wrapper. `rate(...)` has no // grouping concept at all — every entity stays its own summary. - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { child: leaf, family, reduction, .. - } = &child.expr + }) = child.asap() else { - panic!("expected inner SummaryAgg for rate, got {:?}", child.expr); + panic!( + "expected inner SummaryAgg for rate, got {:?}", + child.operator + ); }; assert_eq!( family, - &SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) + &FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) ); assert_eq!(reduction, &Reduction::PerEntity); assert_eq!( dtype(&child.schema, "value"), - &SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) + &FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) ); assert_eq!( child.schema.time_index, @@ -992,40 +939,46 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { ); // The leaf: unrewritten pass-through — TimeRange marker over the Scan. - let SummaryExpr::KeepPreAsap(kept_leaf) = &leaf.expr else { - panic!("expected KeepPreAsap leaf, got {:?}", leaf.expr); - }; - let QueryExpr::TimeRange { range, child: scan } = kept_leaf.as_ref() else { - panic!("expected TimeRange leaf, got {kept_leaf:?}"); + // The kept leaf is the non-ASAP sub-DAG itself. + assert!( + !leaf.contains_asap(), + "expected kept leaf, got {:?}", + leaf.operator + ); + let Some(NonASAPOp::TimeRange { + range, child: scan, .. + }) = leaf.non_asap() + else { + panic!("expected TimeRange leaf, got {:?}", leaf.operator); }; assert_eq!(range.as_secs(), 300); - assert!(matches!(scan.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(scan.non_asap(), Some(NonASAPOp::Scan { .. }))); assert!( leaf.schema .fields .iter() - .all(|f| matches!(f.dtype, SummaryFamilyType::Plain(_))), + .all(|f| matches!(f.dtype, FieldDataType::Plain(_))), "logical edges carry only plain columns" ); } /// An exact workload binds zero sketches: `sum by (job) (m)` at /// `AccuracyTarget::Exact` still gets its mergeable exact accumulator, and -/// `avg(m)` (non-mergeable) passes through as a whole logical subtree. +/// `avg(m)` (non-mergeable) passes through as a whole logical sub-DAG. #[test] fn promql_exact_workload_binds_accumulators_not_sketches() { let pre_asap = lower_promql("sum by (job) (http_requests_total)", AccuracyTarget::Exact) .expect("lowering failed"); let root = realize(&pre_asap).expect("binding failed"); - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { family, reduction, .. - } = &root.expr + }) = root.asap() else { - panic!("expected SummaryAgg, got {:?}", root.expr); + panic!("expected SummaryAgg, got {:?}", root.operator); }; assert_eq!( family, - &SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + &FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) ); assert_eq!( reduction, @@ -1034,7 +987,7 @@ fn promql_exact_workload_binds_accumulators_not_sketches() { ); assert_eq!( dtype(&root.schema, "job"), - &SummaryFamilyType::Plain(DataType::Utf8), + &FieldDataType::Plain(DataType::Utf8), "group keys pass through verbatim" ); @@ -1042,21 +995,19 @@ fn promql_exact_workload_binds_accumulators_not_sketches() { lower_promql("avg(http_requests_total)", AccuracyTarget::Exact).expect("lowering failed"); let root = realize(&pre_asap).expect("binding failed"); assert!( - matches!(root.expr, SummaryExpr::KeepPreAsap(_)), + !root.contains_asap(), "avg has no mergeable accumulator — stays logical" ); } #[test] fn promql_sum_of_count_over_time_is_composed_by_default_search() { - let original = Rc::new( - lower_promql( - "sum by (service) (count_over_time(metrics[5m]))", - AccuracyTarget::Exact, - ) - .expect("lowering failed"), - ); - let original_schema = original.output_schema().unwrap(); + let original = lower_promql( + "sum by (service) (count_over_time(metrics[5m]))", + AccuracyTarget::Exact, + ) + .expect("lowering failed"); + let original_schema = original.schema.clone(); let space = search_workload(vec![("query", original)]); let root = &space.roots[0].1; let group = space.candidates_for_target(root).expect("root memo group"); @@ -1065,20 +1016,21 @@ fn promql_sum_of_count_over_time_is_composed_by_default_search() { .iter() .find(|candidate| candidate.strategy == "SemanticEquivalentRewriteStrategy") .expect("default search should compose the lowered PromQL query"); - let Replacement::Rewrite(rewritten) = &candidate.replacement else { + let Replacement::SubDag(rewritten) = &candidate.replacement else { panic!("expected logical rewrite") }; + assert!(is_logical_rewrite(rewritten), "expected logical rewrite"); - assert_eq!(rewritten.output_schema().unwrap(), original_schema); - let QueryExpr::Project { child, .. } = rewritten.as_ref() else { + assert_eq!(rewritten.schema, original_schema); + let Some(NonASAPOp::Project { child, .. }) = rewritten.non_asap() else { panic!("sum(count_over_time) needs a Float64 cast Project") }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), measures, child, .. - } = child.as_ref() + }) = child.non_asap() else { panic!("expected one composed aggregate") }; @@ -1090,9 +1042,9 @@ fn promql_sum_of_count_over_time_is_composed_by_default_search() { }] )); assert!(matches!( - child.as_ref(), - QueryExpr::TimeRange { range, child } - if range.as_secs() == 300 && matches!(child.as_ref(), QueryExpr::Scan { .. }) + child.non_asap(), + Some(NonASAPOp::TimeRange { range, child, .. }) + if range.as_secs() == 300 && matches!(child.non_asap(), Some(NonASAPOp::Scan { .. })) )); } @@ -1100,67 +1052,59 @@ fn promql_sum_of_count_over_time_is_composed_by_default_search() { fn nested_summary_explicitly_finalizes_exact_child_at_ingestion_time() { // Real workload selection must expose the state-to-value edge; an outer // sketch must not interpret exact accumulator bytes as input samples. - let pre = Rc::new( - lower_promql( - "quantile(0.9, sum_over_time(m[1m]))", - AccuracyTarget::Epsilon(0.05), - ) - .unwrap(), - ); + let pre = lower_promql( + "quantile(0.9, sum_over_time(m[1m]))", + AccuracyTarget::Epsilon(0.05), + ) + .unwrap(); let space = search_workload(vec![("query", pre)]); let selected = space.global_selection(&DefaultCostModel); let plan = selected .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &plan.expr else { + // Stored timings are gone: time the plan and read the timed copy. + let timed_plan = timed(&plan); + let Some(ASAPOp::SummaryEstimate { summary_input, .. }) = timed_plan.asap() else { panic!("expected selected quantile summary"); }; - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { + let Some(ASAPOp::SummaryAgg { child, .. }) = summary_input.asap() else { panic!("expected maintained outer summary"); }; - let SummaryExpr::ValueOperation { - child: source, - operation, - timing, - } = &child.expr - else { + let Some(ASAPOp::FinalizeExactAccumulator { child: source }) = child.asap() else { panic!( "missing explicit accumulator finalization: {:?}", - child.expr + child.operator ); }; - assert!(matches!( - operation, - ValueOperation::FinalizeExactAccumulator - )); assert_eq!( - *timing, - asap_types::post_asap::ExecutionTiming::IngestionTime + child.timing, + Some(asap_types::post_asap::ExecutionTiming::IngestionTime) ); assert!(matches!( - source.expr, - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + source.asap(), + Some(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Sum, _), .. - } + }) )); assert!(child .schema .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); assert!(child .schema .fields .iter() - .any(|field| matches!(field.dtype, SummaryFamilyType::Plain(DataType::Float64)))); - compile_post_asap_dag(&plan).expect("explicit boundary is a valid post-ASAP DAG"); + .any(|field| matches!(field.dtype, FieldDataType::Plain(DataType::Float64)))); + // Explicit boundary is a valid post-ASAP DAG. + post_asap_dag(&plan); } #[test] fn physical_node_owns_phase_independently_of_binary_payload() { - use asap_types::post_asap::{ExecutionTiming, PostAsapOperatorPayload}; + use asap_types::post_asap::ExecutionTiming; for (query, expected) in [ ( // One selector: both operands cover the same series. @@ -1173,19 +1117,24 @@ fn physical_node_owns_phase_independently_of_binary_payload() { ), ] { let input = lower_promql(query, AccuracyTarget::Epsilon(0.05)).unwrap(); - // Backend lowering carries opaque series identity before candidate export. - let input = asap_types::pre_asap::schema::with_promql_series_identity(&input).unwrap(); - let search = search_workload(vec![("q", Rc::new(input))]); + let search = search_workload(vec![("q", input)]); let choice = search.global_selection(&DefaultCostModel); let plan = choice .assemble_selected_dag(&search.roots[0].1) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&plan).unwrap(); + let dag = post_asap_dag(&plan); let node = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Binary { .. })) + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::BinaryOp { .. } + } + ) + }) .unwrap(); assert_eq!(node.output_state.timing, expected); let wire = serde_json::to_value(&node.payload).unwrap(); @@ -1207,14 +1156,10 @@ fn ddsketch_ratio_without_domain_proof_is_uncertified() { ) .unwrap(); let root = realize(&pre).unwrap(); - assert!(matches!(root.expr, SummaryExpr::BinaryOp { .. })); + assert!(matches!(root.non_asap(), Some(NonASAPOp::BinaryOp { .. }))); assert!(root.guarantee.is_none()); let space = search_workload_with_targets( - vec![( - "unproven", - Rc::new(pre), - Some(AccuracyTarget::Epsilon(0.01)), - )], + vec![("unproven", pre, Some(AccuracyTarget::Epsilon(0.01)))], &asap_aware_mapping::default_strategies(), &DefaultAccuracyModel, ); @@ -1226,8 +1171,8 @@ fn ddsketch_ratio_without_domain_proof_is_uncertified() { root_group.candidates.iter().any(|candidate| { matches!( &candidate.replacement, - Replacement::Summary(node) - if matches!(node.expr, SummaryExpr::BinaryOp { .. }) + Replacement::SubDag(node) + if matches!(node.non_asap(), Some(NonASAPOp::BinaryOp { .. })) && node.guarantee.is_none() ) }), @@ -1247,7 +1192,7 @@ fn ddsketch_ratio_without_domain_proof_is_uncertified() { .assemble_selected_dag(&space.roots[0].1) .unwrap() .expect("materialized root"); - assert!(matches!(materialized.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!materialized.contains_asap()); } struct FixtureQuantileDomain { @@ -1255,7 +1200,7 @@ struct FixtureQuantileDomain { upper: f64, } impl AccuracyEvidenceProvider for FixtureQuantileDomain { - fn quantile_input_domain(&self, _: &QueryExpr) -> Option { + fn quantile_input_domain(&self, _: &OperatorNode) -> Option { Some(QuantileInputDomain { lower: self.lower, upper: self.upper, @@ -1278,14 +1223,12 @@ fn ddsketch_ratio_rejects_unsafe_domains() { (f64::MIN_POSITIVE / 2., f64::MIN_POSITIVE / 2.), ] { let evidence = FixtureQuantileDomain { lower, upper }; - let pre = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let pre = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -1304,8 +1247,8 @@ fn ddsketch_ratio_rejects_unsafe_domains() { fn ddsketch_ratio_rejects_one_invalid_domain_when_the_other_is_missing() { struct PartialUnsafeDomain; impl AccuracyEvidenceProvider for PartialUnsafeDomain { - fn quantile_input_domain(&self, operand: &QueryExpr) -> Option { - let QueryExpr::Aggregate { measures, .. } = operand else { + fn quantile_input_domain(&self, operand: &OperatorNode) -> Option { + let Some(NonASAPOp::Aggregate { measures, .. }) = operand.non_asap() else { return None; }; matches!( @@ -1321,14 +1264,12 @@ fn ddsketch_ratio_rejects_one_invalid_domain_when_the_other_is_missing() { } } - let pre = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let pre = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -1339,40 +1280,38 @@ fn ddsketch_ratio_rejects_one_invalid_domain_when_the_other_is_missing() { /// The committed planner alpha is exercised against the pinned sketch implementation. #[test] -fn ddsketch_ratio_bound_holds_for_signed_pinned_sketch_readouts() { +fn ddsketch_ratio_bound_holds_for_signed_pinned_sketch_evaluations() { for sign in [-1., 1.] { let evidence = FixtureQuantileDomain { lower: if sign < 0. { -100. } else { 1. }, upper: if sign < 0. { -1. } else { 100. }, }; - let pre = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let pre = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, &evidence, ); let candidates = strategy.replacements(&TargetSubDAG::new(&pre)); - let Replacement::Summary(node) = &candidates[0].replacement else { + let Replacement::SubDag(node) = &candidates[0].replacement else { panic!("summary") }; - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &node.expr else { + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = node.non_asap() else { panic!("ratio") }; - let alpha = |node: &SummaryNode| { - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { - panic!("readout") + let alpha = |node: &OperatorNode| { + let Some(ASAPOp::SummaryEstimate { summary_input, .. }) = node.asap() else { + panic!("evaluation") }; - let SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + let Some(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), .. - } = &summary_input.expr + }) = summary_input.asap() else { panic!("sketch") }; @@ -1407,12 +1346,12 @@ fn ddsketch_ratio_bound_holds_for_signed_pinned_sketch_readouts() { } } -/// Empty or overlarge population contracts cannot promise a supported readout. +/// Empty or overlarge population contracts cannot promise a supported evaluation. #[test] fn ddsketch_ratio_requires_a_supported_population_size() { struct PopulationEvidence(u64); impl AccuracyEvidenceProvider for PopulationEvidence { - fn quantile_input_domain(&self, _: &QueryExpr) -> Option { + fn quantile_input_domain(&self, _: &OperatorNode) -> Option { Some(QuantileInputDomain { lower: 1., upper: 10., @@ -1421,16 +1360,14 @@ fn ddsketch_ratio_requires_a_supported_population_size() { }) } } - let pre = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); + let pre = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); for count in [0, (1u64 << 53) + 1] { let evidence = PopulationEvidence(count); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -1441,7 +1378,7 @@ fn ddsketch_ratio_requires_a_supported_population_size() { } // Every `without` aggregation candidate exports a valid DAG: its summary state -// column carries the family instead of the readout's Float64 value. +// column carries the family instead of the evaluation's Float64 value. #[test] fn without_aggregation_candidates_export_valid_dags() { for accuracy in [ @@ -1452,7 +1389,7 @@ fn without_aggregation_candidates_export_valid_dags() { }, ] { for query in ["sum without (pod) (m)", "quantile without (pod) (0.5, m)"] { - let root = Rc::new(lower_promql(query, accuracy.clone()).unwrap()); + let root = lower_promql(query, accuracy.clone()).unwrap(); let space = search_workload_with_targets( vec![(0, root, Some(accuracy.clone()))], &asap_aware_mapping::default_strategies(), @@ -1461,7 +1398,7 @@ fn without_aggregation_candidates_export_valid_dags() { let inventory = space.enumerate_candidate_dags_for_root(&0, 65_536).unwrap(); assert!(!inventory.candidates.is_empty(), "{query}"); for (_, node) in inventory.candidates.iter().flatten() { - compile_post_asap_dag(node).unwrap_or_else(|e| panic!("{query}: {e}")); + post_asap_dag(node); } } } diff --git a/crates/integration-tests/tests/scan.rs b/crates/integration-tests/tests/scan.rs index bb7988d7a..fdb72f9a1 100644 --- a/crates/integration-tests/tests/scan.rs +++ b/crates/integration-tests/tests/scan.rs @@ -1,4 +1,4 @@ -//! `QueryExpr::Scan` — label matcher / predicate tests. +//! `NonASAPOp::Scan` — label matcher / predicate tests. //! //! The Scan schema is always [ts(0), value(1), label_a(2), label_b(3), …] //! where labels are appended alphabetically after dedup by the SchemaResolver. @@ -11,60 +11,61 @@ use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{CompareOpKind, Predicate, QueryExpr, ScalarValue, Source}; +use asap_types::ir::{ + ExprSemantics, NonASAPOp, OperatorNode, Predicate, ScalarExpr, TimeRangeKind, +}; +use asap_types::pre_asap::{CompareOpKind, ScalarValue, Source}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn bare_scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::Scan { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::non_asap_node(op).expect("fixture node derives its schema") +} + +fn bare_scan(metric: &str, labels: &[&str]) -> Rc { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: metric_schema(labels), - } + }) } -fn instant(child: QueryExpr) -> QueryExpr { - QueryExpr::TimeRange { +fn instant(child: Rc) -> Rc { + node(NonASAPOp::TimeRange { range: Duration::from_secs(1), - child: Rc::new(child), - } + kind: TimeRangeKind::Instant, + child, + }) +} + +fn label_pred(col_id: usize, op: CompareOpKind, value: &str) -> Predicate { + Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(col_id)), + op, + right: Box::new(ScalarExpr::Literal(ScalarValue::Utf8(value.into()))), + semantics: ExprSemantics::Promql, + }) } fn eq_pred(col_id: usize, value: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(value.into()))), - })) + label_pred(col_id, CompareOpKind::Eq, value) } fn ne_pred(col_id: usize, value: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), - op: CompareOpKind::Ne, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(value.into()))), - })) + label_pred(col_id, CompareOpKind::Ne, value) } fn regex_pred(col_id: usize, pattern: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), - op: CompareOpKind::Regex, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(pattern.into()))), - })) + label_pred(col_id, CompareOpKind::Regex, pattern) } fn notregex_pred(col_id: usize, pattern: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), - op: CompareOpKind::NotRegex, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(pattern.into()))), - })) + label_pred(col_id, CompareOpKind::NotRegex, pattern) } // #1 — bare metric name, no matchers @@ -80,13 +81,13 @@ fn q01_bare_scan() { // schema: [ts(0), value(1), job(2)] #[test] fn q02_equality_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![eq_pred(2, "api-server")], schema: metric_schema(&["job"]), - }); + })); assert_eq!(lower(r#"http_requests_total{job="api-server"}"#), expected); } @@ -94,13 +95,13 @@ fn q02_equality_predicate() { // schema: [ts(0), value(1), status(2)] #[test] fn q03_inequality_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![ne_pred(2, "500")], schema: metric_schema(&["status"]), - }); + })); assert_eq!(lower(r#"http_requests_total{status!="500"}"#), expected); } @@ -108,13 +109,13 @@ fn q03_inequality_predicate() { // schema: [ts(0), value(1), job(2)] #[test] fn q04_regex_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![regex_pred(2, "api.*")], schema: metric_schema(&["job"]), - }); + })); assert_eq!(lower(r#"http_requests_total{job=~"api.*"}"#), expected); } @@ -122,13 +123,13 @@ fn q04_regex_predicate() { // schema: [ts(0), value(1), job(2)] #[test] fn q_notregex_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![notregex_pred(2, "internal.*")], schema: metric_schema(&["job"]), - }); + })); assert_eq!(lower(r#"http_requests_total{job!~"internal.*"}"#), expected); } @@ -137,13 +138,13 @@ fn q_notregex_predicate() { // predicates in same alphabetical order: job first, then status #[test] fn q_multi_two_predicates() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![eq_pred(2, "api-server"), ne_pred(3, "500")], schema: metric_schema(&["job", "status"]), - }); + })); assert_eq!( lower(r#"http_requests_total{job="api-server",status!="500"}"#), expected, diff --git a/crates/integration-tests/tests/schema.rs b/crates/integration-tests/tests/schema.rs index 6024619cc..c616c0d5f 100644 --- a/crates/integration-tests/tests/schema.rs +++ b/crates/integration-tests/tests/schema.rs @@ -1,6 +1,6 @@ //! `Schema::closed` propagation — open/closed invariant tests. //! -//! Verifies that `QueryExpr::output_schema()` propagates the open/closed +//! Verifies that the derived `OperatorNode::schema` propagates the open/closed //! completeness flag correctly through a lowered query tree. //! //! Key invariant: a PromQL scan is always `closed: false` (open) because its @@ -12,14 +12,14 @@ use asap_integration_tests::fixtures::lower_promql; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> asap_types::pre_asap::QueryExpr { +fn lower(q: &str) -> std::rc::Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } // bare scan is open — the metric's full label set is unknown at plan time #[test] fn schema_bare_scan_is_open() { - let s = lower("http_requests_total").output_schema().unwrap(); + let s = lower("http_requests_total").schema.clone(); assert!(!s.closed, "PromQL scan must be open"); } @@ -27,18 +27,16 @@ fn schema_bare_scan_is_open() { #[test] fn schema_filtered_scan_is_open() { let s = lower(r#"http_requests_total{job="api-server"}"#) - .output_schema() - .unwrap(); + .schema + .clone(); assert!(!s.closed, "PromQL scan with predicates must remain open"); - assert_eq!(s.columns.len(), 3, "[ts, value, job]"); + assert_eq!(s.fields.len(), 3, "[ts, value, job]"); } // per-series rate is label-preserving → output stays open #[test] fn schema_rate_stays_open() { - let s = lower("rate(http_requests_total[5m])") - .output_schema() - .unwrap(); + let s = lower("rate(http_requests_total[5m])").schema.clone(); assert!(!s.closed, "per-series rate is label-preserving; stays open"); } @@ -46,24 +44,22 @@ fn schema_rate_stays_open() { #[test] fn schema_count_over_time_stays_open() { let s = lower("count_over_time(http_requests_total[5m])") - .output_schema() - .unwrap(); + .schema + .clone(); assert!(!s.closed, "per-series count_over_time stays open"); } // cross-series sum with no group keys freezes to closed #[test] fn schema_sum_freezes_to_closed() { - let s = lower("sum(http_requests_total)").output_schema().unwrap(); + let s = lower("sum(http_requests_total)").schema.clone(); assert!(s.closed, "cross-series aggregate must freeze to closed"); } // cross-series sum grouped by job also freezes to closed #[test] fn schema_sum_by_job_freezes_to_closed() { - let s = lower("sum by (job) (http_requests_total)") - .output_schema() - .unwrap(); + let s = lower("sum by (job) (http_requests_total)").schema.clone(); assert!( s.closed, "grouped cross-series aggregate must freeze to closed" @@ -74,8 +70,8 @@ fn schema_sum_by_job_freezes_to_closed() { #[test] fn schema_sum_over_rate_freezes_to_closed() { let s = lower("sum by (job) (rate(http_requests_total[5m]))") - .output_schema() - .unwrap(); + .schema + .clone(); assert!( s.closed, "cross-series aggregate over rate must freeze to closed" @@ -86,8 +82,8 @@ fn schema_sum_over_rate_freezes_to_closed() { #[test] fn schema_binary_op_two_open_stays_open() { let s = lower("http_requests_total / http_errors_total") - .output_schema() - .unwrap(); + .schema + .clone(); assert!(!s.closed, "binary op over two open scans must stay open"); } @@ -95,8 +91,8 @@ fn schema_binary_op_two_open_stays_open() { #[test] fn schema_binary_op_two_closed_is_closed() { let s = lower("sum by (job) (http_requests_total) / sum by (job) (http_errors_total)") - .output_schema() - .unwrap(); + .schema + .clone(); assert!( s.closed, "binary op over two closed aggregates must be closed" diff --git a/crates/integration-tests/tests/sql_to_physical.rs b/crates/integration-tests/tests/sql_to_physical.rs index be96107ea..1b3683759 100644 --- a/crates/integration-tests/tests/sql_to_physical.rs +++ b/crates/integration-tests/tests/sql_to_physical.rs @@ -1,4 +1,5 @@ //! SQL frontend, candidate selection, physical compilation and fresh-run execution. +mod physical_common; use asap_aware_mapping::{search_workload, DefaultCostModel}; use asap_frontend_sql::{lower_sql, SqlCatalog}; use asap_physical_operators::{ @@ -7,13 +8,15 @@ use asap_physical_operators::{ sources::{DataSources, MemorySource}, values::{Batch, Value}, }; +use asap_types::ir::export::PostAsapOperatorPayload; use asap_types::{ - post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, - pre_asap::{Column, DataType, QueryExpr, Schema}, + post_asap::FieldDataType, + pre_asap::{DataType, Field, Schema}, types::AccuracyTarget, }; use futures::StreamExt; -use std::{collections::BTreeMap, rc::Rc, sync::Arc}; +use physical_common::compile_post_asap_dag; +use std::{collections::BTreeMap, sync::Arc}; /// SQL filtering and grouped aggregation survive logical/physical lowering; /// rebinding the compiled DAG runs against new data rather than cached results. @@ -22,19 +25,17 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { let catalog = SqlCatalog::new().with_table( "metrics", Schema::new(vec![ - Column::new("service", DataType::Utf8, false), - Column::new("value", DataType::Float64, true), + Field::plain("service", DataType::Utf8, false), + Field::plain("value", DataType::Float64, true), ]), ); for query in [ "SELECT service, SUM(value) AS total FROM metrics WHERE value > 1 GROUP BY service", "SELECT service, SUM(value) AS total FROM metrics GROUP BY service", ] { - let logical = Rc::new( - lower_sql(query, &catalog, AccuracyTarget::Exact) - .await - .unwrap(), - ); + let logical = lower_sql(query, &catalog, AccuracyTarget::Exact) + .await + .unwrap(); let space = search_workload(vec![("sql", logical)]); let selected = space .global_selection(&DefaultCostModel) @@ -48,8 +49,8 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::Scan { .. } + PostAsapOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::Scan { .. } } ) }) @@ -58,7 +59,7 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { assert!(schema .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); let plan = compile( &dag, BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), @@ -88,12 +89,18 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { .collect() }) .collect(); - let PostAsapOperatorPayload::Fallback { expression } = &scan.payload else { - unreachable!() - }; - let QueryExpr::Scan { source, .. } = expression else { + let PostAsapOperatorPayload::Relational { + operator: + asap_types::ir::export::NonASAPOpKind::Scan { + source, + predicates: _, + schema: _scan_schema, + }, + } = &scan.payload + else { unreachable!() }; + let expression = asap_types::ir::OperatorNode::reachable(&selected).into_iter().find(|n| matches!(n.non_asap(), Some(asap_types::ir::NonASAPOp::Scan { source: s, .. }) if s == source)).unwrap(); let mut sources = DataSources::default(); sources .register( @@ -110,7 +117,7 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { let bound = plan .instantiate(BTreeMap::from([( u64::from(scan.id.0), - Box::new(sources.bind(expression).unwrap()) as Source<'_>, + Box::new(sources.bind(&expression).unwrap()) as Source<'_>, )])) .unwrap(); let mut stream = bound diff --git a/crates/integration-tests/tests/sql_to_post_asap.rs b/crates/integration-tests/tests/sql_to_post_asap.rs index d2d418222..be2787016 100644 --- a/crates/integration-tests/tests/sql_to_post_asap.rs +++ b/crates/integration-tests/tests/sql_to_post_asap.rs @@ -1,64 +1,98 @@ //! End-to-end SQL query-string → post-ASAP IR pin (issue #191). //! //! The SQL counterpart of `promql_to_post_asap.rs`: drives SQL text — -//! `lower_sql` (text → pre-ASAP `QueryExpr`) → -//! `SketchAlgorithmStrategy::replacements` (pre-ASAP → post-ASAP -//! `SummaryExpr`, see [`realize`] below) — and pins the resulting -//! sketch-vs-exact-accumulator shape node by node, the way -//! `promql_to_post_asap.rs` does for PromQL. +//! `lower_sql` (text → non-ASAP `OperatorNode` tree) → +//! `ASAPStrategies::replacements` (→ a tree with ASAP operators, +//! see [`realize`] below) — and pins the resulting sketch-vs-exact-accumulator +//! shape node by node, the way `promql_to_post_asap.rs` does for PromQL. //! //! ## A structural wrinkle PromQL doesn't have //! -//! `lower_promql` returns a *bare* `QueryExpr::Aggregate` for a top-level +//! `lower_promql` returns a *bare* `NonASAPOp::Aggregate` for a top-level //! aggregation (`sum by (job) (m)`, `quantile(0.99, …)`), so [`realize`] can //! bind it directly at the tree root. `lower_sql` never does: DataFusion's //! planner always wraps even a single, unaliased aggregate in an identity //! `Project` (confirmed below), so a SQL tree's *root* is normally `Project { //! child: Aggregate { .. } }`. Final materialization retains that projection -//! as a query-time value operation and independently plans its child, keeping +//! as a query-time non-ASAP node and independently plans its child, keeping //! both SELECT-list semantics and the summary-bound aggregate visible. use std::rc::Rc; -use asap_aware_mapping::replacement::{keep_pre_asap, RealizationError}; +use asap_aware_mapping::replacement::{retain_exact, RealizationError}; use asap_aware_mapping::{ - search_workload, DefaultCostModel, Replacement, ReplacementStrategy, ReplacementSubDAG, - SketchAlgorithmStrategy, TargetSubDAG, + search_workload, ASAPStrategies, DefaultCostModel, Replacement, ReplacementStrategy, + ReplacementSubDAG, TargetSubDAG, }; use asap_frontend_sql::{lower_sql, lower_sql_dialect, SqlCatalog}; +use asap_integration_tests::post_asap::post_asap_dag; +use asap_types::ir::export::{ + EdgeRole, NonASAPOpKind, PostAsapNodeId, PostAsapOperatorPayload, WirePredicate, WireScalarExpr, +}; +use asap_types::ir::operator_properties::Reduction; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, Predicate, ScalarExpr}; use asap_types::post_asap::{ - compile_post_asap_dag, EdgeRole, ExactKind, ExactParams, GroupingStrategy, - PostAsapOperatorPayload, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, SummaryExpr, - SummaryFamilyType, SummaryNode, SummarySchema, SummaryUpdate, ValueOperation, + ExactKind, ExactParams, FieldDataType, GroupingStrategy, SketchAlgorithm, SketchKind, + SketchParams, SketchStatistic, SummaryUpdate, }; use asap_types::pre_asap::expr_ir::ColumnRef; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; /// This crate has no "bind me one tree" public API any more — -/// `SketchAlgorithmStrategy::replacements` always returns every candidate, and +/// `ASAPStrategies::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the -/// take-the-first-(`cost_model`-preferred)-candidate pattern so the +/// take-the-first-(`cost_model`-preferred)-summary-candidate pattern so the /// single-answer pins below don't all repeat it by hand. -fn realize(expr: &QueryExpr) -> Result, RealizationError> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - match SketchAlgorithmStrategy::default_cost_model() - .replacements(&target) +fn realize(target: &Rc) -> Result, RealizationError> { + let target_dag = TargetSubDAG::new(target); + match ASAPStrategies::default_cost_model() + .replacements(&target_dag) .into_iter() .next() { Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDag(node), .. - }) => Ok(node), - _ => keep_pre_asap(&root), + }) if node.contains_asap() => Ok(node), + _ => retain_exact(target), + } + .inspect(|node| { + node.validate_structure() + .expect("planned graph satisfies the unified IR contract") + }) +} + +/// The single input of a unary non-ASAP node (Project, Filter, Sort, ...) or +/// of a `FinalizeExactAccumulator`; `None` for anything else. +fn unary_child(node: &OperatorNode) -> Option<&Rc> { + match &node.operator { + Operator::NonASAP(op) => match op.children().as_slice() { + [child] => Some(*child), + _ => None, + }, + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) => Some(child), + Operator::ASAP(_) => None, } } -fn dtype<'a>(schema: &'a SummarySchema, name: &str) -> &'a SummaryFamilyType { +/// A sub-DAG kept as plain (non-ASAP) work: no ASAP operator anywhere below. +fn is_kept_non_asap(node: &OperatorNode) -> bool { + node.non_asap().is_some() && !node.contains_asap() +} + +/// Mirror a scalar-only predicate (no operator references) to its wire form. +fn wire_pred(pred: &Predicate) -> WirePredicate { + WirePredicate(WireScalarExpr::from_expr( + &pred.0, + &mut |_: &Rc| -> PostAsapNodeId { + panic!("fixture predicate references no operator") + }, + )) +} + +fn dtype<'a>(schema: &'a Schema, name: &str) -> &'a FieldDataType { &schema .fields .iter() @@ -67,8 +101,8 @@ fn dtype<'a>(schema: &'a SummarySchema, name: &str) -> &'a SummaryFamilyType { .dtype } -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } /// `metrics(ts, service, latency, bytes)` — mirrors @@ -89,7 +123,7 @@ fn catalog() -> SqlCatalog { ) } -async fn lower(sql: &str, accuracy: AccuracyTarget) -> QueryExpr { +async fn lower(sql: &str, accuracy: AccuracyTarget) -> Rc { lower_sql(sql, &catalog(), accuracy) .await .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) @@ -100,11 +134,11 @@ async fn clickhouse_temporal_sql_reuses_rate_and_increase_physical_summaries() { for (function, expected) in [ ( "asap_rate", - SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), + FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), ), ( "asap_increase", - SummaryFamilyType::ExactAggregate(ExactKind::Increase, ExactParams::Increase), + FieldDataType::ExactAggregate(ExactKind::Increase, ExactParams::Increase), ), ] { let sql = format!( @@ -121,26 +155,27 @@ async fn clickhouse_temporal_sql_reuses_rate_and_increase_physical_summaries() { .expect("explicit temporal SQL must lower"); let physical = realize(inner_aggregate(&pre_asap)).expect("temporal reducer must be planned"); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, reduction, child, .. - } = &physical.expr + }) = &physical.operator else { - panic!("expected a shared SummaryAgg, got {:?}", physical.expr); + panic!("expected a shared SummaryAgg, got {:?}", physical.operator); }; assert_eq!(family, &expected); assert_eq!(reduction, &Reduction::PerEntity); - let SummaryExpr::KeepPreAsap(raw) = &child.expr else { - panic!( - "expected a retained temporal SQL input, got {:?}", - child.expr - ); - }; - assert!(matches!(raw.as_ref(), QueryExpr::TimeRange { range, child } + assert!( + is_kept_non_asap(child), + "expected a retained temporal SQL input, got {:?}", + child.operator + ); + assert!( + matches!(child.non_asap(), Some(NonASAPOp::TimeRange { range, child, .. }) if *range == std::time::Duration::from_secs(300) - && matches!(child.as_ref(), QueryExpr::Project { .. }))); + && matches!(child.non_asap(), Some(NonASAPOp::Project { .. }))) + ); } } @@ -156,16 +191,14 @@ async fn clickhouse_outer_sum_recursively_binds_inner_temporal_aggregate() { SELECT service, {function}(latency, ts, {window_ms}) AS v \ FROM metrics GROUP BY service)" ); - let pre_asap = Rc::new( - lower_sql_dialect( - &sql, - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect("nested temporal SQL must lower"), - ); + let pre_asap = lower_sql_dialect( + &sql, + &catalog(), + SqlDialect::ClickhouseSQL, + AccuracyTarget::Exact, + ) + .await + .expect("nested temporal SQL must lower"); let space = search_workload(vec![("nested", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection @@ -173,31 +206,27 @@ async fn clickhouse_outer_sum_recursively_binds_inner_temporal_aggregate() { .expect("materialization failed") .expect("root must be discovered"); - fn has_temporal_summary(node: &SummaryNode) -> bool { - match &node.expr { - SummaryExpr::SummaryAgg { - family: - SummaryFamilyType::ExactAggregate(ExactKind::Rate | ExactKind::Increase, _), + fn has_temporal_summary(node: &OperatorNode) -> bool { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Rate | ExactKind::Increase, _), .. - } => true, - SummaryExpr::ValueOperation { child, .. } - | SummaryExpr::SummaryEstimate { - summary_input: child, - .. - } => has_temporal_summary(child), - _ => false, + }) => true, + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + has_temporal_summary(summary_input) + } + _ => unary_child(node).is_some_and(|child| has_temporal_summary(child)), } } assert!( has_temporal_summary(&root), "inner {function} was hidden: {root:?}" ); - let dag = compile_post_asap_dag(&root).expect("nested SQL DAG must compile"); + let dag = post_asap_dag(&root); assert!(dag.nodes.iter().any(|node| matches!( node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Exact(_), - .. + PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::Aggregate { .. }, } ))); } @@ -205,11 +234,14 @@ async fn clickhouse_outer_sum_recursively_binds_inner_temporal_aggregate() { /// The `Aggregate` node beneath the identity `Project` DataFusion's planner /// always wraps a top-level aggregate in — see the module docs above. -fn inner_aggregate(qe: &QueryExpr) -> &QueryExpr { - match qe { - QueryExpr::Project { child, .. } => inner_aggregate(child), - QueryExpr::Aggregate { .. } => qe, - other => panic!("expected a Project{{Aggregate}} shape, got {other:?}"), +fn inner_aggregate(node: &Rc) -> &Rc { + match node.non_asap() { + Some(NonASAPOp::Project { child, .. }) => inner_aggregate(child), + Some(NonASAPOp::Aggregate { .. }) => node, + _ => panic!( + "expected a Project{{Aggregate}} shape, got {:?}", + node.operator + ), } } @@ -222,42 +254,40 @@ async fn sql_full_query_retains_project_and_binds_inner_aggregate() { AccuracyTarget::Epsilon(0.01), ) .await; - assert!( - matches!(pre_asap, QueryExpr::Project { .. }), - "sanity: a SQL root is a Project, unlike lower_promql's bare Aggregate" - ); - let pre_asap = Rc::new(pre_asap); + let Some(NonASAPOp::Project { + cols: expected_cols, + qualifier: expected_qualifier, + .. + }) = pre_asap.non_asap() + else { + panic!("sanity: a SQL root is a Project, unlike lower_promql's bare Aggregate"); + }; let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") .expect("root must be discovered"); - let QueryExpr::Project { - cols: expected_cols, - qualifier: expected_qualifier, - .. - } = pre_asap.as_ref() - else { - unreachable!() - }; - let SummaryExpr::ValueOperation { + let Some(NonASAPOp::Project { child, - operation: asap_types::post_asap::ValueOperation::Project { cols, qualifier }, - .. - } = &root.expr + cols, + qualifier, + }) = root.non_asap() else { - panic!("expected retained Project root, got {:?}", root.expr); + panic!("expected retained Project root, got {:?}", root.operator); }; assert_eq!(cols, expected_cols, "projection expressions and aliases"); assert_eq!(qualifier, expected_qualifier, "projection qualifier"); assert_eq!(root.schema.fields[0].name, "p99", "project output schema"); assert_eq!( root.schema.fields[0].dtype, - SummaryFamilyType::Plain(DataType::Float64) + FieldDataType::Plain(DataType::Float64) ); assert!( - matches!(child.expr, SummaryExpr::SummaryEstimate { .. }), + matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + ), "the Aggregate under Project must be summary-bound" ); } @@ -266,63 +296,62 @@ async fn sql_full_query_retains_project_and_binds_inner_aggregate() { /// aggregates are independently selected as physical summaries. #[tokio::test] async fn sql_join_recursively_binds_both_temporal_aggregate_children() { - let pre_asap = Rc::new( - lower_sql_dialect( - "SELECT a.service, a.v / b.v AS ratio FROM \ - (SELECT service, asap_rate(latency, ts, 300000) AS v FROM metrics WHERE service='errors' GROUP BY service) a \ - INNER JOIN \ - (SELECT service, asap_rate(latency, ts, 300000) AS v FROM metrics WHERE service='requests' GROUP BY service) b \ - ON b.service=a.service", - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect("two-subquery rate ratio must lower"), - ); + let pre_asap = lower_sql_dialect( + "SELECT a.service, a.v / b.v AS ratio FROM \ + (SELECT service, asap_rate(latency, ts, 300000) AS v FROM metrics WHERE service='errors' GROUP BY service) a \ + INNER JOIN \ + (SELECT service, asap_rate(latency, ts, 300000) AS v FROM metrics WHERE service='requests' GROUP BY service) b \ + ON b.service=a.service", + &catalog(), + SqlDialect::ClickhouseSQL, + AccuracyTarget::Exact, + ) + .await + .expect("two-subquery rate ratio must lower"); let space = search_workload(vec![("ratio", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") .expect("root must be discovered"); - let SummaryExpr::ValueOperation { - child: join, - operation: ValueOperation::Project { cols, .. }, - .. - } = &root.expr + let Some(NonASAPOp::Project { + child: join, cols, .. + }) = root.non_asap() else { panic!( "expected Project above relational join, got {:?}", - root.expr + root.operator ); }; assert!(matches!( &cols[1].expr, - QueryExpr::Arithmetic { + ScalarExpr::Arithmetic { op: asap_types::pre_asap::ArithmeticOpKind::Div, .. } )); - let SummaryExpr::RelationalJoin { + let Some(NonASAPOp::Join { left, right, kind, pred, - pruning: None, - } = &join.expr + }) = join.non_asap() else { - panic!("expected read-time relational join, got {:?}", join.expr); + panic!( + "expected read-time relational join, got {:?}", + join.operator + ); }; assert_eq!(kind, &asap_types::pre_asap::JoinKind::Inner); assert!(matches!( - pred.0.as_ref(), - QueryExpr::Compare { + &pred.0, + ScalarExpr::Compare { left, op: asap_types::pre_asap::CompareOpKind::Eq, right, - } if matches!(left.as_ref(), QueryExpr::Column(0)) - && matches!(right.as_ref(), QueryExpr::Column(2)) + .. + } if matches!(left.as_ref(), ScalarExpr::Column(0)) + && matches!(right.as_ref(), ScalarExpr::Column(2)) )); assert_eq!( join.schema @@ -333,39 +362,44 @@ async fn sql_join_recursively_binds_both_temporal_aggregate_children() { vec!["service", "v", "service", "v"] ); for child in [left, right] { - let SummaryExpr::ValueOperation { - child: aggregate, - operation: ValueOperation::Project { .. }, - .. - } = &child.expr + let Some(NonASAPOp::Project { + child: aggregate, .. + }) = child.non_asap() else { - panic!("derived table Project was not retained: {:?}", child.expr); + panic!( + "derived table Project was not retained: {:?}", + child.operator + ); }; - let SummaryExpr::ValueOperation { - child: aggregate, - operation: ValueOperation::FinalizeExactAccumulator, - .. - } = &aggregate.expr + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: aggregate }) = + &aggregate.operator else { panic!("derived table Project must consume finalized exact values"); }; assert!(matches!( - aggregate.expr, - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), + aggregate.operator, + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), .. - } + }) )); } assert!(join .guarantee .as_ref() .is_some_and(|value| value.is_exact())); - let dag = compile_post_asap_dag(&root).expect("join DAG must compile"); + let dag = post_asap_dag(&root); let join_id = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::RelationalJoin { .. })) + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::Join { .. }, + } + ) + }) .expect("relational join node") .id; let roles = dag @@ -384,29 +418,27 @@ async fn unsupported_sql_join_shapes_remain_fail_closed() { "SELECT a.service FROM (SELECT service, asap_rate(latency, ts, 300000) v FROM metrics GROUP BY service) a INNER JOIN (SELECT service, asap_rate(latency, ts, 300000) v FROM metrics GROUP BY service) b ON a.v>b.v", "SELECT a.service FROM (SELECT service, asap_rate(latency, ts, 300000) v FROM metrics GROUP BY service) a INNER JOIN (SELECT service, asap_rate(latency, ts, 300000) v FROM metrics GROUP BY service) b ON a.service=a.service", ] { - let pre_asap = Rc::new( - lower_sql_dialect( - sql, - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap_or_else(|error| panic!("join must lower before fail-closed mapping: {error}")), - ); + let pre_asap = lower_sql_dialect( + sql, + &catalog(), + SqlDialect::ClickhouseSQL, + AccuracyTarget::Exact, + ) + .await + .unwrap_or_else(|error| panic!("join must lower before fail-closed mapping: {error}")); let space = search_workload(vec![("unsupported-join", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") .expect("root must be discovered"); - let SummaryExpr::ValueOperation { child, .. } = &root.expr else { - panic!("SQL projection must remain explicit: {:?}", root.expr); + let Some(NonASAPOp::Project { child, .. }) = root.non_asap() else { + panic!("SQL projection must remain explicit: {:?}", root.operator); }; assert!( - matches!(child.expr, SummaryExpr::KeepPreAsap(_)), + is_kept_non_asap(child), "unsupported join was partially accelerated: {:?}", - child.expr + child.operator ); } } @@ -415,16 +447,14 @@ async fn unsupported_sql_join_shapes_remain_fail_closed() { /// explicit read-time nodes while the aggregate is summary-bound. #[tokio::test] async fn sql_relational_parents_retain_summary_bound_aggregate() { - let pre_asap = Rc::new( - lower( - "SELECT t.service, t.p FROM \ - (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ - FROM metrics GROUP BY service) t \ - WHERE t.p > 100 ORDER BY t.p DESC LIMIT 5", - AccuracyTarget::Epsilon(0.01), - ) - .await, - ); + let pre_asap = lower( + "SELECT t.service, t.p FROM \ + (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ + FROM metrics GROUP BY service) t \ + WHERE t.p > 100 ORDER BY t.p DESC LIMIT 5", + AccuracyTarget::Epsilon(0.01), + ) + .await; let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection @@ -438,28 +468,29 @@ async fn sql_relational_parents_retain_summary_bound_aggregate() { let mut saw_sort = false; let mut saw_limit = false; loop { - match &node.expr { - SummaryExpr::ValueOperation { - child, operation, .. - } => { - match operation { - asap_types::post_asap::ValueOperation::Project { .. } => saw_project = true, - asap_types::post_asap::ValueOperation::Filter { .. } => saw_filter = true, - asap_types::post_asap::ValueOperation::Sort { .. } => saw_sort = true, - asap_types::post_asap::ValueOperation::Limit { n, offset, .. } => { - assert_eq!((*n, *offset), (5, 0)); - saw_limit = true; - } - _ => {} - } - node = child; - } - SummaryExpr::SummaryEstimate { summary_input, .. } => { - assert!(matches!(summary_input.expr, SummaryExpr::SummaryAgg { .. })); - break; + if let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator { + assert!(matches!( + summary_input.operator, + Operator::ASAP(ASAPOp::SummaryAgg { .. }) + )); + break; + } + match node.non_asap() { + Some(NonASAPOp::Project { .. }) => saw_project = true, + Some(NonASAPOp::Filter { .. }) => saw_filter = true, + Some(NonASAPOp::Sort { .. }) => saw_sort = true, + Some(NonASAPOp::Limit { n, offset, .. }) => { + assert_eq!((*n, *offset), (Some(5), 0)); + saw_limit = true; } - other => panic!("expected relational parents over SummaryEstimate, got {other:?}"), + _ => {} } + node = unary_child(node).unwrap_or_else(|| { + panic!( + "expected relational parents over SummaryEstimate, got {:?}", + node.operator + ) + }); } assert!(saw_project && saw_filter && saw_sort && saw_limit); } @@ -469,39 +500,47 @@ async fn sql_relational_parents_retain_summary_bound_aggregate() { /// may be dropped or moved across the aggregation boundary. #[tokio::test] async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { - let pre_asap = Rc::new( - lower( - "SELECT t.service, t.p FROM \ - (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ - FROM metrics WHERE service = 'api' GROUP BY service) t \ - WHERE t.p > 100", - AccuracyTarget::Epsilon(0.01), - ) - .await, - ); + let pre_asap = lower( + "SELECT t.service, t.p FROM \ + (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ + FROM metrics WHERE service = 'api' GROUP BY service) t \ + WHERE t.p > 100", + AccuracyTarget::Epsilon(0.01), + ) + .await; let expected_read_predicate = { - let mut node = pre_asap.as_ref(); + let mut node = &pre_asap; loop { - match node { - QueryExpr::Filter { pred, .. } => break pred.clone(), - QueryExpr::Project { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => node = child, - other => panic!("expected a Filter above the aggregate, got {other:?}"), + match node.non_asap() { + Some(NonASAPOp::Filter { pred, .. }) => break pred.clone(), + Some( + NonASAPOp::Project { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. }, + ) => node = child, + _ => panic!( + "expected a Filter above the aggregate, got {:?}", + node.operator + ), } } }; let expected_source_predicates = { - let mut node = pre_asap.as_ref(); + let mut node = &pre_asap; loop { - match node { - QueryExpr::Scan { predicates, .. } => break predicates.clone(), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => node = child, - other => panic!("expected a unary SQL plan over Scan, got {other:?}"), + match node.non_asap() { + Some(NonASAPOp::Scan { predicates, .. }) => break predicates.clone(), + Some( + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. }, + ) => node = child, + _ => panic!( + "expected a unary SQL plan over Scan, got {:?}", + node.operator + ), } } }; @@ -517,41 +556,39 @@ async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { let mut node = root.as_ref(); let mut retained_read_predicate = None; loop { - match &node.expr { - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Filter { pred }, - .. - } => { - retained_read_predicate = Some(pred.clone()); - node = child; - } - SummaryExpr::ValueOperation { child, .. } => node = child, - SummaryExpr::SummaryEstimate { summary_input, .. } => { - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { - panic!("expected SummaryAgg below SummaryEstimate"); - }; - let SummaryExpr::KeepPreAsap(raw_input) = &child.expr else { - panic!("expected raw summary population below SummaryAgg"); - }; - let QueryExpr::Scan { predicates, .. } = raw_input.as_ref() else { - panic!("expected source selection to remain a Scan"); - }; - assert_eq!(predicates, &expected_source_predicates); - break; - } - other => panic!("expected read-time operations over a summary, got {other:?}"), + if let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg below SummaryEstimate"); + }; + assert!( + is_kept_non_asap(child), + "expected raw summary population below SummaryAgg" + ); + let Some(NonASAPOp::Scan { predicates, .. }) = child.non_asap() else { + panic!("expected source selection to remain a Scan"); + }; + assert_eq!(predicates, &expected_source_predicates); + break; + } + if let Some(NonASAPOp::Filter { pred, .. }) = node.non_asap() { + retained_read_predicate = Some(pred.clone()); } + node = unary_child(node).unwrap_or_else(|| { + panic!( + "expected read-time operations over a summary, got {:?}", + node.operator + ) + }); } assert_eq!(retained_read_predicate, Some(expected_read_predicate)); - let dag = compile_post_asap_dag(&root).expect("typed DAG compilation failed"); + let dag = post_asap_dag(&root); + let expected_wire = wire_pred(retained_read_predicate.as_ref().unwrap()); assert!(dag.nodes.iter().any(|node| matches!( &node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Filter { pred }, - .. - } if pred == retained_read_predicate.as_ref().unwrap() + PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::Filter { pred }, + } if *pred == expected_wire ))); } @@ -560,15 +597,13 @@ async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { /// read-time operation. #[tokio::test] async fn sql_filter_preserves_local_fallback_boundary_for_unsupported_child() { - let pre_asap = Rc::new( - lower( - "SELECT t.service, t.avg_bytes FROM \ - (SELECT service, AVG(bytes) AS avg_bytes FROM metrics GROUP BY service) t \ - WHERE t.avg_bytes > 100", - AccuracyTarget::Exact, - ) - .await, - ); + let pre_asap = lower( + "SELECT t.service, t.avg_bytes FROM \ + (SELECT service, AVG(bytes) AS avg_bytes FROM metrics GROUP BY service) t \ + WHERE t.avg_bytes > 100", + AccuracyTarget::Exact, + ) + .await; let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection @@ -579,22 +614,20 @@ async fn sql_filter_preserves_local_fallback_boundary_for_unsupported_child() { let mut node = root.as_ref(); let mut saw_filter = false; loop { - match &node.expr { - SummaryExpr::ValueOperation { - child, operation, .. - } => { - saw_filter |= matches!(operation, ValueOperation::Filter { .. }); - node = child; - } - SummaryExpr::KeepPreAsap(fallback) => { - assert!( - matches!(fallback.as_ref(), QueryExpr::BinaryOp { .. }), - "AVG's unsupported rewritten child should be opaque, got {fallback:?}" - ); - break; - } - other => panic!("expected local value operations over fallback child, got {other:?}"), + if let Some(NonASAPOp::BinaryOp { .. }) = node.non_asap() { + assert!( + is_kept_non_asap(node), + "AVG's unsupported rewritten child should be kept whole, got {node:?}" + ); + break; } + saw_filter |= matches!(node.non_asap(), Some(NonASAPOp::Filter { .. })); + node = unary_child(node).unwrap_or_else(|| { + panic!( + "expected local value operations over fallback child, got {:?}", + node.operator + ) + }); } assert!(saw_filter, "supported Filter must remain explicit"); } @@ -605,7 +638,7 @@ async fn sql_filter_preserves_local_fallback_boundary_for_unsupported_child() { /// ```text /// SummaryEstimate { query: Quantile{0.99} } → {…: Float64} /// └─ SummaryAgg { Kll{k:269}, input: metrics.latency } → {…: Sketch(Kll, {k:269})} -/// └─ KeepPreAsap(Scan) → {ts, service, latency, bytes} +/// └─ Scan (kept non-ASAP) → {ts, service, latency, bytes} /// ``` /// /// The SQL counterpart of `promql_to_post_asap.rs`'s @@ -623,14 +656,14 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { let agg = inner_aggregate(&pre_asap); let root = realize(agg).expect("binding failed"); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; - assert!(matches!(query, SketchQuery::Quantile { q } if *q == 0.99)); + assert!(matches!(query, SketchStatistic::Quantile { q } if *q == 0.99)); assert_eq!( root.schema.fields.len(), 1, @@ -638,23 +671,23 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { ); assert_eq!( root.schema.fields[0].dtype, - SummaryFamilyType::Plain(DataType::Float64), + FieldDataType::Plain(DataType::Float64), "the summary-state type must not propagate past the estimate" ); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, reduction, .. - } = &summary_input.expr + }) = &summary_input.operator else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) @@ -674,22 +707,24 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { ); assert_eq!( summary_input.schema.fields[0].dtype, - SummaryFamilyType::Sketch( + FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) ); - let SummaryExpr::KeepPreAsap(kept_leaf) = &child.expr else { - panic!("expected KeepPreAsap leaf, got {:?}", child.expr); - }; - assert!(matches!(kept_leaf.as_ref(), QueryExpr::Scan { .. })); + assert!( + is_kept_non_asap(child), + "expected a kept non-ASAP leaf, got {:?}", + child.operator + ); + assert!(matches!(child.non_asap(), Some(NonASAPOp::Scan { .. }))); assert!( child .schema .fields .iter() - .all(|f| matches!(f.dtype, SummaryFamilyType::Plain(_))), + .all(|f| matches!(f.dtype, FieldDataType::Plain(_))), "logical edges carry only plain columns" ); } @@ -710,32 +745,32 @@ async fn sql_count_distinct_with_epsilon_binds_hll_rse_over_named_column() { let agg = inner_aggregate(&pre_asap); let root = realize(agg).expect("binding failed"); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; - assert!(matches!(query, SketchQuery::Cardinality)); + assert!(matches!(query, SketchStatistic::Cardinality)); assert_eq!( root.schema.fields[0].dtype, - SummaryFamilyType::Plain(DataType::Int64), + FieldDataType::Plain(DataType::Int64), "COUNT(DISTINCT …) reads back out as an integer count" ); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, input, reduction, .. - } = &summary_input.expr + }) = &summary_input.operator else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Hll, SketchParams::Hll { precision: 14 }), GroupingStrategy::default() ) @@ -752,7 +787,7 @@ async fn sql_count_distinct_with_epsilon_binds_hll_rse_over_named_column() { /// An exact workload binds zero sketches: `SUM(bytes) GROUP BY service` at /// `AccuracyTarget::Exact` still gets its mergeable exact accumulator, and -/// `AVG(bytes)` (non-mergeable) stays a whole logical subtree untouched. SQL +/// `AVG(bytes)` (non-mergeable) stays a whole logical sub-DAG untouched. SQL /// counterpart of `promql_to_post_asap.rs`'s /// `promql_exact_workload_binds_accumulators_not_sketches`. #[tokio::test] @@ -764,15 +799,15 @@ async fn sql_exact_workload_binds_accumulators_not_sketches() { .await; let agg = inner_aggregate(&pre_asap); let root = realize(agg).expect("binding failed"); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, reduction, .. - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryAgg, got {:?}", root.expr); + panic!("expected SummaryAgg, got {:?}", root.operator); }; assert_eq!( family, - &SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + &FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) ); assert_eq!( reduction, @@ -781,7 +816,7 @@ async fn sql_exact_workload_binds_accumulators_not_sketches() { ); assert_eq!( dtype(&root.schema, "service"), - &SummaryFamilyType::Plain(DataType::Utf8), + &FieldDataType::Plain(DataType::Utf8), "group keys pass through verbatim" ); @@ -789,37 +824,42 @@ async fn sql_exact_workload_binds_accumulators_not_sketches() { let agg = inner_aggregate(&pre_asap); let root = realize(agg).expect("binding failed"); assert!( - matches!(root.expr, SummaryExpr::KeepPreAsap(_)), + is_kept_non_asap(&root), "avg has no mergeable accumulator — stays logical" ); + assert!( + root.guarantee.as_ref().is_some_and(|g| g.is_exact()), + "a kept logical subtree is exact" + ); } #[tokio::test] async fn map_projection_export_preserves_unsupported_child_boundary() { - let pre = Rc::new(lower_sql_dialect( + let pre = lower_sql_dialect( "SELECT map('job', t.service) AS labels, t.avg_bytes FROM (SELECT service, AVG(bytes) AS avg_bytes FROM metrics GROUP BY service) t WHERE t.avg_bytes > 100", &catalog(), SqlDialect::ClickhouseSQL, AccuracyTarget::Exact, - ).await.unwrap()); + ).await.unwrap(); let space = search_workload(vec![("map_query", pre)]); let root = space .global_selection(&DefaultCostModel) .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&root).unwrap(); + let dag = post_asap_dag(&root); assert!(dag.nodes.iter().any(|node| matches!(&node.payload, - PostAsapOperatorPayload::Value { operation: ValueOperation::Project { cols, .. }, .. } - if cols.iter().any(|item| matches!(&item.expr, QueryExpr::FunctionCall { name, .. } if name == "map")) + PostAsapOperatorPayload::Relational { operator: NonASAPOpKind::Project { cols, .. } } + if cols.iter().any(|item| matches!(&item.expr, WireScalarExpr::FunctionCall { name, .. } if name == "map")) ))); let mut node = root.as_ref(); loop { - match &node.expr { - SummaryExpr::ValueOperation { child, .. } => node = child, - SummaryExpr::KeepPreAsap(child) => { - assert!(matches!(child.as_ref(), QueryExpr::BinaryOp { .. })); - break; - } - other => panic!("unexpected map/fallback composition: {other:?}"), + if let Some(NonASAPOp::BinaryOp { .. }) = node.non_asap() { + assert!( + is_kept_non_asap(node), + "fallback child must stay whole: {node:?}" + ); + break; } + node = unary_child(node) + .unwrap_or_else(|| panic!("unexpected map/fallback composition: {:?}", node.operator)); } } diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index d3d73121b..d25553812 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -2,6 +2,9 @@ //! source workload -> PromQL lowering -> candidate search -> //! summary-maintenance lifecycle selection -> materialized deployment guarantees. +use asap_types::ir::export::PostAsapOperatorPayload; +use asap_types::ir::ASAPOp; +use physical_common::compile_post_asap_dag; use std::rc::Rc; use asap_aware_mapping::cost_model::Cost; @@ -13,8 +16,9 @@ use asap_aware_mapping::{ SummaryMaintenanceLifecycleCostInputs, SummaryMaintenanceLifecycleRejection, WorkloadDemand, }; use asap_frontend_promql::lower_promql_workload; +use asap_types::ir::OperatorNode; use asap_types::post_asap::{ - EvaluationSchedule, SummaryMaintenanceLifecycle, SummaryMaintenanceMode, SummaryNode, + EvaluationSchedule, SummaryMaintenanceLifecycle, SummaryMaintenanceMode, }; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::types::AccuracyTarget; @@ -32,7 +36,7 @@ struct FullyCostedRuntime; impl CostModel for FullyCostedRuntime { fn raw_query_recompute_total_cost( &self, - _target: &asap_types::pre_asap::QueryExpr, + _target: &OperatorNode, _expected_reads: f64, ) -> Option { Some(Cost(1_000.0)) @@ -48,7 +52,7 @@ impl CostModel for FullyCostedRuntime { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(10.0)), @@ -61,7 +65,7 @@ impl CostModel for FullyCostedRuntime { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -179,8 +183,8 @@ fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() .as_array() .unwrap() .iter() - .find(|node| node["kind"] == "SummaryAgg") - .expect("exported SummaryAgg node"); + .find(|node| node["kind"] == "summary_agg") + .expect("exported summary_agg node"); assert_eq!( summary_node["detail"]["summary_maintenance"]["selected"]["lifecycle"]["kind"], "continuously_maintained" @@ -217,11 +221,11 @@ fn selected_plan_with_horizon( fn selected_plan_for_lowered( workload: &PlanningWorkload, - lowered: asap_types::pre_asap::QueryExpr, + lowered: Rc, model: &dyn CostModel, horizon: Horizon, ) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { - let root = Rc::new(lowered); + let root = lowered; let strategies = asap_aware_mapping::default_strategies_with(model); let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); let target = Rc::clone(&space.roots[0].1); @@ -273,10 +277,7 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { runtime::Scope, values::{Batch, Value}, }; - use asap_types::{ - post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, - pre_asap::DataType, - }; + use asap_types::{post_asap::FieldDataType, pre_asap::DataType}; use std::{collections::BTreeMap, sync::Arc}; let mut workload = dashboard_workload(); workload.query_workload.query_batch.as_mut().unwrap()[0].query = @@ -365,10 +366,8 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { .fields .iter() .map(|field| match field.dtype { - SummaryFamilyType::Plain(DataType::Float64) => { - Value::Float64(f64::from(value)) - } - SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), + FieldDataType::Plain(DataType::Float64) => Value::Float64(f64::from(value)), + FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), _ => panic!("unexpected field {field:?}"), }) .collect() @@ -448,7 +447,7 @@ fn quantile_workload(query: &str) -> PlanningWorkload { fn lifecycle_timed_dag( query: &str, lifecycle: &SummaryMaintenanceLifecycle, -) -> (asap_types::post_asap::PostAsapDag, Vec) { +) -> (asap_types::ir::export::PostAsapDag, Vec) { use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; let workload = quantile_workload(query); let mut lowered = lower_promql_workload(&workload, 0).unwrap().remove(0); @@ -489,7 +488,7 @@ fn lifecycle_timed_dag( /// Compile inputs for a timed DAG: its raw source, available at either phase. fn raw_inputs( - dag: &asap_types::post_asap::PostAsapDag, + dag: &asap_types::ir::export::PostAsapDag, ) -> std::collections::BTreeMap { let raw = dag .nodes @@ -497,7 +496,9 @@ fn raw_inputs( .find(|node| { matches!( node.payload, - asap_types::post_asap::PostAsapOperatorPayload::Fallback { .. } + asap_types::ir::export::PostAsapOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::TimeRange { .. } + } ) }) .unwrap(); @@ -529,7 +530,7 @@ fn planner_lifecycle_selection_reproduces_strategy_timing() { != SummaryMaintenanceLifecycle::Ephemeral }) })); - let strategy = asap_types::post_asap::compile_post_asap_dag(&plan.root).unwrap(); + let strategy = compile_post_asap_dag(&plan.root).unwrap(); assert_eq!(plan.execution_timed_dag().unwrap(), strategy, "{query}"); } } @@ -544,7 +545,7 @@ fn chosen_lifecycle_timing_decides_precompute_contents() { runtime::Scope, values::{Batch, Value}, }; - use asap_types::{post_asap::SummaryFamilyType, pre_asap::DataType}; + use asap_types::{post_asap::FieldDataType, pre_asap::DataType}; use std::collections::BTreeMap; let mut answers = Vec::new(); @@ -568,10 +569,8 @@ fn chosen_lifecycle_timing_decides_precompute_contents() { .fields .iter() .map(|field| match field.dtype { - SummaryFamilyType::Plain(DataType::Float64) => { - Value::Float64(f64::from(value)) - } - SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), + FieldDataType::Plain(DataType::Float64) => Value::Float64(f64::from(value)), + FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), _ => panic!("unexpected field {field:?}"), }) .collect() @@ -705,15 +704,12 @@ fn chosen_population_lifecycle_decides_precompute_contents() { runtime::Scope, values::{Batch, Value}, }; - use asap_types::post_asap::{ - maintained_population::PopulationInput, PostAsapOperatorPayload, ValueOperation, - }; + use asap_types::post_asap::maintained_population::PopulationInput; use std::{collections::BTreeMap, sync::Arc}; let workload = quantile_workload("topk by(job)(1, m)"); - let root = Rc::new( - with_series_identity(&lower_promql_workload(&workload, 0).unwrap().remove(0)).unwrap(), - ); + let root = + with_series_identity(&lower_promql_workload(&workload, 0).unwrap().remove(0)).unwrap(); let root = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) .candidate(&root) .unwrap(); @@ -745,10 +741,7 @@ fn chosen_population_lifecycle_decides_precompute_contents() { .execution_timed_dag() .unwrap(); let population = dag.nodes.iter().find(|node| node.id == id).unwrap(); - let PostAsapOperatorPayload::Value { - operation: ValueOperation::MaintainPopulation { population }, - } = &population.payload - else { + let PostAsapOperatorPayload::MaintainPopulation { population } = &population.payload else { panic!("the deployment is the maintained population"); }; let PopulationInput::CurrentSeries(spec) = &population.input else { @@ -758,7 +751,14 @@ fn chosen_population_lifecycle_decides_precompute_contents() { let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::TimeRange { .. } + } + ) + }) .unwrap(); let (raw_id, schema) = (u64::from(raw.id.0), Arc::new(raw.output_schema.clone())); let frontier = @@ -840,22 +840,18 @@ fn chosen_population_lifecycle_decides_precompute_contents() { fn grouped_rate_sum_placement_is_a_lifecycle_choice() { use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; use asap_physical_operators::physical_planner::{compile_candidate, InputContract}; - use asap_types::post_asap::{ - ExactKind, PostAsapOperatorPayload, SummaryExpr, SummaryFamilyType, - }; + use asap_types::post_asap::{ExactKind, FieldDataType}; use std::{collections::BTreeMap, sync::Arc}; let workload = quantile_workload("sum by(job)(rate(m[1m]))"); - let root = Rc::new( - asap_physical_operators::physical_planner::promql_rows::with_series_identity( - &lower_promql_workload(&workload, 0).unwrap().remove(0), - ) - .unwrap(), - ); - let is_exact = |node: &SummaryNode, kind: ExactKind| { - matches!(&node.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(k, _), .. - } if *k == kind) + let root = asap_physical_operators::physical_planner::promql_rows::with_series_identity( + &lower_promql_workload(&workload, 0).unwrap().remove(0), + ) + .unwrap(); + let is_exact = |node: &OperatorNode, kind: ExactKind| { + matches!(&node.operator, asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(k, _), .. + }) if *k == kind) }; let inventory = asap_aware_mapping::search_workload(vec![("q", root)]) .enumerate_candidate_dags(4096) @@ -865,7 +861,7 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { .into_iter() .map(|mut forest| forest.remove(0).1) .filter(|candidate| { - matches!(&candidate.expr, SummaryExpr::ValueOperation { child, .. } + matches!(&candidate.operator, asap_types::ir::Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) if is_exact(child, ExactKind::Sum)) }) .collect::>(); @@ -911,7 +907,14 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::TimeRange { .. } + } + ) + }) .unwrap(); let frontier = asap_physical_operators::physical_planner::frontier_from_timing(&dag).unwrap(); @@ -952,7 +955,7 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { }; let state = |payload: &PostAsapOperatorPayload, kind: ExactKind| { matches!(payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(k, _), .. + family: FieldDataType::ExactAggregate(k, _), .. } if *k == kind) }; assert!(state(retained, ExactKind::Sum)); @@ -965,8 +968,8 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { /// The lifecycle-timed DAG Planner selects for `query` with upfront series /// typing, and whether it keeps an ingestion-time Binary. -fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDag, bool) { - use asap_types::post_asap::{ExecutionTiming, PostAsapOperatorPayload}; +fn typed_selection(query: &str) -> (asap_types::ir::export::PostAsapDag, bool) { + use asap_types::post_asap::ExecutionTiming; let workload = quantile_workload(query); let lowered = asap_types::pre_asap::schema::with_promql_series_identity( &lower_promql_workload(&workload, 0).unwrap().remove(0), @@ -976,8 +979,12 @@ fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDag, bool) { .execution_timed_dag() .unwrap(); let ingestion_binary = dag.nodes.iter().any(|node| { - matches!(node.payload, PostAsapOperatorPayload::Binary { .. }) - && node.output_state.timing == ExecutionTiming::IngestionTime + matches!( + node.payload, + PostAsapOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::BinaryOp { .. } + } + ) && node.output_state.timing == ExecutionTiming::IngestionTime }); (dag, ingestion_binary) } @@ -985,58 +992,52 @@ fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDag, bool) { /// Execute a timed DAG's precompute and query graphs over `samples` /// (`(metric, job, seconds, value)`) at 300s; returns the root's values. fn execute_timed( - dag: &asap_types::post_asap::PostAsapDag, + dag: &asap_types::ir::export::PostAsapDag, samples: &[(&str, &str, i64, f64)], ) -> Vec { use asap_physical_operators::{ - physical_planner::{ - compile_candidate, frontier_from_timing, promql_fallback, promql_rows, InputContract, - }, + physical_planner::{compile_candidate, frontier_from_timing, promql_rows, InputContract}, runtime::Scope, values::{Batch, Value}, }; - use asap_types::{ - post_asap::PostAsapOperatorPayload, - pre_asap::{QueryExpr, Source}, - }; + use asap_types::{ir::export::PostAsapOperatorPayload, pre_asap::Source}; use std::{collections::BTreeMap, sync::Arc}; // Raw inputs: a selector Fallback is itself the input; a retained // expression reads each of its selectors through its raw-series slots. let mut raw = BTreeMap::new(); for node in &dag.nodes { - let PostAsapOperatorPayload::Fallback { expression } = &node.payload else { + if !matches!( + node.payload, + PostAsapOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::TimeRange { .. } + | asap_types::ir::export::NonASAPOpKind::Scan { .. } + } + ) { continue; - }; - let metric = |selector: &QueryExpr| match selector { - QueryExpr::TimeRange { child, .. } => match child.as_ref() { - QueryExpr::Scan { - source: Source::TimeSeries { metric }, - .. - } => Some(metric.clone()), - _ => None, - }, - QueryExpr::Scan { - source: Source::TimeSeries { metric }, - .. - } => Some(metric.clone()), - _ => None, - }; - if let Some(name) = metric(expression) { - raw.insert( - u64::from(node.id.0), - (Arc::new(node.output_schema.clone()), name), - ); - } else { - for (i, (selector, schema)) in promql_fallback::raw_series(expression) - .unwrap() - .into_iter() - .enumerate() + } + let mut id = node.id; + loop { + let n = dag.nodes.iter().find(|n| n.id == id).unwrap(); + if let PostAsapOperatorPayload::Relational { + operator: + asap_types::ir::export::NonASAPOpKind::Scan { + source: Source::TimeSeries { metric }, + .. + }, + } = &n.payload { raw.insert( - promql_fallback::raw_series_input(u64::from(node.id.0), i), - (schema, metric(&selector).unwrap()), + u64::from(node.id.0), + (Arc::new(node.output_schema.clone()), metric.clone()), ); + break; } + id = dag + .edges + .iter() + .find(|e| e.consumer == id) + .unwrap() + .producer; } } let batch = |schema: &asap_physical_operators::values::Schema, name: &str| { @@ -1132,7 +1133,7 @@ fn maintained_arithmetic_over_different_selectors_matches_prometheus() { } /// Arithmetic over one selector keeps its maintained layout and adds each -/// series' two readouts before the quantile. +/// series' two evaluations before the quantile. #[test] fn maintained_arithmetic_over_one_selector_executes() { let (dag, ingestion_binary) = diff --git a/crates/integration-tests/tests/time_range.rs b/crates/integration-tests/tests/time_range.rs index d3ab732fe..68d1ee17a 100644 --- a/crates/integration-tests/tests/time_range.rs +++ b/crates/integration-tests/tests/time_range.rs @@ -1,46 +1,52 @@ -//! `QueryExpr::TimeRange` — range / streaming function tests. +//! `NonASAPOp::TimeRange` — range / streaming function tests. //! -//! All range functions lower to `Aggregate { child: TimeRange { range, child: Scan } }`. +//! All range functions lower to `Aggregate { child: TimeRange { range, kind: Range, child: Scan } }`. //! The temporal range lives on the `TimeRange` node, not in the `AggIntent`. //! `rate` / `increase` use `AggIntent::Rate` / `AggIntent::Increase` (no window field). //! `*_over_time` functions reuse the corresponding cross-series intents -//! (`Count`, `Sum`, `Quantile`, …) — the `TimeRange` child is what marks them -//! as per-series reductions. +//! (`Count`, `Sum`, `Quantile`, …) — the `Range` selector child is what marks +//! them as per-series reductions. use std::rc::Rc; use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction, Source}; +use asap_types::ir::{NonASAPOp, OperatorNode, TimeRangeKind}; +use asap_types::pre_asap::{AggIntent, Reduction, Source}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn scan(metric: &str) -> QueryExpr { - QueryExpr::Scan { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::non_asap_node(op).expect("fixture node derives its schema") +} + +fn scan(metric: &str) -> Rc { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: metric_schema(&[]), - } + }) } -fn range_agg(range_secs: u64, intent: AggIntent, metric: &str) -> QueryExpr { - QueryExpr::Aggregate { +fn range_agg(range_secs: u64, intent: AggIntent, metric: &str) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: node(NonASAPOp::TimeRange { range: Duration::from_secs(range_secs), - child: Rc::new(scan(metric)), + kind: TimeRangeKind::Range, + child: scan(metric), }), - } + }) } // #13 — rate: counter-reset-aware per-second rate; range on TimeRange node diff --git a/crates/metricsql-parser-vendored/src/optimizer/const_evaluator.rs b/crates/metricsql-parser-vendored/src/optimizer/const_evaluator.rs index 3ac1170e8..6befd206d 100644 --- a/crates/metricsql-parser-vendored/src/optimizer/const_evaluator.rs +++ b/crates/metricsql-parser-vendored/src/optimizer/const_evaluator.rs @@ -12,7 +12,7 @@ use crate::functions::{BuiltinFunction, TransformFunction}; use crate::parser::{parse_number, ParseError, ParseResult}; #[allow(rustdoc::private_intra_doc_links)] -/// Partially evaluate `Expr`s so constant subtrees are evaluated at plan time. +/// Partially evaluate `Expr`s so constant sub-DAGs are evaluated at plan time. /// /// Note it does not handle algebraic rewrites such as `(a or false)` /// --> `a`, which is handled by [`Simplifier`] diff --git a/crates/planner/src/lib.rs b/crates/planner/src/lib.rs index 88a085bf8..4f527f148 100644 --- a/crates/planner/src/lib.rs +++ b/crates/planner/src/lib.rs @@ -11,15 +11,13 @@ //! and a catalog — skips this crate and calls //! [`asap_aware_mapping::optimize`] directly. -use std::rc::Rc; - use asap_types::parsed_workload::{ParsedWorkload, ParsedWorkloadError}; -use asap_types::pre_asap::query_expr::QueryExpr; use asap_types::workload::{PlanningWorkload, QueryLanguage, SqlDialect, WorkloadError}; -use asap_frontend_metricsql::{lower_metricsql, MetricsqlError}; +use asap_frontend_metricsql::{lower_metricsql_query, MetricsqlError}; use asap_frontend_promql::{ - lower_promql_workload, lower_promql_workload_with_histograms, HistogramCatalog, PromqlError, + lower_promql_query_workload, lower_promql_query_workload_with_histograms, HistogramCatalog, + PromqlError, }; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog, SqlError}; @@ -191,7 +189,7 @@ pub async fn e2e_plan(input: UserInput<'_>) -> Result { input.validate()?; let exprs = lower(&input).await?; - let parsed = ParsedWorkload::new(input.workload.clone(), exprs)?; + let parsed = ParsedWorkload::from_roots(input.workload.clone(), exprs)?; let fallback = MajorPass; let pass: &dyn OptimizationPass = input.pass.unwrap_or(&fallback); @@ -206,7 +204,7 @@ pub async fn e2e_plan(input: UserInput<'_>) -> Result { /// through `lower_sql_batch`, which walks `query_batch` alone and would drop /// every repeating query — exactly the entries whose recurrence the lifecycle /// stage needs. -async fn lower(input: &UserInput<'_>) -> Result>, PlanError> { +async fn lower(input: &UserInput<'_>) -> Result, PlanError> { let entries = || input.workload.query_workload.entries(); match &input.frontend_specific { @@ -229,7 +227,7 @@ async fn lower(input: &UserInput<'_>) -> Result>, PlanError> { entry_index: Some(index), source: LoweringError::Sql(source), })?; - lowered.push(Rc::new(expr)); + lowered.push(expr.into()); } Ok(lowered) } @@ -237,28 +235,29 @@ async fn lower(input: &UserInput<'_>) -> Result>, PlanError> { now_ms, histograms, .. } => { let lowered = match histograms { - Some(histograms) => lower_promql_workload_with_histograms( + Some(histograms) => lower_promql_query_workload_with_histograms( input.workload, histograms.clone(), *now_ms, ), - None => lower_promql_workload(input.workload, *now_ms), + None => lower_promql_query_workload(input.workload, *now_ms), } .map_err(|source| PlanError::Lowering { entry_index: None, source: LoweringError::Promql(source), })?; - Ok(lowered.into_iter().map(Rc::new).collect()) + Ok(lowered) } FrontendInput::Metricsql => { let mut lowered = Vec::new(); for (index, entry) in entries().enumerate() { - let expr = lower_metricsql(&entry.query.0, entry.requirements.accuracy.target()) - .map_err(|source| PlanError::Lowering { - entry_index: Some(index), - source: LoweringError::Metricsql(source), - })?; - lowered.push(Rc::new(expr)); + let expr = + lower_metricsql_query(&entry.query.0, entry.requirements.accuracy.target()) + .map_err(|source| PlanError::Lowering { + entry_index: Some(index), + source: LoweringError::Metricsql(source), + })?; + lowered.push(expr); } Ok(lowered) } diff --git a/crates/planner/tests/e2e_plan.rs b/crates/planner/tests/e2e_plan.rs index c3ff85d88..3ab55b101 100644 --- a/crates/planner/tests/e2e_plan.rs +++ b/crates/planner/tests/e2e_plan.rs @@ -12,8 +12,7 @@ use asap_aware_mapping::{ }; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; use asap_planner::{e2e_plan, FrontendInput, PlanError, UserInput, UserInputError}; -use asap_types::post_asap::SummaryExpr; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataArrival, DataWorkload, DurationMs, Evidence, @@ -51,8 +50,8 @@ fn lineitem_catalog() -> SqlCatalog { SqlCatalog::new().with_table( "lineitem", Schema::new(vec![ - Column::new("l_orderkey", DataType::Int64, false), - Column::new("l_extendedprice", DataType::Float64, false), + Field::plain("l_orderkey", DataType::Int64, false), + Field::plain("l_extendedprice", DataType::Float64, false), ]), ) } @@ -137,7 +136,7 @@ async fn builtin_cost_model_cannot_price_lifecycles_and_falls_back_to_raw_recomp ) .await .expect("lowers"); - roots.push((index, Rc::new(expr), Some(accuracy))); + roots.push((index, expr, Some(accuracy))); } let strategies = default_strategies_with_evidence(models.cost, models.evidence); let space = search_workload_with_targets(roots, &strategies, models.accuracy); @@ -150,12 +149,12 @@ async fn builtin_cost_model_cannot_price_lifecycles_and_falls_back_to_raw_recomp .expect("assembles") .expect("root has a group"); assert!( - !matches!(cost_only.expr, SummaryExpr::KeepPreAsap(_)), + cost_only.contains_asap(), "entry {}: cost-only selection was expected to pick a summary", plan.entry_index ); assert!( - matches!(plan.plan.root.expr, SummaryExpr::KeepPreAsap(_)) + !plan.plan.root.contains_asap() && plan.plan.selected_raw_recompute && plan.plan.deployments.is_empty() && plan.plan.summary_total_cost.is_none() @@ -390,7 +389,7 @@ async fn lifecycle_decisions_ride_inside_each_plan() { assert_eq!(output.plans.len(), 1); assert_eq!(output.plans[0].entry_index, 0); let _: &Rc<_> = &output.plans[0].plan.root; - assert_eq!(output.dags().len(), 1); + assert_eq!(output.operator_roots().len(), 1); } /// Each root's lifecycle is planned against the entries that read it: a @@ -437,3 +436,54 @@ async fn each_plan_counts_only_its_own_entries_reads() { let reads: Vec<_> = output.plans.iter().map(|p| p.plan.expected_reads).collect(); assert_eq!(reads, vec![Some(60.0), Some(6.0)]); } + +/// Scalar-only and mixed workloads preserve entry bindings without wrapper nodes. +#[tokio::test] +async fn scalar_roots_survive_planning_in_workload_order() { + for queries in [ + vec!["2", "time()"], + vec!["2", "up * 2", "scalar(sum(up)) + 1"], + ] { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(queries.iter().map(|q| batch(q)).collect()), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let output = e2e_plan(UserInput::new( + &workload, + FrontendInput::Promql { + now_ms: NOW_MS, + histograms: None, + }, + PlanningModels::builtin(), + lifecycle(), + )) + .await + .unwrap(); + assert_eq!( + output.entry_indices(), + (0..queries.len()).collect::>() + ); + assert!(matches!( + output.roots()[0], + asap_types::ir::QueryRoot::Scalar(_) + )); + assert_eq!(output.roots().len(), queries.len()); + if queries.len() == 3 { + assert_eq!(output.plans[0].entry_index, 1); + let asap_types::ir::QueryRoot::Scalar(expr) = &output.roots()[2] else { + panic!() + }; + assert_eq!(expr.operator_refs().len(), 1); + } + } +} diff --git a/crates/planner/tests/summary_sharing.rs b/crates/planner/tests/summary_sharing.rs index 829c38a9c..ebd6e2250 100644 --- a/crates/planner/tests/summary_sharing.rs +++ b/crates/planner/tests/summary_sharing.rs @@ -1,6 +1,8 @@ //! Structurally identical summary producers chosen by different queries are -//! shared after Pass 1: one `Rc` across their plans, costed once. +//! shared after Pass 1: one `Rc` across their plans, costed once. +use asap_types::ir::cse::share_common_subdags; +use asap_types::ir::{ASAPOp, OperatorNode}; use std::rc::Rc; use asap_aware_mapping::accuracy::{ @@ -11,7 +13,7 @@ use asap_aware_mapping::pass::{PlanOutput, PlanningModels}; use asap_aware_mapping::replacement::{default_size_params, DEFAULT_DELTA}; use asap_aware_mapping::{ global_selection_with_summary_maintenance_lifecycles, search_workload_with_targets, - ReplacementStrategy, SketchAlgorithmStrategy, WorkloadDemand, + ASAPStrategies, ReplacementStrategy, WorkloadDemand, }; use asap_aware_mapping::{ CostModel, CostRate, DefaultCostModel, Horizon, LifecycleInput, SummaryMaintenanceCapabilities, @@ -21,15 +23,13 @@ use asap_frontend_promql::lower_promql_workload; use asap_frontend_sql::SqlCatalog; use asap_planner::{e2e_plan, FrontendInput, UserInput}; use asap_types::post_asap::{ - share_common_summary_subtrees, AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, - ProbabilityExpr, ResultGuarantee, SketchQuery, -}; -use asap_types::post_asap::{ - SketchAlgorithm, SketchParams, SummaryExpr, SummaryFamilyType, SummaryNode, + AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ProbabilityExpr, ResultGuarantee, + SketchStatistic, }; +use asap_types::post_asap::{FieldDataType, SketchAlgorithm, SketchParams}; use asap_types::pre_asap::agg_intent::default_quantile; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; -use asap_types::pre_asap::{AggIntent, QueryExpr}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::pre_asap::AggIntent; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, DataArrival, DataWorkload, DurationMs, Evidence, LatencyRequirement, @@ -58,7 +58,7 @@ impl CostModel for FixedCosts { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(self.build)), @@ -71,7 +71,7 @@ impl CostModel for FixedCosts { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -80,7 +80,7 @@ impl CostModel for FixedCosts { } } - fn raw_query_recompute_cost(&self, _target: &QueryExpr) -> Option { + fn raw_query_recompute_cost(&self, _target: &OperatorNode) -> Option { Some(Cost(self.raw_per_read)) } } @@ -171,8 +171,8 @@ async fn plan_sql(queries: &[&str], costs: &FixedCosts) -> PlanOutput { let catalog = SqlCatalog::new().with_table( "lineitem", Schema::new(vec![ - Column::new("l_orderkey", DataType::Int64, false), - Column::new("l_extendedprice", DataType::Float64, false), + Field::plain("l_orderkey", DataType::Int64, false), + Field::plain("l_extendedprice", DataType::Float64, false), ]), ); let input = UserInput::new( @@ -185,7 +185,7 @@ async fn plan_sql(queries: &[&str], costs: &FixedCosts) -> PlanOutput { } /// Every summary state each plan deploys. -fn states(output: &PlanOutput) -> Vec>> { +fn states(output: &PlanOutput) -> Vec>> { output .plans .iter() @@ -202,7 +202,7 @@ fn states(output: &PlanOutput) -> Vec>> { } /// Whether the two plans deploy exactly the same states, by pointer. -fn same_states(states: &[Vec>]) -> bool { +fn same_states(states: &[Vec>]) -> bool { states[0].len() == states[1].len() && states[0] .iter() @@ -212,7 +212,7 @@ fn same_states(states: &[Vec>]) -> bool { /// The deployments a consumer would run, deduplicated by pointer. fn unique_deployments(output: &PlanOutput) -> usize { - let mut seen: Vec<*const SummaryNode> = Vec::new(); + let mut seen: Vec<*const OperatorNode> = Vec::new(); for plan in &output.plans { for deployment in &plan.plan.deployments { let ptr = Rc::as_ptr(&deployment.summary); @@ -297,12 +297,12 @@ fn kll_k(plan: &asap_aware_mapping::pass::QueryLifecyclePlan) -> u32 { let [deployment] = plan.plan.deployments.as_slice() else { panic!("one state: {:?}", plan.plan.deployments.len()); }; - let SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + let asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), .. - } = &deployment.summary.expr + }) = &deployment.summary.operator else { - panic!("sketch state: {:?}", deployment.summary.expr); + panic!("sketch state: {:?}", deployment.summary.operator); }; let SketchParams::Kll { k } = kind.params() else { panic!("KLL state: {kind:?}"); @@ -382,7 +382,7 @@ async fn identical_sql_percentiles_share_one_producer() { assert_eq!(unique_deployments(&output), 1); } -/// The quantile is a readout parameter: SQL p50 and p99 over one filtered +/// The quantile is a evaluation parameter: SQL p50 and p99 over one filtered /// column build one KLL, named after its input, while each query keeps its /// own output column. #[tokio::test] @@ -451,17 +451,17 @@ async fn shared_amortization_alone_can_beat_raw_recompute() { assert_eq!(unique_deployments(&output), 1); } -/// Synthetic evidence certifying UnivMon readouts; it exercises sharing, never +/// Synthetic evidence certifying UnivMon evaluations; it exercises sharing, never /// runtime accuracy. struct UnivMonEvidence; impl AccuracyModel for UnivMonEvidence { fn local_guarantee( &self, - family: &SummaryFamilyType, - query: &SketchQuery, + family: &FieldDataType, + query: &SketchStatistic, ) -> Option { - if matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) + if matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) { let mut guarantee = ResultGuarantee::exact("SYNTHETIC test evidence; not measured"); guarantee.metric = ErrorMetric::RelativeValue; @@ -493,7 +493,7 @@ impl AccuracyModel for UnivMonEvidence { /// when the states are identical. `MajorPass` builds candidates with the /// built-in accuracy model, so this runs its pipeline with the test model. #[test] -fn certified_frequency_readouts_share_one_univmon_state() { +fn certified_frequency_evaluations_share_one_univmon_state() { let queries = [ ("distinct_over_time(m[5m])", 0.02), ("entropy_over_time(m[5m])", 0.02), @@ -505,12 +505,10 @@ fn certified_frequency_readouts_share_one_univmon_state() { .into_iter() .zip(queries) .enumerate() - .map(|(index, (expr, (_, epsilon)))| { - (index, Rc::new(expr), Some(AccuracyTarget::Epsilon(epsilon))) - }) + .map(|(index, (expr, (_, epsilon)))| (index, expr, Some(AccuracyTarget::Epsilon(epsilon)))) .collect(); let strategies: Vec> = - vec![Box::new(SketchAlgorithmStrategy::new_with_planning_inputs( + vec![Box::new(ASAPStrategies::new_with_planning_inputs( &CHEAP_SUMMARY, &UnivMonEvidence, &EqualSplitAllocator, @@ -541,15 +539,17 @@ fn certified_frequency_readouts_share_one_univmon_state() { (*index, dag) }) .collect(); - let mut states: Vec> = Vec::new(); - for (_, root) in share_common_summary_subtrees(assembled) { - assert!(root.guarantee.is_some(), "{:?}", root.expr); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("summary readout: {:?}", root.expr); + let mut states: Vec> = Vec::new(); + for (_, root) in share_common_subdags(assembled) { + assert!(root.guarantee.is_some(), "{:?}", root.operator); + let asap_types::ir::Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = + &root.operator + else { + panic!("summary evaluation: {:?}", root.operator); }; assert!(matches!( - &summary_input.expr, - SummaryExpr::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } + &summary_input.operator, + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) if kind.algorithm() == &SketchAlgorithm::UnivMon )); states.push(Rc::clone(summary_input)); diff --git a/crates/sql-function-catalog/src/lib.rs b/crates/sql-function-catalog/src/lib.rs index a95a2a632..bfaaac613 100644 --- a/crates/sql-function-catalog/src/lib.rs +++ b/crates/sql-function-catalog/src/lib.rs @@ -277,7 +277,7 @@ pub struct ClickHouseBuiltin { pub const CLICKHOUSE_BUILTINS: &[ClickHouseBuiltin] = &[ // Explicit time-series reducers. These deliberately survive under their // own names: the SQL frontend validates (value, timestamp, window_ms) and - // lowers the window to QueryExpr::TimeRange rather than pretending these + // lowers the window to NonASAPOp::TimeRange rather than pretending these // are ordinary tabular aggregates. ClickHouseBuiltin { name: "asap_rate", diff --git a/crates/types/src/dag_export.rs b/crates/types/src/dag_export.rs index 0fe6977f9..22241a7ef 100644 --- a/crates/types/src/dag_export.rs +++ b/crates/types/src/dag_export.rs @@ -1,31 +1,46 @@ -//! Export the pre-ASAP [`QueryExpr`] tree as a generic node/edge graph, for tools -//! that need to render or diff the IR (the `dag_export` example + the -//! `tools/dag-viewer` viewer — see issue #133) rather than walk it in Rust. +//! Export an [`OperatorNode`] DAG as a generic node/edge graph, for tools +//! that need to render or diff the IR (the `dag_export` devtools binary + +//! the `tools/dag-viewer` viewer — see issue #133) rather than walk it in +//! Rust. //! -//! `QueryExpr` already derives `Serialize`, but as a Rust-shaped tagged tree -//! (`Rc` children nested inside each variant's own field). This module -//! flattens that into an explicit node list + child-id edges — the shape a -//! generic graph renderer wants — and additionally tags each node with -//! [`structural_hash`](crate::pre_asap::cse::structural_hash), so a caller -//! with several exported queries can spot identical subtrees (a -//! shared `Scan`, a repeated `Aggregate` shape, …) by comparing hashes -//! rather than re-implementing `QueryExpr: PartialEq` structural comparison -//! client-side. +//! `OperatorNode` already derives `Serialize`, but as a Rust-shaped tagged +//! tree (`Rc` children nested inside each variant's own field, repeated once +//! per reference). This module flattens that into an explicit node list + +//! child-id edges — one entry per unique node, deduplicated by `Rc` pointer +//! identity, so a shared sub-DAG stays one node with several parents — and +//! additionally tags each node with +//! [`structural_hash`](crate::ir::cse::structural_hash), so a caller with +//! several exported queries can spot identical sub-DAGs (a shared `Scan`, a +//! repeated `Aggregate` shape, …) by comparing hashes rather than +//! re-implementing structural comparison client-side. //! //! This is literally the same hashing -//! [`share_common_subtrees`](crate::pre_asap::cse::share_common_subtrees) -//! uses to bucket candidates in its `InternTable` (issue #223 stage 3) — not -//! a parallel reimplementation. `tools/dag-viewer`'s "shared subtree" +//! [`share_common_subdags`](crate::ir::cse::share_common_subdags) uses to +//! bucket candidates in its `InternTable` (issue #223 stage 3) — not a +//! parallel reimplementation. `tools/dag-viewer`'s "shared sub-DAG" //! highlighting is still a *proxy* for real CSE, though: a hash match here //! only means two nodes are legal `InternTable` bucket-mates (same coarse -//! hash), the same candidate-narrowing step `structural_hash` performs -//! inside `InternTable::intern` — it does not mean `share_common_subtrees` -//! actually ran on this data and merged them onto one `Rc` (that also -//! requires the `PartialEq` check `InternTable::intern` performs, and the +//! hash) — it does not mean `share_common_subdags` actually ran on this +//! data and merged them onto one `Rc` (that also requires the structural +//! equality check `InternTable::intern` performs, and the //! `Schema::has_unique_key` legality gate, neither of which this export //! step evaluates). See `tools/dag-viewer/README.md` for the up-to-date //! caveat. //! +//! There is one IR before and after ASAP optimization, so there is one +//! exporter: an ordinary operator and an ASAP summary operator are both +//! rendered by the same per-variant [`shape`] match, whichever entry point +//! ([`export`], [`export_summary`], [`export_post_asap`]) reached them. +//! +//! ## Scalar expressions +//! +//! A [`ScalarExpr`] is owned by value by an operator field (`Filter.pred`, +//! `Project.cols`, …) and is rendered into that operator's `detail`, not as +//! a node of its own. The operator nodes a scalar expression reads +//! (`scalar(v)`, `EXISTS (subquery)`, …) *are* nodes of the graph — they are +//! in [`OperatorNode::children`] — so inside `detail` each such reference is +//! rendered as `{"scalar_ref": }` rather than inlined. +//! //! ## `DagNode::notes` — a layering seam, not a feature this module implements //! //! [`DagNode`] also carries `notes: Vec<`[`DagNote`]`>`, always empty coming @@ -33,13 +48,12 @@ //! `asap_types`, never the reverse — can annotate an already-exported graph //! after the fact without this module needing to know anything about that //! layer's concepts. Concretely: `asap-aware-mapping`'s `explanation` module -//! (issue #257) computes `structural_hash` over the same `QueryExpr` -//! subtrees this module does (via the identical function). The devtools -//! exporter uses that hash to narrow candidates, then compares -//! `ReplacementExplanation::target` with [`DagNode::source_expr`] for a -//! collision-safe match before pushing a [`DagNote`] onto the node. -//! `asap_types` itself never constructs a `DagNote` — see [`DagNode::notes`] -//! for the layering rule this keeps. +//! (issue #257) computes `structural_hash` over the same nodes this module +//! does (via the identical function). The devtools exporter uses that hash +//! to narrow candidates, then compares its target with +//! [`DagNode::source_node`] for a collision-safe match before pushing a +//! [`DagNote`] onto the node. `asap_types` itself never constructs a +//! `DagNote` — see [`DagNode::notes`] for the layering rule this keeps. use std::collections::HashMap; use std::rc::Rc; @@ -47,9 +61,11 @@ use std::rc::Rc; use serde::Serialize; use crate::cost::CostAnnotation; -use crate::post_asap::{AccuracyError, ResultGuarantee, SummaryExpr, SummaryNode}; -use crate::pre_asap::cse::{structural_hash, HashCache}; -use crate::pre_asap::query_expr::{QueryExpr, Source}; +use crate::ir::cse::{structural_hash, HashCache}; +use crate::ir::operator_properties::Source; +use crate::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, ScalarExpr}; +use crate::post_asap::{AccuracyError, ResultGuarantee}; +use crate::pre_asap::schema::FieldDataType; /// One flattened IR node. `detail` holds this node's own scalar fields /// (predicates, aggregate funcs, schema, sort keys, …) — everything except @@ -57,56 +73,48 @@ use crate::pre_asap::query_expr::{QueryExpr, Source}; #[derive(Debug, Clone, Serialize)] pub struct DagNode { pub id: u32, - /// The `QueryExpr` variant name (e.g. `"Aggregate"`). + /// The operator variant name — [`Operator::kind_name`] (e.g. + /// `"Aggregate"`, `"SummaryAgg"`). pub kind: &'static str, /// Short human-readable summary for a node's collapsed on-graph label. pub label: String, pub detail: serde_json::Value, - /// Output schema carried by every exported node. Edge renderers use the - /// child node's schema as the schema flowing along child → consumer. + /// Output schema carried by every exported node ([`OperatorNode::schema`] + /// as JSON). Edge renderers use the child node's schema as the schema + /// flowing along child → consumer. #[serde(skip_serializing_if = "Option::is_none")] pub schema: Option, - /// Child node ids, in the variant's field order (e.g. `Join` is - /// `[left, right]`). + /// Child node ids in [`OperatorNode::children`] order: the operator's + /// own inputs in field order (e.g. `Join` is `[left, right]`), then the + /// nodes referenced from its scalar expressions. pub children: Vec, /// Explicit workload-wide identity assigned by a higher-level exporter. /// Viewers use this field to union nodes and must not reconstruct a /// structural signature client-side. #[serde(skip_serializing_if = "Option::is_none")] pub workload_node_id: Option, - /// [`structural_hash`](crate::pre_asap::cse::structural_hash) of the - /// subtree rooted at this node — the exact same function `cse`'s - /// `InternTable` uses to bucket CSE candidates, so two nodes hash - /// equally here iff they would land in the same `InternTable` bucket. - /// See the module doc for what a hash match here does and doesn't - /// guarantee. - /// - /// `None` for the same reason `source_expr` is `None` — a post-ASAP- - /// originated node in an [`export_post_asap`] merged graph has no - /// `QueryExpr` to hash. Omitted from JSON entirely (rather than, say, - /// serialized as `0`) so a consumer's shared-subtree-by-hash pass can - /// tell "no hash" apart from a real hash that happens to collide with a - /// placeholder — `0` is a legal `structural_hash` output, not a safe - /// sentinel. + /// [`structural_hash`](crate::ir::cse::structural_hash) of the sub-DAG + /// rooted at this node — the exact same function `cse`'s `InternTable` + /// uses to bucket CSE candidates, so two nodes hash equally here iff they + /// would land in the same `InternTable` bucket. See the module doc for + /// what a hash match here does and doesn't guarantee. Always `Some` + /// for a node this module produces; the `Option` is retained for the + /// JSON shape (`None` is omitted rather than serialized as a sentinel, + /// since `0` is a legal hash). #[serde(skip_serializing_if = "Option::is_none")] pub hash: Option, - /// Exact source expression for in-process annotation matching. It is not - /// part of the JSON format: callers first narrow by `hash`, then compare - /// this value structurally to avoid treating a hash collision as node - /// identity. - /// - /// `None` for a node with no corresponding pre-ASAP `QueryExpr` at all — - /// only possible for a post-ASAP-originated node inside a merged - /// [`export_post_asap`] graph (a `SummaryAgg`/`SummaryJoin`/… node has no - /// single `QueryExpr` it corresponds to). Every node [`export`] itself - /// produces is pre-ASAP by construction and always carries `Some`. + /// The exported node itself, for in-process annotation matching. Not + /// part of the JSON format: callers first narrow by `hash`, then + /// compare this value (by pointer or structurally) to avoid treating a + /// hash collision as node identity. Always `Some` for a node this + /// module produces. #[serde(skip)] - pub source_expr: Option, - /// In-process identity of the source `QueryExpr`. Unlike `source_expr`'s - /// structural value, this preserves an `Rc` child reached from multiple - /// parents so post-ASAP flattening can retain true DAG sharing. + pub source_node: Option>, + /// In-process identity of `source_node` (`Rc::as_ptr` as an address): + /// the key the builder deduplicates on, so a node reached from several + /// parents is exported once. Not part of the JSON format. #[serde(skip)] - source_ptr: Option, + pub source_ptr: Option, /// Arbitrary reporting-layer annotations for this node — e.g. why a /// replacement exists here. `asap_types` never populates this itself /// (it has no notion of a "replacement" at all — see the module doc's @@ -164,10 +172,10 @@ pub struct DagDecision { #[serde(skip_serializing_if = "Option::is_none")] pub selected_cost: Option, /// `baseline_cost.value - selected_cost.value` under `baseline_cost`'s - /// own baseline — for a winning `SharedSubtreeStrategy`/`CseShare` + /// own baseline — for a winning `SharedSubDagStrategy`/`CseShare` /// decision this *is* "avoided recomputation for a shared sub-DAG" (one /// of `dag_export`'s issue #286 granularity items): the baseline is - /// exactly the cost of recomputing this subtree independently at every + /// exactly the cost of recomputing this sub-DAG independently at every /// consumer, so the benefit is exactly what sharing avoided. #[serde(skip_serializing_if = "Option::is_none")] pub benefit: Option, @@ -187,7 +195,7 @@ pub struct EdgeCostAnnotation { pub cost: CostAnnotation, } -/// One query's exported graph. `nodes[root as usize]` is the tree's root. +/// One query's exported graph. `nodes[root as usize]` is the DAG's root. #[derive(Debug, Clone, Serialize)] pub struct DagGraph { pub nodes: Vec, @@ -195,8 +203,7 @@ pub struct DagGraph { /// See [`EdgeCostAnnotation`]. Always empty unless a higher layer /// explicitly populated it (same layering rule as [`DagNode::notes`]); /// omitted from JSON entirely when empty, so every existing producer of - /// [`DagGraph`] (every call to [`export`]/[`export_summary`]) is - /// unaffected. + /// [`DagGraph`] is unaffected. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub edge_annotations: Vec, } @@ -207,8 +214,8 @@ pub struct NamedGraph { pub name: String, /// The original query text (SQL or PromQL) this graph was lowered from, /// for display alongside the graph — not used by `export` itself, since - /// that only sees the already-lowered `QueryExpr`. Optional because not - /// every producer of a `NamedGraph` has the source text on hand. + /// that only sees the already-lowered DAG. Optional because not every + /// producer of a `NamedGraph` has the source text on hand. #[serde(skip_serializing_if = "Option::is_none")] pub source: Option, pub graph: DagGraph, @@ -228,15 +235,13 @@ pub struct NamedGraph { /// [`TargetReplacement::before`]/`::after` (small, self-contained /// before/after pairs, one per independently-discovered replacement /// site), this is a single flattened [`DagGraph`] spanning the whole - /// query: every node that has no winning replacement renders as an - /// ordinary pre-ASAP [`DagNode`] (same shape [`export`] itself - /// produces), and every node that does splices in its winning - /// candidate's shape instead — a rewritten [`QueryExpr`] subtree, or a - /// bound `SummaryNode` subtree, rendered inline in the very same node - /// list. `None` unless a higher layer explicitly built one (e.g. the - /// `dag_export` devtools binary's `--post-asap` flag); omitted from the - /// JSON entirely when absent, so every existing producer/consumer of - /// `NamedGraph` is unaffected. + /// query: every node that has no winning replacement renders as it does + /// in [`export`], and every node that does splices in its winning + /// candidate's sub-DAG instead, in the very same node list. `None` + /// unless a higher layer explicitly built one (e.g. the `dag_export` + /// devtools binary's `--post-asap` flag); omitted from the JSON entirely + /// when absent, so every existing producer/consumer of `NamedGraph` is + /// unaffected. #[serde(default, skip_serializing_if = "Option::is_none")] pub post_graph: Option, /// This query's own selected-workload cost/benefit — one of issue @@ -266,7 +271,7 @@ pub struct NamedGraph { } /// A batch of named queries — the shape the viewer's multi-query / compare -/// mode reads (each query starts its own `DagGraph`; shared-subtree +/// mode reads (each query starts its own `DagGraph`; shared-sub-DAG /// highlighting is done by the viewer, matching `DagNode::hash` across /// queries). #[derive(Debug, Clone, Serialize)] @@ -282,81 +287,51 @@ pub struct WorkloadGraph { pub workload_cost: Option, } -// ── Post-ASAP replacement export — a second, layering-seam-shaped feature ── +// ── Post-ASAP replacement export — a layering-seam-shaped feature ────────── // -// Everything below this point is the post-ASAP counterpart of the pre-ASAP -// flattening above: [`export_summary`] flattens a `SummaryNode` the same way -// [`export`] flattens a `QueryExpr`, and [`TargetReplacement`] is the -// generic, crate-agnostic "one replacement site, before and after" shape a -// higher layer (`asap-aware-mapping`, via the `dag_export` devtools binary's -// `--post-asap` flag) populates after running its own search — the exact -// same layering rule [`DagNode::notes`]'s doc above already states: this -// module never runs `asap_aware_mapping::replacement::search_workload_with` -// itself, never picks a "winning" candidate, and has no opinion on what a +// [`TargetReplacement`] is the generic, crate-agnostic "one replacement +// site, before and after" shape a higher layer (`asap-aware-mapping`, via +// the `dag_export` devtools binary's `--post-asap` flag) populates after +// running its own search — the exact same layering rule [`DagNode::notes`]'s +// doc above already states: this module never runs +// `asap_aware_mapping::replacement::search_workload_with` itself, never +// picks a "winning" candidate, and has no opinion on what a // `ReplacementProvenance` or a cost model even is. It only defines shapes // concrete and serializable enough for a higher layer to fill in, and for // `tools/dag-viewer` to render without needing to know anything about // `asap-aware-mapping`'s own vocabulary. -// -// A single whole-query "post-ASAP tree" isn't attempted here, and isn't -// representable in the current type system either: `SummaryExpr` has no -// variant letting a `SummaryNode` be embedded back inside a plain -// `QueryExpr`'s child slot (`QueryExpr`'s own children are always -// `Rc`, never `Rc`), so there is no way to splice a -// post-ASAP binding back into its original pre-ASAP tree in place. Inventing -// a bridge type for that is a real `asap_types`/`asap-aware-mapping` IR -// design decision, well beyond what a devtools visualization export should -// decide unilaterally. Instead, each independently-discovered replacement -// target gets its own small, self-contained `before`/`after` pair — the -// target's own pre-ASAP subtree, and either the winning `SummaryNode` or the -// winning rewritten `QueryExpr`, both of which *are* fully representable -// today via [`export`]/[`export_summary`] as-is. - -/// One flattened post-ASAP node — the [`SummaryExpr`] analogue of -/// [`DagNode`]. `detail` holds this node's own scalar fields (the summarized -/// column, the summary family, grouping strategy, sketch-query kind, …) — -/// everything except its `SummaryNode` children, which live in `children` -/// instead. -/// -/// Unlike [`DagNode`], this carries no `hash`/`source_expr` pair: nothing in -/// this module ever needs to re-identify a particular `SummaryDagNode` the -/// way `DagNode::hash` lets a higher layer re-identify a pre-ASAP node (a -/// `SummaryNode` is always freshly exported for exactly one -/// [`TargetReplacementAfter::Summary`] site, never matched back against a -/// separately-exported graph the way pre-ASAP notes are). -/// -/// Several of `SummaryExpr`'s own fields (`SummaryFamilyType`, -/// `GroupingStrategy`, `SketchQuery`) derive neither `Serialize` nor -/// `Deserialize` in `asap_types::post_asap` — they carry no reporting -/// obligation there, since nothing before this module ever needed to -/// serialize a post-ASAP node. Rather than adding `Serialize` impls to -/// `post_asap`'s own core types purely for this devtools-facing export (a -/// change to that module's own public API contract, out of scope for a -/// reporting concern), this module renders those particular fields into -/// `detail` via their `Debug` formatting instead — human-readable, and -/// sufficient for the display purpose `detail` exists for on every other -/// node in this file (see [`DagNode::detail`]'s own doc), at the cost of -/// those particular fields being opaque strings rather than structured JSON -/// on the `SummaryDagNode` side of the export. + +/// One flattened node of a [`SummaryDagGraph`] — the same node as a +/// [`DagNode`], in the shape the summary-maintenance consumers read: +/// snake_case `kind`, the accuracy guarantee as its own field, no +/// hash/annotation seams. #[derive(Debug, Clone, Serialize)] pub struct SummaryDagNode { pub id: u32, - /// The `SummaryExpr` variant name (e.g. `"SummaryAgg"`). + /// The operator variant name in snake_case (e.g. `"summary_agg"`, + /// `"scan"`) — see [`snake_case_kind`]. pub kind: &'static str, /// Short human-readable summary for a node's collapsed on-graph label. pub label: String, pub detail: serde_json::Value, - /// Child node ids, in the variant's field order (e.g. `SummaryJoin` is - /// `[outer, inner]`). + /// [`OperatorNode::schema`] as JSON. + #[serde(skip_serializing_if = "Option::is_none")] + pub schema: Option, + /// Child node ids in [`OperatorNode::children`] order. pub children: Vec, /// The value's machine-readable accuracy guarantee (issue #172) — - /// [`SummaryNode::guarantee`] serialized structurally (metric, symbolic + /// [`OperatorNode::guarantee`] serialized structurally (metric, symbolic /// bound, failure probability, provenance including any budget /// allocation), not as prose. Omitted when the node carries none (raw /// summary state, or a family with no error model), so every consumer /// predating this field parses the same shape it always has. #[serde(default, skip_serializing_if = "Option::is_none")] pub guarantee: Option, + /// The exported node itself, so a caller annotating the graph can find + /// a node by `Rc` pointer identity rather than by walk order. Not part + /// of the JSON format. Always `Some`. + #[serde(skip)] + pub source_node: Option>, } /// One accuracy-illegal candidate a higher layer's search refused for a @@ -378,239 +353,14 @@ pub struct TargetRejection { pub error: AccuracyError, } -/// One post-ASAP `SummaryNode` tree, flattened the same way [`DagGraph`] -/// flattens a pre-ASAP `QueryExpr` tree. +/// A DAG flattened into [`SummaryDagNode`]s — the same graph [`DagGraph`] +/// holds, in the summary-maintenance consumers' node shape. #[derive(Debug, Clone, Serialize)] pub struct SummaryDagGraph { pub nodes: Vec, pub root: u32, } -/// Flatten a [`SummaryNode`] the same way [`export`] flattens a `QueryExpr` -/// — post-order, one [`SummaryDagNode`] per [`SummaryExpr`] variant, no -/// memoization of repeated `Rc` references (a shared -/// sub-expression reachable through two parents is flattened twice, into two -/// separate node entries — the same "this is a flattened tree view, not a -/// pointer-identity-preserving graph" behavior [`build`] already has for -/// `QueryExpr`). -/// -/// A `KeepPreAsap(inner)` leaf embeds the *whole* pre-ASAP subtree beneath it -/// as a nested [`DagGraph`] (via [`export(inner)`](export)) inside its own -/// `detail` field (`{"pre_asap_subgraph": }`) rather than trying to -/// flatten it into this same node list — [`DagNode`] and [`SummaryDagNode`] -/// are different types with different id spaces, so mixing them into one -/// `Vec` isn't type-safe; nesting is. `label` for a `KeepPreAsap` node is -/// `format!("KeepPreAsap({kind})")`, where `kind` is the inner subtree's own -/// top-level `DagNode::kind`. -pub fn export_summary(node: &SummaryNode) -> SummaryDagGraph { - let mut nodes = Vec::new(); - let root = build_summary(node, &mut nodes); - SummaryDagGraph { nodes, root } -} - -fn push_summary_node( - nodes: &mut Vec, - kind: &'static str, - label: String, - detail: serde_json::Value, - children: Vec, - guarantee: Option, -) -> u32 { - let id = nodes.len() as u32; - nodes.push(SummaryDagNode { - id, - kind, - label, - detail, - children, - guarantee, - }); - id -} - -/// A short, human-readable label for a [`crate::post_asap::SummaryFamilyType`] -/// (e.g. `"Sketch(Kll)"`, `"ExactAggregate(Sum)"`) — for -/// [`SummaryDagNode::label`] text on a `SummaryAgg`/`SummaryJoin` node. Not -/// exhaustive prose (mirrors `asap_aware_mapping::replacement::describe_intent`'s -/// own "this is a label, not a decision" stance) — every variant is covered, -/// but via `Debug` for the inner kind rather than hand-written prose per -/// algorithm. -fn family_label(family: &crate::post_asap::SummaryFamilyType) -> String { - use crate::post_asap::SummaryFamilyType; - match family { - SummaryFamilyType::Plain(dtype) => format!("Plain({dtype:?})"), - SummaryFamilyType::ExactAggregate(kind, _) => format!("ExactAggregate({kind:?})"), - SummaryFamilyType::Sketch(kind, _grouping) => format!("Sketch({:?})", kind.algorithm()), - SummaryFamilyType::Sample(kind, _) => format!("Sample({kind:?})"), - SummaryFamilyType::Wavelet(kind, _) => format!("Wavelet({kind:?})"), - SummaryFamilyType::StatModel(kind, _) => format!("StatModel({kind:?})"), - } -} - -/// `(kind, label, detail)` for every [`SummaryExpr`] variant *except* -/// [`SummaryExpr::KeepPreAsap`] — that variant has no `SummaryDagNode`/ -/// `DagNode` of its own (see [`build_summary`]/[`build_summary_hybrid`], its -/// only two callers, both of which special-case it before ever reaching -/// this function). Factored out so [`build_summary`] (nests a `KeepPreAsap` -/// leaf's pre-ASAP subtree as its own [`SummaryDagGraph`]) and -/// [`build_summary_hybrid`] (splices that same subtree directly into a -/// shared [`DagGraph`] node list — see [`export_post_asap`]) can't drift -/// apart on how every *other* variant's own shape is described, since -/// nothing about that description differs between the two. -macro_rules! define_summary_kind_tags { - ($($pattern:pat => $tag:literal),+ $(,)?) => { - #[cfg(test)] - const SUMMARY_KIND_TAGS: &[&str] = &[$($tag),+]; - - fn summary_kind_tag(expr: &SummaryExpr) -> &'static str { - match expr { - SummaryExpr::KeepPreAsap(_) => unreachable!( - "summary_kind_tag's callers special-case KeepPreAsap" - ), - $($pattern => $tag),+ - } - } - }; -} - -define_summary_kind_tags! { - SummaryExpr::BinaryOp { .. } => "SummaryBinaryOp", - - SummaryExpr::ValueOperation { .. } => "ValueOperation", - SummaryExpr::RelationalJoin { .. } => "RelationalJoin", - SummaryExpr::SummaryAgg { .. } => "SummaryAgg", - SummaryExpr::SummaryJoin { .. } => "SummaryJoin", - SummaryExpr::SummarySubtract { .. } => "SummarySubtract", - SummaryExpr::SummaryDelete { .. } => "SummaryDelete", - SummaryExpr::SummaryEstimate { .. } => "SummaryEstimate", - SummaryExpr::SummaryMerge { .. } => "SummaryMerge", -} - -fn summary_shape(expr: &SummaryExpr) -> (&'static str, String, serde_json::Value) { - let kind = summary_kind_tag(expr); - match expr { - SummaryExpr::KeepPreAsap(_) => { - unreachable!("summary_shape's callers special-case KeepPreAsap before calling it") - } - SummaryExpr::BinaryOp { operator, .. } => { - let label = format!("BinaryOp({:?})", operator.kind); - let detail = serde_json::json!({ - "kind": format!("{:?}", operator.kind), - "vector_match": operator.vector_match, - }); - (kind, label, detail) - } - - SummaryExpr::ValueOperation { - operation, timing, .. - } => ( - kind, - format!("ValueOperation({operation:?})"), - serde_json::json!({ - "operation": format!("{operation:?}"), - "timing": timing.as_str(), - }), - ), - SummaryExpr::RelationalJoin { - kind: join_kind, - pred, - .. - } => ( - kind, - format!("RelationalJoin({join_kind:?})"), - serde_json::json!({ "join_kind": join_kind, "predicate": pred }), - ), - SummaryExpr::SummaryAgg { - family, - input, - reduction, - grouping, - .. - } => { - let label = format!("SummaryAgg({})", family_label(family)); - let detail = serde_json::json!({ - "family": format!("{family:?}"), - "input": input, - "reduction": reduction, - "grouping": format!("{grouping:?}"), - }); - (kind, label, detail) - } - SummaryExpr::SummaryJoin { key, family, .. } => { - let label = format!("SummaryJoin({})", family_label(family)); - let detail = serde_json::json!({ - "key": key, - "family": format!("{family:?}"), - }); - (kind, label, detail) - } - SummaryExpr::SummarySubtract { .. } => { - (kind, "SummarySubtract".into(), serde_json::json!({})) - } - SummaryExpr::SummaryDelete { key, .. } => { - let detail = serde_json::json!({ "key": key }); - (kind, "SummaryDelete".into(), detail) - } - SummaryExpr::SummaryEstimate { query, .. } => { - let label = format!("SummaryEstimate({query:?})"); - let detail = serde_json::json!({ "query": format!("{query:?}") }); - (kind, label, detail) - } - SummaryExpr::SummaryMerge { children, .. } => { - let label = format!("SummaryMerge({} children)", children.len()); - (kind, label, serde_json::json!({})) - } - } -} - -/// `expr`'s own `Rc` children, in the variant's field order -/// (e.g. `SummaryJoin` is `[outer, inner]`) — empty for -/// [`SummaryExpr::KeepPreAsap`], which has no `SummaryNode` children at all -/// (only a boxed pre-ASAP `QueryExpr`). Shared by [`build_summary`] and -/// [`build_summary_hybrid`] for the same reason [`summary_shape`] is. -fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { - match expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::BinaryOp { lhs, rhs, .. } => vec![lhs, rhs], - - SummaryExpr::ValueOperation { child, .. } => vec![child], - SummaryExpr::RelationalJoin { left, right, .. } => vec![left, right], - SummaryExpr::SummaryAgg { child, .. } => vec![child], - SummaryExpr::SummaryJoin { outer, inner, .. } => vec![outer, inner], - SummaryExpr::SummarySubtract { left, right } => vec![left, right], - SummaryExpr::SummaryDelete { summary_input, .. } => vec![summary_input], - SummaryExpr::SummaryEstimate { summary_input, .. } => vec![summary_input], - SummaryExpr::SummaryMerge { children, .. } => children.iter().collect(), - } -} - -/// Recursively flatten `node`, appending [`SummaryDagNode`]s to `nodes` in -/// post-order (children pushed before their parent), and return the pushed -/// root's id. Exhaustive over every [`SummaryExpr`] variant, matching this -/// file's own exhaustive style for `QueryExpr` in [`build`]. -fn build_summary(node: &SummaryNode, nodes: &mut Vec) -> u32 { - if let SummaryExpr::KeepPreAsap(inner) = &node.expr { - let pre_asap_subgraph = export(inner); - let inner_kind = pre_asap_subgraph.nodes[pre_asap_subgraph.root as usize].kind; - let label = format!("KeepPreAsap({inner_kind})"); - let detail = serde_json::json!({ "pre_asap_subgraph": pre_asap_subgraph }); - return push_summary_node( - nodes, - "KeepPreAsap", - label, - detail, - vec![], - node.guarantee.clone(), - ); - } - let children: Vec = summary_children(&node.expr) - .into_iter() - .map(|child| build_summary(child, nodes)) - .collect(); - let (kind, label, detail) = summary_shape(&node.expr); - push_summary_node(nodes, kind, label, detail, children, node.guarantee.clone()) -} - /// One replacement site a higher layer (the `dag_export` binary) found by /// running `asap_aware_mapping::replacement::search_workload_with` + /// `CandidateLogicalASAPDAGs::cost_sorted` and picking the best-ranked candidate for one @@ -626,7 +376,7 @@ pub struct TargetReplacement { pub decision_id: u32, /// Id of the [`DagNode`] (in this query's own `graph.nodes`, i.e. the /// [`NamedGraph`] this `TargetReplacement` is attached to) this - /// replacement's `before` subtree is rooted at. + /// replacement's `before` sub-DAG is rooted at. pub target_pre_id: u32, /// Human label for which strategy proposed the winning candidate — /// e.g. `"Sketch"` / `"HydraGrouping"` / `"SharedSubtree"` / @@ -648,7 +398,7 @@ pub struct TargetReplacement { /// doesn't estimate a numeric cost for this candidate shape (see that /// field's own doc upstream). pub cost: f64, - /// The target's own pre-ASAP subtree, before replacement — literally + /// The target's own sub-DAG, before replacement — literally /// `export(target)` for the `TargetSubDAGCandidates`'s own `target`, reused as-is. pub before: DagGraph, pub after: TargetReplacementAfter, @@ -668,8 +418,10 @@ pub struct TargetReplacement { } /// What a [`TargetReplacement`] became — either a genuine post-ASAP binding -/// or a still-pre-ASAP-shaped structural rewrite, mirroring -/// `asap_aware_mapping::replacement::Replacement`'s own two variants. +/// or a still-relational structural rewrite, mirroring +/// `asap_aware_mapping::replacement::Replacement`'s own two variants. Both +/// carry an ordinary [`DagGraph`]: the unified IR renders a summary sub-DAG +/// and a rewritten relational sub-DAG through the same [`export`]. /// /// Serializes as `{"kind": "Summary"|"Rewrite", "graph": {...}}` (serde's /// adjacently-tagged representation for a `#[serde(tag = "kind", content = @@ -680,73 +432,83 @@ pub struct TargetReplacement { #[serde(tag = "kind", content = "graph")] pub enum TargetReplacementAfter { /// A `Replacement::Summary` candidate — a genuine post-ASAP binding. - Summary(SummaryDagGraph), - /// A `Replacement::Rewrite` candidate — still pre-ASAP shaped (CSE + Summary(DagGraph), + /// A `Replacement::Rewrite` candidate — still relational (CSE /// share/recompute, `AvgToSumOverCountStrategy`, and `RollupStrategy` - /// all produce this kind), so this reuses [`DagGraph`]/[`export`] too, - /// not a new type. + /// all produce this kind). Rewrite(DagGraph), } -/// Flatten `expr` into a [`DagGraph`]. -pub fn export(expr: &QueryExpr) -> DagGraph { - let mut nodes = Vec::new(); - // One cache for the whole export — persisted across every `build`/ - // `push_node` call, not reset per node, so `structural_hash` memoizes - // real work across this pass instead of re-walking an already-hashed - // shared descendant once per node that references it. - let mut cache = HashCache::new(); - // No substitution: an ordinary pre-ASAP export never splices anything - // in — see `build`'s own doc for why it always takes a `find_winner` - // callback regardless (so `export_post_asap` can share this exact - // per-variant traversal instead of duplicating it). - let root = build(expr, &mut nodes, &mut cache, &mut |_| None); - DagGraph { - nodes, - root, - edge_annotations: Vec::new(), - } -} - -/// What a higher layer found for one specific pre-ASAP node when building a -/// merged post-ASAP graph via [`export_post_asap`] — see that function's own -/// doc for the full design. `asap_types` has no opinion on *how* this is +/// What a higher layer found for one specific node when building a merged +/// post-ASAP graph via [`export_post_asap`] — see that function's own doc +/// for the full design. `asap_types` has no opinion on *how* this is /// decided (that's `asap_aware_mapping::replacement::search_workload_with` + /// `CandidateLogicalASAPDAGs::cost_sorted`'s job, a higher layer, exactly the layering rule /// [`DagNode::notes`] already states); it only defines the shape a decision -/// comes back in. +/// comes back in. Both variants render identically (one IR, one builder); +/// they are kept apart so the caller's `Replacement` maps one-to-one. #[derive(Debug, Clone)] pub enum PostAsapSubstitution { /// This exact node has a winning `Replacement::Rewrite` — keep building - /// from `.0` instead of the original node. Still pre-ASAP shaped, so - /// [`build`] renders it via the same ordinary `DagNode` path — see - /// [`build`]'s own doc for why `.0`'s own top level is rendered without - /// re-querying `find_winner` on it (its descendants still are). + /// from `replacement` instead of the original node. Rewrite { - replacement: Rc, + replacement: Rc, decision: DagDecision, }, - /// This exact node has a winning `Replacement::Summary` — switch to - /// rendering `.0`'s bound `SummaryNode` shape from here down, via - /// [`build_summary_hybrid`]. + /// This exact node has a winning `Replacement::Summary` — keep building + /// from `replacement` (a summary-bound sub-DAG) instead of the original + /// node. Summary { - replacement: Rc, + replacement: Rc, decision: DagDecision, }, } +/// Flatten the DAG rooted at `root` into a [`DagGraph`]: one [`DagNode`] +/// per unique reachable node, children pushed before their parents. +pub fn export(root: &Rc) -> DagGraph { + let mut no_substitution = |_: &Rc| None; + let mut builder = Builder::new(&mut no_substitution); + let root = builder.build(root); + builder.finish(root) +} + +/// Flatten the DAG rooted at `node` into a [`SummaryDagGraph`] — the same +/// nodes [`export`] produces, in the [`SummaryDagNode`] shape (snake_case +/// `kind`, `guarantee` as its own field). +pub fn export_summary(node: &Rc) -> SummaryDagGraph { + let graph = export(node); + let nodes = graph + .nodes + .into_iter() + .map(|node| { + let source = node + .source_node + .expect("every exported node carries its source"); + SummaryDagNode { + id: node.id, + kind: snake_case_kind(&source.operator), + label: node.label, + detail: node.detail, + schema: node.schema, + children: node.children, + guarantee: source.guarantee.clone(), + source_node: Some(source), + } + }) + .collect(); + SummaryDagGraph { + nodes, + root: graph.root, + } +} + /// Build one merged "whole query, but post-ASAP" [`DagGraph`] by walking -/// `root`'s ordinary pre-ASAP shape and, at every node, asking `find_winner` -/// whether *that exact node* has a winning replacement — if so, splicing -/// the replacement's own shape in at that position instead, in the very -/// same flattened node list (not a nested sub-graph the way -/// [`TargetReplacement::before`]/`::after` — small, independent, per-site -/// before/after pairs — already do; see this file's "Post-ASAP replacement -/// export" section doc for why *that* design doesn't attempt a single -/// whole-query composite, and why this one can: this is a synthetic -/// id/edge list, the same kind of thing [`DagGraph`] already is for the -/// pre-ASAP side, not a real `QueryExpr`/`SummaryNode` value with a type -/// system to satisfy). +/// `root` and, at every node, asking `find_winner` whether *that exact +/// node* has a winning replacement — if so, splicing the replacement's own +/// sub-DAG in at that position instead, in the very same flattened node +/// list (not a nested sub-graph the way [`TargetReplacement::before`]/ +/// `::after` — small, independent, per-site before/after pairs — do). /// /// `find_winner` is the whole layering seam: `asap_types` never runs /// `asap_aware_mapping::replacement::search_workload_with` or @@ -758,11 +520,11 @@ pub enum PostAsapSubstitution { /// for [`TargetReplacement`] discovery, and passes it in here unchanged. /// /// `find_winner` is deliberately consulted only once per node, at the -/// moment [`build`] first reaches it — **not** re-consulted on a +/// moment the builder first reaches it — **not** re-consulted on a /// substitution's own immediate top level (only on that substitution's /// *descendants*, which get an ordinary fresh call same as any other node). /// This matters for correctness, not just efficiency: -/// `SharedSubtreeStrategy`'s own "build once and share" candidate is +/// `SharedSubDagStrategy`'s own "build once and share" candidate is /// `Replacement::Rewrite(Rc::clone(target))` — literally the *same* value /// as the target it's a candidate for. Re-querying `find_winner` on that /// candidate's own top level would find the identical group and its @@ -770,225 +532,189 @@ pub enum PostAsapSubstitution { /// re-query at exactly that one level is what makes this termination-safe /// for every registered strategy, not just the ones that happen not to /// return the target itself as a candidate. +/// +/// Every node a substitution introduced carries the substitution's +/// [`DagDecision`] (`role = "replacement_root"` on the spliced-in root, +/// `"replacement_region"` on its newly exported descendants); a descendant +/// that was already exported before the splice (a shared input the +/// replacement reuses) keeps whatever it already had. pub fn export_post_asap( - root: &QueryExpr, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, + root: &Rc, + find_winner: &mut dyn FnMut(&Rc) -> Option, ) -> DagGraph { - let mut nodes = Vec::new(); - let mut cache = HashCache::new(); - let root_id = build(root, &mut nodes, &mut cache, find_winner); - deduplicate_pointer_shared_nodes(nodes, root_id) + let mut builder = Builder::new(find_winner); + let root = builder.build(root); + builder.finish(root) } -fn deduplicate_pointer_shared_nodes(nodes: Vec, root: u32) -> DagGraph { - let mut by_source_ptr = HashMap::::new(); - let mut old_to_new = vec![0_u32; nodes.len()]; - let mut deduplicated = Vec::with_capacity(nodes.len()); - for mut node in nodes { - node.children = node - .children - .into_iter() - .map(|child| old_to_new[child as usize]) - .collect(); - if let Some(existing) = node - .source_ptr - .and_then(|source_ptr| by_source_ptr.get(&source_ptr).copied()) - { - old_to_new[node.id as usize] = existing; - continue; - } - let old_id = node.id; - let new_id = deduplicated.len() as u32; - node.id = new_id; - if let Some(source_ptr) = node.source_ptr { - by_source_ptr.insert(source_ptr, new_id); +/// The one flattening pass behind every entry point. Nodes are memoized by +/// `Rc` pointer identity: a node reached from several parents (an operator +/// input shared with a scalar reference, say) is exported once. +struct Builder<'a> { + nodes: Vec, + /// `Rc::as_ptr` of every node already exported (or substituted) → its id. + ids: HashMap<*const OperatorNode, u32>, + /// One cache for the whole export — persisted across every node, not + /// reset per node, so `structural_hash` memoizes real work across this + /// pass instead of re-walking an already-hashed shared descendant once + /// per node that references it. + cache: HashCache, + find_winner: &'a mut dyn FnMut(&Rc) -> Option, +} + +impl<'a> Builder<'a> { + fn new( + find_winner: &'a mut dyn FnMut(&Rc) -> Option, + ) -> Self { + Self { + nodes: Vec::new(), + ids: HashMap::new(), + cache: HashCache::new(), + find_winner, } - old_to_new[old_id as usize] = new_id; - deduplicated.push(node); } - DagGraph { - nodes: deduplicated, - root: old_to_new[root as usize], - edge_annotations: Vec::new(), + fn finish(self, root: u32) -> DagGraph { + DagGraph { + nodes: self.nodes, + root, + edge_annotations: Vec::new(), + } } -} -macro_rules! define_query_kind_tags { - ($($pattern:pat => $tag:literal),+ $(,)?) => { - #[cfg(test)] - const QUERY_KIND_TAGS: &[&str] = &[$($tag),+]; - - fn kind_tag(expr: &QueryExpr) -> &'static str { - match expr { - $($pattern => $tag),+, - other @ (QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. }) => unreachable!( - "kind_tag reached a scalar QueryExpr variant directly: {other:?}" - ), + /// Export `node` (or, when `find_winner` has a substitution for it, the + /// substitution's sub-DAG in its place) and return its id. + fn build(&mut self, node: &Rc) -> u32 { + let ptr = Rc::as_ptr(node); + if let Some(&id) = self.ids.get(&ptr) { + return id; + } + let (replacement, decision) = match (self.find_winner)(node) { + None => return self.build_node(node), + Some(PostAsapSubstitution::Rewrite { + replacement, + decision, + }) + | Some(PostAsapSubstitution::Summary { + replacement, + decision, + }) => (replacement, decision), + }; + let first = self.nodes.len(); + let root = self.build_node(&replacement); + for exported in &mut self.nodes[first..] { + if exported.decision.is_none() { + let mut node_decision = decision.clone(); + node_decision.role = if exported.id == root { + "replacement_root" + } else { + "replacement_region" + }; + exported.decision = Some(node_decision); } } - }; -} - -define_query_kind_tags! { - QueryExpr::Scan { .. } => "Scan", - QueryExpr::PromqlScalarBridge(_) => "PromqlScalarBridge", - QueryExpr::EvalTimestamp => "EvalTimestamp", - QueryExpr::CurrentTimestamp => "CurrentTimestamp", - QueryExpr::PromqlVectorFromScalar(_) => "PromqlVectorFromScalar", - QueryExpr::PromqlScalarFromVector(_) => "PromqlScalarFromVector", - QueryExpr::PromqlRelabel { .. } => "PromqlRelabel", - QueryExpr::PromqlInfoEnrich { .. } => "PromqlInfoEnrich", - QueryExpr::PromqlSeriesSample { .. } => "PromqlSeriesSample", - QueryExpr::Filter { .. } => "Filter", - QueryExpr::Project { .. } => "Project", - QueryExpr::Aggregate { .. } => "Aggregate", - QueryExpr::Dedup { .. } => "Dedup", - QueryExpr::Concat { .. } => "Concat", - QueryExpr::Join { .. } => "Join", - QueryExpr::SetOp { .. } => "SetOp", - QueryExpr::Sort { .. } => "Sort", - QueryExpr::Limit { .. } => "Limit", - QueryExpr::PromqlSubquery { .. } => "PromqlSubquery", - QueryExpr::TimeRange { .. } => "TimeRange", - QueryExpr::TimeShift { .. } => "TimeShift", - QueryExpr::SQLWindowFunc { .. } => "SQLWindowFunc", - QueryExpr::BinaryOp { .. } => "BinaryOp", -} - -/// Push one flattened node for `expr`. `expr` is the *whole* subtree this -/// node represents (not just its own fields) — `hash` is -/// [`structural_hash(expr)`](structural_hash), the identical function and -/// the identical input `InternTable::intern` would hash for this same -/// subtree, so this node's `hash` matches what `cse::share_common_subtrees` -/// would bucket it under. `kind` is [`kind_tag(expr)`](kind_tag), not a -/// caller-supplied argument — see that function's doc for why. -fn push_node( - nodes: &mut Vec, - expr: &QueryExpr, - cache: &mut HashCache, - label: String, - detail: serde_json::Value, - children: Vec, -) -> u32 { - let id = nodes.len() as u32; - let hash = Some(structural_hash(expr, cache)); - nodes.push(DagNode { - id, - kind: kind_tag(expr), - label, - detail, - schema: expr - .output_schema() - .ok() - .and_then(|schema| serde_json::to_value(schema).ok()), - children, - workload_node_id: None, - hash, - source_expr: Some(expr.clone()), - source_ptr: Some(expr as *const QueryExpr as usize), - notes: Vec::new(), - decision: None, - }); - id -} + // The original node now resolves to the substitution: another + // parent of the same `Rc` reuses the spliced-in sub-DAG. + self.ids.insert(ptr, root); + root + } -/// Push one flattened node with no corresponding pre-ASAP `QueryExpr` at -/// all — a post-ASAP-originated node inside [`export_post_asap`]'s merged -/// graph (a `SummaryAgg`/`SummaryJoin`/… node, via [`build_summary_hybrid`]). -/// `hash`/`source_expr`-based re-identification (see [`DagNode::hash`]'s own -/// doc) has no meaning for a node with no `QueryExpr` behind it, so this -/// pushes a fixed placeholder hash (`0`) and `source_expr: None` rather than -/// inventing a hash over `SummaryExpr` (which, unlike `QueryExpr`, has no -/// [`structural_hash`]-equivalent function at all — see [`SummaryDagNode`]'s -/// own doc on why `SummaryExpr`'s fields don't even derive `Hash`/`PartialEq` -/// consistently enough to build one). -fn push_summary_originated_node( - nodes: &mut Vec, - kind: &'static str, - label: String, - detail: serde_json::Value, - children: Vec, -) -> u32 { - let id = nodes.len() as u32; - nodes.push(DagNode { - id, - kind, - label, - detail, - schema: None, - children, - workload_node_id: None, - hash: None, - source_expr: None, - source_ptr: None, - notes: Vec::new(), - decision: None, - }); - id + /// Export `node` itself (no substitution check at this level; children + /// still go through [`Self::build`]) and return its id. + fn build_node(&mut self, node: &Rc) -> u32 { + let ptr = Rc::as_ptr(node); + if let Some(&id) = self.ids.get(&ptr) { + return id; + } + let children: Vec = node.children().into_iter().map(|c| self.build(c)).collect(); + let (label, mut detail) = shape(node, &self.ids); + if let serde_json::Value::Object(map) = &mut detail { + if let Some(timing) = node.timing { + map.insert("timing".into(), serde_json::json!(timing.as_str())); + } + if let Some(guarantee) = &node.guarantee { + if let Ok(value) = serde_json::to_value(guarantee) { + map.insert("guarantee".into(), value); + } + } + } + let hash = structural_hash(node, &mut self.cache); + self.cache.insert(ptr, hash); + let id = self.nodes.len() as u32; + self.nodes.push(DagNode { + id, + kind: node.operator.kind_name(), + label, + detail, + schema: serde_json::to_value(&node.schema).ok(), + children, + workload_node_id: None, + hash: Some(hash), + source_node: Some(Rc::clone(node)), + source_ptr: Some(ptr as usize), + notes: Vec::new(), + decision: None, + }); + self.ids.insert(ptr, id); + id + } } -/// The [`build_summary`]/[`build_summary_hybrid`] counterpart of [`build`] -/// for a bound [`SummaryNode`] reached while building -/// [`export_post_asap`]'s merged graph: appends into the *same* `nodes: -/// Vec` list `build` itself is filling, instead of a separate -/// [`SummaryDagGraph`]. A `KeepPreAsap(inner)` leaf recurses back into -/// [`build`] on `inner` (the general pre-ASAP entry, `find_winner` included) -/// rather than nesting a `{"pre_asap_subgraph": ...}` blob the way -/// [`build_summary`] does — so the merged graph reads as one seamless graph -/// with no dead ends, and so a target reachable underneath a `KeepPreAsap` -/// wrapper (a nested aggregate a strategy independently found a -/// replacement for, say) still gets spliced in correctly. -fn build_summary_hybrid( - node: &SummaryNode, - nodes: &mut Vec, - cache: &mut HashCache, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, -) -> u32 { - if let SummaryExpr::KeepPreAsap(inner) = &node.expr { - return build(inner, nodes, cache, find_winner); +/// [`Operator::kind_name`] in snake_case, for [`SummaryDagNode::kind`]. +/// Exhaustive so a new operator variant fails to compile here until it is +/// named. +fn snake_case_kind(operator: &Operator) -> &'static str { + match operator { + Operator::NonASAP(op) => match op { + NonASAPOp::Scan { .. } => "scan", + NonASAPOp::Values { .. } => "values", + NonASAPOp::Filter { .. } => "filter", + NonASAPOp::Project { .. } => "project", + NonASAPOp::Aggregate { .. } => "aggregate", + NonASAPOp::Join { .. } => "join", + NonASAPOp::SetOp { .. } => "set_op", + NonASAPOp::Concat { .. } => "concat", + NonASAPOp::Dedup { .. } => "dedup", + NonASAPOp::Sort { .. } => "sort", + NonASAPOp::Limit { .. } => "limit", + NonASAPOp::BinaryOp { .. } => "binary_op", + NonASAPOp::SQLWindowFunc { .. } => "sql_window_func", + NonASAPOp::TimeRange { .. } => "time_range", + NonASAPOp::TimeShift { .. } => "time_shift", + NonASAPOp::PromqlVectorFromScalar(_) => "promql_vector_from_scalar", + NonASAPOp::PromqlRelabel { .. } => "promql_relabel", + NonASAPOp::PromqlInfoEnrich { .. } => "promql_info_enrich", + NonASAPOp::PromqlSeriesSample { .. } => "promql_series_sample", + NonASAPOp::PromqlSubquery { .. } => "promql_subquery", + }, + Operator::ASAP(op) => match op { + ASAPOp::SummaryAgg { .. } => "summary_agg", + ASAPOp::SummaryEstimate { .. } => "summary_estimate", + ASAPOp::FinalizeExactAccumulator { .. } => "finalize_exact_accumulator", + ASAPOp::MaintainPopulation { .. } => "maintain_population", + ASAPOp::EvaluatePopulation { .. } => "read_population", + ASAPOp::SummaryMerge { .. } => "summary_merge", + ASAPOp::SummarySubtract { .. } => "summary_subtract", + ASAPOp::SummaryDelete { .. } => "summary_delete", + ASAPOp::SummaryJoin { .. } => "summary_join", + ASAPOp::Extension { .. } => "extension", + }, } - let children: Vec = summary_children(&node.expr) - .into_iter() - .map(|child| build_summary_hybrid(child, nodes, cache, find_winner)) - .collect(); - let (kind, label, mut detail) = summary_shape(&node.expr); - // The merged graph's `DagNode` has no dedicated guarantee field (it is - // the pre-ASAP node shape); the guarantee rides in `detail` under the - // same key/shape `SummaryDagNode::guarantee` uses, additively. - if let Some(guarantee) = &node.guarantee { - if let (serde_json::Value::Object(map), Ok(value)) = - (&mut detail, serde_json::to_value(guarantee)) - { - map.insert("guarantee".into(), value); - } - } - let id = push_summary_originated_node(nodes, kind, label, detail, children); - nodes[id as usize].schema = Some(summary_schema_json(&node.schema)); - id } -fn summary_schema_json(schema: &crate::post_asap::SummarySchema) -> serde_json::Value { - serde_json::json!({ - "fields": schema.fields.iter().map(|field| serde_json::json!({ - "name": field.name, - "dtype": format!("{:?}", field.dtype), - "nullable": field.nullable, - })).collect::>(), - "time_index": schema.time_index, - }) +/// A short, human-readable label for a [`FieldDataType`] (e.g. +/// `"Sketch(Kll)"`, `"ExactAggregate(Sum)"`) — for the label text on a +/// `SummaryAgg`/`SummaryJoin` node. Every variant is covered, via `Debug` +/// for the inner kind rather than hand-written prose per algorithm. +fn family_label(family: &FieldDataType) -> String { + match family { + FieldDataType::Plain(dtype) => format!("Plain({dtype:?})"), + FieldDataType::ExactAggregate(kind, _) => format!("ExactAggregate({kind:?})"), + FieldDataType::Sketch(kind, _grouping) => format!("Sketch({:?})", kind.algorithm()), + FieldDataType::Sample(kind, _) => format!("Sample({kind:?})"), + FieldDataType::Wavelet(kind, _) => format!("Wavelet({kind:?})"), + FieldDataType::StatModel(kind, _) => format!("StatModel({kind:?})"), + } } fn source_label(source: &Source) -> String { @@ -998,407 +724,336 @@ fn source_label(source: &Source) -> String { } } -/// Recursively flatten `expr`, appending nodes to `nodes` in post-order -/// (children pushed before their parent), and return the id of the pushed -/// root node. Exhaustive over every **operator** `QueryExpr` variant — a new -/// one fails to compile here until this match is extended, matching the rest -/// of the IR's exhaustive-match style (e.g. `output_schema`). The scalar -/// variants (issue #205) are never passed to `build` directly: every operator -/// arm that carries one (`Filter.pred`, `Project.cols`, `Aggregate.having`, …) -/// serializes it as opaque `detail` JSON via `Predicate`/`ProjectItem`/ -/// `AggIntent`'s own `Serialize` impl, same as before the merge — a scalar -/// subtree was never a separate DAG node, so this doesn't change that. -/// -/// `find_winner` is [`export_post_asap`]'s substitution seam, threaded -/// through every recursive call (including [`export`]'s own, which always -/// passes a closure that returns `None`) so both entry points share this -/// exact traversal instead of maintaining two copies of it. `build` itself -/// only ever calls `find_winner` once, right here at the top, before -/// dispatching into the ordinary per-variant match below — see -/// [`export_post_asap`]'s own doc for why a substitution's own immediate -/// result is rendered via that match directly (recursing into its children -/// through `build` again, so *they* still get a fresh `find_winner` call) -/// rather than by looping back through this check a second time. -fn build( - expr: &QueryExpr, - nodes: &mut Vec, - cache: &mut HashCache, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, -) -> u32 { - match find_winner(expr) { - Some(PostAsapSubstitution::Rewrite { - replacement, - decision, - }) => { - let first = nodes.len(); - let root = build_no_recheck(&replacement, nodes, cache, find_winner); - for node in &mut nodes[first..] { - if node.decision.is_none() { - let mut node_decision = decision.clone(); - node_decision.role = if node.id == root { - "replacement_root" - } else { - "replacement_region" - }; - node.decision = Some(node_decision); - } +/// `(label, detail)` for one node: its own fields, never its children. +/// Exhaustive over every operator variant — a new one fails to compile +/// here until this match is extended, matching the rest of the IR's +/// exhaustive-match style. Scalar expressions are rendered through +/// [`scalar_json`] with `ids` resolving their operator references. +fn shape( + node: &OperatorNode, + ids: &HashMap<*const OperatorNode, u32>, +) -> (String, serde_json::Value) { + let scalar = |expr: &ScalarExpr| scalar_json(expr, ids); + let scalars = + |exprs: &[ScalarExpr]| -> Vec { exprs.iter().map(scalar).collect() }; + let predicate = |pred: &crate::ir::Predicate| scalar(&pred.0); + let sort_keys = |keys: &[crate::ir::SortKey]| -> Vec { + keys.iter() + .map(|key| { + serde_json::json!({ + "expr": scalar(&key.expr), + "ascending": key.ascending, + "nulls_first": key.nulls_first, + }) + }) + .collect() + }; + match &node.operator { + Operator::NonASAP(op) => match op { + NonASAPOp::Scan { + source, + predicates, + schema, + } => ( + format!("Scan({})", source_label(source)), + serde_json::json!({ + "source": source, + "predicates": predicates.iter().map(predicate).collect::>(), + "schema": schema, + }), + ), + NonASAPOp::Values { rows, schema } => ( + format!("Values({} rows)", rows.len()), + serde_json::json!({ + "rows": rows.iter().map(|row| scalars(row)).collect::>(), + "schema": schema, + }), + ), + NonASAPOp::Filter { pred, .. } => ( + "Filter".into(), + serde_json::json!({ "pred": predicate(pred) }), + ), + NonASAPOp::Project { + cols, qualifier, .. + } => ( + format!("Project({} cols)", cols.len()), + serde_json::json!({ + "cols": cols.iter().map(|item| serde_json::json!({ + "alias": item.alias, + "expr": scalar(&item.expr), + })).collect::>(), + "qualifier": qualifier, + }), + ), + NonASAPOp::Aggregate { + reduction, + measures, + output_names, + having, + .. + } => ( + format!("Aggregate({} measures)", measures.len()), + serde_json::json!({ + "reduction": reduction, + "measures": measures, + "output_names": output_names, + "having": having.as_ref().map(predicate), + }), + ), + NonASAPOp::Join { kind, pred, .. } => ( + format!("Join({kind:?})"), + serde_json::json!({ "kind": kind, "pred": predicate(pred) }), + ), + NonASAPOp::SetOp { kind, all, .. } => ( + format!("SetOp({kind:?})"), + serde_json::json!({ "kind": kind, "all": all }), + ), + NonASAPOp::Concat { + children, + discriminator_unique_key, + } => ( + format!("Concat({} branches)", children.len()), + serde_json::json!({ "discriminator_unique_key": discriminator_unique_key }), + ), + NonASAPOp::Dedup { cols, .. } => ( + format!("Dedup({} cols)", cols.len()), + serde_json::json!({ "cols": cols }), + ), + NonASAPOp::Sort { + keys, partition_by, .. + } => ( + format!("Sort({} keys)", keys.len()), + serde_json::json!({ "keys": sort_keys(keys), "partition_by": partition_by }), + ), + NonASAPOp::Limit { + n, + offset, + partition_by, + .. + } => ( + match n { + Some(n) => format!("Limit({n})"), + None => format!("Limit(offset {offset})"), + }, + serde_json::json!({ "n": n, "offset": offset, "partition_by": partition_by }), + ), + NonASAPOp::BinaryOp { + operator, + return_bool, + .. + } => ( + format!("BinaryOp({})", operator.kind), + serde_json::json!({ + "op": operator.kind.to_string(), + "vector_match": operator.vector_match, + "checked_relative_division": operator.checked_relative_division, + "checked_finite_division": operator.checked_finite_division, + "return_bool": return_bool, + }), + ), + NonASAPOp::SQLWindowFunc { + func, + args, + partition_by, + order_by, + frame, + output_name, + .. + } => ( + format!("SQLWindowFunc({func:?})"), + serde_json::json!({ + "func": func, + "args": scalars(args), + "partition_by": partition_by, + "order_by": sort_keys(order_by), + "frame": frame, + "output_name": output_name, + }), + ), + NonASAPOp::TimeRange { range, kind, .. } => ( + format!("TimeRange({kind:?}, {range:?})"), + serde_json::json!({ "range": range, "kind": kind }), + ), + NonASAPOp::TimeShift { shift, .. } => { + ("TimeShift".into(), serde_json::json!({ "shift": shift })) } - return root; - } - Some(PostAsapSubstitution::Summary { - replacement, - decision, - }) => { - let first = nodes.len(); - let root = build_summary_hybrid(&replacement, nodes, cache, find_winner); - for node in &mut nodes[first..] { - if node.decision.is_none() { - let mut node_decision = decision.clone(); - node_decision.role = if node.id == root { - "replacement_root" - } else { - "replacement_region" - }; - node.decision = Some(node_decision); - } + NonASAPOp::PromqlVectorFromScalar(value) => ( + "vector()".into(), + serde_json::json!({ "value": scalar(value) }), + ), + NonASAPOp::PromqlRelabel { dst, value, .. } => ( + format!("PromqlRelabel(dst={dst})"), + serde_json::json!({ "dst": dst, "value": scalar(value) }), + ), + NonASAPOp::PromqlInfoEnrich { selector, .. } => ( + "PromqlInfoEnrich".into(), + serde_json::json!({ "selector": selector }), + ), + NonASAPOp::PromqlSeriesSample { by, kind, .. } => ( + format!("PromqlSeriesSample({kind:?})"), + serde_json::json!({ "by": by, "kind": kind }), + ), + NonASAPOp::PromqlSubquery { + range, resolution, .. + } => ( + "PromqlSubquery".into(), + serde_json::json!({ "range": range, "resolution": resolution }), + ), + }, + Operator::ASAP(op) => match op { + ASAPOp::SummaryAgg { + family, + input, + reduction, + grouping, + .. + } => ( + format!("SummaryAgg({})", family_label(family)), + serde_json::json!({ + "family": format!("{family:?}"), + "input": input, + "reduction": reduction, + "grouping": format!("{grouping:?}"), + }), + ), + ASAPOp::SummaryEstimate { query, .. } => ( + format!("SummaryEstimate({query:?})"), + serde_json::json!({ "query": format!("{query:?}") }), + ), + ASAPOp::FinalizeExactAccumulator { .. } => { + ("FinalizeExactAccumulator".into(), serde_json::json!({})) } - return root; - } - None => {} + ASAPOp::MaintainPopulation { population, .. } => ( + format!("MaintainPopulation(max_k={})", population.max_k), + serde_json::json!({ "population": population }), + ), + ASAPOp::EvaluatePopulation { evaluation, .. } => ( + format!("EvaluatePopulation({evaluation:?})"), + serde_json::json!({ "evaluation": evaluation }), + ), + ASAPOp::SummaryMerge { children } => ( + format!("SummaryMerge({} children)", children.len()), + serde_json::json!({}), + ), + ASAPOp::SummarySubtract { .. } => ("SummarySubtract".into(), serde_json::json!({})), + ASAPOp::SummaryDelete { key, .. } => { + ("SummaryDelete".into(), serde_json::json!({ "key": key })) + } + ASAPOp::SummaryJoin { key, family, .. } => ( + format!("SummaryJoin({})", family_label(family)), + serde_json::json!({ "key": key, "family": format!("{family:?}") }), + ), + ASAPOp::Extension { name, .. } => ( + format!("Extension({name})"), + serde_json::json!({ "name": name }), + ), + }, } - build_no_recheck(expr, nodes, cache, find_winner) } -/// The actual per-variant match [`build`] dispatches to once it has decided -/// (by consulting `find_winner` exactly once) which `QueryExpr` value to -/// render at this position — either `expr` itself (unchanged), or a winning -/// `Replacement::Rewrite`'s own target. Every recursive call here goes back -/// through [`build`] (not this function), so every child gets its own fresh -/// `find_winner` query. -fn build_no_recheck( - expr: &QueryExpr, - nodes: &mut Vec, - cache: &mut HashCache, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, -) -> u32 { +/// `{"scalar_ref": }` for an operator node a scalar expression reads. +/// The node is one of the owning operator's children, so it has already +/// been exported by the time its parent's `detail` is built. +fn scalar_ref( + node: &Rc, + ids: &HashMap<*const OperatorNode, u32>, +) -> serde_json::Value { + serde_json::json!({ "scalar_ref": ids.get(&Rc::as_ptr(node)).copied() }) +} + +/// `expr` as JSON in `ScalarExpr`'s own serde shape (externally tagged +/// variants), except that every operator reference is rendered via +/// [`scalar_ref`] instead of inlining the referenced sub-DAG. Exhaustive so +/// a new variant fails to compile here until it is rendered. +fn scalar_json(expr: &ScalarExpr, ids: &HashMap<*const OperatorNode, u32>) -> serde_json::Value { + let sub = |e: &ScalarExpr| scalar_json(e, ids); + let list = |es: &[ScalarExpr]| -> Vec { es.iter().map(sub).collect() }; match expr { - QueryExpr::Scan { - source, - predicates, - schema, - } => { - let label = format!("Scan({})", source_label(source)); - let detail = serde_json::json!({ - "source": source, - "predicates": predicates, - "schema": schema, - }); - push_node(nodes, expr, cache, label, detail, vec![]) - } - // The bridged child is a scalar-sub-language node (issue #220), not - // an operator node `build` can recurse into — serialize it as opaque - // `detail` JSON, same as every other scalar-typed field - // (`Filter.pred`, `Project.cols`, …) rather than pushing it as a - // separate DAG node. - QueryExpr::PromqlScalarBridge(inner) => { - let detail = serde_json::json!({ "value": inner }); - push_node( - nodes, - expr, - cache, - format!("PromqlScalarBridge({inner:?})"), - detail, - vec![], - ) - } - QueryExpr::EvalTimestamp => push_node( - nodes, - expr, - cache, - "EvalTimestamp".into(), - serde_json::json!({}), - vec![], - ), - QueryExpr::CurrentTimestamp => push_node( - nodes, - expr, - cache, - "CurrentTimestamp".into(), - serde_json::json!({}), - vec![], - ), - QueryExpr::PromqlVectorFromScalar(child) => { - let c = build(child, nodes, cache, find_winner); - push_node( - nodes, - expr, - cache, - "vector()".into(), - serde_json::json!({}), - vec![c], - ) - } - QueryExpr::PromqlScalarFromVector(child) => { - let c = build(child, nodes, cache, find_winner); - push_node( - nodes, - expr, - cache, - "scalar()".into(), - serde_json::json!({}), - vec![c], - ) - } - QueryExpr::PromqlRelabel { dst, value, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "dst": dst, "value": value }); - push_node( - nodes, - expr, - cache, - format!("PromqlRelabel(dst={dst})"), - detail, - vec![c], - ) - } - QueryExpr::PromqlInfoEnrich { selector, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "selector": selector }); - push_node( - nodes, - expr, - cache, - "PromqlInfoEnrich".into(), - detail, - vec![c], - ) - } - QueryExpr::PromqlSeriesSample { by, kind, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "by": by, "kind": kind }); - push_node( - nodes, - expr, - cache, - format!("PromqlSeriesSample({kind:?})"), - detail, - vec![c], - ) - } - QueryExpr::Filter { pred, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "pred": pred }); - push_node(nodes, expr, cache, "Filter".into(), detail, vec![c]) - } - QueryExpr::Project { - cols, - qualifier, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "cols": cols, "qualifier": qualifier }); - push_node( - nodes, - expr, - cache, - format!("Project({} cols)", cols.len()), - detail, - vec![c], - ) - } - QueryExpr::Aggregate { - reduction, - measures, - output_names, - filters, - having, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ - "reduction": reduction, - "measures": measures, - "output_names": output_names, - "filters": filters, - "having": having, - }); - push_node( - nodes, - expr, - cache, - format!("Aggregate({} measures)", measures.len()), - detail, - vec![c], - ) - } - QueryExpr::Dedup { cols, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "cols": cols }); - push_node( - nodes, - expr, - cache, - format!("Dedup({} cols)", cols.len()), - detail, - vec![c], - ) - } - QueryExpr::Concat { - children, - discriminator_unique_key, - } => { - let ids: Vec = children - .iter() - .map(|c| build(c, nodes, cache, find_winner)) - .collect(); - let label = format!("Concat({} branches)", ids.len()); - let detail = - serde_json::json!({ "discriminator_unique_key": discriminator_unique_key }); - push_node(nodes, expr, cache, label, detail, ids) - } - QueryExpr::Join { - kind, - pred, + ScalarExpr::Column(id) => serde_json::json!({ "Column": id }), + ScalarExpr::Literal(value) => serde_json::json!({ "Literal": value }), + ScalarExpr::Negative { expr, semantics } => serde_json::json!({ + "Negative": { "expr": sub(expr), "semantics": semantics } + }), + ScalarExpr::Compare { left, + op, right, - } => { - let l = build(left, nodes, cache, find_winner); - let r = build(right, nodes, cache, find_winner); - let detail = serde_json::json!({ "kind": kind, "pred": pred }); - push_node( - nodes, - expr, - cache, - format!("Join({kind:?})"), - detail, - vec![l, r], - ) - } - QueryExpr::SetOp { - kind, - all, + semantics, + } => serde_json::json!({ + "Compare": { + "left": sub(left), + "op": op, + "right": sub(right), + "semantics": semantics, + } + }), + ScalarExpr::BoolAnd(parts) => serde_json::json!({ "BoolAnd": list(parts) }), + ScalarExpr::BoolOr(parts) => serde_json::json!({ "BoolOr": list(parts) }), + ScalarExpr::Not(e) => serde_json::json!({ "Not": sub(e) }), + ScalarExpr::IsNull(e) => serde_json::json!({ "IsNull": sub(e) }), + ScalarExpr::IsNotNull(e) => serde_json::json!({ "IsNotNull": sub(e) }), + ScalarExpr::Cast { expr, to, try_cast } => serde_json::json!({ + "Cast": { "expr": sub(expr), "to": to, "try_cast": try_cast } + }), + ScalarExpr::InList { + expr, + list: items, + negated, + } => serde_json::json!({ + "InList": { "expr": sub(expr), "list": list(items), "negated": negated } + }), + ScalarExpr::FunctionCall { name, args } => serde_json::json!({ + "FunctionCall": { "name": name, "args": list(args) } + }), + ScalarExpr::Arithmetic { + op, left, right, - } => { - let l = build(left, nodes, cache, find_winner); - let r = build(right, nodes, cache, find_winner); - let detail = serde_json::json!({ "kind": kind, "all": all }); - push_node( - nodes, - expr, - cache, - format!("SetOp({kind:?})"), - detail, - vec![l, r], - ) - } - QueryExpr::Sort { - keys, - partition_by, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "keys": keys, "partition_by": partition_by }); - push_node( - nodes, - expr, - cache, - format!("Sort({} keys)", keys.len()), - detail, - vec![c], - ) - } - QueryExpr::Limit { n, offset, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "n": n, "offset": offset }); - push_node(nodes, expr, cache, format!("Limit({n})"), detail, vec![c]) - } - QueryExpr::PromqlSubquery { - range, - resolution, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "range": range, "resolution": resolution }); - push_node(nodes, expr, cache, "PromqlSubquery".into(), detail, vec![c]) - } - QueryExpr::TimeRange { range, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "range": range }); - push_node( - nodes, - expr, - cache, - format!("TimeRange({range:?})"), - detail, - vec![c], - ) - } - QueryExpr::TimeShift { shift, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "shift": shift }); - push_node(nodes, expr, cache, "TimeShift".into(), detail, vec![c]) - } - QueryExpr::SQLWindowFunc { - func, - args, - partition_by, - order_by, - frame, - output_name, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ - "func": func, - "args": args, - "partition_by": partition_by, - "order_by": order_by, - "frame": frame, - "output_name": output_name, - }); - push_node( - nodes, - expr, - cache, - format!("SQLWindowFunc({func:?})"), - detail, - vec![c], - ) - } - QueryExpr::BinaryOp { - op, - lhs, - rhs, - vector_match, - } => { - let l = build(lhs, nodes, cache, find_winner); - let r = build(rhs, nodes, cache, find_winner); - let detail = serde_json::json!({ "op": op.to_string(), "vector_match": vector_match }); - push_node( - nodes, - expr, - cache, - format!("BinaryOp({op})"), - detail, - vec![l, r], - ) - } - other @ (QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. }) => { - unreachable!("dag_export::build reached a scalar QueryExpr variant directly: {other:?}") - } + semantics, + } => serde_json::json!({ + "Arithmetic": { + "op": op, + "left": sub(left), + "right": sub(right), + "semantics": semantics, + } + }), + ScalarExpr::Case { + operand, + branches, + else_expr, + } => serde_json::json!({ + "Case": { + "operand": operand.as_deref().map(sub), + "branches": branches + .iter() + .map(|(when, then)| serde_json::json!([sub(when), sub(then)])) + .collect::>(), + "else_expr": else_expr.as_deref().map(sub), + } + }), + ScalarExpr::CurrentTimestamp => serde_json::json!("CurrentTimestamp"), + ScalarExpr::EvalTimestamp => serde_json::json!("EvalTimestamp"), + ScalarExpr::PromqlScalarFromVector(node) => serde_json::json!({ + "PromqlScalarFromVector": scalar_ref(node, ids) + }), + ScalarExpr::ScalarSubquery(node) => serde_json::json!({ + "ScalarSubquery": scalar_ref(node, ids) + }), + ScalarExpr::Exists { subquery, negated } => serde_json::json!({ + "Exists": { "subquery": scalar_ref(subquery, ids), "negated": negated } + }), + ScalarExpr::InSubquery { + expr, + subquery, + negated, + } => serde_json::json!({ + "InSubquery": { + "expr": sub(expr), + "subquery": scalar_ref(subquery, ids), + "negated": negated, + } + }), } } @@ -1407,29 +1062,99 @@ mod tests { use std::rc::Rc; use super::*; + use crate::ir::operator_properties::{GroupKeys, JoinKind, Reduction}; + use crate::ir::Predicate; + use crate::post_asap::{ + BoundExpr, CompositionOperator, ErrorMetric, GroupingStrategy, GuaranteeSource, + ProbabilityExpr, SketchAlgorithm, SketchKind, SketchParams, SketchStatistic, SummaryUpdate, + }; use crate::pre_asap::agg_intent::AggIntent; - use crate::pre_asap::expr_ir::ScalarValue; - use crate::pre_asap::query_expr::{GroupKeys, Predicate, Reduction}; - use crate::pre_asap::schema::{Column, DataType, Schema}; + use crate::pre_asap::expr_ir::{ColumnRef, ScalarValue}; + use crate::pre_asap::schema::{DataType, Field, Schema}; + use crate::types::AccuracyTarget; - fn scan(table: &str, columns: Vec) -> QueryExpr { - QueryExpr::Scan { + fn scan(table: &str, columns: Vec) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Scan { source: Source::Table { table_ref: table.into(), }, predicates: vec![], schema: Schema { - columns, + fields: columns, time_index: None, unique_keys: vec![], closed: true, }, - } + }) + .unwrap() + } + + fn value_col() -> Vec { + vec![Field::plain("value", DataType::Float64, false)] + } + + fn true_pred() -> Predicate { + Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))) + } + + fn count_agg(child: Rc) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { + reduction: Reduction::Reduce(GroupKeys::none()), + measures: vec![AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }], + output_names: vec![], + filters: vec![], + having: None, + child, + }) + .unwrap() } - fn value_col() -> Vec { - vec![Column::new("value", DataType::Float64, false)] + fn join(left: Rc, right: Rc) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Join { + kind: JoinKind::Inner, + pred: true_pred(), + left, + right, + }) + .unwrap() + } + + /// A KLL `SummaryAgg` over `leaf`'s `v` column, read out as a quantile. + fn quantile_evaluation( + leaf: Rc, + guarantee: Option, + ) -> (Rc, Rc) { + let family = FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 40 }), + GroupingStrategy::default(), + ); + let agg = OperatorNode::asap_node( + ASAPOp::SummaryAgg { + child: leaf, + family: family.clone(), + input: SummaryUpdate::column(ColumnRef::Named("v".into())), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + }, + Schema::lifted(vec![Field::new("state", family, false)], None), + None, + ); + let evaluation = OperatorNode::asap_node( + ASAPOp::SummaryEstimate { + summary_input: Rc::clone(&agg), + query: SketchStatistic::Quantile { q: 0.99 }, + }, + Schema::lifted( + vec![Field::plain("quantile", DataType::Float64, false)], + None, + ), + guarantee, + ); + (agg, evaluation) } #[test] @@ -1438,12 +1163,14 @@ mod tests { assert_eq!(graph.nodes.len(), 1); assert_eq!(graph.root, 0); assert_eq!(graph.nodes[0].kind, "Scan"); + assert_eq!(graph.nodes[0].label, "Scan(metrics)"); assert!(graph.nodes[0].children.is_empty()); + assert!(graph.nodes[0].source_node.is_some()); } /// `export` itself never populates higher-layer annotations. Empty - /// annotations must not appear in serialized JSON, so ordinary (non-ASAP) - /// exports retain their existing shape. + /// annotations must not appear in serialized JSON, so ordinary exports + /// retain their existing shape. #[test] fn export_omits_empty_higher_layer_annotations() { let graph = export(&scan("metrics", value_col())); @@ -1460,6 +1187,10 @@ mod tests { !json.contains("decision"), "empty `decision` must be skipped, not serialized as `null`: {json}" ); + assert!( + !json.contains("source_node"), + "`source_node` is in-process only: {json}" + ); let graph_json = serde_json::to_string(&graph).unwrap(); assert!( !graph_json.contains("edge_annotations"), @@ -1473,22 +1204,27 @@ mod tests { // single child slot) share the exact same `Rc` Scan — // `export_post_asap` must merge them onto one node id. Sharing alone // is not physical cost evidence, so no edge cost may be fabricated. - let shared_scan = Rc::new(scan("metrics", value_col())); - let left_branch = QueryExpr::Dedup { + let shared_scan = scan("metrics", value_col()); + let left_branch = OperatorNode::non_asap_node(NonASAPOp::Dedup { cols: vec![0], child: Rc::clone(&shared_scan), - }; - let right_branch = QueryExpr::Limit { - n: 5, + }) + .unwrap(); + let right_branch = OperatorNode::non_asap_node(NonASAPOp::Limit { + n: Some(5), offset: 0, + partition_by: GroupKeys::none(), child: Rc::clone(&shared_scan), - }; - let root = QueryExpr::Concat { + }) + .unwrap(); + let root = OperatorNode::non_asap_node(NonASAPOp::Concat { children: vec![left_branch, right_branch], discriminator_unique_key: None, - }; + }) + .unwrap(); let graph = export_post_asap(&root, &mut |_| None); + assert_eq!(graph.nodes.len(), 4, "Scan, Dedup, Limit, Concat"); assert_eq!( graph.nodes.iter().filter(|n| n.kind == "Scan").count(), 1, @@ -1497,20 +1233,15 @@ mod tests { assert!(graph.edge_annotations.is_empty()); } - /// Regression test: a single parent referencing the same shared child - /// from two of its own operand slots at once (a `Join` whose left and - /// right sides are the exact same `Rc`, post pointer-dedup) is *one* - /// downstream consumer, not two — this must not inflate - /// produce an edge-cost annotation without explicit physical evidence. + /// A single parent referencing the same shared child from two of its + /// own operand slots at once (a `Join` whose left and right sides are + /// the exact same `Rc`) is *one* downstream consumer, not two — this + /// must not produce an edge-cost annotation without explicit physical + /// evidence. #[test] fn a_single_parent_referencing_a_shared_child_twice_is_one_consumer_not_two() { - let shared_scan = Rc::new(scan("metrics", value_col())); - let root = QueryExpr::Join { - kind: crate::pre_asap::query_expr::JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - left: Rc::clone(&shared_scan), - right: Rc::clone(&shared_scan), - }; + let shared_scan = scan("metrics", value_col()); + let root = join(Rc::clone(&shared_scan), Rc::clone(&shared_scan)); let graph = export_post_asap(&root, &mut |_| None); assert_eq!( @@ -1518,6 +1249,7 @@ mod tests { 1, "the shared Scan must be merged onto one node, not duplicated" ); + assert_eq!(graph.nodes[graph.root as usize].children, vec![0, 0]); assert!( graph.edge_annotations.is_empty(), "a single parent referencing the same child twice is one consumer, not a genuine \ @@ -1527,38 +1259,33 @@ mod tests { } #[test] - fn export_never_produces_edge_annotations_since_it_never_shares_nodes() { - // Plain `export` (no `export_post_asap`) never deduplicates by `Rc` - // pointer identity — even a workload-level shared subtree renders as - // two independent tree nodes here, so there is nothing to annotate. - let shared_scan = Rc::new(scan("metrics", value_col())); - let root = QueryExpr::Join { - kind: crate::pre_asap::query_expr::JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - left: Rc::clone(&shared_scan), - right: Rc::clone(&shared_scan), - }; - let graph = export(&root); - assert_eq!(graph.nodes.iter().filter(|n| n.kind == "Scan").count(), 2); + fn export_merges_pointer_shared_nodes_but_not_equal_copies() { + // Plain `export` deduplicates by `Rc` pointer identity: the same + // `Rc` reached twice is one node ... + let shared_scan = scan("metrics", value_col()); + let graph = export(&join(Rc::clone(&shared_scan), Rc::clone(&shared_scan))); + assert_eq!(graph.nodes.iter().filter(|n| n.kind == "Scan").count(), 1); assert!(graph.edge_annotations.is_empty()); + + // ... while two structurally equal but distinct `Rc`s stay two + // nodes (with equal hashes — that is CSE's job, not the export's). + let graph = export(&join( + scan("metrics", value_col()), + scan("metrics", value_col()), + )); + let scans: Vec<_> = graph.nodes.iter().filter(|n| n.kind == "Scan").collect(); + assert_eq!(scans.len(), 2); + assert_eq!(scans[0].hash, scans[1].hash); } #[test] fn chain_preserves_shape_and_child_links() { - let expr = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(QueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::none()), - measures: vec![AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan("metrics", value_col())), - }), - }; - let graph = export(&expr); + let root = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: true_pred(), + child: count_agg(scan("metrics", value_col())), + }) + .unwrap(); + let graph = export(&root); assert_eq!(graph.nodes.len(), 3, "Filter -> Aggregate -> Scan"); let filter = &graph.nodes[graph.root as usize]; @@ -1567,6 +1294,7 @@ mod tests { let agg = &graph.nodes[filter.children[0] as usize]; assert_eq!(agg.kind, "Aggregate"); + assert_eq!(agg.label, "Aggregate(1 measures)"); assert_eq!(agg.children.len(), 1); let leaf = &graph.nodes[agg.children[0] as usize]; @@ -1576,18 +1304,76 @@ mod tests { #[test] fn merge_keeps_every_branch_as_a_child() { - let expr = QueryExpr::concat(vec![ - scan("a", value_col()), - scan("b", value_col()), - scan("c", value_col()), - ]); - let graph = export(&expr); + let root = OperatorNode::non_asap_node(NonASAPOp::Concat { + children: vec![ + scan("a", value_col()), + scan("b", value_col()), + scan("c", value_col()), + ], + discriminator_unique_key: None, + }) + .unwrap(); + let graph = export(&root); assert_eq!(graph.nodes.len(), 4, "3 branches + the Concat node"); let merge = &graph.nodes[graph.root as usize]; assert_eq!(merge.kind, "Concat"); assert_eq!(merge.children.len(), 3); } + /// An operator node read from a scalar expression is a child of the + /// owning operator (after its operator inputs), and the expression's + /// `detail` points at it by id instead of inlining it. + #[test] + fn scalar_operator_references_are_children_rendered_as_scalar_refs() { + let subquery = scan("other", value_col()); + let root = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Exists { + subquery: Rc::clone(&subquery), + negated: false, + }), + child: scan("metrics", value_col()), + }) + .unwrap(); + let graph = export(&root); + assert_eq!(graph.nodes.len(), 3); + let filter = &graph.nodes[graph.root as usize]; + assert_eq!( + filter.children.len(), + 2, + "operator input, then the scalar reference" + ); + let input = &graph.nodes[filter.children[0] as usize]; + let referenced = &graph.nodes[filter.children[1] as usize]; + assert_eq!(input.label, "Scan(metrics)"); + assert_eq!(referenced.label, "Scan(other)"); + assert_eq!( + filter.detail["pred"]["Exists"]["subquery"]["scalar_ref"], + serde_json::json!(referenced.id) + ); + assert_eq!(filter.detail["pred"]["Exists"]["negated"], false); + let json = serde_json::to_string(&filter.detail).unwrap(); + assert!( + !json.contains("other"), + "the referenced subtree must not be inlined into detail: {json}" + ); + } + + #[test] + fn limit_without_n_is_offset_only() { + let root = OperatorNode::non_asap_node(NonASAPOp::Limit { + n: None, + offset: 3, + partition_by: GroupKeys::none(), + child: scan("metrics", value_col()), + }) + .unwrap(); + let graph = export(&root); + let limit = &graph.nodes[graph.root as usize]; + assert_eq!(limit.label, "Limit(offset 3)"); + assert_eq!(limit.detail["n"], serde_json::Value::Null); + assert_eq!(limit.detail["offset"], 3); + } + #[test] fn identical_subtrees_hash_equal_and_differing_ones_dont() { let left = scan("metrics", value_col()); @@ -1615,17 +1401,18 @@ mod tests { // Two roots that each wrap the *same* Scan shape in a different outer // node — the exported hash should still flag the shared Scan even // though it's embedded at different depths / under different parents. - let shared_shape = || scan("metrics", value_col()); - - let q1 = QueryExpr::Limit { - n: 10, + let q1 = OperatorNode::non_asap_node(NonASAPOp::Limit { + n: Some(10), offset: 0, - child: Rc::new(shared_shape()), - }; - let q2 = QueryExpr::Dedup { + partition_by: GroupKeys::none(), + child: scan("metrics", value_col()), + }) + .unwrap(); + let q2 = OperatorNode::non_asap_node(NonASAPOp::Dedup { cols: vec![0], - child: Rc::new(shared_shape()), - }; + child: scan("metrics", value_col()), + }) + .unwrap(); let g1 = export(&q1); let g2 = export(&q2); @@ -1647,9 +1434,9 @@ mod tests { fn root_hash_matches_cse_structural_hash_for_the_same_node() { // Not just "hashes equal for equal inputs" (any two consistent hash // functions would do that) — the exported root's `hash` must be the - // literal `u64` `crate::pre_asap::cse::structural_hash` produces for - // this exact node, because it's the same function call, not a - // parallel reimplementation that happens to agree. + // literal `u64` `crate::ir::cse::structural_hash` produces for this + // exact node, because it's the same function call, not a parallel + // reimplementation that happens to agree. let leaf = scan("metrics", value_col()); let graph = export(&leaf); assert_eq!( @@ -1662,23 +1449,14 @@ mod tests { #[test] fn every_node_hash_matches_cse_structural_hash_on_its_own_subtree() { // A multi-level tree: check the parity holds at every depth, not - // just the root — each `DagNode::hash` must equal - // `structural_hash` applied to the actual `QueryExpr` subtree that - // node represents. - let agg = QueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::none()), - measures: vec![AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan("metrics", value_col())), - }; - let root = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(agg.clone()), - }; + // just the root — each `DagNode::hash` must equal `structural_hash` + // applied to the actual node it represents. + let agg = count_agg(scan("metrics", value_col())); + let root = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: true_pred(), + child: Rc::clone(&agg), + }) + .unwrap(); let graph = export(&root); assert_eq!( @@ -1697,45 +1475,23 @@ mod tests { ); } - /// Issue #172: a readout's guarantee is exported structurally — metric, + /// Issue #172: a evaluation's guarantee is exported structurally — metric, /// symbolic bound, failure probability, provenance (allocation - /// included) — and a rejection carries its typed reason. + /// included) — and a rejection carries its typed reason. A relational + /// node below a summary is its own node, in the same graph. #[test] fn export_carries_guarantee_allocation_and_rejection_reason() { - use crate::post_asap::{ - BoundExpr, CompositionOperator, ErrorMetric, GroupingStrategy, GuaranteeSource, - ProbabilityExpr, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, - SummaryFamilyType, SummarySchema, - }; - let leaf = Rc::new(scan("t", vec![Column::new("v", DataType::Float64, false)])); - let kept = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::clone(&leaf)), - schema: SummarySchema { - fields: vec![], - time_index: None, - }, - guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), - }); - let agg = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: kept, - family: SummaryFamilyType::Sketch( - SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 40 }), - GroupingStrategy::default(), - ), - input: crate::post_asap::SummaryUpdate::column( - crate::pre_asap::expr_ir::ColumnRef::Named("v".into()), - ), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: SummarySchema { - fields: vec![], - time_index: None, - }, - guarantee: None, - }); + let leaf = Rc::new( + OperatorNode::new(Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { + table_ref: "t".into(), + }, + predicates: vec![], + schema: Schema::lifted(vec![Field::plain("v", DataType::Float64, false)], None), + })) + .unwrap() + .with_guarantee(Some(ResultGuarantee::exact("Scan"))), + ); let guarantee = ResultGuarantee { metric: ErrorMetric::Rank, bound: BoundExpr::Sum { @@ -1761,18 +1517,15 @@ mod tests { }, ], }; - let root = SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: agg, - query: SketchQuery::Quantile { q: 0.99 }, - }, - schema: SummarySchema { - fields: vec![], - time_index: None, - }, - guarantee: Some(guarantee), - }; + let (_, root) = quantile_evaluation(Rc::clone(&leaf), Some(guarantee)); let graph = export_summary(&root); + assert_eq!( + graph.nodes.iter().map(|n| n.kind).collect::>(), + ["scan", "summary_agg", "summary_estimate"] + ); + assert_eq!(graph.nodes[1].label, "SummaryAgg(Sketch(Kll))"); + assert!(graph.nodes[2].label.starts_with("SummaryEstimate(Quantile")); + assert!(graph.nodes.iter().all(|n| n.schema.is_some())); let json = serde_json::to_value(&graph).unwrap(); let root_json = &json["nodes"][graph.root as usize]; assert_eq!(root_json["guarantee"]["metric"], "rank"); @@ -1788,10 +1541,18 @@ mod tests { assert!(provenance.iter().any(|s| s["kind"] == "composition_step")); // Raw sketch state carries none; the exact leaf carries zero error. let state = &json["nodes"][1]; - assert_eq!(state["kind"], "SummaryAgg"); assert!(state.get("guarantee").is_none()); assert_eq!(json["nodes"][0]["guarantee"]["bound"]["op"], "zero"); + // The `DagNode` shape carries the same guarantee inside `detail`. + let dag = export(&root); + assert_eq!( + dag.nodes.iter().map(|n| n.kind).collect::>(), + ["Scan", "SummaryAgg", "SummaryEstimate"] + ); + assert_eq!(dag.nodes[2].detail["guarantee"]["metric"], "rank"); + assert!(dag.nodes[1].detail.get("guarantee").is_none()); + let named = NamedGraph { name: "q".into(), source: None, @@ -1801,7 +1562,7 @@ mod tests { workload_cost: None, rejections: vec![TargetRejection { target_pre_id: 0, - strategy: "SketchAlgorithmStrategy".into(), + strategy: "ASAPStrategies".into(), description: "quantile over quantile".into(), error: AccuracyError::UnsupportedComposition { operator: CompositionOperator::ApproximateAggregate, @@ -1828,38 +1589,163 @@ mod tests { .is_none()); } - fn viewer_kind_categories() -> std::collections::BTreeMap { - const START: &str = "const KIND_CATEGORY_JSON = `"; - let source = include_str!(concat!( - env!("CARGO_MANIFEST_DIR"), - "/../../tools/dag-viewer/node-style.js" - )); - let json = source - .split_once(START) - .expect("node-style.js must declare KIND_CATEGORY_JSON") - .1 - .split_once("`;") - .expect("KIND_CATEGORY_JSON must be a template literal") - .0; - serde_json::from_str(json).expect("KIND_CATEGORY_JSON must be valid JSON") + /// `export_post_asap` splices a winning summary in place of its target, + /// tags every node the splice introduced with the decision, and leaves + /// the rest of the query — including an input the summary reuses that + /// was already exported — untagged and shared. + #[test] + fn export_post_asap_splices_a_summary_substitution_in_place() { + let leaf = scan("t", vec![Field::plain("v", DataType::Float64, false)]); + let target = count_agg(Rc::clone(&leaf)); + // `leaf` is exported through the Join's left side before the target + // (its right side) is reached and substituted. + let root = join(Rc::clone(&leaf), Rc::clone(&target)); + let (_, evaluation) = quantile_evaluation(Rc::clone(&leaf), None); + let decision = DagDecision { + id: 7, + strategy: "Sketch".into(), + rationale: "quantile via KLL".into(), + rank: 0, + cost: 1.0, + role: "", + baseline_cost: None, + selected_cost: None, + benefit: None, + }; + let mut calls = Vec::new(); + let graph = export_post_asap(&root, &mut |node| { + calls.push(node.operator.kind_name()); + Rc::ptr_eq(node, &target).then(|| PostAsapSubstitution::Summary { + replacement: Rc::clone(&evaluation), + decision: decision.clone(), + }) + }); + + let kinds: Vec<_> = graph.nodes.iter().map(|n| n.kind).collect(); + assert_eq!(kinds, ["Scan", "SummaryAgg", "SummaryEstimate", "Join"]); + assert!(!kinds.contains(&"Aggregate"), "the target itself is gone"); + let join_node = &graph.nodes[graph.root as usize]; + assert_eq!(join_node.children, vec![0, 2]); + assert!(join_node.decision.is_none()); + let estimate = &graph.nodes[2]; + assert_eq!(estimate.kind, "SummaryEstimate"); + assert_eq!( + estimate.decision.as_ref().map(|d| (d.id, d.role)), + Some((7, "replacement_root")) + ); + let agg = &graph.nodes[estimate.children[0] as usize]; + assert_eq!( + agg.decision.as_ref().map(|d| (d.id, d.role)), + Some((7, "replacement_region")) + ); + let scan_node = &graph.nodes[agg.children[0] as usize]; + assert_eq!( + scan_node.id, 0, + "the summary reuses the already-exported input" + ); + assert!( + scan_node.decision.is_none(), + "a node exported before the splice is not tagged by it" + ); + assert_eq!( + calls, + ["Join", "Scan", "Aggregate", "SummaryAgg"], + "the substitution's own top level (SummaryEstimate) is never re-queried; its \ + descendants are, except the input already exported" + ); } + /// A `SharedSubDagStrategy`-shaped substitution returns the target + /// itself as its replacement; the walk must still terminate and render + /// the target once. #[test] - fn viewer_categorizes_exactly_the_exported_node_kinds() { - let expected: std::collections::BTreeSet<_> = QUERY_KIND_TAGS - .iter() - .chain(SUMMARY_KIND_TAGS) - .copied() - .chain(std::iter::once("KeepPreAsap")) - .collect(); + fn export_post_asap_terminates_when_the_replacement_is_the_target() { + let target = count_agg(scan("t", value_col())); + let decision = DagDecision { + id: 1, + strategy: "SharedSubtree".into(), + rationale: "share".into(), + rank: 0, + cost: f64::NAN, + role: "", + baseline_cost: None, + selected_cost: None, + benefit: None, + }; + let graph = export_post_asap(&target, &mut |node| { + Rc::ptr_eq(node, &target).then(|| PostAsapSubstitution::Rewrite { + replacement: Rc::clone(&target), + decision: decision.clone(), + }) + }); + assert_eq!(graph.nodes.len(), 2); + assert_eq!(graph.nodes[graph.root as usize].kind, "Aggregate"); assert_eq!( - expected.len(), - QUERY_KIND_TAGS.len() + SUMMARY_KIND_TAGS.len() + 1, - "exported kind tags must be unique" + graph.nodes[graph.root as usize] + .decision + .as_ref() + .map(|d| d.role), + Some("replacement_root") ); - let categories = viewer_kind_categories(); - let actual: std::collections::BTreeSet<_> = categories.keys().map(String::as_str).collect(); + } - assert_eq!(actual, expected); + /// The snake_case `kind` table is exactly `kind_name` re-cased, for + /// every variant: a `SummaryDagNode` and a `DagNode` for the same node + /// never disagree on what it is. + #[test] + fn summary_kind_is_the_operator_kind_name_in_snake_case() { + fn to_snake(name: &str) -> String { + let mut out = String::new(); + let chars: Vec = name.chars().collect(); + for (i, &c) in chars.iter().enumerate() { + if c.is_ascii_uppercase() { + let prev_lower = i > 0 && !chars[i - 1].is_ascii_uppercase(); + let next_lower = chars.get(i + 1).is_some_and(|n| n.is_ascii_lowercase()); + if i > 0 && (prev_lower || next_lower) { + out.push('_'); + } + out.push(c.to_ascii_lowercase()); + } else { + out.push(c); + } + } + out + } + let leaf = scan("t", vec![Field::plain("v", DataType::Float64, false)]); + let (_, evaluation) = quantile_evaluation(Rc::clone(&leaf), None); + let finalize = OperatorNode::asap_node( + ASAPOp::FinalizeExactAccumulator { child: evaluation }, + Schema::lifted(vec![], None), + None, + ); + let root = OperatorNode::non_asap_node(NonASAPOp::SQLWindowFunc { + func: crate::ir::operator_properties::WindowFuncKind::RowNumber, + args: vec![], + partition_by: GroupKeys::none(), + order_by: vec![], + frame: None, + output_name: "rn".into(), + child: finalize, + }) + .unwrap(); + let dag = export(&root); + let summary = export_summary(&root); + assert_eq!(dag.nodes.len(), summary.nodes.len()); + for (a, b) in dag.nodes.iter().zip(&summary.nodes) { + assert_eq!(a.id, b.id); + assert_eq!(b.kind, to_snake(a.kind), "{}", a.kind); + assert_eq!(a.children, b.children); + assert_eq!(a.label, b.label); + } + assert_eq!( + summary.nodes.iter().map(|n| n.kind).collect::>(), + [ + "scan", + "summary_agg", + "summary_estimate", + "finalize_exact_accumulator", + "sql_window_func", + ] + ); } } diff --git a/crates/types/src/ir/aggregate_schema.rs b/crates/types/src/ir/aggregate_schema.rs new file mode 100644 index 000000000..10b33d2ae --- /dev/null +++ b/crates/types/src/ir/aggregate_schema.rs @@ -0,0 +1,359 @@ +//! Derive aggregate output columns and types from input schema and reduction. +//! +//! [`aggregate_output_schema`] handles grouping keys and aggregate results. +//! For example, SQL `GROUP BY host` retains the grouping column and adds the +//! aggregate result; a PromQL per-series range reduction preserves labels +//! and produces a Float64 sample value. This module derives schemas, not +//! aggregate values or summary candidates. +use super::operator_properties::*; +use super::SchemaDerivationError; +use crate::pre_asap::{AggIntent, ColumnId, ColumnRef, DataType, Field, FieldDataType, Schema}; +/// Output schema of a *per-series* window/range reduction (`rate`/`increase`, +/// or an `*_over_time` reducer under a time `Window`). Such a reduction emits +/// one value per series, so every label column of `input` is preserved and only +/// the sample value is replaced — kept named `value` so the PromQL sample-value +/// convention (and any outer `SampleValue` reference) still resolves it by name. +fn per_series_reduction_schema( + input: &Schema, + agg: &AggIntent, +) -> Result { + let vi = if let Some(index) = agg.input_cols().first() { + *index + } else { + crate::pre_asap::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, input) + .map_err(|error| SchemaDerivationError::InvalidSampleColumn(error.to_string()))? + }; + if !matches!( + input.fields.get(vi).map(|column| &column.dtype), + Some(FieldDataType::Plain(DataType::Float64 | DataType::Int64)) + ) { + return Err(SchemaDerivationError::InvalidSampleColumn(format!( + "column {vi} is not numeric" + ))); + } + let mut columns = input.fields.clone(); + { + let mut out = agg.output_column(&columns[vi]); + out.name = "value".into(); + // A per-series range reduction produces a PromQL sample value, which is + // always `float64` — override the reducer's own output dtype so + // `count_over_time` (whose `Count` intent types `Int64`) matches every + // other range reducer instead of leaking an `Int64` value column (#69). + out.dtype = FieldDataType::Plain(DataType::Float64); + columns[vi] = out; + } + Ok(Schema { + fields: columns, + time_index: input.time_index, + unique_keys: input.unique_keys.clone(), + // Per-series reduction is label-preserving: it inherits its input's + // completeness (an open scan stays open; a closed one stays closed). + closed: input.closed, + }) +} + +/// The output schema of an `Aggregate { reduction, measures }` over `in_schema` — +/// the **single** canonical derivation shared by +/// [`NonASAPOp::output_schema`](crate::ir::NonASAPOp::output_schema)'s +/// `Aggregate` arm and the HAVING-resolution path (`column_resolution::output_schema_for_aggregate`), +/// so the two can never drift (issue #41). +/// +/// `Reduction::PerEntity` selects the label-preserving +/// [`per_series_reduction_schema`] (`rate`/`increase`/`*_over_time`) instead +/// of the cross-series `by ++ measures` shape. Which one applies is read directly +/// off `reduction` — decided once, at construction, by whoever built the +/// `Aggregate` node (issue #165) — not re-derived here from `by`/child shape. +pub fn aggregate_output_schema( + in_schema: &Schema, + reduction: &Reduction, + measures: &[AggIntent], + output_names: &[String], +) -> Result { + let by = match reduction { + Reduction::PerEntity => { + debug_assert_eq!( + measures.len(), + 1, + "a per-entity reduction is single-aggregate" + ); + return per_series_reduction_schema(in_schema, &measures[0]); + } + Reduction::Reduce(by) => by, + }; + + // `without(excluded)` groups by every label *except* those listed: the kept + // labels are the input's label columns minus the excluded positions (and the + // ts / sample-value columns), and the schema stays **open** because the full + // runtime label set isn't known. The `by(...)` path instead enumerates its + // kept columns and freezes to closed (issue #39). + if by.is_without() { + return without_output_schema(in_schema, by.keys(), measures, output_names); + } + + let mut out_cols: Vec = Vec::with_capacity(by.len() + measures.len()); + for &id in by.keys() { + let c = in_schema + .fields + .get(id) + .ok_or(SchemaDerivationError::InvalidGroupByColumn( + id, + in_schema.fields.len(), + ))?; + out_cols.push(c.clone()); + } + let value_col_idx = + crate::pre_asap::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, in_schema) + .ok() + .or_else(|| (0..in_schema.fields.len()).find(|i| !by.contains(i))); + let probe = value_col_idx + .and_then(|i| in_schema.fields.get(i)) + .cloned() + .unwrap_or_else(|| Field::plain("value", DataType::Float64, false)); + // Each reducer types off its own input column (`SUM(bytes)` vs `AVG(latency)` + // in one node); `None` falls back to the sample-value probe (PromQL's + // single-column convention). A non-empty `output_names[i]` overrides the + // synthetic output column name. + for (i, intent) in measures.iter().enumerate() { + // `count_values("l", v)` emits TWO columns: the synthesized `Utf8` label + // `l` (the stringified sample value it groups by) and the per-value + // count. If `l` collides with a group-by key of the same name, PromQL's + // synthesized label takes precedence — emit a single column, never a + // duplicate. + if let AggIntent::CountValues { label } = intent { + if !out_cols.iter().any(|c| c.name == *label) { + out_cols.push(Field::plain(label.clone(), DataType::Utf8, false)); + } + let mut cnt = intent.output_column(&probe); + if let Some(name) = output_names.get(i).filter(|s| !s.is_empty()) { + cnt.name = name.clone(); + } + out_cols.push(cnt); + continue; + } + // Only the output *type* is read from here, so the leading column is + // enough for the multi-column intents: `Cardinality` and `PearsonCorr` + // both have a fixed output type that ignores it. + let in_col = intent + .input_cols() + .first() + .and_then(|id| in_schema.fields.get(*id)) + .unwrap_or(&probe); + let mut out = intent.output_column(in_col); + // A global extremum emits NULL for an empty input, even if its input + // column is non-nullable. Grouped extrema only emit existing groups. + if by.is_empty() && matches!(intent, AggIntent::Min { .. } | AggIntent::Max { .. }) { + out.nullable = true; + } + if let Some((arg, _)) = intent + .arg_selector_columns(in_schema) + .map_err(SchemaDerivationError::InvalidScalarSignature)? + { + out.dtype = in_schema.fields[arg].dtype.clone(); + out.nullable = in_schema.fields[arg].nullable; + } + if let Some(name) = output_names.get(i).filter(|s| !s.is_empty()) { + out.name = name.clone(); + } + out_cols.push(out); + } + // `count_values` groups by (by-keys ∪ the synthesized value label), so the + // by-keys alone are not a unique key — be conservative and claim none. + let has_count_values = measures + .iter() + .any(|a| matches!(a, AggIntent::CountValues { .. })); + let unique_keys = if by.is_empty() || has_count_values { + Vec::new() + } else { + vec![(0..by.len()).collect()] + }; + Ok(Schema { + fields: out_cols, + time_index: None, + unique_keys, + // A cross-series aggregate enumerates exactly `by ++ measures`, so its output + // is closed even over an open input — this is where an open schema + // freezes to closed. + closed: true, + }) +} + +/// Output schema of a `without(excluded)` aggregate: the kept labels (every +/// input label column except the `excluded` positions, the time axis, and the +/// sample-value column) followed by the aggregate output column(s). Unlike the +/// `by` path this stays **open** — the excluded set is enumerable but the kept +/// set is not (the runtime carries labels the usage-derived schema never saw), +/// so the schema can't freeze to closed and claims no unique key (issue #39). +fn without_output_schema( + in_schema: &Schema, + excluded: &[ColumnId], + measures: &[AggIntent], + output_names: &[String], +) -> Result { + for &id in excluded { + if id >= in_schema.fields.len() { + return Err(SchemaDerivationError::InvalidGroupByColumn( + id, + in_schema.fields.len(), + )); + } + } + let mut out_cols: Vec = Vec::new(); + for (i, col) in in_schema.fields.iter().enumerate() { + let is_time = in_schema.time_index == Some(i); + let is_value = crate::pre_asap::column_resolution::resolve_column_ref( + &ColumnRef::SampleValue, + in_schema, + ) + .ok() + == Some(i); + if !is_time && !is_value && !excluded.contains(&i) { + out_cols.push(col.clone()); + } + } + let probe = in_schema + .column_id("value") + .and_then(|i| in_schema.fields.get(i)) + .cloned() + .unwrap_or_else(|| Field::plain("value", DataType::Float64, false)); + for (i, intent) in measures.iter().enumerate() { + // Only the output *type* is read from here, so the leading column is + // enough for the multi-column intents: `Cardinality` and `PearsonCorr` + // both have a fixed output type that ignores it. + let in_col = intent + .input_cols() + .first() + .and_then(|id| in_schema.fields.get(*id)) + .unwrap_or(&probe); + let mut out = intent.output_column(in_col); + if let Some((arg, _)) = intent + .arg_selector_columns(in_schema) + .map_err(SchemaDerivationError::InvalidScalarSignature)? + { + out.dtype = in_schema.fields[arg].dtype.clone(); + out.nullable = in_schema.fields[arg].nullable; + } + if let Some(name) = output_names.get(i).filter(|s| !s.is_empty()) { + out.name = name.clone(); + } + out_cols.push(out); + } + Ok(Schema { + fields: out_cols, + time_index: None, + unique_keys: Vec::new(), + // The kept label set is runtime-only, so — unlike `by` — this does not + // freeze the open schema to closed. + closed: false, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn col(name: &str, dtype: DataType, nullable: bool) -> Field { + Field::plain(name, dtype, nullable) + } + + #[test] + fn group_keys_by_vs_without_semantics() { + let by = GroupKeys::by(vec![1, 2]); + let without = GroupKeys::without(vec![1, 2]); + assert!(!by.is_without()); + assert!(without.is_without()); + // Deref / iteration expose the stored keys regardless of mode. + assert_eq!(by.len(), 2); + assert_eq!(without.keys(), &[1, 2]); + // A `by` compares equal to its bare vec; a `without` never does. + assert_eq!(by, vec![1, 2]); + assert_ne!(without, vec![1, 2]); + assert_ne!(by, without); + } + + #[test] + fn group_keys_serde_by_is_bare_array_without_is_tagged() { + // `by` keeps the pre-#39 bare-array wire format; `without` uses an object. + let by = serde_json::to_string(&GroupKeys::by(vec![2, 3])).unwrap(); + assert_eq!(by, "[2,3]"); + let without = serde_json::to_string(&GroupKeys::without(vec![2])).unwrap(); + assert_eq!(without, r#"{"without":[2]}"#); + // Round-trip both. + for g in [GroupKeys::by(vec![2, 3]), GroupKeys::without(vec![2])] { + let json = serde_json::to_string(&g).unwrap(); + let back: GroupKeys = serde_json::from_str(&json).unwrap(); + assert_eq!(back, g); + } + } + + #[test] + fn time_shift_identity_and_serde() { + let offset_only = TimeShift { + offset_ms: 1, + at: None, + }; + let at_only = TimeShift { + offset_ms: 0, + at: Some(AtModifier::End), + }; + assert!(TimeShift::default().is_identity()); + assert!(!offset_only.is_identity()); + assert!(!at_only.is_identity()); + // Round-trip the shift + anchor. + let s = TimeShift { + offset_ms: -300_000, + at: Some(AtModifier::Timestamp(60_000)), + }; + let back: TimeShift = serde_json::from_str(&serde_json::to_string(&s).unwrap()).unwrap(); + assert_eq!(back, s); + } + + // Nested temporal aggregation must replace the sample, never the grouping label. + #[test] + fn temporal_reduction_of_grouped_sum_preserves_job() { + let input = Schema::new(vec![ + col("job", DataType::Utf8, true), + col("sum", DataType::Float64, false), + ]); + for aggregate in [ + AggIntent::Avg { col: None }, + AggIntent::Avg { col: Some(1) }, + AggIntent::Rate, + ] { + let output = + aggregate_output_schema(&input, &Reduction::PerEntity, &[aggregate], &[]).unwrap(); + assert_eq!(output.fields[0], input.fields[0]); + assert_eq!(output.fields[1].name, "value"); + assert_eq!(output.fields[1].dtype, DataType::Float64); + } + } + + #[test] + fn discriminator_assertion_rejects_unknown_wire_fields() { + let json = r#"{"discriminator":1,"inner_key":[0],"unverified":true}"#; + assert!(serde_json::from_str::(json).is_err()); + } + + #[test] + fn aggregate_strips_time_and_keeps_unique_keys() { + let input = Schema::with_time_index( + vec![ + col("ts", DataType::Timestamp, false), + col("value", DataType::Float64, false), + col("host", DataType::Utf8, false), + ], + 0, + Vec::new(), + ); + let out = aggregate_output_schema( + &input, + &Reduction::by(vec![2]), + &[AggIntent::Sum { col: None }], + &[], + ) + .expect("valid group-by column"); + let names: Vec<_> = out.fields.iter().map(|c| c.name.as_str()).collect(); + assert_eq!(names, vec!["host", "sum"]); + assert!(out.time_index.is_none()); + assert_eq!(out.unique_keys, vec![vec![0]]); + } +} diff --git a/crates/types/src/ir/asap.rs b/crates/types/src/ir/asap.rs new file mode 100644 index 000000000..4ccbe7b44 --- /dev/null +++ b/crates/types/src/ir/asap.rs @@ -0,0 +1,594 @@ +//! ASAP operators: summary-state construction, state operations and evaluations. +//! The summary family, kind/algorithm and parameters are committed here. + +use std::rc::Rc; + +use serde::{Deserialize, Serialize}; + +use super::node::{OperatorNode, OperatorResultKind}; +use crate::ir::operator_properties::Reduction; +use crate::ir::SchemaDerivationError; +use crate::post_asap::maintained_population::{MaintainedPopulation, PopulationStatistic}; +use crate::post_asap::sketch::{GroupingStrategy, SketchStatistic, SummaryUpdate}; +use crate::pre_asap::schema::{ColumnId, DataType, Field, FieldDataType, Schema}; + +/// Why an ASAP operator cannot be used yet. +pub const UNIMPLEMENTED_ASAP_OP: &str = + "this ASAP operator is reserved: schema, accuracy, timing and export are not implemented"; + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum ASAPOp { + /// Summary aggregation. Output: grouping columns + one field carrying + /// partial summary state per group, typed `family`. + SummaryAgg { + child: Rc, + /// Which summary family realizes this aggregation. Never + /// `FieldDataType::Plain`. + family: FieldDataType, + input: SummaryUpdate, + reduction: Reduction, + grouping: GroupingStrategy, + #[serde(default)] + filter: Option, + }, + /// Read out a query result from built summary state. Output is a + /// row-shaped schema. + SummaryEstimate { + summary_input: Rc, + query: SketchStatistic, + }, + /// Read an exact accumulator's state as its finalized value: the + /// maintenance-to-read boundary before query-time operators. + FinalizeExactAccumulator { + child: Rc, + }, + /// Maintain the full declared population, including membership changes. + MaintainPopulation { + child: Rc, + population: MaintainedPopulation, + }, + /// Read an aggregate or TopK prefix from the maintained population. + EvaluatePopulation { + child: Rc, + evaluation: PopulationStatistic, + }, + // ── Reserved: migrated but unimplemented (§1.3 of the proposal) ── + SummaryMerge { + children: Vec>, + }, + SummarySubtract { + left: Rc, + right: Rc, + }, + SummaryDelete { + summary_input: Rc, + key: ColumnId, + }, + SummaryJoin { + outer: Rc, + inner: Rc, + key: ColumnId, + family: FieldDataType, + }, + Extension { + child: Rc, + name: String, + }, +} + +impl ASAPOp { + pub fn children(&self) -> Vec<&Rc> { + use ASAPOp::*; + match self { + SummaryAgg { child, filter, .. } => { + let mut inputs = vec![child]; + if let Some(filter) = filter { + inputs.extend(filter.0.operator_refs()); + } + inputs + } + FinalizeExactAccumulator { child } + | MaintainPopulation { child, .. } + | EvaluatePopulation { child, .. } + | Extension { child, .. } => vec![child], + SummaryEstimate { summary_input, .. } | SummaryDelete { summary_input, .. } => { + vec![summary_input] + } + SummarySubtract { left, right } => vec![left, right], + SummaryJoin { outer, inner, .. } => vec![outer, inner], + SummaryMerge { children } => children.iter().collect(), + } + } + + pub fn map_children(&self, mut f: impl FnMut(&Rc) -> Rc) -> Self { + use ASAPOp::*; + match self { + SummaryAgg { + child, + family, + input, + reduction, + grouping, + filter, + } => SummaryAgg { + child: f(child), + family: family.clone(), + input: input.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + filter: filter + .as_ref() + .map(|p| super::scalar::Predicate(p.0.map_operator_refs(&mut f))), + }, + SummaryEstimate { + summary_input, + query, + } => SummaryEstimate { + summary_input: f(summary_input), + query: query.clone(), + }, + FinalizeExactAccumulator { child } => FinalizeExactAccumulator { child: f(child) }, + MaintainPopulation { child, population } => MaintainPopulation { + child: f(child), + population: population.clone(), + }, + EvaluatePopulation { child, evaluation } => EvaluatePopulation { + child: f(child), + evaluation: evaluation.clone(), + }, + SummaryMerge { children } => SummaryMerge { + children: children.iter().map(&mut f).collect(), + }, + SummarySubtract { left, right } => SummarySubtract { + left: f(left), + right: f(right), + }, + SummaryDelete { summary_input, key } => SummaryDelete { + summary_input: f(summary_input), + key: *key, + }, + SummaryJoin { + outer, + inner, + key, + family, + } => SummaryJoin { + outer: f(outer), + inner: f(inner), + key: *key, + family: family.clone(), + }, + Extension { child, name } => Extension { + child: f(child), + name: name.clone(), + }, + } + } + + pub fn kind_name(&self) -> &'static str { + use ASAPOp::*; + match self { + SummaryAgg { .. } => "SummaryAgg", + SummaryEstimate { .. } => "SummaryEstimate", + FinalizeExactAccumulator { .. } => "FinalizeExactAccumulator", + MaintainPopulation { .. } => "MaintainPopulation", + EvaluatePopulation { .. } => "EvaluatePopulation", + SummaryMerge { .. } => "SummaryMerge", + SummarySubtract { .. } => "SummarySubtract", + SummaryDelete { .. } => "SummaryDelete", + SummaryJoin { .. } => "SummaryJoin", + Extension { .. } => "Extension", + } + } + + /// Reserved variants that are migrated but not implemented. + pub fn is_unimplemented(&self) -> bool { + use ASAPOp::*; + matches!( + self, + SummaryMerge { .. } + | SummarySubtract { .. } + | SummaryDelete { .. } + | SummaryJoin { .. } + | Extension { .. } + ) + } + + fn unimplemented() -> SchemaDerivationError { + SchemaDerivationError::InvalidScalarSignature(UNIMPLEMENTED_ASAP_OP.into()) + } + + /// The summary state this operator produces, if it produces state. + pub fn produced_state(&self) -> Option<&FieldDataType> { + match self { + ASAPOp::SummaryAgg { family, .. } | ASAPOp::SummaryJoin { family, .. } => Some(family), + _ => None, + } + } + + /// Output schema derived from the operator and its children. Summary + /// planning may retain a more specific schema (evaluation column naming) + /// through [`OperatorNode::with_schema`]; the derived shape agrees with it + /// in field types. + pub fn output_schema(&self) -> Result { + use ASAPOp::*; + Ok(match self { + SummaryAgg { + child, + family, + reduction, + .. + } => { + let mut schema = crate::pre_asap::aggregate_output_schema( + &child.schema, + reduction, + &[crate::pre_asap::AggIntent::Sum { col: None }], + &[], + )?; + let index = match reduction { + Reduction::PerEntity => crate::pre_asap::resolve_column_ref( + &crate::pre_asap::ColumnRef::SampleValue, + &schema, + ) + .map_err(|e| SchemaDerivationError::InvalidScalarSignature(e.to_string()))?, + Reduction::Reduce(_) => schema.fields.len() - 1, + }; + schema.fields[index] = Field::new("state", family.clone(), false); + schema + } + SummaryEstimate { + summary_input, + query, + } => { + let input = &summary_input.schema; + let (name, mut dtype) = match query { + SketchStatistic::Quantile { .. } => ("quantile", DataType::Float64), + SketchStatistic::Cardinality => ("cardinality", DataType::Int64), + SketchStatistic::PointCount { .. } => ("count", DataType::Int64), + SketchStatistic::FrequencyL2 => ("frequency_l2", DataType::Float64), + SketchStatistic::FrequencyEntropy => ("frequency_entropy", DataType::Float64), + SketchStatistic::TopK { .. } => ("topk", DataType::Utf8), + }; + if matches!( + summary_input.asap(), + Some(SummaryAgg { + reduction: Reduction::PerEntity, + .. + }) + ) && dtype == DataType::Int64 + { + dtype = DataType::Float64; + } + let mut schema = input.clone(); + for field in &mut schema.fields { + if !field.is_plain() { + *field = Field::plain(name, dtype.clone(), false); + } + } + schema + } + FinalizeExactAccumulator { child } => { + let value_result = if let Some(ASAPOp::SummaryAgg { + child: source, + family: FieldDataType::ExactAggregate(kind, _), + input, + reduction, + .. + }) = child.asap() + { + { + use crate::post_asap::{ExactKind, SummaryInputExpr}; + use crate::pre_asap::AggIntent; + let column = match &input.weight { + SummaryInputExpr::Column(col) => Some( + crate::pre_asap::column_resolution::resolve_column_ref( + col, + &source.schema, + ) + .map_err(|e| { + SchemaDerivationError::InvalidScalarSignature(e.to_string()) + })?, + ), + _ => None, + }; + let measure = match kind { + ExactKind::Sum => Some(AggIntent::Sum { col: column }), + ExactKind::Min => Some(AggIntent::Min { col: column }), + ExactKind::Max => Some(AggIntent::Max { col: column }), + ExactKind::Count => Some(AggIntent::Count { + accuracy: crate::types::AccuracyTarget::Exact, + }), + ExactKind::Rate => Some(AggIntent::Rate), + ExactKind::IRate => Some(AggIntent::IRate), + ExactKind::Increase => Some(AggIntent::Increase), + }; + measure + .map(|measure| { + super::NonASAPOp::Aggregate { + child: Rc::clone(source), + reduction: reduction.clone(), + measures: vec![measure], + output_names: vec![], + filters: vec![], + having: None, + } + .output_schema() + }) + .transpose()? + .and_then(|s| match reduction { + Reduction::PerEntity => { + s.column_id("value").and_then(|i| s.fields.get(i).cloned()) + } + Reduction::Reduce(_) => s.fields.last().cloned(), + }) + } + } else { + None + }; + let mut out = Schema::lifted(child.schema.fields.clone(), child.schema.time_index); + for f in &mut out.fields { + if let FieldDataType::ExactAggregate(kind, _) = &f.dtype { + if let Some(result) = &value_result { + f.dtype = result.dtype.clone(); + f.nullable = result.nullable; + } else { + f.dtype = FieldDataType::Plain(finalized_data_type(kind)); + } + } + } + out + } + MaintainPopulation { child, .. } => { + Schema::lifted(child.schema.fields.clone(), child.schema.time_index) + } + EvaluatePopulation { child, evaluation } => { + use crate::post_asap::maintained_population::PopulationInput; + use crate::pre_asap::{AggIntent, GroupKeys}; + let Some(MaintainPopulation { + child: source, + population, + }) = child.asap() + else { + return Err(SchemaDerivationError::InvalidScalarSignature( + "population evaluation requires maintained membership".into(), + )); + }; + if matches!(evaluation, PopulationStatistic::TopK { .. }) { + source.schema.clone() + } else { + let (keys, column) = match &population.input { + PopulationInput::Rows { + grouping, + value_column, + .. + } => (grouping.clone(), Some(*value_column)), + PopulationInput::CurrentSeries(spec) => { + let keys = spec + .grouping + .iter() + .map(|name| { + source.schema.column_id(name).ok_or_else(|| { + SchemaDerivationError::InvalidScalarSignature( + "population grouping column is absent".into(), + ) + }) + }) + .collect::, _>>()?; + ( + if spec.without { + GroupKeys::without(keys) + } else { + GroupKeys::by(keys) + }, + None, + ) + } + }; + let accuracy = crate::types::AccuracyTarget::Exact; + let measure = match evaluation { + PopulationStatistic::Quantile { q } => AggIntent::Quantile { + q: *q, + col: column, + accuracy, + }, + PopulationStatistic::Sum => AggIntent::Sum { col: column }, + PopulationStatistic::Count => AggIntent::Count { accuracy }, + PopulationStatistic::Average => AggIntent::Avg { col: column }, + PopulationStatistic::TopK { .. } => unreachable!(), + }; + super::NonASAPOp::Aggregate { + child: source.clone(), + reduction: Reduction::Reduce(keys), + measures: vec![measure], + output_names: vec![], + filters: vec![], + having: None, + } + .output_schema()? + } + } + SummaryMerge { .. } + | SummarySubtract { .. } + | SummaryDelete { .. } + | SummaryJoin { .. } + | Extension { .. } => return Err(Self::unimplemented()), + }) + } + + pub fn output_kind(&self) -> OperatorResultKind { + use ASAPOp::*; + match self { + SummaryAgg { .. } + | MaintainPopulation { .. } + | SummaryMerge { .. } + | SummarySubtract { .. } + | SummaryDelete { .. } + | SummaryJoin { .. } + | Extension { .. } => OperatorResultKind::State, + SummaryEstimate { summary_input, .. } => source_kind(summary_input), + FinalizeExactAccumulator { child } | EvaluatePopulation { child, .. } => { + source_kind(child) + } + } + } + + /// Local input-contract checks. + pub fn validate_inputs(&self) -> Result<(), SchemaDerivationError> { + use ASAPOp::*; + let needs_state = |node: &OperatorNode, what: &str| { + if node.result_kind != OperatorResultKind::State { + Err(SchemaDerivationError::InvalidScalarSignature(format!( + "{what} requires summary state as input, got {:?}", + node.result_kind + ))) + } else { + Ok(()) + } + }; + match self { + SummaryEstimate { + summary_input, + query, + } => { + needs_state(summary_input, "SummaryEstimate")?; + use crate::post_asap::sketch::SketchCategory as C; + let states: Vec<_> = summary_input + .schema + .fields + .iter() + .filter(|f| !f.is_plain()) + .collect(); + let valid = match states.as_slice() { + [field] => match &field.dtype { + FieldDataType::Sketch(kind, _) => matches!( + (kind.category(), query), + (C::Quantile, SketchStatistic::Quantile { .. }) + | (C::Cardinality | C::Universal, SketchStatistic::Cardinality) + | ( + C::Frequency | C::Universal, + SketchStatistic::PointCount { .. } + ) + | ( + C::Universal, + SketchStatistic::FrequencyL2 + | SketchStatistic::FrequencyEntropy + ) + | (C::TopK | C::Universal, SketchStatistic::TopK { .. }) + ), + _ => false, + }, + _ => false, + }; + if !valid { + return Err(SchemaDerivationError::InvalidScalarSignature( + "evaluation does not match its summary family".into(), + )); + } + Ok(()) + } + FinalizeExactAccumulator { child } => { + needs_state(child, "FinalizeExactAccumulator")?; + if child + .schema + .fields + .iter() + .all(|f| !matches!(f.dtype, FieldDataType::ExactAggregate(..))) + { + return Err(SchemaDerivationError::InvalidScalarSignature( + "FinalizeExactAccumulator requires exact accumulator state".into(), + )); + } + Ok(()) + } + EvaluatePopulation { child, evaluation } => { + needs_state(child, "EvaluatePopulation")?; + if !matches!(child.asap(), Some(MaintainPopulation { population, .. }) if population.supports(evaluation)) + { + return Err(SchemaDerivationError::InvalidScalarSignature( + "population evaluation requires compatible maintained membership".into(), + )); + } + Ok(()) + } + SummaryAgg { + child, + family, + input, + filter, + .. + } => { + if family.is_plain() || child.result_kind == OperatorResultKind::State { + return Err(SchemaDerivationError::InvalidScalarSignature( + "summary aggregation requires values and produces a state family".into(), + )); + } + fn check( + expr: &crate::post_asap::SummaryInputExpr, + schema: &Schema, + ) -> Result<(), SchemaDerivationError> { + use crate::post_asap::SummaryInputExpr; + match expr { + SummaryInputExpr::Column(col) => { + crate::pre_asap::resolve_column_ref(col, schema).map_err(|e| { + SchemaDerivationError::InvalidScalarSignature(e.to_string()) + })?; + } + SummaryInputExpr::Tuple(items) => { + for item in items { + check(item, schema)?; + } + } + _ => {} + } + Ok(()) + } + check(&input.weight, &child.schema)?; + if let Some(item) = &input.item { + check(item, &child.schema)?; + } + if let Some(filter) = filter { + if filter.0.scalar_type(&child.schema)?.0 != DataType::Bool { + return Err(SchemaDerivationError::InvalidScalarSignature( + "summary filter must be boolean".into(), + )); + } + } + Ok(()) + } + MaintainPopulation { child, population } => { + if !population.matches_node(child) { + return Err(SchemaDerivationError::InvalidScalarSignature( + "population input differs from its membership contract".into(), + )); + } + Ok(()) + } + _ => Err(Self::unimplemented()), + } + } +} + +/// The plain value an exact accumulator finalizes to. +fn finalized_data_type(kind: &crate::post_asap::sketch::ExactKind) -> DataType { + use crate::post_asap::sketch::ExactKind; + match kind { + ExactKind::Count => DataType::Int64, + _ => DataType::Float64, + } +} + +/// The category of the values a evaluation of `state` produces: the category +/// of the relational input the state was built from. +fn source_kind(node: &OperatorNode) -> OperatorResultKind { + match &node.operator { + super::node::Operator::ASAP(op) => match op.children().first() { + Some(child) => source_kind(child), + None => OperatorResultKind::Relation, + }, + super::node::Operator::NonASAP(_) => match node.result_kind { + OperatorResultKind::RangeVector => OperatorResultKind::InstantVector, + OperatorResultKind::State => OperatorResultKind::Relation, + other => other, + }, + } +} diff --git a/crates/types/src/ir/canonicalize.rs b/crates/types/src/ir/canonicalize.rs new file mode 100644 index 000000000..1b8d48ff5 --- /dev/null +++ b/crates/types/src/ir/canonicalize.rs @@ -0,0 +1,1077 @@ +//! Post-lowering canonicalization of the operator DAG. +//! +//! Erases *structural* differences between semantically identical queries so +//! a post-ASAP binding rule matching on the intent algebra sees one canonical +//! spelling regardless of source language (issue #34). +//! +//! ## Heavy-hitter promotion +//! +//! An additive-ranked "order by the aggregate, take the top k" is a +//! heavy-hitter represented by [`AggIntent::TopK`]. Front ends may emit it as +//! an ordinary `Limit { Sort { … Aggregate } }`; this pass promotes that shape +//! to the canonical +//! +//! ```text +//! Aggregate { reduction: Reduce(), measures: [TopK{k}], +//! child: Aggregate { measures: [Count | Sum], … } } +//! ``` +//! +//! Count supplies unit weights and Sum supplies value weights. Because the +//! match is positional, aliases do not affect it. Other ranked expressions +//! retain Sort + Limit. +//! +//! ## Subquery lowering +//! +//! EXISTS/NOT EXISTS and positive IN filter conjuncts may use semi/anti joins. +//! Scalar subqueries remain explicit: a cross join does not preserve their +//! zero-row NULL or multiple-row error semantics. All scalar plan references +//! participate in DAG traversal and canonicalization. + +use std::collections::HashMap; +use std::rc::Rc; + +use super::node::{Operator, OperatorNode}; +use super::non_asap::NonASAPOp; +use super::scalar::{ExprSemantics, Predicate, ProjectItem, ScalarExpr, SortKey}; +use crate::ir::operator_properties::{JoinKind, Reduction}; +use crate::ir::SchemaDerivationError; +use crate::pre_asap::agg_intent::{topk, AggIntent}; +use crate::pre_asap::expr_ir::{CompareOpKind, ScalarValue}; +use crate::types::AccuracyTarget; + +/// Rewrite the DAG under `root` into its canonical form (bottom-up). +/// Idempotent: an already-canonical DAG comes back as the same `Rc`. Only +/// nodes that change (or whose inputs change) are rebuilt; every untouched +/// sub-DAG keeps its pointer identity, and a shared sub-DAG that is rewritten +/// stays shared. +pub fn canonicalize(root: Rc) -> Result, SchemaDerivationError> { + canon(&root, &mut HashMap::new()) +} + +fn canon( + node: &Rc, + memo: &mut HashMap<*const OperatorNode, Rc>, +) -> Result, SchemaDerivationError> { + if let Some(done) = memo.get(&Rc::as_ptr(node)) { + return Ok(Rc::clone(done)); + } + + // A `Concat` asserting a caller-proven `discriminator_unique_key` (issue + // #228) had that key's `ColumnId`s resolved against exactly the first + // branch's output schema *as it stood before this pass ran*. The rewrites + // below can restructure that branch (anywhere within it) into a shape + // with a different output schema, which would leave those `ColumnId`s + // pointing at the wrong column, or out of bounds. Snapshot the schema the + // key was resolved against before recursing into the children. + let discriminator_branch_schema_before = match &node.operator { + Operator::NonASAP(NonASAPOp::Concat { + children, + discriminator_unique_key: Some(_), + }) => children.first().map(|c| c.schema.clone()), + _ => None, + }; + + // Bottom-up: canonicalize every operator input before matching at this + // node, so an inner heavy-hitter is promoted before an enclosing rewrite + // inspects it. + let mut rebuilt: Vec<(*const OperatorNode, Rc)> = Vec::new(); + let mut changed = false; + for child in operator_children(&node.operator) { + let new = canon(child, memo)?; + changed |= !Rc::ptr_eq(&new, child); + rebuilt.push((Rc::as_ptr(child), new)); + } + let mut current = if changed { + // `map_children` also visits the operator nodes referenced from + // scalar expressions; those are not in `rebuilt` and pass through + // unchanged. (A node that is both an operator input and a scalar + // reference is one shared node, so it takes its canonical form in + // both places.) + let rebuilt_child = |c: &Rc| { + rebuilt + .iter() + .find(|(ptr, _)| *ptr == Rc::as_ptr(c)) + .map_or_else(|| Rc::clone(c), |(_, new)| Rc::clone(new)) + }; + Rc::new(node.map_children(rebuilt_child)?) + } else { + Rc::clone(node) + }; + + // If the first branch's output schema moved out from under the asserted + // key, the key can no longer be trusted — drop it (never re-derive it by + // guessing at name/position). A wrong `unique_keys` claim is a wrong + // query answer, not a missed optimization, so any difference at all + // drops the key. + if let Operator::NonASAP(NonASAPOp::Concat { + children, + discriminator_unique_key: Some(_), + }) = ¤t.operator + { + let after = children.first().map(|c| &c.schema); + if discriminator_branch_schema_before.as_ref() != after { + current = OperatorNode::non_asap_node(NonASAPOp::Concat { + children: children.clone(), + discriminator_unique_key: None, + })?; + } + } + + let current = apply_local_rules(current, memo)?; + + memo.insert(Rc::as_ptr(node), Rc::clone(¤t)); + Ok(current) +} + +type Memo = HashMap<*const OperatorNode, Rc>; + +/// Apply the local rewrite rules at `node` (whose inputs are already +/// canonical) until none matches. The rules chain: a `ROW_NUMBER()`- +/// partitioned top-k rewrites to a `Limit{Sort}`, which the heavy-hitter +/// rule may then promote to an `Aggregate([TopK])`; a `Filter` with several +/// subquery conjuncts sheds one per round. Each rule strictly simplifies the +/// node (one fewer idiom, or one fewer subquery reference), so the loop +/// terminates. +fn apply_local_rules( + mut current: Rc, + memo: &mut Memo, +) -> Result, SchemaDerivationError> { + loop { + let next = if let Some(next) = try_promote_additive_top_ranking(¤t)? { + next + } else if let Some(next) = try_lower_subquery_conjunct(¤t, memo)? { + next + } else { + break; + }; + current = next; + } + Ok(current) +} + +/// The direct **operator** inputs of a node — the relational skeleton only. +/// Operator nodes referenced from a scalar position (`ScalarSubquery`, +/// `Exists`, …) are not visited here: a subquery that the lowering rules +/// lift into a join is canonicalized at that point, and one they leave in +/// place (`NOT IN`, an `EXISTS` outside a `Filter` conjunct) stays as the +/// front end emitted it. +fn operator_children(op: &Operator) -> Vec<&Rc> { + op.children() +} + +/// Recognise an additive-ranked +/// `Limit { Sort { [Project] Aggregate([Count | Sum]) } }` and rewrite it to +/// the canonical heavy-hitter `Aggregate([TopK])` over the explicit inner +/// aggregate. Returns `None` when the shape does not match. +fn try_promote_additive_top_ranking( + node: &OperatorNode, +) -> Result>, SchemaDerivationError> { + // Limit k, no offset (an OFFSET means "not the top k"). + let Some(NonASAPOp::Limit { + n: Some(k), + offset: 0, + partition_by: limit_partition, + child, + }) = node.non_asap() + else { + return Ok(None); + }; + // A single ordering key on a column. + let Some(NonASAPOp::Sort { + keys, + partition_by, + child: sort_child, + }) = child.non_asap() + else { + return Ok(None); + }; + // A per-group `Limit` must agree with its `Sort`'s partition: the + // ranking's partition is what the outer `TopK` groups by. + if !limit_partition.is_empty() && limit_partition != partition_by { + return Ok(None); + } + let [SortKey { + expr: ScalarExpr::Column(sort_col), + ascending, + .. + }] = keys.as_slice() + else { + return Ok(None); + }; + + // The ordered relation is an `Aggregate`, optionally behind a passthrough + // projection (a bare-column SELECT list). Map the sort key through the + // projection to the aggregate's own output column. + let (agg_node, ranked_col) = match sort_child.non_asap() { + Some(NonASAPOp::Project { cols, child, .. }) => { + let Some(ProjectItem { + expr: ScalarExpr::Column(underlying), + .. + }) = cols.get(*sort_col) + else { + return Ok(None); + }; + (child, *underlying) + } + _ => (sort_child, *sort_col), + }; + + // Exactly one aggregate, ranked by *its* output column — the measure sits + // at index `by.len()` (after the group keys). A `PerEntity` reduction has + // no `by` to rank a measure against, so it is a non-match. + let Some(NonASAPOp::Aggregate { + reduction, + measures, + child: aggregate_child, + .. + }) = agg_node.non_asap() + else { + return Ok(None); + }; + let Reduction::Reduce(by) = reduction else { + return Ok(None); + }; + let [ranked_agg] = measures.as_slice() else { + return Ok(None); + }; + if ranked_col != by.len() { + return Ok(None); + } + // The heavy-hitter decision — descending, over a measure with a realised + // heavy-hitter sketch — is the shared rule both front ends consult (issue + // #38). An ascending additive-ranked limit (bottom-k) stays generic. + if !topk::Ranking::from_aggregate(ranked_agg).is_supported(!ascending) { + return Ok(None); + } + // A direct Sum is a stream of additive observation weights. A Sum over a + // derived child such as Rate/Increase still needs exact reset-aware + // values to rerank sketch candidates, and the post-ASAP IR has no + // candidate-sidecar + exact-rerank node, so that shape keeps Sort + Limit. + if matches!(ranked_agg, AggIntent::Sum { .. }) + && matches!( + aggregate_child.non_asap(), + Some(NonASAPOp::Aggregate { .. }) + ) + { + return Ok(None); + } + // Count ranks unit updates; a direct Sum ranks weighted updates. + let accuracy = match ranked_agg { + AggIntent::Count { accuracy } => accuracy.clone(), + AggIntent::Sum { .. } => AccuracyTarget::Exact, + _ => unreachable!("additive ranking gate admitted a non-additive measure"), + }; + + // Outer heavy-hitter `TopK`, grouped by the ranking's partition (empty for + // a global `ORDER BY … LIMIT k`; the `by` labels for a partitioned `topk + // by`), over the unchanged inner additive aggregate. + OperatorNode::non_asap_node(NonASAPOp::Aggregate { + reduction: Reduction::by(partition_by.to_vec()), + measures: vec![AggIntent::TopK { k: *k, accuracy }], + output_names: Vec::new(), + filters: vec![], + having: None, + child: Rc::clone(agg_node), + }) + .map(Some) +} + +// ROW_NUMBER filters retain the window output. Eliminating it without a +// consumer-aware rewrite drops a visible column and invalidates outer scopes. + +/// `Predicate(true)`: the unconditional join predicate the SQL front end +/// emits for an uncorrelated `EXISTS` and for a `CROSS JOIN`. +fn always_true() -> Predicate { + Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))) +} + +/// Whether `conjunct` is one this pass lowers to a semi/anti join. +fn is_join_conjunct(conjunct: &ScalarExpr) -> bool { + match conjunct { + ScalarExpr::Exists { .. } => true, + // An `IN` whose probe expression itself reads a scalar subquery is + // lowered only after that subquery has been joined in by + // `try_lower_scalar_subquery` (a `Join` predicate is not a place that + // rule looks). `NOT IN` is never lowered — see the module docs. + ScalarExpr::InSubquery { + expr, + negated: false, + .. + } => find_scalar_subquery(expr).is_none(), + _ => false, + } +} + +/// Lower one `[NOT] EXISTS (s)` / `x IN (s)` conjunct of a `Filter` to the +/// semi-/anti-join the SQL front end used to emit directly. The remaining +/// conjuncts stay in an outer `Filter` over the join: a semi/anti join's +/// output schema is the left's, so their column ids are unchanged. One +/// conjunct per call; the fixpoint loop picks up the next. +fn try_lower_subquery_conjunct( + node: &OperatorNode, + memo: &mut Memo, +) -> Result>, SchemaDerivationError> { + let Some(NonASAPOp::Filter { + pred: Predicate(pred), + child, + }) = node.non_asap() + else { + return Ok(None); + }; + let conjuncts = pred.conjuncts(); + let Some(idx) = conjuncts.iter().position(is_join_conjunct) else { + return Ok(None); + }; + let left_width = child.schema.fields.len(); + let (kind, subquery, join_pred) = match &conjuncts[idx] { + // Uncorrelated by construction (the IR's `Exists` carries no outer + // column references), so the join condition is unconditionally true. + ScalarExpr::Exists { subquery, negated } => { + let kind = if *negated { + JoinKind::Anti + } else { + JoinKind::Semi + }; + (kind, subquery, always_true()) + } + // `x = `, which sits right after the + // left's columns in the `left ++ right` scope the predicate resolves + // against. + ScalarExpr::InSubquery { expr, subquery, .. } => ( + JoinKind::Semi, + subquery, + Predicate(ScalarExpr::Compare { + left: expr.clone(), + op: CompareOpKind::Eq, + right: Box::new(ScalarExpr::Column(left_width)), + semantics: ExprSemantics::Sql, + }), + ), + _ => unreachable!("`is_join_conjunct` admitted a non-subquery conjunct"), + }; + let join = OperatorNode::non_asap_node(NonASAPOp::Join { + kind, + pred: join_pred, + left: Rc::clone(child), + right: canon(subquery, memo)?, + })?; + let mut rest: Vec = conjuncts + .iter() + .enumerate() + .filter(|(i, _)| *i != idx) + .map(|(_, c)| c.clone()) + .collect(); + let out = match rest.len() { + 0 => join, + 1 => filter(rest.remove(0), join)?, + _ => filter(ScalarExpr::BoolAnd(rest), join)?, + }; + Ok(Some(out)) +} + +/// Lower one scalar subquery read by a `Project` item or a `Filter` +/// predicate: the owner reads it through a cross join against the subquery, +/// whose single column is appended after the left's (`Column(|left|)`), and +/// every occurrence of that subquery node in the owner is replaced by that +/// column reference. One subquery node per call; the fixpoint loop handles +/// the rest, each getting its own cross join further out (so earlier column +/// ids are never shifted). For a `Filter` the output schema is restored to +/// the left's columns by a positional `Project` over the result. +/// +/// Not representable in the IR, and therefore not checked here: SQL raises +/// an error when a scalar subquery yields more than one row (the cross join +/// would duplicate the left's rows instead), and yields NULL when it yields +/// none (the cross join yields no rows instead). +fn filter( + pred: ScalarExpr, + child: Rc, +) -> Result, SchemaDerivationError> { + OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: Predicate(pred), + child, + }) +} + +/// The first `ScalarSubquery` node read by `expr` (pre-order over its scalar +/// children; referenced operator subgraphs are their own scope and are not +/// entered). +fn find_scalar_subquery(expr: &ScalarExpr) -> Option<&Rc> { + if let ScalarExpr::ScalarSubquery(node) = expr { + return Some(node); + } + expr.children().into_iter().find_map(find_scalar_subquery) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ir::operator_properties::WindowFuncKind; + use crate::ir::operator_properties::{ + ConcatDiscriminatorKey, GroupKeys, Source, WindowFrame, WindowFrameBound, + WindowFrameOffset, WindowFrameUnits, + }; + use crate::pre_asap::schema::{DataType, Field, Schema}; + + fn node(op: NonASAPOp) -> Rc { + Rc::new(OperatorNode::new(Operator::NonASAP(op)).expect("fixture derives a schema")) + } + + fn scan() -> Rc { + node(NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("value", DataType::Float64, false), + ], + 0, + vec![], + ), + }) + } + + fn aggregate( + reduction: Reduction, + agg: AggIntent, + child: Rc, + ) -> Rc { + node(NonASAPOp::Aggregate { + reduction, + measures: vec![agg], + output_names: vec![], + filters: vec![], + having: None, + child, + }) + } + + fn count() -> AggIntent { + AggIntent::Count { + accuracy: AccuracyTarget::Exact, + } + } + + /// `Aggregate{ by: [1], [Count] }` over the scan — output cols `[service, count]`. + fn count_by_service() -> Rc { + aggregate(Reduction::by(vec![1]), count(), scan()) + } + + fn key(col: usize, ascending: bool) -> Vec { + vec![SortKey { + expr: ScalarExpr::Column(col), + ascending, + nulls_first: false, + }] + } + + fn desc(col: usize) -> Vec { + key(col, false) + } + + fn limit(n: usize, offset: usize, child: Rc) -> Rc { + node(NonASAPOp::Limit { + n: Some(n), + offset, + partition_by: GroupKeys::none(), + child, + }) + } + + fn sort(keys: Vec, child: Rc) -> Rc { + node(NonASAPOp::Sort { + keys, + partition_by: GroupKeys::none(), + child, + }) + } + + fn passthrough_project(child: Rc) -> Rc { + node(NonASAPOp::Project { + cols: vec![ + ProjectItem { + alias: None, + expr: ScalarExpr::Column(0), + }, + ProjectItem { + alias: Some("c".into()), + expr: ScalarExpr::Column(1), + }, + ], + qualifier: None, + child, + }) + } + + fn concat( + children: Vec>, + key: Option, + ) -> Rc { + node(NonASAPOp::Concat { + children, + discriminator_unique_key: key, + }) + } + + fn measures(n: &OperatorNode) -> &[AggIntent] { + match n.non_asap() { + Some(NonASAPOp::Aggregate { measures, .. }) => measures, + _ => &[], + } + } + + fn is_topk_over_count(n: &OperatorNode) -> bool { + let Some(NonASAPOp::Aggregate { + measures, child, .. + }) = n.non_asap() + else { + return false; + }; + matches!(measures.as_slice(), [AggIntent::TopK { k: 5, .. }]) + && matches!(self::measures(child), [AggIntent::Count { .. }]) + } + + #[test] + fn promotes_count_ranked_limit_sort() { + // Limit 5 { Sort DESC by count-col (1) { Aggregate[Count] by [1] } }. + let q = limit(5, 0, sort(desc(1), count_by_service())); + assert!(is_topk_over_count(&canonicalize(q).unwrap())); + } + + #[test] + fn promotes_through_a_passthrough_projection() { + // …with a `SELECT service, count` projection between the Sort and the Agg. + let q = limit(5, 0, sort(desc(1), passthrough_project(count_by_service()))); + assert!(is_topk_over_count(&canonicalize(q).unwrap())); + } + + #[test] + fn promoted_topk_reuses_the_inner_aggregate_node() { + // The inner aggregate is untouched, so the rewrite shares it rather + // than copying it. + let agg = count_by_service(); + let out = canonicalize(limit(5, 0, sort(desc(1), Rc::clone(&agg)))).unwrap(); + let Some(NonASAPOp::Aggregate { child, .. }) = out.non_asap() else { + panic!("expected TopK aggregate"); + }; + assert!(Rc::ptr_eq(child, &agg)); + } + + #[test] + fn is_idempotent() { + let q = limit(5, 0, sort(desc(1), count_by_service())); + let once = canonicalize(q).unwrap(); + let twice = canonicalize(Rc::clone(&once)).unwrap(); + assert!(Rc::ptr_eq(&once, &twice), "canonicalize must be idempotent"); + } + + #[test] + fn untouched_dag_is_returned_pointer_equal() { + // Nothing here matches a rewrite: a Concat of two projections over + // one shared aggregate. The root (and everything under it) must come + // back as the same `Rc`. + let agg = count_by_service(); + let q = concat( + vec![ + passthrough_project(Rc::clone(&agg)), + passthrough_project(Rc::clone(&agg)), + ], + None, + ); + let out = canonicalize(Rc::clone(&q)).unwrap(); + assert!(Rc::ptr_eq(&out, &q)); + } + + #[test] + fn rewritten_shared_subtree_stays_shared() { + // One promotable sub-DAG referenced twice is rewritten once. + let branch = limit(5, 0, sort(desc(1), count_by_service())); + let q = concat(vec![Rc::clone(&branch), Rc::clone(&branch)], None); + let out = canonicalize(q).unwrap(); + let Some(NonASAPOp::Concat { children, .. }) = out.non_asap() else { + panic!("expected Concat"); + }; + assert!(is_topk_over_count(&children[0])); + assert!(Rc::ptr_eq(&children[0], &children[1])); + } + + // ── Concat's discriminator_unique_key vs. canonicalize (issue #228) ── + // + // `discriminator_unique_key`'s `ColumnId`s were resolved against the + // first branch's *pre-canonicalize* output schema. The key is dropped + // whenever that branch's schema actually changed, and survives untouched + // otherwise. Never guessed at. + + fn discriminator_key(n: &OperatorNode) -> &Option { + match n.non_asap() { + Some(NonASAPOp::Concat { + discriminator_unique_key, + .. + }) => discriminator_unique_key, + _ => panic!("expected Concat"), + } + } + + #[test] + fn concat_discriminator_key_survives_canonicalize_when_first_branch_is_unaffected() { + // A plain `Aggregate` first branch matches neither rewrite trigger, + // so its schema is identical before and after canonicalize. + let q = concat( + vec![count_by_service(), count_by_service()], + Some(ConcatDiscriminatorKey::new(0, vec![1])), + ); + let out = canonicalize(Rc::clone(&q)).unwrap(); + assert!( + discriminator_key(&out).is_some(), + "an untouched first branch's discriminator key must survive canonicalize" + ); + assert!(Rc::ptr_eq(&out, &q)); + } + + #[test] + fn concat_discriminator_key_is_dropped_when_first_branch_gets_rewritten() { + // The first branch is exactly the heavy-hitter promotion trigger, so + // canonicalize rewrites it to `Aggregate{TopK}`, whose own output is + // a single column, not the original two (`[service, count]`). A key + // resolved against the 2-column shape must not survive pointing at + // the new 1-column schema. + let promotable_branch = limit(5, 0, sort(desc(1), count_by_service())); + let q = concat( + vec![promotable_branch, count_by_service()], + Some(ConcatDiscriminatorKey::new(0, vec![1])), + ); + let out = canonicalize(q).unwrap(); + let Some(NonASAPOp::Concat { + children, + discriminator_unique_key, + }) = out.non_asap() + else { + panic!("expected Concat"); + }; + assert!( + is_topk_over_count(&children[0]), + "the first branch is still promoted normally" + ); + assert!( + discriminator_unique_key.is_none(), + "a stale discriminator key must be dropped, never silently kept wrong" + ); + assert!( + out.schema.unique_keys.is_empty(), + "the dropped key leaves the schema" + ); + } + + #[test] + fn does_not_promote_ascending_sort() { + // Ascending = bottom-k: the Top-K ranking rule rejects it (needs + // descending), so it stays a generic Sort+Limit (issue #38). + let q = limit(5, 0, sort(key(1, true), count_by_service())); + assert!(!is_topk_over_count(&canonicalize(q).unwrap())); + } + + #[test] + fn does_not_promote_with_offset() { + let q = limit(5, 2, sort(desc(1), count_by_service())); + assert!(!is_topk_over_count(&canonicalize(q).unwrap())); + } + + #[test] + fn does_not_promote_ranking_by_a_group_key() { + // DESC by col 0 (the `service` group key), not the count → not a + // frequency heavy-hitter. + let q = limit(5, 0, sort(desc(0), count_by_service())); + assert!(!is_topk_over_count(&canonicalize(q).unwrap())); + } + + #[test] + fn does_not_promote_when_limit_partition_disagrees_with_sort() { + // A per-group Limit partitioned differently from its Sort is not the + // top-k shape. + let q = node(NonASAPOp::Limit { + n: Some(5), + offset: 0, + partition_by: GroupKeys::by(vec![0]), + child: sort(desc(1), count_by_service()), + }); + assert!(!is_topk_over_count(&canonicalize(q).unwrap())); + } + + #[test] + fn promotes_sum_ranked_limit_sort_as_weighted_heavy_hitter() { + let sum = aggregate(Reduction::by(vec![1]), AggIntent::Sum { col: None }, scan()); + let out = canonicalize(limit(5, 0, sort(desc(1), sum))).unwrap(); + let Some(NonASAPOp::Aggregate { + measures, child, .. + }) = out.non_asap() + else { + panic!("expected weighted TopK aggregate"); + }; + assert!(matches!( + measures.as_slice(), + [AggIntent::TopK { k: 5, .. }] + )); + assert!(matches!(self::measures(child), [AggIntent::Sum { .. }])); + } + + #[test] + fn keeps_sum_over_counter_reduction_as_exact_value_ranking() { + for counter in [AggIntent::Rate, AggIntent::Increase] { + let derived = aggregate(Reduction::PerEntity, counter, scan()); + let sum = aggregate( + Reduction::by(vec![1]), + AggIntent::Sum { col: None }, + derived, + ); + let out = canonicalize(limit(5, 0, sort(desc(1), sum))).unwrap(); + let Some(NonASAPOp::Limit { child, .. }) = out.non_asap() else { + panic!("expected Limit, got {out:?}"); + }; + let Some(NonASAPOp::Sort { child, .. }) = child.non_asap() else { + panic!("expected Sort under the Limit"); + }; + let Some(NonASAPOp::Aggregate { + measures, child, .. + }) = child.non_asap() + else { + panic!("expected Aggregate under the Sort"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!(matches!( + child.non_asap(), + Some(NonASAPOp::Aggregate { .. }) + )); + } + } + + // ── ROW_NUMBER() partitioned top-k (issue #24) ────────────────────────── + + /// A scan with `[ts, service, region, value]`. + fn scan4() -> Rc { + node(NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("region", DataType::Utf8, false), + Field::plain("value", DataType::Float64, false), + ], + 0, + vec![], + ), + }) + } + + /// `Aggregate{ by: [1,2] (service, region), [agg] }` — output `[service, + /// region, ]` (3 cols), so a ROW_NUMBER over it appends `rn` at index 3. + fn grouped(agg: AggIntent) -> Rc { + aggregate(Reduction::by(vec![1, 2]), agg, scan4()) + } + + /// `ROW_NUMBER` ignores its frame clause; any concrete frame works. + fn rownumber_frame() -> WindowFrame { + WindowFrame { + units: WindowFrameUnits::Rows, + start_bound: WindowFrameBound::Preceding(WindowFrameOffset::Scalar(ScalarValue::Null)), + end_bound: WindowFrameBound::Following(WindowFrameOffset::Scalar(ScalarValue::Null)), + } + } + + /// `SQLWindowFunc{ RowNumber, PARTITION BY region(2), ORDER BY col(2) DESC } { agg }`. + fn rownumber_window(agg: Rc) -> Rc { + node(NonASAPOp::SQLWindowFunc { + func: WindowFuncKind::RowNumber, + args: vec![], + partition_by: GroupKeys::by(vec![2]), // region + order_by: vec![SortKey { + expr: ScalarExpr::Column(2), // the aggregate output column + ascending: false, + nulls_first: true, + }], + frame: Some(rownumber_frame()), + output_name: "rn".into(), + child: agg, + }) + } + + /// `Filter{ col <= 5 } { child }`. + fn filter_le_5(col: usize, child: Rc) -> Rc { + node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(col)), + op: CompareOpKind::Le, + right: Box::new(ScalarExpr::Literal(ScalarValue::Int64(5))), + semantics: ExprSemantics::Sql, + }), + child, + }) + } + + /// `Filter{ rn(3) <= 5 } { ROW_NUMBER window { agg } }`. + fn rownumber_topk(agg: Rc) -> Rc { + filter_le_5(3, rownumber_window(agg)) + } + + #[test] + fn rownumber_count_topk_becomes_a_partitioned_heavy_hitter() { + let original = rownumber_topk(grouped(count())); + let out = canonicalize(Rc::clone(&original)).unwrap(); + assert_eq!(out.schema, original.schema); + assert!( + matches!(out.non_asap(),Some(NonASAPOp::Filter { child,.. }) if matches!(child.non_asap(),Some(NonASAPOp::SQLWindowFunc { .. }))) + ); + assert_idempotent(&out); + } + + #[test] + fn rownumber_avg_topk_becomes_a_partitioned_sort_limit() { + let original = rownumber_topk(grouped(AggIntent::Avg { col: None })); + let out = canonicalize(Rc::clone(&original)).unwrap(); + assert_eq!(out.schema, original.schema); + assert!( + matches!(out.non_asap(),Some(NonASAPOp::Filter { child,.. }) if matches!(child.non_asap(),Some(NonASAPOp::SQLWindowFunc { .. }))) + ); + assert_idempotent(&out); + } + + #[test] + fn filter_on_a_non_rownumber_column_is_left_alone() { + // `WHERE service_len <= 5` (col 0, not the rn window column) must not + // be mistaken for a top-k. + let q = filter_le_5(0, rownumber_window(grouped(count()))); + let out = canonicalize(Rc::clone(&q)).unwrap(); + assert!(Rc::ptr_eq(&out, &q), "left as the same Filter"); + } + + // ── Subquery lowering ─────────────────────────────────────────────────── + + fn filter_of(pred: ScalarExpr, child: Rc) -> Rc { + node(NonASAPOp::Filter { + pred: Predicate(pred), + child, + }) + } + + /// `SELECT service FROM scan` — a one-column subquery. + /// Scalar reads retain cardinality/null semantics and a shared producer. + #[test] + fn scalar_subqueries_remain_explicit_and_shared() { + let sub = one_column_subquery(); + let root = node(NonASAPOp::Project { + child: scan(), + qualifier: None, + cols: vec![ + ProjectItem { + alias: Some("a".into()), + expr: ScalarExpr::ScalarSubquery(Rc::clone(&sub)), + }, + ProjectItem { + alias: Some("b".into()), + expr: ScalarExpr::ScalarSubquery(Rc::clone(&sub)), + }, + ], + }); + let out = canonicalize(root).unwrap(); + let NonASAPOp::Project { cols, child, .. } = out.expect_non_asap() else { + panic!() + }; + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); + for col in cols { + assert!(matches!(&col.expr,ScalarExpr::ScalarSubquery(node) if Rc::ptr_eq(node,&sub))); + } + assert!(out.schema.fields.iter().all(|f| f.nullable)); + assert_idempotent(&out); + } + + fn one_column_subquery() -> Rc { + node(NonASAPOp::Scan { + source: Source::Table { + table_ref: "sub".into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), + }) + } + + fn exists(subquery: Rc, negated: bool) -> ScalarExpr { + ScalarExpr::Exists { subquery, negated } + } + + fn in_subquery(expr: ScalarExpr, subquery: Rc, negated: bool) -> ScalarExpr { + ScalarExpr::InSubquery { + expr: Box::new(expr), + subquery, + negated, + } + } + + /// `value(2) > 1`. + fn value_gt_1() -> ScalarExpr { + ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(2)), + op: CompareOpKind::Gt, + right: Box::new(ScalarExpr::Literal(ScalarValue::Int64(1))), + semantics: ExprSemantics::Sql, + } + } + + fn literal_true() -> ScalarExpr { + ScalarExpr::Literal(ScalarValue::Boolean(true)) + } + + fn join_parts( + n: &OperatorNode, + ) -> (JoinKind, &ScalarExpr, &Rc, &Rc) { + match n.non_asap() { + Some(NonASAPOp::Join { + kind, + pred: Predicate(pred), + left, + right, + }) => (kind.clone(), pred, left, right), + _ => panic!("expected a Join, got {n:?}"), + } + } + + fn assert_idempotent(once: &Rc) { + let twice = canonicalize(Rc::clone(once)).unwrap(); + assert!(Rc::ptr_eq(once, &twice), "canonicalize must be idempotent"); + } + + #[test] + fn exists_filter_becomes_semi_join() { + let (left, sub) = (scan(), one_column_subquery()); + let q = filter_of(exists(Rc::clone(&sub), false), Rc::clone(&left)); + let out = canonicalize(q).unwrap(); + let (kind, pred, l, r) = join_parts(&out); + assert_eq!(kind, JoinKind::Semi); + assert_eq!(*pred, literal_true()); + assert!(Rc::ptr_eq(l, &left) && Rc::ptr_eq(r, &sub)); + assert_eq!( + out.schema.fields, left.schema.fields, + "a semi join outputs the left's columns" + ); + assert_idempotent(&out); + } + + #[test] + fn not_exists_becomes_anti_join() { + let (left, sub) = (scan(), one_column_subquery()); + let q = filter_of(exists(Rc::clone(&sub), true), Rc::clone(&left)); + let out = canonicalize(q).unwrap(); + let (kind, pred, l, r) = join_parts(&out); + assert_eq!(kind, JoinKind::Anti); + assert_eq!(*pred, literal_true()); + assert!(Rc::ptr_eq(l, &left) && Rc::ptr_eq(r, &sub)); + assert_idempotent(&out); + } + + #[test] + fn in_subquery_becomes_semi_join_on_the_subquery_column() { + // `WHERE service IN (SELECT service …)` over a 3-column left: the + // subquery's column is `Column(3)` in the `left ++ right` scope. + let (left, sub) = (scan(), one_column_subquery()); + let q = filter_of( + in_subquery(ScalarExpr::Column(1), Rc::clone(&sub), false), + Rc::clone(&left), + ); + let out = canonicalize(q).unwrap(); + let (kind, pred, l, r) = join_parts(&out); + assert_eq!(kind, JoinKind::Semi); + assert_eq!( + *pred, + ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(1)), + op: CompareOpKind::Eq, + right: Box::new(ScalarExpr::Column(3)), + semantics: ExprSemantics::Sql, + } + ); + assert!(Rc::ptr_eq(l, &left) && Rc::ptr_eq(r, &sub)); + assert_eq!( + out.schema.fields, left.schema.fields, + "a semi join outputs the left's columns" + ); + assert_idempotent(&out); + } + + #[test] + fn exists_with_other_conjuncts_keeps_an_outer_filter() { + // `WHERE value > 1 AND EXISTS (…)` → Filter{ value > 1 }{ Semi }. + let (left, sub) = (scan(), one_column_subquery()); + let q = filter_of( + ScalarExpr::BoolAnd(vec![value_gt_1(), exists(Rc::clone(&sub), false)]), + Rc::clone(&left), + ); + let out = canonicalize(q).unwrap(); + let Some(NonASAPOp::Filter { + pred: Predicate(pred), + child, + }) = out.non_asap() + else { + panic!("expected an outer Filter, got {out:?}"); + }; + assert_eq!(*pred, value_gt_1()); + let (kind, _, l, r) = join_parts(child); + assert_eq!(kind, JoinKind::Semi); + assert!(Rc::ptr_eq(l, &left) && Rc::ptr_eq(r, &sub)); + assert_idempotent(&out); + } + + #[test] + fn two_subquery_conjuncts_become_nested_joins() { + // `WHERE EXISTS (a) AND service NOT EXISTS (b) AND value > 1` sheds + // one conjunct per round: Filter{ value > 1 }{ Anti{ Semi{ l, a }, b } }. + let (left, a, b) = (scan(), one_column_subquery(), one_column_subquery()); + let q = filter_of( + ScalarExpr::BoolAnd(vec![ + exists(Rc::clone(&a), false), + exists(Rc::clone(&b), true), + value_gt_1(), + ]), + Rc::clone(&left), + ); + let out = canonicalize(q).unwrap(); + let Some(NonASAPOp::Filter { + pred: Predicate(pred), + child, + }) = out.non_asap() + else { + panic!("expected an outer Filter, got {out:?}"); + }; + assert_eq!(*pred, value_gt_1()); + let (kind, _, inner, r) = join_parts(child); + assert_eq!(kind, JoinKind::Anti); + assert!(Rc::ptr_eq(r, &b)); + let (kind, _, l, r) = join_parts(inner); + assert_eq!(kind, JoinKind::Semi); + assert!(Rc::ptr_eq(l, &left) && Rc::ptr_eq(r, &a)); + assert_idempotent(&out); + } + + #[test] + fn not_in_subquery_is_left_alone() { + let q = filter_of( + in_subquery(ScalarExpr::Column(1), one_column_subquery(), true), + scan(), + ); + let out = canonicalize(Rc::clone(&q)).unwrap(); + assert!(Rc::ptr_eq(&out, &q), "NOT IN keeps its Filter"); + assert_idempotent(&out); + } + + #[test] + fn lifted_subquery_is_canonicalized() { + // The subquery is itself a promotable heavy-hitter; once lifted into + // the join it is canonical, so a second pass finds nothing to do. + let sub = limit(5, 0, sort(desc(1), count_by_service())); + let q = filter_of(exists(sub, false), scan()); + let out = canonicalize(q).unwrap(); + let (_, _, _, r) = join_parts(&out); + assert!(is_topk_over_count(r)); + assert_idempotent(&out); + } +} diff --git a/crates/types/src/ir/cse.rs b/crates/types/src/ir/cse.rs new file mode 100644 index 000000000..656b3a965 --- /dev/null +++ b/crates/types/src/ir/cse.rs @@ -0,0 +1,786 @@ +//! Structural common-subexpression elimination over the unified operator IR: +//! bottom-up hash-consing of [`OperatorNode`] DAGs across a workload's roots. +//! +//! CSE only runs on already-bound, already-canonicalized plans — structural +//! matching is meaningless before canonicalization has converged +//! semantically-equivalent queries onto one shape. [`share_common_subdags`] +//! is the single entry point, run once per workload batch (a batch of one +//! still deduplicates a query's own repeated sub-DAGs, see below). +//! +//! ## Algorithm: classic hash-consing / value-numbering +//! +//! Bottom-up: every child is interned before its parent, so two parents whose +//! children were independently deduplicated down to the same `Rc`s are +//! structurally identical iff their own fields also match, without re-walking +//! the sub-DAGs. "Child" means everything [`OperatorNode::children`] returns: +//! the operator inputs *and* the operator nodes a scalar expression reads +//! (`PromqlScalarFromVector`, `ScalarSubquery`, `Exists`, `InSubquery`), so a +//! vector read by `scalar(v)` in two queries is shared like any other input. +//! The scalar expressions themselves stay opaque data on their owning node. +//! +//! ## Correctness: hash is a filter, `PartialEq` is the decision +//! +//! This is the one non-negotiable rule. A **false positive** here — two +//! sub-DAGs wrongly judged shareable — is a wrong query answer, not a missed +//! optimization: two different queries would read each other's data. +//! [`structural_hash`] (SipHash over a canonical serialization, no +//! collision-freedom guarantee) may only narrow the candidate set within one +//! bucket; the typed equality check on that bucket ([`same_node`]) is what +//! actually decides sharing, every time, no exceptions for "the hash probably +//! didn't collide." Equality is intentionally conservative: it recognizes +//! *exact* structural matches only, never "a stricter-accuracy summary could +//! also answer a looser request" (that subsumption question belongs to the +//! ASAP matcher, not here). +//! +//! ## Legality +//! +//! Structural equality is necessary but not sufficient. A non-ASAP node is +//! only ever *returned* as a match for another when its output has a provable +//! unique key (`Schema::has_unique_key()`): a producer's output can only be +//! shared across consumers when its row identity is stable across reads, so +//! an ungrouped aggregate, a `without(..)` grouping, a `Concat`/`SetOp` that +//! drops its keys, … is always inserted fresh even when it is structurally +//! identical to something already interned. An ASAP node (summary state and +//! its evaluations) has no such gate: equal operator, schema and guarantee make +//! it shareable, exactly as post-ASAP sharing decided before this IR. +//! +//! ## Single-query CSE falls out for free +//! +//! A repeated sub-expression within *one* query (the same grouped aggregate on +//! both `BinaryOp` branches) is deduplicated by the same bottom-up interning — +//! a workload of size one still interns bottom-up within that one DAG. + +use std::collections::hash_map::DefaultHasher; +use std::collections::HashMap; +use std::hash::{Hash, Hasher}; +use std::rc::Rc; + +use super::node::{Operator, OperatorNode}; +use super::non_asap::NonASAPOp; +use crate::pre_asap::schema::Schema; + +/// [`structural_hash`]'s memoization cache: an already-hashed node's `Rc` +/// pointer to its hash. A fresh cache is always correct; what matters is +/// letting it persist across every node of one bottom-up pass rather than +/// starting a new one per call. The caller must keep every cached node alive +/// for the cache's lifetime, or a reused address would alias a stale entry. +pub type HashCache = HashMap<*const OperatorNode, u64>; + +/// A constant stand-in for every child position. Substituting it before +/// serializing or comparing a node leaves exactly the node's own fields. +fn placeholder() -> Rc { + Rc::new(OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Values { + rows: vec![], + schema: Schema::lifted(vec![], None), + }), + Schema::lifted(vec![], None), + )) +} + +/// The operator with every child — operator inputs and the operator nodes +/// referenced from its scalar expressions alike — replaced by +/// [`placeholder`]. What remains is the node's own data: variant tag, scalar +/// expressions (with their operator references blanked), parameters. +fn own_fields(node: &OperatorNode) -> Operator { + let placeholder = placeholder(); + node.operator.map_children(|_| Rc::clone(&placeholder)) +} + +/// Coarse structural hash used only to bucket [`InternTable::intern`]'s +/// candidate search — never the sharing decision ([`same_node`] is). +/// +/// `OperatorNode` carries `f64`s (`ScalarValue::Float64`, quantile targets, +/// `ResultGuarantee` bounds, …), so it cannot derive `std::hash::Hash`. The +/// hash is SipHash over two parts: +/// +/// 1. the canonical JSON of [`own_fields`] plus `result_kind`, `schema`, +/// `guarantee` and `timing` — every field `PartialEq` compares except the +/// children. A scalar expression is serialized as data with each operator +/// node it reads replaced by a constant placeholder, so a reference to an +/// interned sub-DAG contributes nothing of its own here; +/// 2. for every child in [`OperatorNode::children`] order (operator inputs, +/// then scalar-referenced nodes), the child's own `structural_hash`, +/// memoized in `cache` by `Rc` pointer identity. +/// +/// Part 2 is what makes equal sub-DAGs hash equal whether they are reached +/// through an operator input or through a `scalar(v)`, and what keeps the +/// pass linear: a node is generally a DAG, and re-serializing a shared +/// descendant once per parent would cost `O(sub-DAG)` per node instead of +/// `O(1)` beyond the children's already-known hashes. A non-finite `f64` +/// serializes as `null`, merely widening one (still equality-checked) bucket. +pub fn structural_hash(node: &OperatorNode, cache: &mut HashCache) -> u64 { + fn child_hash(child: &Rc, cache: &mut HashCache) -> u64 { + let ptr = Rc::as_ptr(child); + if let Some(&h) = cache.get(&ptr) { + return h; + } + let h = structural_hash(child, cache); + cache.insert(ptr, h); + h + } + + let mut hasher = DefaultHasher::new(); + let own = ( + own_fields(node), + node.result_kind, + &node.schema, + &node.guarantee, + node.timing, + ); + serde_json::to_string(&own) + .unwrap_or_default() + .hash(&mut hasher); + for child in node.children() { + child_hash(child, cache).hash(&mut hasher); + } + hasher.finish() +} + +/// Numeric `PartialEq` alone conflates signed zeros. The serialized check is +/// additional evidence, never a replacement for typed equality (JSON maps +/// non-finite floats to `null`). Used for the guarantee, whose bounds are +/// floats a shared node must preserve bit-for-bit. +fn same_value(left: &T, right: &T) -> bool { + left == right + && match (serde_json::to_string(left), serde_json::to_string(right)) { + (Ok(left), Ok(right)) => left == right, + _ => false, + } +} + +/// Memo of child-pair comparisons already decided by [`same_node`], keyed by +/// pointer pair. Only interned (table-owned, hence alive) nodes are keys. +type EqMemo = HashMap<(*const OperatorNode, *const OperatorNode), bool>; + +/// The sharing decision: typed equality of two nodes. +/// +/// `OperatorNode`'s derived `PartialEq` would recurse into children by value +/// even when both sides hold the same `Rc` (`OperatorNode` is not `Eq`, so +/// `Rc` gets no pointer shortcut), expanding a shared diamond once per path. +/// Children are therefore compared by pointer first; only when the pointers +/// differ (an equal child that was not legal to share) are the values +/// compared, memoized per pair so a diamond is still walked once. +fn same_node(left: &OperatorNode, right: &OperatorNode, memo: &mut EqMemo) -> bool { + let (lc, rc) = (left.children(), right.children()); + if lc.len() != rc.len() { + return false; + } + let children_equal = lc.iter().zip(&rc).all(|(a, b)| { + if Rc::ptr_eq(a, b) { + return true; + } + let key = (Rc::as_ptr(a), Rc::as_ptr(b)); + if let Some(&eq) = memo.get(&key) { + return eq; + } + let eq = same_node(a, b, memo); + memo.insert(key, eq); + eq + }); + children_equal + && left.result_kind == right.result_kind + && left.schema == right.schema + && left.timing == right.timing + && same_value(&left.guarantee, &right.guarantee) + && same_value(&own_fields(left), &own_fields(right)) +} + +/// Bottom-up hash-consing table: structurally-equal, sharing-legal nodes +/// collapse onto one `Rc`. +/// +/// `buckets` is keyed by [`structural_hash`] — a coarse candidate filter +/// only. Every entry within one bucket is a full node kept around for the +/// [`same_node`] comparison that actually decides a match; a hash collision +/// between structurally different nodes just means a harmless linear scan of +/// a few extra candidates. +struct InternTable { + buckets: HashMap>>, + /// Persisted for the table's whole lifetime so hashing is `O(1)` per node + /// beyond its children; every cached node is owned by `buckets`. + hash_cache: HashCache, + eq_memo: EqMemo, +} + +impl InternTable { + fn new() -> Self { + Self { + buckets: HashMap::new(), + hash_cache: HashMap::new(), + eq_memo: HashMap::new(), + } + } + + /// Intern one node whose children are already interned: look it up by + /// [`structural_hash`], confirm with [`same_node`], and — only when + /// sharing is legal (module doc, "Legality") — return the existing `Rc` + /// instead of allocating a new one. + fn intern(&mut self, node: OperatorNode) -> Rc { + let hash = structural_hash(&node, &mut self.hash_cache); + // A node that is not legal to share is never *returned* as a match + // for something else; it still occupies a fresh slot in the bucket + // (harmless: later scans require legality of the new node too). + let reusable = node.is_asap() || node.schema.has_unique_key(); + let bucket = self.buckets.entry(hash).or_default(); + if reusable { + if let Some(existing) = bucket + .iter() + .find(|candidate| same_node(candidate, &node, &mut self.eq_memo)) + { + return Rc::clone(existing); + } + } + let rc = Rc::new(node); + bucket.push(Rc::clone(&rc)); + rc + } +} + +/// Count of *unique* nodes reachable from `root` (pointer identity, +/// following [`OperatorNode::children`]): the real size of the DAG, not a +/// tree-walk count that re-counts a shared descendant once per parent. +pub fn dag_node_count(root: &Rc) -> usize { + OperatorNode::reachable(root).len() +} + +/// Input pointer → (input `Rc`, interned result). The input `Rc` is retained +/// so its address cannot be freed and reused by a fresh allocation while the +/// memo still maps it. +type Visited = HashMap<*const OperatorNode, (Rc, Rc)>; + +/// Intern `node`'s children (recursively), then `node` itself. The rebuilt +/// node keeps `node`'s retained schema, result kind, guarantee and timing: +/// every child is replaced by an equal node, so each derived property stays +/// valid, and the result is `PartialEq`-equal to the input. +fn intern_bottom_up( + table: &mut InternTable, + visited: &mut Visited, + node: &Rc, +) -> Rc { + if let Some((_, interned)) = visited.get(&Rc::as_ptr(node)) { + return Rc::clone(interned); + } + let operator = node + .operator + .map_children(|child| intern_bottom_up(table, visited, child)); + let rebuilt = OperatorNode { + operator, + result_kind: node.result_kind, + schema: node.schema.clone(), + guarantee: node.guarantee.clone(), + timing: node.timing, + }; + let interned = table.intern(rebuilt); + visited.insert(Rc::as_ptr(node), (Rc::clone(node), Rc::clone(&interned))); + interned +} + +/// Share structurally-identical, sharing-legal sub-DAGs across a workload's +/// roots (or within one root). Every root's *value* is unchanged +/// (`PartialEq`-equal to its input) — only its internal `Rc` structure may +/// now alias another root's, or another part of its own DAG. A node already +/// reached through two paths is visited once. +/// +/// `Id` is caller-chosen — a workload entry's key, an index, a query name. +pub fn share_common_subdags(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> { + let mut table = InternTable::new(); + let mut visited = Visited::new(); + roots + .into_iter() + .map(|(id, root)| (id, intern_bottom_up(&mut table, &mut visited, &root))) + .collect() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ir::asap::ASAPOp; + use crate::ir::operator_properties::{BinaryOpKind, GroupKeys, Reduction, Source}; + use crate::ir::BinaryOperator; + use crate::ir::ScalarExpr; + use crate::post_asap::guarantee::ResultGuarantee; + use crate::post_asap::sketch::{ + GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate, + }; + use crate::pre_asap::agg_intent::AggIntent; + use crate::pre_asap::expr_ir::{ColumnRef, CompareOpKind}; + use crate::pre_asap::schema::{DataType, Field, FieldDataType, Schema}; + + use crate::types::AccuracyTarget; + + fn node(op: NonASAPOp) -> Rc { + OperatorNode::non_asap_node(op).unwrap() + } + + /// `[ts, service, value, latency]`, no unique key. + fn scan() -> Rc { + node(NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("value", DataType::Float64, false), + Field::plain("latency", DataType::Float64, false), + ], + 0, + vec![], + ), + }) + } + + fn quantile_agg(by: Vec, col: Option, q: f64) -> Rc { + node(NonASAPOp::Aggregate { + reduction: Reduction::by(by), + measures: vec![AggIntent::Quantile { + col, + q, + accuracy: AccuracyTarget::Exact, + }], + output_names: vec![], + filters: vec![], + having: None, + child: scan(), + }) + } + + fn compare(lhs: Rc, rhs: Rc) -> Rc { + node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: BinaryOpKind::Compare(CompareOpKind::Eq), + vector_match: None, + }, + return_bool: false, + lhs, + rhs, + }) + } + + fn two_roots(a: Rc, b: Rc) -> (Rc, Rc) { + let shared = share_common_subdags(vec![("a", a), ("b", b)]); + let [(_, ra), (_, rb)] = shared.as_slice() else { + panic!("expected 2 roots"); + }; + (Rc::clone(ra), Rc::clone(rb)) + } + + #[test] + fn distinct_column_quantiles_do_not_merge() { + // Grouped (unique key present) so only the differing `col` blocks it. + let (ra, rb) = two_roots( + quantile_agg(vec![1], Some(2), 0.5), + quantile_agg(vec![1], Some(3), 0.5), + ); + assert!(!Rc::ptr_eq(&ra, &rb)); + assert_ne!(ra, rb); + } + + #[test] + fn no_unique_keys_means_no_merge_even_when_structurally_identical() { + let a = quantile_agg(vec![], Some(2), 0.9); + let b = quantile_agg(vec![], Some(2), 0.9); + assert_eq!(a, b, "fixture sanity: structurally equal"); + assert!( + !a.schema.has_unique_key(), + "fixture sanity: a global aggregate has no provable unique key" + ); + let (ra, rb) = two_roots(a, b); + assert!( + !Rc::ptr_eq(&ra, &rb), + "no unique key ⇒ never hoisted, even for an identical structural match" + ); + } + + #[test] + fn median_and_explicit_half_percentile_merge() { + // Two spellings that lower to the identical grouped `Quantile { q: 0.5 }`. + let (m, p) = two_roots( + quantile_agg(vec![1], Some(2), 0.5), + quantile_agg(vec![1], Some(2), 0.5), + ); + assert!(Rc::ptr_eq(&m, &p)); + } + + #[test] + fn single_query_shares_its_own_repeated_subtree() { + // One root with the same grouped aggregate on both branches, built as + // two separately-allocated sub-DAGs (no sharing yet). + let root = compare( + quantile_agg(vec![1], Some(2), 0.5), + quantile_agg(vec![1], Some(2), 0.5), + ); + let shared = share_common_subdags(vec![("q", root)]); + let [(_, root)] = shared.as_slice() else { + panic!("expected 1 root"); + }; + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = root.non_asap() else { + panic!("expected BinaryOp root, got {root:?}"); + }; + assert!(Rc::ptr_eq(lhs, rhs)); + } + + #[test] + fn shared_root_value_is_unchanged() { + let a = quantile_agg(vec![1], Some(2), 0.5) + .as_ref() + .clone() + .with_guarantee(Some(ResultGuarantee::exact("fixture"))); + let before = Rc::new(a); + let (ra, _) = two_roots(Rc::clone(&before), Rc::clone(&before)); + assert_eq!(ra.as_ref(), before.as_ref()); + assert!( + ra.guarantee.is_some(), + "retained properties survive the rebuild" + ); + } + + // ── scalar-referenced sub-DAGs ────────────────────────────────────── + + /// `vector(scalar(sum by (service) (up)))`. + fn scalar_of_vector() -> Rc { + let sum_up = node(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![1]), + measures: vec![AggIntent::Sum { col: Some(2) }], + output_names: vec![], + filters: vec![], + having: None, + child: scan(), + }); + assert!(sum_up.schema.has_unique_key(), "fixture sanity"); + node(NonASAPOp::PromqlVectorFromScalar( + ScalarExpr::PromqlScalarFromVector(sum_up), + )) + } + + fn bridged_vector(root: &Rc) -> &Rc { + match root.non_asap() { + Some(NonASAPOp::PromqlVectorFromScalar(ScalarExpr::PromqlScalarFromVector(v))) => v, + other => panic!("expected vector(scalar(v)), got {other:?}"), + } + } + + #[test] + fn scalar_referenced_vector_is_shared_across_queries() { + let (ra, rb) = two_roots(scalar_of_vector(), scalar_of_vector()); + assert!( + Rc::ptr_eq(bridged_vector(&ra), bridged_vector(&rb)), + "the vector read by scalar(v) is a child and must be interned" + ); + assert!( + !Rc::ptr_eq(&ra, &rb), + "the scalar bridge itself has no unique key and stays separate" + ); + } + + #[test] + fn structural_hash_sees_through_a_scalar_reference() { + // Two equal bridges must hash equal whether or not their referenced + // vector is the same Rc — the reference contributes the vector's + // memoized hash, not its identity. + let a = scalar_of_vector(); + let b = scalar_of_vector(); + let mut cache = HashMap::new(); + assert_eq!( + structural_hash(&a, &mut cache), + structural_hash(&b, &mut cache) + ); + assert_eq!( + cache.len(), + 4, + "aggregate + scan cached once per root: {cache:?}" + ); + let other = node(NonASAPOp::PromqlVectorFromScalar( + ScalarExpr::PromqlScalarFromVector(quantile_agg(vec![1], Some(2), 0.5)), + )); + assert_ne!( + structural_hash(&a, &mut cache), + structural_hash(&other, &mut cache) + ); + } + + // ── ASAP nodes ────────────────────────────────────────────────────── + + fn summary_agg(alpha: f64, guarantee: Option) -> Rc { + let family = FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::DDSketch, SketchParams::DDSketch { alpha }), + GroupingStrategy::default(), + ); + let schema = Schema::lifted(vec![Field::new("state", family.clone(), false)], None); + assert!(!schema.has_unique_key(), "fixture sanity"); + Rc::new( + OperatorNode::with_schema( + Operator::ASAP(ASAPOp::SummaryAgg { + child: scan(), + family, + input: SummaryUpdate::column(ColumnRef::SampleValue), + reduction: Reduction::PerEntity, + grouping: GroupingStrategy::default(), + filter: None, + }), + schema, + ) + .with_guarantee(guarantee), + ) + } + + #[test] + fn asap_nodes_share_without_a_unique_key() { + let exact = || Some(ResultGuarantee::exact("fixture")); + let (ra, rb) = two_roots(summary_agg(0.01, exact()), summary_agg(0.01, exact())); + assert!(Rc::ptr_eq(&ra, &rb)); + assert!(ra.guarantee.is_some()); + } + + #[test] + fn asap_nodes_with_distinct_parameters_or_guarantees_are_not_shared() { + let exact = || Some(ResultGuarantee::exact("fixture")); + let (ra, rb) = two_roots(summary_agg(0.01, exact()), summary_agg(0.001, exact())); + assert!(!Rc::ptr_eq(&ra, &rb), "different sketch parameters"); + let (ra, rb) = two_roots(summary_agg(0.01, exact()), summary_agg(0.01, None)); + assert!( + !Rc::ptr_eq(&ra, &rb), + "an unknown guarantee never borrows an exact one" + ); + assert!(rb.guarantee.is_none()); + } + + #[test] + fn evaluations_share_their_producer_but_not_each_other() { + use crate::post_asap::sketch::SketchStatistic; + let evaluation = |q: f64| { + Rc::new(OperatorNode::with_schema( + Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input: summary_agg(0.01, None), + query: SketchStatistic::Quantile { q }, + }), + Schema::lifted( + vec![Field::plain("quantile", DataType::Float64, false)], + None, + ), + )) + }; + let (p95, p99) = two_roots(evaluation(0.95), evaluation(0.99)); + let producer = |n: &Rc| Rc::clone(n.children()[0]); + assert!(!Rc::ptr_eq(&p95, &p99)); + assert!(Rc::ptr_eq(&producer(&p95), &producer(&p99))); + } + + // ── structural_hash (DAG-aware memoization) ───────────────────────── + + #[test] + fn structural_hash_is_stable_across_cache_states() { + let agg = quantile_agg(vec![1], Some(2), 0.5); + let mut cold = HashMap::new(); + let mut warm = HashMap::new(); + structural_hash(&scan(), &mut warm); + assert_eq!( + structural_hash(&agg, &mut cold), + structural_hash(&agg, &mut warm), + "hash must be independent of unrelated cache state" + ); + } + + #[test] + fn structural_hash_of_an_internally_shared_tree_matches_the_unshared_equivalent() { + let agg = quantile_agg(vec![1], Some(2), 0.5); + let shared_root = compare(Rc::clone(&agg), Rc::clone(&agg)); + let unshared_root = compare( + quantile_agg(vec![1], Some(2), 0.5), + quantile_agg(vec![1], Some(2), 0.5), + ); + assert_eq!( + structural_hash(&shared_root, &mut HashMap::new()), + structural_hash(&unshared_root, &mut HashMap::new()), + ); + } + + #[test] + fn structural_hash_memoizes_a_shared_descendant_exactly_once() { + let agg = quantile_agg(vec![1], Some(2), 0.5); + let root = compare(Rc::clone(&agg), Rc::clone(&agg)); + let mut cache = HashMap::new(); + structural_hash(&root, &mut cache); + assert_eq!( + cache.len(), + 2, + "one entry per unique node in the shared branch (Aggregate + Scan): {cache:?}" + ); + } + + // ── dag_node_count ─────────────────────────────────────────────────── + + #[test] + fn dag_node_count_is_the_naive_count_when_nothing_is_shared() { + assert_eq!(dag_node_count(&scan()), 1); + assert_eq!(dag_node_count(&quantile_agg(vec![1], Some(2), 0.5)), 2); + assert_eq!( + dag_node_count(&scalar_of_vector()), + 3, + "follows scalar references" + ); + } + + #[test] + fn dag_node_count_deduplicates_an_internally_shared_subtree() { + let root = compare( + quantile_agg(vec![1], Some(2), 0.5), + quantile_agg(vec![1], Some(2), 0.5), + ); + assert_eq!( + dag_node_count(&root), + 5, + "fixture sanity: nothing shared yet" + ); + let shared = share_common_subdags(vec![("q", root)]); + let [(_, root)] = shared.as_slice() else { + panic!("expected 1 root"); + }; + assert_eq!( + dag_node_count(root), + 3, + "BinaryOp + one Aggregate + its Scan" + ); + } + + #[test] + fn dag_node_count_deduplicates_across_two_workload_roots() { + let (ra, rb) = two_roots( + quantile_agg(vec![1], Some(2), 0.5), + quantile_agg(vec![1], Some(2), 0.5), + ); + assert!(Rc::ptr_eq(&ra, &rb), "fixture sanity: the two roots merged"); + assert_eq!(dag_node_count(&ra), 2); + assert_eq!(dag_node_count(&rb), 2); + } + + #[test] + fn dedup_gates_sharing_the_same_as_aggregate() { + // `Dedup { cols }` adds `cols` as a unique key, so two identical + // `Dedup`s merge even though their keyless `Scan`s could not. + let dedup = || { + node(NonASAPOp::Dedup { + cols: vec![1], + child: scan(), + }) + }; + let (ra, rb) = two_roots(dedup(), dedup()); + assert!(Rc::ptr_eq(&ra, &rb)); + } + + #[test] + fn group_keys_gate_still_prevented_when_partition_by_without_used() { + let without_agg = || { + node(NonASAPOp::Aggregate { + reduction: Reduction::Reduce(GroupKeys::without(vec![0])), + measures: vec![AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }], + output_names: vec![], + filters: vec![], + having: None, + child: scan(), + }) + }; + let a = without_agg(); + assert!(!a.schema.has_unique_key()); + let (ra, rb) = two_roots(a, without_agg()); + assert!(!Rc::ptr_eq(&ra, &rb)); + } + + #[test] + fn already_shared_nodes_are_visited_once() { + // A diamond already present in the input stays one node and is not + // re-interned per path. + let agg = quantile_agg(vec![1], Some(2), 0.5); + let root = compare(Rc::clone(&agg), Rc::clone(&agg)); + let shared = share_common_subdags(vec![("q", root)]); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = shared[0].1.non_asap() else { + panic!("expected BinaryOp root"); + }; + assert!(Rc::ptr_eq(lhs, rhs)); + assert_eq!(dag_node_count(&shared[0].1), 3); + } + + // Comparing a shareable node whose equal-but-unshareable children form a + // deep diamond must not expand the diamond once per path. The timeout is + // a coarse runaway guard, not a performance SLA. + #[test] + fn shared_diamond_does_not_expand_during_comparison() { + let (done, completion) = std::sync::mpsc::channel(); + let worker = std::thread::spawn(move || { + fn keyed_diamond() -> Rc { + // BinaryOp over a keyless scan has no unique key at any level, + // so none of the 24 levels is shareable; the `Dedup` on top is. + let mut current = scan(); + for _ in 0..24 { + current = compare(Rc::clone(¤t), current); + } + node(NonASAPOp::Dedup { + cols: vec![1], + child: current, + }) + } + let (ra, rb) = two_roots(keyed_diamond(), keyed_diamond()); + assert!(Rc::ptr_eq(&ra, &rb)); + done.send(()).unwrap(); + }); + completion + .recv_timeout(std::time::Duration::from_secs(5)) + .expect("comparison expanded the shared DAG"); + worker.join().unwrap(); + } + + /// A keyed (hence shareable) projection emitting the literal `value`. + fn keyed_literal(value: f64) -> Rc { + let keyed = node(NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + ], + 0, + vec![vec![1]], + ), + }); + node(NonASAPOp::Project { + cols: vec![ + crate::ir::ProjectItem { + alias: None, + expr: ScalarExpr::Column(1), + }, + crate::ir::ProjectItem { + alias: Some("v".into()), + expr: ScalarExpr::literal_f64(value), + }, + ], + qualifier: None, + child: keyed, + }) + } + + /// Sharing preserves IEEE signed zero, and JSON's `null` encoding of + /// non-finite floats never becomes the equality decision. + #[test] + fn signed_zero_and_nonfinite_values_remain_distinct() { + assert!( + keyed_literal(0.0).schema.has_unique_key(), + "fixture is shareable" + ); + for (a, b) in [ + (0.0, -0.0), + (-0.0, 0.0), + (f64::INFINITY, f64::NEG_INFINITY), + (f64::NAN, f64::NAN), + ] { + let (ra, rb) = two_roots(keyed_literal(a), keyed_literal(b)); + assert!(!Rc::ptr_eq(&ra, &rb), "{a} and {b} must not be shared"); + } + let (ra, rb) = two_roots(keyed_literal(f64::INFINITY), keyed_literal(f64::INFINITY)); + assert!(Rc::ptr_eq(&ra, &rb)); + } +} diff --git a/crates/types/src/ir/error.rs b/crates/types/src/ir/error.rs new file mode 100644 index 000000000..4b6849965 --- /dev/null +++ b/crates/types/src/ir/error.rs @@ -0,0 +1,19 @@ +//! Typed failures from operator/scalar schema and type derivation. +//! +//! [`SchemaDerivationError`] distinguishes invalid scalar signatures, out-of-range +//! grouping columns, empty concatenations, and invalid sample columns. +//! Structural DAG and execution-timing validation have separate error types. +use crate::pre_asap::ColumnId; +use thiserror::Error; +/// Errors from schema and type derivation over an operator DAG. +#[derive(Debug, Error)] +pub enum SchemaDerivationError { + #[error("invalid scalar function signature: {0}")] + InvalidScalarSignature(String), + #[error("by-column id {0} out of range (input has {1} columns)")] + InvalidGroupByColumn(ColumnId, usize), + #[error("Concat requires at least one child")] + EmptyConcat, + #[error("invalid per-series sample column: {0}")] + InvalidSampleColumn(String), +} diff --git a/crates/types/src/ir/export.rs b/crates/types/src/ir/export.rs new file mode 100644 index 000000000..6d8626aff --- /dev/null +++ b/crates/types/src/ir/export.rs @@ -0,0 +1,1564 @@ +//! Post-ASAP DAG export (wire version 7) over the unified operator IR. +//! +//! One exported node per operator — relational operators included — with +//! children as edges and no embedded sub-DAGs. The input must already be +//! timed ([`super::timing::apply_lifecycle_timings`]); export reads each +//! node's timing and does not re-run data-state validation. + +use std::collections::{BTreeMap, HashMap, HashSet}; +use std::rc::Rc; +use std::time::Duration; + +use serde::{Deserialize, Serialize}; +use thiserror::Error; + +use super::asap::ASAPOp; +use super::node::{Operator, OperatorNode}; +use super::non_asap::{BinaryOperator, NonASAPOp, TimeRangeKind}; +use super::scalar::{ExprSemantics, Predicate, ProjectItem, ScalarExpr, SortKey}; +use super::timing::data_state; +use crate::ir::operator_properties::{ + ConcatDiscriminatorKey, GroupKeys, InfoMatcher, JoinKind, Reduction, RelationalSetOpKind, + SampleKind, Source, TimeShift, WindowFrame, WindowFuncKind, +}; +use crate::post_asap::execution_data_state::{ + ExecutionDataState, ExecutionDataStateError, ExecutionTiming, +}; +use crate::post_asap::guarantee::ResultGuarantee; +use crate::post_asap::maintained_population::{MaintainedPopulation, PopulationStatistic}; +use crate::post_asap::sketch::{GroupingStrategy, SketchStatistic, SummaryUpdate}; +use crate::pre_asap::agg_intent::AggIntent; +use crate::pre_asap::expr_ir::{ArithmeticOpKind, CompareOpKind, ScalarValue}; +use crate::pre_asap::schema::{ColumnId, DataType, FieldDataType, Schema}; + +pub const POST_ASAP_DAG_WIRE_VERSION: u32 = 7; + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +pub enum EdgeRole { + Input, + Left, + Right, + /// The consumer reads the producer from inside one of its scalar + /// expressions (`scalar(v)`, a scalar subquery, `EXISTS`, `IN`). + ScalarRef, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +pub enum GroupingEdgeCompatibility { + Identical, + ConsumerCoarsensProducer, + Incompatible, + NotApplicable, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +pub enum WindowEdgeCompatibility { + /// Physical lowering must prove equal pane/query phase or install an + /// exact boundary residual. The logical DAG alone cannot make that claim. + #[serde(rename = "RequiresAlignedPanePhaseOrExactBoundaryResidual")] + RequiresAlignedPanePhaseOrExactWindowEdgeResidual, + NotApplicable, +} + +/// Stable identity of a node within one exported post-ASAP semantic DAG. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +#[serde(transparent)] +pub struct PostAsapNodeId(pub u32); + +// ── Wire mirrors of the scalar language ────────────────────────────────── + +/// [`ScalarExpr`] with every operator reference replaced by the id of the +/// exported node (connected to the owner by an [`EdgeRole::ScalarRef`] edge). +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum WireScalarExpr { + Column(ColumnId), + Literal(ScalarValue), + Negative { + expr: Box, + semantics: ExprSemantics, + }, + Compare { + left: Box, + op: CompareOpKind, + right: Box, + semantics: ExprSemantics, + }, + BoolAnd(Vec), + BoolOr(Vec), + Not(Box), + IsNull(Box), + IsNotNull(Box), + Cast { + expr: Box, + to: DataType, + try_cast: bool, + }, + InList { + expr: Box, + list: Vec, + negated: bool, + }, + FunctionCall { + name: String, + args: Vec, + }, + Arithmetic { + op: ArithmeticOpKind, + left: Box, + right: Box, + semantics: ExprSemantics, + }, + Case { + operand: Option>, + branches: Vec<(WireScalarExpr, WireScalarExpr)>, + else_expr: Option>, + }, + CurrentTimestamp, + EvalTimestamp, + PromqlScalarFromVector(PostAsapNodeId), + ScalarSubquery(PostAsapNodeId), + Exists { + subquery: PostAsapNodeId, + negated: bool, + }, + InSubquery { + expr: Box, + subquery: PostAsapNodeId, + negated: bool, + }, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct WirePredicate(pub WireScalarExpr); + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct WireProjectItem { + pub alias: Option, + pub expr: WireScalarExpr, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct WireSortKey { + pub expr: WireScalarExpr, + pub ascending: bool, + pub nulls_first: bool, +} + +impl WireScalarExpr { + /// Mirror `expr`, resolving every operator reference through `id_of`. + pub fn from_expr( + expr: &ScalarExpr, + id_of: &mut impl FnMut(&Rc) -> PostAsapNodeId, + ) -> Self { + fn boxed( + e: &ScalarExpr, + id_of: &mut impl FnMut(&Rc) -> PostAsapNodeId, + ) -> Box { + Box::new(WireScalarExpr::from_expr(e, id_of)) + } + fn list( + es: &[ScalarExpr], + id_of: &mut impl FnMut(&Rc) -> PostAsapNodeId, + ) -> Vec { + es.iter() + .map(|e| WireScalarExpr::from_expr(e, id_of)) + .collect() + } + match expr { + ScalarExpr::Column(id) => WireScalarExpr::Column(*id), + ScalarExpr::Literal(v) => WireScalarExpr::Literal(v.clone()), + ScalarExpr::Negative { expr, semantics } => WireScalarExpr::Negative { + expr: boxed(expr, id_of), + semantics: *semantics, + }, + ScalarExpr::Compare { + left, + op, + right, + semantics, + } => WireScalarExpr::Compare { + left: boxed(left, id_of), + op: op.clone(), + right: boxed(right, id_of), + semantics: *semantics, + }, + ScalarExpr::BoolAnd(parts) => WireScalarExpr::BoolAnd(list(parts, id_of)), + ScalarExpr::BoolOr(parts) => WireScalarExpr::BoolOr(list(parts, id_of)), + ScalarExpr::Not(e) => WireScalarExpr::Not(boxed(e, id_of)), + ScalarExpr::IsNull(e) => WireScalarExpr::IsNull(boxed(e, id_of)), + ScalarExpr::IsNotNull(e) => WireScalarExpr::IsNotNull(boxed(e, id_of)), + ScalarExpr::Cast { expr, to, try_cast } => WireScalarExpr::Cast { + expr: boxed(expr, id_of), + to: to.clone(), + try_cast: *try_cast, + }, + ScalarExpr::InList { + expr, + list: items, + negated, + } => WireScalarExpr::InList { + expr: boxed(expr, id_of), + list: list(items, id_of), + negated: *negated, + }, + ScalarExpr::FunctionCall { name, args } => WireScalarExpr::FunctionCall { + name: name.clone(), + args: list(args, id_of), + }, + ScalarExpr::Arithmetic { + op, + left, + right, + semantics, + } => WireScalarExpr::Arithmetic { + op: op.clone(), + left: boxed(left, id_of), + right: boxed(right, id_of), + semantics: *semantics, + }, + ScalarExpr::Case { + operand, + branches, + else_expr, + } => WireScalarExpr::Case { + operand: operand.as_ref().map(|e| boxed(e, id_of)), + branches: branches + .iter() + .map(|(w, t)| (Self::from_expr(w, id_of), Self::from_expr(t, id_of))) + .collect(), + else_expr: else_expr.as_ref().map(|e| boxed(e, id_of)), + }, + ScalarExpr::CurrentTimestamp => WireScalarExpr::CurrentTimestamp, + ScalarExpr::EvalTimestamp => WireScalarExpr::EvalTimestamp, + ScalarExpr::PromqlScalarFromVector(node) => { + WireScalarExpr::PromqlScalarFromVector(id_of(node)) + } + ScalarExpr::ScalarSubquery(node) => WireScalarExpr::ScalarSubquery(id_of(node)), + ScalarExpr::Exists { subquery, negated } => WireScalarExpr::Exists { + subquery: id_of(subquery), + negated: *negated, + }, + ScalarExpr::InSubquery { + expr, + subquery, + negated, + } => WireScalarExpr::InSubquery { + expr: boxed(expr, id_of), + subquery: id_of(subquery), + negated: *negated, + }, + } + } +} + +impl WirePredicate { + fn from_pred( + p: &Predicate, + id_of: &mut impl FnMut(&Rc) -> PostAsapNodeId, + ) -> Self { + WirePredicate(WireScalarExpr::from_expr(&p.0, id_of)) + } +} + +impl WireSortKey { + fn from_keys( + keys: &[SortKey], + id_of: &mut impl FnMut(&Rc) -> PostAsapNodeId, + ) -> Vec { + keys.iter() + .map(|k| WireSortKey { + expr: WireScalarExpr::from_expr(&k.expr, id_of), + ascending: k.ascending, + nulls_first: k.nulls_first, + }) + .collect() + } +} + +// ── Wire mirror of the non-ASAP operator vocabulary ────────────────────── + +/// [`NonASAPOp`] without its child fields (children are edges) and with +/// every scalar expression mirrored as [`WireScalarExpr`]. Fields named +/// `kind` in the IR are renamed so they do not collide with the variant tag. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(tag = "kind", rename_all = "snake_case")] +pub enum NonASAPOpKind { + Scan { + source: Source, + #[serde(default)] + predicates: Vec, + schema: Schema, + }, + Values { + rows: Vec>, + schema: Schema, + }, + Filter { + pred: WirePredicate, + }, + Project { + cols: Vec, + #[serde(default)] + qualifier: Option, + }, + Aggregate { + reduction: Reduction, + measures: Vec, + #[serde(default)] + output_names: Vec, + #[serde(default)] + filters: Vec>, + #[serde(default)] + having: Option, + }, + Join { + join_kind: JoinKind, + pred: WirePredicate, + }, + SetOp { + set_kind: RelationalSetOpKind, + all: bool, + }, + Concat { + #[serde(default)] + discriminator_unique_key: Option, + }, + Dedup { + cols: Vec, + }, + Sort { + keys: Vec, + #[serde(default)] + partition_by: GroupKeys, + }, + Limit { + n: Option, + offset: usize, + #[serde(default)] + partition_by: GroupKeys, + }, + BinaryOp { + operator: BinaryOperator, + #[serde(default)] + return_bool: bool, + }, + #[serde(rename = "sql_window_func")] + SQLWindowFunc { + func: WindowFuncKind, + args: Vec, + partition_by: GroupKeys, + order_by: Vec, + #[serde(default)] + frame: Option, + output_name: String, + }, + TimeRange { + range: Duration, + range_kind: TimeRangeKind, + }, + TimeShift { + shift: TimeShift, + }, + PromqlVectorFromScalar { + expr: WireScalarExpr, + }, + PromqlRelabel { + dst: String, + value: WireScalarExpr, + }, + PromqlInfoEnrich { + #[serde(default)] + selector: Vec, + }, + PromqlSeriesSample { + #[serde(default)] + by: GroupKeys, + sample_kind: SampleKind, + }, + PromqlSubquery { + range: Duration, + #[serde(default)] + resolution: Option, + }, +} + +impl NonASAPOpKind { + /// Mirror `op`, resolving every operator node its scalar expressions + /// reference through `id_of`. + pub fn from_op( + op: &NonASAPOp, + id_of: &mut impl FnMut(&Rc) -> PostAsapNodeId, + ) -> Self { + use NonASAPOp as Op; + match op { + Op::Scan { + source, + predicates, + schema, + } => NonASAPOpKind::Scan { + source: source.clone(), + predicates: predicates + .iter() + .map(|p| WirePredicate::from_pred(p, id_of)) + .collect(), + schema: schema.clone(), + }, + Op::Values { rows, schema } => NonASAPOpKind::Values { + rows: rows + .iter() + .map(|row| { + row.iter() + .map(|e| WireScalarExpr::from_expr(e, id_of)) + .collect() + }) + .collect(), + schema: schema.clone(), + }, + Op::Filter { pred, .. } => NonASAPOpKind::Filter { + pred: WirePredicate::from_pred(pred, id_of), + }, + Op::Project { + cols, qualifier, .. + } => NonASAPOpKind::Project { + cols: cols + .iter() + .map(|ProjectItem { alias, expr }| WireProjectItem { + alias: alias.clone(), + expr: WireScalarExpr::from_expr(expr, id_of), + }) + .collect(), + qualifier: qualifier.clone(), + }, + Op::Aggregate { + reduction, + measures, + output_names, + filters, + having, + .. + } => NonASAPOpKind::Aggregate { + reduction: reduction.clone(), + measures: measures.clone(), + output_names: output_names.clone(), + filters: filters + .iter() + .map(|p| p.as_ref().map(|p| WirePredicate::from_pred(p, id_of))) + .collect(), + having: having.as_ref().map(|p| WirePredicate::from_pred(p, id_of)), + }, + Op::Join { kind, pred, .. } => NonASAPOpKind::Join { + join_kind: kind.clone(), + pred: WirePredicate::from_pred(pred, id_of), + }, + Op::SetOp { kind, all, .. } => NonASAPOpKind::SetOp { + set_kind: kind.clone(), + all: *all, + }, + Op::Concat { + discriminator_unique_key, + .. + } => NonASAPOpKind::Concat { + discriminator_unique_key: discriminator_unique_key.clone(), + }, + Op::Dedup { cols, .. } => NonASAPOpKind::Dedup { cols: cols.clone() }, + Op::Sort { + keys, partition_by, .. + } => NonASAPOpKind::Sort { + keys: WireSortKey::from_keys(keys, id_of), + partition_by: partition_by.clone(), + }, + Op::Limit { + n, + offset, + partition_by, + .. + } => NonASAPOpKind::Limit { + n: *n, + offset: *offset, + partition_by: partition_by.clone(), + }, + Op::BinaryOp { + operator, + return_bool, + .. + } => NonASAPOpKind::BinaryOp { + operator: operator.clone(), + return_bool: *return_bool, + }, + Op::SQLWindowFunc { + func, + args, + partition_by, + order_by, + frame, + output_name, + .. + } => NonASAPOpKind::SQLWindowFunc { + func: func.clone(), + args: args + .iter() + .map(|e| WireScalarExpr::from_expr(e, id_of)) + .collect(), + partition_by: partition_by.clone(), + order_by: WireSortKey::from_keys(order_by, id_of), + frame: frame.clone(), + output_name: output_name.clone(), + }, + Op::TimeRange { range, kind, .. } => NonASAPOpKind::TimeRange { + range: *range, + range_kind: *kind, + }, + Op::TimeShift { shift, .. } => NonASAPOpKind::TimeShift { shift: *shift }, + Op::PromqlVectorFromScalar(e) => NonASAPOpKind::PromqlVectorFromScalar { + expr: WireScalarExpr::from_expr(e, id_of), + }, + Op::PromqlRelabel { dst, value, .. } => NonASAPOpKind::PromqlRelabel { + dst: dst.clone(), + value: WireScalarExpr::from_expr(value, id_of), + }, + Op::PromqlInfoEnrich { selector, .. } => NonASAPOpKind::PromqlInfoEnrich { + selector: selector.clone(), + }, + Op::PromqlSeriesSample { by, kind, .. } => NonASAPOpKind::PromqlSeriesSample { + by: by.clone(), + sample_kind: *kind, + }, + Op::PromqlSubquery { + range, resolution, .. + } => NonASAPOpKind::PromqlSubquery { + range: *range, + resolution: *resolution, + }, + } + } +} + +// ── The exported DAG ───────────────────────────────────────────────────── + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(tag = "kind", rename_all = "snake_case", deny_unknown_fields)] +pub enum PostAsapOperatorPayload { + Relational { + operator: NonASAPOpKind, + }, + SummaryAgg { + family: FieldDataType, + input: SummaryUpdate, + reduction: Reduction, + grouping: GroupingStrategy, + #[serde(default)] + filter: Option, + }, + SummaryEstimate { + query: SketchStatistic, + }, + FinalizeExactAccumulator, + MaintainPopulation { + population: MaintainedPopulation, + }, + EvaluatePopulation { + evaluation: PopulationStatistic, + }, + SummaryMerge, + SummarySubtract, + SummaryDelete { + key: ColumnId, + }, + SummaryJoin { + key: ColumnId, + family: FieldDataType, + }, + Extension { + name: String, + }, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct PostAsapDagNode { + pub id: PostAsapNodeId, + /// The payload variant is the sole operator identity (`payload.kind` in JSON). + pub payload: PostAsapOperatorPayload, + /// Phase is a placement choice for every operator, independent of payload kind. + pub output_state: ExecutionDataState, + pub output_schema: Schema, + pub guarantee: Option, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct PostAsapDagEdge { + pub producer: PostAsapNodeId, + pub consumer: PostAsapNodeId, + pub role: EdgeRole, + pub intermediate_schema: Schema, + pub data_state: ExecutionDataState, + pub grouping: GroupingEdgeCompatibility, + pub window: WindowEdgeCompatibility, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct PostAsapDag { + pub nodes: Vec, + pub edges: Vec, + /// Semantic workload root. Physical query/precompute sinks are selected + /// downstream by the control plane. + pub root: PostAsapNodeId, +} + +/// Versioned transport envelope for a post-ASAP semantic DAG. +/// +/// Process boundaries exchange this envelope and call [`Self::validate`]. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct PostAsapDagDocument { + pub schema_version: u32, + pub dag: PostAsapDag, +} + +#[derive(Debug, Clone, PartialEq, Eq, Error)] +pub enum PostAsapDagValidationError { + #[error("phase assignment must name every DAG node exactly once")] + IncompletePhaseAssignment, + #[error("ingestion node {consumer:?} depends on query node {producer:?}")] + QueryDependencyInIngestion { + producer: PostAsapNodeId, + consumer: PostAsapNodeId, + }, + #[error("unsupported post-ASAP DAG schema version {0}")] + UnsupportedVersion(u32), + #[error("duplicate post-ASAP node id {0:?}")] + DuplicateNodeId(PostAsapNodeId), + #[error("post-ASAP DAG root {0:?} does not name a node")] + MissingRoot(PostAsapNodeId), + #[error("edge endpoint {0:?} does not name a node")] + MissingEdgeEndpoint(PostAsapNodeId), + #[error("edge {producer:?}->{consumer:?} schema differs from producer output")] + EdgeSchemaMismatch { + producer: PostAsapNodeId, + consumer: PostAsapNodeId, + }, + #[error("edge {producer:?}->{consumer:?} data state differs from producer output")] + EdgeDataStateMismatch { + producer: PostAsapNodeId, + consumer: PostAsapNodeId, + }, + #[error("post-ASAP DAG contains a cycle")] + Cycle, + #[error("post-ASAP node {0:?} is not reachable from the root")] + UnreachableNode(PostAsapNodeId), + #[error("summary aggregate node {node:?} output schema does not contain its declared family")] + SummaryFamilySchemaMismatch { node: PostAsapNodeId }, + #[error( + "summary aggregate node {node:?} declares grouping inconsistent with its sketch state" + )] + SummaryGroupingMismatch { node: PostAsapNodeId }, +} + +impl PostAsapDagDocument { + pub fn new(dag: PostAsapDag) -> Self { + Self { + schema_version: POST_ASAP_DAG_WIRE_VERSION, + dag, + } + } + + pub fn validate(&self) -> Result<(), PostAsapDagValidationError> { + if self.schema_version != POST_ASAP_DAG_WIRE_VERSION { + return Err(PostAsapDagValidationError::UnsupportedVersion( + self.schema_version, + )); + } + self.dag.validate() + } +} + +impl PostAsapDag { + /// Assign execution phases without changing operator semantics. Phase choices + /// do not prove deployment support: callers must bind concrete implementations + /// and storage boundaries before installing this plan. + pub fn with_execution_phases( + &self, + phases: &BTreeMap, + ) -> Result { + self.validate()?; + if phases.len() != self.nodes.len() + || self.nodes.iter().any(|node| !phases.contains_key(&node.id)) + { + return Err(PostAsapDagValidationError::IncompletePhaseAssignment); + } + let mut dag = self.clone(); + for node in &mut dag.nodes { + node.output_state.timing = phases[&node.id]; + } + let states: HashMap<_, _> = dag.nodes.iter().map(|n| (n.id, n.output_state)).collect(); + for edge in &mut dag.edges { + edge.data_state = states[&edge.producer]; + } + dag.validate()?; + Ok(dag) + } + + pub fn validate(&self) -> Result<(), PostAsapDagValidationError> { + let mut nodes = HashMap::new(); + for node in &self.nodes { + if nodes.insert(node.id, node).is_some() { + return Err(PostAsapDagValidationError::DuplicateNodeId(node.id)); + } + if let PostAsapOperatorPayload::SummaryAgg { + family, grouping, .. + } = &node.payload + { + let mut found_family = false; + for field in &node.output_schema.fields { + if &field.dtype == family { + found_family = true; + } + if let FieldDataType::Sketch(_, schema_grouping) = &field.dtype { + if schema_grouping != grouping { + return Err(PostAsapDagValidationError::SummaryGroupingMismatch { + node: node.id, + }); + } + } + } + if !found_family { + return Err(PostAsapDagValidationError::SummaryFamilySchemaMismatch { + node: node.id, + }); + } + } + } + if !nodes.contains_key(&self.root) { + return Err(PostAsapDagValidationError::MissingRoot(self.root)); + } + let mut children: HashMap> = HashMap::new(); + for edge in &self.edges { + let producer = nodes.get(&edge.producer).ok_or( + PostAsapDagValidationError::MissingEdgeEndpoint(edge.producer), + )?; + if !nodes.contains_key(&edge.consumer) { + return Err(PostAsapDagValidationError::MissingEdgeEndpoint( + edge.consumer, + )); + } + if producer.output_state.timing == ExecutionTiming::QueryTime + && nodes[&edge.consumer].output_state.timing == ExecutionTiming::IngestionTime + { + return Err(PostAsapDagValidationError::QueryDependencyInIngestion { + producer: edge.producer, + consumer: edge.consumer, + }); + } + if edge.intermediate_schema != producer.output_schema { + return Err(PostAsapDagValidationError::EdgeSchemaMismatch { + producer: edge.producer, + consumer: edge.consumer, + }); + } + if edge.data_state != producer.output_state { + return Err(PostAsapDagValidationError::EdgeDataStateMismatch { + producer: edge.producer, + consumer: edge.consumer, + }); + } + children + .entry(edge.consumer) + .or_default() + .push(edge.producer); + } + fn visit( + id: PostAsapNodeId, + children: &HashMap>, + visiting: &mut HashSet, + visited: &mut HashSet, + ) -> bool { + if visited.contains(&id) { + return true; + } + if !visiting.insert(id) { + return false; + } + if children + .get(&id) + .into_iter() + .flatten() + .any(|child| !visit(*child, children, visiting, visited)) + { + return false; + } + visiting.remove(&id); + visited.insert(id); + true + } + if !visit( + self.root, + &children, + &mut HashSet::new(), + &mut HashSet::new(), + ) { + return Err(PostAsapDagValidationError::Cycle); + } + fn mark( + id: PostAsapNodeId, + children: &HashMap>, + reachable: &mut HashSet, + ) { + if !reachable.insert(id) { + return; + } + for child in children.get(&id).into_iter().flatten() { + mark(*child, children, reachable); + } + } + let mut reachable = HashSet::new(); + mark(self.root, &children, &mut reachable); + if let Some(id) = nodes.keys().find(|id| !reachable.contains(id)) { + return Err(PostAsapDagValidationError::UnreachableNode(*id)); + } + Ok(()) + } +} + +// ── Compilation from the IR ────────────────────────────────────────────── + +/// Compiler-local identity assignment. It deliberately retains `Rc` handles +/// and is not serialized; deployed artifacts persist the post-ASAP node ID +/// together with their physical materialization/query IDs. +#[derive(Debug, Clone)] +pub struct PostAsapNodeIdentityMap { + nodes_by_id: Vec>, +} + +impl PostAsapNodeIdentityMap { + pub fn node_id(&self, node: &Rc) -> Option { + self.nodes_by_id + .iter() + .position(|candidate| Rc::ptr_eq(candidate, node)) + .map(|id| PostAsapNodeId(id as u32)) + } + + pub fn operator_node(&self, id: PostAsapNodeId) -> Option<&Rc> { + self.nodes_by_id.get(id.0 as usize) + } +} + +#[derive(Debug, Clone)] +pub struct PostAsapDagCompilation { + pub dag: PostAsapDag, + pub node_ids: PostAsapNodeIdentityMap, +} + +pub fn compile_post_asap_dag( + root: &Rc, +) -> Result { + Ok(compile_post_asap_dag_with_node_ids(root)?.dag) +} + +/// Export the timed DAG below `root`. Every reachable node must carry a +/// timing (see [`super::timing::apply_lifecycle_timings`]); the data-state +/// rules were checked by that pass and are not re-run here. +pub fn compile_post_asap_dag_with_node_ids( + root: &Rc, +) -> Result { + let mut exporter = Exporter::default(); + let root = exporter.visit(root)?; + let dag = PostAsapDag { + nodes: exporter.nodes, + edges: exporter.edges, + root, + }; + dag.validate() + .expect("compiler emits a valid post-ASAP DAG"); + Ok(PostAsapDagCompilation { + dag, + node_ids: PostAsapNodeIdentityMap { + nodes_by_id: exporter.nodes_by_id, + }, + }) +} + +#[derive(Default)] +struct Exporter { + ids: HashMap<*const OperatorNode, PostAsapNodeId>, + nodes: Vec, + edges: Vec, + nodes_by_id: Vec>, +} + +impl Exporter { + fn visit( + &mut self, + node: &Rc, + ) -> Result { + if let Some(id) = self.ids.get(&Rc::as_ptr(node)) { + return Ok(*id); + } + let output_state = data_state(node).ok_or(ExecutionDataStateError::UntimedNode { + operator: node.operator.kind_name(), + })?; + // Operator inputs first, then the nodes read from scalar expressions. + let mut producers = Vec::new(); + for (child, role) in input_edges(&node.operator) { + producers.push((self.visit(child)?, child, role)); + } + { + let scalars = match &node.operator { + Operator::NonASAP(op) => op.scalar_exprs(), + Operator::ASAP(ASAPOp::SummaryAgg { + filter: Some(filter), + .. + }) => vec![&filter.0], + _ => vec![], + }; + for expr in scalars { + for referenced in expr.operator_refs() { + producers.push((self.visit(referenced)?, referenced, EdgeRole::ScalarRef)); + } + } + } + let id = PostAsapNodeId(self.nodes.len() as u32); + let payload = { + let ids = &self.ids; + let mut id_of = |n: &Rc| ids[&Rc::as_ptr(n)]; + payload_of(&node.operator, &mut id_of) + }; + self.nodes.push(PostAsapDagNode { + id, + payload, + output_state, + output_schema: node.schema.clone(), + guarantee: node.guarantee.clone(), + }); + self.nodes_by_id.push(Rc::clone(node)); + self.ids.insert(Rc::as_ptr(node), id); + for (producer, child, role) in producers { + let producer_state = self.nodes[producer.0 as usize].output_state; + let maintenance_dependency = producer_state.timing == ExecutionTiming::IngestionTime + && output_state.timing == ExecutionTiming::IngestionTime; + self.edges.push(PostAsapDagEdge { + producer, + consumer: id, + role, + intermediate_schema: child.schema.clone(), + data_state: producer_state, + grouping: grouping_compatibility(&child.operator, &node.operator), + window: if maintenance_dependency { + WindowEdgeCompatibility::RequiresAlignedPanePhaseOrExactWindowEdgeResidual + } else { + WindowEdgeCompatibility::NotApplicable + }, + }); + } + Ok(id) + } +} + +/// The operator's own inputs with their edge roles, in field order. +fn input_edges(operator: &Operator) -> Vec<(&Rc, EdgeRole)> { + match operator { + Operator::NonASAP(op) => match op { + NonASAPOp::Join { left, right, .. } + | NonASAPOp::SetOp { left, right, .. } + | NonASAPOp::BinaryOp { + lhs: left, + rhs: right, + .. + } => vec![(left, EdgeRole::Left), (right, EdgeRole::Right)], + NonASAPOp::Concat { children, .. } => { + children.iter().map(|c| (c, EdgeRole::Input)).collect() + } + NonASAPOp::Filter { child, .. } + | NonASAPOp::Project { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::SQLWindowFunc { child, .. } + | NonASAPOp::TimeRange { child, .. } + | NonASAPOp::TimeShift { child, .. } + | NonASAPOp::PromqlRelabel { child, .. } + | NonASAPOp::PromqlInfoEnrich { child, .. } + | NonASAPOp::PromqlSeriesSample { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => vec![(child, EdgeRole::Input)], + NonASAPOp::Scan { .. } + | NonASAPOp::Values { .. } + | NonASAPOp::PromqlVectorFromScalar(_) => vec![], + }, + Operator::ASAP(op) => match op { + ASAPOp::SummarySubtract { left, right } + | ASAPOp::SummaryJoin { + outer: left, + inner: right, + .. + } => vec![(left, EdgeRole::Left), (right, EdgeRole::Right)], + ASAPOp::SummaryMerge { children } => { + children.iter().map(|c| (c, EdgeRole::Input)).collect() + } + ASAPOp::SummaryAgg { child, .. } + | ASAPOp::FinalizeExactAccumulator { child } + | ASAPOp::MaintainPopulation { child, .. } + | ASAPOp::EvaluatePopulation { child, .. } + | ASAPOp::Extension { child, .. } => vec![(child, EdgeRole::Input)], + ASAPOp::SummaryEstimate { summary_input, .. } + | ASAPOp::SummaryDelete { summary_input, .. } => { + vec![(summary_input, EdgeRole::Input)] + } + }, + } +} + +fn payload_of( + operator: &Operator, + id_of: &mut impl FnMut(&Rc) -> PostAsapNodeId, +) -> PostAsapOperatorPayload { + match operator { + Operator::NonASAP(op) => PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::from_op(op, id_of), + }, + Operator::ASAP(op) => match op { + ASAPOp::SummaryAgg { + family, + input, + reduction, + grouping, + filter, + .. + } => PostAsapOperatorPayload::SummaryAgg { + family: family.clone(), + input: input.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + filter: filter.as_ref().map(|p| WirePredicate::from_pred(p, id_of)), + }, + ASAPOp::SummaryEstimate { query, .. } => PostAsapOperatorPayload::SummaryEstimate { + query: query.clone(), + }, + ASAPOp::FinalizeExactAccumulator { .. } => { + PostAsapOperatorPayload::FinalizeExactAccumulator + } + ASAPOp::MaintainPopulation { population, .. } => { + PostAsapOperatorPayload::MaintainPopulation { + population: population.clone(), + } + } + ASAPOp::EvaluatePopulation { evaluation, .. } => { + PostAsapOperatorPayload::EvaluatePopulation { + evaluation: evaluation.clone(), + } + } + ASAPOp::SummaryMerge { .. } => PostAsapOperatorPayload::SummaryMerge, + ASAPOp::SummarySubtract { .. } => PostAsapOperatorPayload::SummarySubtract, + ASAPOp::SummaryDelete { key, .. } => { + PostAsapOperatorPayload::SummaryDelete { key: *key } + } + ASAPOp::SummaryJoin { key, family, .. } => PostAsapOperatorPayload::SummaryJoin { + key: *key, + family: family.clone(), + }, + ASAPOp::Extension { name, .. } => { + PostAsapOperatorPayload::Extension { name: name.clone() } + } + }, + } +} + +/// Grouping compatibility between two `SummaryAgg`s by their reductions. +fn grouping_compatibility(producer: &Operator, consumer: &Operator) -> GroupingEdgeCompatibility { + let ( + Operator::ASAP(ASAPOp::SummaryAgg { + reduction: producer, + .. + }), + Operator::ASAP(ASAPOp::SummaryAgg { + reduction: consumer, + .. + }), + ) = (producer, consumer) + else { + return GroupingEdgeCompatibility::NotApplicable; + }; + match (producer, consumer) { + (p, c) if p == c => GroupingEdgeCompatibility::Identical, + (Reduction::PerEntity, Reduction::Reduce(_)) => { + GroupingEdgeCompatibility::ConsumerCoarsensProducer + } + (Reduction::Reduce(p), Reduction::Reduce(c)) + if !p.is_without() && !c.is_without() && c.iter().all(|key| p.contains(key)) => + { + GroupingEdgeCompatibility::ConsumerCoarsensProducer + } + _ => GroupingEdgeCompatibility::Incompatible, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ir::timing::{apply_lifecycle_timings, LifecycleAssignment, TimingMemo}; + use crate::post_asap::execution_data_state::DataPrimitive; + use crate::post_asap::maintained_population::{CurrentSeriesInput, PopulationInput}; + use crate::post_asap::sketch::{ExactKind, ExactParams}; + use crate::pre_asap::schema::Field; + use crate::pre_asap::ColumnRef; + + fn timed(root: &Rc) -> Rc { + apply_lifecycle_timings( + root, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .expect("timed") + } + + fn scan(fields: Vec) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::new(fields), + }) + .unwrap() + } + + fn value_scan() -> Rc { + scan(vec![Field::plain("value", DataType::Float64, false)]) + } + + fn true_pred() -> Predicate { + Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))) + } + + fn sum_agg(child: Rc) -> Rc { + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + Rc::new(OperatorNode::with_schema( + Operator::ASAP(ASAPOp::SummaryAgg { + child, + family: family.clone(), + input: SummaryUpdate::column(ColumnRef::SampleValue), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + }), + Schema::lifted(vec![Field::new("value", family, false)], None), + )) + } + + fn finalize(child: Rc) -> Rc { + Rc::new(OperatorNode::with_schema( + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }), + Schema::lifted(vec![Field::plain("value", DataType::Float64, false)], None), + )) + } + + fn relational_count(dag: &PostAsapDag) -> usize { + dag.nodes + .iter() + .filter(|n| matches!(n.payload, PostAsapOperatorPayload::Relational { .. })) + .count() + } + + #[test] + fn every_physical_payload_can_be_assigned_either_phase() { + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let payloads = vec![ + PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::Limit { + n: Some(1), + offset: 0, + partition_by: GroupKeys::none(), + }, + }, + PostAsapOperatorPayload::SummaryAgg { + family: family.clone(), + input: SummaryUpdate::column(ColumnRef::SampleValue), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + }, + PostAsapOperatorPayload::SummaryEstimate { + query: SketchStatistic::Cardinality, + }, + PostAsapOperatorPayload::FinalizeExactAccumulator, + PostAsapOperatorPayload::MaintainPopulation { + population: MaintainedPopulation { + input: PopulationInput::CurrentSeries(CurrentSeriesInput { + metric: "m".into(), + matchers: vec![], + grouping: vec![], + without: false, + lookback_ms: 300_000, + }), + max_k: 10, + quantiles: false, + }, + }, + PostAsapOperatorPayload::EvaluatePopulation { + evaluation: PopulationStatistic::Count, + }, + PostAsapOperatorPayload::SummaryMerge, + PostAsapOperatorPayload::SummarySubtract, + PostAsapOperatorPayload::SummaryDelete { key: 0 }, + PostAsapOperatorPayload::SummaryJoin { + key: 0, + family: family.clone(), + }, + PostAsapOperatorPayload::Extension { name: "ext".into() }, + ]; + for payload in payloads { + // This checks physical identity and placement, not kernel availability. + let primitive = match &payload { + PostAsapOperatorPayload::Relational { .. } + | PostAsapOperatorPayload::SummaryEstimate { .. } + | PostAsapOperatorPayload::FinalizeExactAccumulator + | PostAsapOperatorPayload::EvaluatePopulation { .. } + | PostAsapOperatorPayload::Extension { .. } => DataPrimitive::Raw, + PostAsapOperatorPayload::SummaryAgg { .. } + | PostAsapOperatorPayload::MaintainPopulation { .. } + | PostAsapOperatorPayload::SummaryJoin { .. } + | PostAsapOperatorPayload::SummarySubtract + | PostAsapOperatorPayload::SummaryDelete { .. } + | PostAsapOperatorPayload::SummaryMerge => DataPrimitive::SummaryState, + }; + let dag = PostAsapDag { + root: PostAsapNodeId(0), + edges: vec![], + nodes: vec![PostAsapDagNode { + id: PostAsapNodeId(0), + payload: payload.clone(), + output_state: ExecutionDataState { + timing: ExecutionTiming::QueryTime, + primitive, + }, + output_schema: Schema::lifted( + vec![Field::new("value", family.clone(), false)], + None, + ), + guarantee: None, + }], + }; + for phase in [ExecutionTiming::IngestionTime, ExecutionTiming::QueryTime] { + let placed = dag + .with_execution_phases(&BTreeMap::from([(dag.root, phase)])) + .unwrap(); + assert_eq!(placed.nodes[0].payload, payload); + assert_eq!(placed.nodes[0].output_state.timing, phase); + let wire = serde_json::to_value(&placed).unwrap(); + assert!(wire["nodes"][0]["payload"].get("timing").is_none()); + assert_eq!(serde_json::from_value::(wire).unwrap(), placed); + } + assert!(dag.with_execution_phases(&BTreeMap::new()).is_err()); + } + } + + #[test] + fn phase_assignment_updates_edges_and_rejects_query_dependencies_in_ingestion() { + let schema = Schema::lifted(vec![], None); + let nodes = [0, 1] + .into_iter() + .map(|id| PostAsapDagNode { + id: PostAsapNodeId(id), + payload: PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: schema.clone(), + }, + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: schema.clone(), + guarantee: None, + }) + .collect(); + let dag = PostAsapDag { + nodes, + root: PostAsapNodeId(1), + edges: vec![PostAsapDagEdge { + producer: PostAsapNodeId(0), + consumer: PostAsapNodeId(1), + role: EdgeRole::Input, + intermediate_schema: schema, + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }], + }; + let placed = dag + .with_execution_phases(&BTreeMap::from([ + (PostAsapNodeId(0), ExecutionTiming::IngestionTime), + (PostAsapNodeId(1), ExecutionTiming::QueryTime), + ])) + .unwrap(); + assert_eq!( + placed.edges[0].data_state.timing, + ExecutionTiming::IngestionTime + ); + assert_eq!(dag.edges[0].data_state.timing, ExecutionTiming::QueryTime); + assert!(matches!( + dag.with_execution_phases(&BTreeMap::from([ + (PostAsapNodeId(0), ExecutionTiming::QueryTime), + (PostAsapNodeId(1), ExecutionTiming::IngestionTime), + ])), + Err(PostAsapDagValidationError::QueryDependencyInIngestion { .. }) + )); + } + + #[test] + fn exports_summary_over_summary_as_typed_precompute_edges() { + let inner = sum_agg(value_scan()); + let outer = sum_agg(Rc::clone(&inner)); + let root = timed(&finalize(outer)); + let inner = Rc::clone(root.children()[0].children()[0]); + + let compiled = compile_post_asap_dag_with_node_ids(&root).unwrap(); + assert_eq!(compiled.node_ids.node_id(&root), Some(PostAsapNodeId(3))); + assert!(Rc::ptr_eq( + compiled.node_ids.operator_node(PostAsapNodeId(1)).unwrap(), + &inner + )); + let dag = compiled.dag; + assert_eq!(dag.root, PostAsapNodeId(3)); + assert_eq!( + dag.nodes[0].output_state, + ExecutionDataState::INGESTION_ROWS + ); + assert_eq!( + dag.nodes[1].output_state, + ExecutionDataState::INGESTION_SUMMARY + ); + assert_eq!( + dag.nodes[2].output_state, + ExecutionDataState::INGESTION_SUMMARY + ); + assert_eq!(dag.nodes[3].output_state, ExecutionDataState::QUERY_ROWS); + let dependency = dag + .edges + .iter() + .find(|e| e.producer == PostAsapNodeId(1) && e.consumer == PostAsapNodeId(2)) + .unwrap(); + assert_eq!(dependency.role, EdgeRole::Input); + assert_eq!(dependency.data_state, ExecutionDataState::INGESTION_SUMMARY); + assert_eq!(dependency.grouping, GroupingEdgeCompatibility::Identical); + assert_eq!( + dependency.window, + WindowEdgeCompatibility::RequiresAlignedPanePhaseOrExactWindowEdgeResidual + ); + assert!(matches!( + dependency.intermediate_schema.fields[0].dtype, + FieldDataType::ExactAggregate(ExactKind::Sum, _) + )); + let encoded = serde_json::to_string(&dag).expect("serialize post-ASAP DAG"); + let decoded: PostAsapDag = + serde_json::from_str(&encoded).expect("deserialize post-ASAP DAG"); + assert_eq!(decoded, dag); + let document = PostAsapDagDocument::new(decoded); + document.validate().unwrap(); + let mut invalid = serde_json::to_value(&document).unwrap(); + invalid["dag"]["nodes"][0]["operator"] = serde_json::json!("Binary"); + assert!(serde_json::from_value::(invalid).is_err()); + assert!(document.dag.nodes.iter().all(|node| { + let wire = serde_json::to_value(node).unwrap(); + wire.get("operator").is_none() && wire["payload"]["kind"].is_string() + })); + let mut old_version = document.clone(); + old_version.schema_version = 1; + assert_eq!( + old_version.validate(), + Err(PostAsapDagValidationError::UnsupportedVersion(1)) + ); + let mut unknown = serde_json::to_value(&document).unwrap(); + unknown["unexpected"] = serde_json::json!(true); + assert!(serde_json::from_value::(unknown).is_err()); + assert!(matches!( + dag.nodes[2].payload, + PostAsapOperatorPayload::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + reduction: Reduction::Reduce(_), + .. + } + )); + } + + #[test] + fn relational_operators_above_and_below_a_summary_export_one_node_each() { + // Scan -> Filter -> SummaryAgg -> FinalizeExactAccumulator -> Limit + let filter = OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: true_pred(), + child: value_scan(), + }) + .unwrap(); + let limit = OperatorNode::non_asap_node(NonASAPOp::Limit { + n: Some(1), + offset: 0, + partition_by: GroupKeys::none(), + child: finalize(sum_agg(filter)), + }) + .unwrap(); + let root = timed(&limit); + let dag = compile_post_asap_dag(&root).unwrap(); + assert_eq!(dag.nodes.len(), 5); + assert_eq!(dag.edges.len(), 4); + assert_eq!(relational_count(&dag), 3); + assert_eq!(dag.root, PostAsapNodeId(4)); + assert_eq!( + dag.nodes[0].output_state, + ExecutionDataState::INGESTION_ROWS + ); + assert_eq!( + dag.nodes[1].output_state, + ExecutionDataState::INGESTION_ROWS + ); + assert_eq!(dag.nodes[4].output_state, ExecutionDataState::QUERY_ROWS); + assert!(matches!( + &dag.nodes[1].payload, + PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::Filter { + pred: WirePredicate(WireScalarExpr::Literal(ScalarValue::Boolean(true))) + } + } + )); + assert!(matches!( + &dag.nodes[4].payload, + PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::Limit { n: Some(1), .. } + } + )); + // Filter -> SummaryAgg: both at ingestion time, so pane alignment is + // a lowering obligation; the relational producer has no grouping. + let edge = &dag.edges[1]; + assert_eq!( + (edge.producer, edge.consumer), + (PostAsapNodeId(1), PostAsapNodeId(2)) + ); + assert_eq!(edge.grouping, GroupingEdgeCompatibility::NotApplicable); + assert_eq!( + edge.window, + WindowEdgeCompatibility::RequiresAlignedPanePhaseOrExactWindowEdgeResidual + ); + assert!(dag.edges.iter().all(|e| e.role == EdgeRole::Input)); + + let encoded = serde_json::to_string(&dag).unwrap(); + let decoded: PostAsapDag = serde_json::from_str(&encoded).unwrap(); + assert_eq!(decoded, dag); + } + + #[test] + fn shared_scan_under_two_consumers_is_exported_once() { + let shared = value_scan(); + let branch = |child| { + OperatorNode::non_asap_node(NonASAPOp::Filter { + pred: true_pred(), + child, + }) + .unwrap() + }; + let join = OperatorNode::non_asap_node(NonASAPOp::Join { + kind: JoinKind::Inner, + pred: true_pred(), + left: branch(Rc::clone(&shared)), + right: branch(shared), + }) + .unwrap(); + let root = timed(&join); + let compiled = compile_post_asap_dag_with_node_ids(&root).unwrap(); + let dag = compiled.dag; + assert_eq!(dag.nodes.len(), 4); + assert_eq!(dag.edges.len(), 4); + let scan_id = PostAsapNodeId(0); + assert!(matches!( + dag.nodes[0].payload, + PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::Scan { .. } + } + )); + let from_scan: Vec<_> = dag.edges.iter().filter(|e| e.producer == scan_id).collect(); + assert_eq!(from_scan.len(), 2); + assert_ne!(from_scan[0].consumer, from_scan[1].consumer); + let into_join: Vec<_> = dag + .edges + .iter() + .filter(|e| e.consumer == dag.root) + .map(|e| e.role) + .collect(); + assert_eq!(into_join, vec![EdgeRole::Left, EdgeRole::Right]); + let timed_scan = Rc::clone(root.children()[0].children()[0]); + assert_eq!(compiled.node_ids.node_id(&timed_scan), Some(scan_id)); + } + + #[test] + fn scalar_bridge_reference_exports_a_scalar_ref_edge() { + let vector = scan(vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + ]); + let bridge = OperatorNode::non_asap_node(NonASAPOp::PromqlVectorFromScalar( + ScalarExpr::PromqlScalarFromVector(vector), + )) + .unwrap(); + let root = timed(&bridge); + let dag = compile_post_asap_dag(&root).unwrap(); + assert_eq!(dag.nodes.len(), 2); + assert_eq!(dag.root, PostAsapNodeId(1)); + assert_eq!( + dag.nodes[1].payload, + PostAsapOperatorPayload::Relational { + operator: NonASAPOpKind::PromqlVectorFromScalar { + expr: WireScalarExpr::PromqlScalarFromVector(PostAsapNodeId(0)), + }, + } + ); + assert_eq!(dag.edges.len(), 1); + assert_eq!(dag.edges[0].role, EdgeRole::ScalarRef); + assert_eq!(dag.edges[0].producer, PostAsapNodeId(0)); + assert_eq!(dag.edges[0].consumer, PostAsapNodeId(1)); + let wire = serde_json::to_value(&dag).unwrap(); + assert_eq!(wire["nodes"][1]["payload"]["kind"], "relational"); + assert_eq!( + wire["nodes"][1]["payload"]["operator"]["kind"], + "promql_vector_from_scalar" + ); + assert_eq!( + wire["nodes"][1]["payload"]["operator"]["expr"]["PromqlScalarFromVector"], + 0 + ); + assert_eq!(serde_json::from_value::(wire).unwrap(), dag); + } + + #[test] + fn untimed_input_is_rejected() { + let root = value_scan(); + assert!(matches!( + compile_post_asap_dag(&root), + Err(ExecutionDataStateError::UntimedNode { operator: "Scan" }) + )); + } + + #[test] + fn wire_5_document_is_rejected() { + let root = timed(&finalize(sum_agg(value_scan()))); + let document = PostAsapDagDocument::new(compile_post_asap_dag(&root).unwrap()); + assert_eq!(document.schema_version, 7); + let mut wire = serde_json::to_value(&document).unwrap(); + wire["schema_version"] = serde_json::json!(5); + let old: PostAsapDagDocument = serde_json::from_value(wire).unwrap(); + assert_eq!( + old.validate(), + Err(PostAsapDagValidationError::UnsupportedVersion(5)) + ); + let encoded = serde_json::to_string(&document).unwrap(); + let decoded: PostAsapDagDocument = serde_json::from_str(&encoded).unwrap(); + assert_eq!(decoded, document); + decoded.validate().unwrap(); + } + + #[test] + fn post_asap_node_ids_serialize_in_deterministic_binding_order() { + let mut bindings = BTreeMap::new(); + bindings.insert(PostAsapNodeId(10), "materialization-10"); + bindings.insert(PostAsapNodeId(2), "query-2"); + assert_eq!( + serde_json::to_string(&bindings).unwrap(), + r#"{"2":"query-2","10":"materialization-10"}"# + ); + } +} diff --git a/crates/types/src/ir/mod.rs b/crates/types/src/ir/mod.rs new file mode 100644 index 000000000..641ed0913 --- /dev/null +++ b/crates/types/src/ir/mod.rs @@ -0,0 +1,37 @@ +//! The unified operator IR: one operator language before and after ASAP +//! optimization. +//! +//! - [`node`] — [`OperatorNode`] / [`Operator`]: the DAG node and its two +//! operator categories, with the common planning properties. +//! - [`non_asap`] — [`NonASAPOp`]: ordinary query operators. +//! - [`asap`] — [`ASAPOp`]: summary-state construction, operations and evaluations. +//! - [`scalar`] — [`ScalarExpr`]: value computation owned by operator fields. +//! - [`operator_properties`] — operator parameters, such as grouping and window frames. +//! - [`aggregate_schema`] — aggregate output-column/type derivation. +//! - [`error`] — errors from schema and type derivation. + +pub mod asap; +pub mod canonicalize; +pub mod cse; +pub mod export; +pub mod node; +pub mod query; +pub use query::QueryRoot; +pub mod non_asap; +pub mod scalar; +pub mod timing; + +pub use asap::{ASAPOp, UNIMPLEMENTED_ASAP_OP}; +pub use node::{Operator, OperatorNode, OperatorResultKind}; +pub use non_asap::BinaryOperator; +pub use non_asap::{NonASAPOp, TimeRangeKind}; +pub use scalar::{ExprSemantics, Predicate, ProjectItem, ScalarExpr, SortKey}; +pub use timing::{ + apply_lifecycle_timings, data_state, planned_data_state, split_shared_by_phase, + validate_default, LifecycleAssignment, TimingMemo, +}; + +pub mod aggregate_schema; +pub mod error; +pub mod operator_properties; +pub use error::SchemaDerivationError; diff --git a/crates/types/src/ir/node.rs b/crates/types/src/ir/node.rs new file mode 100644 index 000000000..592be3aef --- /dev/null +++ b/crates/types/src/ir/node.rs @@ -0,0 +1,300 @@ +//! The unified operator node: one node type before and after ASAP +//! optimization. A node is an operator plus the common planning properties +//! every traversal needs (output category and schema, accuracy guarantee, +//! execution timing). + +use std::collections::HashSet; +use std::rc::Rc; + +use serde::{Deserialize, Serialize}; + +use super::asap::ASAPOp; +use super::non_asap::NonASAPOp; +use crate::ir::SchemaDerivationError; +use crate::post_asap::execution_data_state::ExecutionTiming; +use crate::post_asap::guarantee::ResultGuarantee; +use crate::pre_asap::schema::Schema; + +/// The output category of an operator, derived from the operation and its +/// inputs. Matching column schemas do not make categories interchangeable. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum OperatorResultKind { + Relation, + InstantVector, + RangeVector, + /// Unfinalized summary / accumulator state. + State, +} + +/// The operation a node performs: an ordinary query operator or an ASAP +/// summary operator. Either category can consume the other's output. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum Operator { + NonASAP(NonASAPOp), + ASAP(ASAPOp), +} + +impl Operator { + pub fn children(&self) -> Vec<&Rc> { + match self { + Operator::NonASAP(op) => op.children(), + Operator::ASAP(op) => op.children(), + } + } + + pub fn map_children(&self, f: impl FnMut(&Rc) -> Rc) -> Self { + match self { + Operator::NonASAP(op) => Operator::NonASAP(op.map_children(f)), + Operator::ASAP(op) => Operator::ASAP(op.map_children(f)), + } + } + + pub fn output_schema(&self) -> Result { + match self { + Operator::NonASAP(op) => op.output_schema(), + Operator::ASAP(op) => op.output_schema(), + } + } + + pub fn output_kind(&self) -> OperatorResultKind { + match self { + Operator::NonASAP(op) => op.output_kind(), + Operator::ASAP(op) => op.output_kind(), + } + } + + pub fn validate_inputs(&self) -> Result<(), SchemaDerivationError> { + match self { + Operator::NonASAP(op) => op.validate_inputs(), + Operator::ASAP(op) => op.validate_inputs(), + } + } + + pub fn kind_name(&self) -> &'static str { + match self { + Operator::NonASAP(op) => op.kind_name(), + Operator::ASAP(op) => op.kind_name(), + } + } +} + +/// A node of the logical DAG. Nodes are immutable and shared through `Rc`; +/// a structurally identical sub-DAG referenced from several parents is one +/// node. +/// +/// `schema` and `result_kind` are derived from `operator` and its children +/// at construction and retained. `guarantee` is `None` until accuracy +/// assessment establishes one (`None` never means exact). `timing` is `None` +/// until a lifecycle assignment is applied; export rejects an executable +/// node without one. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct OperatorNode { + pub operator: Operator, + pub result_kind: OperatorResultKind, + pub schema: Schema, + pub guarantee: Option, + pub timing: Option, +} + +impl OperatorNode { + /// Build a node, deriving its schema and output category. Fails when the + /// schema cannot be derived (a column reference out of range, a reserved + /// ASAP operator, ...). + pub fn new(operator: Operator) -> Result { + let schema = operator.output_schema()?; + Ok(Self::with_schema(operator, schema)) + } + + /// Build a node with a caller-supplied output schema. Summary planning + /// uses this where its evaluation naming is more specific than the derived + /// shape; the output category is still derived. + pub fn with_schema(operator: Operator, schema: Schema) -> Self { + let result_kind = operator.output_kind(); + Self { + operator, + result_kind, + schema, + guarantee: None, + timing: None, + } + } + + pub fn non_asap_node(op: NonASAPOp) -> Result, SchemaDerivationError> { + Self::new(Operator::NonASAP(op)).map(Rc::new) + } + + /// An ASAP node with a planner-supplied schema and guarantee. + pub fn asap_node(op: ASAPOp, schema: Schema, guarantee: Option) -> Rc { + Rc::new(Self::with_schema(Operator::ASAP(op), schema).with_guarantee(guarantee)) + } + + pub fn with_guarantee(mut self, guarantee: Option) -> Self { + self.guarantee = guarantee; + self + } + + pub fn with_timing(mut self, timing: Option) -> Self { + self.timing = timing; + self + } + + pub fn non_asap(&self) -> Option<&NonASAPOp> { + match &self.operator { + Operator::NonASAP(op) => Some(op), + Operator::ASAP(_) => None, + } + } + + pub fn asap(&self) -> Option<&ASAPOp> { + match &self.operator { + Operator::ASAP(op) => Some(op), + Operator::NonASAP(_) => None, + } + } + + pub fn is_asap(&self) -> bool { + matches!(self.operator, Operator::ASAP(_)) + } + + /// The non-ASAP operator of a node that is known to be one; a front-end + /// DAG never contains an ASAP node, so one here is a caller bug. + pub fn expect_non_asap(&self) -> &NonASAPOp { + self.non_asap().unwrap_or_else(|| { + panic!( + "expected a non-ASAP operator, found {}", + self.operator.kind_name() + ) + }) + } + + /// Direct inputs, including the operator nodes referenced from this + /// node's scalar expressions. + pub fn children(&self) -> Vec<&Rc> { + self.operator.children() + } + + /// Rebuild this node with `f` applied to every direct input. The schema + /// is re-derived for a non-ASAP operator (its shape follows its inputs); + /// an ASAP node keeps its retained schema. `guarantee` and `timing` are + /// cleared: both depend on the inputs and must be re-established. + pub fn map_children( + &self, + f: impl FnMut(&Rc) -> Rc, + ) -> Result { + let operator = self.operator.map_children(f); + match &operator { + Operator::NonASAP(_) => Self::new(operator), + Operator::ASAP(_) => Ok(Self::with_schema(operator, self.schema.clone())), + } + } + + /// Whether any node reachable from this one (including itself) is an + /// ASAP operator. Visits each shared node once. + pub fn contains_asap(&self) -> bool { + fn walk(node: &OperatorNode, seen: &mut HashSet<*const OperatorNode>) -> bool { + if node.is_asap() { + return true; + } + node.children() + .iter() + .any(|child| seen.insert(Rc::as_ptr(child)) && walk(child, seen)) + } + walk(self, &mut HashSet::new()) + } + + /// Every unique node reachable from `root`, parents before children + /// (pre-order, deduplicated by pointer identity). + pub fn reachable(root: &Rc) -> Vec> { + fn walk( + node: &Rc, + seen: &mut HashSet<*const OperatorNode>, + out: &mut Vec>, + ) { + if !seen.insert(Rc::as_ptr(node)) { + return; + } + out.push(Rc::clone(node)); + for child in node.children() { + walk(child, seen, out); + } + } + let mut out = Vec::new(); + walk(root, &mut HashSet::new(), &mut out); + out + } + + /// Validate assigned phases without imposing any particular runtime implementation. + pub fn validate_execution_timing(self: &Rc) -> Result<(), SchemaDerivationError> { + self.validate_structure()?; + for node in Self::reachable(self) { + let timing = node.timing.ok_or_else(|| { + SchemaDerivationError::InvalidScalarSignature( + "execution timing is unassigned".into(), + ) + })?; + if timing == crate::post_asap::ExecutionTiming::IngestionTime + && node.children().iter().any(|child| { + child.timing != Some(crate::post_asap::ExecutionTiming::IngestionTime) + }) + { + return Err(SchemaDerivationError::InvalidScalarSignature( + "ingestion-time operation depends on a query-time or unassigned input".into(), + )); + } + } + Ok(()) + } + + /// Validate the whole DAG reachable from this node: every operator's + /// input contract, scalar typing against the owning operator's input + /// schema, and agreement between each retained schema and the one + /// derived from the operator. For an ASAP node the planner may retain + /// more specific column names, so only the field types must agree. + /// `timing` may be `None`. + pub fn validate_structure(self: &Rc) -> Result<(), SchemaDerivationError> { + for node in Self::reachable(self) { + if node.schema.time_index.is_some_and(|i| { + node.schema + .fields + .get(i) + .is_none_or(|f| f.plain_dtype() != Some(&crate::pre_asap::DataType::Timestamp)) + }) || node + .schema + .unique_keys + .iter() + .flatten() + .any(|i| *i >= node.schema.fields.len()) + { + return Err(SchemaDerivationError::InvalidScalarSignature( + "invalid time or identity column in schema".into(), + )); + } + node.operator.validate_inputs()?; + if node.result_kind != node.operator.output_kind() { + return Err(SchemaDerivationError::InvalidScalarSignature( + "retained result kind disagrees with operation".into(), + )); + } + let derived = node.operator.output_schema()?; + let agree = match &node.operator { + Operator::NonASAP(_) => derived == node.schema, + Operator::ASAP(_) => { + derived.fields.len() == node.schema.fields.len() + && derived + .fields + .iter() + .zip(&node.schema.fields) + .all(|(d, r)| d.dtype == r.dtype && d.nullable == r.nullable) + } + }; + if !agree { + return Err(SchemaDerivationError::InvalidScalarSignature(format!( + "retained schema of {} disagrees with its derived schema", + node.operator.kind_name() + ))); + } + } + Ok(()) + } +} diff --git a/crates/types/src/ir/non_asap.rs b/crates/types/src/ir/non_asap.rs new file mode 100644 index 000000000..2fd970b0b --- /dev/null +++ b/crates/types/src/ir/non_asap.rs @@ -0,0 +1,1624 @@ +//! Ordinary (non-ASAP) query operators: everything a front end emits and +//! everything that survives ASAP optimization unchanged. + +use std::rc::Rc; +use std::time::Duration; + +use serde::{Deserialize, Serialize}; + +use super::node::{OperatorNode, OperatorResultKind}; +use super::scalar::{Predicate, ProjectItem, ScalarExpr, SortKey}; +use crate::ir::aggregate_schema::aggregate_output_schema; +use crate::ir::operator_properties::{ + BinaryOpKind, ConcatDiscriminatorKey, GroupKeys, InfoMatcher, JoinKind, Reduction, + RelationalSetOpKind, SampleKind, Source, TimeShift, VectorMatch, WindowFrame, WindowFuncKind, +}; +use crate::ir::SchemaDerivationError; +use crate::pre_asap::agg_intent::AggIntent; +use crate::pre_asap::schema::{ColumnId, DataType, Field, FieldDataType, Schema}; + +/// All semantics owned by a binary operator. +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +pub struct BinaryOperator { + /// Execute division only for finite operands, a nonzero divisor, and a + /// normal finite result; otherwise use exact execution. Required by the + /// relative-value division certificate, including floating-point range. + #[serde(default)] + pub checked_relative_division: bool, + /// Conditional exact rewrites (such as temporal average from sum/count) + /// require finite operands and quotient. Zero/subnormal results are valid; + /// overflow must fall back to the original query rather than emit infinity. + #[serde(default)] + pub checked_finite_division: bool, + pub kind: BinaryOpKind, + /// `None` is the only currently supported vector/vector matching mode. + /// The field is retained so execution never has to recover semantics by + /// re-parsing PromQL. + pub vector_match: Option, +} + +/// Which samples a PromQL selector reads. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum TimeRangeKind { + /// An instant selector: `range` is the lookback horizon and the latest + /// eligible sample per series is selected. + Instant, + /// A range selector (`m[5m]`): every sample in the window. + Range, +} + +/// The non-ASAP operator vocabulary. Children are [`Rc`], so an +/// ordinary operator can read a summary evaluation and a summary can read any +/// relational sub-DAG. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum NonASAPOp { + /// Leaf. `schema` is the binding schema every positional `ColumnId` in + /// the tree indexes into; `predicates` are leaf-level row filters (PromQL + /// label matchers, pushed-down `WHERE` conjuncts). + Scan { + source: Source, + #[serde(default)] + predicates: Vec, + schema: Schema, + }, + /// SQL `VALUES` rows, or the one empty row of a `SELECT` without `FROM`. + /// Row expressions have no input-column scope. + Values { + rows: Vec>, + schema: Schema, + }, + /// σ — row-level filter. Output schema = child schema. + Filter { + pred: Predicate, + child: Rc, + }, + /// π — projection. + Project { + cols: Vec, + /// Re-qualifies every output column with this table alias (a derived + /// table / inline view). `None` for an ordinary SELECT list. + #[serde(default)] + qualifier: Option, + child: Rc, + }, + /// γ + α — grouping + aggregate intents. + Aggregate { + reduction: Reduction, + measures: Vec, + /// Output column names parallel to `measures`; an empty entry falls + /// back to the intent's synthetic name. + #[serde(default)] + output_names: Vec, + #[serde(default)] + filters: Vec>, + #[serde(default)] + having: Option, + child: Rc, + }, + Join { + kind: JoinKind, + pred: Predicate, + left: Rc, + right: Rc, + }, + SetOp { + kind: RelationalSetOpKind, + all: bool, + left: Rc, + right: Rc, + }, + /// ⊕ — exact n-ary `UNION ALL` of union-compatible branches. The output + /// schema is the first child's. + Concat { + children: Vec>, + #[serde(default)] + discriminator_unique_key: Option, + }, + /// δ — deduplication; empty `cols` = all columns. + Dedup { + cols: Vec, + child: Rc, + }, + /// Order-by, per `partition_by` group when non-empty. + Sort { + keys: Vec, + #[serde(default)] + partition_by: GroupKeys, + child: Rc, + }, + /// Row selection; `n = None` is offset-only. `partition_by` applies the + /// limit per group (PromQL `topk by (..)`). + Limit { + n: Option, + offset: usize, + #[serde(default)] + partition_by: GroupKeys, + child: Rc, + }, + /// Arithmetic / comparison / set composition of two operands (PromQL + /// binary operators). Mixed scalar/vector operations use Project or Filter. + BinaryOp { + operator: BinaryOperator, + /// PromQL `bool` modifier: a comparison returns `0`/`1` instead of + /// filtering. Valid only for comparison operators. + #[serde(default)] + return_bool: bool, + lhs: Rc, + rhs: Rc, + }, + /// SQL analytic window function. Output schema = child schema + one + /// column named `output_name`. + SQLWindowFunc { + func: WindowFuncKind, + args: Vec, + partition_by: GroupKeys, + order_by: Vec, + #[serde(default)] + frame: Option, + output_name: String, + child: Rc, + }, + /// Temporal selection over a time-series input. + TimeRange { + range: Duration, + kind: TimeRangeKind, + child: Rc, + }, + /// PromQL `offset` / `@`: moves when `child` is evaluated. + TimeShift { + shift: TimeShift, + child: Rc, + }, + /// PromQL `vector(s)`: a label-less instant vector carrying a scalar. + PromqlVectorFromScalar(ScalarExpr), + /// ρ — PromQL `label_replace` / `label_join`. + PromqlRelabel { + dst: String, + value: ScalarExpr, + child: Rc, + }, + /// PromQL `info(v, selector)` label enrichment. + PromqlInfoEnrich { + #[serde(default)] + selector: Vec, + child: Rc, + }, + /// PromQL `limitk` / `limit_ratio`. + PromqlSeriesSample { + #[serde(default)] + by: GroupKeys, + kind: SampleKind, + child: Rc, + }, + /// PromQL subquery `[range:resolution]`. + PromqlSubquery { + range: Duration, + #[serde(default)] + resolution: Option, + child: Rc, + }, +} + +impl NonASAPOp { + /// The direct operator inputs, in field order, followed by the operator + /// nodes referenced from this operator's scalar expressions. + pub fn children(&self) -> Vec<&Rc> { + use NonASAPOp::*; + let mut out: Vec<&Rc> = match self { + Scan { .. } | Values { .. } | PromqlVectorFromScalar(_) => vec![], + Filter { child, .. } + | Project { child, .. } + | Aggregate { child, .. } + | Dedup { child, .. } + | Sort { child, .. } + | Limit { child, .. } + | SQLWindowFunc { child, .. } + | TimeRange { child, .. } + | TimeShift { child, .. } + | PromqlRelabel { child, .. } + | PromqlInfoEnrich { child, .. } + | PromqlSeriesSample { child, .. } + | PromqlSubquery { child, .. } => vec![child], + Join { left, right, .. } | SetOp { left, right, .. } => vec![left, right], + BinaryOp { lhs, rhs, .. } => vec![lhs, rhs], + Concat { children, .. } => children.iter().collect(), + }; + for expr in self.scalar_exprs() { + out.extend(expr.operator_refs()); + } + out + } + + /// Every scalar expression this operator owns. + pub fn scalar_exprs(&self) -> Vec<&ScalarExpr> { + use NonASAPOp::*; + match self { + Scan { predicates, .. } => predicates.iter().map(|p| &p.0).collect(), + Values { rows, .. } => rows.iter().flatten().collect(), + Filter { pred, .. } | Join { pred, .. } => vec![&pred.0], + Project { cols, .. } => cols.iter().map(|c| &c.expr).collect(), + Aggregate { + filters, having, .. + } => filters + .iter() + .flatten() + .chain(having.iter()) + .map(|p| &p.0) + .collect(), + Sort { keys, .. } => keys.iter().map(|k| &k.expr).collect(), + SQLWindowFunc { args, order_by, .. } => args + .iter() + .chain(order_by.iter().map(|k| &k.expr)) + .collect(), + PromqlVectorFromScalar(e) => vec![e], + PromqlRelabel { value, .. } => vec![value], + SetOp { .. } + | Concat { .. } + | Dedup { .. } + | Limit { .. } + | BinaryOp { .. } + | TimeRange { .. } + | TimeShift { .. } + | PromqlInfoEnrich { .. } + | PromqlSeriesSample { .. } + | PromqlSubquery { .. } => vec![], + } + } + + /// Rebuild this operator with `f` applied to every child, including the + /// operator nodes referenced from scalar expressions. Every other field + /// is cloned. + pub fn map_children(&self, mut f: impl FnMut(&Rc) -> Rc) -> Self { + use NonASAPOp::*; + let mut map_scalar = |e: &ScalarExpr| e.map_operator_refs(&mut f); + fn map_pred(p: &Predicate, f: &mut impl FnMut(&ScalarExpr) -> ScalarExpr) -> Predicate { + Predicate(f(&p.0)) + } + fn map_keys( + keys: &[SortKey], + f: &mut impl FnMut(&ScalarExpr) -> ScalarExpr, + ) -> Vec { + keys.iter() + .map(|k| SortKey { + expr: f(&k.expr), + ascending: k.ascending, + nulls_first: k.nulls_first, + }) + .collect() + } + match self { + Scan { + source, + predicates, + schema, + } => Scan { + source: source.clone(), + predicates: predicates + .iter() + .map(|p| map_pred(p, &mut map_scalar)) + .collect(), + schema: schema.clone(), + }, + Values { rows, schema } => Values { + rows: rows + .iter() + .map(|row| row.iter().map(&mut map_scalar).collect()) + .collect(), + schema: schema.clone(), + }, + PromqlVectorFromScalar(e) => PromqlVectorFromScalar(map_scalar(e)), + Filter { pred, child } => { + let pred = map_pred(pred, &mut map_scalar); + Filter { + pred, + child: f(child), + } + } + Project { + cols, + qualifier, + child, + } => { + let cols = cols + .iter() + .map(|c| ProjectItem { + alias: c.alias.clone(), + expr: map_scalar(&c.expr), + }) + .collect(); + Project { + cols, + qualifier: qualifier.clone(), + child: f(child), + } + } + Aggregate { + reduction, + measures, + output_names, + filters, + having, + child, + } => { + let filters = filters + .iter() + .map(|p| p.as_ref().map(|p| map_pred(p, &mut map_scalar))) + .collect(); + let having = having.as_ref().map(|p| map_pred(p, &mut map_scalar)); + Aggregate { + reduction: reduction.clone(), + measures: measures.clone(), + output_names: output_names.clone(), + filters, + having, + child: f(child), + } + } + Join { + kind, + pred, + left, + right, + } => { + let pred = map_pred(pred, &mut map_scalar); + Join { + kind: kind.clone(), + pred, + left: f(left), + right: f(right), + } + } + SetOp { + kind, + all, + left, + right, + } => SetOp { + kind: kind.clone(), + all: *all, + left: f(left), + right: f(right), + }, + Concat { + children, + discriminator_unique_key, + } => Concat { + children: children.iter().map(&mut f).collect(), + discriminator_unique_key: discriminator_unique_key.clone(), + }, + Dedup { cols, child } => Dedup { + cols: cols.clone(), + child: f(child), + }, + Sort { + keys, + partition_by, + child, + } => { + let keys = map_keys(keys, &mut map_scalar); + Sort { + keys, + partition_by: partition_by.clone(), + child: f(child), + } + } + Limit { + n, + offset, + partition_by, + child, + } => Limit { + n: *n, + offset: *offset, + partition_by: partition_by.clone(), + child: f(child), + }, + BinaryOp { + operator, + return_bool, + lhs, + rhs, + } => BinaryOp { + operator: operator.clone(), + return_bool: *return_bool, + lhs: f(lhs), + rhs: f(rhs), + }, + SQLWindowFunc { + func, + args, + partition_by, + order_by, + frame, + output_name, + child, + } => { + let args = args.iter().map(&mut map_scalar).collect(); + let order_by = map_keys(order_by, &mut map_scalar); + SQLWindowFunc { + func: func.clone(), + args, + partition_by: partition_by.clone(), + order_by, + frame: frame.clone(), + output_name: output_name.clone(), + child: f(child), + } + } + TimeRange { range, kind, child } => TimeRange { + range: *range, + kind: *kind, + child: f(child), + }, + TimeShift { shift, child } => TimeShift { + shift: *shift, + child: f(child), + }, + PromqlRelabel { dst, value, child } => { + let value = map_scalar(value); + PromqlRelabel { + dst: dst.clone(), + value, + child: f(child), + } + } + PromqlInfoEnrich { selector, child } => PromqlInfoEnrich { + selector: selector.clone(), + child: f(child), + }, + PromqlSeriesSample { by, kind, child } => PromqlSeriesSample { + by: by.clone(), + kind: *kind, + child: f(child), + }, + PromqlSubquery { + range, + resolution, + child, + } => PromqlSubquery { + range: *range, + resolution: *resolution, + child: f(child), + }, + } + } + + /// The variant name, for diagnostics and export. + pub fn kind_name(&self) -> &'static str { + use NonASAPOp::*; + match self { + Scan { .. } => "Scan", + Values { .. } => "Values", + Filter { .. } => "Filter", + Project { .. } => "Project", + Aggregate { .. } => "Aggregate", + Join { .. } => "Join", + SetOp { .. } => "SetOp", + Concat { .. } => "Concat", + Dedup { .. } => "Dedup", + Sort { .. } => "Sort", + Limit { .. } => "Limit", + BinaryOp { .. } => "BinaryOp", + SQLWindowFunc { .. } => "SQLWindowFunc", + TimeRange { .. } => "TimeRange", + TimeShift { .. } => "TimeShift", + PromqlVectorFromScalar(_) => "PromqlVectorFromScalar", + PromqlRelabel { .. } => "PromqlRelabel", + PromqlInfoEnrich { .. } => "PromqlInfoEnrich", + PromqlSeriesSample { .. } => "PromqlSeriesSample", + PromqlSubquery { .. } => "PromqlSubquery", + } + } + + /// Output schema derived from this operator's parameters and its + /// children's (already derived) schemas. + pub fn output_schema(&self) -> Result { + use NonASAPOp::*; + Ok(match self { + Scan { schema, .. } | Values { schema, .. } => schema.clone(), + + Aggregate { + reduction, + measures, + output_names, + child, + filters, + .. + } => { + let mut output = + aggregate_output_schema(&child.schema, reduction, measures, output_names)?; + if child.result_kind == OperatorResultKind::Relation { + let offset = reduction.group_keys().map_or(0, |keys| keys.len()); + for (index, measure) in measures.iter().enumerate() { + if matches!( + measure, + AggIntent::Sum { .. } + | AggIntent::Avg { .. } + | AggIntent::Min { .. } + | AggIntent::Max { .. } + ) { + let nullable = offset == 0 + || filters.get(index).is_some_and(Option::is_some) + || measure + .input_cols() + .iter() + .any(|i| child.schema.fields[*i].nullable); + if let Some(field) = output.fields.get_mut(offset + index) { + field.nullable = nullable; + } + } + } + } + output + } + + Filter { child, .. } + | Sort { child, .. } + | Limit { child, .. } + | PromqlSubquery { child, .. } + | PromqlSeriesSample { child, .. } + | PromqlInfoEnrich { child, .. } + | TimeRange { child, .. } + | TimeShift { child, .. } => child.schema.clone(), + + // ρ — relabel preserves every input column and writes one label + // `dst` (Utf8): overwritten in place if it already exists, else + // appended (nullable). Row-uniqueness is no longer provable. + PromqlRelabel { dst, child, .. } => { + let mut out = child.schema.clone(); + if let Some(existing) = out.fields.iter_mut().find(|c| c.name == *dst) { + existing.dtype = FieldDataType::Plain(DataType::Utf8); + existing.nullable = true; + } else { + out.fields + .push(Field::plain(dst.clone(), DataType::Utf8, true)); + } + out.unique_keys.clear(); + out + } + + // π — one output column per projection item. A bare column item + // keeps its field verbatim, so an `ExactAggregate` state column + // can pass through a projection unchanged; any other expression + // is typed against the input and must read plain values. + Project { + cols, + qualifier, + child, + } => { + let in_schema = &child.schema; + let fields: Vec = cols + .iter() + .enumerate() + .map(|(i, item)| { + let mut field = match &item.expr { + ScalarExpr::Column(id) if in_schema.fields.get(*id).is_some() => { + let mut f = in_schema.fields[*id].clone(); + f.table = None; + f + } + expr => { + let (dtype, nullable) = expr.scalar_type(in_schema)?; + Field::plain(String::new(), dtype, nullable) + } + }; + field.name = item + .alias + .clone() + .unwrap_or_else(|| default_proj_name(&item.expr, i, in_schema)); + Ok(match qualifier { + Some(q) => field.with_table(q), + None => field, + }) + }) + .collect::, SchemaDerivationError>>()?; + let time_index = fields.iter().position(|c| c.name == "ts"); + let unique_keys = in_schema + .unique_keys + .iter() + .filter_map(|key| { + key.iter() + .map(|input_col| { + cols.iter().position(|item| { + matches!(&item.expr, ScalarExpr::Column(col) if col == input_col) + }) + }) + .collect::>>() + }) + .collect(); + Schema { + fields, + time_index, + unique_keys, + closed: true, + } + } + + Dedup { cols, child } => { + let mut out = child.schema.clone(); + if !cols.is_empty() { + out.add_unique_key(cols.clone()); + } + out + } + + Concat { + children, + discriminator_unique_key, + } => { + let mut s = children + .first() + .ok_or(SchemaDerivationError::EmptyConcat)? + .schema + .clone(); + s.unique_keys.clear(); + if let Some(key) = discriminator_unique_key { + let mut compound = vec![*key.discriminator()]; + compound.extend(key.inner_key().iter().copied()); + s.add_unique_key(compound); + } + s + } + SetOp { left, .. } => { + let mut s = left.schema.clone(); + s.unique_keys.clear(); + s + } + Join { + kind, left, right, .. + } => { + let l = &left.schema; + let r = &right.schema; + if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + return Ok(Schema { + unique_keys: Vec::new(), + ..l.clone() + }); + } + let (left_null, right_null) = match kind { + JoinKind::Left => (false, true), + JoinKind::Right => (true, false), + JoinKind::Full => (true, true), + JoinKind::Inner | JoinKind::Cross => (false, false), + JoinKind::Semi | JoinKind::Anti => unreachable!("handled above"), + }; + let l_len = l.fields.len(); + let mut fields = Vec::with_capacity(l_len + r.fields.len()); + fields.extend(l.fields.iter().cloned().map(|mut c| { + c.nullable |= left_null; + c + })); + fields.extend(r.fields.iter().cloned().map(|mut c| { + c.nullable |= right_null; + c + })); + let time_index = l.time_index.or(r.time_index.map(|i| i + l_len)); + Schema { + fields, + time_index, + unique_keys: Vec::new(), + closed: l.closed && r.closed, + } + } + SQLWindowFunc { + func, + args, + output_name, + child, + .. + } => { + let mut out = child.schema.clone(); + let arg = args.first().map(|a| a.scalar_type(&out)).transpose()?; + let arg_dtype = || { + arg.as_ref() + .map(|(ty, _)| ty.clone()) + .unwrap_or(DataType::Null) + }; + let (dtype, nullable) = match func { + WindowFuncKind::RowNumber + | WindowFuncKind::Rank + | WindowFuncKind::DenseRank + | WindowFuncKind::Count => (DataType::Int64, false), + WindowFuncKind::Sum => (arg_dtype(), true), + WindowFuncKind::Avg => (DataType::Float64, true), + WindowFuncKind::Lag + | WindowFuncKind::Lead + | WindowFuncKind::LagInFrame + | WindowFuncKind::LeadInFrame + | WindowFuncKind::FirstValue + | WindowFuncKind::LastValue + | WindowFuncKind::NthValue(_) => (arg_dtype(), true), + WindowFuncKind::Min | WindowFuncKind::Max => (arg_dtype(), true), + }; + out.fields + .push(Field::plain(output_name.clone(), dtype, nullable)); + out + } + + // `vector(s)` yields a label-less instant vector: the (ts, value) + // floor and nothing else; its full label set (empty) is known. + PromqlVectorFromScalar(_) => Schema { + fields: vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + ], + time_index: Some(0), + unique_keys: Vec::new(), + closed: true, + }, + + // The output shape of ` op ` is the vector side's: + // a scalar operand contributes only its value, no labels. A `bool` + // comparison still produces the vector's shape (values 0/1). + BinaryOp { + lhs, rhs, operator, .. + } => { + let mut output = lhs.schema.clone(); + let grouping = operator + .vector_match + .as_ref() + .and_then(|m| m.grouping.as_ref()); + let right_rows = matches!( + operator.kind, + BinaryOpKind::Set(crate::pre_asap::PromQLVectorSetOpKind::Or) + ) || matches!(grouping, Some(g) if g.side == crate::pre_asap::GroupSide::Right); + let mut additions = Vec::new(); + if right_rows { + additions.extend( + rhs.schema + .fields + .iter() + .filter(|c| c.plain_dtype() == Some(&DataType::Utf8)) + .cloned(), + ); + } + if let Some(grouping) = grouping { + additions.extend( + grouping + .labels + .iter() + .map(|name| Field::plain(name.clone(), DataType::Utf8, true)), + ); + } + for column in additions { + if !output.fields.iter().any(|c| c.name == column.name) { + output.fields.push(column); + } + } + output + } + }) + } + + /// The output category derived from this operator and its children. + pub fn output_kind(&self) -> OperatorResultKind { + use NonASAPOp::*; + match self { + Scan { source, .. } => match source { + Source::TimeSeries { .. } => OperatorResultKind::InstantVector, + Source::Table { .. } => OperatorResultKind::Relation, + }, + Values { .. } | SQLWindowFunc { .. } => OperatorResultKind::Relation, + TimeRange { child, .. } if child.result_kind == OperatorResultKind::Relation => { + OperatorResultKind::Relation + } + TimeRange { kind, .. } => match kind { + TimeRangeKind::Instant => OperatorResultKind::InstantVector, + TimeRangeKind::Range => OperatorResultKind::RangeVector, + }, + PromqlSubquery { .. } => OperatorResultKind::RangeVector, + PromqlVectorFromScalar(_) => OperatorResultKind::InstantVector, + // A per-entity range reduction turns a range vector into an + // instant vector; a cross-series reduction keeps its input's + // category (a SQL GROUP BY stays a relation). + Aggregate { + reduction, child, .. + } => match (reduction, child.result_kind) { + (Reduction::PerEntity, OperatorResultKind::RangeVector) + | (_, OperatorResultKind::State) => OperatorResultKind::InstantVector, + (_, kind) => kind, + }, + Project { child, cols, .. } if cols.iter().any(|item| matches!(item.expr, ScalarExpr::Column(i) if child.schema.fields.get(i).is_some_and(|f| !f.is_plain()))) => OperatorResultKind::State, + Filter { child, .. } + | Project { child, .. } + | Dedup { child, .. } + | Sort { child, .. } + | Limit { child, .. } + | TimeShift { child, .. } + | PromqlRelabel { child, .. } + | PromqlInfoEnrich { child, .. } + | PromqlSeriesSample { child, .. } => readable(child.result_kind), + Join { left, .. } | SetOp { left, .. } => readable(left.result_kind), + Concat { children, .. } => children + .first() + .map_or(OperatorResultKind::Relation, |c| readable(c.result_kind)), + BinaryOp { lhs, .. } => readable(lhs.result_kind), + } + } + + /// Local producer/consumer contract checks that need only this operator + /// and its children's output categories. + pub fn validate_inputs(&self) -> Result<(), SchemaDerivationError> { + use NonASAPOp::*; + let no_state = |node: &OperatorNode, what: &str| { + if node.result_kind == OperatorResultKind::State { + Err(SchemaDerivationError::InvalidScalarSignature(format!( + "{what} consumes summary state; read it out first" + ))) + } else { + Ok(()) + } + }; + let invalid = |message: &str| SchemaDerivationError::InvalidScalarSignature(message.into()); + let predicate = |pred: &Predicate, scope: &Schema| -> Result<(), SchemaDerivationError> { + if matches!( + pred.0.scalar_type(scope)?.0, + DataType::Bool | DataType::Null + ) { + Ok(()) + } else { + Err(invalid("predicate must be boolean")) + } + }; + let instant = |child: &OperatorNode| -> Result<(), SchemaDerivationError> { + if child.result_kind == OperatorResultKind::InstantVector { + Ok(()) + } else { + Err(invalid("operation requires an instant vector")) + } + }; + let columns = |cols: &[usize], scope: &Schema| -> Result<(), SchemaDerivationError> { + if cols.iter().any(|i| *i >= scope.fields.len()) { + Err(invalid("column outside operator input scope")) + } else { + Ok(()) + } + }; + match self { + Scan { + predicates, schema, .. + } => { + for pred in predicates { + predicate(pred, schema)?; + } + } + Values { rows, schema } => { + for row in rows { + if row.len() != schema.fields.len() { + return Err(invalid("Values row width differs from its schema")); + } + for (expr, field) in row.iter().zip(&schema.fields) { + let (ty, nullable) = expr.scalar_type(&Schema::default())?; + if field + .plain_dtype() + .is_none_or(|declared| *declared != ty && ty != DataType::Null) + || nullable && !field.nullable + { + return Err(invalid( + "Values expression differs from declared type/nullability", + )); + } + } + } + } + Filter { child, pred } => { + no_state(child, "Filter")?; + predicate(pred, &child.schema)?; + } + Project { child, cols, .. } => { + for col in cols { + if let ScalarExpr::Column(index) = col.expr { + columns(&[index], &child.schema)?; + } else { + col.expr.scalar_type(&child.schema)?; + } + } + } + BinaryOp { + lhs, + rhs, + operator, + return_bool, + } => { + no_state(lhs, "BinaryOp")?; + no_state(rhs, "BinaryOp")?; + if lhs.result_kind != rhs.result_kind + || lhs.result_kind == OperatorResultKind::RangeVector + { + return Err(invalid("binary operands have incompatible result kinds")); + } + if *return_bool && !matches!(operator.kind, BinaryOpKind::Compare(_)) { + return Err(invalid("bool mode requires a comparison")); + } + } + Join { + left, right, pred, .. + } => { + no_state(left, "Join")?; + no_state(right, "Join")?; + if left.result_kind != OperatorResultKind::Relation + || right.result_kind != OperatorResultKind::Relation + { + return Err(invalid("SQL join requires relations")); + } + let scope = Schema::new( + left.schema + .fields + .iter() + .chain(&right.schema.fields) + .cloned() + .collect(), + ); + predicate(pred, &scope)?; + } + SetOp { left, right, .. } => { + no_state(left, "SetOp")?; + no_state(right, "SetOp")?; + if left.result_kind != OperatorResultKind::Relation + || right.result_kind != OperatorResultKind::Relation + { + return Err(invalid("SQL set operation requires relations")); + } + } + Concat { children, .. } => { + for child in children { + no_state(child, "Concat")?; + } + } + Aggregate { + child, + filters, + measures, + having, + reduction, + .. + } => { + no_state(child, "Aggregate")?; + if let Reduction::Reduce(keys) = reduction { + columns(keys.keys(), &child.schema)?; + } + for measure in measures { + columns(&measure.input_cols(), &child.schema)?; + } + if !filters.is_empty() && filters.len() != measures.len() { + return Err(invalid("aggregate filter count differs from measure count")); + } + for pred in filters.iter().flatten() { + predicate(pred, &child.schema)?; + } + if let Some(pred) = having { + predicate(pred, &self.output_schema()?)?; + } + } + Dedup { child, cols } => { + no_state(child, "Dedup")?; + columns(cols, &child.schema)?; + } + Sort { + child, + keys, + partition_by, + } => { + columns(partition_by.keys(), &child.schema)?; + for key in keys { + key.expr.scalar_type(&child.schema)?; + } + } + Limit { + child, + partition_by, + .. + } => columns(partition_by.keys(), &child.schema)?, + SQLWindowFunc { + child, + args, + order_by, + partition_by, + .. + } => { + if child.result_kind != OperatorResultKind::Relation { + return Err(invalid("SQL window requires a relation")); + } + columns(partition_by.keys(), &child.schema)?; + for expr in args.iter().chain(order_by.iter().map(|k| &k.expr)) { + expr.scalar_type(&child.schema)?; + } + } + PromqlVectorFromScalar(expr) => { + if expr.scalar_type(&Schema::default())? != (DataType::Float64, false) { + return Err(invalid("vector() requires a non-null float scalar")); + } + } + PromqlSubquery { child, .. } + | PromqlInfoEnrich { child, .. } + | PromqlSeriesSample { child, .. } => instant(child)?, + PromqlRelabel { child, value, .. } => { + instant(child)?; + if value.scalar_type(&child.schema)?.0 != DataType::Utf8 { + return Err(invalid("label expression requires a string")); + } + } + TimeRange { child, .. } => { + if child.result_kind != OperatorResultKind::Relation { + instant(child)?; + } + } + TimeShift { child, .. } => { + if !matches!( + child.result_kind, + OperatorResultKind::InstantVector | OperatorResultKind::RangeVector + ) { + return Err(invalid("time shift requires a vector")); + } + } + } + Ok(()) + } +} + +/// A evaluation-shaped category for a value-level operator over `kind`: state +/// never flows through an ordinary operator unchanged in category. +fn readable(kind: OperatorResultKind) -> OperatorResultKind { + match kind { + OperatorResultKind::State => OperatorResultKind::Relation, + other => other, + } +} + +/// Default output-column name for a projection item with no explicit alias: +/// a bare column keeps its (schema) name; anything else gets `col_{i}`. +fn default_proj_name(expr: &ScalarExpr, idx: usize, schema: &Schema) -> String { + match expr { + ScalarExpr::Column(id) => schema + .fields + .get(*id) + .map(|c| c.name.clone()) + .unwrap_or_else(|| format!("col_{idx}")), + _ => format!("col_{idx}"), + } +} + +/// Whether any aggregate measure has its own input predicate. +pub fn any_measure_filtered(filters: &[Option]) -> bool { + filters.iter().any(Option::is_some) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ir::operator_properties::{ + AtModifier, VectorMatchKind, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, + }; + use crate::ir::scalar::ExprSemantics; + use crate::pre_asap::expr_ir::{ArithmeticOpKind, CompareOpKind, ScalarValue}; + + fn col(name: &str, dtype: DataType, nullable: bool) -> Field { + Field::plain(name, dtype, nullable) + } + + fn node(op: NonASAPOp) -> Rc { + OperatorNode::non_asap_node(op).unwrap() + } + + fn scan( + columns: Vec, + time_index: Option, + uk: Vec>, + ) -> NonASAPOp { + NonASAPOp::Scan { + source: Source::Table { + table_ref: "t".into(), + }, + predicates: vec![], + schema: Schema { + fields: columns, + time_index, + unique_keys: uk, + closed: true, + }, + } + } + + /// `[ts, value, job]` time-series leaf; open, as a PromQL leaf is. + fn series_scan() -> NonASAPOp { + NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + col("ts", DataType::Timestamp, false), + col("value", DataType::Float64, false), + col("job", DataType::Utf8, true), + ], + 0, + vec![], + ), + } + } + + fn item(alias: Option<&str>, expr: ScalarExpr) -> ProjectItem { + ProjectItem { + alias: alias.map(Into::into), + expr, + } + } + + fn add(left: ScalarExpr, right: ScalarExpr) -> ScalarExpr { + ScalarExpr::Arithmetic { + op: ArithmeticOpKind::Add, + left: Box::new(left), + right: Box::new(right), + semantics: ExprSemantics::Sql, + } + } + + fn project(cols: Vec, child: NonASAPOp) -> NonASAPOp { + NonASAPOp::Project { + cols, + qualifier: None, + child: node(child), + } + } + + fn dedup_branch(columns: Vec) -> Rc { + node(NonASAPOp::Dedup { + cols: vec![0], + child: node(scan(columns, None, vec![])), + }) + } + + fn concat( + children: Vec>, + discriminator_unique_key: Option, + ) -> NonASAPOp { + NonASAPOp::Concat { + children, + discriminator_unique_key, + } + } + + fn rate_over(child: NonASAPOp) -> NonASAPOp { + NonASAPOp::Aggregate { + reduction: Reduction::PerEntity, + measures: vec![AggIntent::Rate], + output_names: vec![], + filters: vec![], + having: None, + child: node(child), + } + } + + #[test] + fn project_preserves_unique_keys_that_are_passed_through() { + let input = scan( + vec![ + col("tenant", DataType::Utf8, false), + col("region", DataType::Utf8, false), + col("value", DataType::Int64, false), + ], + None, + vec![vec![0, 1]], + ); + let projected = project( + vec![ + item(Some("r"), ScalarExpr::Column(1)), + item(Some("t"), ScalarExpr::Column(0)), + item( + None, + add( + ScalarExpr::Column(2), + ScalarExpr::Literal(ScalarValue::Int64(1)), + ), + ), + ], + input, + ); + assert_eq!( + projected.output_schema().unwrap().unique_keys, + vec![vec![1, 0]] + ); + } + + #[test] + fn project_drops_a_unique_key_when_a_key_column_is_omitted() { + let input = scan( + vec![ + col("tenant", DataType::Utf8, false), + col("region", DataType::Utf8, false), + ], + None, + vec![vec![0, 1]], + ); + let projected = project(vec![item(None, ScalarExpr::Column(0))], input); + assert!(projected.output_schema().unwrap().unique_keys.is_empty()); + } + + #[test] + fn project_retypes_and_renames_per_item() { + let child = scan( + vec![ + col("ts", DataType::Timestamp, false), + col("host", DataType::Utf8, false), + col("value", DataType::Float64, false), + ], + Some(0), + vec![vec![0, 1]], + ); + let q = project( + vec![ + // A bare column keeps its name and type. + item(None, ScalarExpr::Column(1)), + item( + Some("dbl"), + add(ScalarExpr::Column(2), ScalarExpr::Column(2)), + ), + // A comparison is a nullable Bool under 3-valued logic. + item( + Some("flag"), + ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(2)), + op: CompareOpKind::Gt, + right: Box::new(ScalarExpr::Literal(ScalarValue::Float64(0.0))), + semantics: ExprSemantics::Sql, + }, + ), + ], + child, + ); + let s = q.output_schema().unwrap(); + assert_eq!(s.fields.len(), 3); + assert_eq!(s.fields[0], col("host", DataType::Utf8, false)); + assert_eq!(s.fields[1], col("dbl", DataType::Float64, false)); + assert_eq!(s.fields[2], col("flag", DataType::Bool, false)); + // `ts` is not retained: no time axis, and the key is lost. + assert!(s.time_index.is_none()); + assert!(s.unique_keys.is_empty()); + } + + #[test] + fn project_keeps_time_index_when_ts_passed_through() { + let child = scan( + vec![ + col("ts", DataType::Timestamp, false), + col("value", DataType::Float64, false), + ], + Some(0), + vec![], + ); + let q = project( + vec![ + item(None, ScalarExpr::Column(1)), + item(None, ScalarExpr::Column(0)), + ], + child, + ); + let s = q.output_schema().unwrap(); + assert_eq!(s.fields[0].name, "value"); + assert_eq!(s.fields[1].name, "ts"); + assert_eq!(s.time_index, Some(1)); + } + + #[test] + fn legacy_window_json_without_frame_deserializes_as_unspecified() { + let window = NonASAPOp::SQLWindowFunc { + func: WindowFuncKind::RowNumber, + args: vec![], + partition_by: GroupKeys::by(vec![]), + order_by: vec![], + frame: Some(WindowFrame { + units: WindowFrameUnits::Range, + start_bound: WindowFrameBound::Preceding(WindowFrameOffset::Scalar( + ScalarValue::Null, + )), + end_bound: WindowFrameBound::CurrentRow, + }), + output_name: "row_number".into(), + child: node(scan(vec![col("v", DataType::Int64, false)], None, vec![])), + }; + let mut json = serde_json::to_value(window).unwrap(); + json.get_mut("SQLWindowFunc") + .and_then(serde_json::Value::as_object_mut) + .unwrap() + .remove("frame"); + let decoded: NonASAPOp = serde_json::from_value(json).unwrap(); + assert!(matches!( + decoded, + NonASAPOp::SQLWindowFunc { frame: None, .. } + )); + } + + /// A row can appear in more than one branch, so no branch's unique key is + /// a key of the union — `unique_keys` feeds CSE's sharing legality check. + #[test] + fn merge_drops_the_branches_unique_keys() { + let branch = || { + dedup_branch(vec![ + col("k", DataType::Utf8, false), + col("v", DataType::Int64, false), + ]) + }; + assert_eq!( + branch().schema.unique_keys, + vec![vec![0]], + "a Dedup branch does have a unique key on its own" + ); + let schema = concat(vec![branch(), branch()], None) + .output_schema() + .unwrap(); + assert!( + schema.unique_keys.is_empty(), + "the union of two deduplicated branches is not deduplicated" + ); + assert_eq!(schema.fields.len(), 2, "column shape is the first branch's"); + } + + #[test] + fn merge_and_setop_agree_on_unique_keys() { + let branch = || dedup_branch(vec![col("k", DataType::Utf8, false)]); + let merged = concat(vec![branch(), branch()], None); + let setop = NonASAPOp::SetOp { + kind: RelationalSetOpKind::Union, + all: true, + left: branch(), + right: branch(), + }; + assert_eq!( + merged.output_schema().unwrap().unique_keys, + setop.output_schema().unwrap().unique_keys, + ); + } + + #[test] + fn an_empty_merge_has_no_schema() { + assert!(matches!( + concat(vec![], None).output_schema(), + Err(SchemaDerivationError::EmptyConcat) + )); + } + + /// Issue #228: an asserted discriminator yields the compound + /// `(discriminator, inner_key)` unique key, although each branch's own + /// `inner_key` repeats across branches. + #[test] + fn discriminator_override_produces_a_compound_unique_key() { + let branch = || { + dedup_branch(vec![ + col("k", DataType::Utf8, false), + col("branch_id", DataType::Int64, false), + ]) + }; + let schema = concat( + vec![branch(), branch()], + Some(ConcatDiscriminatorKey::new(1, vec![0])), + ) + .output_schema() + .unwrap(); + assert_eq!(schema.unique_keys, vec![vec![1, 0]]); + assert_eq!(schema.fields.len(), 2, "column shape is the first branch's"); + } + + /// Without a named discriminator a `Concat` never claims a unique key. + #[test] + fn no_way_to_fabricate_a_unique_key_without_naming_a_discriminator() { + let branch = || dedup_branch(vec![col("k", DataType::Utf8, false)]); + assert!(concat(vec![branch(), branch()], None) + .output_schema() + .unwrap() + .unique_keys + .is_empty()); + } + + #[test] + fn without_aggregate_keeps_open_schema_minus_excluded() { + // `sum without (instance) (m)` over `[ts, value, instance, job]`. + let leaf = NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + col("ts", DataType::Timestamp, false), + col("value", DataType::Float64, false), + col("instance", DataType::Utf8, true), + col("job", DataType::Utf8, true), + ], + 0, + vec![], + ), + }; + let agg = NonASAPOp::Aggregate { + reduction: Reduction::Reduce(GroupKeys::without(vec![2])), + measures: vec![AggIntent::Sum { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: node(leaf), + }; + let s = agg.output_schema().unwrap(); + let names: Vec<_> = s.fields.iter().map(|c| c.name.as_str()).collect(); + assert_eq!(names, vec!["job", "sum"], "kept `job`, dropped `instance`"); + assert!(!s.closed, "a `without` result stays open"); + assert!(s.time_index.is_none()); + assert!(s.unique_keys.is_empty(), "kept set unknown → no unique key"); + } + + #[test] + fn time_shift_is_schema_pass_through() { + let leaf = node(scan( + vec![ + col("ts", DataType::Timestamp, false), + col("value", DataType::Float64, false), + col("job", DataType::Utf8, true), + ], + Some(0), + vec![], + )); + let shifted = NonASAPOp::TimeShift { + shift: TimeShift { + offset_ms: 3_600_000, + at: Some(AtModifier::Timestamp(1_609_746_000_000)), + }, + child: Rc::clone(&leaf), + }; + assert_eq!(shifted.output_schema().unwrap(), leaf.schema); + } + + /// `rate` and `*_over_time` are per-series: every label survives and only + /// the sample value is replaced (kept named `value`). + #[test] + fn per_series_reductions_preserve_labels() { + for measure in [AggIntent::Rate, AggIntent::Avg { col: None }] { + let reduced = NonASAPOp::Aggregate { + reduction: Reduction::PerEntity, + measures: vec![measure], + output_names: vec![], + filters: vec![], + having: None, + child: node(NonASAPOp::TimeRange { + range: Duration::from_secs(300), + kind: TimeRangeKind::Range, + child: node(series_scan()), + }), + }; + let s = reduced.output_schema().unwrap(); + let names: Vec<_> = s.fields.iter().map(|c| c.name.as_str()).collect(); + assert_eq!(names, vec!["ts", "value", "job"]); + assert_eq!(s.time_index, Some(0)); + } + } + + /// An open leaf stays open through a per-series `rate` and is frozen to + /// closed by a cross-series aggregate. + #[test] + fn completeness_open_leaf_freezes_to_closed_at_cross_series_aggregate() { + let leaf = series_scan(); + assert!( + !leaf.output_schema().unwrap().closed, + "schemaless leaf is open" + ); + let rate = rate_over(leaf); + assert!(!rate.output_schema().unwrap().closed, "rate stays open"); + let sum_by_job = NonASAPOp::Aggregate { + reduction: Reduction::by(vec![2]), + measures: vec![AggIntent::Sum { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: node(rate), + }; + assert!(sum_by_job.output_schema().unwrap().closed); + } + + fn join(kind: JoinKind) -> NonASAPOp { + NonASAPOp::Join { + kind, + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + left: node(scan( + vec![col("a", DataType::Int64, false)], + None, + vec![vec![0]], + )), + right: node(scan(vec![col("b", DataType::Utf8, false)], None, vec![])), + } + } + + #[test] + fn inner_join_concatenates_both_sides() { + let s = join(JoinKind::Inner).output_schema().unwrap(); + assert_eq!( + s.fields, + vec![ + col("a", DataType::Int64, false), + col("b", DataType::Utf8, false) + ] + ); + assert!( + s.unique_keys.is_empty(), + "post-join row identity not provable" + ); + } + + #[test] + fn left_join_makes_right_side_nullable() { + let s = join(JoinKind::Left).output_schema().unwrap(); + assert!(!s.fields[0].nullable); + assert!(s.fields[1].nullable); + } + + #[test] + fn full_join_makes_both_sides_nullable() { + let s = join(JoinKind::Full).output_schema().unwrap(); + assert!(s.fields[0].nullable); + assert!(s.fields[1].nullable); + } + + #[test] + fn setop_takes_left_shape_and_drops_unique_keys() { + let side = || { + node(scan( + vec![ + col("k", DataType::Utf8, false), + col("v", DataType::Int64, false), + ], + None, + vec![vec![0]], + )) + }; + let s = NonASAPOp::SetOp { + kind: RelationalSetOpKind::Union, + all: false, + left: side(), + right: side(), + } + .output_schema() + .unwrap(); + assert_eq!(s.fields.len(), 2); + assert_eq!(s.fields[0].name, "k"); + assert!( + s.unique_keys.is_empty(), + "UNION does not preserve row identity" + ); + } + + /// Standalone constants are scalar roots and carry no operator schema. + #[test] + fn constant_is_a_scalar_root() { + let root = crate::ir::QueryRoot::Scalar(ScalarExpr::literal_f64(42.0)); + assert!(root.as_operator().is_none()); + } + + /// ` op ` takes the vector side's schema; the vector + /// match modifier is kept on the operator. + #[test] + fn binary_op_schema_follows_the_vector_side_over_a_scalar_bridge() { + let vector = node(scan( + vec![ + col("host", DataType::Utf8, false), + col("value", DataType::Float64, false), + ], + None, + vec![], + )); + let vm = VectorMatch { + kind: VectorMatchKind::On, + labels: vec!["host".into()], + grouping: None, + }; + let op = NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: BinaryOpKind::Compare(CompareOpKind::Gt), + vector_match: Some(vm.clone()), + }, + return_bool: false, + lhs: Rc::clone(&vector), + rhs: Rc::clone(&vector), + }; + assert_eq!(op.output_schema().unwrap(), vector.schema); + let NonASAPOp::BinaryOp { operator, .. } = &op else { + unreachable!() + }; + assert_eq!(operator.vector_match.as_ref(), Some(&vm)); + } +} diff --git a/crates/types/src/ir/operator_properties.rs b/crates/types/src/ir/operator_properties.rs new file mode 100644 index 000000000..ef0c7baef --- /dev/null +++ b/crates/types/src/ir/operator_properties.rs @@ -0,0 +1,558 @@ +//! Supporting parameter types used inside operator payloads. +//! +//! For example, `Aggregate.by` uses [`GroupKeys`], a join chooses [`JoinKind`], +//! and a SQL window carries [`WindowFrame`]. These types describe what an +//! operator does. Derived node metadata (schema, guarantee, timing) lives on +//! [`super::OperatorNode`], not in this module. +use crate::pre_asap::{ArithmeticOpKind, ColumnId, ColumnRef, CompareOpKind, ScalarValue}; +use serde::{Deserialize, Serialize}; +/// The column-reference type an operator parameter is generic over: +/// positional [`ColumnId`] once bound, name-based [`ColumnRef`] before. +pub trait ColState: + Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de> +{ +} + +impl ColState for ColumnId {} + +impl ColState for ColumnRef {} + +// ── Leaf / supporting types ─────────────────────────────────────────────────── + +/// Positional grouping keys, shared by every "operate per group" operator: +/// `Aggregate.by` (reduce per group), `Sort.partition_by` (rank per group — +/// including generic `topk`/`bottomk`), and `SQLWindowFunc.partition_by` (window +/// per group). One spelling so grouping has a single home to evolve. Empty +/// (and `by`) = no grouping (a global operation). +/// +/// Heavy-hitter `AggIntent::TopK` carries its grouping here too, via the +/// enclosing `Aggregate.by` (issue #13) — so reduce, rank, and window groupings +/// all share this one type. +/// +/// ## `by` vs `without` (issue #39) +/// +/// The stored [`keys`](Self::keys) are **kept** labels for `by(...)` and +/// **excluded** labels for `without(...)`. PromQL's `without(labels)` groups by +/// every label *except* those listed; the complement can't be enumerated at +/// lowering time under an open (usage-derived) schema, so it is deferred to the +/// runtime — the excluded positions are stored, the kept set stays open. Only +/// `Aggregate` ever produces the `without` form; `Sort` / `SQLWindowFunc` / +/// `PromqlSeriesSample` groupings are always `by`. +/// +/// Serialises as a bare array for the (overwhelmingly common) `by` case — +/// wire-compatible with the `Vec` this field held before — and as +/// `{"without": [...]}` for the exclusion case. +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub struct GroupKeys { + keys: Vec, + without: bool, +} + +// Not `#[derive(Default)]`: derive would add a `C: Default` bound, but an +// empty key set needs nothing from `C` — `ColumnRef` has no meaningful +// default anyway. +impl Default for GroupKeys { + fn default() -> Self { + Self { + keys: Vec::new(), + without: false, + } + } +} + +impl GroupKeys { + /// An empty key set — a global (ungrouped) operation. + pub fn none() -> Self { + Self::default() + } + /// `by(keys)` — group by exactly these columns. + pub fn by(keys: Vec) -> Self { + Self { + keys, + without: false, + } + } + /// `without(keys)` — group by every label *except* these (issue #39). The + /// kept set is runtime-resolved; only the excluded positions are stored. + pub fn without(keys: Vec) -> Self { + Self { + keys, + without: true, + } + } + /// Whether this is a `without(...)` exclusion grouping. + pub fn is_without(&self) -> bool { + self.without + } + /// The named keys — kept labels for `by`, excluded labels for `without`. + pub fn keys(&self) -> &[C] { + &self.keys + } +} + +impl std::ops::Deref for GroupKeys { + type Target = [C]; + fn deref(&self) -> &Self::Target { + &self.keys + } +} + +impl From> for GroupKeys { + fn from(keys: Vec) -> Self { + Self::by(keys) + } +} + +impl FromIterator for GroupKeys { + fn from_iter>(iter: I) -> Self { + Self::by(iter.into_iter().collect()) + } +} + +impl<'a, C> IntoIterator for &'a GroupKeys { + type Item = &'a C; + type IntoIter = std::slice::Iter<'a, C>; + fn into_iter(self) -> Self::IntoIter { + self.keys.iter() + } +} + +/// Compare directly against a `Vec` so call sites and tests can keep +/// writing `keys == vec![..]` / `assert_eq!(keys, &vec![..])`. A `without` +/// grouping never equals a bare `by` list. +impl PartialEq> for GroupKeys { + fn eq(&self, other: &Vec) -> bool { + !self.without && &self.keys == other + } +} + +/// (De)serialise as a bare array for `by`, or `{"without": [...]}` for the +/// exclusion form — keeping the `by` wire format identical to the old newtype. +/// Borrowed for `Serialize` (no `C: Clone` needed to write one out), owned for +/// `Deserialize` (there's nothing to borrow from). +#[derive(Serialize)] +#[serde(untagged)] +enum GroupKeysReprRef<'a, C> { + By(&'a [C]), + Without { without: &'a [C] }, +} + +#[derive(Deserialize)] +#[serde(untagged)] +enum GroupKeysRepr { + By(Vec), + Without { without: Vec }, +} + +impl Serialize for GroupKeys { + fn serialize(&self, serializer: S) -> Result { + if self.without { + GroupKeysReprRef::Without { + without: self.keys.as_slice(), + } + .serialize(serializer) + } else { + GroupKeysReprRef::By(self.keys.as_slice()).serialize(serializer) + } + } +} + +impl<'de, C: Deserialize<'de>> Deserialize<'de> for GroupKeys { + fn deserialize>(deserializer: D) -> Result { + Ok(match GroupKeysRepr::deserialize(deserializer)? { + GroupKeysRepr::By(keys) => Self::by(keys), + GroupKeysRepr::Without { without } => Self::without(without), + }) + } +} + +/// Which data model a `Source` / `AggIntent` operates over. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum DataModel { + TimeSeries, + Tabular, + Any, +} + +/// The leaf data source of a `Scan`. The schema itself rides on the +/// `Scan.schema` field (SchemaResolver-built); `Source` carries only the leaf's +/// identity. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum Source { + /// Time-series leaf — PromQL / DC lifecycle. Produces `(ts, value, *labels)`. + TimeSeries { metric: String }, + /// Tabular leaf — asap-fusion / future OLAP. Columns ride on `Scan.schema`. + Table { table_ref: String }, +} + +impl Source { + pub fn data_model(&self) -> DataModel { + match self { + Source::TimeSeries { .. } => DataModel::TimeSeries, + Source::Table { .. } => DataModel::Tabular, + } + } +} + +/// Operator on the query-level `BinaryOp` node. Reuses the scalar IR's +/// [`ArithmeticOpKind`] / [`CompareOpKind`] so every arithmetic/comparison +/// operator has exactly one representation (and one `Display`) across the IR. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum BinaryOpKind { + /// Arithmetic — `Add/Sub/Mul/Div/Mod` (shared with `ScalarExpr::Arithmetic`). + Arithmetic(ArithmeticOpKind), + /// Comparison — `Eq/Ne/Lt/Le/Gt/Ge` + `Like/ILike/Regex` family (shared + /// with `ScalarExpr::Compare`). + Compare(CompareOpKind), + /// PromQL vector-set operation. + Set(PromQLVectorSetOpKind), +} + +impl std::fmt::Display for BinaryOpKind { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + BinaryOpKind::Arithmetic(op) => write!(f, "{op}"), + BinaryOpKind::Compare(op) => write!(f, "{op}"), + BinaryOpKind::Set(PromQLVectorSetOpKind::And) => f.write_str("AND"), + BinaryOpKind::Set(PromQLVectorSetOpKind::Or) => f.write_str("OR"), + BinaryOpKind::Set(PromQLVectorSetOpKind::Unless) => f.write_str("unless"), + } + } +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum JoinKind { + Inner, + Left, + Right, + Full, + Cross, + /// Left semi-join — each left row that has **at least one** match, once. + /// `WHERE c IN (SELECT …)` / `WHERE EXISTS (…)` (issue #111). + /// + /// Output schema is the **left's alone**; the right side is a filter, not a + /// source of columns. The join predicate still resolves against the + /// concatenated `left ++ right` schema — its scope is deliberately wider + /// than the node's output. + Semi, + /// Left anti-join — each left row with **no** match. `WHERE NOT EXISTS (…)`. + /// Same schema rule as [`JoinKind::Semi`]. + /// + /// Note this is *not* `NOT IN (SELECT …)`: under SQL's three-valued logic a + /// NULL on the right makes `NOT IN` yield no rows at all, where an anti-join + /// yields every left row. The SQL front end rejects `NOT IN (subquery)` + /// rather than lower it here. + Anti, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum RelationalSetOpKind { + Union, + Intersect, + Except, +} + +/// PromQL vector-set operator used by [`BinaryOpKind::Set`]. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum PromQLVectorSetOpKind { + And, + Or, + Unless, +} + +/// SQL analytic window function (`fn(...) OVER (…)`). Distinct from a streaming +/// time `Window`: this is an analytic frame over already-materialised rows. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum WindowFuncKind { + RowNumber, + Rank, + DenseRank, + Lag, + Lead, + /// ClickHouse `lagInFrame`/`leadInFrame`: unlike [`Lag`](Self::Lag)/[`Lead`](Self::Lead), + /// these respect the window frame bounds (NULL/default past the frame edge) + /// rather than reaching arbitrarily far back/forward. Kept as distinct + /// variants so the frame clause is never silently discarded by conflating + /// them with `Lag`/`Lead` (#267). `WindowFuncKind` still has no frame + /// representation, so today these lower and behave exactly like + /// `Lag`/`Lead` — the tag is correct, the frame-respecting behavior isn't + /// implemented yet. See #231 for modeling window frames properly. + LagInFrame, + LeadInFrame, + FirstValue, + LastValue, + /// `NTH_VALUE(expr, n)` — `n` is resolved from the (literal) 2nd argument. + NthValue(Option), + Sum, + Avg, + Count, + Min, + Max, +} + +/// A window's frame-spec (`ROWS`/`RANGE BETWEEN … AND …`) — which rows around +/// the current one an analytic window function reads. `GROUPS` is rejected at +/// lowering time (issue #268): every SQL corpus in this repo uses only `ROWS`, +/// and nothing downstream interprets frame semantics yet, so it isn't worth +/// modelling untested. +/// +/// Meaningless (but harmless) on the rank-only and navigation functions +/// (`ROW_NUMBER`/`RANK`/`DENSE_RANK`/`LAG`/`LEAD`), which ignore the frame per +/// SQL semantics — DataFusion still attaches one, stored here verbatim. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct WindowFrame { + pub units: WindowFrameUnits, + pub start_bound: WindowFrameBound, + pub end_bound: WindowFrameBound, +} + +/// A finite window-frame displacement. Intervals are normalized to Arrow's +/// month/day/nanosecond representation so SQL `RANGE INTERVAL ...` bounds +/// survive lowering without leaking DataFusion types into the canonical IR. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum WindowFrameOffset { + Scalar(ScalarValue), + Interval { + months: i32, + days: i32, + nanoseconds: i64, + }, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum WindowFrameUnits { + /// Boundaries count physical rows: `ROWS BETWEEN 2 PRECEDING AND CURRENT ROW`. + Rows, + /// Boundaries count by value-distance on the (single) `ORDER BY` column: + /// `RANGE BETWEEN INTERVAL '1' HOUR PRECEDING AND CURRENT ROW`. + Range, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum WindowFrameBound { + /// `UNBOUNDED PRECEDING` is + /// `Preceding(WindowFrameOffset::Scalar(ScalarValue::Null))`. + Preceding(WindowFrameOffset), + CurrentRow, + /// `UNBOUNDED FOLLOWING` is + /// `Following(WindowFrameOffset::Scalar(ScalarValue::Null))`. + Following(WindowFrameOffset), +} + +/// A symbolic label matcher on the **info metric** side of an +/// [`crate::ir::NonASAPOp::PromqlInfoEnrich`] (issue #84). Unlike a `Scan` predicate it is not +/// resolved positionally — it references the info metric's labels (`__name__` +/// picks the metric, the rest constrain data labels), which aren't in the input +/// vector's schema; the post-ASAP realization pass applies it against the info metric. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct InfoMatcher { + pub label: String, + /// One of `Eq` / `Ne` / `Regex` / `NotRegex` (PromQL `=`/`!=`/`=~`/`!~`). + pub op: CompareOpKind, + pub value: String, +} + +/// Series-sampling selection mode (PromQL `limitk` / `limit_ratio`, issue #86). +/// A [`crate::ir::NonASAPOp::PromqlSeriesSample`] keeps a *subset of whole series*, unchanged — it does +/// not rank or reduce, so it is distinct from `TopK` and from `Sort → Limit`. +#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum SampleKind { + /// `limitk(k, v)` — up to `k` series per group. Which series survive is + /// deterministic across evaluations but otherwise unspecified (no ordering). + LimitK(usize), + /// `limit_ratio(r, v)` — a deterministic `r`-fraction of series per group. + /// `r ∈ [-1, 1]`; a negative `r` selects the complementary fraction. + LimitRatio(f64), +} + +/// PromQL vector-match modifier (`on`/`ignoring` + `group_left`/`group_right`). +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct VectorMatch { + pub kind: VectorMatchKind, + pub labels: Vec, + pub grouping: Option, +} + +/// PromQL `@` modifier — pins a selector's evaluation time to an anchor instead +/// of the query evaluation time (issue #40). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +pub enum AtModifier { + /// `@ start()` — the query range's start instant. + Start, + /// `@ end()` — the query range's end instant. + End, + /// `@ ` — an absolute instant, milliseconds since the Unix epoch (may be + /// negative). PromQL writes the timestamp in seconds; the front end scales it. + Timestamp(i64), +} + +/// PromQL per-selector **time-shift** modifiers — `offset` and `@` (issue #40). +/// Neither changes a selector's *schema*; both move *when* it is evaluated, so +/// the shift is a pass-through wrapper ([`crate::ir::NonASAPOp::TimeShift`]) over the +/// selector rather than a new leaf shape. The runtime resolves the anchor and +/// applies the offset. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)] +pub struct TimeShift { + /// `offset ` as signed milliseconds — a positive value shifts the + /// lookback *back* in time (`offset 5m`), a negative value shifts it + /// *forward* (`offset -5m`). `0` = no offset. + pub offset_ms: i64, + /// `@` anchor; `None` = evaluate at the query time. + pub at: Option, +} + +impl TimeShift { + /// Whether this shift is the identity (no `offset`, no `@`) — the state of + /// every selector that carries neither modifier. + pub fn is_identity(&self) -> bool { + self.offset_ms == 0 && self.at.is_none() + } +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum VectorMatchKind { + On, + Ignoring, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct VectorGrouping { + pub side: GroupSide, + pub labels: Vec, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum GroupSide { + Left, + Right, +} + +// ── Intent algebra IR ──────────────────────────────────────────────────────── + +/// What kind of computation an `Aggregate` node performs — orthogonal to +/// *which* columns it groups by (that's still [`GroupKeys`], inside +/// `Reduce`). Explicit, decided once by whichever pass constructs the node +/// (structural, at front-end lowering time), rather than inferred downstream from +/// whether a grouping-key list happens to be empty or from a neighboring +/// node's shape. See design proposal #165. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum Reduction { + /// Collapses input rows via `by` — `by`/`without` semantics are exactly + /// [`GroupKeys`]'s. May still collapse every row into one (an empty, + /// non-`without` `by`) — that's a genuine reduction with zero grouping + /// columns, not "no grouping concept." + Reduce(GroupKeys), + /// No grouping concept at all: preserves one output row per input + /// entity (e.g. a per-series windowed computation with no `by(...)` + /// clause to begin with, because there's no aggregation operator here + /// for such a clause to attach to). Never merges across entities, and + /// never collapses an entity's own row structure (e.g. a time axis) — + /// unlike `Reduce(GroupKeys::without(vec![]))` ("group by every + /// label"), which is still a genuine reduction and does collapse it. + PerEntity, +} + +impl Reduction { + /// Shorthand for the common case — group by these (possibly empty) + /// keys, kept rather than excluded. + pub fn by(keys: Vec) -> Self { + Self::Reduce(GroupKeys::by(keys)) + } + + /// The grouping keys, if this is a genuine reduction — `None` for + /// `PerEntity`, which has no grouping-keys concept to report. + pub fn group_keys(&self) -> Option<&GroupKeys> { + match self { + Self::Reduce(by) => Some(by), + Self::PerEntity => None, + } + } + + /// The grouping keys, panicking if this is `PerEntity` — for call sites + /// (tests, mostly) that already know, from the shape they built or are + /// asserting on, that this must be a genuine reduction. Prefer + /// [`group_keys`](Self::group_keys) wherever the caller can't assume that. + pub fn expect_reduce(&self) -> &GroupKeys { + match self { + Self::Reduce(by) => by, + Self::PerEntity => panic!("expected Reduction::Reduce, got PerEntity"), + } + } +} + +/// A caller-proven compound unique key for a [`crate::ir::NonASAPOp::Concat`] (issue +/// #228) — built only via [`ConcatDiscriminatorKey::new`], never by naming `discriminator` directly +/// in a struct literal (both fields are private): from *other Rust code*, +/// the only way to end up with one of these is to hand over a specific +/// column as the discriminator, by name, at the call site. +/// +/// Caveat: this is a Rust-API-level guarantee, not a data-level one. The +/// derived `Deserialize` impl below builds a `ConcatDiscriminatorKey` +/// directly from field values, bypassing `new()`. Deserialization is therefore +/// equivalent to a caller supplying the assertion directly; it does not prove +/// either fact below. An external boundary accepting IR data must +/// reject this field or validate both obligations before treating it as +/// uniqueness evidence. +/// +/// # Soundness +/// +/// `Concat`'s default (see its own doc) is to drop `unique_keys` +/// unconditionally, because a key unique **within** one branch is not unique +/// **across** the concatenation unless the branches' value sets for that key +/// are provably disjoint — nothing about matching schemas or matching +/// per-branch keys establishes that on its own. Two different branches can +/// trivially emit the same `inner_key` value (e.g. two PromQL +/// `histogram_quantiles` branches keyed on `(host, le)` can both produce a +/// `(host, le)` pair for different φ). +/// +/// Prepending `discriminator` restores a compound key only when two facts +/// hold: `inner_key` uniquely identifies rows **within every branch**, and +/// `discriminator`'s value is **guaranteed to differ between branches** — a +/// literal the producer just tagged the branch with (PromQL φ riding along via +/// [`crate::ir::NonASAPOp::PromqlRelabel`], a Postgres-style synthetic `GROUPING()` id +/// for `ROLLUP`/`CUBE`, …), never something inferred structurally from the +/// branches' own data — then `discriminator` alone partitions rows into +/// disjoint sets independent of what the branches actually contain, so +/// `(discriminator, inner_key)` is sound even when otherwise-identical +/// `inner_key` values occur in different branches. Neither fact is verified +/// here; both are part of the caller-proven claim. +/// +/// This is a **caller-proven claim, not something `Concat` can verify**: +/// nothing stops a caller from asserting a discriminator that in fact +/// repeats across branches, in which case the resulting `unique_keys` claim +/// is simply wrong — `output_schema` trusts it without checking. The +/// obligation is on the constructor call site, exactly as it is on +/// [`crate::ir::NonASAPOp::Dedup`]'s `cols` or any other unverified `unique_keys` +/// producer in this module. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +#[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] +pub struct ConcatDiscriminatorKey { + discriminator: C, + inner_key: Vec, +} + +impl ConcatDiscriminatorKey { + /// The only constructor — `discriminator` must be named explicitly by + /// the caller. See the type's doc for the soundness obligation this + /// puts on that caller. + pub fn new(discriminator: C, inner_key: Vec) -> Self { + Self { + discriminator, + inner_key, + } + } + + pub fn discriminator(&self) -> &C { + &self.discriminator + } + + pub fn inner_key(&self) -> &[C] { + &self.inner_key + } +} diff --git a/crates/types/src/ir/query.rs b/crates/types/src/ir/query.rs new file mode 100644 index 000000000..1540d6105 --- /dev/null +++ b/crates/types/src/ir/query.rs @@ -0,0 +1,36 @@ +//! Query results are either an operator result or a standalone scalar expression. +//! The root discriminator is not an operator and never creates a graph node. +use super::{OperatorNode, ScalarExpr}; +use serde::{Deserialize, Serialize}; +use std::rc::Rc; +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum QueryRoot { + Operator(Rc), + Scalar(ScalarExpr), +} +impl From> for QueryRoot { + fn from(node: Rc) -> Self { + Self::Operator(node) + } +} +impl QueryRoot { + pub fn validate_structure(&self) -> Result<(), crate::pre_asap::SchemaDerivationError> { + match self { + Self::Operator(node) => node.validate_structure(), + Self::Scalar(expr) => { + expr.scalar_type(&crate::pre_asap::Schema::default())?; + for node in expr.operator_refs() { + node.validate_structure()?; + } + Ok(()) + } + } + } + + pub fn as_operator(&self) -> Option<&Rc> { + match self { + Self::Operator(node) => Some(node), + Self::Scalar(_) => None, + } + } +} diff --git a/crates/types/src/ir/scalar.rs b/crates/types/src/ir/scalar.rs new file mode 100644 index 000000000..998f82bf9 --- /dev/null +++ b/crates/types/src/ir/scalar.rs @@ -0,0 +1,990 @@ +//! Scalar expressions: value computation evaluated within the schema chosen by +//! the operator that owns them. +//! +//! A [`ScalarExpr`] never produces a table. It is owned by value by an operator +//! field (`Filter.pred`, `ProjectItem.expr`, `SortKey.expr`, `HAVING`, window +//! arguments, relabel values) or by a [`super::QueryRoot::Scalar`] query root. +//! The only operator references inside a scalar tree are the explicit +//! plan-reading variants (`PromqlScalarFromVector`, `ScalarSubquery`, `Exists`, +//! `InSubquery`); every traversal of the operator DAG follows them. + +use std::rc::Rc; + +use serde::{Deserialize, Serialize}; + +use super::node::OperatorNode; +use crate::ir::SchemaDerivationError; +use crate::pre_asap::expr_ir::{ArithmeticOpKind, CompareOpKind, ScalarValue}; +use crate::pre_asap::scalar_signature::MapScalarFunction; +use crate::pre_asap::schema::{ColumnId, DataType, Schema}; + +/// Which language's numeric and comparison rules an expression follows. +/// Both languages use `Float64`, so a result type alone does not preserve +/// NaN, ordering or error rules. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ExprSemantics { + Sql, + Promql, +} + +/// A scalar expression over the owning operator's input schema. Column +/// references are positional [`ColumnId`]s. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum ScalarExpr { + Column(ColumnId), + Literal(ScalarValue), + /// Unary minus. + Negative { + expr: Box, + semantics: ExprSemantics, + }, + Compare { + left: Box, + op: CompareOpKind, + right: Box, + semantics: ExprSemantics, + }, + /// Flat conjunction (logical AND). An empty list is vacuously true. + BoolAnd(Vec), + /// Flat disjunction (logical OR). An empty list is vacuously false. + BoolOr(Vec), + Not(Box), + IsNull(Box), + IsNotNull(Box), + /// `CAST(expr AS to)`; `try_cast` for SQL `TRY_CAST` (NULL on failure). + Cast { + expr: Box, + to: DataType, + try_cast: bool, + }, + /// `expr [NOT] IN (v1, v2, …)`. + InList { + expr: Box, + list: Vec, + negated: bool, + }, + /// Scalar function call, e.g. `LOWER(col)`, `ABS(x)`. + FunctionCall { + name: String, + args: Vec, + }, + Arithmetic { + op: ArithmeticOpKind, + left: Box, + right: Box, + semantics: ExprSemantics, + }, + /// SQL `CASE` (both searched and simple forms). `operand` present for the + /// simple form (`CASE expr WHEN …`), absent for searched. + Case { + operand: Option>, + branches: Vec<(ScalarExpr, ScalarExpr)>, + else_expr: Option>, + }, + /// SQL `NOW()` / `CURRENT_TIMESTAMP`: the statement evaluation time. + CurrentTimestamp, + /// PromQL `time()`: the evaluation instant as Unix seconds (`Float64`). + EvalTimestamp, + /// PromQL `scalar(v)`: the single sample of an instant vector, NaN + /// otherwise. The referenced operator is a real plan dependency. + PromqlScalarFromVector(Rc), + /// An uncorrelated SQL scalar subquery: one column; zero rows is NULL, + /// more than one row is an error. + ScalarSubquery(Rc), + /// SQL `[NOT] EXISTS (subquery)`. + Exists { + subquery: Rc, + negated: bool, + }, + /// SQL `expr [NOT] IN (subquery)` over a one-column relation. + InSubquery { + expr: Box, + subquery: Rc, + negated: bool, + }, +} + +/// A row-level filter predicate (WHERE clause / PromQL label matcher). +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct Predicate(pub ScalarExpr); + +/// One item in a SELECT projection list. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct ProjectItem { + pub alias: Option, + pub expr: ScalarExpr, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct SortKey { + pub expr: ScalarExpr, + pub ascending: bool, + pub nulls_first: bool, +} + +impl ScalarExpr { + pub fn literal_f64(v: f64) -> Self { + ScalarExpr::Literal(ScalarValue::Float64(v)) + } + + pub fn column(id: ColumnId) -> Self { + ScalarExpr::Column(id) + } + + /// If this expression is a `BoolAnd`, its elements; otherwise `self` alone. + pub fn conjuncts(&self) -> &[ScalarExpr] { + match self { + ScalarExpr::BoolAnd(v) => v.as_slice(), + _ => std::slice::from_ref(self), + } + } + + /// If this expression is a `BoolOr`, its elements; otherwise `self` alone. + pub fn disjuncts(&self) -> &[ScalarExpr] { + match self { + ScalarExpr::BoolOr(v) => v.as_slice(), + _ => std::slice::from_ref(self), + } + } + + /// The direct scalar sub-expressions. + pub fn children(&self) -> Vec<&ScalarExpr> { + match self { + ScalarExpr::Column(_) + | ScalarExpr::Literal(_) + | ScalarExpr::CurrentTimestamp + | ScalarExpr::EvalTimestamp + | ScalarExpr::PromqlScalarFromVector(_) + | ScalarExpr::ScalarSubquery(_) + | ScalarExpr::Exists { .. } => vec![], + ScalarExpr::Negative { expr, .. } + | ScalarExpr::Not(expr) + | ScalarExpr::IsNull(expr) + | ScalarExpr::IsNotNull(expr) + | ScalarExpr::Cast { expr, .. } + | ScalarExpr::InSubquery { expr, .. } => vec![expr], + ScalarExpr::Compare { left, right, .. } + | ScalarExpr::Arithmetic { left, right, .. } => { + vec![left, right] + } + ScalarExpr::BoolAnd(parts) | ScalarExpr::BoolOr(parts) => parts.iter().collect(), + ScalarExpr::InList { expr, list, .. } => { + let mut v = vec![expr.as_ref()]; + v.extend(list.iter()); + v + } + ScalarExpr::FunctionCall { args, .. } => args.iter().collect(), + ScalarExpr::Case { + operand, + branches, + else_expr, + } => { + let mut v = Vec::new(); + if let Some(op) = operand { + v.push(op.as_ref()); + } + for (when, then) in branches { + v.push(when); + v.push(then); + } + if let Some(e) = else_expr { + v.push(e.as_ref()); + } + v + } + } + } + + /// The operator nodes this expression (transitively) reads: the explicit + /// plan-reading variants. Every DAG traversal must follow these. + pub fn operator_refs(&self) -> Vec<&Rc> { + let mut out = Vec::new(); + self.collect_operator_refs(&mut out); + out + } + + fn collect_operator_refs<'a>(&'a self, out: &mut Vec<&'a Rc>) { + match self { + ScalarExpr::PromqlScalarFromVector(node) | ScalarExpr::ScalarSubquery(node) => { + out.push(node) + } + ScalarExpr::Exists { subquery, .. } => out.push(subquery), + ScalarExpr::InSubquery { subquery, .. } => out.push(subquery), + _ => {} + } + for child in self.children() { + child.collect_operator_refs(out); + } + } + + /// Rebuild this expression with `f` applied to every operator node it + /// reads (recursively through scalar children). + pub fn map_operator_refs( + &self, + f: &mut impl FnMut(&Rc) -> Rc, + ) -> ScalarExpr { + fn map_box) -> Rc>( + e: &ScalarExpr, + f: &mut F, + ) -> Box { + Box::new(e.map_operator_refs(f)) + } + match self { + ScalarExpr::Column(_) + | ScalarExpr::Literal(_) + | ScalarExpr::CurrentTimestamp + | ScalarExpr::EvalTimestamp => self.clone(), + ScalarExpr::PromqlScalarFromVector(node) => ScalarExpr::PromqlScalarFromVector(f(node)), + ScalarExpr::ScalarSubquery(node) => ScalarExpr::ScalarSubquery(f(node)), + ScalarExpr::Exists { subquery, negated } => ScalarExpr::Exists { + subquery: f(subquery), + negated: *negated, + }, + ScalarExpr::InSubquery { + expr, + subquery, + negated, + } => ScalarExpr::InSubquery { + expr: map_box(expr, f), + subquery: f(subquery), + negated: *negated, + }, + ScalarExpr::Negative { expr, semantics } => ScalarExpr::Negative { + expr: map_box(expr, f), + semantics: *semantics, + }, + ScalarExpr::Compare { + left, + op, + right, + semantics, + } => ScalarExpr::Compare { + left: map_box(left, f), + op: op.clone(), + right: map_box(right, f), + semantics: *semantics, + }, + ScalarExpr::BoolAnd(parts) => { + ScalarExpr::BoolAnd(parts.iter().map(|p| p.map_operator_refs(f)).collect()) + } + ScalarExpr::BoolOr(parts) => { + ScalarExpr::BoolOr(parts.iter().map(|p| p.map_operator_refs(f)).collect()) + } + ScalarExpr::Not(e) => ScalarExpr::Not(map_box(e, f)), + ScalarExpr::IsNull(e) => ScalarExpr::IsNull(map_box(e, f)), + ScalarExpr::IsNotNull(e) => ScalarExpr::IsNotNull(map_box(e, f)), + ScalarExpr::Cast { expr, to, try_cast } => ScalarExpr::Cast { + expr: map_box(expr, f), + to: to.clone(), + try_cast: *try_cast, + }, + ScalarExpr::InList { + expr, + list, + negated, + } => ScalarExpr::InList { + expr: map_box(expr, f), + list: list.iter().map(|p| p.map_operator_refs(f)).collect(), + negated: *negated, + }, + ScalarExpr::FunctionCall { name, args } => ScalarExpr::FunctionCall { + name: name.clone(), + args: args.iter().map(|p| p.map_operator_refs(f)).collect(), + }, + ScalarExpr::Arithmetic { + op, + left, + right, + semantics, + } => ScalarExpr::Arithmetic { + op: op.clone(), + left: map_box(left, f), + right: map_box(right, f), + semantics: *semantics, + }, + ScalarExpr::Case { + operand, + branches, + else_expr, + } => ScalarExpr::Case { + operand: operand.as_ref().map(|e| map_box(e, f)), + branches: branches + .iter() + .map(|(w, t)| (w.map_operator_refs(f), t.map_operator_refs(f))) + .collect(), + else_expr: else_expr.as_ref().map(|e| map_box(e, f)), + }, + } + } + + /// Every column referenced anywhere in this expression (not inside + /// referenced operator subgraphs, which have their own scope). + pub fn columns_referenced(&self) -> Vec { + let mut out = Vec::new(); + self.collect_columns(&mut out); + out + } + + fn collect_columns(&self, out: &mut Vec) { + if let ScalarExpr::Column(id) = self { + out.push(*id); + } + for child in self.children() { + child.collect_columns(out); + } + } + + /// Infer the `(DataType, nullable)` this expression produces against the + /// input schema its owner evaluates it in. Unregistered functions are + /// rejected. A reference to a field carrying summary state is an error: + /// state must be read out before an expression can use it. + pub fn scalar_type(&self, schema: &Schema) -> Result<(DataType, bool), SchemaDerivationError> { + Ok(match self { + ScalarExpr::CurrentTimestamp => (DataType::Timestamp, false), + ScalarExpr::EvalTimestamp => (DataType::Float64, false), + ScalarExpr::Column(id) => match schema.fields.get(*id) { + Some(c) => match c.plain_dtype() { + Some(dtype) => (dtype.clone(), c.nullable), + None => { + return Err(SchemaDerivationError::InvalidScalarSignature(format!( + "column `{}` carries summary state and cannot be read as a value", + c.name + ))) + } + }, + None => return Err(signature("column outside scalar input scope")), + }, + ScalarExpr::Literal(s) => match s { + ScalarValue::Int64(_) => (DataType::Int64, false), + ScalarValue::Float64(_) => (DataType::Float64, false), + ScalarValue::Utf8(_) => (DataType::Utf8, false), + ScalarValue::Boolean(_) => (DataType::Bool, false), + ScalarValue::Null => (DataType::Null, true), + ScalarValue::Interval { .. } => (DataType::Interval, false), + }, + ScalarExpr::Compare { + left, right, op, .. + } => { + let (lt, ln) = left.scalar_type(schema)?; + let (rt, rn) = right.scalar_type(schema)?; + common_scalar_type(<, &rt)?; + if matches!( + op, + CompareOpKind::Regex + | CompareOpKind::NotRegex + | CompareOpKind::Like + | CompareOpKind::NotLike + | CompareOpKind::ILike + | CompareOpKind::NotILike + ) && (lt != DataType::Utf8 || rt != DataType::Utf8) + { + return Err(signature("pattern comparison requires strings")); + } + (DataType::Bool, ln || rn) + } + ScalarExpr::BoolAnd(parts) | ScalarExpr::BoolOr(parts) => { + let mut nullable = false; + for part in parts { + let (ty, n) = part.scalar_type(schema)?; + require_bool(&ty)?; + nullable |= n; + } + (DataType::Bool, nullable) + } + ScalarExpr::Not(expr) => { + let (ty, n) = expr.scalar_type(schema)?; + require_bool(&ty)?; + (DataType::Bool, n) + } + ScalarExpr::IsNull(expr) | ScalarExpr::IsNotNull(expr) => { + expr.scalar_type(schema)?; + (DataType::Bool, false) + } + ScalarExpr::InList { expr, list, .. } => { + let (ty, mut nullable) = expr.scalar_type(schema)?; + for item in list { + let (other, n) = item.scalar_type(schema)?; + common_scalar_type(&ty, &other)?; + nullable |= n; + } + (DataType::Bool, nullable) + } + ScalarExpr::Exists { subquery, .. } => { + relation(subquery)?; + (DataType::Bool, false) + } + ScalarExpr::InSubquery { expr, subquery, .. } => { + let (ty, _) = expr.scalar_type(schema)?; + let field = scalar_subquery_field(subquery)?; + common_scalar_type( + &ty, + field + .plain_dtype() + .ok_or_else(|| signature("subquery returns state"))?, + )?; + (DataType::Bool, true) + } + ScalarExpr::Negative { expr, .. } => { + let (dtype, nullable) = expr.scalar_type(schema)?; + if !numeric(&dtype) && dtype != DataType::Interval { + return Err(signature("negation requires a number or interval")); + } + (dtype, nullable) + } + ScalarExpr::Arithmetic { + op, left, right, .. + } => { + let (lt, ln) = left.scalar_type(schema)?; + let (rt, rn) = right.scalar_type(schema)?; + // Temporal subtraction yields a fixed duration with a unit, not a + // calendar interval or a floating-point number. Until the IR can + // preserve that unit, fail instead of publishing a numeric schema. + if matches!(op, ArithmeticOpKind::Sub) + && matches!(lt, DataType::Date | DataType::Timestamp) + && matches!(rt, DataType::Date | DataType::Timestamp) + { + return Err(SchemaDerivationError::InvalidScalarSignature( + "temporal subtraction produces an unsupported duration type".into(), + )); + } + let dtype = match (<, &rt) { + (DataType::Int64, DataType::Interval) + | (DataType::Interval, DataType::Int64) + if matches!(op, ArithmeticOpKind::Mul) => + { + DataType::Interval + } + (DataType::Timestamp, DataType::Interval) + if matches!(op, ArithmeticOpKind::Add | ArithmeticOpKind::Sub) => + { + DataType::Timestamp + } + (DataType::Interval, DataType::Timestamp) + if matches!(op, ArithmeticOpKind::Add) => + { + DataType::Timestamp + } + (DataType::Date, DataType::Interval) + if matches!(op, ArithmeticOpKind::Add | ArithmeticOpKind::Sub) => + { + DataType::Date + } + (DataType::Interval, DataType::Date) if matches!(op, ArithmeticOpKind::Add) => { + DataType::Date + } + (DataType::Interval, DataType::Interval) + if matches!(op, ArithmeticOpKind::Add | ArithmeticOpKind::Sub) => + { + DataType::Interval + } + (DataType::Int64, DataType::Int64) => DataType::Int64, + _ if numeric(<) && numeric(&rt) => common_scalar_type(<, &rt)?, + _ => return Err(signature("invalid arithmetic operand types")), + }; + (dtype, ln || rn) + } + ScalarExpr::Cast { to, try_cast, expr } => { + let (_, nullable) = expr.scalar_type(schema)?; + (to.clone(), *try_cast || nullable) + } + ScalarExpr::FunctionCall { name, args } => { + if let Some(arity) = crate::pre_asap::scalar_signature::promql_function_arity(name) + { + if args.len() != arity + || args + .iter() + .map(|a| a.scalar_type(schema)) + .collect::, _>>()? + .iter() + .any(|t| *t != (DataType::Float64, false)) + { + return Err(signature("PromQL function requires non-null float arguments of the declared arity")); + } + (DataType::Float64, false) + } else if name == "promql_drop_metric_name" { + if args.len() != 1 || args[0].scalar_type(schema)? != (DataType::Utf8, false) { + return Err(SchemaDerivationError::InvalidScalarSignature( + "metric-name removal requires one non-null series identity".into(), + )); + } + (DataType::Utf8, false) + } else if name == "asap_element_access" { + element_access_type(args, schema) + .map_err(SchemaDerivationError::InvalidScalarSignature)? + } else if name == "asap_struct_field" { + struct_field_type(args, schema) + .map_err(SchemaDerivationError::InvalidScalarSignature)? + } else if let Some(function) = MapScalarFunction::from_name(name) { + let arguments = args + .iter() + .map(|arg| arg.scalar_type(schema)) + .collect::, _>>()?; + function + .output_type(&arguments) + .map_err(SchemaDerivationError::InvalidScalarSignature)? + } else { + sql_function_type(name, args, schema)? + } + } + ScalarExpr::Case { + operand, + branches, + else_expr, + } => { + let operand = operand + .as_ref() + .map(|e| e.scalar_type(schema)) + .transpose()?; + let mut dtype = DataType::Null; + let mut nullable = else_expr.is_none(); + for (condition, value) in branches { + let (condition, _) = condition.scalar_type(schema)?; + if let Some((ty, _)) = &operand { + common_scalar_type(ty, &condition)?; + } else { + require_bool(&condition)?; + } + let (ty, n) = value.scalar_type(schema)?; + dtype = common_scalar_type(&dtype, &ty)?; + nullable |= n; + } + if let Some(value) = else_expr { + let (ty, n) = value.scalar_type(schema)?; + dtype = common_scalar_type(&dtype, &ty)?; + nullable |= n; + } + (dtype, nullable) + } + // `scalar(v)` is one float sample (NaN when the vector is not + // exactly one series); a scalar subquery is its single column. + ScalarExpr::PromqlScalarFromVector(node) => { + if node.result_kind != super::OperatorResultKind::InstantVector { + return Err(signature("scalar() requires an instant vector")); + } + (DataType::Float64, false) + } + ScalarExpr::ScalarSubquery(node) => { + let field = scalar_subquery_field(node)?; + ( + field + .plain_dtype() + .ok_or_else(|| signature("scalar subquery returns state"))? + .clone(), + true, + ) + } + }) + } +} + +fn signature(message: &str) -> SchemaDerivationError { + SchemaDerivationError::InvalidScalarSignature(message.into()) +} +fn numeric(ty: &DataType) -> bool { + matches!(ty, DataType::Int64 | DataType::Float64 | DataType::Null) +} +fn require_bool(ty: &DataType) -> Result<(), SchemaDerivationError> { + if matches!(ty, DataType::Bool | DataType::Null) { + Ok(()) + } else { + Err(signature("boolean expression required")) + } +} +fn common_scalar_type(a: &DataType, b: &DataType) -> Result { + if a == b || *b == DataType::Null { + Ok(a.clone()) + } else if *a == DataType::Null { + Ok(b.clone()) + } else if numeric(a) && numeric(b) { + Ok(DataType::Float64) + } else { + Err(signature("incompatible scalar types")) + } +} +fn relation(node: &OperatorNode) -> Result<(), SchemaDerivationError> { + if node.result_kind == super::OperatorResultKind::Relation { + Ok(()) + } else { + Err(signature("SQL subquery requires a relation")) + } +} +fn scalar_subquery_field( + node: &OperatorNode, +) -> Result<&crate::pre_asap::Field, SchemaDerivationError> { + relation(node)?; + match node.schema.fields.as_slice() { + [field] => Ok(field), + _ => Err(signature("scalar subquery requires exactly one column")), + } +} +fn sql_function_type( + name: &str, + args: &[ScalarExpr], + schema: &Schema, +) -> Result<(DataType, bool), SchemaDerivationError> { + let types = args + .iter() + .map(|a| a.scalar_type(schema)) + .collect::, _>>()?; + let nullable = types.iter().any(|(_, n)| *n); + let name = name.to_ascii_lowercase(); + match (name.as_str(), types.as_slice()) { + ("abs" | "ceil" | "floor" | "round", [(ty, _)]) if numeric(ty) => { + Ok((ty.clone(), nullable)) + } + ("sqrt" | "exp" | "ln" | "log2" | "log10" | "sin" | "cos" | "tan", [(ty, _)]) + if numeric(ty) => + { + Ok((DataType::Float64, nullable)) + } + ("lower" | "upper" | "trim" | "btrim" | "ltrim" | "rtrim", [(DataType::Utf8, _)]) => { + Ok((DataType::Utf8, nullable)) + } + ("length" | "char_length" | "character_length" | "octet_length", [(DataType::Utf8, _)]) => { + Ok((DataType::Int64, nullable)) + } + ("label_replace", [(DataType::Utf8, _), (DataType::Utf8, _), (DataType::Utf8, _)]) => { + Ok((DataType::Utf8, false)) + } + ("label_join" | "concat", [_, ..]) if types.iter().all(|(t, _)| *t == DataType::Utf8) => { + Ok((DataType::Utf8, nullable)) + } + ("date_trunc", [(DataType::Utf8, _), (DataType::Timestamp, _)]) => { + Ok((DataType::Timestamp, nullable)) + } + ("regexp_like", [(DataType::Utf8, _), (DataType::Utf8, _)]) => { + Ok((DataType::Bool, nullable)) + } + ("nullif", [(a, _), (b, _)]) => { + common_scalar_type(a, b)?; + Ok((a.clone(), true)) + } + ("coalesce", [_, ..]) => { + let mut ty = DataType::Null; + for (arg, _) in &types { + ty = common_scalar_type(&ty, arg)?; + } + Ok((ty, types.iter().all(|(_, n)| *n))) + } + _ => Err(signature(&format!( + "unregistered scalar function or invalid signature: {name}" + ))), + } +} + +/// Resolve the bounded canonical `asap_struct_field(struct, selector)` operation. +/// Selectors are positive 1-based literal ordinals or exact literal field names. +pub fn struct_field_type(args: &[ScalarExpr], schema: &Schema) -> Result<(DataType, bool), String> { + let [input, selector] = args else { + return Err("struct field access requires a struct and constant selector".into()); + }; + let (dtype, nullable) = input + .scalar_type(schema) + .map_err(|error| error.to_string())?; + if nullable { + return Err("nullable struct container access is unsupported".into()); + } + let DataType::Struct { fields } = dtype else { + return Err("struct field access requires a Struct input".into()); + }; + let field = match selector { + ScalarExpr::Literal(ScalarValue::Int64(index)) if *index > 0 => usize::try_from(*index - 1) + .ok() + .and_then(|index| fields.get(index)) + .ok_or("struct field ordinal is out of bounds")?, + ScalarExpr::Literal(ScalarValue::Utf8(name)) => { + let mut matches = fields.iter().filter(|field| field.name == *name); + let field = matches.next().ok_or("struct field name does not exist")?; + if matches.next().is_some() { + return Err("struct field name is ambiguous".into()); + } + field + } + _ => { + return Err( + "struct field selector must be a positive ordinal or field-name literal".into(), + ) + } + }; + Ok((field.dtype.clone(), field.nullable)) +} + +/// Resolve `asap_element_access(collection, index)`: Map access through the +/// map function contract, List access with integer indices. +pub fn element_access_type( + args: &[ScalarExpr], + schema: &Schema, +) -> Result<(DataType, bool), String> { + let [input, index] = args else { + return Err("element access requires a collection and index".into()); + }; + let source = input.scalar_type(schema).map_err(|e| e.to_string())?; + let key = index.scalar_type(schema).map_err(|e| e.to_string())?; + match &source.0 { + DataType::Map { .. } => MapScalarFunction::Access.output_type(&[source, key]), + DataType::List { element } => { + if source.1 { + return Err("nullable List container access is unsupported".into()); + } + if !matches!(key.0, DataType::Int64 | DataType::Null) { + return Err("List index must have integer type".into()); + } + if matches!(index, ScalarExpr::Literal(ScalarValue::Int64(0))) { + return Err( + "literal zero List index is unsupported without constant-array proof".into(), + ); + } + Ok(( + element.dtype.clone(), + element.nullable || key.1 || key.0 == DataType::Null, + )) + } + _ => Err("element access requires a Map or List".into()), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ir::operator_properties::Source; + use crate::ir::{NonASAPOp, OperatorNode}; + use crate::pre_asap::schema::{Field, FieldDataType}; + + fn call(name: &str, args: Vec) -> ScalarExpr { + ScalarExpr::FunctionCall { + name: name.into(), + args, + } + } + + fn int(v: i64) -> ScalarExpr { + ScalarExpr::Literal(ScalarValue::Int64(v)) + } + + fn utf8(v: &str) -> ScalarExpr { + ScalarExpr::Literal(ScalarValue::Utf8(v.into())) + } + + /// Shifting an instant by a duration stays an instant and shifting a date + /// stays a date (`l_shipdate + INTERVAL '30' DAY`), never numeric. + #[test] + fn interval_arithmetic_keeps_the_temporal_type() { + let schema = Schema::new(vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("d", DataType::Date, false), + ]); + let thirty_days = || { + Box::new(ScalarExpr::Literal(ScalarValue::Interval { + months: 0, + days: 30, + nanos: 0, + })) + }; + let shift = |left: Box, op| ScalarExpr::Arithmetic { + op, + left, + right: thirty_days(), + semantics: ExprSemantics::Sql, + }; + let ty = |e: ScalarExpr| e.scalar_type(&schema).unwrap().0; + assert_eq!( + ty(shift( + Box::new(ScalarExpr::Column(0)), + ArithmeticOpKind::Add + )), + DataType::Timestamp + ); + assert_eq!( + ty(shift( + Box::new(ScalarExpr::Column(1)), + ArithmeticOpKind::Sub + )), + DataType::Date + ); + assert_eq!( + ty(shift(thirty_days(), ArithmeticOpKind::Add)), + DataType::Interval + ); + } + + #[test] + fn canonical_projection_uses_map_signature_and_rejects_invalid_arity() { + let project = |expr: ScalarExpr| NonASAPOp::Project { + cols: vec![ProjectItem { + alias: Some("result".into()), + expr, + }], + qualifier: None, + child: OperatorNode::non_asap_node(NonASAPOp::Scan { + source: Source::Table { + table_ref: "t".into(), + }, + predicates: vec![], + schema: Schema::new(vec![ + Field::plain("k", DataType::Utf8, false), + Field::plain("v", DataType::Int64, true), + ]), + }) + .unwrap(), + }; + let map = call("map", vec![ScalarExpr::Column(0), ScalarExpr::Column(1)]); + let schema = project(map.clone()).output_schema().unwrap(); + assert_eq!( + schema.fields[0].dtype, + DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Int64), + value_nullable: true + } + ); + assert!(!schema.fields[0].nullable); + let lookup = call("asap_map_access", vec![map, utf8("missing")]); + assert_eq!( + project(lookup).output_schema().unwrap().fields[0], + Field::plain("result", DataType::Int64, true) + ); + assert!(project(call("map", vec![ScalarExpr::Column(0)])) + .output_schema() + .is_err()); + } + + fn record_schema() -> Schema { + Schema::new(vec![Field::plain( + "record", + DataType::Struct { + fields: vec![ + Field::new("ts", DataType::Int64, false), + Field::new( + "values", + DataType::List { + element: Box::new(Field::new("item", DataType::Float64, true)), + }, + true, + ), + ], + }, + false, + )]) + } + + fn field_access(selector: ScalarExpr) -> ScalarExpr { + call("asap_struct_field", vec![ScalarExpr::Column(0), selector]) + } + + #[test] + fn field_access_reuses_nested_field_type_and_nullability() { + let schema = record_schema(); + assert_eq!( + field_access(int(1)).scalar_type(&schema).unwrap(), + (DataType::Int64, false) + ); + let named = field_access(utf8("values")); + let ordinal = field_access(int(2)); + assert_eq!( + named.scalar_type(&schema).unwrap(), + ordinal.scalar_type(&schema).unwrap() + ); + assert_eq!( + named.scalar_type(&schema).unwrap(), + ( + DataType::List { + element: Box::new(Field::new("item", DataType::Float64, true)) + }, + true + ) + ); + let roundtrip: ScalarExpr = + serde_json::from_str(&serde_json::to_string(&named).unwrap()).unwrap(); + assert_eq!(roundtrip, named); + } + + #[test] + fn unsupported_field_access_is_an_error_not_placeholder_typing() { + for selector in [ + ScalarExpr::Column(0), + int(0), + int(-1), + int(3), + utf8("missing"), + ] { + assert!(field_access(selector) + .scalar_type(&record_schema()) + .is_err()); + } + let mut ambiguous = record_schema(); + if let FieldDataType::Plain(DataType::Struct { fields }) = &mut ambiguous.fields[0].dtype { + fields.push(Field::new("ts", DataType::Utf8, false)); + } + assert!(field_access(utf8("ts")).scalar_type(&ambiguous).is_err()); + let mut nullable = record_schema(); + nullable.fields[0].nullable = true; + assert!(field_access(int(1)).scalar_type(&nullable).is_err()); + } + + fn element_access(index: ScalarExpr) -> ScalarExpr { + call("asap_element_access", vec![ScalarExpr::Column(0), index]) + } + + #[test] + fn list_index_preserves_nested_element_metadata() { + let element = DataType::Struct { + fields: vec![ + Field::new("ts", DataType::Int64, false), + Field::new("value", DataType::Float64, true), + ], + }; + let schema = Schema::new(vec![ + Field::plain( + "samples", + DataType::List { + element: Box::new(Field::new("item", element.clone(), false)), + }, + false, + ), + Field::plain("i", DataType::Int64, true), + ]); + for index in [1, -1, 100] { + assert_eq!( + element_access(int(index)).scalar_type(&schema).unwrap(), + (element.clone(), false) + ); + } + assert_eq!( + element_access(ScalarExpr::Column(1)) + .scalar_type(&schema) + .unwrap(), + (element.clone(), true) + ); + assert!(element_access(int(0)).scalar_type(&schema).is_err()); + assert!(element_access(ScalarExpr::literal_f64(1.0)) + .scalar_type(&schema) + .is_err()); + let nested = call("asap_struct_field", vec![element_access(int(1)), int(2)]); + assert_eq!( + nested.scalar_type(&schema).unwrap(), + (DataType::Float64, true) + ); + let roundtrip: ScalarExpr = + serde_json::from_value(serde_json::to_value(&nested).unwrap()).unwrap(); + assert_eq!(roundtrip, nested); + } + + #[test] + fn generic_map_lookup_reuses_legacy_signature() { + let schema = Schema::new(vec![Field::plain( + "m", + DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Int64), + value_nullable: false, + }, + false, + )]); + let legacy = call("asap_map_access", vec![ScalarExpr::Column(0), utf8("k")]); + assert_eq!( + element_access(utf8("k")).scalar_type(&schema).unwrap(), + legacy.scalar_type(&schema).unwrap() + ); + } +} diff --git a/crates/types/src/ir/timing.rs b/crates/types/src/ir/timing.rs new file mode 100644 index 000000000..ff36fbfa4 --- /dev/null +++ b/crates/types/src/ir/timing.rs @@ -0,0 +1,898 @@ +//! Execution timing: written into every node from a lifecycle assignment, +//! then validated against each operator's kind and its consuming edges. +//! +//! The logical DAG carries no timing. Summary materialization chooses a +//! lifecycle per summary state; [`LifecycleAssignment`] records that choice +//! (ingestion-time maintenance or query-time recomputation per `SummaryAgg`) +//! and [`apply_lifecycle_timings`] expands it into a timing on every node: +//! +//! - a node of fixed kind takes its kind's timing (`SummaryEstimate` and +//! `EvaluatePopulation` run at query time, `MaintainPopulation` at ingestion +//! time); +//! - a `SummaryAgg` takes the assignment's timing (default: ingestion time), +//! unless something below it can only exist at query time; +//! - every other node runs when its consumer runs: everything that feeds a +//! maintained state runs at ingestion time, everything above a evaluation at +//! query time. +//! +//! A node reached from two consumers that need different timings cannot be +//! executed once for both; [`split_shared_by_phase`] copies such a sub-DAG +//! for one side before the assignment is applied, and the pass itself +//! rejects a conflict it still finds. +//! +//! ## Edge rules (checked after the write) +//! +//! | Consumer | Accepts from an input | +//! |---|---| +//! | `SummaryAgg.child` | Rows, or exact-accumulator state, never a query-time value when the state is maintained | +//! | `SummaryEstimate.summary_input` | Summary state at either phase | +//! | `FinalizeExactAccumulator.child` | Exact-accumulator state | +//! | `EvaluatePopulation.child` | A `MaintainPopulation` at ingestion time | +//! | `MaintainPopulation.child` | Ingestion-time rows matching the population's input | +//! | any `NonASAP` consumer | Rows (or exact-accumulator state for a projection-like operator) at the consumer's own timing; ingestion work never reads a query-time value | + +use std::collections::HashMap; +use std::rc::Rc; + +use super::asap::ASAPOp; +use super::node::{Operator, OperatorNode}; +use super::non_asap::NonASAPOp; +use crate::ir::operator_properties::BinaryOpKind; +use crate::post_asap::execution_data_state::{ + DataPrimitive, ExecutionDataState, ExecutionDataStateError, ExecutionTiming, +}; +use crate::pre_asap::schema::{DataType, FieldDataType, Schema}; + +/// The per-state lifecycle choice summary materialization made: for each +/// `SummaryAgg` node (by identity), whether its state is maintained at +/// ingestion time or recomputed at query time. A state absent from the map +/// takes the default, ingestion-time maintenance. +#[derive(Debug, Clone, Default)] +pub struct LifecycleAssignment { + summary_timings: HashMap<*const OperatorNode, ExecutionTiming>, +} + +impl LifecycleAssignment { + /// The assignment under which every summary state is maintained at + /// ingestion time — the timings every plan carried before lifecycles + /// became a planning choice. + pub fn default_maintained() -> Self { + Self::default() + } + + pub fn set(&mut self, summary: &Rc, timing: ExecutionTiming) { + self.summary_timings.insert(Rc::as_ptr(summary), timing); + } + + pub fn summary_timing(&self, summary: &Rc) -> ExecutionTiming { + self.summary_timings + .get(&Rc::as_ptr(summary)) + .copied() + .unwrap_or(ExecutionTiming::IngestionTime) + } +} + +/// Memo of one [`apply_lifecycle_timings`] pass: `input node → timed node`, +/// shared by every root of a workload so a node shared by two roots stays +/// one `Rc`. Re-reaching a node with a different timing is a conflict. +#[derive(Default)] +pub struct TimingMemo { + done: HashMap<*const OperatorNode, Rc>, +} + +impl TimingMemo { + pub fn new() -> Self { + Self::default() + } + + /// The timed node produced for `input`, if the pass has reached it. + pub fn timed(&self, input: &Rc) -> Option<&Rc> { + self.done.get(&Rc::as_ptr(input)) + } +} + +/// The data state a timed node's output carries. +pub fn data_state(node: &OperatorNode) -> Option { + Some(ExecutionDataState { + timing: node.timing?, + primitive: match &node.operator { + Operator::ASAP(op) if op.produced_state().is_some() => DataPrimitive::SummaryState, + Operator::ASAP(ASAPOp::MaintainPopulation { .. }) => DataPrimitive::SummaryState, + _ => DataPrimitive::Raw, + }, + }) +} + +/// Whether the sub-DAG below `node` contains a node that can only run at +/// query time (a evaluation), which forces every consumer above it to query +/// time as well. +fn forces_query_time(node: &OperatorNode, seen: &mut HashMap<*const OperatorNode, bool>) -> bool { + let key = node as *const OperatorNode; + if let Some(&cached) = seen.get(&key) { + return cached; + } + let forced = match &node.operator { + _ if node.timing == Some(ExecutionTiming::QueryTime) => true, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + | Operator::ASAP(ASAPOp::EvaluatePopulation { .. }) => true, + Operator::NonASAP(NonASAPOp::BinaryOp { lhs, rhs, .. }) + if node + .schema + .fields + .iter() + .any(|f| f.name == crate::pre_asap::schema::PROMQL_SERIES_IDENTITY) + && per_series_rows(lhs).is_none_or(|rows| per_series_rows(rhs) != Some(rows)) => + { + true + } + _ => node + .children() + .iter() + .any(|child| forces_query_time(child, seen)), + }; + seen.insert(key, forced); + forced +} + +/// Write the timings of `assignment` into every node reachable from `root`, +/// top-down, then validate every edge. Returns the timed copy of `root`; +/// `memo` carries the sharing across the roots of one workload. +pub fn apply_lifecycle_timings( + root: &Rc, + assignment: &LifecycleAssignment, + memo: &mut TimingMemo, +) -> Result, ExecutionDataStateError> { + let mut forced = HashMap::new(); + let timed = write( + root, + ExecutionTiming::QueryTime, + assignment, + memo, + &mut forced, + )?; + if timed.timing == Some(ExecutionTiming::IngestionTime) + && data_state(&timed).map(|s| s.primitive) == Some(DataPrimitive::Raw) + { + return Err(ExecutionDataStateError::MaintenanceRowsAtRoot); + } + validate(&timed, &mut HashMap::new())?; + Ok(timed) +} + +/// Validate the sub-DAG below `root` under the default (every summary +/// maintained) assignment, with `root` consumed at `root_timing`. For +/// planning-time legality checks of a candidate before it is assembled into +/// a workload DAG; nothing is kept. +pub fn validate_default( + root: &Rc, + root_timing: ExecutionTiming, +) -> Result<(), ExecutionDataStateError> { + let assignment = LifecycleAssignment::default_maintained(); + let mut memo = TimingMemo::new(); + let mut forced = HashMap::new(); + let timed = write(root, root_timing, &assignment, &mut memo, &mut forced)?; + validate(&timed, &mut HashMap::new()) +} + +/// The data state `node` produces under the default assignment when its +/// consumer runs at `consumer` — the planning-time answer to "what does this +/// candidate's output look like" before any assignment is applied. +pub fn planned_data_state( + node: &Rc, + consumer: ExecutionTiming, +) -> ExecutionDataState { + let mut forced = HashMap::new(); + let timing = own_timing( + node, + consumer, + &LifecycleAssignment::default_maintained(), + &mut forced, + ); + ExecutionDataState { + timing, + primitive: match &node.operator { + Operator::ASAP(op) if op.produced_state().is_some() => DataPrimitive::SummaryState, + Operator::ASAP(ASAPOp::MaintainPopulation { .. }) => DataPrimitive::SummaryState, + _ => DataPrimitive::Raw, + }, + } +} + +/// The timing `node` takes when its consumer runs at `consumer`. +fn own_timing( + node: &Rc, + consumer: ExecutionTiming, + assignment: &LifecycleAssignment, + forced: &mut HashMap<*const OperatorNode, bool>, +) -> ExecutionTiming { + // A placement fixed when the candidate was built (an exact-state read + // boundary that must run at query time, or one that feeds maintenance) + // is honored; a conflicting consumer is rejected by validation. + if let Some(placed) = node.timing { + return placed; + } + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + | Operator::ASAP(ASAPOp::EvaluatePopulation { .. }) => ExecutionTiming::QueryTime, + Operator::ASAP(ASAPOp::MaintainPopulation { .. }) => ExecutionTiming::IngestionTime, + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) => { + if forces_query_time(child, forced) { + ExecutionTiming::QueryTime + } else { + assignment.summary_timing(node) + } + } + _ => consumer, + } +} + +fn write( + node: &Rc, + consumer: ExecutionTiming, + assignment: &LifecycleAssignment, + memo: &mut TimingMemo, + forced: &mut HashMap<*const OperatorNode, bool>, +) -> Result, ExecutionDataStateError> { + let timing = own_timing(node, consumer, assignment, forced); + if let Some(done) = memo.done.get(&Rc::as_ptr(node)) { + let previous = done.timing.expect("memoized node is timed"); + if previous != timing { + return Err(ExecutionDataStateError::ConflictingTiming { + first: ExecutionDataState { + timing: previous, + primitive: data_state(done).map_or(DataPrimitive::Raw, |s| s.primitive), + }, + second: ExecutionDataState { + timing, + primitive: data_state(done).map_or(DataPrimitive::Raw, |s| s.primitive), + }, + }); + } + return Ok(Rc::clone(done)); + } + let mut error = None; + let operator = + node.operator.map_children( + |child| match write(child, timing, assignment, memo, forced) { + Ok(timed) => timed, + Err(e) => { + error.get_or_insert(e); + Rc::clone(child) + } + }, + ); + if let Some(e) = error { + return Err(e); + } + let timed = Rc::new(OperatorNode { + operator, + result_kind: node.result_kind, + schema: node.schema.clone(), + guarantee: node.guarantee.clone(), + timing: Some(timing), + }); + memo.done.insert(Rc::as_ptr(node), Rc::clone(&timed)); + Ok(timed) +} + +fn state_of(node: &OperatorNode) -> ExecutionDataState { + data_state(node).expect("timed node") +} + +/// Check every edge below `node` against the module-level rules. +fn validate( + node: &Rc, + seen: &mut HashMap<*const OperatorNode, ()>, +) -> Result<(), ExecutionDataStateError> { + if seen.insert(Rc::as_ptr(node), ()).is_some() { + return Ok(()); + } + let timing = node.timing.expect("timed node"); + match &node.operator { + Operator::ASAP(op) => validate_asap(node, op, timing)?, + Operator::NonASAP(op) => validate_non_asap(node, op, timing)?, + } + for child in node.children() { + validate(child, seen)?; + } + Ok(()) +} + +fn is_exact_accumulator_state(schema: &Schema) -> Result<(), ExecutionDataStateError> { + for field in &schema.fields { + match &field.dtype { + FieldDataType::Plain(_) | FieldDataType::ExactAggregate(..) => {} + other => { + return Err(ExecutionDataStateError::UnsupportedStateComposition { + family: format!("{other:?}"), + }) + } + } + } + Ok(()) +} + +fn validate_asap( + node: &OperatorNode, + op: &ASAPOp, + timing: ExecutionTiming, +) -> Result<(), ExecutionDataStateError> { + match op { + ASAPOp::SummaryAgg { child, .. } => { + let avail = state_of(child); + match avail { + ExecutionDataState::INGESTION_ROWS | ExecutionDataState::QUERY_ROWS => {} + s if s.primitive == DataPrimitive::SummaryState => { + is_exact_accumulator_state(&child.schema)? + } + other => { + return Err(ExecutionDataStateError::EvaluationUnderMaintenance { + edge: "SummaryAgg.child", + child: other, + }) + } + } + if timing == ExecutionTiming::IngestionTime + && avail.timing == ExecutionTiming::QueryTime + { + return Err(ExecutionDataStateError::EvaluationUnderMaintenance { + edge: "SummaryAgg.child", + child: avail, + }); + } + Ok(()) + } + ASAPOp::SummaryEstimate { summary_input, .. } => { + let s = state_of(summary_input); + if s.primitive != DataPrimitive::SummaryState { + return Err(ExecutionDataStateError::IllegalChildDataState { + edge: "SummaryEstimate.summary_input", + child: s, + }); + } + if timing != ExecutionTiming::QueryTime { + return Err(ExecutionDataStateError::IllegalChildDataState { + edge: "SummaryEstimate", + child: state_of(node), + }); + } + Ok(()) + } + ASAPOp::FinalizeExactAccumulator { child } => { + let s = state_of(child); + if s.primitive != DataPrimitive::SummaryState + || is_exact_accumulator_state(&child.schema).is_err() + || (timing == ExecutionTiming::IngestionTime && s.timing != timing) + { + return Err(ExecutionDataStateError::IllegalChildDataState { + edge: "FinalizeExactAccumulator.child", + child: s, + }); + } + Ok(()) + } + ASAPOp::MaintainPopulation { child, population } => { + let valid = population.matches_node(child) + && state_of(child) + == ExecutionDataState { + timing, + primitive: DataPrimitive::Raw, + }; + if !valid { + return Err(ExecutionDataStateError::InvalidMaintainedPopulation); + } + Ok(()) + } + ASAPOp::EvaluatePopulation { child, evaluation } => { + let valid = timing == ExecutionTiming::QueryTime + && matches!( + &child.operator, + Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) + if population.supports(evaluation) + && child.timing.is_some() + ); + if !valid { + return Err(ExecutionDataStateError::InvalidMaintainedPopulation); + } + Ok(()) + } + ASAPOp::SummaryMerge { .. } + | ASAPOp::SummarySubtract { .. } + | ASAPOp::SummaryDelete { .. } + | ASAPOp::SummaryJoin { .. } + | ASAPOp::Extension { .. } => Err(ExecutionDataStateError::UnimplementedOperator { + operator: op.kind_name(), + }), + } +} + +fn check_plain_or_exact_values(input: &Schema) -> Result<(), ExecutionDataStateError> { + for field in &input.fields { + if !matches!( + field.dtype, + FieldDataType::Plain(_) | FieldDataType::ExactAggregate(..) + ) { + return Err(ExecutionDataStateError::NonPlainOperand { + column: field.name.clone(), + dtype: format!("{:?}", field.dtype), + }); + } + } + Ok(()) +} + +fn check_all_plain(input: &Schema) -> Result<(), ExecutionDataStateError> { + for field in &input.fields { + if !field.is_plain() { + return Err(ExecutionDataStateError::NonPlainOperand { + column: field.name.clone(), + dtype: format!("{:?}", field.dtype), + }); + } + } + Ok(()) +} + +fn validate_non_asap( + node: &OperatorNode, + op: &NonASAPOp, + timing: ExecutionTiming, +) -> Result<(), ExecutionDataStateError> { + // Every input is rows at this node's own timing. Ingestion work never + // reads a query-time value; exact-accumulator state may pass through + // the projection-like operators unchanged. + for child in op.children() { + let s = state_of(child); + let passes_state = matches!( + op, + NonASAPOp::Project { .. } + | NonASAPOp::Filter { .. } + | NonASAPOp::Sort { .. } + | NonASAPOp::Limit { .. } + ) && s.primitive == DataPrimitive::SummaryState + && is_exact_accumulator_state(&child.schema).is_ok(); + if s.timing != timing || (s.primitive != DataPrimitive::Raw && !passes_state) { + return Err(ExecutionDataStateError::IllegalChildDataState { + edge: op.kind_name(), + child: s, + }); + } + } + match op { + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => check_plain_or_exact_values(&child.schema)?, + NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } => { + let mut referenced: Vec = reduction + .group_keys() + .map(|keys| keys.keys().to_vec()) + .unwrap_or_default(); + for m in measures { + referenced.extend(m.input_cols()); + } + let implicit = measures.iter().any(|m| m.input_cols().is_empty()); + for (i, field) in child.schema.fields.iter().enumerate() { + if (implicit || referenced.contains(&i)) && !field.is_plain() { + return Err(ExecutionDataStateError::NonPlainOperand { + column: field.name.clone(), + dtype: format!("{:?}", field.dtype), + }); + } + } + } + NonASAPOp::BinaryOp { + operator, lhs, rhs, .. + } => { + let is_div = matches!( + operator.kind, + BinaryOpKind::Arithmetic(crate::pre_asap::ArithmeticOpKind::Div) + ); + if (operator.checked_relative_division && operator.checked_finite_division) + || ((operator.checked_relative_division || operator.checked_finite_division) + && (timing != ExecutionTiming::QueryTime || !is_div)) + { + return Err(ExecutionDataStateError::InvalidCheckedDivision); + } + if timing == ExecutionTiming::IngestionTime { + let plain_float_or_ts = |schema: &Schema| { + schema.fields.iter().all(|field| { + !field.nullable + && (matches!( + field.dtype, + FieldDataType::Plain(DataType::Float64 | DataType::Timestamp) + ) || (field.name + == crate::pre_asap::schema::PROMQL_SERIES_IDENTITY + && field.dtype == FieldDataType::Plain(DataType::Utf8))) + }) + }; + let float_count = node + .schema + .fields + .iter() + .filter(|f| matches!(f.dtype, FieldDataType::Plain(DataType::Float64))) + .count(); + if operator.vector_match.is_some() + || !matches!(operator.kind, BinaryOpKind::Arithmetic(_)) + || lhs.schema != rhs.schema + || lhs.schema != node.schema + || node + .schema + .fields + .iter() + .filter(|f| f.name == crate::pre_asap::schema::PROMQL_SERIES_IDENTITY) + .count() + > 1 + || (node + .schema + .fields + .iter() + .any(|f| f.name == crate::pre_asap::schema::PROMQL_SERIES_IDENTITY) + && per_series_rows(lhs) + .is_none_or(|rows| per_series_rows(rhs) != Some(rows))) + || !plain_float_or_ts(&node.schema) + || float_count != 1 + { + return Err(ExecutionDataStateError::InvalidMaintenanceBinary); + } + } + } + _ => { + for child in op.children() { + check_all_plain(&child.schema)?; + } + } + } + Ok(()) +} + +/// Copy, for one consumer, every sub-DAG that `assignment` would reach with +/// two different timings, so that a workload whose CSE shared a `Scan` +/// between an ingestion-time summary and a query-time computation can still +/// be timed. Only the conflicting sub-DAGs are copied; a sub-DAG reached with +/// one timing stays one `Rc`. Returns the (possibly rewritten) root. +pub fn split_shared_by_phase( + root: &Rc, + assignment: &LifecycleAssignment, +) -> Rc { + // First pass: the set of timings each node is reached with. + let mut reached: HashMap<*const OperatorNode, Vec> = HashMap::new(); + let mut forced = HashMap::new(); + fn collect( + node: &Rc, + consumer: ExecutionTiming, + assignment: &LifecycleAssignment, + reached: &mut HashMap<*const OperatorNode, Vec>, + forced: &mut HashMap<*const OperatorNode, bool>, + ) { + let timing = own_timing(node, consumer, assignment, forced); + let entry = reached.entry(Rc::as_ptr(node)).or_default(); + if entry.contains(&timing) { + return; + } + entry.push(timing); + for child in node.children() { + collect(child, timing, assignment, reached, forced); + } + } + collect( + root, + ExecutionTiming::QueryTime, + assignment, + &mut reached, + &mut forced, + ); + if reached.values().all(|timings| timings.len() <= 1) { + return Rc::clone(root); + } + // Second pass: rebuild, giving each (node, timing) pair its own copy. + let mut copies: HashMap<(*const OperatorNode, ExecutionTiming), Rc> = + HashMap::new(); + fn rebuild( + node: &Rc, + consumer: ExecutionTiming, + assignment: &LifecycleAssignment, + reached: &HashMap<*const OperatorNode, Vec>, + copies: &mut HashMap<(*const OperatorNode, ExecutionTiming), Rc>, + forced: &mut HashMap<*const OperatorNode, bool>, + ) -> Rc { + let timing = own_timing(node, consumer, assignment, forced); + let key = (Rc::as_ptr(node), timing); + if let Some(done) = copies.get(&key) { + return Rc::clone(done); + } + let conflicted = reached + .get(&Rc::as_ptr(node)) + .is_some_and(|timings| timings.len() > 1); + let mut changed = conflicted; + let operator = node.operator.map_children(|child| { + let rebuilt = rebuild(child, timing, assignment, reached, copies, forced); + changed |= !Rc::ptr_eq(&rebuilt, child); + rebuilt + }); + let out = if changed { + Rc::new(OperatorNode { + operator, + result_kind: node.result_kind, + schema: node.schema.clone(), + guarantee: node.guarantee.clone(), + timing: node.timing, + }) + } else { + Rc::clone(node) + }; + copies.insert(key, Rc::clone(&out)); + out + } + rebuild( + root, + ExecutionTiming::QueryTime, + assignment, + &reached, + &mut copies, + &mut forced, + ) +} + +/// Maintenance arithmetic needs the same per-series population on both sides. +fn per_series_rows(node: &OperatorNode) -> Option<&OperatorNode> { + use crate::post_asap::ExactKind; + match &node.operator { + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) => match &child.operator { + Operator::ASAP(ASAPOp::SummaryAgg { + child, + family: FieldDataType::ExactAggregate(ExactKind::Sum | ExactKind::Count, _), + reduction: crate::pre_asap::Reduction::PerEntity, + filter: None, + .. + }) => Some(child), + _ => None, + }, + Operator::NonASAP(NonASAPOp::BinaryOp { lhs, rhs, .. }) => { + let rows = per_series_rows(lhs)?; + (per_series_rows(rhs) == Some(rows)).then_some(rows) + } + _ => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ir::non_asap::NonASAPOp; + use crate::ir::operator_properties::{Reduction, Source}; + use crate::post_asap::sketch::{ + ExactKind, ExactParams, GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, + SketchStatistic, SummaryUpdate, + }; + use crate::pre_asap::agg_intent::AggIntent; + use crate::pre_asap::expr_ir::ColumnRef; + use crate::pre_asap::schema::Field; + + fn scan_with(fields: Vec) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index(fields, 0, vec![]), + }) + .unwrap() + } + + fn scan() -> Rc { + scan_with(vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("zone", DataType::Utf8, true), + ]) + } + + fn kll() -> FieldDataType { + FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 200 }), + GroupingStrategy::default(), + ) + } + + fn exact_sum() -> FieldDataType { + FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + } + + fn agg(child: Rc, family: FieldDataType) -> Rc { + OperatorNode::asap_node( + ASAPOp::SummaryAgg { + child, + family: family.clone(), + input: SummaryUpdate::column(ColumnRef::SampleValue), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + }, + Schema::lifted(vec![Field::new("state", family, false)], None), + None, + ) + } + + fn estimate(child: Rc) -> Rc { + OperatorNode::asap_node( + ASAPOp::SummaryEstimate { + summary_input: child, + query: SketchStatistic::Quantile { q: 0.99 }, + }, + Schema::lifted( + vec![Field::plain("quantile_0_99", DataType::Float64, false)], + None, + ), + None, + ) + } + + fn aggregate(measure: AggIntent, child: Rc) -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![]), + measures: vec![measure], + output_names: vec![], + filters: vec![], + having: None, + child, + }) + .unwrap() + } + + fn max(child: Rc) -> Rc { + aggregate(AggIntent::Max { col: None }, child) + } + + /// `node` with its timing fixed in advance, as a candidate builder does + /// for an operator that must feed maintenance. + fn placed_at_ingestion(node: Rc) -> Rc { + Rc::new( + (*node) + .clone() + .with_timing(Some(ExecutionTiming::IngestionTime)), + ) + } + + fn apply(root: &Rc) -> Result, ExecutionDataStateError> { + apply_lifecycle_timings( + root, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + } + + fn child(node: &Rc) -> Rc { + Rc::clone(node.children()[0]) + } + + #[test] + fn summary_agg_input_runs_at_ingestion_time() { + let root = apply(&agg(scan(), kll())).unwrap(); + assert_eq!( + data_state(&root), + Some(ExecutionDataState::INGESTION_SUMMARY) + ); + assert_eq!( + data_state(&child(&root)), + Some(ExecutionDataState::INGESTION_ROWS) + ); + } + + #[test] + fn exact_accumulator_state_may_feed_another_summary_agg() { + let inner = agg(scan(), exact_sum()); + assert!(apply(&estimate(agg(inner, kll()))).is_ok()); + } + + #[test] + fn evaluation_can_feed_summary_construction_at_query_time() { + let inner = estimate(agg(scan(), kll())); + let root = apply(&estimate(agg(inner, kll()))).unwrap(); + assert_eq!(child(&root).timing, Some(ExecutionTiming::QueryTime)); + } + + /// Any non-ASAP operator over a evaluation runs at query time. + #[test] + fn query_time_operation_over_evaluation_is_legal_and_root_is_evaluation() { + let evaluation = || estimate(agg(scan(), kll())); + let sorted = OperatorNode::non_asap_node(NonASAPOp::Sort { + keys: vec![], + partition_by: Default::default(), + child: evaluation(), + }) + .unwrap(); + for root in [max(evaluation()), sorted] { + let root = apply(&root).unwrap(); + assert_eq!(data_state(&root), Some(ExecutionDataState::QUERY_ROWS)); + } + } + + #[test] + fn query_time_values_can_feed_query_time_summary_construction() { + let post = max(estimate(agg(scan(), kll()))); + let root = apply(&estimate(agg(post, kll()))).unwrap(); + assert_eq!(child(&root).timing, Some(ExecutionTiming::QueryTime)); + } + + #[test] + fn function_under_summary_agg_is_legal_but_not_at_root() { + let operation = placed_at_ingestion(max(scan())); + assert_eq!( + apply(&operation).err(), + Some(ExecutionDataStateError::MaintenanceRowsAtRoot) + ); + let root = apply(&estimate(agg(operation, kll()))).unwrap(); + let timed_operation = child(&child(&root)); + assert_eq!( + data_state(&timed_operation), + Some(ExecutionDataState::INGESTION_ROWS) + ); + } + + #[test] + fn function_over_evaluation_is_rejected() { + let operation = placed_at_ingestion(max(estimate(agg(scan(), kll())))); + assert!(matches!( + apply(&estimate(agg(operation, kll()))), + Err(ExecutionDataStateError::IllegalChildDataState { + child: ExecutionDataState::QUERY_ROWS, + .. + }) + )); + } + + /// One shared sub-DAG reached as maintenance input and as query-time + /// input cannot be executed once for both; splitting it by phase first + /// makes the plan timeable. + #[test] + fn a_shared_subtree_reached_at_two_timings_conflicts() { + let shared = scan(); + let root = OperatorNode::non_asap_node(NonASAPOp::Concat { + children: vec![ + max(estimate(agg(Rc::clone(&shared), kll()))), + max(Rc::clone(&shared)), + ], + discriminator_unique_key: None, + }) + .unwrap(); + assert_eq!( + apply(&root).err(), + Some(ExecutionDataStateError::ConflictingTiming { + first: ExecutionDataState::INGESTION_ROWS, + second: ExecutionDataState::QUERY_ROWS, + }) + ); + let split = split_shared_by_phase(&root, &LifecycleAssignment::default_maintained()); + assert!(apply(&split).is_ok()); + } + + /// Both paired operands must be plain; an unrelated state column is not + /// an input. + #[test] + fn pearson_corr_checks_both_operand_states() { + let corr_over = |state_column: usize| { + let mut fields = vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("x", DataType::Float64, false), + Field::plain("y", DataType::Float64, false), + Field::plain("unused", DataType::Float64, false), + ]; + fields[state_column].dtype = kll(); + aggregate( + AggIntent::PearsonCorr { left: 1, right: 2 }, + scan_with(fields), + ) + }; + for operand in [1, 2] { + assert!(matches!( + validate_default(&corr_over(operand), ExecutionTiming::QueryTime), + Err(ExecutionDataStateError::NonPlainOperand { .. }) + )); + } + validate_default(&corr_over(3), ExecutionTiming::QueryTime).unwrap(); + } +} diff --git a/crates/types/src/lib.rs b/crates/types/src/lib.rs index caacfd441..9252ac806 100644 --- a/crates/types/src/lib.rs +++ b/crates/types/src/lib.rs @@ -1,26 +1,22 @@ //! `asap-types` — shared vocabulary for the whole workspace. //! -//! Merges the former `asap-ir` crate (the pre-ASAP intent algebra, -//! workload/batch types, and DAG export) with the data-type-only modules of -//! the former `asap-sketch` crate (the post-ASAP sketch-bound IR types, -//! under [`post_asap`]). -//! -//! - [`pre_asap`] / [`types`] / [`workload`] / [`dag_export`] — the -//! pre-ASAP IR: language-agnostic query intent, independent of any -//! sketch decision. -//! - [`post_asap`] — the post-ASAP IR: sketch-bound types -//! ([`post_asap::sketch`], [`post_asap::expr`], [`post_asap::schema`]) -//! that commit to a concrete `SummaryKind`/`SummaryParams` realization. -//! No execution logic lives in this workspace (see issue #190) — a -//! downstream deployment crate is expected to supply that. -//! [`post_asap::query_time`] is the one exception, folder-separated from -//! the rest of `post_asap` on purpose: pure, sketch-object-agnostic -//! posterior error-bound math (issue #239) that a future real sketch -//! runtime's readout path can call directly — see that module's docs -//! for the planning-time/execution-time boundary and why it's unwired -//! today. +//! - [`ir`] — the unified operator IR: one operator language before and +//! after ASAP optimization ([`ir::OperatorNode`]), plus its passes +//! (canonicalize, CSE, timing) and the wire export ([`ir::export`]). +//! - [`pre_asap`] — the shared field vocabulary the IR's operators are +//! built from (grouping keys, reductions, sources, aggregation intents, +//! scalar literal / operator kinds, [`pre_asap::Schema`]). +//! - [`post_asap`] — summary-state types (families, kinds, parameters, +//! grouping strategy), accuracy guarantees, and the execution-timing +//! vocabulary. No execution logic lives in this workspace (issue #190). +//! [`post_asap::query_time`] holds pure posterior error-bound math +//! (issue #239) a future sketch runtime's evaluation path can call; see its +//! docs for why it is unwired today. +//! - [`types`] / [`workload`] / [`parsed_workload`] / [`dag_export`] / +//! [`cost`] / [`resources`] — workload, batch, export and cost types. pub mod cost; pub mod dag_export; +pub mod ir; pub mod parsed_workload; pub mod post_asap; pub mod pre_asap; diff --git a/crates/types/src/parsed_workload.rs b/crates/types/src/parsed_workload.rs index ff955e6a7..9dcf21b83 100644 --- a/crates/types/src/parsed_workload.rs +++ b/crates/types/src/parsed_workload.rs @@ -1,5 +1,5 @@ //! [`ParsedWorkload`] — a [`PlanningWorkload`] whose queries have been lowered -//! to pre-ASAP IR. +//! to the operator IR. //! //! This is the boundary between the frontend stage and the optimization stage //! (issues #429, #430). Everything downstream of lowering consumes this type @@ -10,7 +10,7 @@ use std::rc::Rc; -use crate::pre_asap::query_expr::QueryExpr; +use crate::ir::{OperatorNode, QueryRoot, ScalarExpr}; use crate::workload::{ DataWorkload, PlanningWorkload, QueryWorkload, QueryWorkloadEntry, WorkloadError, }; @@ -33,7 +33,9 @@ pub enum ParsedWorkloadError { #[derive(Debug, Clone)] pub struct ParsedWorkload { workload: PlanningWorkload, - exprs: Vec>, + exprs: Vec>, + operator_indices: Vec, + scalars: Vec<(usize, ScalarExpr)>, } impl ParsedWorkload { @@ -41,16 +43,43 @@ impl ParsedWorkload { /// `i`-th entry. pub fn new( workload: PlanningWorkload, - exprs: Vec>, + exprs: Vec>, + ) -> Result { + Self::from_roots( + workload, + exprs.into_iter().map(QueryRoot::Operator).collect(), + ) + } + + pub fn from_roots( + workload: PlanningWorkload, + roots: Vec, ) -> Result { let entries = workload.query_workload.entries().count(); - if entries != exprs.len() { + if entries != roots.len() { return Err(ParsedWorkloadError::LengthMismatch { entries, - lowered: exprs.len(), + lowered: roots.len(), }); } - Ok(Self { workload, exprs }) + let mut exprs = Vec::new(); + let mut operator_indices = Vec::new(); + let mut scalars = Vec::new(); + for (index, root) in roots.into_iter().enumerate() { + match root { + QueryRoot::Operator(node) => { + operator_indices.push(index); + exprs.push(node); + } + QueryRoot::Scalar(expr) => scalars.push((index, expr)), + } + } + Ok(Self { + workload, + exprs, + operator_indices, + scalars, + }) } pub fn planning_workload(&self) -> &PlanningWorkload { @@ -65,26 +94,37 @@ impl ParsedWorkload { self.workload.data_workload.as_ref() } - pub fn exprs(&self) -> &[Rc] { + pub fn exprs(&self) -> &[Rc] { &self.exprs } pub fn len(&self) -> usize { - self.exprs.len() + self.exprs.len() + self.scalars.len() } pub fn is_empty(&self) -> bool { - self.exprs.is_empty() + self.len() == 0 } /// Normalized entries paired with their lowered expression. - pub fn entries(&self) -> impl Iterator)> + '_ { + pub fn entries(&self) -> impl Iterator)> + '_ { self.workload .query_workload .entries() + .enumerate() + .filter(|(index, _)| self.operator_indices.binary_search(index).is_ok()) + .map(|(_, entry)| entry) .zip(self.exprs.iter()) } + pub fn operator_indices(&self) -> &[usize] { + &self.operator_indices + } + + pub fn scalar_roots(&self) -> &[(usize, ScalarExpr)] { + &self.scalars + } + /// The retained workload's own validation — entry legality and data-workload /// consistency. The PromQL-specific checks it also runs were already a /// precondition of the lowering that produced `self`. diff --git a/crates/types/src/post_asap/cse.rs b/crates/types/src/post_asap/cse.rs deleted file mode 100644 index 1872243f8..000000000 --- a/crates/types/src/post_asap/cse.rs +++ /dev/null @@ -1,457 +0,0 @@ -//! Structural sharing for a selected workload in one execution/data scope. -//! -//! This is not candidate selection or a cross-request cache. Callers opt into -//! common producer execution only after agreeing on lifecycle and data scope. -//! Typed equality includes schemas, guarantees and complete source expressions. - -use std::collections::HashMap; -use std::rc::Rc; - -use super::{SummaryExpr, SummaryNode}; - -/// Numeric PartialEq alone conflates signed zeros. The serialized check is -/// additional evidence, never a replacement for typed equality (JSON maps -/// nonfinite floats to null). Keep this rule local to structural sharing. -fn same_value(left: &T, right: &T) -> bool { - left == right - && match (serde_json::to_string(left), serde_json::to_string(right)) { - (Ok(left), Ok(right)) => left == right, - _ => false, - } -} - -/// Children have already been interned. Comparing their identities avoids -/// recursively expanding a shared DAG once for every path to each descendant. -fn same_node(left: &SummaryNode, right: &SummaryNode) -> bool { - use SummaryExpr::*; - let expression_equal = match (&left.expr, &right.expr) { - (KeepPreAsap(a), KeepPreAsap(b)) => Rc::ptr_eq(a, b) || same_value(a, b), - ( - BinaryOp { - lhs: al, - rhs: ar, - operator: ao, - timing: at, - }, - BinaryOp { - lhs: bl, - rhs: br, - operator: bo, - timing: bt, - }, - ) => Rc::ptr_eq(al, bl) && Rc::ptr_eq(ar, br) && ao == bo && at == bt, - ( - ValueOperation { - child: ac, - operation: ao, - timing: at, - }, - ValueOperation { - child: bc, - operation: bo, - timing: bt, - }, - ) => Rc::ptr_eq(ac, bc) && same_value(ao, bo) && at == bt, - ( - RelationalJoin { - left: al, - right: ar, - kind: ak, - pred: ap, - pruning: ax, - }, - RelationalJoin { - left: bl, - right: br, - kind: bk, - pred: bp, - pruning: bx, - }, - ) => { - Rc::ptr_eq(al, bl) - && Rc::ptr_eq(ar, br) - && ak == bk - && same_value(ap, bp) - && same_value(ax, bx) - } - ( - SummaryAgg { - child: ac, - family: af, - input: ai, - reduction: ar, - grouping: ag, - filter: afl, - }, - SummaryAgg { - child: bc, - family: bf, - input: bi, - reduction: br, - grouping: bg, - filter: bfl, - }, - ) => { - Rc::ptr_eq(ac, bc) - && af == bf - && same_value(ai, bi) - && ar == br - && ag == bg - && same_value(afl, bfl) - } - ( - SummaryJoin { - outer: ao, - inner: ai, - key: ak, - family: af, - }, - SummaryJoin { - outer: bo, - inner: bi, - key: bk, - family: bf, - }, - ) => Rc::ptr_eq(ao, bo) && Rc::ptr_eq(ai, bi) && ak == bk && af == bf, - ( - SummarySubtract { - left: al, - right: ar, - }, - SummarySubtract { - left: bl, - right: br, - }, - ) => Rc::ptr_eq(al, bl) && Rc::ptr_eq(ar, br), - ( - SummaryEstimate { - summary_input: ai, - query: aq, - }, - SummaryEstimate { - summary_input: bi, - query: bq, - }, - ) => Rc::ptr_eq(ai, bi) && same_value(aq, bq), - ( - SummaryDelete { - summary_input: ai, - key: ak, - }, - SummaryDelete { - summary_input: bi, - key: bk, - }, - ) => Rc::ptr_eq(ai, bi) && ak == bk, - ( - SummaryMerge { - children: a, - timing: at, - }, - SummaryMerge { - children: b, - timing: bt, - }, - ) => at == bt && a.len() == b.len() && a.iter().zip(b).all(|(a, b)| Rc::ptr_eq(a, b)), - // Keep this exhaustive on the left: new variants require a sharing rule. - ( - KeepPreAsap(_) - | BinaryOp { .. } - | ValueOperation { .. } - | RelationalJoin { .. } - | SummaryAgg { .. } - | SummaryJoin { .. } - | SummarySubtract { .. } - | SummaryEstimate { .. } - | SummaryDelete { .. } - | SummaryMerge { .. }, - _, - ) => false, - }; - expression_equal && left.schema == right.schema && same_value(&left.guarantee, &right.guarantee) -} - -/// Intern equal selected subtrees across roots while preserving every root ID. -/// -/// Only structural equality is used: no grouping, parameter, accuracy or source -/// coercions are performed. All roots must belong to the same data snapshot or -/// maintenance scope. Downstream realization must still check physical -/// implementation compatibility. Use separate calls for independent executions. -/// -/// When the selected states are identical, this is the planner's -/// summary-capability rule (#509 Pass 2): one summary build node feeds every -/// readout it supports, e.g. one KLL for p50 and p99, or one UnivMon for -/// distinct count, entropy and L2. Candidate generation sizes a variant for -/// the strictest sibling consumer so differing accuracy targets can reach -/// identical states here. -pub fn share_common_summary_subtrees( - roots: Vec<(Id, Rc)>, -) -> Vec<(Id, Rc)> { - fn visit( - node: &Rc, - seen: &mut HashMap>, - pool: &mut Vec>, - ) -> Rc { - let identity = Rc::as_ptr(node) as usize; - if let Some(node) = seen.get(&identity) { - return Rc::clone(node); - } - let mut result = node.as_ref().clone(); - match &mut result.expr { - SummaryExpr::KeepPreAsap(_) => {} - SummaryExpr::SummaryAgg { child, .. } => *child = visit(child, seen, pool), - SummaryExpr::BinaryOp { lhs, rhs, .. } => { - *lhs = visit(lhs, seen, pool); - *rhs = visit(rhs, seen, pool); - } - - SummaryExpr::ValueOperation { child, .. } => *child = visit(child, seen, pool), - SummaryExpr::RelationalJoin { left, right, .. } => { - *left = visit(left, seen, pool); - *right = visit(right, seen, pool); - } - SummaryExpr::SummaryJoin { outer, inner, .. } => { - *outer = visit(outer, seen, pool); - *inner = visit(inner, seen, pool); - } - SummaryExpr::SummarySubtract { left, right } => { - *left = visit(left, seen, pool); - *right = visit(right, seen, pool); - } - SummaryExpr::SummaryEstimate { summary_input, .. } - | SummaryExpr::SummaryDelete { summary_input, .. } => { - *summary_input = visit(summary_input, seen, pool); - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - *child = visit(child, seen, pool); - } - } - } - let result = match pool.iter().find(|existing| same_node(existing, &result)) { - Some(existing) => Rc::clone(existing), - None => { - let result = Rc::new(result); - pool.push(Rc::clone(&result)); - result - } - }; - seen.insert(identity, Rc::clone(&result)); - result - } - let mut seen = HashMap::new(); - let mut pool = Vec::new(); - roots - .into_iter() - .map(|(id, root)| (id, visit(&root, &mut seen, &mut pool))) - .collect() -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::post_asap::{ResultGuarantee, SummarySchema}; - use crate::pre_asap::{QueryExpr, ScalarValue}; - - fn leaf(value: f64) -> Rc { - Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(QueryExpr::Literal(ScalarValue::Float64( - value, - )))), - schema: SummarySchema { - fields: vec![], - time_index: None, - }, - guarantee: Some(ResultGuarantee::exact("fixture")), - }) - } - - // Equal separately constructed roots preserve both IDs but share identity. - #[test] - fn shares_equal_roots_and_preserves_ids() { - let roots = share_common_summary_subtrees(vec![("a", leaf(1.0)), ("b", leaf(1.0))]); - assert_eq!(roots[0].0, "a"); - assert_eq!(roots[1].0, "b"); - assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); - } - - // A diamond is retained across the returned roots, not copied per consumer. - #[test] - fn shares_children_across_distinct_roots() { - let merge = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: crate::post_asap::ExecutionTiming::IngestionTime, - children: vec![leaf(1.0), leaf(2.0)], - }, - schema: SummarySchema { - fields: vec![], - time_index: None, - }, - guarantee: None, - }); - let roots = share_common_summary_subtrees(vec![(0, leaf(1.0)), (1, merge)]); - let SummaryExpr::SummaryMerge { children, .. } = &roots[1].1.expr else { - panic!() - }; - assert!(Rc::ptr_eq(&roots[0].1, &children[0])); - assert!(!Rc::ptr_eq(&children[0], &children[1])); - } - - // Unknown guarantees must not be replaced by an equal expression's exact guarantee. - #[test] - fn distinct_guarantees_and_values_are_not_shared() { - let mut unknown = leaf(1.0).as_ref().clone(); - unknown.guarantee = None; - let roots = share_common_summary_subtrees(vec![ - (0, leaf(1.0)), - (1, Rc::new(unknown)), - (2, leaf(2.0)), - ]); - assert!(!Rc::ptr_eq(&roots[0].1, &roots[1].1)); - assert!(!Rc::ptr_eq(&roots[0].1, &roots[2].1)); - assert!(roots[1].1.guarantee.is_none()); - } - - // Sharing must preserve IEEE signed zero, including inside exact expressions. - #[test] - fn signed_zero_is_not_coalesced() { - for values in [[0.0, -0.0], [-0.0, 0.0]] { - let roots = - share_common_summary_subtrees(vec![(0, leaf(values[0])), (1, leaf(values[1]))]); - assert!(!Rc::ptr_eq(&roots[0].1, &roots[1].1)); - for ((_, root), expected) in roots.iter().zip(values) { - let SummaryExpr::KeepPreAsap(expr) = &root.expr else { - panic!() - }; - let QueryExpr::Literal(ScalarValue::Float64(actual)) = expr.as_ref() else { - panic!() - }; - assert_eq!(actual.to_bits(), expected.to_bits()); - assert_eq!(1.0 / actual, 1.0 / expected); - } - } - } - - // Exact expression wrappers must retain signed zero too; JSON's null - // encoding of nonfinite floats must never become the equality decision. - #[test] - fn nested_values_and_nonfinite_values_remain_distinct() { - let wrapped = |value| { - Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(QueryExpr::promql_scalar(value))), - ..leaf(1.0).as_ref().clone() - }) - }; - for (a, b) in [ - (0.0, -0.0), - (f64::INFINITY, f64::NEG_INFINITY), - (f64::NAN, f64::NAN), - ] { - let roots = share_common_summary_subtrees(vec![(0, wrapped(a)), (1, wrapped(b))]); - assert!(!Rc::ptr_eq(&roots[0].1, &roots[1].1)); - } - let roots = share_common_summary_subtrees(vec![ - (0, wrapped(f64::INFINITY)), - (1, wrapped(f64::INFINITY)), - ]); - assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); - } - - // Distinct quantile readouts share only a compatible typed sketch producer. - #[test] - fn quantile_roots_share_producer_but_not_readout_or_parameters() { - use crate::post_asap::{ - GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, - SummaryFamilyType, SummaryUpdate, - }; - use crate::pre_asap::{ColumnRef, Reduction}; - fn readout(q: f64, alpha: f64) -> Rc { - let producer = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: leaf(1.0), - family: SummaryFamilyType::Sketch( - SketchKind::new( - SketchAlgorithm::DDSketch, - SketchParams::DDSketch { alpha }, - ), - GroupingStrategy::default(), - ), - input: SummaryUpdate::column(ColumnRef::SampleValue), - reduction: Reduction::PerEntity, - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: SummarySchema { - fields: vec![], - time_index: None, - }, - guarantee: None, - }); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: producer, - query: SketchQuery::Quantile { q }, - }, - schema: SummarySchema { - fields: vec![], - time_index: None, - }, - guarantee: None, - }) - } - let roots = share_common_summary_subtrees(vec![ - ("p95", readout(0.95, 0.01)), - ("p99", readout(0.99, 0.01)), - ("strict", readout(0.95, 0.001)), - ]); - let producer = |root: &Rc| match &root.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => Rc::clone(summary_input), - _ => panic!(), - }; - assert!(!Rc::ptr_eq(&roots[0].1, &roots[1].1)); - assert!(Rc::ptr_eq(&producer(&roots[0].1), &producer(&roots[1].1))); - assert!(!Rc::ptr_eq(&producer(&roots[0].1), &producer(&roots[2].1))); - } - - // Fifty unique input nodes must not require walking an expanded 2^24 tree. - // The timeout is a coarse runaway guard, not a performance SLA. - #[test] - fn shared_diamond_does_not_expand_during_comparison() { - let (done, completion) = std::sync::mpsc::channel(); - let worker = std::thread::spawn(move || { - fn diamond() -> Rc { - let mut current = leaf(1.0); - for _ in 0..24 { - current = Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - timing: super::super::ExecutionTiming::QueryTime, - lhs: Rc::clone(¤t), - rhs: current, - operator: super::super::BinaryOperator { - checked_relative_division: false, - checked_finite_division: false, - kind: crate::pre_asap::BinaryOpKind::Arithmetic( - crate::pre_asap::ArithmeticOpKind::Add, - ), - vector_match: None, - }, - }, - schema: super::super::SummarySchema { - fields: vec![], - time_index: None, - }, - guarantee: None, - }); - } - current - } - let roots = share_common_summary_subtrees(vec![(0, diamond()), (1, diamond())]); - assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); - done.send(()).unwrap(); - }); - completion - .recv_timeout(std::time::Duration::from_secs(5)) - .expect("comparison expanded the shared DAG"); - worker.join().unwrap(); - } -} diff --git a/crates/types/src/post_asap/execution_data_state.rs b/crates/types/src/post_asap/execution_data_state.rs index 3c1c3e0c1..de764ea2c 100644 --- a/crates/types/src/post_asap/execution_data_state.rs +++ b/crates/types/src/post_asap/execution_data_state.rs @@ -1,56 +1,15 @@ -//! Execution-data-state contract for mixed exact/summary plans (issue #171). +//! Execution timing and data-state vocabulary of the operator IR. //! -//! A post-ASAP DAG mixes two very different moments of execution: the -//! **update/ingest path** (rows arrive, maintained summary state is updated) -//! and **query evaluation** (maintained state is read out and a final result -//! is produced). A plan that places a query-time residual *underneath* a -//! maintained summary is not merely expensive — it is unexecutable, because -//! the maintenance loop has no readout values to feed into that summary. -//! [`SummaryExpr::ValueOperation`] represents such work without inventing a -//! node per function or use case. Its [`ExecutionTiming`] makes placement -//! explicit and independent of the semantic [`ValueOperation`]. -//! -//! [`ExecutionDataState`] is what a node's output *is*, at which data_state; -//! [`validate_execution_data_states`] checks every edge of a DAG against the -//! rules below at plan construction, returning a typed [`ExecutionDataStateError`] rather -//! than deferring to a runtime failure. -//! -//! ## Edge rules -//! -//! | Parent | Accepts from `child` | -//! |---|---| -//! | `SummaryAgg.child` | Rows or exact accumulator state at either phase. The initial construction phase follows the input; deployment assigns final phases. | -//! | `SummaryEstimate.summary_input` | Summary state at either phase (any family). Initial readout produces `QUERY_ROWS`. | -//! | `SummaryJoin.outer/inner` | `INGESTION_ROWS` or `INGESTION_SUMMARY`; never a read-time data_state. | -//! | `SummarySubtract`/`SummaryDelete` | `INGESTION_SUMMARY`. | -//! | `SummaryMerge` | Summary state at its explicit ingestion or read timing. | -//! | `ValueOperation.child` with `IngestionTime` | `INGESTION_ROWS`; explicit `FinalizeExactAccumulator` also accepts exact accumulator state. Produces `INGESTION_ROWS`. | -//! | `ValueOperation.child` with `QueryTime` | `QUERY_ROWS`. Produces `QUERY_ROWS`. | -//! -//! ## `KeepPreAsap` declares its data_state through the derivation -//! -//! A [`SummaryExpr::KeepPreAsap`] leaf is a raw pre-ASAP computation that a -//! runtime can execute at either time: as maintenance input beneath a -//! `SummaryAgg`/maintenance-time `ValueOperation`, or as a query-time fallback -//! beneath a read-time `ValueOperation` (or at the root). It carries no timing -//! field of its own -//! — every existing consumer pattern-matches the one-field shape — so its -//! data_state is *assigned* by [`validate_execution_data_states`] from the edge that -//! reaches it and reported in the returned [`ExecutionDataStateAssignment`]. What it may -//! not do is stay ambiguous inside one mixed plan: the same `Rc` -//! reached once as update input and once as query-time fallback is -//! [`ExecutionDataStateError::AmbiguousKeepPreAsap`], because no single execution of that -//! subtree can serve both roles. - -use std::collections::HashMap; -use std::rc::Rc; +//! [`ExecutionTiming`] says when a node's value is produced (ingestion vs. +//! query time); [`ExecutionDataState`] pairs it with the [`DataPrimitive`] +//! the edge carries (raw values vs. summary state). The rules that assign +//! and check them over a DAG live in [`crate::ir::timing`], which reports +//! violations as [`ExecutionDataStateError`]. use thiserror::Error; -use super::expr::{ExactOperation, SummaryExpr, SummaryNode, ValueOperation}; -use super::schema::{SummaryFamilyType, SummaryField, SummarySchema}; -use crate::pre_asap::query_expr::{aggregate_output_schema, Predicate, QueryExprError}; -use crate::pre_asap::schema::{Column, Schema}; +use crate::ir::SchemaDerivationError; +use crate::pre_asap::schema::Schema; /// When a post-ASAP value is produced. #[derive( @@ -78,7 +37,7 @@ impl ExecutionTiming { /// The primitive representation carried by a post-ASAP edge. #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] pub enum DataPrimitive { - /// Directly usable values, including approximate summary readouts. + /// Directly usable values, including approximate summary evaluations. /// This does not imply original input data or an exact guarantee. Raw, SummaryState, @@ -122,55 +81,27 @@ impl std::fmt::Display for ExecutionDataState { } } -/// Which parent/edge a [`ExecutionDataStateError`] is about — the variant name of the -/// parent `SummaryExpr` plus its field, for a message a plan author can act -/// on. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum ExecutionDataStateEdge { - SummaryAggChild, - SummaryEstimateInput, - SummaryJoinInput, - SummarySubtractInput, - SummaryDeleteInput, - SummaryMergeInput, - ValueOperationChild, -} - -impl ExecutionDataStateEdge { - fn describe(self) -> &'static str { - match self { - Self::SummaryAggChild => "SummaryAgg.child", - Self::SummaryEstimateInput => "SummaryEstimate.summary_input", - Self::SummaryJoinInput => "SummaryJoin.{outer,inner}", - Self::SummarySubtractInput => "SummarySubtract.{left,right}", - Self::SummaryDeleteInput => "SummaryDelete.summary_input", - Self::SummaryMergeInput => "SummaryMerge.children[]", - Self::ValueOperationChild => "ValueOperation.child", - } - } -} - /// A plan-construction-time data_state violation. Typed (not a string) so a /// strategy can degrade to a conservative fallback on the specific variant /// it expects, and so tests can assert the *reason* a plan was rejected. #[derive(Debug, Clone, PartialEq, Eq, Error)] pub enum ExecutionDataStateError { - #[error("invalid maintained-population maintenance/readout contract")] + #[error("invalid maintained-population maintenance/evaluation contract")] InvalidMaintainedPopulation, - /// A query-time value (`SummaryEstimate` / read-time `ValueOperation` output) + /// A query-time value (a `SummaryEstimate` or query-time operator output) /// placed beneath a maintained summary — the one shape issue #171's /// data_state split exists to make unrepresentable. #[error( - "readout value under maintenance: {edge} received a {child} input, but a maintained \ + "evaluation value under maintenance: {edge} received a {child} input, but a maintained \ summary can only consume update-path values (or exact accumulator state)" )] - ReadoutUnderMaintenance { + EvaluationUnderMaintenance { edge: &'static str, child: ExecutionDataState, }, /// Any other edge whose child data_state the parent does not accept /// (e.g. plain update rows fed straight into a `SummaryEstimate`, or a - /// sketch's opaque state fed into a read-time `ValueOperation`). + /// sketch's opaque state fed into a query-time operator). #[error("{edge} does not accept a {child} input")] IllegalChildDataState { edge: &'static str, @@ -184,13 +115,10 @@ pub enum ExecutionDataStateError { composed into another maintained summary" )] UnsupportedStateComposition { family: String }, - /// One shared `KeepPreAsap` node reached both as update-path raw input - /// and as a query-time fallback — see the module docs. - #[error( - "KeepPreAsap subtree is data_state-ambiguous: reached as {first} and as {second} in the same \ - plan" - )] - AmbiguousKeepPreAsap { + /// One shared node assigned two different execution timings by its + /// consumers; no single execution of it can serve both. + #[error("shared node is assigned conflicting timings: {first} and {second}")] + ConflictingTiming { first: ExecutionDataState, second: ExecutionDataState, }, @@ -202,634 +130,38 @@ pub enum ExecutionDataStateError { InvalidMaintenanceBinary, #[error("checked division requires one valid guard on a read-time division operator")] InvalidCheckedDivision, - /// An `ExactOperation` whose input columns are not all `Plain` at its + /// An exact operator whose input columns are not all `Plain` at its /// declared data_state. #[error("exact operator consumes non-plain column {column:?} ({dtype})")] NonPlainOperand { column: String, dtype: String }, + /// A reserved ASAP operator (`SummaryMerge`, `SummarySubtract`, + /// `SummaryDelete`, `SummaryJoin`, `Extension`) in an executable plan. + #[error("{operator} is a reserved operator with no execution contract yet")] + UnimplementedOperator { operator: &'static str }, + /// A node reached by export without a timing: the lifecycle timing pass + /// was not applied to the DAG first. + #[error("{operator} node has no execution timing; apply lifecycle timings before export")] + UntimedNode { operator: &'static str }, } -/// The data_state assigned to every node of a validated plan, keyed by -/// `Rc` pointer identity — the explicit per-node "execution_data_state" a -/// runtime or a DAG export reads instead of re-deriving it. For every -/// non-`KeepPreAsap` node this equals [`produced_data_state`]; for a -/// `KeepPreAsap` leaf it is the data_state the reaching edge assigned. -#[derive(Debug, Clone, Default)] -pub struct ExecutionDataStateAssignment { - domains: HashMap<*const SummaryNode, ExecutionDataState>, -} - -impl ExecutionDataStateAssignment { - /// The data_state assigned to `node`, if it was part of the validated plan. - pub fn data_state_of(&self, node: &Rc) -> Option { - self.domains.get(&Rc::as_ptr(node)).copied() - } - - /// The data_state assigned to the node at `ptr` — for callers walking a plan - /// by reference rather than by `Rc`. - pub fn data_state_of_ptr(&self, ptr: *const SummaryNode) -> Option { - self.domains.get(&ptr).copied() - } -} - -/// Initial layout proposed by semantic realization, not a restriction on physical -/// operator placement. `PostAsapDag::with_execution_phases` assigns the final -/// phase independently of payload kind. Returns `None` for -/// [`SummaryExpr::KeepPreAsap`], whose data_state is assigned by the edge reaching -/// it (see the module docs). -pub fn produced_data_state(expr: &SummaryExpr) -> Option { - Some(match expr { - SummaryExpr::KeepPreAsap(_) => return None, - SummaryExpr::BinaryOp { timing, .. } => ExecutionDataState { - timing: *timing, - primitive: DataPrimitive::Raw, - }, - SummaryExpr::RelationalJoin { .. } => ExecutionDataState::QUERY_ROWS, - SummaryExpr::SummaryAgg { child, .. } => ExecutionDataState { - timing: produced_data_state(&child.expr) - .map_or(ExecutionTiming::IngestionTime, |state| state.timing), - primitive: DataPrimitive::SummaryState, - }, - SummaryExpr::SummaryJoin { .. } - | SummaryExpr::SummarySubtract { .. } - | SummaryExpr::SummaryDelete { .. } => ExecutionDataState::INGESTION_SUMMARY, - SummaryExpr::SummaryMerge { timing, .. } => ExecutionDataState { - timing: *timing, - primitive: DataPrimitive::SummaryState, - }, - SummaryExpr::SummaryEstimate { .. } => ExecutionDataState::QUERY_ROWS, - SummaryExpr::ValueOperation { timing, .. } => match timing { - ExecutionTiming::IngestionTime => ExecutionDataState::INGESTION_ROWS, - ExecutionTiming::QueryTime => ExecutionDataState::QUERY_ROWS, - }, - }) -} - -/// Is `family` the exact-accumulator family whose partial state *is* the -/// value — the one summary state a `SummaryAgg` may re-accumulate? -fn is_exact_accumulator_state(schema: &SummarySchema) -> Result<(), ExecutionDataStateError> { - for field in &schema.fields { - match &field.dtype { - SummaryFamilyType::Plain(_) | SummaryFamilyType::ExactAggregate(..) => {} - other => { - return Err(ExecutionDataStateError::UnsupportedStateComposition { - family: format!("{other:?}"), - }) - } - } - } - Ok(()) -} - -/// Validate every edge of the DAG rooted at `root` against the module-level -/// rules, returning each node's assigned data_state on success. Shared -/// `Rc`s are visited once per reaching edge (the assignment is -/// per node, so a conflict between two edges is what -/// [`ExecutionDataStateError::AmbiguousKeepPreAsap`] detects). -pub fn validate_execution_data_states( - root: &Rc, -) -> Result { - // The root may be a readable value or bare maintained state (a - // deployment may hand an `ExactAggregate` accumulator straight to a - // consumer) — only an update-path-only root is meaningless. - let root_domain = match produced_data_state(&root.expr) { - None => ExecutionDataState::QUERY_ROWS, - Some(ExecutionDataState::INGESTION_ROWS) => { - return Err(ExecutionDataStateError::MaintenanceRowsAtRoot) - } - Some(data_state) => data_state, - }; - validate_execution_data_states_at(root, root_domain) -} - -/// [`validate_execution_data_states`] for a *sub*-plan whose root is known to -/// sit at `data_state` — e.g. a maintenance-time `ValueOperation` about to be placed beneath a -/// `SummaryAgg`, which would be rejected as a whole-plan root but is a -/// legal update-path input. Validates every edge beneath `root` exactly -/// as the whole-plan entry point does. -pub fn validate_execution_data_states_at( - root: &Rc, - data_state: ExecutionDataState, -) -> Result { - let mut assignment = ExecutionDataStateAssignment::default(); - visit(root, data_state, &mut assignment)?; - Ok(assignment) -} - -/// The source rows whose series a maintenance operand has one row for: a -/// finalized per-series Sum or Count of those rows, or aligned arithmetic of -/// operands over the same rows. Each emits exactly the series with a sample. -fn per_series_rows(node: &SummaryNode) -> Option<&crate::pre_asap::QueryExpr> { - use crate::post_asap::ExactKind; - match &node.expr { - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - .. - } => match &child.expr { - SummaryExpr::SummaryAgg { - child, - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum | ExactKind::Count, _), - reduction: crate::pre_asap::query_expr::Reduction::PerEntity, - .. - } => match &child.expr { - SummaryExpr::KeepPreAsap(rows) => Some(rows.as_ref()), - _ => None, - }, - _ => None, - }, - SummaryExpr::BinaryOp { - lhs, - rhs, - timing: ExecutionTiming::IngestionTime, - .. - } => { - let rows = per_series_rows(lhs)?; - (per_series_rows(rhs) == Some(rows)).then_some(rows) - } - _ => None, - } -} - -/// Record `data_state` for `node` (detecting a conflicting earlier assignment -/// for a `KeepPreAsap`), then check and recurse into every child edge. -fn visit( - node: &Rc, - data_state: ExecutionDataState, - assignment: &mut ExecutionDataStateAssignment, -) -> Result<(), ExecutionDataStateError> { - let ptr = Rc::as_ptr(node); - if let Some(previous) = assignment.domains.get(&ptr) { - if *previous != data_state { - return Err(ExecutionDataStateError::AmbiguousKeepPreAsap { - first: *previous, - second: data_state, - }); - } - // Already validated through another edge with the same data_state. - return Ok(()); - } - assignment.domains.insert(ptr, data_state); - - match &node.expr { - SummaryExpr::KeepPreAsap(_) => Ok(()), - SummaryExpr::BinaryOp { - lhs, - rhs, - timing, - operator, - } => { - if (operator.checked_relative_division && operator.checked_finite_division) - || (operator.checked_relative_division || operator.checked_finite_division) - && (*timing != ExecutionTiming::QueryTime - || !matches!( - operator.kind, - crate::pre_asap::BinaryOpKind::Arithmetic( - crate::pre_asap::ArithmeticOpKind::Div - ) - )) - { - return Err(ExecutionDataStateError::InvalidCheckedDivision); - } - if *timing == ExecutionTiming::IngestionTime { - use crate::pre_asap::{BinaryOpKind, DataType}; - if operator.vector_match.is_some() - || !matches!(operator.kind, BinaryOpKind::Arithmetic(_)) - || lhs.schema != rhs.schema - || lhs.schema != node.schema - // The opaque identity is a key, not an extra maintenance value. - || node.schema.fields.iter().filter(|field| { - field.name == crate::pre_asap::schema::PROMQL_SERIES_IDENTITY - }).count() > 1 - // Maintenance arithmetic pairs every row by identity, while - // Prometheus drops unmatched series; it is exact only when - // both operands provably produce the same series. - || node.schema.fields.iter().any(|field| { - field.name == crate::pre_asap::schema::PROMQL_SERIES_IDENTITY - }) && per_series_rows(lhs).is_none_or(|rows| per_series_rows(rhs) != Some(rows)) - || !node.schema.fields.iter().all(|field| { - !field.nullable - && if field.name == crate::pre_asap::schema::PROMQL_SERIES_IDENTITY { - field.dtype == SummaryFamilyType::Plain(DataType::Utf8) - } else { - matches!( - field.dtype, - SummaryFamilyType::Plain(DataType::Float64 | DataType::Timestamp) - ) - } - }) - || node - .schema - .fields - .iter() - .filter(|field| { - matches!(field.dtype, SummaryFamilyType::Plain(DataType::Float64)) - }) - .count() - != 1 - { - return Err(ExecutionDataStateError::InvalidMaintenanceBinary); - } - } - let expected = ExecutionDataState { - timing: *timing, - primitive: DataPrimitive::Raw, - }; - for input in [lhs, rhs] { - let state = produced_data_state(&input.expr).unwrap_or(expected); - if state != expected { - return Err(ExecutionDataStateError::IllegalChildDataState { - edge: "BinaryOp operand", - child: state, - }); - } - visit(input, state, assignment)?; - } - Ok(()) - } - - SummaryExpr::RelationalJoin { left, right, .. } => { - for input in [left, right] { - let state = - produced_data_state(&input.expr).unwrap_or(ExecutionDataState::QUERY_ROWS); - if state != ExecutionDataState::QUERY_ROWS { - return Err(ExecutionDataStateError::IllegalChildDataState { - edge: "RelationalJoin input", - child: state, - }); - } - visit(input, state, assignment)?; - } - Ok(()) - } - SummaryExpr::SummaryAgg { child, .. } => { - let child_domain = child_domain( - child, - ExecutionDataStateEdge::SummaryAggChild, - |avail| match avail { - ExecutionDataState::INGESTION_ROWS | ExecutionDataState::QUERY_ROWS => Ok(()), - state if state.primitive == DataPrimitive::SummaryState => { - is_exact_accumulator_state(&child.schema) - } - other => Err(ExecutionDataStateError::ReadoutUnderMaintenance { - edge: ExecutionDataStateEdge::SummaryAggChild.describe(), - child: other, - }), - }, - )?; - visit(child, child_domain, assignment) - } - SummaryExpr::SummaryJoin { outer, inner, .. } => { - for input in [outer, inner] { - let s = child_domain(input, ExecutionDataStateEdge::SummaryJoinInput, |avail| { - match avail { - ExecutionDataState::INGESTION_ROWS - | ExecutionDataState::INGESTION_SUMMARY => Ok(()), - other => Err(ExecutionDataStateError::ReadoutUnderMaintenance { - edge: ExecutionDataStateEdge::SummaryJoinInput.describe(), - child: other, - }), - } - })?; - visit(input, s, assignment)?; - } - Ok(()) - } - SummaryExpr::SummarySubtract { left, right } => { - for input in [left, right] { - let s = state_only(input, ExecutionDataStateEdge::SummarySubtractInput)?; - visit(input, s, assignment)?; - } - Ok(()) - } - SummaryExpr::SummaryDelete { summary_input, .. } => { - let s = state_only(summary_input, ExecutionDataStateEdge::SummaryDeleteInput)?; - visit(summary_input, s, assignment) - } - SummaryExpr::SummaryMerge { children, timing } => { - for input in children { - let s = child_domain(input, ExecutionDataStateEdge::SummaryMergeInput, |state| { - if state.primitive == DataPrimitive::SummaryState - && (*timing == ExecutionTiming::QueryTime || state.timing == *timing) - { - Ok(()) - } else { - Err(ExecutionDataStateError::IllegalChildDataState { - edge: ExecutionDataStateEdge::SummaryMergeInput.describe(), - child: state, - }) - } - })?; - visit(input, s, assignment)?; - } - Ok(()) - } - SummaryExpr::SummaryEstimate { summary_input, .. } => { - let s = state_only(summary_input, ExecutionDataStateEdge::SummaryEstimateInput)?; - visit(summary_input, s, assignment) - } - SummaryExpr::ValueOperation { - child, - operation, - timing, - } => { - // Population timing is a lifecycle decision: a retained population - // is maintained at ingestion time, an ephemeral one is rebuilt - // from raw input per query. Its input and readout contracts are - // structural and hold either way. - let valid_population = match operation { - ValueOperation::MaintainPopulation { population } => { - matches!(&child.expr, SummaryExpr::KeepPreAsap(input) if population.matches_input(input)) - } - ValueOperation::ReadPopulation { readout } => { - *timing == ExecutionTiming::QueryTime - && matches!(&child.expr, SummaryExpr::ValueOperation { operation: ValueOperation::MaintainPopulation { population }, .. } if population.supports(readout)) - } - _ => true, - }; - if !valid_population { - return Err(ExecutionDataStateError::InvalidMaintainedPopulation); - } - let required = match timing { - ExecutionTiming::IngestionTime => ExecutionDataState::INGESTION_ROWS, - ExecutionTiming::QueryTime => ExecutionDataState::QUERY_ROWS, - }; - let s = produced_data_state(&child.expr).unwrap_or(required); - let exact_readout = (*timing == ExecutionTiming::QueryTime - || matches!(operation, ValueOperation::FinalizeExactAccumulator)) - && s.primitive == DataPrimitive::SummaryState - && (*timing == ExecutionTiming::QueryTime || s.timing == *timing) - && is_exact_accumulator_state(&child.schema).is_ok(); - // A query-time readout may read a population retained at ingestion. - let population_readout = matches!(operation, ValueOperation::ReadPopulation { .. }) - && *timing == ExecutionTiming::QueryTime - && matches!( - &child.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { .. }, - .. - } - ); - if s != required && !exact_readout && !population_readout { - return Err(ExecutionDataStateError::IllegalChildDataState { - edge: ExecutionDataStateEdge::ValueOperationChild.describe(), - child: s, - }); - } - check_plain_operands(operation, &child.schema)?; - visit(child, s, assignment) - } - } -} - -/// The data_state `child` takes as a direct input of `parent`, without -/// validating legality — `child`'s own produced data_state, or for a -/// `KeepPreAsap` leaf the data_state `parent`'s edge assigns it (update-path raw -/// input under maintenance-time operation edges, query-time fallback under a -/// a read-time operation, and — meaninglessly, but for a stable answer — maintenance rows -/// under a state-only edge). For DAG export and other reporting that needs -/// an explicit per-node data_state even on a plan that -/// [`validate_execution_data_states`] would reject. -pub fn assigned_child_data_state(parent: &SummaryExpr, child: &SummaryNode) -> ExecutionDataState { - if let Some(avail) = produced_data_state(&child.expr) { - return avail; - } - match parent { - SummaryExpr::ValueOperation { - timing: ExecutionTiming::QueryTime, - .. - } - | SummaryExpr::BinaryOp { - timing: ExecutionTiming::QueryTime, - .. - } => ExecutionDataState::QUERY_ROWS, - SummaryExpr::KeepPreAsap(_) - | SummaryExpr::BinaryOp { - timing: ExecutionTiming::IngestionTime, - .. - } - | SummaryExpr::RelationalJoin { .. } - | SummaryExpr::SummaryAgg { .. } - | SummaryExpr::SummaryJoin { .. } - | SummaryExpr::SummarySubtract { .. } - | SummaryExpr::SummaryDelete { .. } - | SummaryExpr::SummaryEstimate { .. } - | SummaryExpr::SummaryMerge { .. } - | SummaryExpr::ValueOperation { - timing: ExecutionTiming::IngestionTime, - .. - } => ExecutionDataState::INGESTION_ROWS, - } -} - -/// The data_state `child` takes on `edge`: its own produced data_state -/// (checked via `accept`), or — for a `KeepPreAsap` leaf — the data_state the -/// edge assigns it, derived from what that edge accepts. -fn child_domain( - child: &Rc, - edge: ExecutionDataStateEdge, - accept: impl Fn(ExecutionDataState) -> Result<(), ExecutionDataStateError>, -) -> Result { - match produced_data_state(&child.expr) { - Some(avail) => { - accept(avail)?; - Ok(avail) - } - None => { - // A raw pre-ASAP subtree executes at whichever data_state its consumer - // needs: update-path input for maintenance-time operation edges, - // query-time fallback for a read-time edge. State-only edges - // can't consume plain rows at all. - let assigned = match edge { - ExecutionDataStateEdge::SummaryAggChild - | ExecutionDataStateEdge::SummaryJoinInput - | ExecutionDataStateEdge::ValueOperationChild => ExecutionDataState::INGESTION_ROWS, - ExecutionDataStateEdge::SummaryEstimateInput - | ExecutionDataStateEdge::SummarySubtractInput - | ExecutionDataStateEdge::SummaryDeleteInput - | ExecutionDataStateEdge::SummaryMergeInput => { - return Err(ExecutionDataStateError::IllegalChildDataState { - edge: edge.describe(), - child: ExecutionDataState::INGESTION_ROWS, - }) - } - }; - accept(assigned)?; - Ok(assigned) - } - } -} - -fn state_only( - child: &Rc, - edge: ExecutionDataStateEdge, -) -> Result { - child_domain(child, edge, |avail| match avail { - state - if state.primitive == DataPrimitive::SummaryState - && (state.timing == ExecutionTiming::IngestionTime - || matches!(edge, ExecutionDataStateEdge::SummaryEstimateInput)) => - { - Ok(()) - } - other => Err(ExecutionDataStateError::IllegalChildDataState { - edge: edge.describe(), - child: other, - }), - }) -} - -/// The exact operator must consume only `Plain` columns of its input: for -/// an `Aggregate` payload, every grouping key and every measure's input -/// column. -fn check_plain_operands( - op: &ValueOperation, - input: &SummarySchema, -) -> Result<(), ExecutionDataStateError> { - if matches!( - op, - ValueOperation::Sort { .. } - | ValueOperation::Limit { .. } - | ValueOperation::Project { .. } - | ValueOperation::Filter { .. } - | ValueOperation::FinalizeExactAccumulator - ) { - return check_plain_or_exact_values(input); - } - let ValueOperation::Exact(op) = op else { - return check_all_plain(input); - }; - let ExactOperation::Aggregate { - reduction, - measures, - filters, - .. - } = op; - let mut referenced: Vec = reduction - .group_keys() - .map(|keys| keys.keys().to_vec()) - .unwrap_or_default(); - for m in measures { - referenced.extend(m.input_cols()); - } - for Predicate(f) in filters.iter().flatten() { - referenced.extend(f.columns_referenced().into_iter().copied()); - } - // With no explicit input column (the PromQL sample-value convention) - // the operator reads every non-key column, so all must be plain. - let implicit = measures.iter().any(|m| m.input_cols().is_empty()); - for (i, field) in input.fields.iter().enumerate() { - if !(implicit || referenced.contains(&i)) { - continue; - } - if !matches!(field.dtype, SummaryFamilyType::Plain(_)) { - return Err(ExecutionDataStateError::NonPlainOperand { - column: field.name.clone(), - dtype: format!("{:?}", field.dtype), - }); - } - } - Ok(()) -} - -fn check_plain_or_exact_values(input: &SummarySchema) -> Result<(), ExecutionDataStateError> { - for field in &input.fields { - if !matches!( - field.dtype, - SummaryFamilyType::Plain(_) | SummaryFamilyType::ExactAggregate(..) - ) { - return Err(ExecutionDataStateError::NonPlainOperand { - column: field.name.clone(), - dtype: format!("{:?}", field.dtype), - }); - } - } - Ok(()) -} - -fn check_all_plain(input: &SummarySchema) -> Result<(), ExecutionDataStateError> { - for field in &input.fields { - if !matches!(field.dtype, SummaryFamilyType::Plain(_)) { - return Err(ExecutionDataStateError::NonPlainOperand { - column: field.name.clone(), - dtype: format!("{:?}", field.dtype), - }); - } - } - Ok(()) -} - -/// The plain pre-ASAP `Schema` underlying an all-`Plain` `SummarySchema`, or -/// `None` if any column carries summary state. -pub fn plain_schema(schema: &SummarySchema) -> Option { - let mut columns = Vec::with_capacity(schema.fields.len()); - for field in &schema.fields { - let SummaryFamilyType::Plain(dtype) = &field.dtype else { - return None; - }; - columns.push(Column::new(&field.name, dtype.clone(), field.nullable)); - } - Some(Schema { - columns, - time_index: schema.time_index, - unique_keys: Vec::new(), - closed: true, - }) -} - -/// Lift a plain pre-ASAP schema to a `SummarySchema` with every column -/// `Plain` — the output of every exact operator. -pub fn lift_plain(schema: &Schema) -> SummarySchema { - SummarySchema { - fields: schema - .columns - .iter() - .map(|c| SummaryField { - name: c.name.clone(), - dtype: SummaryFamilyType::Plain(c.dtype.clone()), - nullable: c.nullable, - }) - .collect(), - time_index: schema.time_index, - } -} - -/// Output schema of `op` applied to a child whose edge carries `input` — -/// the same canonical derivation the pre-ASAP `Aggregate` node uses, so an -/// exact `ValueOperation` never disagrees with the pre-ASAP -/// target it was lowered from. `Err` when the child carries non-plain -/// state the operator cannot read. -pub fn exact_operation_output_schema( - op: &ExactOperation, - input: &SummarySchema, -) -> Result { - let plain = plain_schema(input).ok_or(ExactOperationSchemaError::NonPlainInput)?; - let ExactOperation::Aggregate { - reduction, - measures, - output_names, - .. - } = op; - let out = aggregate_output_schema(&plain, reduction, measures, output_names)?; - Ok(lift_plain(&out)) +/// `schema` as a summary-planning node output: fields and time axis kept, +/// unique keys dropped, closed. +pub fn lift_plain(schema: &Schema) -> Schema { + Schema::lifted(schema.fields.clone(), schema.time_index) } -/// Why [`exact_operation_output_schema`] could not derive a schema. +/// Why an exact operator's output schema could not be derived. #[derive(Debug, Error)] pub enum ExactOperationSchemaError { #[error("exact operator input carries summary state, not plain columns")] NonPlainInput, #[error("schema derivation failed: {0}")] - Schema(#[from] QueryExprError), + Schema(#[from] SchemaDerivationError), } #[cfg(test)] mod tests { use super::*; - use crate::post_asap::{ExactKind, ExactParams, GroupingStrategy, SketchQuery}; - use crate::pre_asap::agg_intent::AggIntent; - use crate::pre_asap::expr_ir::ColumnRef; - use crate::pre_asap::query_expr::{QueryExpr, Reduction, Source}; - use crate::pre_asap::schema::DataType; /// Both execution phases use raw values, distinct from maintained state. #[test] @@ -847,344 +179,6 @@ mod tests { assert_eq!(DataPrimitive::SummaryState.as_str(), "summary_state"); } - fn scan() -> Rc { - Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - Column::new("zone", DataType::Utf8, true), - ], - 0, - vec![], - ), - }) - } - - fn keep() -> Rc { - let s = scan(); - let schema = lift_plain(&s.output_schema().unwrap()); - Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(s), - schema, - guarantee: None, - }) - } - - fn plain(names: &[&str]) -> SummarySchema { - SummarySchema { - fields: names - .iter() - .map(|n| SummaryField { - name: (*n).into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), - nullable: false, - }) - .collect(), - time_index: None, - } - } - - fn agg(child: Rc, family: SummaryFamilyType) -> Rc { - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child, - family: family.clone(), - input: crate::post_asap::SummaryUpdate::column(ColumnRef::SampleValue), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "state".into(), - dtype: family, - nullable: false, - }], - time_index: None, - }, - guarantee: None, - }) - } - - fn kll() -> SummaryFamilyType { - use crate::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; - SummaryFamilyType::Sketch( - SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 200 }), - GroupingStrategy::default(), - ) - } - - fn estimate(child: Rc) -> Rc { - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: child, - query: SketchQuery::Quantile { q: 0.99 }, - }, - schema: plain(&["quantile_0_99"]), - guarantee: None, - }) - } - - fn max_op() -> ExactOperation { - ExactOperation::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::Max { col: None }], - output_names: vec![], - having: None, - filters: vec![], - } - } - - #[test] - fn keep_pre_asap_under_summary_agg_is_update_input() { - let leaf = keep(); - let root = agg(Rc::clone(&leaf), kll()); - let assignment = validate_execution_data_states(&root).unwrap(); - assert_eq!( - assignment.data_state_of(&leaf), - Some(ExecutionDataState::INGESTION_ROWS) - ); - assert_eq!( - assignment.data_state_of(&root), - Some(ExecutionDataState::INGESTION_SUMMARY) - ); - } - - // Typed derived updates retain one opaque series identity only when both - // operands cover the same series; arbitrary labels are never admitted. - #[test] - fn maintenance_binary_accepts_only_well_typed_series_identity() { - use crate::pre_asap::{ArithmeticOpKind, BinaryOpKind}; - let identity = crate::pre_asap::schema::PROMQL_SERIES_IDENTITY; - // A finalized per-series Sum of `metric`'s rows. - let operand = |metric: &str, schema: &SummarySchema| { - let rows = Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: metric.into(), - }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - ], - 0, - vec![], - ), - }); - let mut state = schema.clone(); - state.fields[0].dtype = - SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let sum = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(rows), - schema: schema.clone(), - guarantee: None, - }), - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), - input: crate::post_asap::SummaryUpdate::column(ColumnRef::SampleValue), - reduction: Reduction::PerEntity, - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: state, - guarantee: None, - }); - Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: sum, - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::IngestionTime, - }, - schema: schema.clone(), - guarantee: None, - }) - }; - let validate = |schema: SummarySchema, rhs: &str| { - let binary = Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - lhs: operand("m", &schema), - rhs: operand(rhs, &schema), - timing: ExecutionTiming::IngestionTime, - operator: crate::post_asap::BinaryOperator { - kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }, - }, - schema, - guarantee: None, - }); - validate_execution_data_states(&estimate(agg(binary, kll()))).map(|_| ()) - }; - let mut schema = plain(&["value"]); - schema.fields.push(SummaryField { - name: "ts".into(), - dtype: SummaryFamilyType::Plain(DataType::Timestamp), - nullable: false, - }); - schema.time_index = Some(1); - assert!(validate(schema.clone(), "m").is_ok()); - assert!( - validate(schema.clone(), "n").is_ok(), - "no identity to align" - ); - schema.fields.push(SummaryField { - name: identity.into(), - dtype: SummaryFamilyType::Plain(DataType::Utf8), - nullable: false, - }); - assert!(validate(schema.clone(), "m").is_ok()); - assert_eq!( - validate(schema.clone(), "n"), - Err(ExecutionDataStateError::InvalidMaintenanceBinary), - "different selectors may cover different series" - ); - for mutation in 0..4 { - let mut invalid = schema.clone(); - match mutation { - 0 => invalid.fields[2].nullable = true, - 1 => invalid.fields[2].dtype = SummaryFamilyType::Plain(DataType::Timestamp), - 2 => invalid.fields.push(invalid.fields[2].clone()), - _ => invalid.fields[2].name = "label".into(), - } - assert_eq!( - validate(invalid, "m"), - Err(ExecutionDataStateError::InvalidMaintenanceBinary) - ); - } - } - - #[test] - fn exact_accumulator_state_may_feed_another_summary_agg() { - let inner = agg( - keep(), - SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), - ); - let root = estimate(agg(inner, kll())); - assert!(validate_execution_data_states(&root).is_ok()); - } - - #[test] - fn readout_can_feed_summary_construction_at_query_time() { - let inner = estimate(agg(keep(), kll())); - let summary = agg(inner, kll()); - let root = estimate(summary.clone()); - let assignment = validate_execution_data_states(&root).unwrap(); - assert_eq!( - assignment.data_state_of(&summary).unwrap().timing, - ExecutionTiming::QueryTime - ); - } - - #[test] - fn query_time_operation_over_readout_is_legal_and_root_is_readout() { - let inner = estimate(agg(keep(), kll())); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: inner, - operation: ValueOperation::Exact(max_op()), - timing: ExecutionTiming::QueryTime, - }, - schema: plain(&["max"]), - guarantee: None, - }); - let assignment = validate_execution_data_states(&root).unwrap(); - assert_eq!( - assignment.data_state_of(&root), - Some(ExecutionDataState::QUERY_ROWS) - ); - } - - #[test] - fn non_exact_operator_uses_the_same_read_domain_contract() { - let inner = estimate(agg(keep(), kll())); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: inner, - operation: ValueOperation::Extension { - name: "approximate_calibration".into(), - }, - timing: ExecutionTiming::QueryTime, - }, - schema: plain(&["calibrated"]), - guarantee: None, - }); - - let assignment = validate_execution_data_states(&root).unwrap(); - assert_eq!( - assignment.data_state_of(&root), - Some(ExecutionDataState::QUERY_ROWS) - ); - } - - #[test] - fn query_time_values_can_feed_query_time_summary_construction() { - let inner = estimate(agg(keep(), kll())); - let post = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: inner, - operation: ValueOperation::Exact(max_op()), - timing: ExecutionTiming::QueryTime, - }, - schema: plain(&["max"]), - guarantee: None, - }); - let root = agg(post, kll()); - let root = estimate(root); - validate_execution_data_states(&root).unwrap(); - } - - #[test] - fn function_under_summary_agg_is_legal_but_not_at_root() { - let operation = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: keep(), - operation: ValueOperation::Exact(max_op()), - timing: ExecutionTiming::IngestionTime, - }, - schema: plain(&["max"]), - guarantee: None, - }); - assert_eq!( - validate_execution_data_states(&operation).err(), - Some(ExecutionDataStateError::MaintenanceRowsAtRoot) - ); - let root = estimate(agg(Rc::clone(&operation), kll())); - let assignment = validate_execution_data_states(&root).unwrap(); - assert_eq!( - assignment.data_state_of(&operation), - Some(ExecutionDataState::INGESTION_ROWS) - ); - } - - #[test] - fn function_over_readout_is_rejected() { - let inner = estimate(agg(keep(), kll())); - let operation = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: inner, - operation: ValueOperation::Exact(max_op()), - timing: ExecutionTiming::IngestionTime, - }, - schema: plain(&["max"]), - guarantee: None, - }); - let root = agg(operation, kll()); - assert!(matches!( - validate_execution_data_states(&root), - Err(ExecutionDataStateError::IllegalChildDataState { - edge: "ValueOperation.child", - child: ExecutionDataState::QUERY_ROWS - }) - )); - } - #[test] fn execution_phase_wire_names_are_ingestion_and_query_time() { for (phase, name) in [ @@ -1201,154 +195,4 @@ mod tests { assert!(serde_json::from_str::("\"maintenance_time\"").is_err()); assert!(serde_json::from_str::("\"MaintenanceTime\"").is_err()); } - - #[test] - fn summary_merge_runs_at_ingestion_or_query_time() { - for timing in [ExecutionTiming::IngestionTime, ExecutionTiming::QueryTime] { - let input = agg(keep(), kll()); - let merged = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - children: vec![input.clone()], - timing, - }, - schema: input.schema.clone(), - guarantee: None, - }); - let root = estimate(merged.clone()); - let assignment = validate_execution_data_states(&root).unwrap(); - assert_eq!( - assignment.data_state_of(&merged), - Some(ExecutionDataState { - timing, - primitive: DataPrimitive::SummaryState, - }) - ); - let exported = crate::post_asap::compile_post_asap_dag(&root).unwrap(); - assert!(exported.nodes.iter().any(|node| matches!(node.payload, - crate::post_asap::PostAsapOperatorPayload::SummaryMerge - if node.output_state.timing == timing))); - } - } - - #[test] - fn ingestion_merge_cannot_depend_on_query_execution() { - let input = agg(keep(), kll()); - let query_merge = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - children: vec![input.clone()], - timing: ExecutionTiming::QueryTime, - }, - schema: input.schema.clone(), - guarantee: None, - }); - let ingestion_merge = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - children: vec![query_merge], - timing: ExecutionTiming::IngestionTime, - }, - schema: input.schema.clone(), - guarantee: None, - }); - assert!(validate_execution_data_states(&ingestion_merge).is_err()); - } - - #[test] - fn a_shared_keep_pre_asap_reached_in_two_domains_is_ambiguous() { - // One raw subtree used both as update input (under a SummaryAgg) and - // as a query-time fallback (under an ExactRead) — no single - // execution can serve both, so the plan is rejected. - let shared = keep(); - let maintained = estimate(agg(Rc::clone(&shared), kll())); - let post_over_raw = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: Rc::clone(&shared), - operation: ValueOperation::Exact(max_op()), - timing: ExecutionTiming::QueryTime, - }, - schema: plain(&["max"]), - guarantee: None, - }); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: ExecutionTiming::IngestionTime, - children: vec![ - Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: maintained, - operation: ValueOperation::Exact(max_op()), - timing: ExecutionTiming::QueryTime, - }, - schema: plain(&["max"]), - guarantee: None, - }), - post_over_raw, - ], - }, - schema: plain(&["max"]), - guarantee: None, - }); - // SummaryMerge only accepts state, so this fails earlier for a - // different reason; probe the ambiguity through a direct visit. - let mut assignment = ExecutionDataStateAssignment::default(); - visit(&shared, ExecutionDataState::INGESTION_ROWS, &mut assignment).unwrap(); - assert_eq!( - visit(&shared, ExecutionDataState::QUERY_ROWS, &mut assignment), - Err(ExecutionDataStateError::AmbiguousKeepPreAsap { - first: ExecutionDataState::INGESTION_ROWS, - second: ExecutionDataState::QUERY_ROWS, - }) - ); - assert!(validate_execution_data_states(&root).is_err()); - } - - // Both paired operands must be plain; an unrelated state column is not an input. - #[test] - fn pearson_corr_checks_both_operand_states() { - let operation = ValueOperation::Exact(ExactOperation::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::PearsonCorr { left: 0, right: 1 }], - output_names: vec![], - having: None, - filters: vec![], - }); - for operand in [0, 1] { - let mut input = plain(&["x", "y", "unused"]); - input.fields[operand].dtype = kll(); - assert!(matches!( - check_plain_operands(&operation, &input), - Err(ExecutionDataStateError::NonPlainOperand { .. }) - )); - } - let mut input = plain(&["x", "y", "unused"]); - input.fields[2].dtype = kll(); - check_plain_operands(&operation, &input).unwrap(); - } - - #[test] - fn exact_operator_schema_matches_pre_asap_aggregate_derivation() { - let child_schema = lift_plain(&scan().output_schema().unwrap()); - let op = ExactOperation::Aggregate { - reduction: Reduction::by(vec![2]), - measures: vec![AggIntent::Max { col: None }], - output_names: vec![], - having: None, - filters: vec![], - }; - let out = exact_operation_output_schema(&op, &child_schema).unwrap(); - let names: Vec<_> = out.fields.iter().map(|f| f.name.as_str()).collect(); - assert_eq!(names, vec!["zone", "max"]); - assert!(out - .fields - .iter() - .all(|f| matches!(f.dtype, SummaryFamilyType::Plain(_)))); - } - - #[test] - fn exact_operator_rejects_non_plain_input() { - let state = agg(keep(), kll()); - assert!(matches!( - exact_operation_output_schema(&max_op(), &state.schema), - Err(ExactOperationSchemaError::NonPlainInput) - )); - } } diff --git a/crates/types/src/post_asap/expr.rs b/crates/types/src/post_asap/expr.rs deleted file mode 100644 index 6c48114a2..000000000 --- a/crates/types/src/post_asap/expr.rs +++ /dev/null @@ -1,285 +0,0 @@ -use super::ExecutionTiming; -use std::rc::Rc; - -use super::guarantee::ResultGuarantee; -use super::schema::{SummaryFamilyType, SummarySchema}; -use super::sketch::{GroupingStrategy, SketchQuery, SummaryUpdate}; -use crate::pre_asap::agg_intent::AggIntent; -use crate::pre_asap::query_expr::Predicate; -use crate::pre_asap::{ - BinaryOpKind, ColumnRef, GroupKeys, JoinKind, ProjectItem, QueryExpr, Reduction, SortKey, - VectorMatch, -}; - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[non_exhaustive] -pub enum ExactOperation { - Aggregate { - reduction: Reduction, - measures: Vec, - output_names: Vec, - /// Per-measure row predicates parallel to `measures`, positional - /// against the child's output rows — the same contract as - /// `QueryExpr::Aggregate.filters` (issue #466). - #[serde(default)] - filters: Vec>, - having: Option, - }, -} - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[non_exhaustive] -pub enum ValueOperation { - /// Maintain the full declared population, including membership changes, - /// so removing a TopK member can promote another. - MaintainPopulation { - population: super::maintained_population::MaintainedPopulation, - }, - /// Read an aggregate or TopK prefix from the maintained population. - ReadPopulation { - readout: super::maintained_population::PopulationReadout, - }, - Exact(ExactOperation), - /// Read an exact accumulator's state as its finalized scalar value. - /// - /// Exact accumulators do not need an estimator, but the explicit node - /// marks the maintenance-to-read boundary before query-time operators - /// such as PromQL binary arithmetic, sorting, and limiting. - FinalizeExactAccumulator, - /// Query-time column projection. SQL lowering retains the SELECT list as - /// a `Project` above its aggregate, so the post-ASAP DAG must preserve - /// its expressions, aliases, and optional derived-table qualifier while - /// allowing the aggregate child to be planned independently. - Project { - cols: Vec, - qualifier: Option, - }, - /// Query-time row filtering. The predicate remains positional against - /// the child's output schema and is evaluated only after any summary - /// state below it has been read out to rows. - Filter { - pred: Predicate, - }, - /// Query-time ordering of the child's value rows. This is deliberately - /// distinct from frequency-sketch heavy-hitter readout: PromQL `topk` - /// ranks the values produced by its child at the evaluation timestamp. - Sort { - keys: Vec, - partition_by: GroupKeys, - }, - /// Query-time row selection, normally composed over [`Self::Sort`] for - /// PromQL `topk`/`bottomk` and SQL `ORDER BY … LIMIT`. - Limit { - n: usize, - offset: usize, - /// Apply the offset and limit independently to each group. - partition_by: GroupKeys, - }, - Extension { - name: String, - }, -} - -/// Whether a candidate-membership sidecar is proven to contain every true -/// top-k key or is an explicitly approximate optimization. -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[serde(tag = "kind", rename_all = "snake_case")] -pub enum CandidateCompleteness { - Certified { guarantee: ResultGuarantee }, - BestEffort { guarantee: Option }, -} - -// ── Post-ASAP DAG node ─────────────────────────────────────────────────────── - -/// A node in the post-ASAP DAG: wraps the expression and its derived output -/// schema so every edge carries a typed schema. `SummarySchema` may contain -/// summary-state-typed columns (`SummaryFamilyType`'s non-`Plain` variants); -/// the pre-ASAP `Schema` cannot. -#[derive(Debug, Clone, PartialEq)] -pub struct SummaryNode { - pub expr: SummaryExpr, - /// Output schema of `expr` — the schema of the data flowing on the edge - /// leading *from* this node to its parent(s). - pub schema: SummarySchema, - /// The machine-readable accuracy guarantee of the *value* this node - /// produces (issue #172) — `Some` on every finalized, caller-visible - /// value: a `SummaryEstimate` readout, an `ExactAggregate`-family - /// `SummaryAgg` (its state *is* the value), or a `KeepPreAsap` subtree - /// (executed exactly). `None` on raw summary state — a sketch-family - /// `SummaryAgg`, `SummaryMerge`, `SummarySubtract`, `SummaryDelete`, - /// `SummaryJoin` — whose guarantee only exists once something reads it - /// out; and `None` on a readout of a family the plugged-in - /// `AccuracyModel` has no local guarantee for (`Sample`/`Wavelet`/ - /// `StatModel`), which a fail-closed consumer must treat as "unknown", - /// never as exact. - pub guarantee: Option, -} - -// ── Post-ASAP sketch-bound IR ──────────────────────────────────────────────── - -/// Sketch-bound IR produced by post-ASAP binding and final selection. Binding -/// rules selectively replace logical aggregates and joins in the pre-ASAP -/// `QueryExpr` with summary-bound counterparts. Final selection can retain -/// supported read-time value operations around independently planned children; -/// other unsupported subtrees pass through as `KeepPreAsap(Rc)`. -/// -/// Traversing from the root node yields a DAG; shared sub-expressions appear -/// as multiple `Rc` references to the same `SummaryNode`. -#[derive(Debug, Clone, PartialEq)] -pub enum SummaryExpr { - /// A pre-ASAP subtree kept as-is because it has no selected implementation - /// or supported residual decomposition. Output schema is the inner node's - /// schema, lifted to `SummarySchema` with all fields as - /// `SummaryFamilyType::Plain`. - KeepPreAsap(Rc), - - /// A PromQL binary operation whose operands were planned independently. - /// This keeps realizable summary/readout leaves visible instead of - /// hiding the complete expression inside `KeepPreAsap`. - BinaryOp { - timing: ExecutionTiming, - lhs: Rc, - rhs: Rc, - operator: BinaryOperator, - }, - - /// Plain-row semantics composed with a post-ASAP child. Timing is an - /// independent physical choice, not part of the operation's identity. - ValueOperation { - child: Rc, - operation: ValueOperation, - timing: super::execution_data_state::ExecutionTiming, - }, - - /// Read-time relational join over two row-producing children. This is - /// distinct from [`SummaryJoin`](Self::SummaryJoin), which combines - /// summary states for join estimation during maintenance. - RelationalJoin { - left: Rc, - right: Rc, - kind: JoinKind, - pred: Predicate, - /// Optional proof for candidate pruning; ranking remains a separate operation. - pruning: Option, - }, - - /// Summary aggregation. Post-ASAP binding chose `family` — which - /// summary family (exact accumulator, sketch, sample, wavelet, or - /// statistical model) and its `(kind, params)` — from the catalog for - /// `AggIntent` under `DeploymentConstraints`. - /// Output schema: grouping columns (verbatim) + one field carrying - /// partial summary state per group, typed `family`. - SummaryAgg { - child: Rc, - /// Which summary family realizes this aggregation, and that - /// family's own `(kind, params)`. Never `SummaryFamilyType::Plain` - /// — this node always produces summary state, not a plain value. - family: SummaryFamilyType, - /// Optional multidimensional item identity and the observation/update - /// weight fed into each state update. Subpopulation semantics remain - /// on `reduction`; physical sharing remains on `grouping`. - input: SummaryUpdate, - /// How this aggregation's output rows relate to `child`'s — the - /// same [`Reduction`] the pre-ASAP `Aggregate` node it was bound - /// from carried (issue #165), reused verbatim rather than - /// flattened to a bare `Vec`. `Reduction::Reduce(by)` - /// with an empty `by` is a genuine full reduction (merge every - /// candidate into one group); `Reduction::PerEntity` has no - /// grouping concept at all (never merge across entities) — the - /// two collapsed to the same ambiguous `by: []` before this field - /// existed (issue #163). - reduction: Reduction, - /// How this aggregation's summary state is physically instantiated - /// across `reduction`'s subpopulations — one independent instance - /// per `by` key (today's only behavior, and this field's default), - /// or one shared Hydra-family structure serving all of them (issue - /// #256). Lives here, next to `reduction`, for planning, and is also - /// encoded in sketch-valued `family`/output-schema state so merges - /// can reject incompatible layouts. `reduction` is the field that - /// carries the `by` keys this axis's legality depends on (a - /// `SharedMultiSubpopulation` choice only makes sense when - /// `reduction` actually has a subpopulation concept — see - /// `asap_aware_mapping::grouping`'s module docs for the legality - /// rules). Every existing producer of a `SummaryAgg` sets this to - /// `GroupingStrategy::PerSubpopulationInstance` (its `Default`), - /// so no existing behavior changes. - grouping: GroupingStrategy, - /// Row predicate gating this summary's updates (issue #466): only - /// rows where it is `TRUE` update the state; grouping keys are - /// still read from every row. Positional against `child`'s output. - /// A field rather than a `Filter` child so summaries that differ - /// only in predicate can still share one child. No binding rule - /// sets it yet — a filtered pre-ASAP measure stays `KeepPreAsap` — - /// so every producer today writes `None`. - filter: Option, - }, - - /// Summary-aware join (KMV / theta for join-cardinality; join-sample for - /// sampling). Emitted only when a `Bind*OnJoin` rule fires. - /// Output schema: one field typed `family`, read by a downstream - /// `SummaryEstimate`. - SummaryJoin { - outer: Rc, - inner: Rc, - key: ColumnRef, - /// Never `SummaryFamilyType::Plain` — see [`SummaryAgg::family`](SummaryExpr::SummaryAgg). - family: SummaryFamilyType, - }, - - /// Subtract one summary from another. Valid only for families with a - /// linear-inverse property (CMS, theta, count-based). Catalog flag - /// `subtractable` must be true for the family. - /// Output schema: one field (same family + params as inputs). - SummarySubtract { - left: Rc, - right: Rc, - }, - - /// Delete a key from a summary (CMS update with −1, deletable Bloom - /// filter). Catalog flag `deletable` must be true. Output schema = - /// input schema unchanged in type (same field type as input). - SummaryDelete { - summary_input: Rc, - key: ColumnRef, - }, - - /// Read out a query result from a built summary. The summary-state field - /// type does *not* propagate downstream of an estimate — the output - /// schema is a regular row-shaped schema (Float64 for quantile, Int64 - /// for count/cardinality, `[(key, count)]` for top-k). - SummaryEstimate { - summary_input: Rc, - query: SketchQuery, - }, - - /// ⊕ — union of summaries across stages / shards. Distinct from the - /// pre-ASAP `Concat` because summary union has type constraints: all - /// inputs must agree on `family` (kind + params) and the catalog flag - /// `mergeable` must be true. Inserted by a deployment's own stage - /// allocator (not modeled in this crate) on cut edges. - /// Output schema: one field (same family + params as inputs). - SummaryMerge { - children: Vec>, - timing: ExecutionTiming, - }, -} - -/// All semantics owned by a post-ASAP binary operator. -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -pub struct BinaryOperator { - /// Execute division only for finite operands, a nonzero divisor, and a - /// normal finite result; otherwise use exact execution. Required by the - /// relative-value division certificate, including floating-point range. - #[serde(default)] - pub checked_relative_division: bool, - /// Conditional exact rewrites (such as temporal average from sum/count) - /// require finite operands and quotient. Zero/subnormal results are valid; - /// overflow must fall back to the original query rather than emit infinity. - #[serde(default)] - pub checked_finite_division: bool, - pub kind: BinaryOpKind, - /// `None` is the only currently supported vector/vector matching mode. - /// The field is retained so execution never has to recover semantics by - /// re-parsing PromQL. - pub vector_match: Option, -} diff --git a/crates/types/src/post_asap/guarantee.rs b/crates/types/src/post_asap/guarantee.rs index 88e070ce9..8b1f3ee9d 100644 --- a/crates/types/src/post_asap/guarantee.rs +++ b/crates/types/src/post_asap/guarantee.rs @@ -16,9 +16,9 @@ //! ## What a guarantee says //! //! [`ResultGuarantee`] is attached to a finalized, caller-visible value — -//! [`super::SummaryNode::guarantee`] on a `SummaryEstimate` readout, an -//! exact accumulator, or a kept pre-ASAP subtree — never to raw summary -//! state (a `SummaryAgg` sketch node carries `None`; its readout carries the +//! [`crate::ir::OperatorNode::guarantee`] on a `SummaryEstimate` evaluation, an +//! exact accumulator, or a kept pre-ASAP sub-DAG — never to raw summary +//! state (a `SummaryAgg` sketch node carries `None`; its evaluation carries the //! guarantee). Its statement is: //! //! ```text @@ -325,13 +325,13 @@ pub enum GuaranteeSource { /// Deterministic exact computation — zero error by construction. Exact { /// What made it exact (e.g. `"ExactAggregate(Sum)"`, - /// `"KeepPreAsap"`). + /// `"RetainedExact"`). reason: String, }, - /// The target this readout's sketch was sized against. + /// The target this evaluation's sketch was sized against. AccuracyTarget { target: AccuracyTarget }, - /// The concrete sketch a readout's local guarantee was derived from. - SketchReadout { + /// The concrete sketch a evaluation's local guarantee was derived from. + SketchEvaluation { algorithm: String, /// Stable estimator/analysis contract used to derive this guarantee. #[serde(default)] @@ -421,8 +421,8 @@ impl ResultGuarantee { self.bound.is_zero() && self.failure_probability.is_zero() } - /// How many approximate sketch readouts contributed to this value — - /// `1` for a plain readout, `0` for an exact value, and the transitive + /// How many approximate sketch evaluations contributed to this value — + /// `1` for a plain evaluation, `0` for an exact value, and the transitive /// count through every [`GuaranteeSource::ChildGuarantee`] for a /// composition. An `AccuracyBudgetAllocator` uses this as the number /// of layers a budget must be split across. @@ -430,7 +430,7 @@ impl ResultGuarantee { self.provenance .iter() .map(|source| match source { - GuaranteeSource::SketchReadout { .. } => 1, + GuaranteeSource::SketchEvaluation { .. } => 1, GuaranteeSource::ChildGuarantee { guarantee, .. } => { guarantee.approximate_layer_count() } diff --git a/crates/types/src/post_asap/maintained_population.rs b/crates/types/src/post_asap/maintained_population.rs index 3939c7c13..34ca37afe 100644 --- a/crates/types/src/post_asap/maintained_population.rs +++ b/crates/types/src/post_asap/maintained_population.rs @@ -1,4 +1,4 @@ -//! Language-independent maintained populations and their readouts. +//! Language-independent maintained populations and their evaluations. //! Resource limits, ingestion placement and data structures belong to the executor. use serde::{Deserialize, Serialize}; @@ -27,7 +27,7 @@ pub enum CurrentSeriesMatch { } #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub enum PopulationReadout { +pub enum PopulationStatistic { Quantile { q: f64 }, TopK { k: usize }, Sum, @@ -37,27 +37,28 @@ pub enum PopulationReadout { impl CurrentSeriesInput { /// Verify the named contract against the canonical maintenance input. - pub fn matches_input(&self, input: &crate::pre_asap::QueryExpr) -> bool { - use crate::pre_asap::{CompareOpKind, DataType, QueryExpr, ScalarValue, Source}; - // PromQL instant selectors carry an ingestion-interval `TimeRange` as - // their input scope. The population must use the same expiry horizon; - // shifted and otherwise transformed inputs still fail below. - let input = match input { - QueryExpr::TimeRange { range, child } - if self.lookback_ms > 0 - && *range == std::time::Duration::from_millis(self.lookback_ms) => + pub fn matches_node(&self, input: &crate::ir::OperatorNode) -> bool { + use crate::ir::{NonASAPOp, Operator, ScalarExpr, TimeRangeKind}; + use crate::pre_asap::{CompareOpKind, DataType, ScalarValue, Source}; + let input = match &input.operator { + Operator::NonASAP(NonASAPOp::TimeRange { + range, + kind: TimeRangeKind::Instant, + child, + }) if self.lookback_ms > 0 + && *range == std::time::Duration::from_millis(self.lookback_ms) => { child.as_ref() } - QueryExpr::TimeRange { .. } => return false, - other if self.lookback_ms == 300_000 => other, + Operator::NonASAP(NonASAPOp::TimeRange { .. }) => return false, + _ if self.lookback_ms == 300_000 => input, _ => return false, }; - let QueryExpr::Scan { + let Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric }, predicates, schema, - } = input + }) = &input.operator else { return false; }; @@ -70,7 +71,7 @@ impl CurrentSeriesInput { } if self.grouping.iter().any(|label| { !schema - .columns + .fields .iter() .any(|c| c.name == *label && c.dtype == DataType::Utf8) }) { @@ -78,15 +79,18 @@ impl CurrentSeriesInput { } let mut matchers = Vec::new(); for predicate in predicates { - let QueryExpr::Compare { left, op, right } = predicate.0.as_ref() else { + let ScalarExpr::Compare { + left, op, right, .. + } = &predicate.0 + else { return false; }; - let (QueryExpr::Column(col), QueryExpr::Literal(ScalarValue::Utf8(value))) = + let (ScalarExpr::Column(col), ScalarExpr::Literal(ScalarValue::Utf8(value))) = (left.as_ref(), right.as_ref()) else { return false; }; - let Some(column) = schema.columns.get(*col) else { + let Some(column) = schema.fields.get(*col) else { return false; }; if column.dtype != DataType::Utf8 { @@ -117,7 +121,7 @@ impl CurrentSeriesInput { pub enum PopulationInput { CurrentSeries(CurrentSeriesInput), Rows { - input: std::rc::Rc, + input: std::rc::Rc, value_column: usize, grouping: crate::pre_asap::GroupKeys, }, @@ -131,28 +135,48 @@ pub struct MaintainedPopulation { } impl MaintainedPopulation { - pub fn matches_input(&self, input: &crate::pre_asap::QueryExpr) -> bool { + /// Whether `input` is the maintenance input this population declares. + pub fn matches_node(&self, input: &crate::ir::OperatorNode) -> bool { + use crate::ir::{NonASAPOp, Operator}; + use crate::pre_asap::{DataType, Source}; match &self.input { - PopulationInput::CurrentSeries(spec) => spec.matches_input(input), + PopulationInput::CurrentSeries(spec) => spec.matches_node(input), PopulationInput::Rows { input: expected, value_column, grouping, } => { - use crate::pre_asap::{DataType, QueryExpr, Source}; - expected.as_ref() == input - && matches!(input, QueryExpr::Scan { source: Source::Table { .. }, schema, .. } - if schema.closed && schema.columns.get(*value_column).is_some_and(|c| c.dtype == DataType::Float64 && !c.nullable) - && !grouping.is_without() && grouping.keys().iter().all(|k| *k < schema.columns.len())) + let Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { .. }, + schema, + .. + }) = &input.operator + else { + return false; + }; + // The same computation, whatever accuracy or timing has + // been attached to the node since. + let same_source = + expected.operator == input.operator && expected.schema == input.schema; + same_source + && schema.closed + && schema + .fields + .get(*value_column) + .is_some_and(|c| c.dtype == DataType::Float64 && !c.nullable) + && !grouping.is_without() + && grouping.keys().iter().all(|k| *k < schema.fields.len()) } } } - pub fn supports(&self, readout: &PopulationReadout) -> bool { - match readout { - PopulationReadout::Quantile { q } => self.quantiles && q.is_finite(), - PopulationReadout::TopK { k } => *k <= self.max_k, - PopulationReadout::Sum | PopulationReadout::Count | PopulationReadout::Average => true, + pub fn supports(&self, evaluation: &PopulationStatistic) -> bool { + match evaluation { + PopulationStatistic::Quantile { q } => self.quantiles && q.is_finite(), + PopulationStatistic::TopK { k } => *k <= self.max_k, + PopulationStatistic::Sum + | PopulationStatistic::Count + | PopulationStatistic::Average => true, } } } diff --git a/crates/types/src/post_asap/mod.rs b/crates/types/src/post_asap/mod.rs index 6bf317420..e282fccad 100644 --- a/crates/types/src/post_asap/mod.rs +++ b/crates/types/src/post_asap/mod.rs @@ -1,75 +1,54 @@ -//! The post-ASAP IR: summary-bound types, distinct from -//! [`crate::pre_asap`]'s pre-ASAP IR. +//! Summary-state types and the execution-timing vocabulary of the operator +//! IR ([`crate::ir`]). //! -//! Where [`crate::pre_asap`] carries *intent* only ("compute a -//! quantile to ε accuracy"), this module is the summary-bound IR: the -//! summary family, kind/algorithm, and parameters are committed (one -//! `(Kind, Params)` pair per family — [`sketch::ExactKind`]/[`sketch::ExactParams`], +//! Where an intent ([`crate::pre_asap::AggIntent`]) says *what* to compute +//! ("a quantile to ε accuracy"), these types say *how* a summary realizes it: +//! the family, kind/algorithm and parameters are committed — one +//! `(Kind, Params)` pair per family ([`sketch::ExactKind`]/[`sketch::ExactParams`], //! [`sketch::SamplingKind`]/[`sketch::SamplingParams`], //! [`sketch::WaveletKind`]/[`sketch::WaveletParams`], -//! [`sketch::StatModelKind`]/[`sketch::StatModelParams`]), and -//! [`expr::SummaryNode`] / [`expr::SummaryExpr`] describe the summary -//! computation. The `Sketch` family is the one exception to that -//! one-pair-per-family shape: it nests a third level, [`sketch::SketchKind`] -//! (quantile/cardinality/frequency/top-k), which itself carries the -//! committed [`sketch::SketchAlgorithm`] and [`sketch::SketchParams`] — -//! `SummaryFamilyType::Sketch(SketchKind, GroupingStrategy)`, not a flat -//! `(kind, params)` pair -//! — because `Sketch` is the one family with more than one algorithm per -//! purpose today; no other family needs that extra level yet. +//! [`sketch::StatModelKind`]/[`sketch::StatModelParams`]). The `Sketch` family +//! nests a third level, [`sketch::SketchKind`] (quantile/cardinality/ +//! frequency/top-k), carrying the committed [`sketch::SketchAlgorithm`] and +//! [`sketch::SketchParams`], because it is the one family with more than one +//! algorithm per purpose. //! -//! A second, orthogonal axis lives here too: [`sketch::GroupingStrategy`] -//! (issue #256) — *how many* physical instances of a chosen family/kind -//! exist across a grouped aggregate's `by` subpopulations -//! (`PerSubpopulationInstance`, today's only behavior, vs. -//! `SharedMultiSubpopulation`/Hydra — see [`sketch::HydraKind`]/ -//! [`sketch::HydraParams`]), carried on [`expr::SummaryExpr::SummaryAgg`] -//! alongside `reduction` and on sketch-valued edge types -//! — see `asap_aware_mapping::grouping`'s module docs for why. +//! [`sketch::GroupingStrategy`] is a second, orthogonal axis: how many +//! physical instances of a summary exist across a grouped aggregate's `by` +//! subpopulations (per-subpopulation vs. one shared Hydra instance — see +//! `asap_aware_mapping::grouping`). It rides on `ASAPOp::SummaryAgg` and on +//! sketch-valued edge types. +//! +//! The rest: accuracy guarantees ([`guarantee`]), maintained populations, +//! summary windows and maintenance lifecycle, and the execution timing / +//! data-state vocabulary ([`execution_data_state`]). -pub mod cse; pub mod execution_data_state; -pub mod expr; pub mod guarantee; pub mod maintained_population; -pub mod post_asap_dag; pub mod query_time; -pub mod schema; pub mod sketch; pub mod summary_maintenance; pub mod summary_maintenance_lifecycle; pub mod summary_window; -pub use cse::share_common_summary_subtrees; +pub use crate::pre_asap::schema::{Field, FieldDataType, Schema}; pub use execution_data_state::{ - assigned_child_data_state, exact_operation_output_schema, produced_data_state, - validate_execution_data_states, validate_execution_data_states_at, DataPrimitive, - ExactOperationSchemaError, ExecutionDataState, ExecutionDataStateAssignment, + lift_plain, DataPrimitive, ExactOperationSchemaError, ExecutionDataState, ExecutionDataStateError, ExecutionTiming, }; -pub use expr::{ - BinaryOperator, CandidateCompleteness, ExactOperation, SummaryExpr, SummaryNode, ValueOperation, -}; pub use guarantee::{ AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, GuaranteeSource, ProbabilityExpr, ResultGuarantee, }; -pub use post_asap_dag::{ - compile_post_asap_dag, compile_post_asap_dag_with_node_ids, EdgeRole, - GroupingEdgeCompatibility, PostAsapDag, PostAsapDagCompilation, PostAsapDagDocument, - PostAsapDagEdge, PostAsapDagNode, PostAsapDagValidationError, PostAsapNodeId, - PostAsapNodeIdentityMap, PostAsapOperatorPayload, WindowEdgeCompatibility, - POST_ASAP_DAG_WIRE_VERSION, -}; pub use query_time::{ classic_cms_sizing, cms_posterior_error_bound, count_sketch_posterior_error_bound, cu_sketch_posterior_error_bound, traditional_a_priori_bound, }; -pub use schema::{SummaryFamilyType, SummaryField, SummarySchema}; pub use sketch::{ default_hydra_params, hydra_kind_for, EntityIdentity, ExactKind, ExactParams, GroupingStrategy, HydraKind, HydraParams, NonNegativeWeightProof, SamplingKind, SamplingParams, SketchAlgorithm, - SketchCategory, SketchKind, SketchParams, SketchQuery, StatModelKind, StatModelParams, + SketchCategory, SketchKind, SketchParams, SketchStatistic, StatModelKind, StatModelParams, SummaryInputExpr, SummaryUpdate, WaveletKind, WaveletParams, WeightDomain, }; pub use summary_maintenance::SummaryMaintenanceMode; diff --git a/crates/types/src/post_asap/post_asap_dag.rs b/crates/types/src/post_asap/post_asap_dag.rs deleted file mode 100644 index adaabb1b3..000000000 --- a/crates/types/src/post_asap/post_asap_dag.rs +++ /dev/null @@ -1,866 +0,0 @@ -//! Runtime-neutral post-ASAP DAG contract shared by precompute and query engines. - -use std::collections::HashMap; -use std::rc::Rc; - -use super::{ - validate_execution_data_states, ExecutionDataState, ExecutionDataStateError, ResultGuarantee, - SummaryExpr, SummaryNode, SummarySchema, -}; -use super::{ - BinaryOperator, CandidateCompleteness, ExecutionTiming, GroupingStrategy, SketchQuery, - SummaryFamilyType, SummaryUpdate, ValueOperation, -}; -use crate::pre_asap::{ColumnRef, JoinKind, Predicate, QueryExpr, Reduction}; -use thiserror::Error; - -pub const POST_ASAP_DAG_WIRE_VERSION: u32 = 6; - -#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)] -pub enum EdgeRole { - Input, - Left, - Right, -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)] -pub enum GroupingEdgeCompatibility { - Identical, - ConsumerCoarsensProducer, - Incompatible, - NotApplicable, -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)] -pub enum WindowEdgeCompatibility { - /// Physical lowering must prove equal pane/query phase or install an - /// exact boundary residual. The logical DAG alone cannot make that claim. - #[serde(rename = "RequiresAlignedPanePhaseOrExactBoundaryResidual")] - RequiresAlignedPanePhaseOrExactWindowEdgeResidual, - NotApplicable, -} - -/// Stable identity of a node within one exported post-ASAP semantic DAG. -#[derive( - Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, serde::Serialize, serde::Deserialize, -)] -#[serde(transparent)] -pub struct PostAsapNodeId(pub u32); - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[serde(tag = "kind", rename_all = "snake_case", deny_unknown_fields)] -pub enum PostAsapOperatorPayload { - Fallback { - expression: QueryExpr, - }, - Binary { - operator: BinaryOperator, - }, - Value { - operation: ValueOperation, - }, - RelationalJoin { - join_kind: JoinKind, - pred: Predicate, - pruning: Option, - }, - SummaryAgg { - family: SummaryFamilyType, - input: SummaryUpdate, - reduction: Reduction, - grouping: GroupingStrategy, - /// See `SummaryExpr::SummaryAgg::filter`. Wire version 6 added it; - /// a version-5 reader would otherwise take a filtered summary as - /// unfiltered. - filter: Option, - }, - SummaryJoin { - key: ColumnRef, - family: SummaryFamilyType, - }, - SummarySubtract, - SummaryDelete { - key: ColumnRef, - }, - SummaryEstimate { - query: SketchQuery, - }, - SummaryMerge, -} - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[serde(deny_unknown_fields)] -pub struct PostAsapDagNode { - pub id: PostAsapNodeId, - /// The payload variant is the sole operator identity (`payload.kind` in JSON). - pub payload: PostAsapOperatorPayload, - /// Phase is a placement choice for every operator, independent of payload kind. - pub output_state: ExecutionDataState, - pub output_schema: SummarySchema, - pub guarantee: Option, -} - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[serde(deny_unknown_fields)] -pub struct PostAsapDagEdge { - pub producer: PostAsapNodeId, - pub consumer: PostAsapNodeId, - pub role: EdgeRole, - pub intermediate_schema: SummarySchema, - pub data_state: ExecutionDataState, - pub grouping: GroupingEdgeCompatibility, - pub window: WindowEdgeCompatibility, -} - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[serde(deny_unknown_fields)] -pub struct PostAsapDag { - pub nodes: Vec, - pub edges: Vec, - /// Semantic workload root. Physical query/precompute sinks are selected - /// downstream by the control plane. - pub root: PostAsapNodeId, -} - -/// Versioned transport envelope for a post-ASAP semantic DAG. -/// -/// Process boundaries exchange this envelope and call [`Self::validate`]. -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[serde(deny_unknown_fields)] -pub struct PostAsapDagDocument { - pub schema_version: u32, - pub dag: PostAsapDag, -} - -#[derive(Debug, Clone, PartialEq, Eq, Error)] -pub enum PostAsapDagValidationError { - #[error("phase assignment must name every DAG node exactly once")] - IncompletePhaseAssignment, - #[error("ingestion node {consumer:?} depends on query node {producer:?}")] - QueryDependencyInIngestion { - producer: PostAsapNodeId, - consumer: PostAsapNodeId, - }, - #[error("unsupported post-ASAP DAG schema version {0}")] - UnsupportedVersion(u32), - #[error("duplicate post-ASAP node id {0:?}")] - DuplicateNodeId(PostAsapNodeId), - #[error("post-ASAP DAG root {0:?} does not name a node")] - MissingRoot(PostAsapNodeId), - #[error("edge endpoint {0:?} does not name a node")] - MissingEdgeEndpoint(PostAsapNodeId), - #[error("edge {producer:?}->{consumer:?} schema differs from producer output")] - EdgeSchemaMismatch { - producer: PostAsapNodeId, - consumer: PostAsapNodeId, - }, - #[error("edge {producer:?}->{consumer:?} data state differs from producer output")] - EdgeDataStateMismatch { - producer: PostAsapNodeId, - consumer: PostAsapNodeId, - }, - #[error("post-ASAP DAG contains a cycle")] - Cycle, - #[error("post-ASAP node {0:?} is not reachable from the root")] - UnreachableNode(PostAsapNodeId), - #[error("summary aggregate node {node:?} output schema does not contain its declared family")] - SummaryFamilySchemaMismatch { node: PostAsapNodeId }, - #[error( - "summary aggregate node {node:?} declares grouping inconsistent with its sketch state" - )] - SummaryGroupingMismatch { node: PostAsapNodeId }, -} - -impl PostAsapDagDocument { - pub fn new(dag: PostAsapDag) -> Self { - Self { - schema_version: POST_ASAP_DAG_WIRE_VERSION, - dag, - } - } - - pub fn validate(&self) -> Result<(), PostAsapDagValidationError> { - if self.schema_version != POST_ASAP_DAG_WIRE_VERSION { - return Err(PostAsapDagValidationError::UnsupportedVersion( - self.schema_version, - )); - } - self.dag.validate() - } -} - -impl PostAsapDag { - /// Assign execution phases without changing operator semantics. Phase choices - /// do not prove deployment support: callers must bind concrete implementations - /// and storage boundaries before installing this plan. - pub fn with_execution_phases( - &self, - phases: &std::collections::BTreeMap, - ) -> Result { - self.validate()?; - if phases.len() != self.nodes.len() - || self.nodes.iter().any(|node| !phases.contains_key(&node.id)) - { - return Err(PostAsapDagValidationError::IncompletePhaseAssignment); - } - let mut dag = self.clone(); - for node in &mut dag.nodes { - node.output_state.timing = phases[&node.id]; - } - let states: HashMap<_, _> = dag.nodes.iter().map(|n| (n.id, n.output_state)).collect(); - for edge in &mut dag.edges { - edge.data_state = states[&edge.producer]; - } - dag.validate()?; - Ok(dag) - } - - pub fn validate(&self) -> Result<(), PostAsapDagValidationError> { - use std::collections::{HashMap, HashSet}; - let mut nodes = HashMap::new(); - for node in &self.nodes { - if nodes.insert(node.id, node).is_some() { - return Err(PostAsapDagValidationError::DuplicateNodeId(node.id)); - } - if let PostAsapOperatorPayload::SummaryAgg { - family, grouping, .. - } = &node.payload - { - let mut found_family = false; - for field in &node.output_schema.fields { - if &field.dtype == family { - found_family = true; - } - if let SummaryFamilyType::Sketch(_, schema_grouping) = &field.dtype { - if schema_grouping != grouping { - return Err(PostAsapDagValidationError::SummaryGroupingMismatch { - node: node.id, - }); - } - } - } - if !found_family { - return Err(PostAsapDagValidationError::SummaryFamilySchemaMismatch { - node: node.id, - }); - } - } - } - if !nodes.contains_key(&self.root) { - return Err(PostAsapDagValidationError::MissingRoot(self.root)); - } - let mut children: HashMap> = HashMap::new(); - for edge in &self.edges { - let producer = nodes.get(&edge.producer).ok_or( - PostAsapDagValidationError::MissingEdgeEndpoint(edge.producer), - )?; - if !nodes.contains_key(&edge.consumer) { - return Err(PostAsapDagValidationError::MissingEdgeEndpoint( - edge.consumer, - )); - } - if producer.output_state.timing == ExecutionTiming::QueryTime - && nodes[&edge.consumer].output_state.timing == ExecutionTiming::IngestionTime - { - return Err(PostAsapDagValidationError::QueryDependencyInIngestion { - producer: edge.producer, - consumer: edge.consumer, - }); - } - if edge.intermediate_schema != producer.output_schema { - return Err(PostAsapDagValidationError::EdgeSchemaMismatch { - producer: edge.producer, - consumer: edge.consumer, - }); - } - if edge.data_state != producer.output_state { - return Err(PostAsapDagValidationError::EdgeDataStateMismatch { - producer: edge.producer, - consumer: edge.consumer, - }); - } - children - .entry(edge.consumer) - .or_default() - .push(edge.producer); - } - fn visit( - id: PostAsapNodeId, - children: &HashMap>, - visiting: &mut HashSet, - visited: &mut HashSet, - ) -> bool { - if visited.contains(&id) { - return true; - } - if !visiting.insert(id) { - return false; - } - if children - .get(&id) - .into_iter() - .flatten() - .any(|child| !visit(*child, children, visiting, visited)) - { - return false; - } - visiting.remove(&id); - visited.insert(id); - true - } - if !visit( - self.root, - &children, - &mut HashSet::new(), - &mut HashSet::new(), - ) { - return Err(PostAsapDagValidationError::Cycle); - } - let mut reachable = HashSet::new(); - fn mark( - id: PostAsapNodeId, - children: &HashMap>, - reachable: &mut HashSet, - ) { - if !reachable.insert(id) { - return; - } - for child in children.get(&id).into_iter().flatten() { - mark(*child, children, reachable); - } - } - mark(self.root, &children, &mut reachable); - if let Some(id) = nodes.keys().find(|id| !reachable.contains(id)) { - return Err(PostAsapDagValidationError::UnreachableNode(*id)); - } - Ok(()) - } -} - -/// Compiler-local identity assignment. It deliberately retains `Rc` handles -/// and is not serialized; deployed artifacts persist the post-ASAP node ID -/// together with their physical materialization/query IDs. -#[derive(Debug, Clone)] -pub struct PostAsapNodeIdentityMap { - nodes_by_id: Vec>, -} - -impl PostAsapNodeIdentityMap { - pub fn node_id(&self, node: &Rc) -> Option { - self.nodes_by_id - .iter() - .position(|candidate| Rc::ptr_eq(candidate, node)) - .map(|id| PostAsapNodeId(id as u32)) - } - - pub fn summary_node(&self, id: PostAsapNodeId) -> Option<&Rc> { - self.nodes_by_id.get(id.0 as usize) - } -} - -#[derive(Debug, Clone)] -pub struct PostAsapDagCompilation { - pub dag: PostAsapDag, - pub node_ids: PostAsapNodeIdentityMap, -} - -pub fn compile_post_asap_dag( - root: &Rc, -) -> Result { - Ok(compile_post_asap_dag_with_node_ids(root)?.dag) -} - -pub fn compile_post_asap_dag_with_node_ids( - root: &Rc, -) -> Result { - let assignment = validate_execution_data_states(root)?; - let mut nodes = Vec::new(); - let mut edges = Vec::new(); - let mut ids = HashMap::new(); - let mut nodes_by_id = Vec::new(); - - fn visit( - node: &Rc, - assignment: &super::ExecutionDataStateAssignment, - ids: &mut HashMap<*const SummaryNode, PostAsapNodeId>, - nodes: &mut Vec, - edges: &mut Vec, - nodes_by_id: &mut Vec>, - ) -> PostAsapNodeId { - if let Some(id) = ids.get(&Rc::as_ptr(node)) { - return *id; - } - let children: Vec<(&Rc, EdgeRole)> = match &node.expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::BinaryOp { lhs, rhs, .. } => { - vec![(lhs, EdgeRole::Left), (rhs, EdgeRole::Right)] - } - - SummaryExpr::ValueOperation { child, .. } | SummaryExpr::SummaryAgg { child, .. } => { - vec![(child, EdgeRole::Input)] - } - SummaryExpr::RelationalJoin { left, right, .. } => { - vec![(left, EdgeRole::Left), (right, EdgeRole::Right)] - } - SummaryExpr::SummaryJoin { outer, inner, .. } => { - vec![(outer, EdgeRole::Left), (inner, EdgeRole::Right)] - } - SummaryExpr::SummarySubtract { left, right } => { - vec![(left, EdgeRole::Left), (right, EdgeRole::Right)] - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - vec![(summary_input, EdgeRole::Input)] - } - SummaryExpr::SummaryMerge { children, .. } => { - children.iter().map(|c| (c, EdgeRole::Input)).collect() - } - }; - let child_ids: Vec<_> = children - .iter() - .map(|(c, r)| (visit(c, assignment, ids, nodes, edges, nodes_by_id), *c, *r)) - .collect(); - let id = PostAsapNodeId(nodes.len() as u32); - let state = assignment - .data_state_of(node) - .expect("validated node has state"); - let payload = match &node.expr { - SummaryExpr::KeepPreAsap(expression) => PostAsapOperatorPayload::Fallback { - expression: (**expression).clone(), - }, - SummaryExpr::BinaryOp { operator, .. } => PostAsapOperatorPayload::Binary { - operator: operator.clone(), - }, - - SummaryExpr::ValueOperation { operation, .. } => PostAsapOperatorPayload::Value { - operation: operation.clone(), - }, - SummaryExpr::RelationalJoin { - kind, - pred, - pruning, - .. - } => PostAsapOperatorPayload::RelationalJoin { - join_kind: kind.clone(), - pred: pred.clone(), - pruning: pruning.clone(), - }, - SummaryExpr::SummaryAgg { - family, - input, - reduction, - grouping, - filter, - .. - } => PostAsapOperatorPayload::SummaryAgg { - family: family.clone(), - input: input.clone(), - reduction: reduction.clone(), - grouping: grouping.clone(), - filter: filter.clone(), - }, - SummaryExpr::SummaryJoin { key, family, .. } => PostAsapOperatorPayload::SummaryJoin { - key: key.clone(), - family: family.clone(), - }, - SummaryExpr::SummarySubtract { .. } => PostAsapOperatorPayload::SummarySubtract, - SummaryExpr::SummaryDelete { key, .. } => { - PostAsapOperatorPayload::SummaryDelete { key: key.clone() } - } - SummaryExpr::SummaryEstimate { query, .. } => { - PostAsapOperatorPayload::SummaryEstimate { - query: query.clone(), - } - } - SummaryExpr::SummaryMerge { .. } => PostAsapOperatorPayload::SummaryMerge, - }; - nodes.push(PostAsapDagNode { - id, - payload, - output_state: state, - output_schema: node.schema.clone(), - guarantee: node.guarantee.clone(), - }); - nodes_by_id.push(Rc::clone(node)); - ids.insert(Rc::as_ptr(node), id); - for (producer, child, role) in child_ids { - let maintenance_dependency = nodes[producer.0 as usize].output_state.timing - == ExecutionTiming::IngestionTime - && nodes[id.0 as usize].output_state.timing == ExecutionTiming::IngestionTime; - let grouping = match (&child.expr, &node.expr) { - ( - SummaryExpr::SummaryAgg { - reduction: producer, - .. - }, - SummaryExpr::SummaryAgg { - reduction: consumer, - .. - }, - ) if producer == consumer => GroupingEdgeCompatibility::Identical, - ( - SummaryExpr::SummaryAgg { - reduction: crate::pre_asap::Reduction::PerEntity, - .. - }, - SummaryExpr::SummaryAgg { - reduction: crate::pre_asap::Reduction::Reduce(_), - .. - }, - ) => GroupingEdgeCompatibility::ConsumerCoarsensProducer, - ( - SummaryExpr::SummaryAgg { - reduction: crate::pre_asap::Reduction::Reduce(producer), - .. - }, - SummaryExpr::SummaryAgg { - reduction: crate::pre_asap::Reduction::Reduce(consumer), - .. - }, - ) if !producer.is_without() - && !consumer.is_without() - && consumer.iter().all(|key| producer.contains(key)) => - { - GroupingEdgeCompatibility::ConsumerCoarsensProducer - } - (SummaryExpr::SummaryAgg { .. }, SummaryExpr::SummaryAgg { .. }) => { - GroupingEdgeCompatibility::Incompatible - } - _ => GroupingEdgeCompatibility::NotApplicable, - }; - edges.push(PostAsapDagEdge { - producer, - consumer: id, - role, - intermediate_schema: child.schema.clone(), - // The whole-graph validator owns contextual state assignment, - // especially for shared KeepPreAsap leaves. Export that - // authoritative result instead of independently deriving the - // edge state a second time. - data_state: assignment - .data_state_of(child) - .expect("validated child has data state"), - grouping, - window: if maintenance_dependency { - WindowEdgeCompatibility::RequiresAlignedPanePhaseOrExactWindowEdgeResidual - } else { - WindowEdgeCompatibility::NotApplicable - }, - }); - } - id - } - - let root = visit( - root, - &assignment, - &mut ids, - &mut nodes, - &mut edges, - &mut nodes_by_id, - ); - let dag = PostAsapDag { nodes, edges, root }; - dag.validate() - .expect("compiler emits a valid post-ASAP DAG"); - Ok(PostAsapDagCompilation { - dag, - node_ids: PostAsapNodeIdentityMap { nodes_by_id }, - }) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::post_asap::{ - ExactKind, ExactParams, ExecutionTiming, GroupingStrategy, SummaryFamilyType, SummaryField, - SummaryUpdate, ValueOperation, - }; - use crate::pre_asap::schema::{Column, Schema}; - use crate::pre_asap::{ColumnRef, DataType, QueryExpr, Reduction, Source}; - use std::collections::BTreeMap; - - #[test] - fn every_physical_payload_can_be_assigned_either_phase() { - use crate::post_asap::DataPrimitive; - use crate::pre_asap::{ArithmeticOpKind, BinaryOpKind, JoinKind, Predicate, ScalarValue}; - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let predicate = Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))); - let payloads = vec![ - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::Literal(ScalarValue::Int64(1)), - }, - PostAsapOperatorPayload::Binary { - operator: BinaryOperator { - checked_relative_division: false, - checked_finite_division: false, - kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - vector_match: None, - }, - }, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Limit { - n: 1, - offset: 0, - partition_by: Default::default(), - }, - }, - PostAsapOperatorPayload::RelationalJoin { - join_kind: JoinKind::Semi, - pred: predicate, - pruning: None, - }, - PostAsapOperatorPayload::SummaryAgg { - family: family.clone(), - input: SummaryUpdate::column(ColumnRef::SampleValue), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - PostAsapOperatorPayload::SummaryJoin { - key: ColumnRef::SampleValue, - family: family.clone(), - }, - PostAsapOperatorPayload::SummarySubtract, - PostAsapOperatorPayload::SummaryDelete { - key: ColumnRef::SampleValue, - }, - PostAsapOperatorPayload::SummaryEstimate { - query: SketchQuery::Cardinality, - }, - PostAsapOperatorPayload::SummaryMerge, - ]; - for payload in payloads { - // This checks physical identity and placement, not kernel availability. - let primitive = match &payload { - PostAsapOperatorPayload::Fallback { .. } - | PostAsapOperatorPayload::Binary { .. } - | PostAsapOperatorPayload::Value { .. } - | PostAsapOperatorPayload::RelationalJoin { .. } - | PostAsapOperatorPayload::SummaryEstimate { .. } => DataPrimitive::Raw, - PostAsapOperatorPayload::SummaryAgg { .. } - | PostAsapOperatorPayload::SummaryJoin { .. } - | PostAsapOperatorPayload::SummarySubtract - | PostAsapOperatorPayload::SummaryDelete { .. } - | PostAsapOperatorPayload::SummaryMerge => DataPrimitive::SummaryState, - }; - let dag = PostAsapDag { - root: PostAsapNodeId(0), - edges: vec![], - nodes: vec![PostAsapDagNode { - id: PostAsapNodeId(0), - payload: payload.clone(), - output_state: ExecutionDataState { - timing: ExecutionTiming::QueryTime, - primitive, - }, - output_schema: SummarySchema { - fields: vec![SummaryField { - name: "value".into(), - dtype: family.clone(), - nullable: false, - }], - time_index: None, - }, - guarantee: None, - }], - }; - for phase in [ExecutionTiming::IngestionTime, ExecutionTiming::QueryTime] { - let placed = dag - .with_execution_phases(&BTreeMap::from([(dag.root, phase)])) - .unwrap(); - assert_eq!(placed.nodes[0].payload, payload); - assert_eq!(placed.nodes[0].output_state.timing, phase); - let wire = serde_json::to_value(&placed).unwrap(); - assert!(wire["nodes"][0]["payload"].get("timing").is_none()); - assert_eq!(serde_json::from_value::(wire).unwrap(), placed); - } - assert!(dag.with_execution_phases(&BTreeMap::new()).is_err()); - } - } - - #[test] - fn phase_assignment_updates_edges_and_rejects_query_dependencies_in_ingestion() { - use crate::pre_asap::ScalarValue; - let schema = SummarySchema { - fields: vec![], - time_index: None, - }; - let nodes = [0, 1] - .into_iter() - .map(|id| PostAsapDagNode { - id: PostAsapNodeId(id), - payload: PostAsapOperatorPayload::Fallback { - expression: QueryExpr::Literal(ScalarValue::Int64(1)), - }, - output_state: ExecutionDataState::QUERY_ROWS, - output_schema: schema.clone(), - guarantee: None, - }) - .collect(); - let dag = PostAsapDag { - nodes, - root: PostAsapNodeId(1), - edges: vec![PostAsapDagEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), - role: EdgeRole::Input, - intermediate_schema: schema, - data_state: ExecutionDataState::QUERY_ROWS, - grouping: GroupingEdgeCompatibility::NotApplicable, - window: WindowEdgeCompatibility::NotApplicable, - }], - }; - let placed = dag - .with_execution_phases(&BTreeMap::from([ - (PostAsapNodeId(0), ExecutionTiming::IngestionTime), - (PostAsapNodeId(1), ExecutionTiming::QueryTime), - ])) - .unwrap(); - assert_eq!( - placed.edges[0].data_state.timing, - ExecutionTiming::IngestionTime - ); - assert_eq!(dag.edges[0].data_state.timing, ExecutionTiming::QueryTime); - assert!(matches!( - dag.with_execution_phases(&BTreeMap::from([ - (PostAsapNodeId(0), ExecutionTiming::QueryTime), - (PostAsapNodeId(1), ExecutionTiming::IngestionTime), - ])), - Err(PostAsapDagValidationError::QueryDependencyInIngestion { .. }) - )); - } - - #[test] - fn exports_summary_over_summary_as_typed_precompute_edges() { - let scan = Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), - }); - let raw = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(scan), - schema: SummarySchema { - fields: vec![SummaryField { - name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), - nullable: false, - }], - time_index: None, - }, - guarantee: None, - }); - let make_agg = |child: Rc, kind, params| { - let family = SummaryFamilyType::ExactAggregate(kind, params); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child, - family: family.clone(), - input: SummaryUpdate::column(ColumnRef::SampleValue), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "value".into(), - dtype: family, - nullable: false, - }], - time_index: None, - }, - guarantee: None, - }) - }; - let inner = make_agg(raw, ExactKind::Sum, ExactParams::Sum); - let outer = make_agg(Rc::clone(&inner), ExactKind::Sum, ExactParams::Sum); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: outer, - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::QueryTime, - }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), - nullable: false, - }], - time_index: None, - }, - guarantee: None, - }); - - let compiled = compile_post_asap_dag_with_node_ids(&root).unwrap(); - assert_eq!(compiled.node_ids.node_id(&root), Some(PostAsapNodeId(3))); - assert!(Rc::ptr_eq( - compiled.node_ids.summary_node(PostAsapNodeId(1)).unwrap(), - &inner - )); - let dag = compiled.dag; - assert_eq!(dag.root, PostAsapNodeId(3)); - assert_eq!( - dag.nodes[1].output_state, - ExecutionDataState::INGESTION_SUMMARY - ); - assert_eq!( - dag.nodes[2].output_state, - ExecutionDataState::INGESTION_SUMMARY - ); - let dependency = dag - .edges - .iter() - .find(|e| e.producer == PostAsapNodeId(1) && e.consumer == PostAsapNodeId(2)) - .unwrap(); - assert_eq!(dependency.data_state, ExecutionDataState::INGESTION_SUMMARY); - assert_eq!(dependency.grouping, GroupingEdgeCompatibility::Identical); - assert_eq!( - dependency.window, - WindowEdgeCompatibility::RequiresAlignedPanePhaseOrExactWindowEdgeResidual - ); - assert!(matches!( - dependency.intermediate_schema.fields[0].dtype, - SummaryFamilyType::ExactAggregate(ExactKind::Sum, _) - )); - let encoded = serde_json::to_string(&dag).expect("serialize post-ASAP DAG"); - let decoded: PostAsapDag = - serde_json::from_str(&encoded).expect("deserialize post-ASAP DAG"); - assert_eq!(decoded, dag); - let document = PostAsapDagDocument::new(decoded); - document.validate().unwrap(); - let mut invalid = serde_json::to_value(&document).unwrap(); - invalid["dag"]["nodes"][0]["operator"] = serde_json::json!("Binary"); - assert!(serde_json::from_value::(invalid).is_err()); - assert!(document.dag.nodes.iter().all(|node| { - let wire = serde_json::to_value(node).unwrap(); - wire.get("operator").is_none() && wire["payload"]["kind"].is_string() - })); - let mut old_version = document.clone(); - old_version.schema_version = 1; - assert_eq!( - old_version.validate(), - Err(PostAsapDagValidationError::UnsupportedVersion(1)) - ); - let mut unknown = serde_json::to_value(&document).unwrap(); - unknown["unexpected"] = serde_json::json!(true); - assert!(serde_json::from_value::(unknown).is_err()); - assert!(matches!( - dag.nodes[2].payload, - PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), - reduction: Reduction::Reduce(_), - .. - } - )); - } - - #[test] - fn post_asap_node_ids_serialize_in_deterministic_binding_order() { - let mut bindings = BTreeMap::new(); - bindings.insert(PostAsapNodeId(10), "materialization-10"); - bindings.insert(PostAsapNodeId(2), "query-2"); - assert_eq!( - serde_json::to_string(&bindings).unwrap(), - r#"{"2":"query-2","10":"materialization-10"}"# - ); - } -} diff --git a/crates/types/src/post_asap/query_time/error_estimation.rs b/crates/types/src/post_asap/query_time/error_estimation.rs index 663d2da12..f2bf229da 100644 --- a/crates/types/src/post_asap/query_time/error_estimation.rs +++ b/crates/types/src/post_asap/query_time/error_estimation.rs @@ -41,7 +41,7 @@ //! //! ## What this is *not* — no runtime sketch exists yet to wire this into //! -//! This issue names two possible integration points: (1) runtime/readout-time +//! This issue names two possible integration points: (1) runtime/evaluation-time //! accuracy reporting from a sketch's *actual* counters, and (2) tighter //! plan-time sizing. As of this module landing, **this repository has no //! vendored CMS/CountSketch/CU-Sketch runtime and no counter-array data @@ -52,12 +52,12 @@ //! planning-time sizing metadata. There is no `A[row][col]` counter matrix //! anywhere in the workspace for these functions to be handed at query //! time. So integration point (1) — reporting an *actual* query's posterior -//! error from real counters at readout — has nothing to wire into today. +//! error from real counters at evaluation — has nothing to wire into today. //! //! The functions here are deliberately **sketch-object-agnostic**: they take //! plain counter slices (`&[u64]` / `&[i64]`) and numeric parameters, not a //! concrete sketch type, specifically so that the moment a real CMS/ -//! Count-Sketch/CU-Sketch runtime lands in this workspace, its readout path +//! Count-Sketch/CU-Sketch runtime lands in this workspace, its evaluation path //! can call these functions directly on its real counter arrays with zero //! changes needed here. That wiring is out of scope for this module — see //! issue #239. @@ -255,8 +255,8 @@ fn posterior_rank(w: usize, rows: u32, delta: f64) -> Option { /// The `k`-th largest value in `values` (1-indexed: `k=1` is the max). /// `select_nth_unstable_by` partitions in O(w) average instead of fully /// sorting in O(w log w) — this only ever needs one rank, not a total -/// order, and both call sites (this module's per-query readout math) are -/// documented as meant to run on a future runtime's hot readout path. +/// order, and both call sites (this module's per-query evaluation math) are +/// documented as meant to run on a future runtime's hot evaluation path. fn kth_largest(values: &[u64], k: usize) -> u64 { let mut buf: Vec = values.to_vec(); let idx = k - 1; diff --git a/crates/types/src/post_asap/schema.rs b/crates/types/src/post_asap/schema.rs deleted file mode 100644 index d67e249df..000000000 --- a/crates/types/src/post_asap/schema.rs +++ /dev/null @@ -1,64 +0,0 @@ -use super::sketch::{ - ExactKind, ExactParams, GroupingStrategy, SamplingKind, SamplingParams, SketchKind, - StatModelKind, StatModelParams, WaveletKind, WaveletParams, -}; -use crate::pre_asap::DataType; - -// ── Post-ASAP data types ──────────────────────────────────────────────────── - -/// Column types that may appear on a post-ASAP DAG edge. A strict superset of -/// the pre-ASAP [`DataType`]: adds one variant per summary *family* for -/// edges that carry partial summary state between a `SummaryAgg` and a -/// downstream `SummaryEstimate` or `SummaryMerge`. -/// -/// Every non-`Plain` variant carries the physical state identity required by -/// that family (`Sketch` additionally carries its grouping layout), so the -/// type system can reject merges of incompatible -/// summaries at plan construction time — a `SummaryMerge` over -/// `Sketch(Kll, …)` and `Sketch(Cms, …)` inputs is a plan-time error, and a -/// `Sketch(…)` can never be confused for a `Sample(…)` even though both are -/// "opaque summary state" at a glance. -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -pub enum SummaryFamilyType { - /// An ordinary, readable value — the same closed vocabulary as the - /// pre-ASAP `DataType` (`Int64`/`Float64`/`Utf8`/`Bool`/`Timestamp`), - /// passed through unchanged from a pre-ASAP edge. - Plain(DataType), - /// Exact, mergeable accumulator state (`Sum`/`Count`/`Min`/`Max`/`Rate`/ - /// `Increase`). Value consumers require an explicit finalization boundary. - ExactAggregate(ExactKind, ExactParams), - /// Approximate sketch state (KLL/CMS/HLL/…), read out via a - /// `SummaryEstimate`. A [`SketchKind`] already carries the concrete - /// algorithm, params, and grouping layout committed to, not just its - /// category — a bound node needs to know it's specifically independent - /// KLL or shared Hydra-backed CMS, not merely "some sketch". - Sketch(SketchKind, GroupingStrategy), - /// Sampling-based summary state (a retained row subset). - Sample(SamplingKind, SamplingParams), - /// Wavelet-transform summary state (a coefficient vector). - Wavelet(WaveletKind, WaveletParams), - /// Fitted statistical/parametric-model summary state. - StatModel(StatModelKind, StatModelParams), -} - -// ── Post-ASAP schema ───────────────────────────────────────────────────────── - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -pub struct SummaryField { - pub name: String, - pub dtype: SummaryFamilyType, - pub nullable: bool, -} - -/// Schema carried on every edge of the post-ASAP DAG. Extends the pre-ASAP -/// `Schema` with the ability to express summary-state columns. The two are -/// separate types so a pre-ASAP node structurally cannot carry a -/// summary-state-typed column — any attempt to do so is a compile-time type -/// error. -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -pub struct SummarySchema { - pub fields: Vec, - /// Index into `fields` for the time axis, if any (same semantics as the - /// pre-ASAP `Schema::time_index`). - pub time_index: Option, -} diff --git a/crates/types/src/post_asap/sketch.rs b/crates/types/src/post_asap/sketch.rs index 391d12f0f..b5942a5b9 100644 --- a/crates/types/src/post_asap/sketch.rs +++ b/crates/types/src/post_asap/sketch.rs @@ -6,7 +6,7 @@ use crate::pre_asap::ColumnRef; /// An exact, mergeable accumulator family — zero approximation error. The /// partial state built for one of these *is* the answer; no -/// `SummaryEstimate` readout is needed to get a value out of it. +/// `SummaryEstimate` evaluation is needed to get a value out of it. #[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] pub enum ExactKind { /// Exact sum accumulator (mergeable by addition). @@ -48,7 +48,7 @@ pub enum ExactParams { /// [`SketchKind::new`] is where that classification is made. #[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] pub enum SketchAlgorithm { - /// Universal frequency-vector summary with shared statistic readouts. + /// Universal frequency-vector summary with shared statistic evaluations. UnivMon, /// KLL quantile sketch (mergeable, ε-accurate rank queries). Kll, @@ -134,7 +134,7 @@ pub enum SketchCategory { /// quantile-style, cardinality-style, frequency-style, or heavy-hitter/ /// top-k-style estimation — together with the concrete [`SketchAlgorithm`] /// and [`SketchParams`] realizing it. Sits between -/// [`SummaryFamilyType::Sketch`](super::schema::SummaryFamilyType::Sketch) +/// [`FieldDataType::Sketch`](super::schema::FieldDataType::Sketch) /// (the `Sketch` family as a whole, sibling to `Sample`/`Wavelet`/ /// `StatModel`) and the bare algorithm: `Kll` vs. `DDSketch` is a choice /// *within* `Quantile`, not a choice *of* `SketchKind` — every `Quantile` @@ -345,7 +345,7 @@ pub enum StatModelParams { /// really the universal-sketch composition (L layers of Count-Sketch plus a /// heavy-hitter heap, Theorems 1+2 combined) estimating entropy/L1-norm/ /// L2-norm/cardinality/frequency-moments as one instance. Standalone UnivMon -/// and its frequency readouts are represented here, but sharing a Hydra +/// and its frequency evaluations are represented here, but sharing a Hydra /// grid across populations still needs its own collision/error contract; /// standalone support does not establish that contract. #[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] @@ -495,13 +495,13 @@ pub fn default_hydra_params( /// How a grouped aggregate's summary state is physically instantiated /// across its `by` subpopulations — orthogonal to *which* /// `SketchKind`/`SamplingKind`/`WaveletKind`/`StatModelKind` answers the -/// intent (that choice lives alongside it on `SummaryFamilyType`). Lives here, +/// intent (that choice lives alongside it on `FieldDataType`). Lives here, /// alongside `SketchKind`/`SketchParams` etc., rather than on any of those /// enums themselves, for exactly the reason explained in this section's /// module docs above. /// -/// Carried both on `SummaryExpr::SummaryAgg` (where planning consults it) -/// and on sketch-valued `SummaryFamilyType` edges (where it prevents +/// Carried both on `ASAPOp::SummaryAgg` (where planning consults it) +/// and on sketch-valued `FieldDataType` edges (where it prevents /// incompatible shared and independent physical states from type-checking /// as merge-compatible). #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] @@ -590,9 +590,10 @@ pub enum SummaryInputExpr { EntityIdentity(EntityIdentity), } -/// What to extract from a built summary. Carried by `SummaryEstimate`. +/// The statistic computed from summary state by `SummaryEstimate`, for example +/// `Quantile { q: 0.99 }`. This is a result operation, not a workload query. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub enum SketchQuery { +pub enum SketchStatistic { /// sqrt(sum_v frequency(v)^2), not the norm of numeric input values. FrequencyL2, /// Shannon entropy of the value-frequency distribution, in bits. @@ -605,8 +606,8 @@ pub enum SketchQuery { /// `value: Some(v)` is a per-item point lookup (e.g. /// `count(cms_metric{item="checkout"})` — `key` is `item`, `value` is /// `"checkout"`). `value` is carried here rather than resolved by the - /// `SummaryExecutor` from a `Filter` predicate because `readout`'s - /// trait signature has no tree access — see `CostModel::readout_extension`. + /// `SummaryExecutor` from a `Filter` predicate because `evaluation`'s + /// trait signature has no graph access — see `CostModel::evaluation_extension`. PointCount { key: ColumnRef, value: Option, @@ -622,7 +623,7 @@ mod tests { use super::*; #[test] - fn keyed_summary_input_and_topk_readout_round_trip() { + fn keyed_summary_input_and_topk_evaluation_round_trip() { for input in [ SummaryUpdate { item: Some(SummaryInputExpr::EntityIdentity( @@ -646,9 +647,12 @@ mod tests { serde_json::from_str::(&input_json).unwrap(), input ); - let query = SketchQuery::TopK { k: 10 }; + let query = SketchStatistic::TopK { k: 10 }; let json = serde_json::to_string(&query).unwrap(); - assert_eq!(serde_json::from_str::(&json).unwrap(), query); + assert_eq!( + serde_json::from_str::(&json).unwrap(), + query + ); } } diff --git a/crates/types/src/post_asap/summary_maintenance.rs b/crates/types/src/post_asap/summary_maintenance.rs index d1e50d7e5..4e7eaf205 100644 --- a/crates/types/src/post_asap/summary_maintenance.rs +++ b/crates/types/src/post_asap/summary_maintenance.rs @@ -1,6 +1,6 @@ //! Planner-level construction mode for a materialized summary. //! -//! A [`super::SummaryNode`] is a logical summary expression and deliberately +//! An ASAP [`crate::ir::OperatorNode`] is a logical summary expression and deliberately //! does not carry this choice: the same candidate may be built directly for //! one workload or maintained incrementally for another. Planner search //! attaches the selected mode to its lifecycle guarantee; downstream physical diff --git a/crates/types/src/post_asap/summary_window.rs b/crates/types/src/post_asap/summary_window.rs index 0e344c1ed..6d45649b4 100644 --- a/crates/types/src/post_asap/summary_window.rs +++ b/crates/types/src/post_asap/summary_window.rs @@ -57,7 +57,7 @@ pub enum PaneCoverageError { }, } -/// Validate that a pane-only readout covers a query exactly. A mismatched +/// Validate that a pane-only evaluation covers a query exactly. A mismatched /// phase is sound only when the physical plan explicitly supplies an exact /// residual for the partial edge panes. pub fn validate_pane_coverage( @@ -147,7 +147,7 @@ mod tests { } #[test] - fn pane_only_readout_rejects_source_and_query_phase_mismatch() { + fn pane_only_evaluation_rejects_source_and_query_phase_mismatch() { let layout = PaneLayout { pane_width_ms: 60_000, pane_origin_ms: Some(26_000), diff --git a/crates/types/src/pre_asap/agg_intent.rs b/crates/types/src/pre_asap/agg_intent.rs index 2203f4ec7..d3885f279 100644 --- a/crates/types/src/pre_asap/agg_intent.rs +++ b/crates/types/src/pre_asap/agg_intent.rs @@ -10,28 +10,28 @@ //! heavy-hitter sketch when approximate — is a post-ASAP cost-aware decision, //! not encoded here. The semantic distinction that *is* made at lowering is //! intent vs operator: a heavy-hitter aggregate becomes `TopK`, whereas a -//! generic `ORDER BY value LIMIT k` stays as the `QueryExpr::Sort + Limit` +//! generic `ORDER BY value LIMIT k` stays as the `NonASAPOp::Sort + Limit` //! operator pair. use serde::{Deserialize, Serialize}; -use crate::pre_asap::query_expr::DataModel; -use crate::pre_asap::schema::{Column, ColumnId, DataType}; +use crate::ir::operator_properties::DataModel; +use crate::pre_asap::schema::{ColumnId, DataType, Field, FieldDataType}; use crate::types::AccuracyTarget; /// "What to compute" — the vocabulary the planner pivots on. /// -/// Grouping for `TopK` rides on the enclosing `QueryExpr::Aggregate.by` +/// Grouping for `TopK` rides on the enclosing `NonASAPOp::Aggregate`'s `reduction` /// (positional `ColumnId`s), like every other aggregate; the intent itself /// carries only `k` + the accuracy target. /// /// The single-column reducers (`Sum` / `Min` / `Max` / `Avg` / `StdDev` / /// `Variance` / `Quantile`) carry `col: Option` — the input /// column they reduce, generic over the column-reference state the same way -/// [`QueryExpr`](super::query_expr::QueryExpr) is: positional `ColumnId` once -/// bound (the default, and every existing use of the bare `AggIntent` name), -/// or an unresolved name-based `ColumnRef` for a front end constructing this -/// intent directly, before the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has run. +/// the rest of the vocabulary is: positional `ColumnId` once bound (the +/// default, and every existing use of the bare `AggIntent` name), or an +/// unresolved name-based `ColumnRef` for a front end constructing this +/// intent directly, before name resolution (`asap_frontend_common`) has run. /// `None` is the PromQL convention "the time-series sample value"; SQL /// `SUM(bytes), AVG(latency)` sets distinct `Some(_)`s so a multi-aggregate /// node binds each reducer to the right column, and `plan::bind` knows which @@ -128,7 +128,7 @@ pub enum AggIntent { // ── Time-series streaming derivatives ──────────────────────────────── // Counter-reset adjustment; not equivalent to Sum/Count over a window. - // The temporal range lives on the enclosing `QueryExpr::TimeRange` node, + // The temporal range lives on the enclosing `NonASAPOp::TimeRange` node, // not in the intent — this keeps the intent vocabulary range-agnostic. Rate, /// PromQL `irate(v[w])` — reset-aware rate from the final two samples. @@ -234,7 +234,7 @@ pub enum AggIntent { /// A time / calendar accessor (issue #46) — `timestamp`, `minute`, `hour`, /// `day_of_week`, … over each sample's timestamp (or, for the no-arg forms, /// over the evaluation time). Label-preserving per-series value transform. - /// (`time()` is the evaluation time itself — a `QueryExpr::EvalTimestamp` leaf, + /// (`time()` is the evaluation time itself — a `ScalarExpr::EvalTimestamp` leaf, /// not this.) TimeFn(TimeFunc), @@ -378,9 +378,8 @@ pub enum MathFunc { } // `requires` / `is_per_series` / `output_column` never read `col`'s value — -// only its presence via a `{ .. }` pattern — so, unlike -// `QueryExpr::output_schema` (which genuinely cannot compile for an -// unresolved tree — see its own doc), nothing stops these from being generic +// only its presence via a `{ .. }` pattern — so, unlike schema derivation +// (which needs bound positions), nothing stops these from being generic // over every `C`. And a front end constructing `AggIntent` // directly (issue #179) does need `is_per_series` pre-binding — it decides // the `PerEntity`/`Reduce` reduction shape right at construction time (see @@ -529,10 +528,10 @@ impl AggIntent { impl AggIntent { /// Output column name + type produced by this intent over `input`. - /// Used by `QueryExpr::Aggregate`'s schema-derivation rule. The PromQL + /// Used by `NonASAPOp::Aggregate`'s schema-derivation rule. The PromQL /// convention names the column after the intent kind so consumers can /// locate it without an alias lookup. - pub fn output_column(&self, input: &Column) -> Column { + pub fn output_column(&self, input: &Field) -> Field { match self { AggIntent::Count { .. } => col("count", DataType::Int64, false), AggIntent::Sum { .. } => col("sum", input.dtype.clone(), false), @@ -613,8 +612,8 @@ impl AggIntent { } } -fn col(name: &str, dtype: DataType, nullable: bool) -> Column { - Column::new(name, dtype, nullable) +fn col(name: &str, dtype: impl Into, nullable: bool) -> Field { + Field::new(name, dtype.into(), nullable) } /// `0.99` → `"0_99"`, `0.5` → `"0_5"`. Used by `Quantile` output naming so @@ -739,10 +738,10 @@ pub fn default_quantile(q: f64) -> AggIntent { #[cfg(test)] mod tests { use super::*; - use crate::pre_asap::schema::{Column, DataType}; + use crate::pre_asap::schema::{DataType, Field}; - fn c(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) + fn c(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } // Correlation exposes both dependencies but cannot merge final scalar results. @@ -813,7 +812,7 @@ mod tests { AggIntent::::Sum { col: None } .output_column(&c("c", DataType::Int64)) .dtype, - DataType::Int64 + FieldDataType::Plain(DataType::Int64) )); } @@ -970,7 +969,7 @@ mod arg_selector_contract_tests { use crate::pre_asap::{ColumnRef, Schema}; #[test] fn arg_selector_rejects_missing_or_unresolved_arguments() { - let schema = Schema::new(vec![Column::new("value", DataType::Float64, false)]); + let schema = Schema::new(vec![Field::plain("value", DataType::Float64, false)]); for payload in [ serde_json::json!({"arg_col": ColumnRef::Named("value".into())}), serde_json::json!({"arg_col": ColumnRef::Named("value".into()), "val_col": ColumnRef::Named("missing".into())}), diff --git a/crates/types/src/pre_asap/canonicalize.rs b/crates/types/src/pre_asap/canonicalize.rs deleted file mode 100644 index 3a66c5c4d..000000000 --- a/crates/types/src/pre_asap/canonicalize.rs +++ /dev/null @@ -1,782 +0,0 @@ -//! Shared post-lowering canonicalization of the resolved [`QueryExpr`]. -//! -//! Both language front ends funnel through [`resolve_root`](super::resolve::resolve_root), -//! which runs this pass over the resolved tree. Its job is to erase -//! *structural* differences between semantically identical queries so a -//! post-ASAP binding rule matching on the intent algebra sees one canonical -//! spelling regardless of source language (issue #34). -//! -//! ## Heavy-hitter promotion -//! -//! An additive-ranked "order by the aggregate, take the top k" is a -//! heavy-hitter represented by [`AggIntent::TopK`]. Front ends may -//! emit it as an ordinary `Limit { Sort { … Aggregate } }`; this pass promotes -//! that shape to the canonical -//! -//! ```text -//! Aggregate { reduction: Reduce(), measures: [TopK{k}], -//! child: Aggregate { measures: [Count | Sum], … } } -//! ``` -//! -//! Count supplies unit weights and Sum supplies value weights. Because the -//! match is positional, aliases do not affect it. Other ranked expressions -//! retain Sort + Limit. - -use std::rc::Rc; - -use super::agg_intent::{topk, AggIntent}; -use super::expr_ir::{CompareOpKind, ScalarValue}; -use super::query_expr::{Predicate, QueryExpr, Reduction, SortKey, WindowFuncKind}; -use crate::types::AccuracyTarget; - -/// Rewrite `expr` into its canonical form (bottom-up). Idempotent: a tree that -/// is already canonical is returned unchanged. -pub fn canonicalize(mut expr: QueryExpr) -> QueryExpr { - canon(&mut expr); - expr -} - -fn canon(expr: &mut QueryExpr) { - // A `Concat` asserting a caller-proven `discriminator_unique_key` (issue - // #228) had that key's `ColumnId`s resolved, in `resolve.rs`, against - // exactly the first branch's output schema *as it stood before this - // pass ran*. `try_promote_additive_top_ranking`/`try_rewrite_rownumber_topk` - // below can restructure that branch (anywhere within it — not only at - // its own top level, since the same recursive walk can rewrite a node - // nested under a pass-through wrapper too) into a shape with a - // different output schema, which would leave those `ColumnId`s - // pointing at the wrong column, or out of bounds, of the - // post-canonicalize schema. Snapshot the schema the discriminator key - // was actually resolved against, right here, before recursing into the - // children — this is the exact tree state `resolve.rs` saw. - let discriminator_branch_schema_before = match expr { - QueryExpr::Concat { - children, - discriminator_unique_key: Some(_), - } => children.first().and_then(|c| c.output_schema().ok()), - _ => None, - }; - - // Bottom-up: canonicalize every child before matching at this node, so an - // inner heavy-hitter is promoted before an enclosing rewrite inspects it. - for child in children_mut(expr) { - canon(child); - } - - // If the first branch's output schema moved out from under the asserted - // key, the key can no longer be trusted — drop it (never re-derive it by - // guessing at name/position: the two rewrites above don't preserve - // column identity in a way that's safe to infer). A wrong `unique_keys` - // claim is a wrong query answer, not a missed optimization — see - // `ConcatDiscriminatorKey`'s soundness doc — so this errs conservatively: - // any difference at all (not just a column-count/type change) drops the - // key, including the schema becoming undecidable in either direction. - if let QueryExpr::Concat { - children, - discriminator_unique_key: key @ Some(_), - } = expr - { - let discriminator_branch_schema_after = - children.first().and_then(|c| c.output_schema().ok()); - if discriminator_branch_schema_before != discriminator_branch_schema_after { - *key = None; - } - } - - // Local rewrites chain: a `ROW_NUMBER()`-partitioned top-k rewrites to a - // `Limit{Sort}`, which the heavy-hitter rule may then promote to an - // `Aggregate([TopK])`. Each rule strictly simplifies the node, so applying - // them to a fixpoint terminates. - while let Some(rewritten) = - try_rewrite_rownumber_topk(expr).or_else(|| try_promote_additive_top_ranking(expr)) - { - *expr = rewritten; - } -} - -/// A `&mut QueryExpr` out of a child `Rc` — clone-on-write via -/// [`Rc::make_mut`]: free (no clone) while `r` is uniquely owned, which is -/// the overwhelmingly common case (a tree `canonicalize` was just handed by -/// value); falls back to cloning just *this* node (its own fields — the -/// grandchildren stay shared `Rc`s, not deep-copied) only when some other -/// owner still holds the same `Rc`, e.g. a caller that kept its own clone -/// around (`once.clone()` in `is_idempotent` below — `QueryExpr::clone()` is -/// now a cheap `Rc`-bump, not a deep copy, so that clone shares structure -/// with `once` until a rewrite here needs to touch it). `Rc::get_mut` would -/// panic on exactly that case; `make_mut` degrades to a shallow copy instead -/// of requiring sole ownership as a precondition. Once a workload-level CSE -/// pass runs (issue #212, #222) and canonicalize sees an already-shared -/// subtree from a *different* query, this is also the mechanism that keeps -/// canonicalizing one query from silently corrupting another's view of the -/// same shared node. -fn rc_mut(r: &mut Rc) -> &mut QueryExpr { - Rc::make_mut(r) -} - -/// Mutable references to the direct **operator** `QueryExpr` children of a -/// node — `canon`'s own top-down/bottom-up walk only ever visits the -/// relational skeleton, never descending into a scalar position (`Filter.pred`, -/// `ProjectItem.expr`, …): none of the three rewrite rules rewrite anything -/// inside a scalar subtree, so there's nothing to gain by recursing into one, -/// and every scalar variant (issue #205) hits the catch-all below. -fn children_mut(expr: &mut QueryExpr) -> Vec<&mut QueryExpr> { - use QueryExpr::*; - match expr { - // `PromqlScalarBridge`'s child is a scalar-sub-language node (issue - // #220), not the relational skeleton — same "no children to recurse - // into" treatment as the scalar variants below. - Scan { .. } | EvalTimestamp | CurrentTimestamp | PromqlScalarBridge(_) => vec![], - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => vec![rc_mut(c)], - PromqlRelabel { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | PromqlSeriesSample { child, .. } - | PromqlInfoEnrich { child, .. } - | Sort { child, .. } - | Limit { child, .. } => vec![rc_mut(child)], - Concat { children, .. } => children.iter_mut().collect(), - Join { left, right, .. } | SetOp { left, right, .. } => { - vec![rc_mut(left), rc_mut(right)] - } - BinaryOp { lhs, rhs, .. } => vec![rc_mut(lhs), rc_mut(rhs)], - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => vec![], - } -} - -/// Recognise an additive-ranked -/// `Limit { Sort { [Project] Aggregate([Count | Sum]) } }` and rewrite it to -/// the canonical heavy-hitter `Aggregate([TopK])` over the explicit inner -/// aggregate. Returns `None` when the shape does not match. -fn try_promote_additive_top_ranking(expr: &QueryExpr) -> Option { - // Limit k, no offset (an OFFSET means "not the top k"). - let QueryExpr::Limit { - n: k, - offset: 0, - child, - } = expr - else { - return None; - }; - // A single ordering key on a column. - let QueryExpr::Sort { - keys, - partition_by, - child: sort_child, - } = child.as_ref() - else { - return None; - }; - let [SortKey { - expr: QueryExpr::Column(sort_col), - ascending, - .. - }] = keys.as_slice() - else { - return None; - }; - - // The ordered relation is an `Aggregate`, optionally behind a passthrough - // projection (a bare-column SELECT list). Map the sort key through the - // projection to the aggregate's own output column. - let (agg_expr, ranked_col) = match sort_child.as_ref() { - QueryExpr::Project { cols, child, .. } => { - let QueryExpr::Column(underlying) = &cols.get(*sort_col)?.expr else { - return None; - }; - (child.as_ref(), *underlying) - } - other => (other, *sort_col), - }; - - // Exactly one aggregate, ranked by *its* output column — the measure sits at - // index `by.len()` (after the group keys). A `PerEntity` reduction has no - // `by` to rank a measure against — this shape can't be heavy-hitter - // promoted, so it's a non-match rather than an error. - let QueryExpr::Aggregate { - reduction, - measures, - filters, - child: aggregate_child, - .. - } = agg_expr - else { - return None; - }; - let Reduction::Reduce(by) = reduction else { - return None; - }; - let [ranked_agg] = measures.as_slice() else { - return None; - }; - // A heavy-hitter sketch ranks the raw update stream; a filtered measure - // only counts part of it, and no binding rule applies the filter (#466). - if filters.iter().any(Option::is_some) { - return None; - } - if ranked_col != by.len() { - return None; - } - // The heavy-hitter decision — descending, over a measure with a realised - // heavy-hitter sketch — is the shared rule both front ends' promotions - // consult (issue #38). So an ascending additive-ranked limit - // (`ORDER BY COUNT(*) ASC LIMIT k` = bottom-k) stays generic, exactly as - // PromQL `bottomk(k, count_over_time(…))` does. - if !topk::Ranking::from_aggregate(ranked_agg).is_supported(!ascending) { - return None; - } - // A direct Sum is a stream of additive observation weights. A Sum over a - // derived child such as Rate/Increase is different: a heavy-hitter sketch - // may propose candidate membership, but PromQL still requires exact - // reset-aware/extrapolated values to rerank those candidates. The current - // post-ASAP IR has no candidate-sidecar + exact-rerank node, so keep that - // shape as Sort + Limit instead of treating a sketch estimate as final. - if matches!(ranked_agg, AggIntent::Sum { .. }) - && matches!(aggregate_child.as_ref(), QueryExpr::Aggregate { .. }) - { - return None; - } - // Count ranks unit updates; a direct Sum ranks weighted updates. - let accuracy = match ranked_agg { - AggIntent::Count { accuracy } => accuracy.clone(), - AggIntent::Sum { .. } => AccuracyTarget::Exact, - _ => unreachable!("additive ranking gate admitted a non-additive measure"), - }; - - // Outer heavy-hitter `TopK`, grouped by the ranking's partition (empty for a - // global `ORDER BY … LIMIT k`; the `by` labels for a partitioned `topk by`), - // over the unchanged inner additive aggregate. - Some(QueryExpr::Aggregate { - reduction: Reduction::by(partition_by.to_vec()), - measures: vec![AggIntent::TopK { k: *k, accuracy }], - output_names: Vec::new(), - filters: Vec::new(), - having: None, - child: Rc::new(agg_expr.clone()), - }) -} - -/// Recognise the SQL partitioned-top-k idiom — `WHERE rn <= k` over a -/// `ROW_NUMBER() OVER (PARTITION BY p ORDER BY o)` — and rewrite it to the -/// generic partitioned top-k `Limit{k} { Sort{ o, partition_by: p } }` (issue -/// #24). The count-ranked case is then promoted to a heavy-hitter `TopK` by -/// [`try_promote_additive_top_ranking`], so a SQL `ROW_NUMBER` top-k and the PromQL -/// `topk by (…)` it mirrors converge on the same canonical shape. -fn try_rewrite_rownumber_topk(expr: &QueryExpr) -> Option { - // Filter { pred: `Column(rn) <= k` }. - let QueryExpr::Filter { pred, child } = expr else { - return None; - }; - let Predicate(pred_expr) = pred; - let QueryExpr::Compare { left, op, right } = pred_expr.as_ref() else { - return None; - }; - // `rn <= k` (top-k). `rn < k` would be off-by-one; require `<=`. - if *op != CompareOpKind::Le { - return None; - } - let (QueryExpr::Column(rn_col), QueryExpr::Literal(ScalarValue::Int64(k))) = - (left.as_ref(), right.as_ref()) - else { - return None; - }; - if *k < 0 { - return None; - } - - // Optionally strip a passthrough projection (the derived table's SELECT that - // re-exposes the aggregate columns + rn), mapping the rn column through it. - let (wf_expr, rn_in_wf) = match child.as_ref() { - QueryExpr::Project { cols, child, .. } => { - let QueryExpr::Column(underlying) = &cols.get(*rn_col)?.expr else { - return None; - }; - (child.as_ref(), *underlying) - } - other => (other, *rn_col), - }; - - // The filtered column must be a `ROW_NUMBER()` window output — the single - // column the SQLWindowFunc appends after its input, i.e. the last one. - let QueryExpr::SQLWindowFunc { - func: WindowFuncKind::RowNumber, - partition_by, - order_by, - child: inner, - .. - } = wf_expr - else { - return None; - }; - if order_by.is_empty() { - return None; - } - let inner_cols = inner.output_schema().ok()?.columns.len(); - if rn_in_wf != inner_cols { - return None; // the predicate ranks some other column, not the row number - } - - // Generic partitioned top-k. The window's ORDER BY keys are relative to its - // input (`inner`), so they transfer directly to a `Sort` over `inner`. - Some(QueryExpr::Limit { - n: *k as usize, - offset: 0, - child: Rc::new(QueryExpr::Sort { - keys: order_by.clone(), - partition_by: partition_by.clone(), - child: Rc::new(inner.as_ref().clone()), - }), - }) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::pre_asap::query_expr::{ - GroupKeys, ProjectItem, Source, WindowFrame, WindowFrameBound, WindowFrameOffset, - WindowFrameUnits, - }; - use crate::pre_asap::schema::{Column, DataType, Schema}; - use crate::types::AccuracyTarget; - - fn scan() -> QueryExpr { - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("service", DataType::Utf8, false), - Column::new("value", DataType::Float64, false), - ], - 0, - vec![], - ), - } - } - - /// `Aggregate{ by: [1], [Count] }` over the scan — output cols `[service, count]`. - fn count_by_service() -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::by(vec![1]), - measures: vec![AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan()), - } - } - - fn desc(col: usize) -> Vec { - vec![SortKey { - expr: QueryExpr::Column(col), - ascending: false, - nulls_first: false, - }] - } - - fn limit(n: usize, offset: usize, child: QueryExpr) -> QueryExpr { - QueryExpr::Limit { - n, - offset, - child: Rc::new(child), - } - } - - fn sort(keys: Vec, child: QueryExpr) -> QueryExpr { - QueryExpr::Sort { - keys, - partition_by: GroupKeys::by(vec![]), - child: Rc::new(child), - } - } - - fn is_topk_over_count(qe: &QueryExpr) -> bool { - matches!(qe, - QueryExpr::Aggregate { measures, child, .. } - if matches!(measures.as_slice(), [AggIntent::TopK { k: 5, .. }]) - && matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } - if matches!(measures.as_slice(), [AggIntent::Count { .. }]))) - } - - #[test] - fn promotes_count_ranked_limit_sort() { - // Limit 5 { Sort DESC by count-col (1) { Aggregate[Count] by [1] } }. - let q = limit(5, 0, sort(desc(1), count_by_service())); - assert!(is_topk_over_count(&canonicalize(q))); - } - - // A heavy-hitter sketch ranks every row; a count that only counts some - // rows (#466) is not that, so the generic Sort + Limit stays. - #[test] - fn does_not_promote_a_filtered_count_ranking() { - let mut filtered = count_by_service(); - let QueryExpr::Aggregate { filters, .. } = &mut filtered else { - unreachable!() - }; - *filters = vec![Some(Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(2)), - op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(1.0))), - })))]; - let q = limit(5, 0, sort(desc(1), filtered)); - let canonical = canonicalize(q.clone()); - assert!(!is_topk_over_count(&canonical)); - assert_eq!(canonical, q); - } - - #[test] - fn promotes_through_a_passthrough_projection() { - // …with a `SELECT service, count` projection between the Sort and the Agg. - let proj = QueryExpr::Project { - cols: vec![ - ProjectItem { - alias: None, - expr: QueryExpr::Column(0), - }, - ProjectItem { - alias: Some("c".into()), - expr: QueryExpr::Column(1), - }, - ], - qualifier: None, - child: Rc::new(count_by_service()), - }; - let q = limit(5, 0, sort(desc(1), proj)); - assert!(is_topk_over_count(&canonicalize(q))); - } - - #[test] - fn is_idempotent() { - let q = limit(5, 0, sort(desc(1), count_by_service())); - let once = canonicalize(q); - let twice = canonicalize(once.clone()); - assert_eq!(once, twice, "canonicalize must be idempotent"); - } - - // ── Concat's discriminator_unique_key vs. canonicalize (issue #228 review) ── - // - // `resolve.rs` resolves `discriminator_unique_key`'s `ColumnId`s against - // the first branch's *pre-canonicalize* output schema. If canonicalize - // then restructures that branch (heavy-hitter promotion, the - // `ROW_NUMBER()` top-k rewrite), those `ColumnId`s can end up pointing at - // the wrong column — or out of bounds — of the new schema. The two tests - // below pin the fix: the key is dropped whenever the branch's schema - // actually changed, and survives untouched otherwise. Never guessed at. - - #[test] - fn concat_discriminator_key_survives_canonicalize_when_first_branch_is_unaffected() { - // A plain `Aggregate` first branch matches neither rewrite trigger, - // so its schema is identical before and after canonicalize. - let q = QueryExpr::concat_with_discriminator( - vec![count_by_service(), count_by_service()], - /* discriminator */ 0, - /* inner_key */ vec![1], - ); - let QueryExpr::Concat { - discriminator_unique_key, - .. - } = canonicalize(q) - else { - panic!("expected Concat"); - }; - assert!( - discriminator_unique_key.is_some(), - "an untouched first branch's discriminator key must survive canonicalize" - ); - } - - #[test] - fn concat_discriminator_key_is_dropped_when_first_branch_gets_rewritten() { - // The first branch is exactly the heavy-hitter promotion trigger — - // `Limit{Sort{Aggregate([Count])}}`, with an empty (global) - // `partition_by` — so canonicalize rewrites it in place to - // `Aggregate{TopK}`, whose own output is a single column, not the - // original two (`[service, count]`). A discriminator key resolved - // against the original 2-column shape (`discriminator` = `service` - // at index 0, `inner_key` = `count` at index 1) must not silently - // survive pointing at the new 1-column schema. - let promotable_branch = limit(5, 0, sort(desc(1), count_by_service())); - let q = QueryExpr::concat_with_discriminator( - vec![promotable_branch, count_by_service()], - /* discriminator */ 0, - /* inner_key */ vec![1], - ); - let QueryExpr::Concat { - children, - discriminator_unique_key, - } = canonicalize(q) - else { - panic!("expected Concat"); - }; - assert!( - is_topk_over_count(&children[0]), - "the first branch is still promoted normally" - ); - assert!( - discriminator_unique_key.is_none(), - "a stale discriminator key must be dropped, never silently kept wrong" - ); - } - - #[test] - fn does_not_promote_ascending_sort() { - // Ascending = bottom-k: the Top-K operator's ranking rule - // rejects it (needs descending), so it stays a generic Sort+Limit — the - // same call PromQL `bottomk` makes (issue #38). - let asc = vec![SortKey { - expr: QueryExpr::Column(1), - ascending: true, - nulls_first: false, - }]; - let q = limit(5, 0, sort(asc, count_by_service())); - assert!(!is_topk_over_count(&canonicalize(q))); - } - - #[test] - fn does_not_promote_with_offset() { - let q = limit(5, 2, sort(desc(1), count_by_service())); - assert!(!is_topk_over_count(&canonicalize(q))); - } - - #[test] - fn does_not_promote_ranking_by_a_group_key() { - // DESC by col 0 (the `service` group key), not the count → not a - // frequency heavy-hitter. - let q = limit(5, 0, sort(desc(0), count_by_service())); - assert!(!is_topk_over_count(&canonicalize(q))); - } - - #[test] - fn promotes_sum_ranked_limit_sort_as_weighted_heavy_hitter() { - let sum = QueryExpr::Aggregate { - reduction: Reduction::by(vec![1]), - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan()), - }; - let q = limit(5, 0, sort(desc(1), sum)); - let out = canonicalize(q); - let QueryExpr::Aggregate { - measures, child, .. - } = out - else { - panic!("expected weighted TopK aggregate"); - }; - assert!(matches!( - measures.as_slice(), - [AggIntent::TopK { k: 5, .. }] - )); - assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } - if matches!(measures.as_slice(), [AggIntent::Sum { .. }])) - ); - } - - #[test] - fn keeps_sum_over_counter_reduction_as_exact_value_ranking() { - for counter in [AggIntent::Rate, AggIntent::Increase] { - let derived = QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![counter], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan()), - }; - let sum = QueryExpr::Aggregate { - reduction: Reduction::by(vec![1]), - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(derived), - }; - let out = canonicalize(limit(5, 0, sort(desc(1), sum))); - assert!(matches!(out, QueryExpr::Limit { child, .. } - if matches!(child.as_ref(), QueryExpr::Sort { child, .. } - if matches!(child.as_ref(), QueryExpr::Aggregate { measures, child, .. } - if matches!(measures.as_slice(), [AggIntent::Sum { .. }]) - && matches!(child.as_ref(), QueryExpr::Aggregate { .. }))))); - } - } - - // ── ROW_NUMBER() partitioned top-k (issue #24) ────────────────────────── - - /// A scan with `[ts, service, region, value]`. - fn scan4() -> QueryExpr { - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("service", DataType::Utf8, false), - Column::new("region", DataType::Utf8, false), - Column::new("value", DataType::Float64, false), - ], - 0, - vec![], - ), - } - } - - /// `Aggregate{ by: [1,2] (service, region), [agg] }` — output `[service, - /// region, ]` (3 cols), so a ROW_NUMBER over it appends `rn` at index 3. - fn grouped(agg: AggIntent) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::by(vec![1, 2]), - measures: vec![agg], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan4()), - } - } - - /// `ROW_NUMBER` ignores its frame clause, so the top-k rewrite doesn't care - /// what's in it; any concrete frame works as fixture data. - fn rownumber_frame() -> WindowFrame { - WindowFrame { - units: WindowFrameUnits::Rows, - start_bound: WindowFrameBound::Preceding(WindowFrameOffset::Scalar(ScalarValue::Null)), - end_bound: WindowFrameBound::Following(WindowFrameOffset::Scalar(ScalarValue::Null)), - } - } - - /// `Filter{ rn(3) <= 5 } { SQLWindowFunc{ RowNumber, PARTITION BY region(2), - /// ORDER BY col(2) DESC } { agg } }`. - fn rownumber_topk(agg: QueryExpr) -> QueryExpr { - let wf = QueryExpr::SQLWindowFunc { - func: WindowFuncKind::RowNumber, - args: vec![], - partition_by: GroupKeys::by(vec![2]), // region - order_by: vec![SortKey { - expr: QueryExpr::Column(2), // the aggregate output column - ascending: false, - nulls_first: true, - }], - frame: Some(rownumber_frame()), - output_name: "rn".into(), - child: Rc::new(agg), - }; - QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), // rn = the appended window column - op: CompareOpKind::Le, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(5))), - })), - child: Rc::new(wf), - } - } - - #[test] - fn rownumber_count_topk_becomes_a_partitioned_heavy_hitter() { - // Count-ranked ROW_NUMBER top-k → outer TopK grouped by the partition - // (region, col 2) over the explicit inner Count. - let q = rownumber_topk(grouped(AggIntent::Count { - accuracy: AccuracyTarget::Exact, - })); - let out = canonicalize(q); - let QueryExpr::Aggregate { - reduction, - measures, - child, - .. - } = &out - else { - panic!("expected outer Aggregate([TopK]), got {out:?}"); - }; - let Reduction::Reduce(by) = reduction else { - panic!("expected a Reduce grouping, got {reduction:?}"); - }; - assert!(matches!( - measures.as_slice(), - [AggIntent::TopK { k: 5, .. }] - )); - assert_eq!(**by, vec![2], "outer TopK partitioned by region"); - assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } - if matches!(measures.as_slice(), [AggIntent::Count { .. }])) - ); - } - - #[test] - fn rownumber_avg_topk_becomes_a_partitioned_sort_limit() { - // Avg-ranked (not a frequency heavy-hitter) → generic partitioned - // top-k: Limit{5}{ Sort{ partition_by: [region] } }. - let q = rownumber_topk(grouped(AggIntent::Avg { col: None })); - let out = canonicalize(q); - let QueryExpr::Limit { n, child, .. } = &out else { - panic!("expected a Limit, got {out:?}"); - }; - assert_eq!(*n, 5); - let QueryExpr::Sort { - partition_by, - child, - .. - } = child.as_ref() - else { - panic!("expected a Sort under the Limit"); - }; - assert_eq!(**partition_by, vec![2], "partitioned by region"); - assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } - if matches!(measures.as_slice(), [AggIntent::Avg { .. }])) - ); - } - - #[test] - fn filter_on_a_non_rownumber_column_is_left_alone() { - // `WHERE service_len <= 5` (col 0, not the rn window column) must not be - // mistaken for a top-k. - let wf = QueryExpr::SQLWindowFunc { - func: WindowFuncKind::RowNumber, - args: vec![], - partition_by: GroupKeys::by(vec![2]), - order_by: vec![SortKey { - expr: QueryExpr::Column(2), - ascending: false, - nulls_first: true, - }], - frame: Some(rownumber_frame()), - output_name: "rn".into(), - child: Rc::new(grouped(AggIntent::Count { - accuracy: AccuracyTarget::Exact, - })), - }; - let q = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), // NOT the rn column (index 3) - op: CompareOpKind::Le, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(5))), - })), - child: Rc::new(wf), - }; - assert!( - matches!(canonicalize(q), QueryExpr::Filter { .. }), - "left as a Filter" - ); - } -} diff --git a/crates/types/src/pre_asap/column_resolution.rs b/crates/types/src/pre_asap/column_resolution.rs index 6f974922e..e2d8120d6 100644 --- a/crates/types/src/pre_asap/column_resolution.rs +++ b/crates/types/src/pre_asap/column_resolution.rs @@ -1,23 +1,14 @@ //! Schema-driven column resolution. //! -//! Front ends (issue #179) emit `ColumnRef` (name-based, optionally -//! table-qualified); the canonical tree uses positional [`ColumnId`] resolved -//! against a per-node [`Schema`]. These helpers bridge the two — the -//! [`SchemaResolver`](super::schema_resolver) builds the schema, and [`resolve_column_refs`] -//! turns name-based refs (group keys, dedup columns) into positional ids, -//! qualifier-aware. - -use std::rc::Rc; +//! Front ends emit `ColumnRef` (name-based, optionally table-qualified); the +//! IR uses positional [`ColumnId`] resolved against a per-node [`Schema`]. +//! These helpers turn name-based refs into positional ids, qualifier-aware. +//! Front-end name resolution (`asap_frontend_common::resolve`) calls them. use thiserror::Error; -use super::agg_intent::AggIntent; use super::expr_ir::ColumnRef; -use super::query_expr::{ - aggregate_output_schema, GroupKeys, QueryExpr, QueryExprError, Reduction, ResolvedQueryExpr, - UnresolvedQueryExpr, -}; -use super::schema::{ColumnId, DataType, Schema}; +use super::schema::{ColumnId, DataType, FieldDataType, Schema}; /// Errors returned by the resolution helpers. #[derive(Debug, Error, PartialEq, Eq)] @@ -40,7 +31,7 @@ pub fn resolve_column_ref(col: &ColumnRef, schema: &Schema) -> Result Result schema .column_id("value") @@ -62,16 +53,19 @@ pub fn resolve_column_ref(col: &ColumnRef, schema: &Schema) -> Result = (0..schema.columns.len()) + let numeric: Vec = (0..schema.fields.len()) .filter(|&i| Some(i) != schema.time_index) .filter(|&i| { - matches!(schema.columns[i].dtype, DataType::Float64 | DataType::Int64) + matches!( + schema.fields[i].dtype, + FieldDataType::Plain(DataType::Float64 | DataType::Int64) + ) }) .collect(); (numeric.len() == 1).then(|| numeric[0]) }) .ok_or_else(|| ResolveError::NoSampleValue { - available: schema.columns.iter().map(|c| c.name.clone()).collect(), + available: schema.fields.iter().map(|c| c.name.clone()).collect(), }), ColumnRef::Wildcard => Err(ResolveError::WildcardNotPositional), } @@ -112,113 +106,17 @@ pub fn resolve_group_keys_promql( .collect() } -/// Resolve a name-based scalar [`UnresolvedQueryExpr`] (one of `QueryExpr`'s scalar -/// variants, issue #205) into a positional [`ResolvedQueryExpr`] by resolving every -/// column reference against `schema`. Structural otherwise. `expr` must be -/// one of the scalar variants — an operator variant here is a construction -/// bug, not a shape this needs to handle silently. -pub fn resolve_expr( - expr: &UnresolvedQueryExpr, - schema: &Schema, -) -> Result { - let rc = |e: &UnresolvedQueryExpr| -> Result, ResolveError> { - Ok(Rc::new(resolve_expr(e, schema)?)) - }; - let each = |es: &[UnresolvedQueryExpr]| -> Result, ResolveError> { - es.iter().map(|e| resolve_expr(e, schema)).collect() - }; - Ok(match expr { - QueryExpr::Column(c) => QueryExpr::Column(resolve_column_ref(c, schema)?), - QueryExpr::Literal(s) => QueryExpr::Literal(s.clone()), - QueryExpr::EvalTimestamp => QueryExpr::EvalTimestamp, - QueryExpr::CurrentTimestamp => QueryExpr::CurrentTimestamp, - QueryExpr::Compare { left, op, right } => QueryExpr::Compare { - left: rc(left)?, - op: op.clone(), - right: rc(right)?, - }, - QueryExpr::BoolAnd(v) => QueryExpr::BoolAnd(each(v)?), - QueryExpr::BoolOr(v) => QueryExpr::BoolOr(each(v)?), - QueryExpr::Not(e) => QueryExpr::Not(rc(e)?), - QueryExpr::IsNull(e) => QueryExpr::IsNull(rc(e)?), - QueryExpr::IsNotNull(e) => QueryExpr::IsNotNull(rc(e)?), - QueryExpr::Cast { expr, to, try_cast } => QueryExpr::Cast { - expr: rc(expr)?, - to: to.clone(), - try_cast: *try_cast, - }, - QueryExpr::InList { - expr, - list, - negated, - } => QueryExpr::InList { - expr: rc(expr)?, - list: each(list)?, - negated: *negated, - }, - QueryExpr::FunctionCall { name, args } => QueryExpr::FunctionCall { - name: name.clone(), - args: each(args)?, - }, - QueryExpr::Arithmetic { op, left, right } => QueryExpr::Arithmetic { - op: op.clone(), - left: rc(left)?, - right: rc(right)?, - }, - QueryExpr::Case { - operand, - branches, - else_expr, - } => QueryExpr::Case { - operand: operand.as_deref().map(rc).transpose()?, - branches: branches - .iter() - .map(|(w, t)| Ok((resolve_expr(w, schema)?, resolve_expr(t, schema)?))) - .collect::, ResolveError>>()?, - else_expr: else_expr.as_deref().map(rc).transpose()?, - }, - other => unreachable!("resolve_expr called on a non-scalar QueryExpr variant: {other:?}"), - }) -} - -/// Output schema produced by an `Aggregate { by, measures }` over `input`. -/// Mirrors `QueryExpr::output_schema_in`'s `Aggregate` arm; out-of-range `by` -/// ids are silently dropped (callers needing the strict check resolve `by` -/// via [`resolve_column_refs`], which surfaces `NotFound`). -pub fn output_schema_for_aggregate( - input: &Schema, - by: &GroupKeys, - measures: &[AggIntent], - output_names: &[String], -) -> Result { - // Delegate to the single canonical derivation so HAVING resolution can never - // drift from `QueryExpr::output_schema_in` (issue #41). HAVING is SQL-only - // and cross-series (SQL has no `without`), but detect the child-independent - // per-entity case anyway (a lone `rate`/`increase`/`*_over_time` intent) so - // the two agree on every shared input — the range-window child marker the - // canonical arm also keys off is not visible here, and never co-occurs with - // HAVING. - let per_entity = - by.is_empty() && !by.is_without() && measures.len() == 1 && measures[0].is_per_series(); - let reduction = if per_entity { - Reduction::PerEntity - } else { - Reduction::Reduce(by.clone()) - }; - aggregate_output_schema(input, &reduction, measures, output_names) -} - #[cfg(test)] mod tests { use super::*; - use crate::pre_asap::schema::Column; + use crate::pre_asap::schema::Field; /// The conventional PromQL leaf shape: `(ts: Timestamp, value: Float64)`. fn ts_value_schema() -> Schema { Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], 0, Vec::new(), @@ -237,8 +135,8 @@ mod tests { // column and two non-ts columns (ambiguous), but exactly one numeric // column — the sample value an outer `topk` ranks by. let s = Schema::new(vec![ - Column::new("job", DataType::Utf8, true), - Column::new("sum", DataType::Float64, false), + Field::plain("job", DataType::Utf8, true), + Field::plain("sum", DataType::Float64, false), ]); assert_eq!(resolve_column_ref(&ColumnRef::SampleValue, &s), Ok(1)); } @@ -250,8 +148,8 @@ mod tests { // bind to it (#70) — resolution fails cleanly instead of picking a label. let s = Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("host", DataType::Utf8, true), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("host", DataType::Utf8, true), ], 0, vec![], @@ -266,8 +164,8 @@ mod tests { fn sample_value_ambiguous_when_two_numeric_columns() { // Two numeric non-ts columns → genuinely ambiguous → NoSampleValue. let s = Schema::new(vec![ - Column::new("a", DataType::Float64, false), - Column::new("b", DataType::Int64, false), + Field::plain("a", DataType::Float64, false), + Field::plain("b", DataType::Int64, false), ]); assert!(matches!( resolve_column_ref(&ColumnRef::SampleValue, &s), @@ -287,9 +185,9 @@ mod tests { // The output of a nested cross-series aggregate: closed `[group, sum]`. // `by (job)` — `job` is provably absent → dropped, not rejected (#53). let s = Schema { - columns: vec![ - Column::new("group", DataType::Utf8, true), - Column::new("sum", DataType::Float64, false), + fields: vec![ + Field::plain("group", DataType::Utf8, true), + Field::plain("sum", DataType::Float64, false), ], time_index: None, unique_keys: vec![], @@ -323,77 +221,4 @@ mod tests { Err(ResolveError::NotFound { .. }) )); } - - #[test] - fn aggregate_strips_time_and_keeps_unique_keys() { - let mut input = ts_value_schema(); - input - .columns - .push(Column::new("host", DataType::Utf8, false)); - let out = output_schema_for_aggregate( - &input, - &GroupKeys::by(vec![2]), - &[AggIntent::Sum { col: None }], - &[], - ) - .expect("valid group-by column"); - assert_eq!(out.columns.len(), 2); // host, sum - assert_eq!(out.columns[0].name, "host"); - assert_eq!(out.columns[1].name, "sum"); - assert!(out.time_index.is_none()); - assert_eq!(out.unique_keys, vec![vec![0]]); - } - - #[test] - fn having_schema_agrees_with_canonical_for_a_per_series_reduction() { - // Issue #41: `output_schema_for_aggregate` (HAVING resolution) and the - // canonical `QueryExpr::output_schema_in` must produce identical schemas - // for the same aggregate. Before the dedup this diverged on a per-series - // reduction — the HAVING mirror lacked the per-series branch and would - // collapse `[ts, value]` to a single `rate` column. - use crate::pre_asap::query_expr::Source; - use std::time::Duration; - - let leaf_schema = Schema::with_time_index( - vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - ], - 0, - vec![], - ); - let scan = QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: leaf_schema.clone(), - }; - // Aggregate{ reduction: PerEntity, [Rate], child: TimeRange{ Scan } } — - // a per-series reduction (label-preserving). - let agg = QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![AggIntent::Rate], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }), - }; - let canonical = agg.output_schema().expect("canonical schema"); - - // The HAVING-resolution derivation gets only the input schema (the - // TimeRange passes the leaf schema through). - let having_side = - output_schema_for_aggregate(&leaf_schema, &GroupKeys::none(), &[AggIntent::Rate], &[]) - .unwrap(); - - assert_eq!( - canonical, having_side, - "the two aggregate-schema derivations must agree (issue #41)" - ); - // Sanity: it really is the label-preserving per-series shape, not `[rate]`. - assert!(having_side.columns.iter().any(|c| c.name == "value")); - assert!(having_side.time_index.is_some()); - } } diff --git a/crates/types/src/pre_asap/cse.rs b/crates/types/src/pre_asap/cse.rs deleted file mode 100644 index 1a2758737..000000000 --- a/crates/types/src/pre_asap/cse.rs +++ /dev/null @@ -1,1111 +0,0 @@ -//! Pre-ASAP structural common-subexpression elimination: bottom-up -//! hash-consing over an already-`resolve_root`'d [`QueryExpr`] tree (issue -//! #212, #222, #223). -//! -//! CSE only runs on an already-bound, already-canonicalized tree — -//! structural matching is meaningless before canonicalization has converged -//! semantically-equivalent queries onto one shape (`docs/develop_docs/pre-asap-ir.md` -//! design principle 3; `median(latency)` and `approx_percentile_cont(latency, -//! 0.5)` already lower to an identical `AggIntent::Quantile` today, per -//! `sql_lowering.rs`'s `median_is_the_same_intent_as_an_explicit_half_percentile` -//! test). [`share_common_subtrees`] is the single entry point, run once per -//! workload batch (or once per query — see "Single-query CSE" below) *after* -//! `resolve_root`, *before* the pre-ASAP → post-ASAP replacement/search pass -//! (`asap_aware_mapping::replacement`). -//! -//! ## Algorithm: classic hash-consing / value-numbering -//! -//! Bottom-up: every child is interned before its parent, so a parent's -//! candidacy for sharing naturally incorporates whether its own children were -//! themselves shared — two parents whose children were independently -//! deduplicated down to the same `Rc`s are structurally identical iff their -//! own fields also match, without re-walking the subtrees. -//! -//! Only the **relational skeleton** participates — the same set of "operator" -//! children [`canonicalize`](super::canonicalize)'s `children_mut` walks -//! (`Filter`/`Project`/`Aggregate`/`Concat`/`Join`/`BinaryOp`/…). A scalar -//! subexpression reachable only through a wrapper position (`Predicate`, -//! `ProjectItem.expr`, `Aggregate.having`, `SQLWindowFunc.args`, …) stays -//! embedded as opaque data on its owning operator node, compared by -//! `QueryExpr`'s derived `PartialEq` along with the rest of that node's -//! fields, rather than separately hash-consed — the same scope -//! `canonicalize.rs` settled on ("none of the rewrite rules touch a scalar -//! subtree, so there's nothing to gain by recursing into one"). Widening this -//! to scalar positions is future work, not attempted here. -//! -//! ## Correctness: hash is a filter, `PartialEq` is the decision -//! -//! This is the one non-negotiable rule. A **false positive** here — two -//! subtrees wrongly judged shareable — is a wrong query answer, not a missed -//! optimization: two different queries would read each other's data. -//! [`structural_hash`] (`DefaultHasher`/SipHash over a canonical -//! serialization, no collision-freedom guarantee) may only narrow the -//! candidate set within one bucket; [`InternTable::intern`]'s `PartialEq` -//! check on that bucket is what actually decides sharing, every time, no -//! exceptions for "the hash probably didn't collide." -//! -//! This also means the pass is safe by construction against the case #212 -//! flagged as a real historical bug (issue #115): `AggIntent::Quantile` -//! carries its input column and its `AccuracyTarget`, both `PartialEq` -//! fields, so `Quantile(x, 0.99, ε=0.01)` and `Quantile(x, 0.99, ε=0.001)` — -//! or `Quantile(x, ..)` vs `Quantile(y, ..)` — are never merged. This is -//! intentionally conservative: it only recognizes *exact* structural -//! matches, not "a stricter-accuracy summary could also answer a looser -//! request." That subsumption question already has a documented, -//! deliberately-unfilled home (`asap_aware_mapping::Matcher`) — -//! CSE here does not attempt it. -//! -//! ## Legality: gated by `Schema::unique_keys` -//! -//! Structural equality alone is necessary but not sufficient. Per -//! [`Schema::unique_keys`](super::schema::Schema::unique_keys)'s own doc: "a -//! producer's output can only be safely shared across consumers when its row -//! identity is provably stable across reads." A candidate node with no -//! provable unique key (`Schema::has_unique_key()` false, or `output_schema` -//! not even defined for that node, e.g. a `Concat`/`SetOp` branch whose union -//! drops `unique_keys`, or an ungrouped/global `Aggregate`, whose empty `by` -//! also reports no unique key today) is **never** hoisted, even when it is -//! structurally identical to something already interned — it is always -//! inserted fresh, matching the rule the (now-deleted) prior CSE attempt -//! already encoded and the doc comment on `Aggregate`'s `child` field -//! ("`unique_keys` feeds CSE's producer-sharing legality check"). -//! -//! ## Single-query CSE falls out for free -//! -//! A repeated sub-expression within *one* query (e.g. the same grouped -//! `Aggregate` referenced twice on two `BinaryOp` branches) is deduplicated -//! by the exact same bottom-up interning — a workload of size one still -//! interns bottom-up within that one tree. No separate mechanism is needed; -//! see the `single_query_shares_its_own_repeated_subtree` test below. -//! -//! ## Landing plan (issue #223) -//! -//! This module is stage 1 of a 4-stage plan. Stage 2 -//! (`asap_aware_mapping::replacement::search_workload_with`, which runs -//! [`share_common_subtrees`] itself before searching) is a real caller, -//! wired at the same time so this never becomes unwired dead code again -//! (the original `asap-plan::cse::dedupe_subtrees` was deleted in #192 for -//! exactly that). Stage 3 — [`dag_export`](crate::dag_export) computing its -//! per-node `hash` by calling this module's [`structural_hash`] directly, -//! instead of a parallel reimplementation — is also done, so -//! `tools/dag-viewer`'s "shared subtree" highlighting now flags exactly the -//! candidate pairs this module's own `InternTable` would bucket together -//! (still only a hash match, not a guarantee of -//! `share_common_subtrees`-actual sharing — see `dag_export`'s module doc). -//! Stage 4 (issue #237) is implemented in -//! `asap_aware_mapping::cost_model::CostModel::cse_share_decision`, called -//! from `asap_aware_mapping::replacement::CandidateLogicalASAPDAGs::cost_sorted` (via that -//! module's own `cse_preference`) — a real, Volcano/Cascades-style cost -//! comparison over what this module detects, not a fixed rule. See -//! `docs/design_docs/cost-model.md`. This module's own -//! unconditional "share whenever legal" behavior is unchanged: detection -//! stays cost-agnostic by construction (this crate cannot depend on -//! `asap-aware-mapping`'s `CostModel`), and the cost-aware decision is -//! applied downstream, after detection, over what this module finds. - -use std::collections::HashMap; -use std::hash::{Hash, Hasher}; -use std::rc::Rc; - -use super::query_expr::QueryExpr; - -/// Bottom-up hash-consing table: structurally-equal, sharing-legal -/// [`QueryExpr`] nodes collapse onto one `Rc`. -/// -/// `buckets` is keyed by [`structural_hash`] — a coarse candidate filter -/// only (see the module-level "Correctness" section). Every entry within one -/// bucket is a full node kept around for the `PartialEq` comparison that -/// actually decides a match; a hash collision between structurally different -/// nodes just means a (harmless) linear scan of a few extra candidates. -struct InternTable { - buckets: HashMap>>, - /// Memoizes [`structural_hash`] per already-hashed `Rc` pointer, shared - /// across every [`intern`](Self::intern) call for the table's whole - /// lifetime — see [`structural_hash`]'s own doc on why this matters: - /// without it, hashing an `N`-node bottom-up pass costs `O(N)` work - /// *per node* (every already-interned descendant gets re-walked), not - /// `O(1)` amortized. - hash_cache: HashCache, -} - -impl InternTable { - fn new() -> Self { - Self { - buckets: HashMap::new(), - hash_cache: HashMap::new(), - } - } - - /// Intern one already-children-rebuilt node: look it up by - /// [`structural_hash`], confirm with `PartialEq`, and — only when - /// sharing is legal (see "Legality" above) — return the existing `Rc` - /// instead of allocating a new one. - fn intern(&mut self, node: QueryExpr) -> Rc { - let hash = structural_hash(&node, &mut self.hash_cache); - // A node with no provable unique key is never *returned* as a match - // for something else — it may still go on to occupy a fresh slot in - // the bucket (harmless; it just never gets found by a later - // `PartialEq` scan that also requires `reusable`). - let reusable = node - .output_schema() - .is_ok_and(|schema| schema.has_unique_key()); - let bucket = self.buckets.entry(hash).or_default(); - if reusable { - if let Some(existing) = bucket.iter().find(|candidate| candidate.as_ref() == &node) { - return Rc::clone(existing); - } - } - let rc = Rc::new(node); - bucket.push(Rc::clone(&rc)); - rc - } -} - -/// [`structural_hash`]'s memoization cache: maps an already-hashed node's -/// `Rc` pointer to its computed hash. Not tied to any one `QueryExpr` — a -/// fresh, empty cache is correct to start with anywhere; what matters is -/// letting it *persist* across every node in one bottom-up pass (as -/// [`InternTable`] does via its own `hash_cache` field), rather than -/// starting a new one per call. -/// -/// `pub` (not `pub(crate)`) so `asap_aware_mapping`'s workload-search MEMO -/// engine (`replacement::is_duplicate_rewrite`) can reuse this exact -/// candidate-narrowing filter for its own dedup, instead of maintaining a -/// parallel reimplementation — the same "one real hash, reused everywhere -/// it's needed" rationale [`structural_hash`]'s own doc gives for -/// [`dag_export`](crate::dag_export)'s `pub(crate)` reuse. -pub type HashCache = HashMap<*const QueryExpr, u64>; - -/// Coarse structural hash used only to bucket [`InternTable::intern`]'s -/// candidate search — never the actual sharing decision (`PartialEq` is). -/// -/// `QueryExpr` carries `f64`s (`Literal(ScalarValue::Float64)`, `AggIntent::Quantile.q`, …), so it -/// cannot derive `std::hash::Hash`. Serializing to a canonical JSON string -/// and hashing that sidesteps the `f64` problem — but only for `node`'s own -/// tag and non-child fields, *not* its children's full values: each -/// `Rc`-backed child's contribution is its own [`structural_hash`], looked -/// up in `cache` if already computed there (memoized by `Rc` pointer -/// identity) rather than recursed into again. -/// -/// This is the DAG-aware fix a naive "just serialize the whole subtree" -/// hash would get wrong: after [`share_common_subtrees`] (or even before -/// it — a front end can emit internal `Rc` sharing directly, e.g. a -/// repeated subexpression within one query), `node` is generally a DAG, -/// not a tree. A full-subtree serialization re-serializes — re-walks — -/// any descendant `node` already shares internally once per parent that -/// references it; called once per node in a bottom-up pass (as -/// [`InternTable::intern`] and [`dag_export`](crate::dag_export) both do), -/// that costs `O(subtree size)` *per node* instead of `O(1)` amortized — -/// quadratic-or-worse for a deep chain, compounding further with any real -/// internal sharing. Memoizing each child's hash by pointer identity in -/// `cache` (persisted across the whole pass by the caller, not reset per -/// node) makes each node's own contribution `O(1)` beyond its children's -/// already-known hashes, giving `O(N)` total for `N` nodes — matching -/// [`dag_node_count`]'s own DAG-vs-tree fix (issue #212/#223/#237's stage -/// 4) in spirit, applied to hashing instead of counting. -/// -/// `pub` (not private) so [`dag_export`](crate::dag_export) can call -/// this exact function for its exported nodes' `hash` field instead of -/// maintaining its own parallel reimplementation — issue #223 stage 3. That -/// makes `tools/dag-viewer`'s "shared subtree" highlighting reflect this -/// module's real hashing, not a lookalike computed a different way; see the -/// module doc's "Landing plan" section. A NaN/infinite `f64` makes JSON -/// serialization fail; falling back to a fixed hash just puts every such -/// node in one (larger, still `PartialEq`-disambiguated) bucket. Made `pub` -/// (rather than staying `pub(crate)`) for one more reuse across the crate -/// boundary: `asap_aware_mapping`'s workload-search MEMO engine -/// (`replacement::is_duplicate_rewrite`) needs the identical -/// candidate-narrowing filter this module's own [`InternTable::intern`] -/// already uses, so it doesn't have to reinvent (and risk drifting from) it. -/// -/// Exhaustive over every `QueryExpr` variant, matching [`rebuild_children`] -/// in which fields count as an operator child (must stay in sync — a new -/// variant fails to compile in both places until both are extended). -pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { - use QueryExpr::*; - - fn child_hash(child: &Rc, cache: &mut HashCache) -> u64 { - let ptr = Rc::as_ptr(child); - if let Some(&h) = cache.get(&ptr) { - return h; - } - let h = structural_hash(child, cache); - cache.insert(ptr, h); - h - } - - /// Hash `own_fields` (this node's own tag and non-child scalar - /// fields — anything JSON-serializable and small, i.e. never a - /// `QueryExpr` subtree) via the same canonical-JSON-string trick the - /// whole-subtree version used, just applied to `O(1)` fields instead - /// of `O(subtree size)`. - fn hash_own_fields(hasher: &mut impl Hasher, own_fields: &impl serde::Serialize) { - serde_json::to_string(own_fields) - .unwrap_or_default() - .hash(hasher); - } - - let mut hasher = std::collections::hash_map::DefaultHasher::new(); - match node { - Scan { - source, - predicates, - schema, - } => hash_own_fields(&mut hasher, &("Scan", source, predicates, schema)), - PromqlVectorFromScalar(c) => { - "PromqlVectorFromScalar".hash(&mut hasher); - child_hash(c, cache).hash(&mut hasher); - } - PromqlScalarFromVector(c) => { - "PromqlScalarFromVector".hash(&mut hasher); - child_hash(c, cache).hash(&mut hasher); - } - PromqlRelabel { dst, value, child } => { - hash_own_fields(&mut hasher, &("PromqlRelabel", dst, value)); - child_hash(child, cache).hash(&mut hasher); - } - PromqlInfoEnrich { selector, child } => { - hash_own_fields(&mut hasher, &("PromqlInfoEnrich", selector)); - child_hash(child, cache).hash(&mut hasher); - } - PromqlSeriesSample { by, kind, child } => { - hash_own_fields(&mut hasher, &("PromqlSeriesSample", by, kind)); - child_hash(child, cache).hash(&mut hasher); - } - Filter { pred, child } => { - hash_own_fields(&mut hasher, &("Filter", pred)); - child_hash(child, cache).hash(&mut hasher); - } - Project { - cols, - qualifier, - child, - } => { - hash_own_fields(&mut hasher, &("Project", cols, qualifier)); - child_hash(child, cache).hash(&mut hasher); - } - Aggregate { - reduction, - measures, - output_names, - filters, - having, - child, - } => { - hash_own_fields( - &mut hasher, - &( - "Aggregate", - reduction, - measures, - output_names, - filters, - having, - ), - ); - child_hash(child, cache).hash(&mut hasher); - } - Dedup { cols, child } => { - hash_own_fields(&mut hasher, &("Dedup", cols)); - child_hash(child, cache).hash(&mut hasher); - } - Concat { - children, - discriminator_unique_key, - } => { - hash_own_fields(&mut hasher, &("Concat", discriminator_unique_key)); - for c in children { - // Stored by value, not `Rc` — see `rebuild_children`'s - // `intern_owned` use for this variant — so there's no - // pointer to memoize on here; recurse directly. Any - // `Rc`-typed descendant beneath `c` still gets memoized - // once this call reaches it. - structural_hash(c, cache).hash(&mut hasher); - } - } - Join { - kind, - pred, - left, - right, - } => { - hash_own_fields(&mut hasher, &("Join", kind, pred)); - child_hash(left, cache).hash(&mut hasher); - child_hash(right, cache).hash(&mut hasher); - } - SetOp { - kind, - all, - left, - right, - } => { - hash_own_fields(&mut hasher, &("SetOp", kind, all)); - child_hash(left, cache).hash(&mut hasher); - child_hash(right, cache).hash(&mut hasher); - } - Sort { - keys, - partition_by, - child, - } => { - hash_own_fields(&mut hasher, &("Sort", keys, partition_by)); - child_hash(child, cache).hash(&mut hasher); - } - Limit { n, offset, child } => { - hash_own_fields(&mut hasher, &("Limit", n, offset)); - child_hash(child, cache).hash(&mut hasher); - } - PromqlSubquery { - range, - resolution, - child, - } => { - hash_own_fields(&mut hasher, &("PromqlSubquery", range, resolution)); - child_hash(child, cache).hash(&mut hasher); - } - TimeRange { range, child } => { - hash_own_fields(&mut hasher, &("TimeRange", range)); - child_hash(child, cache).hash(&mut hasher); - } - TimeShift { shift, child } => { - hash_own_fields(&mut hasher, &("TimeShift", shift)); - child_hash(child, cache).hash(&mut hasher); - } - SQLWindowFunc { - func, - args, - partition_by, - order_by, - frame, - output_name, - child, - } => { - hash_own_fields( - &mut hasher, - &( - "SQLWindowFunc", - func, - args, - partition_by, - order_by, - frame, - output_name, - ), - ); - child_hash(child, cache).hash(&mut hasher); - } - BinaryOp { - op, - lhs, - rhs, - vector_match, - } => { - hash_own_fields(&mut hasher, &("BinaryOp", op, vector_match)); - child_hash(lhs, cache).hash(&mut hasher); - child_hash(rhs, cache).hash(&mut hasher); - } - // `EvalTimestamp`, `PromqlScalarBridge`, and the scalar variants - // (issue #205) are all leaves for this traversal's purposes — none - // has an operator child to look up in `cache` — so hashing the - // whole node via `serde_json` in one shot is already `O(node - // size)`, not `O(subtree size)`: exactly the same cost the - // per-variant `hash_own_fields` calls above pay, just without - // needing to spell out each field individually. Matches - // `rebuild_children`'s and `dag_node_count`'s identical scope - // decision for these variants ("never descended into"). - EvalTimestamp - | CurrentTimestamp - | PromqlScalarBridge(_) - | Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => hash_own_fields(&mut hasher, node), - } - hasher.finish() -} - -/// Count of *unique* nodes reachable from `root`, deduplicated by `Rc` -/// pointer identity (`Rc::as_ptr`) — the real size of the DAG rooted at -/// `root`, not a tree-walk count. -/// -/// After [`share_common_subtrees`] runs (or even before it, for a tree a -/// front end already built with internal `Rc` sharing — e.g. re-running -/// CSE, or a single-query repeated subexpression), `root` is generally a -/// **DAG**, not a tree — that is this whole module's premise. Anything that -/// walks `root` as if every reference were a fresh subtree (a naive -/// recursive walk with no identity tracking, or a naive full -/// `serde_json` serialization — `Rc`'s `Serialize` impl serializes the -/// pointee's *value* at every occurrence, it does not dedupe by identity) -/// re-visits/re-counts an already-shared descendant once per parent that -/// references it, over-counting relative to the actual work of holding it -/// in memory or recomputing it once. This function is the DAG-correct -/// alternative: each unique node is counted exactly once, regardless of -/// how many places within `root` reference it. -/// -/// `pub` so cost-aware callers outside this crate (e.g. -/// `asap_aware_mapping::CostModel::cse_recompute_cost`'s default) have a -/// DAG-correct structural-size proxy available, instead of reaching for -/// something tree-shaped like a raw serialization length. -/// -/// Same operator-child traversal scope as [`share_common_subtrees`] itself -/// (see the module doc's "Algorithm" section, and this module's private -/// `rebuild_children`) — a scalar subexpression embedded in a wrapper -/// position (`Predicate`, `ProjectItem.expr`, `Aggregate.having`, …) is not -/// separately visited, matching this module's own stated scope; it's -/// counted as part of its owning operator node, the same node -/// `rebuild_children` treats as a single opaque leaf for interning -/// purposes. -pub fn dag_node_count(root: &QueryExpr) -> usize { - let mut seen: std::collections::HashSet<*const QueryExpr> = std::collections::HashSet::new(); - count_unique(root, &mut seen) -} - -/// One node's own contribution (`1`) plus each *not-yet-seen* operator -/// child's contribution — exhaustive over every `QueryExpr` variant, -/// enumerating the same fields [`rebuild_children`] does (kept as a -/// separate, read-only traversal rather than threaded through -/// `rebuild_children` itself, since that function consumes and rebuilds -/// its input while this one only ever reads it). -fn count_unique(node: &QueryExpr, seen: &mut std::collections::HashSet<*const QueryExpr>) -> usize { - use QueryExpr::*; - - /// Visit one `Rc`-held child: counts (and recurses into) it only the - /// first time its pointer is seen, `0` on every later occurrence — - /// this is the actual dedup step. - fn visit( - child: &Rc, - seen: &mut std::collections::HashSet<*const QueryExpr>, - ) -> usize { - if seen.insert(Rc::as_ptr(child)) { - count_unique(child, seen) - } else { - 0 - } - } - - 1 + match node { - // `PromqlScalarBridge`'s child is a scalar-sub-language node (issue - // #220), never descended into — same treatment `rebuild_children` - // gives it (see that function's comment on this same variant). - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => 0, - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => visit(c, seen), - PromqlRelabel { child, .. } - | PromqlInfoEnrich { child, .. } - | PromqlSeriesSample { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | Sort { child, .. } - | Limit { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } => visit(child, seen), - // `Concat`'s branches are stored by value (`Vec`, not - // `Rc` — see `rebuild_children`'s `intern_owned` use for - // this variant), so a branch has no `Rc` identity of its own to - // dedup on at this position; still recurse into each in case an - // `Rc`-shared descendant appears further down. - Concat { children, .. } => children.iter().map(|c| count_unique(c, seen)).sum(), - Join { left, right, .. } | SetOp { left, right, .. } => { - visit(left, seen) + visit(right, seen) - } - BinaryOp { lhs, rhs, .. } => visit(lhs, seen) + visit(rhs, seen), - // Scalar variants (issue #205) — never descended into, matching - // `rebuild_children`'s own scope exactly (see its trailing match - // arm and this module's "Algorithm" section). - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => 0, - } -} - -/// Recurse into `child`, then intern the result. `Rc::try_unwrap` recovers -/// the owned node without cloning in the overwhelmingly common case — a -/// tree freshly built by a front end / `resolve_root`, not yet shared by any -/// prior CSE pass, where every `Rc` is uniquely owned. Falls back to cloning -/// this node's own fields (its children stay `Rc`s, not deep-copied) only -/// when `child` is already shared — e.g. re-running CSE over a tree that -/// went through a previous `share_common_subtrees` pass; a structural -/// duplicate collapses right back onto `child` itself via `PartialEq`, an -/// already-optimal no-op. -fn intern_child(table: &mut InternTable, child: Rc) -> Rc { - match Rc::try_unwrap(child) { - Ok(owned) => intern_bottom_up(table, owned), - Err(shared) => intern_bottom_up(table, (*shared).clone()), - } -} - -/// Like [`intern_child`], for a `Concat` branch — stored by value -/// (`Vec`, not `Rc`), so this position itself can never -/// alias another parent. Interning it anyway still lets any `Rc`-typed -/// descendant of the branch participate in sharing, and registers the -/// branch's own hash/value in the table for a *different* `Concat` elsewhere -/// with a structurally identical branch (which — being in its own `Vec` -/// slot too — still can't literally share the `Rc`, but this keeps the -/// interning behavior uniform and the table's bucket contents consistent). -fn intern_owned(table: &mut InternTable, expr: QueryExpr) -> QueryExpr { - let rc = intern_bottom_up(table, expr); - Rc::try_unwrap(rc).unwrap_or_else(|shared| (*shared).clone()) -} - -/// Bottom-up: rebuild `expr`'s children (recursively interning each), then -/// intern the rebuilt node itself. -fn intern_bottom_up(table: &mut InternTable, expr: QueryExpr) -> Rc { - let rebuilt = rebuild_children(table, expr); - table.intern(rebuilt) -} - -/// Rebuild `expr` with each **operator** child (see the module doc on scope) -/// replaced by its interned `Rc`. Exhaustive over every `QueryExpr` variant, -/// matching `canonicalize.rs`'s `children_mut` exactly in which fields count -/// as an operator child — new variants fail to compile here until this match -/// is extended. -fn rebuild_children(table: &mut InternTable, expr: QueryExpr) -> QueryExpr { - use QueryExpr::*; - match expr { - Scan { .. } | EvalTimestamp | CurrentTimestamp => expr, - PromqlVectorFromScalar(c) => PromqlVectorFromScalar(intern_child(table, c)), - PromqlScalarFromVector(c) => PromqlScalarFromVector(intern_child(table, c)), - PromqlRelabel { dst, value, child } => PromqlRelabel { - dst, - value, - child: intern_child(table, child), - }, - PromqlInfoEnrich { selector, child } => PromqlInfoEnrich { - selector, - child: intern_child(table, child), - }, - PromqlSeriesSample { by, kind, child } => PromqlSeriesSample { - by, - kind, - child: intern_child(table, child), - }, - Filter { pred, child } => Filter { - pred, - child: intern_child(table, child), - }, - Project { - cols, - qualifier, - child, - } => Project { - cols, - qualifier, - child: intern_child(table, child), - }, - Aggregate { - reduction, - measures, - output_names, - filters, - having, - child, - } => Aggregate { - reduction, - measures, - output_names, - filters, - having, - child: intern_child(table, child), - }, - Dedup { cols, child } => Dedup { - cols, - child: intern_child(table, child), - }, - Concat { - children, - discriminator_unique_key, - } => Concat { - children: children - .into_iter() - .map(|c| intern_owned(table, c)) - .collect(), - discriminator_unique_key, - }, - Join { - kind, - pred, - left, - right, - } => Join { - kind, - pred, - left: intern_child(table, left), - right: intern_child(table, right), - }, - SetOp { - kind, - all, - left, - right, - } => SetOp { - kind, - all, - left: intern_child(table, left), - right: intern_child(table, right), - }, - Sort { - keys, - partition_by, - child, - } => Sort { - keys, - partition_by, - child: intern_child(table, child), - }, - Limit { n, offset, child } => Limit { - n, - offset, - child: intern_child(table, child), - }, - PromqlSubquery { - range, - resolution, - child, - } => PromqlSubquery { - range, - resolution, - child: intern_child(table, child), - }, - TimeRange { range, child } => TimeRange { - range, - child: intern_child(table, child), - }, - TimeShift { shift, child } => TimeShift { - shift, - child: intern_child(table, child), - }, - SQLWindowFunc { - func, - args, - partition_by, - order_by, - frame, - output_name, - child, - } => SQLWindowFunc { - func, - args, - partition_by, - order_by, - frame, - output_name, - child: intern_child(table, child), - }, - BinaryOp { - op, - lhs, - rhs, - vector_match, - } => BinaryOp { - op, - lhs: intern_child(table, lhs), - rhs: intern_child(table, rhs), - vector_match, - }, - // `PromqlScalarBridge`'s child is a scalar-sub-language node (issue - // #220) — same "never descended into" treatment as the scalar - // variants below; the whole bridge node is still interned as a unit - // by the `table.intern(rebuilt)` call in `intern_bottom_up`. - PromqlScalarBridge(_) => expr, - // Scalar variants (issue #205) — never descended into; see the - // module doc's "Algorithm" section on scope. Left byte-for-byte - // unchanged: predicate / project-list / sort-key / window-arg - // expressions stay embedded as opaque leaf data, compared by the - // enclosing operator node's derived `PartialEq`. - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => expr, - } -} - -/// Share structurally-identical, sharing-legal subtrees across a workload's -/// query roots (or within one query, for `roots.len() == 1` — see the -/// module doc's "Single-query CSE" section). Every root's *value* is -/// unchanged (`PartialEq`-equal to its input) — only its internal `Rc` -/// structure may now alias another root's, or another part of its own tree. -/// -/// `roots` must already be bound + canonicalized (post-`resolve_root`). -/// `Id` is caller-chosen — a `QueryWorkload` entry's own key, an index, a -/// query name, whatever identifies one root through the pipeline; this -/// module has no opinion on its shape. -pub fn share_common_subtrees(roots: Vec<(Id, QueryExpr)>) -> Vec<(Id, Rc)> { - let mut table = InternTable::new(); - roots - .into_iter() - .map(|(id, expr)| (id, intern_bottom_up(&mut table, expr))) - .collect() -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::pre_asap::agg_intent::AggIntent; - use crate::pre_asap::expr_ir::{CompareOpKind, ScalarValue}; - use crate::pre_asap::query_expr::{BinaryOpKind, GroupKeys, Predicate, Reduction, Source}; - use crate::pre_asap::schema::{Column, DataType, Schema}; - use crate::types::AccuracyTarget; - - /// `[ts, service, value, latency]`. - fn scan() -> QueryExpr { - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("service", DataType::Utf8, false), - Column::new("value", DataType::Float64, false), - Column::new("latency", DataType::Float64, false), - ], - 0, - vec![], - ), - } - } - - fn quantile_agg(by: Vec, col: Option, q: f64) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::by(by), - measures: vec![AggIntent::Quantile { - col, - q, - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan()), - } - } - - #[test] - fn distinct_column_quantiles_do_not_merge() { - // Grouped (unique_keys present) so the legality gate isn't what's - // blocking the merge — only the differing `col` is. - let a = quantile_agg(vec![1], Some(2), 0.5); - let b = quantile_agg(vec![1], Some(3), 0.5); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!( - !Rc::ptr_eq(ra, rb), - "distinct-column Quantiles must not be shared" - ); - assert_ne!(ra, rb); - } - - // Two aggregates that differ only in one measure's `FILTER` predicate - // compute different values, so structural sharing must keep them apart. - #[test] - fn filtered_and_unfiltered_aggregates_do_not_merge() { - let a = quantile_agg(vec![1], Some(2), 0.5); - let mut b = quantile_agg(vec![1], Some(2), 0.5); - let QueryExpr::Aggregate { filters, .. } = &mut b else { - unreachable!() - }; - *filters = vec![Some(Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(1.0))), - })))]; - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!(!Rc::ptr_eq(ra, rb), "a filtered measure must not be shared"); - assert_ne!(ra, rb); - } - - #[test] - fn no_unique_keys_means_no_merge_even_when_structurally_identical() { - // Ungrouped (global) aggregate: `by` is empty, so - // `aggregate_output_schema` reports no unique key today — not - // hoistable even though `a` and `b` are structurally identical. - let a = quantile_agg(vec![], Some(2), 0.9); - let b = quantile_agg(vec![], Some(2), 0.9); - assert_eq!(a, b, "fixture sanity: the two trees are structurally equal"); - assert!( - !a.output_schema().unwrap().has_unique_key(), - "fixture sanity: an ungrouped aggregate has no provable unique key" - ); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!( - !Rc::ptr_eq(ra, rb), - "no unique key ⇒ never hoisted, even for an identical structural match" - ); - } - - #[test] - fn median_and_explicit_half_percentile_merge() { - // Two front-end spellings ("median" and "approx_percentile_cont(., - // 0.5)") already lower to the identical `AggIntent::Quantile { q: - // 0.5, .. }` today (see `sql_lowering.rs`'s - // `median_is_the_same_intent_as_an_explicit_half_percentile`) — here - // built directly (grouped, so a unique key is provable) as two - // independently-constructed but structurally identical trees, the - // way two different call sites in a workload would produce them. - let median = quantile_agg(vec![1], Some(2), 0.5); - let approx_percentile_cont_half = quantile_agg(vec![1], Some(2), 0.5); - let shared = share_common_subtrees(vec![ - ("median", median), - ("percentile", approx_percentile_cont_half), - ]); - let [(_, m), (_, p)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!( - Rc::ptr_eq(m, p), - "median and an explicit 0.5 percentile must merge onto one Rc" - ); - } - - #[test] - fn single_query_shares_its_own_repeated_subtree() { - // One query root referencing the same grouped Aggregate on both - // BinaryOp branches — built as two separately-allocated but - // structurally identical subtrees (`.clone()` into two distinct - // `Rc::new` calls), the shape a front end emitting a repeated - // sub-expression would actually produce (no sharing yet). A - // workload of size 1 still interns bottom-up within this one tree — - // no separate single-query mechanism needed. - let agg = quantile_agg(vec![1], Some(2), 0.5); - let root = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), - lhs: Rc::new(agg.clone()), - rhs: Rc::new(agg), - vector_match: None, - }; - let shared = share_common_subtrees(vec![("q", root)]); - let [(_, root)] = shared.as_slice() else { - panic!("expected 1 root"); - }; - let QueryExpr::BinaryOp { lhs, rhs, .. } = root.as_ref() else { - panic!("expected BinaryOp root, got {root:?}"); - }; - assert!( - Rc::ptr_eq(lhs, rhs), - "the two structurally identical branches must collapse onto one Rc" - ); - } - - // ── structural_hash (DAG-aware memoization) ───────────────────────── - - #[test] - fn structural_hash_is_stable_across_cache_states() { - // The hash of a given *value* must not depend on whether its cache - // started warm or cold — memoization changes how much work is - // redone, never what a node's hash actually is. - let agg = quantile_agg(vec![1], Some(2), 0.5); - let mut cold = HashMap::new(); - let mut warm = HashMap::new(); - // Prime `warm` with an unrelated node first, so it's non-empty but - // holds nothing relevant to `agg`. - structural_hash(&scan(), &mut warm); - assert_eq!( - structural_hash(&agg, &mut cold), - structural_hash(&agg, &mut warm), - "hash must be independent of unrelated cache state" - ); - } - - #[test] - fn structural_hash_of_an_internally_shared_tree_matches_the_unshared_equivalent() { - // The same BinaryOp-with-shared-branches shape as - // `dag_node_count_deduplicates_an_internally_shared_subtree` below: - // hashing it (however the memoization internally short-circuits the - // second branch) must produce the exact same value as hashing a - // structurally-identical tree built with *no* sharing at all — the - // whole point of memoization is not changing the answer, only the - // work needed to reach it. - let agg = quantile_agg(vec![1], Some(2), 0.5); - let shared_root = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), - lhs: Rc::new(agg.clone()), - rhs: Rc::new(agg.clone()), - vector_match: None, - }; - let unshared_root = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), - lhs: Rc::new(agg.clone()), - rhs: Rc::new(agg), // a second, independently-allocated Rc with an equal value - vector_match: None, - }; - let mut cache = HashMap::new(); - assert_eq!( - structural_hash(&shared_root, &mut cache), - structural_hash(&unshared_root, &mut HashMap::new()), - ); - } - - #[test] - fn structural_hash_memoizes_a_shared_descendant_exactly_once() { - // Direct proof the cache is actually doing its job: hashing a - // BinaryOp whose two branches are the *same* Rc (2 underlying - // nodes: Scan + Aggregate) should populate the cache with exactly - // 2 entries — the shared branch's nodes, cached once each when - // first reached — not a fresh entry (or a fresh, redundant - // recursive walk) for the second occurrence. - let agg = quantile_agg(vec![1], Some(2), 0.5); - let shared = Rc::new(agg); - let root = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), - lhs: Rc::clone(&shared), - rhs: Rc::clone(&shared), - vector_match: None, - }; - let mut cache = HashMap::new(); - structural_hash(&root, &mut cache); - assert_eq!( - cache.len(), - 2, - "expected exactly one cache entry per unique node in the shared \ - branch (Aggregate + its Scan child), got {} entries: {:?}", - cache.len(), - cache - ); - } - - // ── dag_node_count ─────────────────────────────────────────────────── - - #[test] - fn dag_node_count_is_the_naive_count_when_nothing_is_shared() { - // scan() alone: 1 node. - assert_eq!(dag_node_count(&scan()), 1); - // quantile_agg's own child is a fresh, unshared scan(): 2 nodes. - assert_eq!(dag_node_count(&quantile_agg(vec![1], Some(2), 0.5)), 2); - } - - #[test] - fn dag_node_count_deduplicates_an_internally_shared_subtree() { - // Same shape as `single_query_shares_its_own_repeated_subtree`: a - // BinaryOp whose two branches are the *same* Rc after - // `share_common_subtrees` (2 nodes: Scan + Aggregate) — the root - // itself makes 3 unique nodes total (BinaryOp, Aggregate, Scan), - // not 5 (which a tree-walk / naive serialization, counting the - // shared branch's 2 nodes twice, would report). - let agg = quantile_agg(vec![1], Some(2), 0.5); - let root = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), - lhs: Rc::new(agg.clone()), - rhs: Rc::new(agg), - vector_match: None, - }; - let shared = share_common_subtrees(vec![("q", root)]); - let [(_, root)] = shared.as_slice() else { - panic!("expected 1 root"); - }; - assert_eq!( - dag_node_count(root), - 3, - "the shared branch's 2 nodes must be counted once, not once per \ - occurrence — got {} for {root:?}", - dag_node_count(root) - ); - } - - #[test] - fn dag_node_count_deduplicates_across_two_workload_roots() { - // Two workload roots sharing one Aggregate after - // `share_common_subtrees` (the `duplicate_workload_queries_...` - // shape from `crates/integration-tests/tests/cse.rs`, built - // directly here): each root's own `dag_node_count` must report the - // shared subtree's real size once, not double-count anything — - // there's nothing *to* double-count from a single root's own count - // in this case (no root references the shared node twice), so this - // pins the simpler, more common case that a per-candidate cost - // proxy (`CseCandidate::subtree` in `asap-aware-mapping`) actually - // exercises: counting one occurrence's own reachable DAG size. - let a = quantile_agg(vec![1], Some(2), 0.5); - let b = quantile_agg(vec![1], Some(2), 0.5); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!(Rc::ptr_eq(ra, rb), "fixture sanity: the two roots merged"); - assert_eq!(dag_node_count(ra), 2); - assert_eq!(dag_node_count(rb), 2); - } - - #[test] - fn dedup_gates_sharing_the_same_as_aggregate() { - // `Dedup { cols }` adds `cols` as a unique key — so two identical - // `Dedup` subtrees over a keyed column *do* merge, exercising the - // legality gate on a non-`Aggregate` node. - let dedup = |cols: Vec| QueryExpr::Dedup { - cols, - child: Rc::new(scan()), - }; - let a = dedup(vec![1]); - let b = dedup(vec![1]); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!( - Rc::ptr_eq(ra, rb), - "Dedup on the same cols has a provable unique key and should merge" - ); - } - - #[test] - fn group_keys_gate_still_prevented_when_partition_by_without_used() { - // Sanity on the module's advertised precedent: a `without(...)` - // grouping stays open (no unique key) even though `by` is - // non-empty-shaped structurally, so two identical `without` groups - // do not merge under the same gate that blocks the ungrouped case. - let without_agg = || QueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::without(vec![0])), - measures: vec![AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan()), - }; - let a = without_agg(); - let b = without_agg(); - assert!(!a.output_schema().unwrap().has_unique_key()); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!(!Rc::ptr_eq(ra, rb)); - } -} diff --git a/crates/types/src/pre_asap/expr_ir.rs b/crates/types/src/pre_asap/expr_ir.rs index 2aed64aa2..b1069079b 100644 --- a/crates/types/src/pre_asap/expr_ir.rs +++ b/crates/types/src/pre_asap/expr_ir.rs @@ -1,27 +1,17 @@ -//! Column-reference and scalar-operator vocabulary shared by the whole -//! canonical [`QueryExpr`](super::query_expr::QueryExpr) tree. -//! -//! Issue #205: the scalar expression shapes (`Column`/`Literal`/`Compare`/…) -//! used to live in a separate, self-recursive `Expr` tree here, reachable -//! from `QueryExpr` only through wrapper fields (`Predicate`, `ProjectItem`, -//! `SortKey`). They're variants of `QueryExpr` itself now — one recursive -//! tree, not two type families joined by wrappers — generic over the same -//! column-reference state `C` the rest of `QueryExpr` already carries -//! (issue #179): [`ColumnRef`] (name-based, front-end-emitted) or -//! [`ColumnId`](super::schema::ColumnId) (positional, once bound). -//! -//! What's left here is the vocabulary those scalar variants are built from — -//! [`ScalarValue`], [`CompareOpKind`], [`ArithmeticOpKind`] — the **union** of what the two +//! Field-reference and scalar-operator vocabulary shared by the IR's scalar +//! expressions ([`crate::ir::ScalarExpr`]) and the front ends' unresolved +//! form: [`ColumnRef`] (name-based, front-end-emitted; positional +//! [`ColumnId`](super::schema::ColumnId) once bound), and [`ScalarValue`], +//! [`CompareOpKind`], [`ArithmeticOpKind`] — the **union** of what the two //! front ends need: PromQL contributes `Regex` / `NotRegex` (`=~` / `!~`); SQL //! contributes arithmetic, `CASE`, `IN`, `CAST`, `IS [NOT] NULL`, scalar //! function calls, and the `LIKE` / `ILIKE` comparison family. use serde::{Deserialize, Serialize}; -/// A name-based column reference — the front-end-emitted, unresolved state of -/// [`QueryExpr::Column`](super::query_expr::QueryExpr::Column) (`C = -/// ColumnRef`); the [`SchemaResolver`](super::schema_resolver::SchemaResolver) resolves it to a -/// positional [`ColumnId`](super::schema::ColumnId). Includes the two +/// A name-based column reference — the front-end-emitted, unresolved form of +/// [`ScalarExpr::Column`](crate::ir::ScalarExpr::Column); front-end name +/// resolution turns it into a positional [`ColumnId`](super::schema::ColumnId). Includes the two /// PromQL-conventional synthetic columns. #[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] pub enum ColumnRef { diff --git a/crates/types/src/pre_asap/mod.rs b/crates/types/src/pre_asap/mod.rs index f8f7e3519..37a269c37 100644 --- a/crates/types/src/pre_asap/mod.rs +++ b/crates/types/src/pre_asap/mod.rs @@ -1,64 +1,34 @@ -//! The canonical pre-ASAP intent algebra IR. +//! Shared vocabulary of the operator IR. The operators themselves live in +//! [`crate::ir`]; this module holds the field types they are built from. //! -//! - [`query_expr`] — the canonical, language- and deployment-independent -//! intent algebra: one recursive [`QueryExpr`] tree (relational operators -//! *and* scalar expression shapes both, since issue #205) + [`AggIntent`], -//! generic over the column-reference state (positional [`ColumnId`] once -//! bound, name-based [`ColumnRef`] before). -//! - [`agg_intent`] — the aggregation-intent vocabulary. -//! - [`expr_ir`] — the [`ColumnRef`] column-reference type and the scalar -//! operator/literal vocabulary ([`ScalarValue`], [`CompareOpKind`], [`ArithmeticOpKind`]) -//! [`QueryExpr`]'s scalar variants are built from. +//! Operator parameters and schema derivation live in [`crate::ir`]. +//! - [`agg_intent`] — the aggregation-intent vocabulary ([`AggIntent`]). +//! - [`expr_ir`] — [`ColumnRef`] and the scalar literal / operator kinds +//! ([`ScalarValue`], [`CompareOpKind`], [`ArithmeticOpKind`]). //! - [`schema`] — the per-edge [`Schema`] every node carries. -//! - [`schema_resolver`] / [`column_resolution`] — name resolution: turn a `ColumnRef` -//! into a positional `ColumnId` against an in-scope [`Schema`]. -//! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedQueryExpr`] tree to -//! canonical [`ResolvedQueryExpr`] (issue #179): both front ends -//! (`asap-frontend-promql`, `asap-frontend-sql`) construct `UnresolvedQueryExpr` -//! directly during their own `interpret` step and call -//! [`resolve_root`] on the result — there is no separate per-language -//! relational tree or converter anymore. -//! - [`canonicalize`] — post-lowering structural normalization of [`QueryExpr`] -//! (issue #34), run by [`resolve_root`]. -//! - [`cse`] — workload-level structural common-subexpression elimination -//! over an already-`resolve_root`'d tree (issue #212, #222, #223), run -//! *after* `resolve_root` / `canonicalize` and *before* implementation -//! (`asap_aware_mapping::replacement`). -//! -//! Formerly the separate `asap-l2` crate; folded in here since -//! `schema_resolver`/`column_resolution`/`canonicalize`/`resolve` have no -//! front-end-specific logic — they operate directly on this crate's own -//! `QueryExpr`. +//! - [`column_resolution`] — turn a name-based `ColumnRef` into a positional +//! `ColumnId` against a [`Schema`] (used by front-end name resolution). +//! - [`scalar_signature`] — type rules of the map scalar functions. pub mod agg_intent; -pub mod canonicalize; pub mod column_resolution; -pub mod cse; pub mod expr_ir; -pub mod query_expr; -pub mod resolve; pub mod scalar_signature; pub mod schema; -pub mod schema_resolver; +pub use crate::ir::operator_properties::{ + AtModifier, BinaryOpKind, ColState, ConcatDiscriminatorKey, DataModel, GroupKeys, GroupSide, + InfoMatcher, JoinKind, PromQLVectorSetOpKind, Reduction, RelationalSetOpKind, SampleKind, + Source, TimeShift, VectorGrouping, VectorMatch, VectorMatchKind, WindowFrame, WindowFrameBound, + WindowFrameOffset, WindowFrameUnits, WindowFuncKind, +}; pub use agg_intent::{ agg_accuracy, agg_is_exact, agg_is_mergeable, default_cardinality, default_quantile, AggIntent, MathFunc, TimeFunc, }; -pub use canonicalize::canonicalize; -pub use column_resolution::{ - output_schema_for_aggregate, resolve_column_ref, resolve_column_refs, resolve_expr, - ResolveError, -}; -pub use cse::share_common_subtrees; +pub use column_resolution::{resolve_column_ref, resolve_column_refs, ResolveError}; pub use expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; -pub use query_expr::{ - aggregate_output_schema, any_measure_filtered, AtModifier, BinaryOpKind, ColState, DataModel, - GroupKeys, GroupSide, InfoMatcher, JoinKind, Predicate, ProjectItem, PromQLVectorSetOpKind, - QueryExpr, QueryExprError, Reduction, RelationalSetOpKind, ResolvedQueryExpr, SampleKind, - SortKey, Source, TimeShift, UnresolvedQueryExpr, VectorGrouping, VectorMatch, VectorMatchKind, - WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, -}; -pub use resolve::{resolve_root, ResolveTreeError}; -pub use schema::{Column, ColumnId, DataType, Schema}; -pub use schema_resolver::{SchemaCatalog, SchemaResolver, UsageDerivedCatalog}; +pub use schema::{ColumnId, DataType, Field, FieldDataType, Schema}; + +pub use crate::ir::aggregate_schema::aggregate_output_schema; +pub use crate::ir::SchemaDerivationError; diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs deleted file mode 100644 index 9f0e800c1..000000000 --- a/crates/types/src/pre_asap/query_expr.rs +++ /dev/null @@ -1,2768 +0,0 @@ -//! The canonical pre-ASAP intent algebra IR. -//! -//! Language- and deployment-independent. `Rc`-owned tree — a child field is -//! `Rc>` rather than `Box>` so a structurally -//! identical sub-expression can be shared (the same `Rc`) across more than -//! one parent, within one query or across a `QueryWorkload` batch, instead of -//! being duplicated. Nothing in this module produces that sharing on its -//! own — construction still allocates a fresh `Rc` per node, the same shape -//! as the old `Box` tree — a separate CSE pass is what turns two -//! independently constructed, structurally-equal subtrees into two -//! references to one `Rc` (issue #212, #222). Column identity is -//! **positional** (`Aggregate.reduction: Reduction`, wrapping `GroupKeys` -//! for the grouped case), resolved by the [`SchemaResolver`](super::schema_resolver) against -//! the self-contained [`Schema`] carried on each `Scan`. - -use std::rc::Rc; -use std::time::Duration; - -use serde::{Deserialize, Serialize}; -use thiserror::Error; - -use super::agg_intent::AggIntent; -use super::expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; -use super::schema::{Column, ColumnId, DataType, Schema}; - -/// The column-reference resolution state a [`QueryExpr`] tree carries — -/// [`ColumnId`] (the default, and what the bare `QueryExpr` name has always -/// meant) once the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has resolved every -/// reference positionally, or the front-end-emitted, name-based [`ColumnRef`] -/// before binding. The only place the two states differ in *shape* rather -/// than just in which type fills `C` is [`QueryExpr::Scan`]'s `schema` field: -/// a bound tree's binding schema is always known (the SchemaResolver is total, so -/// [`ScanSchema`](Self::ScanSchema) `= Schema`); an unresolved front-end -/// `Scan` knows its schema only when the front end already has it without -/// binding — a SQL leaf, catalog-backed (`Some`) — `None` (PromQL) defers to -/// the SchemaResolver, so `ScanSchema = Option`. -pub trait ColState: - Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de> -{ - /// What [`QueryExpr::Scan`]'s `schema` field holds for a tree in this state. - type ScanSchema: Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de>; -} - -impl ColState for ColumnId { - type ScanSchema = Schema; -} - -impl ColState for ColumnRef { - type ScanSchema = Option; -} - -/// Errors from schema derivation over a canonical tree. -#[derive(Debug, Error)] -pub enum QueryExprError { - #[error("invalid scalar function signature: {0}")] - InvalidScalarSignature(String), - #[error("by-column id {0} out of range (input has {1} columns)")] - InvalidGroupByColumn(ColumnId, usize), - #[error("Concat requires at least one child")] - EmptyConcat, - /// [`QueryExpr::output_schema`] called on (or reached, while recursing, a - /// child that is) one of the scalar variants (issue #205) — those have no - /// independent row schema of their own; a scalar expression's *type* only - /// makes sense against the schema it's embedded in (see `infer_expr_type`, - /// used by `Project`'s own `output_schema` arm instead). - #[error("a scalar expression has no row schema of its own")] - ScalarHasNoRowSchema, - #[error("invalid per-series sample column: {0}")] - InvalidSampleColumn(String), -} - -// ── Leaf / supporting types ─────────────────────────────────────────────────── - -/// Positional grouping keys, shared by every "operate per group" operator: -/// `Aggregate.by` (reduce per group), `Sort.partition_by` (rank per group — -/// including generic `topk`/`bottomk`), and `SQLWindowFunc.partition_by` (window -/// per group). One spelling so grouping has a single home to evolve. Empty -/// (and `by`) = no grouping (a global operation). -/// -/// Heavy-hitter `AggIntent::TopK` carries its grouping here too, via the -/// enclosing `Aggregate.by` (issue #13) — so reduce, rank, and window groupings -/// all share this one type. -/// -/// ## `by` vs `without` (issue #39) -/// -/// The stored [`keys`](Self::keys) are **kept** labels for `by(...)` and -/// **excluded** labels for `without(...)`. PromQL's `without(labels)` groups by -/// every label *except* those listed; the complement can't be enumerated at -/// lowering time under an open (usage-derived) schema, so it is deferred to the -/// runtime — the excluded positions are stored, the kept set stays open. Only -/// `Aggregate` ever produces the `without` form; `Sort` / `SQLWindowFunc` / -/// `PromqlSeriesSample` groupings are always `by`. -/// -/// Serialises as a bare array for the (overwhelmingly common) `by` case — -/// wire-compatible with the `Vec` this field held before — and as -/// `{"without": [...]}` for the exclusion case. -#[derive(Debug, Clone, PartialEq, Eq, Hash)] -pub struct GroupKeys { - keys: Vec, - without: bool, -} - -// Not `#[derive(Default)]`: derive would add a `C: Default` bound, but an -// empty key set needs nothing from `C` — `ColumnRef` has no meaningful -// default anyway. -impl Default for GroupKeys { - fn default() -> Self { - Self { - keys: Vec::new(), - without: false, - } - } -} - -impl GroupKeys { - /// An empty key set — a global (ungrouped) operation. - pub fn none() -> Self { - Self::default() - } - /// `by(keys)` — group by exactly these columns. - pub fn by(keys: Vec) -> Self { - Self { - keys, - without: false, - } - } - /// `without(keys)` — group by every label *except* these (issue #39). The - /// kept set is runtime-resolved; only the excluded positions are stored. - pub fn without(keys: Vec) -> Self { - Self { - keys, - without: true, - } - } - /// Whether this is a `without(...)` exclusion grouping. - pub fn is_without(&self) -> bool { - self.without - } - /// The named keys — kept labels for `by`, excluded labels for `without`. - pub fn keys(&self) -> &[C] { - &self.keys - } -} - -impl std::ops::Deref for GroupKeys { - type Target = [C]; - fn deref(&self) -> &Self::Target { - &self.keys - } -} - -impl From> for GroupKeys { - fn from(keys: Vec) -> Self { - Self::by(keys) - } -} - -impl FromIterator for GroupKeys { - fn from_iter>(iter: I) -> Self { - Self::by(iter.into_iter().collect()) - } -} - -impl<'a, C> IntoIterator for &'a GroupKeys { - type Item = &'a C; - type IntoIter = std::slice::Iter<'a, C>; - fn into_iter(self) -> Self::IntoIter { - self.keys.iter() - } -} - -/// Compare directly against a `Vec` so call sites and tests can keep -/// writing `keys == vec![..]` / `assert_eq!(keys, &vec![..])`. A `without` -/// grouping never equals a bare `by` list. -impl PartialEq> for GroupKeys { - fn eq(&self, other: &Vec) -> bool { - !self.without && &self.keys == other - } -} - -/// (De)serialise as a bare array for `by`, or `{"without": [...]}` for the -/// exclusion form — keeping the `by` wire format identical to the old newtype. -/// Borrowed for `Serialize` (no `C: Clone` needed to write one out), owned for -/// `Deserialize` (there's nothing to borrow from). -#[derive(Serialize)] -#[serde(untagged)] -enum GroupKeysReprRef<'a, C> { - By(&'a [C]), - Without { without: &'a [C] }, -} - -#[derive(Deserialize)] -#[serde(untagged)] -enum GroupKeysRepr { - By(Vec), - Without { without: Vec }, -} - -impl Serialize for GroupKeys { - fn serialize(&self, serializer: S) -> Result { - if self.without { - GroupKeysReprRef::Without { - without: self.keys.as_slice(), - } - .serialize(serializer) - } else { - GroupKeysReprRef::By(self.keys.as_slice()).serialize(serializer) - } - } -} - -impl<'de, C: Deserialize<'de>> Deserialize<'de> for GroupKeys { - fn deserialize>(deserializer: D) -> Result { - Ok(match GroupKeysRepr::deserialize(deserializer)? { - GroupKeysRepr::By(keys) => Self::by(keys), - GroupKeysRepr::Without { without } => Self::without(without), - }) - } -} - -/// Which data model a `Source` / `AggIntent` operates over. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum DataModel { - TimeSeries, - Tabular, - Any, -} - -/// The leaf data source of a `Scan`. The schema itself rides on the -/// `Scan.schema` field (SchemaResolver-built); `Source` carries only the leaf's -/// identity. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum Source { - /// Time-series leaf — PromQL / DC lifecycle. Produces `(ts, value, *labels)`. - TimeSeries { metric: String }, - /// Tabular leaf — asap-fusion / future OLAP. Columns ride on `Scan.schema`. - Table { table_ref: String }, -} - -impl Source { - pub fn data_model(&self) -> DataModel { - match self { - Source::TimeSeries { .. } => DataModel::TimeSeries, - Source::Table { .. } => DataModel::Tabular, - } - } -} - -/// Operator on the query-level `BinaryOp` node. Reuses the scalar IR's -/// [`ArithmeticOpKind`] / [`CompareOpKind`] so every arithmetic/comparison -/// operator has exactly one representation (and one `Display`) across the IR. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum BinaryOpKind { - /// Arithmetic — `Add/Sub/Mul/Div/Mod` (shared with `QueryExpr::Arithmetic`). - Arithmetic(ArithmeticOpKind), - /// Comparison — `Eq/Ne/Lt/Le/Gt/Ge` + `Like/ILike/Regex` family (shared - /// with `QueryExpr::Compare`). PromQL keeps the matched series whose - /// comparison holds. - Compare(CompareOpKind), - /// PromQL comparison with the `bool` modifier: every matched series - /// yields 1 or 0 and loses its metric name. A separate variant, not a - /// flag, because only comparisons take `bool`. - CompareBool(CompareOpKind), - /// PromQL vector-set operation. - Set(PromQLVectorSetOpKind), -} - -impl std::fmt::Display for BinaryOpKind { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - BinaryOpKind::Arithmetic(op) => write!(f, "{op}"), - BinaryOpKind::Compare(op) => write!(f, "{op}"), - BinaryOpKind::CompareBool(op) => write!(f, "{op} bool"), - BinaryOpKind::Set(PromQLVectorSetOpKind::And) => f.write_str("AND"), - BinaryOpKind::Set(PromQLVectorSetOpKind::Or) => f.write_str("OR"), - BinaryOpKind::Set(PromQLVectorSetOpKind::Unless) => f.write_str("unless"), - } - } -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum JoinKind { - Inner, - Left, - Right, - Full, - Cross, - /// Left semi-join — each left row that has **at least one** match, once. - /// `WHERE c IN (SELECT …)` / `WHERE EXISTS (…)` (issue #111). - /// - /// Output schema is the **left's alone**; the right side is a filter, not a - /// source of columns. The join predicate still resolves against the - /// concatenated `left ++ right` schema — its scope is deliberately wider - /// than the node's output. - Semi, - /// Left anti-join — each left row with **no** match. `WHERE NOT EXISTS (…)`. - /// Same schema rule as [`JoinKind::Semi`]. - /// - /// Note this is *not* `NOT IN (SELECT …)`: under SQL's three-valued logic a - /// NULL on the right makes `NOT IN` yield no rows at all, where an anti-join - /// yields every left row. The SQL front end rejects `NOT IN (subquery)` - /// rather than lower it here. - Anti, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum RelationalSetOpKind { - Union, - Intersect, - Except, -} - -/// PromQL vector-set operator used by [`BinaryOpKind::Set`]. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum PromQLVectorSetOpKind { - And, - Or, - Unless, -} - -/// SQL analytic window function (`fn(...) OVER (…)`). Distinct from a streaming -/// time `Window`: this is an analytic frame over already-materialised rows. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum WindowFuncKind { - RowNumber, - Rank, - DenseRank, - Lag, - Lead, - /// ClickHouse `lagInFrame`/`leadInFrame`: unlike [`Lag`](Self::Lag)/[`Lead`](Self::Lead), - /// these respect the window frame bounds (NULL/default past the frame edge) - /// rather than reaching arbitrarily far back/forward. Kept as distinct - /// variants so the frame clause is never silently discarded by conflating - /// them with `Lag`/`Lead` (#267). `WindowFuncKind` still has no frame - /// representation, so today these lower and behave exactly like - /// `Lag`/`Lead` — the tag is correct, the frame-respecting behavior isn't - /// implemented yet. See #231 for modeling window frames properly. - LagInFrame, - LeadInFrame, - FirstValue, - LastValue, - /// `NTH_VALUE(expr, n)` — `n` is resolved from the (literal) 2nd argument. - NthValue(Option), - Sum, - Avg, - Count, - Min, - Max, -} - -/// A window's frame-spec (`ROWS`/`RANGE BETWEEN … AND …`) — which rows around -/// the current one an analytic window function reads. `GROUPS` is rejected at -/// lowering time (issue #268): every SQL corpus in this repo uses only `ROWS`, -/// and nothing downstream interprets frame semantics yet, so it isn't worth -/// modelling untested. -/// -/// Meaningless (but harmless) on the rank-only and navigation functions -/// (`ROW_NUMBER`/`RANK`/`DENSE_RANK`/`LAG`/`LEAD`), which ignore the frame per -/// SQL semantics — DataFusion still attaches one, stored here verbatim. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct WindowFrame { - pub units: WindowFrameUnits, - pub start_bound: WindowFrameBound, - pub end_bound: WindowFrameBound, -} - -/// A finite window-frame displacement. Intervals are normalized to Arrow's -/// month/day/nanosecond representation so SQL `RANGE INTERVAL ...` bounds -/// survive lowering without leaking DataFusion types into the canonical IR. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum WindowFrameOffset { - Scalar(ScalarValue), - Interval { - months: i32, - days: i32, - nanoseconds: i64, - }, -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum WindowFrameUnits { - /// Boundaries count physical rows: `ROWS BETWEEN 2 PRECEDING AND CURRENT ROW`. - Rows, - /// Boundaries count by value-distance on the (single) `ORDER BY` column: - /// `RANGE BETWEEN INTERVAL '1' HOUR PRECEDING AND CURRENT ROW`. - Range, -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum WindowFrameBound { - /// `UNBOUNDED PRECEDING` is - /// `Preceding(WindowFrameOffset::Scalar(ScalarValue::Null))`. - Preceding(WindowFrameOffset), - CurrentRow, - /// `UNBOUNDED FOLLOWING` is - /// `Following(WindowFrameOffset::Scalar(ScalarValue::Null))`. - Following(WindowFrameOffset), -} - -/// A symbolic label matcher on the **info metric** side of an -/// [`QueryExpr::PromqlInfoEnrich`] (issue #84). Unlike a `Scan` predicate it is not -/// resolved positionally — it references the info metric's labels (`__name__` -/// picks the metric, the rest constrain data labels), which aren't in the input -/// vector's schema; the post-ASAP realization pass applies it against the info metric. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct InfoMatcher { - pub label: String, - /// One of `Eq` / `Ne` / `Regex` / `NotRegex` (PromQL `=`/`!=`/`=~`/`!~`). - pub op: CompareOpKind, - pub value: String, -} - -/// Series-sampling selection mode (PromQL `limitk` / `limit_ratio`, issue #86). -/// A [`QueryExpr::PromqlSeriesSample`] keeps a *subset of whole series*, unchanged — it does -/// not rank or reduce, so it is distinct from `TopK` and from `Sort → Limit`. -#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum SampleKind { - /// `limitk(k, v)` — up to `k` series per group. Which series survive is - /// deterministic across evaluations but otherwise unspecified (no ordering). - LimitK(usize), - /// `limit_ratio(r, v)` — a deterministic `r`-fraction of series per group. - /// `r ∈ [-1, 1]`; a negative `r` selects the complementary fraction. - LimitRatio(f64), -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct SortKey { - pub expr: QueryExpr, - pub ascending: bool, - pub nulls_first: bool, -} - -/// PromQL vector-match modifier (`on`/`ignoring` + `group_left`/`group_right`). -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct VectorMatch { - pub kind: VectorMatchKind, - pub labels: Vec, - pub grouping: Option, -} - -/// PromQL `@` modifier — pins a selector's evaluation time to an anchor instead -/// of the query evaluation time (issue #40). -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -pub enum AtModifier { - /// `@ start()` — the query range's start instant. - Start, - /// `@ end()` — the query range's end instant. - End, - /// `@ ` — an absolute instant, milliseconds since the Unix epoch (may be - /// negative). PromQL writes the timestamp in seconds; the front end scales it. - Timestamp(i64), -} - -/// PromQL per-selector **time-shift** modifiers — `offset` and `@` (issue #40). -/// Neither changes a selector's *schema*; both move *when* it is evaluated, so -/// the shift is a pass-through wrapper ([`QueryExpr::TimeShift`]) over the -/// selector rather than a new leaf shape. The runtime resolves the anchor and -/// applies the offset. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)] -pub struct TimeShift { - /// `offset ` as signed milliseconds — a positive value shifts the - /// lookback *back* in time (`offset 5m`), a negative value shifts it - /// *forward* (`offset -5m`). `0` = no offset. - pub offset_ms: i64, - /// `@` anchor; `None` = evaluate at the query time. - pub at: Option, -} - -impl TimeShift { - /// Whether this shift is the identity (no `offset`, no `@`) — the state of - /// every selector that carries neither modifier. - pub fn is_identity(&self) -> bool { - self.offset_ms == 0 && self.at.is_none() - } -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum VectorMatchKind { - On, - Ignoring, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct VectorGrouping { - pub side: GroupSide, - pub labels: Vec, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum GroupSide { - Left, - Right, -} - -/// A row-level filter predicate (WHERE clause / PromQL label matcher). -/// Boxed: `Predicate` sits directly (not behind a `Vec`) in -/// `Filter.pred`/`Join.pred`/`Aggregate.having`, and `QueryExpr` is -/// self-recursive without further indirection once the scalar variants are -/// part of it — the box is what makes the recursive type's size finite there. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct Predicate(pub Rc>); - -/// Whether any entry of an `Aggregate.filters` vector is set — the shape -/// no binding rule accepts yet (issue #466): a filtered measure stays -/// `KeepPreAsap`, and heavy-hitter promotion skips it. -pub fn any_measure_filtered(filters: &[Option>]) -> bool { - filters.iter().any(Option::is_some) -} - -/// One item in a SELECT projection list. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct ProjectItem { - pub alias: Option, - pub expr: QueryExpr, -} - -// ── Intent algebra IR ──────────────────────────────────────────────────────── - -/// What kind of computation an `Aggregate` node performs — orthogonal to -/// *which* columns it groups by (that's still [`GroupKeys`], inside -/// `Reduce`). Explicit, decided once by whichever pass constructs the node -/// (structural, at front-end lowering time), rather than inferred downstream from -/// whether a grouping-key list happens to be empty or from a neighboring -/// node's shape. See design proposal #165. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub enum Reduction { - /// Collapses input rows via `by` — `by`/`without` semantics are exactly - /// [`GroupKeys`]'s. May still collapse every row into one (an empty, - /// non-`without` `by`) — that's a genuine reduction with zero grouping - /// columns, not "no grouping concept." - Reduce(GroupKeys), - /// No grouping concept at all: preserves one output row per input - /// entity (e.g. a per-series windowed computation with no `by(...)` - /// clause to begin with, because there's no aggregation operator here - /// for such a clause to attach to). Never merges across entities, and - /// never collapses an entity's own row structure (e.g. a time axis) — - /// unlike `Reduce(GroupKeys::without(vec![]))` ("group by every - /// label"), which is still a genuine reduction and does collapse it. - PerEntity, -} - -impl Reduction { - /// Shorthand for the common case — group by these (possibly empty) - /// keys, kept rather than excluded. - pub fn by(keys: Vec) -> Self { - Self::Reduce(GroupKeys::by(keys)) - } - - /// The grouping keys, if this is a genuine reduction — `None` for - /// `PerEntity`, which has no grouping-keys concept to report. - pub fn group_keys(&self) -> Option<&GroupKeys> { - match self { - Self::Reduce(by) => Some(by), - Self::PerEntity => None, - } - } - - /// The grouping keys, panicking if this is `PerEntity` — for call sites - /// (tests, mostly) that already know, from the shape they built or are - /// asserting on, that this must be a genuine reduction. Prefer - /// [`group_keys`](Self::group_keys) wherever the caller can't assume that. - pub fn expect_reduce(&self) -> &GroupKeys { - match self { - Self::Reduce(by) => by, - Self::PerEntity => panic!("expected Reduction::Reduce, got PerEntity"), - } - } -} - -/// A caller-proven compound unique key for a [`QueryExpr::Concat`] (issue -/// #228) — built only via [`QueryExpr::concat_with_discriminator`] / -/// [`ConcatDiscriminatorKey::new`], never by naming `discriminator` directly -/// in a struct literal (both fields are private): from *other Rust code*, -/// the only way to end up with one of these is to hand over a specific -/// column as the discriminator, by name, at the call site. -/// -/// Caveat: this is a Rust-API-level guarantee, not a data-level one. The -/// derived `Deserialize` impl below builds a `ConcatDiscriminatorKey` -/// directly from field values, bypassing `new()`. Deserialization is therefore -/// equivalent to a caller supplying the assertion directly; it does not prove -/// either fact below. An external boundary accepting `QueryExpr` data must -/// reject this field or validate both obligations before treating it as -/// uniqueness evidence. -/// -/// # Soundness -/// -/// `Concat`'s default (see its own doc) is to drop `unique_keys` -/// unconditionally, because a key unique **within** one branch is not unique -/// **across** the concatenation unless the branches' value sets for that key -/// are provably disjoint — nothing about matching schemas or matching -/// per-branch keys establishes that on its own. Two different branches can -/// trivially emit the same `inner_key` value (e.g. two PromQL -/// `histogram_quantiles` branches keyed on `(host, le)` can both produce a -/// `(host, le)` pair for different φ). -/// -/// Prepending `discriminator` restores a compound key only when two facts -/// hold: `inner_key` uniquely identifies rows **within every branch**, and -/// `discriminator`'s value is **guaranteed to differ between branches** — a -/// literal the producer just tagged the branch with (PromQL φ riding along via -/// [`QueryExpr::PromqlRelabel`], a Postgres-style synthetic `GROUPING()` id -/// for `ROLLUP`/`CUBE`, …), never something inferred structurally from the -/// branches' own data — then `discriminator` alone partitions rows into -/// disjoint sets independent of what the branches actually contain, so -/// `(discriminator, inner_key)` is sound even when otherwise-identical -/// `inner_key` values occur in different branches. Neither fact is verified -/// here; both are part of the caller-proven claim. -/// -/// This is a **caller-proven claim, not something `Concat` can verify**: -/// nothing stops a caller from asserting a discriminator that in fact -/// repeats across branches, in which case the resulting `unique_keys` claim -/// is simply wrong — `output_schema` trusts it without checking. The -/// obligation is on the constructor call site, exactly as it is on -/// [`QueryExpr::Dedup`]'s `cols` or any other unverified `unique_keys` -/// producer in this module. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -#[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct ConcatDiscriminatorKey { - discriminator: C, - inner_key: Vec, -} - -impl ConcatDiscriminatorKey { - /// The only constructor — `discriminator` must be named explicitly by - /// the caller. See the type's doc for the soundness obligation this - /// puts on that caller. - pub fn new(discriminator: C, inner_key: Vec) -> Self { - Self { - discriminator, - inner_key, - } - } - - pub fn discriminator(&self) -> &C { - &self.discriminator - } - - pub fn inner_key(&self) -> &[C] { - &self.inner_key - } -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub enum QueryExpr { - /// Outermost leaf. `schema` is the **binding schema** — the resolved column - /// set every positional `ColumnId` in the tree indexes into, *not* a full - /// description of the runtime row — once bound (`schema: Schema`, always - /// present: the [`SchemaResolver`](super::schema_resolver) is total). Before binding, a - /// front-end-emitted `Scan` (`C = ColumnRef`) knows it only when the front - /// end already has it without binding — a catalog-backed SQL leaf — `None` - /// (PromQL) defers to the SchemaResolver; see [`ColState::ScanSchema`]. Complete - /// when catalog-backed (SQL); for schemaless PromQL the bound schema is - /// usage-derived (the `(ts, value)` floor + the labels the query - /// references), since a metric's label set is open and known only at - /// runtime. That distinction is carried explicitly by - /// [`Schema::closed`](super::schema::Schema::closed) (SQL leaf → `true`, - /// PromQL leaf → `false`). `predicates` are leaf-level row filters (PromQL - /// label matchers, pushed-down `WHERE` conjuncts). - Scan { - source: Source, - #[serde(default)] - predicates: Vec>, - schema: C::ScanSchema, - }, - /// A scalar sub-expression sitting in an **operator-tree position** — a - /// [`BinaryOp`](Self::BinaryOp) operand for ` op ` - /// thresholds / unit conversions (#35), a - /// [`PromqlVectorFromScalar`](Self::PromqlVectorFromScalar) child, or a - /// whole query's root (a bare PromQL scalar query, e.g. `5`). - /// - /// Formerly its own leaf variant, `PromqlScalar(f64)`. Issue #220: that - /// variant held exactly the same value [`Literal`](Self::Literal) does - /// (every PromQL scalar is `f64`), duplicating it for no reason but - /// *which tree position* it was allowed to appear in. This wrapper - /// carries that position instead of the value — the inner node is an - /// ordinary scalar sub-language expression (in practice always - /// `Literal(ScalarValue::Float64(_))`, since a front end only ever - /// constructs this fully constant-folded — see - /// [`promql_scalar`](Self::promql_scalar)) — and is what `output_schema`, - /// `canonicalize`, and `resolve` now key off to tell "this operand has - /// its own row schema" from "this is a nested scalar leaf with none," - /// in place of the old `PromqlScalar` vs. `Literal` variant tag. - PromqlScalarBridge(Rc>), - - /// The query **evaluation timestamp** as Unix seconds, exposed by PromQL - /// `time()`. This is not inherently the current wall-clock time: its value - /// is the instant or range-step at which the expression is evaluated. It - /// is also the implicit input of no-argument calendar functions. Issue #46. - EvalTimestamp, - - /// The SQL statement evaluation time (`NOW()` / `CURRENT_TIMESTAMP`) as - /// a SQL [`DataType::Timestamp`]. Kept distinct from [`EvalTimestamp`], - /// whose PromQL `time()` contract is Unix seconds as `Float64`. - CurrentTimestamp, - - /// PromQL `vector(s)` — the scalar→instant-vector bridge. Promotes a - /// scalar-typed child to a single label-less series carrying the scalar's - /// value at every step. Lets a scalar participate where a vector is required - /// (`up or vector(0)` dead-man's-switch). Issue #48. - PromqlVectorFromScalar(Rc>), - - /// PromQL `scalar(v)` — the instant-vector→scalar bridge. Collapses a - /// single-element vector to its value (NaN at runtime if the input is not - /// exactly one series). Lets a vector feed a scalar position (`vector` / - /// aggregation `k` args, thresholds). Issue #48. - PromqlScalarFromVector(Rc>), - - /// ρ — a per-series **label rewrite** (PromQL `label_replace` / - /// `label_join`). Every input row passes through unchanged except for the - /// destination label `dst`, whose new value is computed by `value` — a - /// scalar expression over the child's (source) label columns: - /// `label_replace` → a `label_replace(src, regex, replacement)` function - /// call (regex capture-expansion), `label_join` → a `label_join(sep, srcs…)` - /// concatenation. Sample values and the time axis are untouched. Issue #50. - PromqlRelabel { - /// The label written by this rewrite (PromQL `dst_label`). - dst: String, - value: Rc>, - child: Rc>, - }, - - /// PromQL `info(v, [selector])` — left-join **label enrichment** (#84). Each - /// series in `child` is enriched with labels from the matching info metric(s) - /// (`target_info` by default; `selector`'s `__name__` matchers pick the - /// metric(s), the rest constrain the data labels), joined on their shared - /// identifying labels. Those join keys are the info metric's identifying - /// labels — runtime/metadata-resolved, since an open PromQL schema can't - /// enumerate them — so they are NOT carried here; the post-ASAP realization pass - /// resolves them from the info metric's schema. The output keeps - /// `child`'s (open) schema: the - /// grafted labels appear at runtime. - PromqlInfoEnrich { - #[serde(default)] - selector: Vec, - child: Rc>, - }, - - /// Series-sampling **selection** — PromQL `limitk` / `limit_ratio` (#86). - /// Keeps a subset of whole series per `by` group (empty = global), passing - /// each surviving series through unchanged. Not a ranking (`TopK`) and not a - /// reduction: the output schema equals the child's. - PromqlSeriesSample { - #[serde(default)] - by: GroupKeys, - kind: SampleKind, - child: Rc>, - }, - - /// σ — row-level filter. Output schema = child schema. - Filter { - pred: Predicate, - child: Rc>, - }, - /// π — column projection. - Project { - cols: Vec>, - /// Re-qualifies every output column with this table alias (a derived - /// table / inline view). `None` for an ordinary SELECT list. - #[serde(default)] - qualifier: Option, - child: Rc>, - }, - - /// γ + α — GROUP BY (positional) + aggregate intents. - Aggregate { - reduction: Reduction, - measures: Vec>, - /// Output column names parallel to `measures`. A non-empty entry overrides - /// the synthetic intent-keyed name — SQL threads DataFusion's generated - /// name (e.g. `"sum(metrics.bytes)"`) here so an enclosing `Project` - /// resolves the aggregate output by the name it references. An empty - /// entry (or empty vec) falls back to `AggIntent::output_column`'s name - /// (PromQL's convention). - #[serde(default)] - output_names: Vec, - /// Per-measure row predicates, parallel to `measures` — SQL - /// `FILTER (WHERE …)` semantics (issue #466): only rows where - /// `filters[i]` is `TRUE` update `measures[i]`; groups are still - /// formed from every row. Positional against `child`'s output - /// schema, like `Filter.pred` — not against this node's output like - /// `having`. `None` (or an entry past the end of a shorter vec) is - /// an unfiltered measure, so an empty vec is the pre-#466 shape. - #[serde(default)] - filters: Vec>>, - #[serde(default)] - having: Option>, - child: Rc>, - }, - - /// δ — SQL `DISTINCT` / row deduplication. Positional like every other - /// column reference here; empty = dedup on all columns (`SELECT DISTINCT *`). - Dedup { - cols: Vec, - child: Rc>, - }, - /// ⊕ — exact, n-ary `UNION ALL` of independent branches. Rows are - /// concatenated, never deduplicated; SQL's `UNION`/`INTERSECT`/`EXCEPT` are - /// [`QueryExpr::SetOp`], not this. - /// - /// Used for the branches of one query that a single `Aggregate` cannot - /// express — PromQL `histogram_quantiles` (one branch per φ, issue #109) and - /// SQL `ROLLUP`/`CUBE`/`GROUPING SETS` (one branch per grouping level, issue - /// #118) — as well as for sharded / fan-in plans. - /// - /// **The branches must be union-compatible; nothing here enforces it.** The - /// output schema is the *first* child's, so branches that disagree on a - /// column name or type leave the merged schema silently misdescribing every - /// branch but one. A producer that cannot guarantee compatibility must - /// project the branches into a common shape first. - /// - /// A row may appear in several branches, so no branch's unique key survives - /// the union — `unique_keys` is dropped, as in `SetOp`. **Unless** the - /// constructor asserted `discriminator_unique_key` (issue #228, - /// [`QueryExpr::concat_with_discriminator`]): a caller-proven claim that - /// one column's value is guaranteed distinct per branch, which makes - /// `(discriminator, inner_key)` a sound compound unique key regardless of - /// whether `inner_key` alone repeats across branches. `None` — every - /// ordinary construction path, including the plain struct literal and - /// [`QueryExpr::concat`] — reproduces the old, unconditional-drop - /// behavior exactly; see [`ConcatDiscriminatorKey`]'s doc for the - /// soundness argument and the obligation this puts on whoever asserts it. - /// - /// Empty children is an error ([`QueryExprError::EmptyConcat`]), not an - /// empty relation: there would be no schema to derive. - Concat { - children: Vec>, - /// See the field-level doc above and [`ConcatDiscriminatorKey`]. - #[serde(default)] - discriminator_unique_key: Option>, - }, - - /// Logical join. Post-ASAP binding picks the physical alternative. - Join { - kind: JoinKind, - pred: Predicate, - left: Rc>, - right: Rc>, - }, - SetOp { - kind: RelationalSetOpKind, - all: bool, - left: Rc>, - right: Rc>, - }, - - /// Generic order-by for non-heavy-hitter cases. - /// - /// `partition_by` makes the ordering **per-group**: a non-empty set means - /// "rank within each `partition_by` group" — the semantics behind PromQL - /// `topk by (host) (…)` / SQL `… OVER (PARTITION BY host ORDER BY …)`. It is - /// row-preserving (schema pass-through) and is where the grouping of a - /// generic (non-heavy-hitter) ranking lives, so there is no separate - /// `Partition` node (issue #12: reducing GROUP BY → `Aggregate.by`, per-group - /// ranking → here, parallel sharding → a deployment's own physical - /// stage). Empty = a global order-by. - Sort { - keys: Vec>, - #[serde(default)] - partition_by: GroupKeys, - child: Rc>, - }, - Limit { - n: usize, - offset: usize, - child: Rc>, - }, - - /// PromQL sub-query (`[range:resolution]`). Logical pass-through. - PromqlSubquery { - range: Duration, - #[serde(default)] - resolution: Option, - child: Rc>, - }, - - /// Temporal range selection — "look back `range` of history for this - /// computation." Used for all range-vector functions: `rate`, `increase`, - /// `*_over_time`. The range is distinct from a row-level `Filter`. - /// - /// Structural marker: an `Aggregate` whose direct child is a `TimeRange` - /// is a *per-series* reduction (label-preserving); one whose child is a - /// plain `Scan` or another `Aggregate` is a *cross-series* reduction. - TimeRange { - range: Duration, - child: Rc>, - }, - - /// PromQL `offset` / `@` **time shift** on a selector (issue #40). A - /// pass-through wrapper: it moves *when* `child` is evaluated (the runtime - /// resolves the `@` anchor and applies the offset) but leaves its schema - /// unchanged. Wraps the shifted selector directly — `m offset 1h` → - /// `TimeShift { Scan }`; a ranged selector `m[5m] offset 1h` → - /// `TimeRange { 5m, TimeShift { Scan } }` (the range is taken at the shifted - /// time). A shifted subquery wraps the `PromqlSubquery`, moving its step - /// grid. Never carries the identity shift (the converter emits a bare - /// selector when neither modifier is present). - TimeShift { - shift: TimeShift, - child: Rc>, - }, - - /// SQL analytic window function: `func(args) OVER (PARTITION BY … ORDER BY … - /// ROWS/RANGE BETWEEN …)`. Output schema = child schema + one column named - /// `output_name` (the name the enclosing `Project` references). - SQLWindowFunc { - func: WindowFuncKind, - /// Operand expressions (`LAG(value)` → `[Column(value_id)]`); empty for - /// the rank-only functions (`ROW_NUMBER`/`RANK`/`DENSE_RANK`). - args: Vec>, - partition_by: GroupKeys, - order_by: Vec>, - /// `None` is accepted only for backward compatibility with serialized - /// pre-#268 IR, where the engine's implicit frame was not retained. - /// Newly lowered SQL always carries `Some` with DataFusion's resolved - /// concrete default or explicit frame. - #[serde(default)] - frame: Option, - /// The output column's name — DataFusion's window-expr field name, so a - /// `Project` above resolves it (cf. `Aggregate.output_names`). - output_name: String, - child: Rc>, - }, - - /// Arithmetic / comparison / boolean composition (PromQL binary ops). - BinaryOp { - op: BinaryOpKind, - lhs: Rc>, - rhs: Rc>, - #[serde(default)] - vector_match: Option, - }, - - // ── Scalar expression shapes (issue #205) ─────────────────────────── - // - // Formerly a separate, self-recursive `Expr` tree, reachable from the - // operator variants above only through wrapper fields (`Predicate`, - // `ProjectItem`, `SortKey`). They're variants of this same tree now — a - // scalar sub-expression is only ever reachable through one of those same - // wrapper positions (`Filter.pred`, `ProjectItem.expr`, `Aggregate.having`, - // `PromqlRelabel.value`, `SQLWindowFunc.args`, …), which is a *convention* this - // type no longer enforces at compile time the way the old, closed - // `Expr` variant set did — nothing stops constructing, say, a `Scan` - // where a `Compare`'s `left` operand belongs. `output_schema` and every - // scalar-position consumer (`resolve`, `canonicalize`, `infer_expr_type`) - // reject a non-scalar variant found there instead (a `QueryExprError` or - // an `unreachable!`, depending on the call site) — the accepted - // replacement, since the alternative (a marker-trait/sub-enum bound - // restricting which variants are constructible in a scalar position) adds - // real type-level machinery for a distinction every constructor already - // has to get right structurally anyway (a `Filter` is never built with an - // operator subtree as its `pred`). - /// A column reference — unresolved [`ColumnRef`] (front-end-emitted, `C = - /// ColumnRef`) or positional [`ColumnId`] (once bound, `C = ColumnId`). - Column(C), - /// A constant literal value. - Literal(ScalarValue), - /// `left op right` — binary comparison. - Compare { - left: Rc>, - op: CompareOpKind, - right: Rc>, - }, - /// Flat conjunction (logical AND). An empty list is vacuously true. - BoolAnd(Vec>), - /// Flat disjunction (logical OR). An empty list is vacuously false. - BoolOr(Vec>), - /// Logical NOT. - Not(Rc>), - /// `expr IS NULL`. - IsNull(Rc>), - /// `expr IS NOT NULL`. - IsNotNull(Rc>), - /// `CAST(expr AS to)`; `try_cast` for SQL `TRY_CAST` (NULL on failure). - Cast { - expr: Rc>, - to: DataType, - try_cast: bool, - }, - /// `expr [NOT] IN (v1, v2, …)`. - InList { - expr: Rc>, - list: Vec>, - negated: bool, - }, - /// Scalar function call, e.g. `LOWER(col)`, `ABS(x)`. - FunctionCall { - name: String, - args: Vec>, - }, - /// Binary arithmetic: `left op right`. - Arithmetic { - op: ArithmeticOpKind, - left: Rc>, - right: Rc>, - }, - /// SQL `CASE` (both searched and simple forms). `operand` present for the - /// simple form (`CASE expr WHEN …`), absent for searched. - Case { - operand: Option>>, - branches: Vec<(QueryExpr, QueryExpr)>, - else_expr: Option>>, - }, -} - -impl QueryExpr { - /// Construct the [`PromqlScalarBridge`](Self::PromqlScalarBridge) leaf - /// for a bare PromQL numeric literal / folded constant scalar (issue - /// #220) — `Literal(ScalarValue::Float64(v))` at an operator-tree - /// position. The one constructor every front end / test that used to - /// write `QueryExpr::PromqlScalar(v)` should use instead. - pub fn promql_scalar(v: f64) -> Self { - QueryExpr::PromqlScalarBridge(Rc::new(QueryExpr::Literal(ScalarValue::Float64(v)))) - } - - /// Build an ordinary [`Concat`](Self::Concat) — the ordinary/default - /// construction path every call site should prefer over the bare struct - /// literal: `output_schema` drops `unique_keys` unconditionally, exactly - /// as before issue #228. Use - /// [`concat_with_discriminator`](Self::concat_with_discriminator) instead - /// when the caller can prove branch disjointness via a discriminator - /// column. - pub fn concat(children: Vec>) -> Self { - QueryExpr::Concat { - children, - discriminator_unique_key: None, - } - } - - /// Build a [`Concat`](Self::Concat) whose output schema carries the - /// caller-proven compound unique key `(discriminator, inner_key)` (issue - /// #228). See [`ConcatDiscriminatorKey`]'s doc for the soundness - /// argument and the obligation this puts on the caller — - /// `output_schema` trusts this claim without verifying it: nothing here - /// checks that `inner_key` is unique within every branch or that - /// `discriminator`'s value is distinct between branches. - pub fn concat_with_discriminator( - children: Vec>, - discriminator: C, - inner_key: Vec, - ) -> Self { - QueryExpr::Concat { - children, - discriminator_unique_key: Some(ConcatDiscriminatorKey::new(discriminator, inner_key)), - } - } - - /// The value of a [`PromqlScalarBridge`](Self::PromqlScalarBridge) leaf - /// wrapping a plain `Literal(ScalarValue::Float64(_))` — every one a - /// front end constructs today (see [`promql_scalar`](Self::promql_scalar)). - /// `None` for any other shape, including a `PromqlScalarBridge` wrapping - /// something else (not constructed today, but not precluded by the type). - pub fn as_promql_scalar(&self) -> Option { - match self { - QueryExpr::PromqlScalarBridge(inner) => match inner.as_ref() { - QueryExpr::Literal(ScalarValue::Float64(v)) => Some(*v), - _ => None, - }, - _ => None, - } - } - - /// If this expression is a `BoolAnd`, return its elements; otherwise a - /// single-element slice containing `self`. - pub fn conjuncts(&self) -> &[QueryExpr] { - match self { - QueryExpr::BoolAnd(v) => v.as_slice(), - _ => std::slice::from_ref(self), - } - } - - /// If this expression is a `BoolOr`, return its elements; otherwise a - /// single-element slice containing `self`. - pub fn disjuncts(&self) -> &[QueryExpr] { - match self { - QueryExpr::BoolOr(v) => v.as_slice(), - _ => std::slice::from_ref(self), - } - } - - /// Recursively collect every column reference in a **scalar** subtree — - /// used by the [`SchemaResolver`](super::schema_resolver::SchemaResolver) to seed usage-derived - /// leaf schemas, and available to post-ASAP binding for column-lineage / - /// selectivity. - /// `self` must be one of the scalar variants (see the module doc on - /// [`QueryExpr`]'s scalar shapes) — every caller already only reaches - /// this through a scalar-typed position (`Predicate`, `ProjectItem.expr`, - /// …), so an operator variant here indicates a construction bug, not a - /// shape this needs to handle silently. - pub fn columns_referenced(&self) -> Vec<&C> { - match self { - QueryExpr::Column(c) => vec![c], - QueryExpr::Literal(_) => vec![], - QueryExpr::EvalTimestamp => vec![], - QueryExpr::CurrentTimestamp => vec![], - QueryExpr::Compare { left, right, .. } | QueryExpr::Arithmetic { left, right, .. } => { - let mut v = left.columns_referenced(); - v.extend(right.columns_referenced()); - v - } - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { - parts.iter().flat_map(|e| e.columns_referenced()).collect() - } - QueryExpr::Not(e) | QueryExpr::IsNull(e) | QueryExpr::IsNotNull(e) => { - e.columns_referenced() - } - QueryExpr::Cast { expr, .. } => expr.columns_referenced(), - QueryExpr::InList { expr, list, .. } => { - let mut v = expr.columns_referenced(); - v.extend(list.iter().flat_map(|e| e.columns_referenced())); - v - } - QueryExpr::FunctionCall { args, .. } => { - args.iter().flat_map(|e| e.columns_referenced()).collect() - } - QueryExpr::Case { - operand, - branches, - else_expr, - } => { - let mut v = vec![]; - if let Some(op) = operand { - v.extend(op.columns_referenced()); - } - for (when, then) in branches { - v.extend(when.columns_referenced()); - v.extend(then.columns_referenced()); - } - if let Some(e) = else_expr { - v.extend(e.columns_referenced()); - } - v - } - other => unreachable!( - "columns_referenced called on a non-scalar QueryExpr variant: {other:?}" - ), - } - } -} - -/// The canonical, positional, resolved tree — what the bare `QueryExpr` name -/// has always meant (the default `C = ColumnId`). Every existing consumer -/// keeps using `QueryExpr` unparameterized; this alias exists only to name -/// the resolved state explicitly at a use site that also wants to name -/// [`UnresolvedQueryExpr`] nearby. -pub type ResolvedQueryExpr = QueryExpr; - -/// The front-end-emitted, name-based, unresolved tree — -/// `QueryExpr`: front ends construct this directly during their -/// own `interpret` step (issue #179), and the [`SchemaResolver`](super::schema_resolver) -/// resolves it into [`ResolvedQueryExpr`]. -pub type UnresolvedQueryExpr = QueryExpr; - -// `output_schema` needs a fully bound tree — it reads `Scan.schema` as a plain -// `Schema` and resolves every scalar `Expr::Column` positionally — so it lives -// only on the resolved instantiation, not `impl QueryExpr`. -// Same reasoning as `AggIntent`'s `output_column`/`requires`/`is_per_series` -// (#205): a schema-shaped property that is only meaningful post-binding. -impl QueryExpr { - /// Infer a scalar expression against its input relation using the same - /// canonical rules as projection schema derivation. - pub fn scalar_type(&self, input: &Schema) -> Result<(DataType, bool), QueryExprError> { - infer_expr_type(self, input) - } - - /// Output schema of the root of a canonical tree. - pub fn output_schema(&self) -> Result { - match self { - QueryExpr::Scan { schema, .. } => Ok(schema.clone()), - - QueryExpr::Aggregate { - reduction, - measures, - output_names, - child, - .. - } => { - let in_schema = child.output_schema()?; - aggregate_output_schema(&in_schema, reduction, measures, output_names) - } - - QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - // Series sampling keeps a subset of whole series unchanged, so the - // output schema (and row-uniqueness) is exactly the child's (#86). - | QueryExpr::PromqlSeriesSample { child, .. } - // Info enrichment adds runtime info labels — the statically-known - // schema is the child's (open), so it passes through (#84). - | QueryExpr::PromqlInfoEnrich { child, .. } - | QueryExpr::TimeRange { child, .. } - // A time shift (`offset`/`@`) moves *when* the child is evaluated, - // never its columns — schema passes through (#40). - | QueryExpr::TimeShift { child, .. } => child.output_schema(), - - // ρ — relabel preserves every input column and writes one label - // `dst` (Utf8): overwritten in place if it already exists, else - // appended (nullable — a `label_replace` regex non-match leaves it - // unset). The schema stays open (other labels remain runtime-only). - // A rewrite can collapse two label sets into one, so row-uniqueness - // is no longer provable — drop unique_keys. - QueryExpr::PromqlRelabel { dst, child, .. } => { - let mut out = child.output_schema()?; - if let Some(existing) = out.columns.iter_mut().find(|c| c.name == *dst) { - existing.dtype = DataType::Utf8; - existing.nullable = true; - } else { - out.columns.push(Column::new(dst.clone(), DataType::Utf8, true)); - } - out.unique_keys.clear(); - Ok(out) - } - - // π — one output column per projection item. Each item's type is - // inferred from its expression against the child schema; the name - // is the explicit alias or a derived default. A child unique key - // survives exactly when every one of its columns is passed through - // as a bare `Column` item (possibly reordered or aliased). Derived - // expressions cannot carry key identity. `time_index` is re-found - // by name. - QueryExpr::Project { cols, qualifier, child } => { - let in_schema = child.output_schema()?; - let columns: Vec = cols - .iter() - .enumerate() - .map(|(i, item)| { - let (dtype, nullable) = infer_expr_type(&item.expr, &in_schema)?; - let name = item - .alias - .clone() - .unwrap_or_else(|| default_proj_name(&item.expr, i, &in_schema)); - let c = Column::new(name, dtype, nullable); - // A derived table re-qualifies its output columns with - // its alias, so `t.col` (and a join over two derived - // tables) resolves to the right relation. - Ok(match qualifier { - Some(q) => c.with_table(q), - None => c, - }) - }) - .collect::, QueryExprError>>()?; - let time_index = columns.iter().position(|c| c.name == "ts"); - let unique_keys = in_schema - .unique_keys - .iter() - .filter_map(|key| { - key.iter() - .map(|input_col| { - cols.iter().position(|item| { - matches!(&item.expr, QueryExpr::Column(col) if col == input_col) - }) - }) - .collect::>>() - }) - .collect(); - Ok(Schema { - columns, - time_index, - unique_keys, - // Projection enumerates exactly its items → closed. - closed: true, - }) - } - - QueryExpr::Dedup { cols, child } => { - let mut out = child.output_schema()?; - // Deduplicating on `cols` makes them a unique key of the result. - if !cols.is_empty() { - out.add_unique_key(cols.clone()); - } - Ok(out) - } - - // ⊕ — the branches are union-compatible by construction, so the - // output shape is the first child's. A row can appear in more than - // one branch, so no key of one branch is a key of the union: drop - // unique_keys, exactly as `SetOp` does — unless the constructor - // asserted `discriminator_unique_key` (issue #228), in which case - // `(discriminator, inner_key)` becomes the sole unique key. That - // assertion is trusted verbatim here, never checked: see - // `ConcatDiscriminatorKey`'s doc for the soundness argument and - // whose obligation it is. - QueryExpr::Concat { - children, - discriminator_unique_key, - } => { - let mut s = children - .first() - .ok_or(QueryExprError::EmptyConcat) - .and_then(|c| c.output_schema())?; - s.unique_keys.clear(); - if let Some(key) = discriminator_unique_key { - let mut compound = vec![*key.discriminator()]; - compound.extend(key.inner_key().iter().copied()); - s.add_unique_key(compound); - } - Ok(s) - } - // Set operations are union-compatible: both sides share the left's - // column shape, so the output schema is the left's. (Row identity - // is not preserved across a UNION, so unique_keys are dropped.) - QueryExpr::SetOp { left, .. } => { - let mut s = left.output_schema()?; - s.unique_keys.clear(); - Ok(s) - } - // ⋈ — output is the concatenation of both inputs' columns. Outer - // joins make the non-preserved side nullable. Post-join row - // identity isn't provable in general, so unique_keys reset. - QueryExpr::Join { - kind, left, right, .. - } => { - let l = left.output_schema()?; - let r = right.output_schema()?; - // Semi / anti joins filter the left side; the right contributes - // no columns, so the output is the left's schema unchanged. Row - // identity *is* preserved (each left row appears at most once), - // but a left row can be dropped, so unique_keys still reset. - if matches!(kind, JoinKind::Semi | JoinKind::Anti) { - return Ok(Schema { - unique_keys: Vec::new(), - ..l - }); - } - let (left_null, right_null) = match kind { - JoinKind::Left => (false, true), - JoinKind::Right => (true, false), - JoinKind::Full => (true, true), - JoinKind::Inner | JoinKind::Cross => (false, false), - JoinKind::Semi | JoinKind::Anti => unreachable!("handled above"), - }; - let l_len = l.columns.len(); - let mut columns = Vec::with_capacity(l_len + r.columns.len()); - columns.extend(l.columns.iter().cloned().map(|mut c| { - c.nullable |= left_null; - c - })); - columns.extend(r.columns.iter().cloned().map(|mut c| { - c.nullable |= right_null; - c - })); - let time_index = l.time_index.or(r.time_index.map(|i| i + l_len)); - Ok(Schema { - columns, - time_index, - unique_keys: Vec::new(), - // The concatenation is complete only if both sides are. - closed: l.closed && r.closed, - }) - } - // ψ-analytic — child schema + one appended window-output column. - QueryExpr::SQLWindowFunc { - func, - args, - output_name, - child, - .. - } => { - let mut out = child.output_schema()?; - // First operand's (dtype, nullable) from the child schema, owned - // so the borrow ends before we append. - let arg = args.first().and_then(|a| match a { - QueryExpr::Column(id) => out.columns.get(*id), - _ => None, - }); - let arg_dtype = || arg.map_or(DataType::Float64, |c| c.dtype.clone()); - let (dtype, nullable) = match func { - WindowFuncKind::RowNumber - | WindowFuncKind::Rank - | WindowFuncKind::DenseRank - | WindowFuncKind::Count => (DataType::Int64, false), - WindowFuncKind::Sum | WindowFuncKind::Avg => (DataType::Float64, true), - // Navigation funcs: arg type, nullable (boundary rows are NULL). - WindowFuncKind::Lag - | WindowFuncKind::Lead - | WindowFuncKind::LagInFrame - | WindowFuncKind::LeadInFrame - | WindowFuncKind::FirstValue - | WindowFuncKind::LastValue - | WindowFuncKind::NthValue(_) => (arg_dtype(), true), - WindowFuncKind::Min | WindowFuncKind::Max => { - (arg_dtype(), arg.is_none_or(|c| c.nullable)) - } - }; - out.columns - .push(Column::new(output_name.clone(), dtype, nullable)); - Ok(out) - } - - // A scalar bridge has no series — model it as a single `value` - // column so it can sit as a `BinaryOp` operand. Both scalar - // leaves — a bridged scalar sub-expression and the eval time — - // are a single `value` column with no labels. Every - // `PromqlScalarBridge` constructed today wraps a plain - // `Literal(Float64)` (issue #220), so the schema doesn't need to - // inspect the inner node. - QueryExpr::PromqlScalarBridge(_) | QueryExpr::EvalTimestamp => Ok(Schema { - columns: vec![Column::new("value", DataType::Float64, false)], - time_index: None, - unique_keys: Vec::new(), - closed: true, - }), - - QueryExpr::CurrentTimestamp => Ok(Schema { - columns: vec![Column::new("value", DataType::Timestamp, false)], - time_index: None, - unique_keys: Vec::new(), - closed: true, - }), - - // `vector(s)` yields a label-less instant vector: the (ts, value) - // floor and nothing else. `closed` — its full label set (empty) is - // known statically (#48). - QueryExpr::PromqlVectorFromScalar(_) => Ok(Schema { - columns: vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - ], - time_index: Some(0), - unique_keys: Vec::new(), - closed: true, - }), - - // `scalar(v)` collapses to a single `value`, no time index — the same - // scalar shape as a constant or `time()` (#48). - QueryExpr::PromqlScalarFromVector(_) => Ok(Schema { - columns: vec![Column::new("value", DataType::Float64, false)], - time_index: None, - unique_keys: Vec::new(), - closed: true, - }), - - // The output shape of ` op ` (or ` op - // `) is the vector side's — a scalar operand (a constant or - // `time()`) contributes only its value, no labels. Prefer the - // non-scalar side. - QueryExpr::BinaryOp { lhs, rhs, op, vector_match } => { - fn scalar(expression: &QueryExpr) -> bool { - match expression { - QueryExpr::PromqlScalarBridge(_) | QueryExpr::EvalTimestamp | QueryExpr::PromqlScalarFromVector(_) => true, - QueryExpr::BinaryOp { lhs, rhs, .. } => scalar(lhs) && scalar(rhs), - _ => false, - } - } - let left = lhs.output_schema()?; - let right = rhs.output_schema()?; - if scalar(lhs) { return Ok(right); } - let mut output = left; - let grouping = vector_match.as_ref().and_then(|m| m.grouping.as_ref()); - let right_rows = matches!(op, BinaryOpKind::Set(PromQLVectorSetOpKind::Or)) - || matches!(grouping, Some(g) if g.side == GroupSide::Right); - let mut additions = Vec::new(); - if right_rows { - additions.extend(right.columns.iter().filter(|c| c.dtype == DataType::Utf8).cloned()); - } - if let Some(grouping) = grouping { - additions.extend(grouping.labels.iter().map(|name| Column::new(name.clone(), DataType::Utf8, true))); - } - for column in additions { - if !output.columns.iter().any(|c| c.name == column.name) { - output.columns.push(column); - } - } - Ok(output) - }, - - // The scalar variants (issue #205) — see `QueryExprError::ScalarHasNoRowSchema`. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => Err(QueryExprError::ScalarHasNoRowSchema), - } - } -} - -/// Output schema of a *per-series* window/range reduction (`rate`/`increase`, -/// or an `*_over_time` reducer under a time `Window`). Such a reduction emits -/// one value per series, so every label column of `input` is preserved and only -/// the sample value is replaced — kept named `value` so the PromQL sample-value -/// convention (and any outer `SampleValue` reference) still resolves it by name. -fn per_series_reduction_schema(input: &Schema, agg: &AggIntent) -> Result { - let vi = if let Some(index) = agg.input_cols().first() { - *index - } else { - super::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, input) - .map_err(|error| QueryExprError::InvalidSampleColumn(error.to_string()))? - }; - if !matches!( - input.columns.get(vi).map(|column| &column.dtype), - Some(DataType::Float64 | DataType::Int64) - ) { - return Err(QueryExprError::InvalidSampleColumn(format!( - "column {vi} is not numeric" - ))); - } - let mut columns = input.columns.clone(); - { - let mut out = agg.output_column(&columns[vi]); - out.name = "value".into(); - // A per-series range reduction produces a PromQL sample value, which is - // always `float64` — override the reducer's own output dtype so - // `count_over_time` (whose `Count` intent types `Int64`) matches every - // other range reducer instead of leaking an `Int64` value column (#69). - out.dtype = DataType::Float64; - columns[vi] = out; - } - Ok(Schema { - columns, - time_index: input.time_index, - unique_keys: input.unique_keys.clone(), - // Per-series reduction is label-preserving: it inherits its input's - // completeness (an open scan stays open; a closed one stays closed). - closed: input.closed, - }) -} - -/// The output schema of an `Aggregate { reduction, measures }` over `in_schema` — -/// the **single** canonical derivation shared by -/// [`QueryExpr::output_schema`]'s `Aggregate` arm and the converter's -/// HAVING-resolution path (`column_resolution::output_schema_for_aggregate`), -/// so the two can never drift (issue #41). -/// -/// `Reduction::PerEntity` selects the label-preserving -/// [`per_series_reduction_schema`] (`rate`/`increase`/`*_over_time`) instead -/// of the cross-series `by ++ measures` shape. Which one applies is read directly -/// off `reduction` — decided once, at construction, by whoever built the -/// `Aggregate` node (issue #165) — not re-derived here from `by`/child shape. -pub fn aggregate_output_schema( - in_schema: &Schema, - reduction: &Reduction, - measures: &[AggIntent], - output_names: &[String], -) -> Result { - let by = match reduction { - Reduction::PerEntity => { - debug_assert_eq!( - measures.len(), - 1, - "a per-entity reduction is single-aggregate" - ); - return per_series_reduction_schema(in_schema, &measures[0]); - } - Reduction::Reduce(by) => by, - }; - - // `without(excluded)` groups by every label *except* those listed: the kept - // labels are the input's label columns minus the excluded positions (and the - // ts / sample-value columns), and the schema stays **open** because the full - // runtime label set isn't known. The `by(...)` path instead enumerates its - // kept columns and freezes to closed (issue #39). - if by.is_without() { - return without_output_schema(in_schema, by.keys(), measures, output_names); - } - - let mut out_cols: Vec = Vec::with_capacity(by.len() + measures.len()); - for &id in by.keys() { - let c = in_schema - .columns - .get(id) - .ok_or(QueryExprError::InvalidGroupByColumn( - id, - in_schema.columns.len(), - ))?; - out_cols.push(c.clone()); - } - let value_col_idx = - super::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, in_schema) - .ok() - .or_else(|| (0..in_schema.columns.len()).find(|i| !by.contains(i))); - let probe = value_col_idx - .and_then(|i| in_schema.columns.get(i)) - .cloned() - .unwrap_or_else(|| Column::new("value", DataType::Float64, false)); - // Each reducer types off its own input column (`SUM(bytes)` vs `AVG(latency)` - // in one node); `None` falls back to the sample-value probe (PromQL's - // single-column convention). A non-empty `output_names[i]` overrides the - // synthetic output column name. - for (i, intent) in measures.iter().enumerate() { - // `count_values("l", v)` emits TWO columns: the synthesized `Utf8` label - // `l` (the stringified sample value it groups by) and the per-value - // count. If `l` collides with a group-by key of the same name, PromQL's - // synthesized label takes precedence — emit a single column, never a - // duplicate. - if let AggIntent::CountValues { label } = intent { - if !out_cols.iter().any(|c| c.name == *label) { - out_cols.push(Column::new(label.clone(), DataType::Utf8, false)); - } - let mut cnt = intent.output_column(&probe); - if let Some(name) = output_names.get(i).filter(|s| !s.is_empty()) { - cnt.name = name.clone(); - } - out_cols.push(cnt); - continue; - } - // Only the output *type* is read from here, so the leading column is - // enough for the multi-column intents: `Cardinality` and `PearsonCorr` - // both have a fixed output type that ignores it. - let in_col = intent - .input_cols() - .first() - .and_then(|id| in_schema.columns.get(*id)) - .unwrap_or(&probe); - let mut out = intent.output_column(in_col); - // A global extremum emits NULL for an empty input, even if its input - // column is non-nullable. Grouped extrema only emit existing groups. - if by.is_empty() && matches!(intent, AggIntent::Min { .. } | AggIntent::Max { .. }) { - out.nullable = true; - } - if let Some((arg, _)) = intent - .arg_selector_columns(in_schema) - .map_err(QueryExprError::InvalidScalarSignature)? - { - out.dtype = in_schema.columns[arg].dtype.clone(); - out.nullable = in_schema.columns[arg].nullable; - } - if let Some(name) = output_names.get(i).filter(|s| !s.is_empty()) { - out.name = name.clone(); - } - out_cols.push(out); - } - // `count_values` groups by (by-keys ∪ the synthesized value label), so the - // by-keys alone are not a unique key — be conservative and claim none. - let has_count_values = measures - .iter() - .any(|a| matches!(a, AggIntent::CountValues { .. })); - let unique_keys = if by.is_empty() || has_count_values { - Vec::new() - } else { - vec![(0..by.len()).collect()] - }; - Ok(Schema { - columns: out_cols, - time_index: None, - unique_keys, - // A cross-series aggregate enumerates exactly `by ++ measures`, so its output - // is closed even over an open input — this is where an open schema - // freezes to closed. - closed: true, - }) -} - -/// Output schema of a `without(excluded)` aggregate: the kept labels (every -/// input label column except the `excluded` positions, the time axis, and the -/// sample-value column) followed by the aggregate output column(s). Unlike the -/// `by` path this stays **open** — the excluded set is enumerable but the kept -/// set is not (the runtime carries labels the usage-derived schema never saw), -/// so the schema can't freeze to closed and claims no unique key (issue #39). -fn without_output_schema( - in_schema: &Schema, - excluded: &[ColumnId], - measures: &[AggIntent], - output_names: &[String], -) -> Result { - for &id in excluded { - if id >= in_schema.columns.len() { - return Err(QueryExprError::InvalidGroupByColumn( - id, - in_schema.columns.len(), - )); - } - } - // A nested aggregate renames the sample value (`sum by (le) (…)` → `sum`); - // it is still the value, not a kept label. - let value = - super::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, in_schema).ok(); - let mut out_cols: Vec = Vec::new(); - for (i, col) in in_schema.columns.iter().enumerate() { - let is_time = in_schema.time_index == Some(i); - if !is_time && value != Some(i) && !excluded.contains(&i) { - out_cols.push(col.clone()); - } - } - let probe = value - .and_then(|i| in_schema.columns.get(i)) - .cloned() - .unwrap_or_else(|| Column::new("value", DataType::Float64, false)); - for (i, intent) in measures.iter().enumerate() { - // Only the output *type* is read from here, so the leading column is - // enough for the multi-column intents: `Cardinality` and `PearsonCorr` - // both have a fixed output type that ignores it. - let in_col = intent - .input_cols() - .first() - .and_then(|id| in_schema.columns.get(*id)) - .unwrap_or(&probe); - let mut out = intent.output_column(in_col); - if let Some((arg, _)) = intent - .arg_selector_columns(in_schema) - .map_err(QueryExprError::InvalidScalarSignature)? - { - out.dtype = in_schema.columns[arg].dtype.clone(); - out.nullable = in_schema.columns[arg].nullable; - } - if let Some(name) = output_names.get(i).filter(|s| !s.is_empty()) { - out.name = name.clone(); - } - out_cols.push(out); - } - Ok(Schema { - columns: out_cols, - time_index: None, - unique_keys: Vec::new(), - // The kept label set is runtime-only, so — unlike `by` — this does not - // freeze the open schema to closed. - closed: false, - }) -} - -/// Infer the `(DataType, nullable)` a scalar [`QueryExpr`] produces against an -/// input [`Schema`]. Used by `Project` schema derivation. Approximate here: -/// unknown columns and bare `FunctionCall`s fall back to a permissive default -/// (post-ASAP binding refines with a real function/type registry). `expr` -/// must be one of the scalar variants (issue #205) — an operator variant here -/// is a construction bug, not a shape this needs to handle silently. -fn infer_expr_type( - expr: &QueryExpr, - schema: &Schema, -) -> Result<(DataType, bool), QueryExprError> { - Ok(match expr { - QueryExpr::CurrentTimestamp => (DataType::Timestamp, false), - QueryExpr::Column(id) => schema - .columns - .get(*id) - .map(|c| (c.dtype.clone(), c.nullable)) - .unwrap_or((DataType::Float64, true)), - QueryExpr::Literal(s) => match s { - ScalarValue::Int64(_) => (DataType::Int64, false), - ScalarValue::Float64(_) => (DataType::Float64, false), - ScalarValue::Utf8(_) => (DataType::Utf8, false), - ScalarValue::Boolean(_) => (DataType::Bool, false), - ScalarValue::Null => (DataType::Null, true), - ScalarValue::Interval { .. } => (DataType::Interval, false), - }, - // Boolean-valued expressions (SQL three-valued logic → nullable). - QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::InList { .. } => (DataType::Bool, true), - QueryExpr::Arithmetic { op, left, right } => { - let (lt, ln) = infer_expr_type(left, schema)?; - let (rt, rn) = infer_expr_type(right, schema)?; - // Temporal subtraction yields a fixed duration with a unit, not a - // calendar interval or a floating-point number. Until the IR can - // preserve that unit, fail instead of publishing a numeric schema. - if matches!(op, ArithmeticOpKind::Sub) - && matches!(lt, DataType::Date | DataType::Timestamp) - && matches!(rt, DataType::Date | DataType::Timestamp) - { - return Err(QueryExprError::InvalidScalarSignature( - "temporal subtraction produces an unsupported duration type".into(), - )); - } - - // Operand order is not checked: the orders that are not valid SQL - // (`Interval - Timestamp`) are rejected by the planner upstream, so - // a pair rule stays as small as the numeric one it sits beside. - let dtype = match (<, &rt) { - // SQL unary minus lowers to -1 * expression, including intervals. - (DataType::Int64, DataType::Interval) | (DataType::Interval, DataType::Int64) - if matches!(op, ArithmeticOpKind::Mul) => - { - DataType::Interval - } - (DataType::Timestamp, DataType::Interval) - | (DataType::Interval, DataType::Timestamp) => DataType::Timestamp, - (DataType::Date, DataType::Interval) | (DataType::Interval, DataType::Date) => { - DataType::Date - } - (DataType::Interval, DataType::Interval) => DataType::Interval, - (DataType::Int64, DataType::Int64) => DataType::Int64, - _ => DataType::Float64, - }; - (dtype, ln || rn) - } - QueryExpr::Cast { to, try_cast, expr } => { - let (_, nullable) = infer_expr_type(expr, schema)?; - (to.clone(), *try_cast || nullable) - } - QueryExpr::FunctionCall { name, args } => { - if name == "asap_element_access" { - super::scalar_signature::element_access_type(args, schema) - .map_err(QueryExprError::InvalidScalarSignature)? - } else if name == "asap_struct_field" { - super::scalar_signature::struct_field_type(args, schema) - .map_err(QueryExprError::InvalidScalarSignature)? - } else if let Some(function) = - super::scalar_signature::MapScalarFunction::from_name(name) - { - let arguments = args - .iter() - .map(|arg| infer_expr_type(arg, schema)) - .collect::, _>>()?; - function - .output_type(&arguments) - .map_err(QueryExprError::InvalidScalarSignature)? - } else { - // Legacy unknown functions retain their existing policy. - (DataType::Float64, true) - } - } - QueryExpr::Case { - branches, - else_expr, - .. - } => { - if let Some((_, then)) = branches.first() { - (infer_expr_type(then, schema)?.0, true) - } else if let Some(other) = else_expr { - infer_expr_type(other, schema)? - } else { - (DataType::Null, true) - } - } - other => { - unreachable!("infer_expr_type called on a non-scalar QueryExpr variant: {other:?}") - } - }) -} - -/// Default output-column name for a projection item with no explicit alias: -/// a bare column keeps its (schema) name; anything else gets `col_{i}`. -fn default_proj_name(expr: &QueryExpr, idx: usize, schema: &Schema) -> String { - match expr { - QueryExpr::Column(id) => schema - .columns - .get(*id) - .map(|c| c.name.clone()) - .unwrap_or_else(|| format!("col_{idx}")), - _ => format!("col_{idx}"), - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::pre_asap::expr_ir::{ArithmeticOpKind, CompareOpKind}; - use crate::types::AccuracyTarget; - - fn col(name: &str, dtype: DataType, nullable: bool) -> Column { - Column::new(name, dtype, nullable) - } - - /// Shifting an instant by a duration stays an instant, and shifting a date - /// stays a date — neither falls through to the numeric default, which is - /// what `l_shipdate + INTERVAL '30' DAY` would otherwise be typed as. - #[test] - fn interval_arithmetic_keeps_the_temporal_type() { - let schema = Schema::new(vec![ - col("ts", DataType::Timestamp, false), - col("d", DataType::Date, false), - ]); - let thirty_days = || { - Rc::new(QueryExpr::Literal(ScalarValue::Interval { - months: 0, - days: 30, - nanos: 0, - })) - }; - let shift = |column, op| QueryExpr::Arithmetic { - op, - left: Rc::new(QueryExpr::Column(column)), - right: thirty_days(), - }; - - assert_eq!( - shift(0, ArithmeticOpKind::Add) - .scalar_type(&schema) - .unwrap() - .0, - DataType::Timestamp - ); - assert_eq!( - shift(1, ArithmeticOpKind::Sub) - .scalar_type(&schema) - .unwrap() - .0, - DataType::Date - ); - assert_eq!( - QueryExpr::Arithmetic { - op: ArithmeticOpKind::Add, - left: thirty_days(), - right: thirty_days(), - } - .scalar_type(&schema) - .unwrap() - .0, - DataType::Interval - ); - } - - fn scan( - columns: Vec, - time_index: Option, - uk: Vec>, - ) -> QueryExpr { - QueryExpr::Scan { - source: Source::Table { - table_ref: "t".into(), - }, - predicates: vec![], - schema: Schema { - columns, - time_index, - unique_keys: uk, - closed: true, - }, - } - } - - #[test] - fn project_preserves_unique_keys_that_are_passed_through() { - let input = Rc::new(scan( - vec![ - col("tenant", DataType::Utf8, false), - col("region", DataType::Utf8, false), - col("value", DataType::Int64, false), - ], - None, - vec![vec![0, 1]], - )); - let projected = QueryExpr::Project { - cols: vec![ - ProjectItem { - alias: Some("r".into()), - expr: QueryExpr::Column(1), - }, - ProjectItem { - alias: Some("t".into()), - expr: QueryExpr::Column(0), - }, - ProjectItem { - alias: None, - expr: QueryExpr::Arithmetic { - op: ArithmeticOpKind::Add, - left: Rc::new(QueryExpr::Column(2)), - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), - }, - }, - ], - qualifier: None, - child: input, - }; - - assert_eq!( - projected.output_schema().unwrap().unique_keys, - vec![vec![1, 0]] - ); - } - - #[test] - fn project_drops_a_unique_key_when_a_key_column_is_omitted() { - let input = Rc::new(scan( - vec![ - col("tenant", DataType::Utf8, false), - col("region", DataType::Utf8, false), - ], - None, - vec![vec![0, 1]], - )); - let projected = QueryExpr::Project { - cols: vec![ProjectItem { - alias: None, - expr: QueryExpr::Column(0), - }], - qualifier: None, - child: input, - }; - - assert!(projected.output_schema().unwrap().unique_keys.is_empty()); - } - - #[test] - fn legacy_window_json_without_frame_deserializes_as_unspecified() { - let window = QueryExpr::SQLWindowFunc { - func: WindowFuncKind::RowNumber, - args: vec![], - partition_by: GroupKeys::by(vec![]), - order_by: vec![], - frame: Some(WindowFrame { - units: WindowFrameUnits::Range, - start_bound: WindowFrameBound::Preceding(WindowFrameOffset::Scalar( - ScalarValue::Null, - )), - end_bound: WindowFrameBound::CurrentRow, - }), - output_name: "row_number".into(), - child: Rc::new(scan(vec![col("v", DataType::Int64, false)], None, vec![])), - }; - let mut json = serde_json::to_value(window).unwrap(); - json.get_mut("SQLWindowFunc") - .and_then(serde_json::Value::as_object_mut) - .unwrap() - .remove("frame"); - - let decoded: QueryExpr = serde_json::from_value(json).unwrap(); - assert!(matches!( - decoded, - QueryExpr::SQLWindowFunc { frame: None, .. } - )); - } - - /// A row can appear in more than one branch, so no branch's unique key is a - /// key of the union. `Concat` took the first child's schema verbatim, which - /// let a `Dedup`'s key leak out and claim a uniqueness the merged rows do - /// not have — `unique_keys` feeds CSE's producer-sharing legality check. - #[test] - fn merge_drops_the_branches_unique_keys() { - let branch = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(scan( - vec![ - col("k", DataType::Utf8, false), - col("v", DataType::Int64, false), - ], - None, - vec![], - )), - }; - assert_eq!( - branch().output_schema().unwrap().unique_keys, - vec![vec![0]], - "a Dedup branch does have a unique key on its own" - ); - - let merged = QueryExpr::concat(vec![branch(), branch()]); - let schema = merged.output_schema().unwrap(); - assert!( - schema.unique_keys.is_empty(), - "the union of two deduplicated branches is not deduplicated" - ); - // The column shape is still the first branch's. - assert_eq!(schema.columns.len(), 2); - } - - /// Same rule as `SetOp`, which already dropped them. - #[test] - fn merge_and_setop_agree_on_unique_keys() { - let branch = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(scan(vec![col("k", DataType::Utf8, false)], None, vec![])), - }; - let merged = QueryExpr::concat(vec![branch(), branch()]); - let setop = QueryExpr::SetOp { - kind: RelationalSetOpKind::Union, - all: true, - left: Rc::new(branch()), - right: Rc::new(branch()), - }; - assert_eq!( - merged.output_schema().unwrap().unique_keys, - setop.output_schema().unwrap().unique_keys, - ); - } - - #[test] - fn an_empty_merge_has_no_schema() { - assert!(matches!( - QueryExpr::concat(vec![]).output_schema(), - Err(QueryExprError::EmptyConcat) - )); - } - - /// Issue #228: a `Concat` built via `concat_with_discriminator` gets a - /// sound compound `(discriminator, inner_key)` unique key, even though - /// each branch's own `inner_key` alone repeats across branches (exactly - /// the shape `merge_drops_the_branches_unique_keys` shows is unsafe - /// *without* a discriminator). - #[test] - fn discriminator_override_produces_a_compound_unique_key() { - // Two branches, each individually deduplicated on column 0 (`k`) — - // but, per `merge_drops_the_branches_unique_keys`, that alone proves - // nothing about the union. Column 1 (`branch_id`) stands in for a - // discriminator the constructor has separately proven distinct per - // branch (PromQL φ, a synthetic `GROUPING()` id, ...) — this - // schema-level test only checks the shape `output_schema` derives - // from asserting one, not how a real caller proves distinctness. - let branch = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(scan( - vec![ - col("k", DataType::Utf8, false), - col("branch_id", DataType::Int64, false), - ], - None, - vec![], - )), - }; - let merged = QueryExpr::concat_with_discriminator( - vec![branch(), branch()], - /* discriminator */ 1, - /* inner_key */ vec![0], - ); - let schema = merged.output_schema().unwrap(); - assert_eq!( - schema.unique_keys, - vec![vec![1, 0]], - "(discriminator, inner_key) is the sole asserted unique key" - ); - assert_eq!( - schema.columns.len(), - 2, - "column shape is still the first branch's" - ); - } - - #[test] - fn discriminator_assertion_rejects_unknown_wire_fields() { - let json = r#"{"discriminator":1,"inner_key":[0],"unverified":true}"#; - assert!(serde_json::from_str::(json).is_err()); - } - - /// The override is opt-in: building a `Concat` without asserting a - /// discriminator — via the plain struct literal, exactly like every call - /// site before issue #228 — still drops `unique_keys` by default, - /// unchanged. - #[test] - fn ordinary_concat_struct_literal_still_drops_unique_keys_by_default() { - let branch = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(scan(vec![col("k", DataType::Utf8, false)], None, vec![])), - }; - let merged = QueryExpr::Concat { - children: vec![branch(), branch()], - discriminator_unique_key: None, - }; - assert!(merged.output_schema().unwrap().unique_keys.is_empty()); - } - - /// Misuse check (issue #228): there is no way to end up with a - /// discriminator-backed unique key without a call site literally naming - /// a column as the discriminator. Neither the ordinary `concat` - /// constructor nor a bare struct literal with `discriminator_unique_key: - /// None` can be coaxed into fabricating one — the only path that - /// produces `Some` is `concat_with_discriminator` / - /// `ConcatDiscriminatorKey::new`, both of which require `discriminator` - /// as an explicit, named argument. - #[test] - fn no_way_to_fabricate_a_unique_key_without_naming_a_discriminator() { - let branch = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(scan(vec![col("k", DataType::Utf8, false)], None, vec![])), - }; - // The ordinary builder. - assert_eq!( - QueryExpr::concat(vec![branch(), branch()]) - .output_schema() - .unwrap() - .unique_keys, - Vec::>::new() - ); - // The bare struct literal, explicitly opting out. - assert_eq!( - QueryExpr::Concat { - children: vec![branch(), branch()], - discriminator_unique_key: None, - } - .output_schema() - .unwrap() - .unique_keys, - Vec::>::new() - ); - } - - #[test] - fn project_retypes_and_renames_per_item() { - let child = scan( - vec![ - col("ts", DataType::Timestamp, false), - col("host", DataType::Utf8, false), - col("value", DataType::Float64, false), - ], - Some(0), - vec![vec![0, 1]], - ); - let q = QueryExpr::Project { - qualifier: None, - cols: vec![ - // bare column passthrough keeps its (schema) name + type: host=col 1 - ProjectItem { - alias: None, - expr: QueryExpr::Column(1), - }, - // arithmetic over value (col 2) → Float64 - ProjectItem { - alias: Some("dbl".into()), - expr: QueryExpr::Arithmetic { - op: ArithmeticOpKind::Add, - left: Rc::new(QueryExpr::Column(2)), - right: Rc::new(QueryExpr::Column(2)), - }, - }, - // comparison → Bool (nullable under 3-valued logic) - ProjectItem { - alias: Some("flag".into()), - expr: QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(2)), - op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(0.0))), - }, - }, - ], - child: Rc::new(child), - }; - let s = q.output_schema().unwrap(); - assert_eq!(s.columns.len(), 3); - assert_eq!(s.columns[0], col("host", DataType::Utf8, false)); - assert_eq!(s.columns[1], col("dbl", DataType::Float64, false)); - assert_eq!(s.columns[2], col("flag", DataType::Bool, true)); - // projection drops the time axis + unique keys (ts not retained) - assert!(s.time_index.is_none()); - assert!(s.unique_keys.is_empty()); - } - - #[test] - fn group_keys_by_vs_without_semantics() { - let by = GroupKeys::by(vec![1, 2]); - let without = GroupKeys::without(vec![1, 2]); - assert!(!by.is_without()); - assert!(without.is_without()); - // Deref / iteration expose the stored keys regardless of mode. - assert_eq!(by.len(), 2); - assert_eq!(without.keys(), &[1, 2]); - // A `by` compares equal to its bare vec; a `without` never does. - assert_eq!(by, vec![1, 2]); - assert_ne!(without, vec![1, 2]); - assert_ne!(by, without); - } - - #[test] - fn group_keys_serde_by_is_bare_array_without_is_tagged() { - // `by` keeps the pre-#39 bare-array wire format; `without` uses an object. - let by = serde_json::to_string(&GroupKeys::by(vec![2, 3])).unwrap(); - assert_eq!(by, "[2,3]"); - let without = serde_json::to_string(&GroupKeys::without(vec![2])).unwrap(); - assert_eq!(without, r#"{"without":[2]}"#); - // Round-trip both. - for g in [GroupKeys::by(vec![2, 3]), GroupKeys::without(vec![2])] { - let json = serde_json::to_string(&g).unwrap(); - let back: GroupKeys = serde_json::from_str(&json).unwrap(); - assert_eq!(back, g); - } - } - - #[test] - fn without_aggregate_keeps_open_schema_minus_excluded() { - // `sum without (instance) (m)` over `[ts, value, instance, job]`: the - // kept labels are the input labels minus the excluded `instance` (and ts - // / value), followed by the `sum` column, and the schema stays OPEN - // (issue #39). `job` survives; `instance` is dropped. - let scan_node = QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - col("ts", DataType::Timestamp, false), - col("value", DataType::Float64, false), - col("instance", DataType::Utf8, true), - col("job", DataType::Utf8, true), - ], - 0, - vec![], - ), - }; - let agg = QueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::without(vec![2])), // exclude `instance` - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan_node), - }; - let s = agg.output_schema().unwrap(); - let names: Vec<_> = s.columns.iter().map(|c| c.name.as_str()).collect(); - assert_eq!(names, vec!["job", "sum"], "kept `job`, dropped `instance`"); - assert!(!s.closed, "a `without` result stays open"); - assert!(s.time_index.is_none()); - assert!(s.unique_keys.is_empty(), "kept set unknown → no unique key"); - } - - // A nested aggregate's renamed sample value is not a kept label. - #[test] - fn without_aggregate_drops_a_renamed_sample_value() { - // `sum without (inst) (sum by (inst, job) (m))` over `[inst, job, sum]`. - let inner = QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::new(vec![ - col("inst", DataType::Utf8, true), - col("job", DataType::Utf8, true), - col("sum", DataType::Float64, false), - ]), - }; - let agg = QueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::without(vec![0])), - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(inner), - }; - let s = agg.output_schema().unwrap(); - let names: Vec<_> = s.columns.iter().map(|c| c.name.as_str()).collect(); - assert_eq!(names, vec!["job", "sum"]); - } - - #[test] - fn time_shift_is_schema_pass_through() { - // `offset`/`@` move *when* a selector is evaluated, never its columns — - // a `TimeShift` output schema equals its child's (issue #40). - let scan_node = scan( - vec![ - col("ts", DataType::Timestamp, false), - col("value", DataType::Float64, false), - col("job", DataType::Utf8, true), - ], - Some(0), - vec![], - ); - let shifted = QueryExpr::TimeShift { - shift: TimeShift { - offset_ms: 3_600_000, - at: Some(AtModifier::Timestamp(1_609_746_000_000)), - }, - child: Rc::new(scan_node.clone()), - }; - assert_eq!( - shifted.output_schema().unwrap(), - scan_node.output_schema().unwrap(), - ); - } - - #[test] - fn time_shift_identity_and_serde() { - let offset_only = TimeShift { - offset_ms: 1, - at: None, - }; - let at_only = TimeShift { - offset_ms: 0, - at: Some(AtModifier::End), - }; - assert!(TimeShift::default().is_identity()); - assert!(!offset_only.is_identity()); - assert!(!at_only.is_identity()); - // Round-trip the shift + anchor. - let s = TimeShift { - offset_ms: -300_000, - at: Some(AtModifier::Timestamp(60_000)), - }; - let back: TimeShift = serde_json::from_str(&serde_json::to_string(&s).unwrap()).unwrap(); - assert_eq!(back, s); - } - - // Nested temporal aggregation must replace the sample, never the grouping label. - #[test] - fn temporal_reduction_of_grouped_sum_preserves_job() { - let input = Schema::new(vec![ - col("job", DataType::Utf8, true), - col("sum", DataType::Float64, false), - ]); - for aggregate in [ - AggIntent::Avg { col: None }, - AggIntent::Avg { col: Some(1) }, - AggIntent::Rate, - ] { - let output = - aggregate_output_schema(&input, &Reduction::PerEntity, &[aggregate], &[]).unwrap(); - assert_eq!(output.columns[0], input.columns[0]); - assert_eq!(output.columns[1].name, "value"); - assert_eq!(output.columns[1].dtype, DataType::Float64); - } - } - - #[test] - fn per_series_rate_preserves_labels() { - // A per-series range reduction (`rate`) is label-preserving: it produces - // one value per series, so every label survives and only the sample - // value is replaced (kept named `value`). The TimeRange child is the - // structural marker; the outer Aggregate carries the Rate intent. - let scan_node = scan( - vec![ - col("ts", DataType::Timestamp, false), - col("value", DataType::Float64, false), - col("job", DataType::Utf8, true), - ], - Some(0), - vec![], - ); - let rate = QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![AggIntent::Rate], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan_node), - }), - }; - let s = rate.output_schema().unwrap(); - assert_eq!( - s.columns - .iter() - .map(|c| c.name.as_str()) - .collect::>(), - vec!["ts", "value", "job"], - "rate preserves all labels; only the sample value is replaced" - ); - assert_eq!(s.time_index, Some(0)); - assert!(s.column_id("job").is_some(), "label survives the reduction"); - } - - #[test] - fn over_time_reduction_preserves_labels() { - // `*_over_time` lowers to `Aggregate { by:[], [reducer], TimeRange { Scan } }`: - // a per-series time-range reduction. The TimeRange child confers per-series - // semantics on otherwise cross-series intents like `Avg`, so an outer - // `sum by(job)(avg_over_time(...))` resolves its key positionally. - let scan_node = scan( - vec![ - col("ts", DataType::Timestamp, false), - col("value", DataType::Float64, false), - col("job", DataType::Utf8, true), - ], - Some(0), - vec![], - ); - let avg_over_time = QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![AggIntent::Avg { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan_node), - }), - }; - let s = avg_over_time.output_schema().unwrap(); - assert_eq!( - s.columns - .iter() - .map(|c| c.name.as_str()) - .collect::>(), - vec!["ts", "value", "job"], - "TimeRange-child marks per-series: labels preserved, value renamed" - ); - assert!( - s.column_id("job").is_some(), - "outer Aggregate.by can resolve it" - ); - } - - #[test] - fn completeness_open_leaf_freezes_to_closed_at_cross_series_aggregate() { - // A schemaless (PromQL-style) leaf is *open*; it stays open through a - // per-series reduction (`rate`), then is **frozen to closed** by a - // cross-series aggregate (which enumerates exactly its output columns). - let open_leaf = QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - // `with_time_index` defaults to `closed: false` (open). - schema: Schema::with_time_index( - vec![ - col("ts", DataType::Timestamp, false), - col("value", DataType::Float64, false), - col("job", DataType::Utf8, true), - ], - 0, - vec![], - ), - }; - assert!( - !open_leaf.output_schema().unwrap().closed, - "schemaless leaf is open" - ); - - let rate = QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![AggIntent::Rate], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(open_leaf), - }; - assert!( - !rate.output_schema().unwrap().closed, - "per-series rate is label-preserving → stays open" - ); - - let sum_by_job = QueryExpr::Aggregate { - reduction: Reduction::by(vec![2]), // `job` - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(rate), - }; - assert!( - sum_by_job.output_schema().unwrap().closed, - "cross-series aggregate enumerates `by ++ measures` → frozen to closed" - ); - } - - #[test] - fn project_keeps_time_index_when_ts_passed_through() { - let child = scan( - vec![ - col("ts", DataType::Timestamp, false), - col("value", DataType::Float64, false), - ], - Some(0), - vec![], - ); - let q = QueryExpr::Project { - qualifier: None, - cols: vec![ - // value=col 1, ts=col 0 - ProjectItem { - alias: None, - expr: QueryExpr::Column(1), - }, - ProjectItem { - alias: None, - expr: QueryExpr::Column(0), - }, - ], - child: Rc::new(child), - }; - let s = q.output_schema().unwrap(); - assert_eq!(s.columns[0].name, "value"); - assert_eq!(s.columns[1].name, "ts"); - assert_eq!(s.time_index, Some(1)); - } - - fn join(kind: JoinKind) -> QueryExpr { - let left = scan(vec![col("a", DataType::Int64, false)], None, vec![vec![0]]); - let right = scan(vec![col("b", DataType::Utf8, false)], None, vec![]); - QueryExpr::Join { - kind, - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - left: Rc::new(left), - right: Rc::new(right), - } - } - - #[test] - fn inner_join_concatenates_both_sides() { - let s = join(JoinKind::Inner).output_schema().unwrap(); - assert_eq!(s.columns.len(), 2); - assert_eq!(s.columns[0], col("a", DataType::Int64, false)); - assert_eq!(s.columns[1], col("b", DataType::Utf8, false)); - // post-join row identity not provable → no unique keys - assert!(s.unique_keys.is_empty()); - } - - #[test] - fn left_join_makes_right_side_nullable() { - let s = join(JoinKind::Left).output_schema().unwrap(); - assert!(!s.columns[0].nullable, "preserved left side stays non-null"); - assert!(s.columns[1].nullable, "right side nullable under LEFT JOIN"); - } - - #[test] - fn full_join_makes_both_sides_nullable() { - let s = join(JoinKind::Full).output_schema().unwrap(); - assert!(s.columns[0].nullable); - assert!(s.columns[1].nullable); - } - - #[test] - fn setop_takes_left_shape_and_drops_unique_keys() { - let left = scan( - vec![ - col("k", DataType::Utf8, false), - col("v", DataType::Int64, false), - ], - None, - vec![vec![0]], - ); - let right = scan( - vec![ - col("k", DataType::Utf8, false), - col("v", DataType::Int64, false), - ], - None, - vec![vec![0]], - ); - let q = QueryExpr::SetOp { - kind: RelationalSetOpKind::Union, - all: false, - left: Rc::new(left), - right: Rc::new(right), - }; - let s = q.output_schema().unwrap(); - assert_eq!(s.columns.len(), 2); - assert_eq!(s.columns[0].name, "k"); - assert!( - s.unique_keys.is_empty(), - "UNION does not preserve row identity" - ); - } - - // ── PromqlScalarBridge / Literal dedup (issue #220) ───────────────────── - - /// `QueryExpr::promql_scalar(v)` — what every front end now constructs in - /// place of the old `PromqlScalar(v)` leaf — wraps exactly - /// `Literal(ScalarValue::Float64(v))`: the same value a SQL-emitted typed - /// float literal in a scalar-sub-language position would carry, just at a - /// different tree position. `as_promql_scalar` is the round-trip inverse. - #[test] - fn promql_scalar_bridges_a_literal_float_at_an_operator_position() { - let bridge = QueryExpr::::promql_scalar(2.5); - assert_eq!( - bridge, - QueryExpr::PromqlScalarBridge(Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.5)))) - ); - assert_eq!(bridge.as_promql_scalar(), Some(2.5)); - - // The same value a SQL `Compare`/`Arithmetic` operand would carry, in - // its native (unwrapped, no row schema) scalar-sub-language position — - // no longer a different variant, just not bridged to this tree - // position. - let sql_literal = QueryExpr::::Literal(ScalarValue::Float64(2.5)); - assert_eq!(bridge.as_promql_scalar(), Some(2.5)); - assert_ne!( - bridge, sql_literal, - "bridge and bare literal are distinct nodes" - ); - // Not every shape is a scalar bridge: neither a bare `Literal` nor an - // operator node reports a value. - assert_eq!(sql_literal.as_promql_scalar(), None); - assert_eq!(scan(vec![], None, vec![]).as_promql_scalar(), None); - } - - /// Pins the tree-position distinction issue #220 asks for: the very same - /// `Literal(ScalarValue::Float64(_))` value has a row schema when it sits - /// at the operator-tree position (wrapped in `PromqlScalarBridge` — a - /// `BinaryOp` operand, `PromqlVectorFromScalar` child, or a query root), - /// and has none when it sits bare, in a scalar-sub-language position - /// (`Compare`/`Arithmetic`/… operand) — no longer decided by which of two - /// duplicate variants was used, only by whether the wrapper is present. - #[test] - fn row_schema_rides_on_the_bridge_wrapper_not_the_literal_variant() { - let bridged = QueryExpr::::promql_scalar(42.0); - let schema = bridged.output_schema().expect("bridge has a row schema"); - assert_eq!(schema.columns.len(), 1); - assert_eq!(schema.columns[0].name, "value"); - assert_eq!(schema.columns[0].dtype, DataType::Float64); - assert!(schema.time_index.is_none()); - - // The identical value, unwrapped (the scalar-sub-language position a - // `Compare`/`Arithmetic` operand would occupy) has no row schema of - // its own — it's a construction bug to call `output_schema` on it - // directly, caught as `ScalarHasNoRowSchema` rather than panicking. - let bare = QueryExpr::::Literal(ScalarValue::Float64(42.0)); - assert!(matches!( - bare.output_schema(), - Err(QueryExprError::ScalarHasNoRowSchema) - )); - } - - /// `BinaryOp`'s schema derivation follows the non-scalar (vector) side - /// when the other operand is a `PromqlScalarBridge`, and a `VectorMatch` - /// modifier survives unchanged alongside it — the relational binary-op - /// path (issue #220's Instance 2, left as follow-up) is untouched by the - /// Instance-1 `PromqlScalar` → `PromqlScalarBridge` collapse. - // `filters` (#466) round-trips, and an `Aggregate` serialized before the - // field existed still deserializes as unfiltered. - #[test] - fn aggregate_filters_serde_round_trip_and_default() { - let child = Rc::new(scan( - vec![ - col("service", DataType::Utf8, false), - col("latency", DataType::Float64, false), - ], - None, - vec![], - )); - let filtered = QueryExpr::Aggregate { - reduction: Reduction::by(vec![0]), - measures: vec![ - AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }, - AggIntent::Sum { col: Some(1) }, - ], - output_names: vec![], - filters: vec![ - Some(Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(1)), - op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(1.0))), - }))), - None, - ], - having: None, - child: Rc::clone(&child), - }; - let json = serde_json::to_value(&filtered).unwrap(); - assert_eq!( - serde_json::from_value::(json.clone()).unwrap(), - filtered - ); - - let mut legacy = json; - legacy["Aggregate"] - .as_object_mut() - .unwrap() - .remove("filters") - .expect("fixture sanity: filters was serialized"); - let decoded: QueryExpr = serde_json::from_value(legacy).unwrap(); - let QueryExpr::Aggregate { filters, .. } = &decoded else { - unreachable!() - }; - assert!(filters.is_empty()); - } - - #[test] - fn binary_op_schema_follows_the_vector_side_over_a_scalar_bridge_with_vector_match_intact() { - let vector = scan( - vec![ - col("host", DataType::Utf8, false), - col("value", DataType::Float64, false), - ], - None, - vec![], - ); - let vm = VectorMatch { - kind: VectorMatchKind::On, - labels: vec!["host".into()], - grouping: None, - }; - let op = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Gt), - lhs: Rc::new(vector.clone()), - rhs: Rc::new(QueryExpr::promql_scalar(1.0)), - vector_match: Some(vm.clone()), - }; - assert_eq!(op.output_schema().unwrap(), vector.output_schema().unwrap()); - let QueryExpr::BinaryOp { vector_match, .. } = &op else { - unreachable!() - }; - assert_eq!(vector_match.as_ref(), Some(&vm)); - } -} diff --git a/crates/types/src/pre_asap/resolve.rs b/crates/types/src/pre_asap/resolve.rs deleted file mode 100644 index 95b482e54..000000000 --- a/crates/types/src/pre_asap/resolve.rs +++ /dev/null @@ -1,857 +0,0 @@ -//! Resolve a front-end-emitted, unresolved [`UnresolvedQueryExpr`] (`QueryExpr`) -//! into the canonical, positional [`ResolvedQueryExpr`] (`QueryExpr`). -//! -//! Both front ends (`asap-frontend-promql`, `asap-frontend-sql`) construct -//! canonical `QueryExpr` shapes directly during their own `interpret` step -//! (issue #179) — heavy-hitter `topk` recognition, the window-over-aggregate -//! fold, the `PerEntity`/`Reduce` reduction choice, and every other -//! *structural* decision happen right there, since a front end already knows -//! the answer at parse time. What's left for [`resolve_root`] is exactly the -//! "mechanical, schema-dependent substitution" #179 describes: a single -//! generic, shape-preserving walk — every [`UnresolvedQueryExpr`] variant maps to the -//! identical [`ResolvedQueryExpr`] variant — that resolves every [`ColumnRef`] to -//! the [`SchemaResolver`](super::schema_resolver::SchemaResolver)-computed positional [`ColumnId`]. -//! -//! ## Why positional `ColumnId`, not just carrying names all the way through (issue #216) -//! -//! A mature query engine can legitimately choose either design — DataFusion's -//! own logical plan (what `asap-frontend-sql` walks to build its `QueryExpr`) -//! and Calcite both keep names, with an optional table qualifier, all the way -//! through logical optimization, only going positional once they lower to a -//! physical plan. Resolving once, immediately after each front end's own -//! `interpret` step, is the better trade for *this* codebase's shape — one -//! front-end-facing tree feeding several independent downstream passes -//! (`canonicalize`, the cost model, `dag_export`, schema/type inference, -//! `asap-aware-mapping`'s summary binding) — for three concrete reasons: -//! -//! 1. **Names collide across joins.** Not hypothetical: `join_predicate_disambiguates_shared_column_name` -//! (`crates/frontend-sql/tests/sql_lowering.rs`) exists specifically because -//! `metrics.service` and `hosts.service` are both just `"service"` once their -//! schemas are concatenated. A bare name is ambiguous the moment two sources -//! share one; `ColumnId` is what makes "the second `service`, position 4, not -//! the first" a fact recorded once, instead of a lookup redone at every use site. -//! 2. **A name's meaning changes going up the tree.** `Project` renames/aliases, -//! `Aggregate` collapses columns and introduces synthetic ones, `Join` -//! concatenates two schemas — a name valid at a `Scan` leaf isn't -//! automatically the right binding three nodes up; it has to be reinterpreted -//! against whatever schema is in scope at that node. Resolving bottom-up -//! pins each reference to "this exact column of this exact node's -//! already-derived output schema," so nothing downstream re-derives that scope. -//! 3. **It concentrates scoping logic in one place instead of ~6.** Every -//! downstream pass just compares/indexes `ColumnId`s — O(1), unambiguous. If -//! they worked on names instead, each would need its own qualifier-aware, -//! join-collision-aware name resolver, or risk silently binding to the wrong -//! `"service"`. -//! -//! Removing this resolution step and carrying `ColumnRef` everywhere would -//! therefore be a real regression for this repo's shape, not just a rename — -//! every one of those downstream passes would have to reimplement the scoping -//! this module already centralizes. - -use std::rc::Rc; - -use thiserror::Error; - -use super::agg_intent::AggIntent; -use super::column_resolution::{ - resolve_column_ref, resolve_column_refs, resolve_expr, resolve_group_keys_promql, ResolveError, -}; -use super::expr_ir::ColumnRef; -use super::query_expr::{ - aggregate_output_schema, any_measure_filtered, ConcatDiscriminatorKey, GroupKeys, Predicate, - ProjectItem, QueryExprError, Reduction, ResolvedQueryExpr, SortKey, UnresolvedQueryExpr, -}; -use super::schema::{ColumnId, Schema}; -use super::schema_resolver::SchemaResolver; - -/// Errors from resolving a canonical, unresolved [`UnresolvedQueryExpr`] tree. -#[derive(Debug, Error)] -pub enum ResolveTreeError { - /// A column reference did not resolve against its in-scope schema. - #[error("column resolution failed: {0}")] - Resolve(#[from] ResolveError), - /// Deriving the schema of an already-resolved child failed (needed to - /// resolve positional column references against it). - #[error("schema derivation failed: {0}")] - Schema(#[from] QueryExprError), -} - -/// Resolve a whole [`UnresolvedQueryExpr`] tree rooted at `tree` into canonical -/// [`ResolvedQueryExpr`]: binds every `ColumnRef` to a `ColumnId` via the -/// [`SchemaResolver`], then [`canonicalize`](super::canonicalize::canonicalize)s the -/// result. -pub fn resolve_root(tree: &UnresolvedQueryExpr) -> Result { - resolve_root_with_inherited(tree, &[]) -} - -/// [`resolve_root`] with label names inherited from an enclosing scope seeded -/// into the leaf schema, used when re-binding a `BinaryOp` side (issue #52). -fn resolve_root_with_inherited( - tree: &UnresolvedQueryExpr, - inherited: &[String], -) -> Result { - let fallback = SchemaResolver::new().resolve_schema_with_inherited(tree, inherited); - let l3 = resolve(tree, &fallback)?; - Ok(super::canonicalize::canonicalize(l3)) -} - -/// The generic substitution walk: converts children first (bottom-up), then -/// resolves this node's own `ColumnRef`s against the *converted child's* -/// derived output schema — so a `JOIN`'s concatenated schema and a cross- -/// series aggregate's frozen-closed output bind to the right positions. -fn resolve( - tree: &UnresolvedQueryExpr, - fallback: &Schema, -) -> Result { - use super::query_expr::QueryExpr as QE; - Ok(match tree { - QE::Scan { - source, - predicates, - schema, - } => { - let schema = schema.clone().unwrap_or_else(|| fallback.clone()); - let predicates = predicates - .iter() - .map(|Predicate(e)| Ok(Predicate(Rc::new(resolve_expr(e, &schema)?)))) - .collect::, ResolveError>>()?; - QE::Scan { - source: source.clone(), - predicates, - schema, - } - } - - // `PromqlScalarBridge`'s child is a scalar-sub-language node (issue - // #220) sitting at this operator-tree position — resolved through - // `resolve_expr`, same as every other scalar position (`Predicate`, - // `ProjectItem.expr`, …), not the operator walk. In practice it's - // always a `Literal`, which has no `ColumnRef` to resolve, so - // `fallback` is never actually consulted here. - QE::PromqlScalarBridge(inner) => { - QE::PromqlScalarBridge(Rc::new(resolve_expr(inner, fallback)?)) - } - QE::EvalTimestamp => QE::EvalTimestamp, - QE::CurrentTimestamp => QE::CurrentTimestamp, - - QE::PromqlVectorFromScalar(child) => { - QE::PromqlVectorFromScalar(Rc::new(resolve(child, fallback)?)) - } - QE::PromqlScalarFromVector(child) => { - QE::PromqlScalarFromVector(Rc::new(resolve(child, fallback)?)) - } - - QE::PromqlRelabel { dst, value, child } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - QE::PromqlRelabel { - dst: dst.clone(), - value: Rc::new(resolve_expr(value, &child_schema)?), - child: Rc::new(child), - } - } - - QE::PromqlInfoEnrich { selector, child } => QE::PromqlInfoEnrich { - selector: selector.clone(), - child: Rc::new(resolve(child, fallback)?), - }, - - QE::PromqlSeriesSample { by, kind, child } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - QE::PromqlSeriesSample { - by: resolve_group_keys(by, &child_schema)?, - kind: *kind, - child: Rc::new(child), - } - } - - QE::Filter { pred, child } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - QE::Filter { - pred: Predicate(Rc::new(resolve_expr(&pred.0, &child_schema)?)), - child: Rc::new(child), - } - } - - QE::Project { - cols, - qualifier, - child, - } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - let cols = cols - .iter() - .map(|item| -> Result { - Ok(ProjectItem { - alias: item.alias.clone(), - expr: resolve_expr(&item.expr, &child_schema)?, - }) - }) - .collect::, _>>()?; - QE::Project { - cols, - qualifier: qualifier.clone(), - child: Rc::new(child), - } - } - - QE::Aggregate { - reduction, - measures, - output_names, - filters, - having, - child, - } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - let reduction = resolve_reduction(reduction, &child_schema)?; - let measures = measures - .iter() - .map(|m| resolve_agg_intent(m, &child_schema)) - .collect::, ResolveError>>()?; - // A measure filter reads the rows being aggregated, so it binds - // against the child's schema, not the aggregate's output. - let filters = filters - .iter() - .map(|f| { - f.as_ref() - .map(|Predicate(p)| Ok(Predicate(Rc::new(resolve_expr(p, &child_schema)?)))) - .transpose() - }) - .collect::, ResolveError>>()?; - // One canonical spelling of "unfiltered" (empty), so structural - // equality and CSE never split on `[]` versus `[None, None]`. - let filters = if any_measure_filtered(&filters) { - filters - } else { - Vec::new() - }; - let having = having - .as_ref() - .map(|Predicate(h)| -> Result { - let out_schema = aggregate_output_schema( - &child_schema, - &reduction, - &measures, - output_names, - )?; - Ok(Predicate(Rc::new(resolve_expr(h, &out_schema)?))) - }) - .transpose()?; - QE::Aggregate { - reduction, - measures, - output_names: output_names.clone(), - filters, - having, - child: Rc::new(child), - } - } - - QE::Dedup { cols, child } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - QE::Dedup { - cols: resolve_column_refs(cols, &child_schema)?, - child: Rc::new(child), - } - } - - QE::Concat { - children, - discriminator_unique_key, - } => { - let children: Vec<_> = children - .iter() - .map(|c| resolve(c, fallback)) - .collect::, _>>()?; - // No front end asserts this today (issue #228 shipped the - // extension point ahead of a wired call site) — resolved here - // regardless, against the first resolved branch's own output - // schema, exactly the schema `output_schema`'s `Concat` arm - // derives the merged schema from, so a future direct - // `concat_with_discriminator` caller upstream of `resolve_root` - // gets a correctly positional `ConcatDiscriminatorKey` out the - // other side. - let discriminator_unique_key = discriminator_unique_key - .as_ref() - .map(|key| -> Result<_, ResolveTreeError> { - let schema = children - .first() - .ok_or(QueryExprError::EmptyConcat)? - .output_schema()?; - Ok(ConcatDiscriminatorKey::new( - resolve_column_ref(key.discriminator(), &schema)?, - resolve_column_refs(key.inner_key(), &schema)?, - )) - }) - .transpose()?; - QE::Concat { - children, - discriminator_unique_key, - } - } - - QE::Join { - kind, - pred, - left, - right, - } => { - // Each branch is bound independently, same reasoning as `BinaryOp` - // below — different leaves / label sets. - let left = resolve_root_with_inherited(left, &[])?; - let right = resolve_root_with_inherited(right, &[])?; - let mut concat = left.output_schema()?; - concat.columns.extend(right.output_schema()?.columns); - let pred = Predicate(Rc::new(resolve_expr(&pred.0, &concat)?)); - QE::Join { - kind: kind.clone(), - pred, - left: Rc::new(left), - right: Rc::new(right), - } - } - - QE::SetOp { - kind, - all, - left, - right, - } => QE::SetOp { - kind: kind.clone(), - all: *all, - left: Rc::new(resolve_root_with_inherited(left, &[])?), - right: Rc::new(resolve_root_with_inherited(right, &[])?), - }, - - QE::Sort { - keys, - partition_by, - child, - } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - let keys = keys - .iter() - .map(|k| -> Result { - Ok(SortKey { - expr: resolve_expr(&k.expr, &child_schema)?, - ascending: k.ascending, - nulls_first: k.nulls_first, - }) - }) - .collect::, _>>()?; - let partition_by = resolve_group_keys(partition_by, &child_schema)?; - QE::Sort { - keys, - partition_by, - child: Rc::new(child), - } - } - - QE::Limit { n, offset, child } => QE::Limit { - n: *n, - offset: *offset, - child: Rc::new(resolve(child, fallback)?), - }, - - QE::PromqlSubquery { - range, - resolution, - child, - } => QE::PromqlSubquery { - range: *range, - resolution: *resolution, - child: Rc::new(resolve(child, fallback)?), - }, - - QE::TimeRange { range, child } => QE::TimeRange { - range: *range, - child: Rc::new(resolve(child, fallback)?), - }, - - QE::TimeShift { shift, child } => QE::TimeShift { - shift: *shift, - child: Rc::new(resolve(child, fallback)?), - }, - - QE::SQLWindowFunc { - func, - args, - partition_by, - order_by, - frame, - output_name, - child, - } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - let args = args - .iter() - .map(|a| resolve_expr(a, &child_schema)) - .collect::, _>>()?; - let partition_by = resolve_group_keys(partition_by, &child_schema)?; - let order_by = order_by - .iter() - .map(|k| -> Result { - Ok(SortKey { - expr: resolve_expr(&k.expr, &child_schema)?, - ascending: k.ascending, - nulls_first: k.nulls_first, - }) - }) - .collect::, _>>()?; - QE::SQLWindowFunc { - func: func.clone(), - args, - partition_by, - order_by, - frame: frame.clone(), - output_name: output_name.clone(), - child: Rc::new(child), - } - } - - QE::BinaryOp { - op, - lhs, - rhs, - vector_match, - } => { - // A binary op's two sides may scan different metrics with - // different label sets, so each branch resolves against its OWN - // bound schema; but an independently-bound side still has to see - // label names an *enclosing* node references (issue #52). - let own = super::schema_resolver::collect_referenced_columns(tree); - let inherited: Vec = inherited_names(fallback) - .into_iter() - .filter(|n| !own.contains(n)) - .collect(); - QE::BinaryOp { - op: op.clone(), - lhs: Rc::new(resolve_root_with_inherited(lhs, &inherited)?), - rhs: Rc::new(resolve_root_with_inherited(rhs, &inherited)?), - vector_match: vector_match.clone(), - } - } - - // The scalar variants (issue #205) are never reached here directly — - // `resolve` only ever recurses into `child`/operator positions; - // every scalar position (`Predicate`, `ProjectItem.expr`, …) goes - // through `resolve_expr` instead, at the operator arm that owns it. - other @ (QE::Column(_) - | QE::Literal(_) - | QE::Compare { .. } - | QE::BoolAnd(_) - | QE::BoolOr(_) - | QE::Not(_) - | QE::IsNull(_) - | QE::IsNotNull(_) - | QE::Cast { .. } - | QE::InList { .. } - | QE::FunctionCall { .. } - | QE::Arithmetic { .. } - | QE::Case { .. }) => { - unreachable!("resolve reached a scalar QueryExpr variant directly: {other:?}") - } - }) -} - -/// The label names an enclosing scope's schema carries beyond the `(ts, -/// value)` floor. -fn inherited_names(schema: &Schema) -> Vec { - schema - .columns - .iter() - .filter(|c| c.name != "ts" && c.name != "value") - .map(|c| c.name.clone()) - .collect() -} - -/// Resolve a name-based [`GroupKeys`] into positional -/// [`GroupKeys`], preserving its `by`/`without` mode. -fn resolve_group_keys( - keys: &GroupKeys, - schema: &Schema, -) -> Result, ResolveError> { - let ids = resolve_column_refs(keys.keys(), schema)?; - Ok(if keys.is_without() { - GroupKeys::without(ids) - } else { - GroupKeys::by(ids) - }) -} - -/// Resolve a name-based [`Reduction`] into positional -/// [`Reduction`]. -/// -/// Uses [`resolve_group_keys_promql`] rather than the strict -/// [`resolve_group_keys`], unlike every other group-key site in `resolve` -/// (`PromqlSeriesSample.by`, `Sort.partition_by`, `SQLWindowFunc.partition_by`): a key -/// absent from a **closed** schema (e.g. the output of a nested cross-series -/// aggregate that collapsed the label) is provably absent from every row, so -/// PromQL drops it from the grouping rather than rejecting the query (issue -/// #53) — `sum(sum by (group) (m)) by (job)` is the canonical case, `job` -/// absent from the inner aggregate's closed `[group, sum]` output. Applied -/// uniformly to every `Aggregate`, not just PromQL's: SQL's `GROUP BY` keys -/// are always genuinely present (DataFusion validates the plan), so the -/// "drop instead of reject" branch is simply never exercised there — the -/// lenient resolver is a no-op difference for a SQL tree, not a behavior -/// change. -fn resolve_reduction( - reduction: &Reduction, - schema: &Schema, -) -> Result, ResolveError> { - Ok(match reduction { - Reduction::Reduce(by) => { - let ids = resolve_group_keys_promql(by.keys(), schema)?; - Reduction::Reduce(if by.is_without() { - GroupKeys::without(ids) - } else { - GroupKeys::by(ids) - }) - } - Reduction::PerEntity => Reduction::PerEntity, - }) -} - -/// Resolve a name-based [`AggIntent`] into positional -/// [`AggIntent`] — every `col: Option` resolves to -/// `Option` (`None` stays `None`, the sample-value convention); -/// every other field carries straight through unchanged. -fn resolve_agg_intent( - intent: &AggIntent, - schema: &Schema, -) -> Result, ResolveError> { - let col = |c: &Option| -> Result, ResolveError> { - c.as_ref() - .map(|r| resolve_column_ref(r, schema)) - .transpose() - }; - Ok(match intent { - AggIntent::Count { accuracy } => AggIntent::Count { - accuracy: accuracy.clone(), - }, - AggIntent::PearsonCorr { left, right } => AggIntent::PearsonCorr { - left: resolve_column_ref(left, schema)?, - right: resolve_column_ref(right, schema)?, - }, - AggIntent::Sum { col: c } => AggIntent::Sum { col: col(c)? }, - AggIntent::Min { col: c } => AggIntent::Min { col: col(c)? }, - AggIntent::Max { col: c } => AggIntent::Max { col: col(c)? }, - AggIntent::Avg { col: c } => AggIntent::Avg { col: col(c)? }, - AggIntent::StdDev { col: c, population } => AggIntent::StdDev { - col: col(c)?, - population: *population, - }, - AggIntent::Variance { col: c, population } => AggIntent::Variance { - col: col(c)?, - population: *population, - }, - AggIntent::Quantile { - col: c, - q, - accuracy, - } => AggIntent::Quantile { - col: col(c)?, - q: *q, - accuracy: accuracy.clone(), - }, - AggIntent::TopK { k, accuracy } => AggIntent::TopK { - k: *k, - accuracy: accuracy.clone(), - }, - AggIntent::Cardinality { cols, accuracy } => AggIntent::Cardinality { - cols: cols - .iter() - .map(|c| resolve_column_ref(c, schema)) - .collect::>()?, - accuracy: accuracy.clone(), - }, - AggIntent::FrequencyL2 { col: c, accuracy } => AggIntent::FrequencyL2 { - col: col(c)?, - accuracy: accuracy.clone(), - }, - AggIntent::FrequencyEntropy { col: c, accuracy } => AggIntent::FrequencyEntropy { - col: col(c)?, - accuracy: accuracy.clone(), - }, - AggIntent::Rate => AggIntent::Rate, - AggIntent::IRate => AggIntent::IRate, - AggIntent::Increase => AggIntent::Increase, - AggIntent::Changes => AggIntent::Changes, - AggIntent::Delta => AggIntent::Delta, - AggIntent::IDelta => AggIntent::IDelta, - AggIntent::Deriv => AggIntent::Deriv, - AggIntent::Resets => AggIntent::Resets, - AggIntent::PredictLinear { seconds } => AggIntent::PredictLinear { seconds: *seconds }, - AggIntent::DoubleExpSmoothing { smoothing, trend } => AggIntent::DoubleExpSmoothing { - smoothing: *smoothing, - trend: *trend, - }, - AggIntent::HistogramCount => AggIntent::HistogramCount, - AggIntent::HistogramSum => AggIntent::HistogramSum, - AggIntent::HistogramAvg => AggIntent::HistogramAvg, - AggIntent::HistogramStdDev => AggIntent::HistogramStdDev, - AggIntent::HistogramStdVar => AggIntent::HistogramStdVar, - AggIntent::HistogramFraction { lower, upper } => AggIntent::HistogramFraction { - lower: *lower, - upper: *upper, - }, - AggIntent::HistogramQuantile { q, le } => AggIntent::HistogramQuantile { - q: *q, - le: resolve_column_ref(le, schema)?, - }, - AggIntent::Math(f) => AggIntent::Math(f.clone()), - AggIntent::Absent => AggIntent::Absent, - AggIntent::AbsentOverTime => AggIntent::AbsentOverTime, - AggIntent::PresentOverTime => AggIntent::PresentOverTime, - AggIntent::TimeFn(f) => AggIntent::TimeFn(*f), - AggIntent::Group => AggIntent::Group, - AggIntent::CountValues { label } => AggIntent::CountValues { - label: label.clone(), - }, - AggIntent::LastOverTime => AggIntent::LastOverTime, - AggIntent::FirstOverTime => AggIntent::FirstOverTime, - AggIntent::MadOverTime => AggIntent::MadOverTime, - AggIntent::TsOfMinOverTime => AggIntent::TsOfMinOverTime, - AggIntent::TsOfMaxOverTime => AggIntent::TsOfMaxOverTime, - AggIntent::TsOfFirstOverTime => AggIntent::TsOfFirstOverTime, - AggIntent::TsOfLastOverTime => AggIntent::TsOfLastOverTime, - AggIntent::Extension { ext_kind, payload } => AggIntent::Extension { - ext_kind: ext_kind.clone(), - payload: payload.clone(), - }, - }) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::pre_asap::expr_ir::CompareOpKind; - use crate::pre_asap::query_expr::{ - BinaryOpKind, QueryExpr, Source, VectorMatch, VectorMatchKind, - }; - - // A measure filter (#466) binds positionally against the aggregate's - // input, and a vector with no set entry collapses to the empty spelling. - #[test] - fn resolve_measure_filters_against_the_child_schema() { - use crate::pre_asap::expr_ir::ScalarValue; - use crate::pre_asap::query_expr::Predicate; - use crate::pre_asap::{Column, DataType, GroupKeys}; - use crate::types::AccuracyTarget; - let scan = || UnresolvedQueryExpr::Scan { - source: Source::Table { - table_ref: "metrics".into(), - }, - predicates: vec![], - schema: Some(Schema::new(vec![ - Column::new("service", DataType::Utf8, false), - Column::new("latency", DataType::Float64, false), - Column::new("bytes", DataType::Int64, false), - ])), - }; - let aggregate = |filters| UnresolvedQueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::by(vec![ColumnRef::Named("service".into())])), - measures: vec![ - AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }, - AggIntent::Sum { - col: Some(ColumnRef::Named("bytes".into())), - }, - ], - output_names: vec![], - filters, - having: None, - child: Rc::new(scan()), - }; - let latency_gt_one = Predicate(Rc::new(UnresolvedQueryExpr::Compare { - left: Rc::new(UnresolvedQueryExpr::Column(ColumnRef::Named( - "latency".into(), - ))), - op: CompareOpKind::Gt, - right: Rc::new(UnresolvedQueryExpr::Literal(ScalarValue::Float64(1.0))), - })); - - let resolved = resolve_root(&aggregate(vec![Some(latency_gt_one), None])).unwrap(); - let QueryExpr::Aggregate { filters, .. } = &resolved else { - unreachable!() - }; - let [Some(Predicate(first)), None] = filters.as_slice() else { - panic!("expected one filtered and one unfiltered measure, got {filters:?}"); - }; - assert!( - matches!(first.as_ref(), QueryExpr::Compare { left, .. } - if matches!(left.as_ref(), QueryExpr::Column(1))), - "latency is input column 1, got {first:?}" - ); - - let resolved = resolve_root(&aggregate(vec![None, None])).unwrap(); - let QueryExpr::Aggregate { filters, .. } = &resolved else { - unreachable!() - }; - assert!(filters.is_empty()); - } - - // Both sides resolve with qualifiers; an unknown right input is an error. - #[test] - fn resolve_pearson_corr_inputs() { - use crate::pre_asap::{Column, DataType}; - let schema = Schema::new(vec![ - Column::new("x", DataType::Float64, true).with_table("a"), - Column::new("x", DataType::Float64, true).with_table("b"), - ]); - let intent = AggIntent::PearsonCorr { - left: ColumnRef::Qualified { - table: "a".into(), - name: "x".into(), - }, - right: ColumnRef::Qualified { - table: "b".into(), - name: "x".into(), - }, - }; - assert_eq!( - resolve_agg_intent(&intent, &schema).unwrap(), - AggIntent::PearsonCorr { left: 0, right: 1 } - ); - let missing = AggIntent::PearsonCorr { - left: ColumnRef::Qualified { - table: "a".into(), - name: "x".into(), - }, - right: ColumnRef::Named("missing".into()), - }; - assert!(resolve_agg_intent(&missing, &schema).is_err()); - } - - // Every leg resolves independently, qualifiers included; one unknown leg - // fails rather than silently shortening the tuple. - #[test] - fn resolve_distinct_tuple_columns() { - use crate::pre_asap::{Column, DataType}; - use crate::types::AccuracyTarget; - let schema = Schema::new(vec![ - Column::new("k", DataType::Int64, true).with_table("a"), - Column::new("k", DataType::Int64, true).with_table("b"), - ]); - let qualified = |table: &str| ColumnRef::Qualified { - table: table.into(), - name: "k".into(), - }; - let intent = AggIntent::Cardinality { - cols: vec![qualified("b"), qualified("a")], - accuracy: AccuracyTarget::Exact, - }; - assert_eq!( - resolve_agg_intent(&intent, &schema).unwrap(), - AggIntent::Cardinality { - cols: vec![1, 0], - accuracy: AccuracyTarget::Exact, - } - ); - let missing = AggIntent::Cardinality { - cols: vec![qualified("a"), ColumnRef::Named("missing".into())], - accuracy: AccuracyTarget::Exact, - }; - assert!(resolve_agg_intent(&missing, &schema).is_err()); - } - - /// `resolve_root` over a `BinaryOp { , PromqlScalarBridge, vector_match }` - /// (issue #220): the bridged scalar operand resolves through the same - /// generic walk as every other node (its `Literal` child has no - /// `ColumnRef` to resolve, so it comes through unchanged), the vector - /// side's `ColumnRef`s resolve positionally, and the `VectorMatch` - /// modifier on the relational binary-op path survives resolution - /// untouched — Instance 2 of #220 (`BinaryOp` vs `Compare`/`Arithmetic`) - /// is out of scope for this change, so this pins that its behavior is - /// unaffected by the Instance-1 collapse. - #[test] - fn resolve_root_threads_a_scalar_bridge_operand_and_preserves_vector_match() { - let vm = VectorMatch { - kind: VectorMatchKind::Ignoring, - labels: vec!["job".into()], - grouping: None, - }; - let unresolved: UnresolvedQueryExpr = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Gt), - lhs: Rc::new(UnresolvedQueryExpr::Scan { - source: Source::TimeSeries { - metric: "up".into(), - }, - predicates: vec![], - schema: None, - }), - rhs: Rc::new(UnresolvedQueryExpr::promql_scalar(1.0)), - vector_match: Some(vm.clone()), - }; - - let resolved = resolve_root(&unresolved).expect("resolves"); - let QueryExpr::BinaryOp { - lhs, - rhs, - vector_match, - .. - } = &resolved - else { - panic!("expected a resolved BinaryOp, got {resolved:?}"); - }; - assert!(matches!(lhs.as_ref(), QueryExpr::Scan { .. })); - assert_eq!(rhs.as_promql_scalar(), Some(1.0)); - assert_eq!(vector_match.as_ref(), Some(&vm)); - - // Schema derivation still follows the vector side post-resolution. - assert_eq!( - resolved.output_schema().unwrap(), - lhs.output_schema().unwrap() - ); - } - - /// Issue #228 review, end-to-end: `resolve_root` over a `Concat` whose - /// discriminator column is referenced *nowhere else* in the tree, with a - /// schema-less (usage-derived) leaf `Scan` in the first branch — exactly - /// the scenario the review flagged. Before the `schema_resolver.rs` fix, the - /// SchemaResolver's fallback schema wouldn't contain `phi` at all, and this - /// `resolve_column_ref` call would fail `NotFound` for a column the - /// caller correctly named. It must resolve cleanly, and the resolved - /// `ConcatDiscriminatorKey` must carry the *positional* `ColumnId`s of - /// the branch's own (usage-derived) schema. - #[test] - fn resolve_root_seeds_and_resolves_an_otherwise_unreferenced_discriminator_column() { - let branch = || UnresolvedQueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: None, - }; - let unresolved = UnresolvedQueryExpr::concat_with_discriminator( - vec![branch(), branch()], - ColumnRef::Named("phi".into()), - vec![ColumnRef::Named("host".into())], - ); - - let resolved = resolve_root(&unresolved).expect("resolves"); - let QueryExpr::Concat { - children, - discriminator_unique_key, - } = &resolved - else { - panic!("expected a resolved Concat, got {resolved:?}"); - }; - let schema = children[0].output_schema().unwrap(); - let key = discriminator_unique_key - .as_ref() - .expect("discriminator key survives resolution"); - assert_eq!(*key.discriminator(), schema.column_id("phi").unwrap()); - assert_eq!( - key.inner_key().to_vec(), - vec![schema.column_id("host").unwrap()] - ); - } -} diff --git a/crates/types/src/pre_asap/scalar_signature.rs b/crates/types/src/pre_asap/scalar_signature.rs index 7d61f6247..988422ab4 100644 --- a/crates/types/src/pre_asap/scalar_signature.rs +++ b/crates/types/src/pre_asap/scalar_signature.rs @@ -123,6 +123,21 @@ fn common_type(left: &DataType, right: &DataType) -> Result { )) } +/// Closed, namespaced contracts for PromQL pointwise float functions. +/// Date functions consume Unix seconds; `timestamp` remains a sample-selection +/// operation because its operand is a sample timestamp rather than its value. +pub fn promql_function_arity(name: &str) -> Option { + Some(match name.strip_prefix("promql_")? { + "abs" | "ceil" | "floor" | "exp" | "ln" | "log2" | "log10" | "sqrt" | "sgn" | "sin" + | "cos" | "tan" | "asin" | "acos" | "atan" | "sinh" | "cosh" | "tanh" | "asinh" + | "acosh" | "atanh" | "deg" | "rad" | "minute" | "hour" | "day_of_week" + | "day_of_month" | "day_of_year" | "month" | "year" | "days_in_month" => 1, + "round" | "clamp_min" | "clamp_max" => 2, + "clamp" => 3, + _ => return None, + }) +} + #[cfg(test)] mod tests { use super::*; @@ -196,309 +211,3 @@ mod tests { .is_err()); } } - -#[cfg(test)] -mod projection_tests { - use super::*; - use crate::pre_asap::{Column, ProjectItem, QueryExpr, ScalarValue, Schema, Source}; - use std::rc::Rc; - fn project(expr: QueryExpr) -> QueryExpr { - QueryExpr::Project { - cols: vec![ProjectItem { - alias: Some("result".into()), - expr, - }], - qualifier: None, - child: Rc::new(QueryExpr::Scan { - source: Source::Table { - table_ref: "t".into(), - }, - predicates: vec![], - schema: Schema::new(vec![ - Column::new("k", DataType::Utf8, false), - Column::new("v", DataType::Int64, true), - ]), - }), - } - } - #[test] - fn canonical_projection_uses_map_signature_and_rejects_invalid_arity() { - let map = QueryExpr::FunctionCall { - name: "map".into(), - args: vec![QueryExpr::Column(0), QueryExpr::Column(1)], - }; - let schema = project(map.clone()).output_schema().unwrap(); - assert_eq!( - schema.columns[0].dtype, - DataType::Map { - key: Box::new(DataType::Utf8), - value: Box::new(DataType::Int64), - value_nullable: true - } - ); - assert!(!schema.columns[0].nullable); - let lookup = QueryExpr::FunctionCall { - name: "asap_map_access".into(), - args: vec![map, QueryExpr::Literal(ScalarValue::Utf8("missing".into()))], - }; - assert_eq!( - project(lookup).output_schema().unwrap().columns[0], - Column::new("result", DataType::Int64, true) - ); - assert!(project(QueryExpr::FunctionCall { - name: "map".into(), - args: vec![QueryExpr::Column(0)] - }) - .output_schema() - .is_err()); - } -} - -/// Resolve the bounded canonical `asap_struct_field(struct, selector)` operation. -/// Selectors are positive 1-based literal ordinals or exact literal field names. -/// The existing Struct fields remain the sole authority for type/nullability. -/// Dynamic/negative/defaulted selectors and nullable containers are intentionally -/// unsupported here; this is not a claim of complete native tupleElement support. -pub fn struct_field_type( - args: &[super::QueryExpr], - schema: &super::Schema, -) -> Result<(DataType, bool), String> { - use super::{QueryExpr, ScalarValue}; - let [input, selector] = args else { - return Err("struct field access requires a struct and constant selector".into()); - }; - let (dtype, nullable) = input - .scalar_type(schema) - .map_err(|error| error.to_string())?; - if nullable { - return Err("nullable struct container access is unsupported".into()); - } - let DataType::Struct { fields } = dtype else { - return Err("struct field access requires a Struct input".into()); - }; - let field = match selector { - QueryExpr::Literal(ScalarValue::Int64(index)) if *index > 0 => usize::try_from(*index - 1) - .ok() - .and_then(|index| fields.get(index)) - .ok_or("struct field ordinal is out of bounds")?, - QueryExpr::Literal(ScalarValue::Utf8(name)) => { - let mut matches = fields.iter().filter(|field| field.name == *name); - let field = matches.next().ok_or("struct field name does not exist")?; - if matches.next().is_some() { - return Err("struct field name is ambiguous".into()); - } - field - } - _ => { - return Err( - "struct field selector must be a positive ordinal or field-name literal".into(), - ) - } - }; - Ok((field.dtype.clone(), field.nullable)) -} - -#[cfg(test)] -mod struct_field_tests { - use super::*; - use crate::pre_asap::{Column, QueryExpr, ScalarValue, Schema}; - fn schema() -> Schema { - Schema::new(vec![Column::new( - "record", - DataType::Struct { - fields: vec![ - Column::new("ts", DataType::Int64, false), - Column::new( - "values", - DataType::List { - element: Box::new(Column::new("item", DataType::Float64, true)), - }, - true, - ), - ], - }, - false, - )]) - } - fn access(selector: QueryExpr) -> QueryExpr { - QueryExpr::FunctionCall { - name: "asap_struct_field".into(), - args: vec![QueryExpr::Column(0), selector], - } - } - #[test] - fn field_access_reuses_nested_field_type_and_nullability() { - let schema = schema(); - assert_eq!( - access(QueryExpr::Literal(ScalarValue::Int64(1))) - .scalar_type(&schema) - .unwrap(), - (DataType::Int64, false) - ); - let named = access(QueryExpr::Literal(ScalarValue::Utf8("values".into()))); - let ordinal = access(QueryExpr::Literal(ScalarValue::Int64(2))); - assert_eq!( - named.scalar_type(&schema).unwrap(), - ordinal.scalar_type(&schema).unwrap() - ); - assert_eq!( - named.scalar_type(&schema).unwrap(), - ( - DataType::List { - element: Box::new(Column::new("item", DataType::Float64, true)) - }, - true - ) - ); - let roundtrip: QueryExpr = - serde_json::from_str(&serde_json::to_string(&named).unwrap()).unwrap(); - assert_eq!(roundtrip, named); - } - #[test] - fn unsupported_field_access_is_an_error_not_placeholder_typing() { - for selector in [ - QueryExpr::Column(0), - QueryExpr::Literal(ScalarValue::Int64(0)), - QueryExpr::Literal(ScalarValue::Int64(-1)), - QueryExpr::Literal(ScalarValue::Int64(3)), - QueryExpr::Literal(ScalarValue::Utf8("missing".into())), - ] { - assert!(access(selector).scalar_type(&schema()).is_err()); - } - let mut ambiguous = schema(); - if let DataType::Struct { fields } = &mut ambiguous.columns[0].dtype { - fields.push(Column::new("ts", DataType::Utf8, false)); - } - assert!(access(QueryExpr::Literal(ScalarValue::Utf8("ts".into()))) - .scalar_type(&ambiguous) - .is_err()); - let mut nullable = schema(); - nullable.columns[0].nullable = true; - assert!(access(QueryExpr::Literal(ScalarValue::Int64(1))) - .scalar_type(&nullable) - .is_err()); - } -} - -/// Canonical element lookup over a declared Map or List. Map lookup retains its -/// existing key/default contract. List lookup is one-based, supports negative -/// indices, and returns the declared element default when a dynamic index is -/// out of range. Literal zero is conservatively rejected because native array -/// behavior depends on whether the input array is constant. Nullable containers -/// are unsupported; nullable indices produce nullable results. -pub fn element_access_type( - args: &[super::QueryExpr], - schema: &super::Schema, -) -> Result<(DataType, bool), String> { - use super::{QueryExpr, ScalarValue}; - let [input, index] = args else { - return Err("element access requires a collection and index".into()); - }; - let source = input.scalar_type(schema).map_err(|e| e.to_string())?; - let key = index.scalar_type(schema).map_err(|e| e.to_string())?; - match &source.0 { - DataType::Map { .. } => MapScalarFunction::Access.output_type(&[source, key]), - DataType::List { element } => { - if source.1 { - return Err("nullable List container access is unsupported".into()); - } - if !matches!(key.0, DataType::Int64 | DataType::Null) { - return Err("List index must have integer type".into()); - } - if matches!(index, QueryExpr::Literal(ScalarValue::Int64(0))) { - return Err( - "literal zero List index is unsupported without constant-array proof".into(), - ); - } - Ok(( - element.dtype.clone(), - element.nullable || key.1 || key.0 == DataType::Null, - )) - } - _ => Err("element access requires a Map or List".into()), - } -} - -#[cfg(test)] -mod element_access_tests { - use super::*; - use crate::pre_asap::{Column, QueryExpr, ScalarValue, Schema}; - fn access(index: QueryExpr) -> QueryExpr { - QueryExpr::FunctionCall { - name: "asap_element_access".into(), - args: vec![QueryExpr::Column(0), index], - } - } - #[test] - fn list_index_preserves_nested_element_metadata() { - let element = DataType::Struct { - fields: vec![ - Column::new("ts", DataType::Int64, false), - Column::new("value", DataType::Float64, true), - ], - }; - let schema = Schema::new(vec![ - Column::new( - "samples", - DataType::List { - element: Box::new(Column::new("item", element.clone(), false)), - }, - false, - ), - Column::new("i", DataType::Int64, true), - ]); - for index in [1, -1, 100] { - assert_eq!( - access(QueryExpr::Literal(ScalarValue::Int64(index))) - .scalar_type(&schema) - .unwrap(), - (element.clone(), false) - ); - } - assert_eq!( - access(QueryExpr::Column(1)).scalar_type(&schema).unwrap(), - (element.clone(), true) - ); - assert!(access(QueryExpr::Literal(ScalarValue::Int64(0))) - .scalar_type(&schema) - .is_err()); - assert!(access(QueryExpr::Literal(ScalarValue::Float64(1.0))) - .scalar_type(&schema) - .is_err()); - let nested = QueryExpr::FunctionCall { - name: "asap_struct_field".into(), - args: vec![ - access(QueryExpr::Literal(ScalarValue::Int64(1))), - QueryExpr::Literal(ScalarValue::Int64(2)), - ], - }; - assert_eq!( - nested.scalar_type(&schema).unwrap(), - (DataType::Float64, true) - ); - let roundtrip: QueryExpr = - serde_json::from_value(serde_json::to_value(&nested).unwrap()).unwrap(); - assert_eq!(roundtrip, nested); - } - #[test] - fn generic_map_lookup_reuses_legacy_signature() { - let schema = Schema::new(vec![Column::new( - "m", - DataType::Map { - key: Box::new(DataType::Utf8), - value: Box::new(DataType::Int64), - value_nullable: false, - }, - false, - )]); - let key = QueryExpr::Literal(ScalarValue::Utf8("k".into())); - let legacy = QueryExpr::FunctionCall { - name: "asap_map_access".into(), - args: vec![QueryExpr::Column(0), key.clone()], - }; - assert_eq!( - access(key).scalar_type(&schema).unwrap(), - legacy.scalar_type(&schema).unwrap() - ); - } -} diff --git a/crates/types/src/pre_asap/schema.rs b/crates/types/src/pre_asap/schema.rs index 9d5b2e226..d5c5366ea 100644 --- a/crates/types/src/pre_asap/schema.rs +++ b/crates/types/src/pre_asap/schema.rs @@ -1,56 +1,60 @@ -//! Pre-ASAP IR schema flow — every edge carries a typed `Schema`. +//! Per-edge schema of the operator IR — every edge carries a typed `Schema`. //! -//! Per `control_plane/docs/design.md` §6 "Schema flow — every L3 edge carries -//! a typed schema" (that doc's own layer numbering; this crate no longer uses -//! it). The DAG is type-checked: a node's output schema is a function of its -//! inputs and parameters and is verifiable independently of the surrounding -//! context. +//! One `Schema` type serves every operator, before and after ASAP +//! optimization: a field's [`FieldDataType`] is either a plain readable +//! [`DataType`] or the summary / exact-accumulator state a `SummaryAgg` +//! produces. The DAG is type-checked: a node's output schema is a function of +//! its inputs and parameters and is verifiable independently of the +//! surrounding context. //! //! `Schema::unique_keys` is metadata for reuse-aware planning: a producer's //! output can only be safely shared across consumers when its row identity //! is provably stable across reads, which is what this field records. -//! -//! Single-query plans don't read this field; it lives here so the metadata -//! is available the moment workload-aware planning lands without requiring -//! a pre-ASAP-IR-wide schema change. #![allow(dead_code)] use serde::{Deserialize, Serialize}; -/// Index into [`Schema::columns`] used everywhere a column position is +use crate::post_asap::sketch::{ + ExactKind, ExactParams, GroupingStrategy, SamplingKind, SamplingParams, SketchKind, + StatModelKind, StatModelParams, WaveletKind, WaveletParams, +}; + +/// Index into [`Schema::fields`] used everywhere a column position is /// referenced (group-by keys, unique-key sets, the time axis index). /// -/// Aliased to `usize` to match `design.md`'s `Vec>` for -/// `unique_keys`. Kept as a named type so downstream code can pattern on -/// the intent ("this is a column position, not just any number"). +/// Kept as a named type so downstream code can pattern on the intent ("this +/// is a column position, not just any number"). pub type ColumnId = usize; -/// One column in a [`Schema`]. Mirrors `design.md` §6 `Field` — -/// `name + dtype + nullable`. +/// One field of a [`Schema`]: `name + dtype + nullable`, plus an optional +/// table qualifier. The struct describes a column and holds none of its data. +/// +/// `T` is the type vocabulary: [`FieldDataType`] on an operator edge (the +/// default, and what [`Schema::fields`] holds), plain [`DataType`] for the +/// nested element fields of [`DataType::List`] / [`DataType::Struct`], which +/// can never carry summary state. #[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] -pub struct Column { - /// Column name as it appears in the producer's output. PromQL leaves +pub struct Field { + /// Field name as it appears in the producer's output. PromQL leaves /// produce label-name + the synthetic `value` / `timestamp` columns; /// SQL leaves carry their `information_schema` names. pub name: String, - /// Logical scalar or collection type. Sketch state remains a post-ASAP - /// concern and is intentionally absent here. - pub dtype: DataType, - /// Whether NULL values are allowed in this column. PromQL value + pub dtype: T, + /// Whether NULL values are allowed in this field. PromQL value /// columns are non-nullable; SQL columns inherit their DDL nullability. pub nullable: bool, /// Optional table/alias qualifier (SQL `t.col` / `t AS a` → `a`). Travels - /// with the column through joins so a `ColumnRef::Qualified` can pick the + /// with the field through joins so a `ColumnRef::Qualified` can pick the /// right side when both carry the same `name`. `None` for PromQL labels and /// unqualified columns. #[serde(default)] pub table: Option, } -impl Column { - /// An unqualified column (`table = None`). - pub fn new(name: impl Into, dtype: DataType, nullable: bool) -> Self { +impl Field { + /// An unqualified field (`table = None`). + pub fn new(name: impl Into, dtype: T, nullable: bool) -> Self { Self { name: name.into(), dtype, @@ -59,16 +63,113 @@ impl Column { } } - /// This column re-qualified under `table` (e.g. by a `SubqueryAlias`). + /// This field re-qualified under `table` (e.g. by a `SubqueryAlias`). pub fn with_table(mut self, table: impl Into) -> Self { self.table = Some(table.into()); self } } -/// Pre-ASAP IR column data types. Deliberately narrow: no sketch state at -/// this layer (see `design.md` §6.4 for the post-ASAP `DataType::Sketch(...)` -/// extension). +impl Field { + /// An unqualified field carrying an ordinary readable value. + pub fn plain(name: impl Into, dtype: DataType, nullable: bool) -> Self { + Self::new(name, FieldDataType::Plain(dtype), nullable) + } + + /// The value type of a plain field; `None` for summary / accumulator state. + pub fn plain_dtype(&self) -> Option<&DataType> { + match &self.dtype { + FieldDataType::Plain(dtype) => Some(dtype), + _ => None, + } + } + + /// Whether this field carries an ordinary readable value. + pub fn is_plain(&self) -> bool { + matches!(self.dtype, FieldDataType::Plain(_)) + } + + /// The value type of a plain field; panics on summary / accumulator + /// state. For code that has already established the field is plain + /// (front ends, scalar type inference over value columns). + pub fn expect_plain_dtype(&self) -> &DataType { + self.plain_dtype().unwrap_or_else(|| { + panic!( + "field `{}` carries summary state ({:?}), not a plain value", + self.name, self.dtype + ) + }) + } +} + +impl From> for Field { + fn from(field: Field) -> Self { + Self { + name: field.name, + dtype: FieldDataType::Plain(field.dtype), + nullable: field.nullable, + table: field.table, + } + } +} + +/// What a schema field carries: an ordinary readable value, or the summary / +/// exact-accumulator state produced by a `SummaryAgg`. +/// +/// Every non-`Plain` variant carries the physical state identity required by +/// that family (`Sketch` additionally carries its grouping layout), so the +/// type system can reject merges of incompatible summaries at plan +/// construction time — a `SummaryMerge` over `Sketch(Kll, …)` and +/// `Sketch(Cms, …)` inputs is a plan-time error, and a `Sketch(…)` can never +/// be confused for a `Sample(…)` even though both are "opaque summary state" +/// at a glance. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum FieldDataType { + /// An ordinary, readable value — the closed vocabulary of [`DataType`]. + Plain(DataType), + /// Exact, mergeable accumulator state (`Sum`/`Count`/`Min`/`Max`/`Rate`/ + /// `Increase`). Value consumers require an explicit finalization boundary. + ExactAggregate(ExactKind, ExactParams), + /// Approximate sketch state (KLL/CMS/HLL/…), read out via a + /// `SummaryEstimate`. A [`SketchKind`] already carries the concrete + /// algorithm, params, and grouping layout committed to, not just its + /// category — a bound node needs to know it's specifically independent + /// KLL or shared Hydra-backed CMS, not merely "some sketch". + Sketch(SketchKind, GroupingStrategy), + /// Sampling-based summary state (a retained row subset). + Sample(SamplingKind, SamplingParams), + /// Wavelet-transform summary state (a coefficient vector). + Wavelet(WaveletKind, WaveletParams), + /// Fitted statistical/parametric-model summary state. + StatModel(StatModelKind, StatModelParams), +} + +impl FieldDataType { + pub fn is_plain(&self) -> bool { + matches!(self, FieldDataType::Plain(_)) + } + + pub fn plain(&self) -> Option<&DataType> { + match self { + FieldDataType::Plain(dtype) => Some(dtype), + _ => None, + } + } +} + +impl From for FieldDataType { + fn from(dtype: DataType) -> Self { + FieldDataType::Plain(dtype) + } +} + +impl PartialEq for FieldDataType { + fn eq(&self, other: &DataType) -> bool { + matches!(self, FieldDataType::Plain(dtype) if dtype == other) + } +} + +/// Plain value types. Summary state is a [`FieldDataType`] concern. #[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] #[serde(rename_all = "snake_case")] pub enum DataType { @@ -97,9 +198,9 @@ pub enum DataType { Date, /// Variable-length sequence. The existing column contract preserves the /// element field name, type, and nullability. Nested fields are unqualified. - List { element: Box }, + List { element: Box> }, /// Ordered named fields, including each field's independent nullability. - Struct { fields: Vec }, + Struct { fields: Vec> }, /// SQL map entries with non-null keys and explicitly nullable values. Map { key: Box, @@ -108,8 +209,8 @@ pub enum DataType { }, } -/// Per-edge pre-ASAP IR schema. Flowing between any two operators, on every -/// node's input and output. +/// Per-edge schema. Flowing between any two operators, on every node's +/// input and output. /// /// `unique_keys` is metadata for reuse-aware planning: each inner `Vec` /// is a set of column indices that together uniquely identify rows. The @@ -117,15 +218,11 @@ pub enum DataType { /// unique constraint). Populated by per-node input/output spec — /// `Aggregate { by, .. }` emits `unique_keys = [by]`; `Dedup { cols }` /// adds `cols`; most other nodes pass through. -/// -/// **Consumed by**: a future workload-level reuse pass (not yet shipped). -/// The single-query path, the `Bind*` rules, push-down, and a deployment's -/// own physical emitters do not read this field. -#[derive(Debug, Clone, PartialEq, Eq, Default, Serialize, Deserialize)] +#[derive(Debug, Clone, PartialEq, Default, Serialize, Deserialize)] pub struct Schema { - /// Columns flowing on this edge, in positional order. - pub columns: Vec, - /// Index into `columns` for the time axis, if any. PromQL leaves + /// Fields flowing on this edge, in positional order. + pub fields: Vec, + /// Index into `fields` for the time axis, if any. PromQL leaves /// always carry one; SQL leaves may or may not. #[serde(default)] pub time_index: Option, @@ -167,90 +264,107 @@ pub const PROMQL_SERIES_IDENTITY: &str = "$promql_series_identity"; /// Operators that rewrite or implicitly match dynamic label sets require their /// own realization; they must not accidentally treat the opaque identity as a /// user label or silently discard it. -pub fn with_promql_series_identity(root: &super::QueryExpr) -> Result { - use super::{QueryExpr, Source}; - use std::rc::Rc; - let mut root = root.clone(); - fn visit(node: &mut QueryExpr) -> Result<(), String> { - match node { - QueryExpr::Scan { +pub fn with_promql_series_identity( + root: &std::rc::Rc, +) -> Result, String> { + use super::Source; + use crate::ir::{NonASAPOp, Operator, OperatorNode}; + use std::{collections::HashMap, rc::Rc}; + fn visit( + node: &Rc, + memo: &mut HashMap<*const OperatorNode, Rc>, + ) -> Result, String> { + if let Some(found) = memo.get(&Rc::as_ptr(node)) { + return Ok(Rc::clone(found)); + } + let mut error = None; + let mut operator = node + .operator + .map_children(|child| match visit(child, memo) { + Ok(child) => child, + Err(e) => { + error = Some(e); + Rc::clone(child) + } + }); + if let Some(error) = error { + return Err(error); + } + match &mut operator { + Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { .. }, schema, .. - } => { + }) => { if schema - .columns + .fields .iter() - .any(|column| column.name == PROMQL_SERIES_IDENTITY) + .any(|field| field.name == PROMQL_SERIES_IDENTITY) { - return Err("source already contains a physical series identity".into()); + if !schema.has_promql_series_identity() { + return Err("invalid physical series identity".into()); + } + memo.insert(Rc::as_ptr(node), Rc::clone(node)); + return Ok(Rc::clone(node)); } if schema.closed { return Err("dynamic series identity requires an open PromQL source".into()); } - schema - .columns - .push(Column::new(PROMQL_SERIES_IDENTITY, DataType::Utf8, false)); + schema.fields.push(Field::new( + PROMQL_SERIES_IDENTITY, + FieldDataType::Plain(DataType::Utf8), + false, + )); schema.closed = true; - Ok(()) - } - QueryExpr::TimeRange { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::PromqlScalarFromVector(child) - | QueryExpr::PromqlRelabel { child, .. } => visit(Rc::make_mut(child)), - // Constants read no series. - QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::Literal(super::ScalarValue::Float64(_)) => Ok(()), - QueryExpr::PromqlVectorFromScalar(child) => visit(Rc::make_mut(child)), - QueryExpr::BinaryOp { lhs, rhs, .. } => { - visit(Rc::make_mut(lhs))?; - visit(Rc::make_mut(rhs)) } - QueryExpr::Concat { children, .. } => { - for child in children { - visit(child)?; - } - Ok(()) + Operator::NonASAP(NonASAPOp::Sort { partition_by, .. }) + if partition_by.is_without() => + { + return Err("dynamic without ranking requires label-set projection".into()); } - QueryExpr::Aggregate { child, .. } => visit(Rc::make_mut(child)), - QueryExpr::Sort { - child, - partition_by, - .. - } => { - if partition_by.is_without() { - return Err("dynamic without ranking requires label-set projection".into()); - } - visit(Rc::make_mut(child)) - } - _ => Err("operator has no dynamic series-identity realization".into()), + Operator::NonASAP( + NonASAPOp::TimeRange { .. } + | NonASAPOp::Limit { .. } + | NonASAPOp::Project { .. } + | NonASAPOp::Filter { .. } + | NonASAPOp::TimeShift { .. } + | NonASAPOp::PromqlSubquery { .. } + | NonASAPOp::PromqlRelabel { .. } + | NonASAPOp::PromqlVectorFromScalar(_) + | NonASAPOp::BinaryOp { .. } + | NonASAPOp::Concat { .. } + | NonASAPOp::Aggregate { .. } + | NonASAPOp::Sort { .. }, + ) => {} + _ => return Err("operator has no dynamic series-identity realization".into()), } + let mut rebuilt = OperatorNode::new(operator).map_err(|e| e.to_string())?; + rebuilt.guarantee = node.guarantee.clone(); + rebuilt.timing = node.timing; + let rebuilt = Rc::new(rebuilt); + memo.insert(Rc::as_ptr(node), Rc::clone(&rebuilt)); + Ok(rebuilt) } - visit(&mut root)?; - root.output_schema().map_err(|error| error.to_string())?; - Ok(root) + visit(root, &mut HashMap::new()) } impl Schema { pub fn has_promql_series_identity(&self) -> bool { self.closed - && self.columns.iter().any(|column| { - column.name == PROMQL_SERIES_IDENTITY - && column.dtype == DataType::Utf8 - && !column.nullable - && column.table.is_none() + && self.fields.iter().any(|field| { + field.name == PROMQL_SERIES_IDENTITY + && field.dtype == DataType::Utf8 + && !field.nullable + && field.table.is_none() }) } - /// Construct a `Schema` from columns alone — no time index, no + /// Construct a `Schema` from fields alone — no time index, no /// unique-key constraint. Used by `Scan` over a tabular source /// when the catalog supplies no primary-key metadata. - pub fn new(columns: Vec) -> Self { + pub fn new(fields: Vec) -> Self { Self { - columns, + fields, time_index: None, unique_keys: Vec::new(), closed: false, @@ -260,42 +374,59 @@ impl Schema { /// Construct a `Scan`-style schema with explicit `time_index` + /// inferred unique keys (e.g. PromQL leaves: `[time_index, label_set]`). pub fn with_time_index( - columns: Vec, + fields: Vec, time_index: ColumnId, unique_keys: Vec>, ) -> Self { Self { - columns, + fields, time_index: Some(time_index), unique_keys, closed: false, } } - /// Look up a column by name (first match). `None` if not present. + /// The schema of a summary-planning node: `fields` and a time axis, no + /// unique-key claim, closed. The shape every post-ASAP operator output + /// carried before pre- and post-ASAP schemas were one type. + pub fn lifted(fields: Vec, time_index: Option) -> Self { + Self { + fields, + time_index, + unique_keys: Vec::new(), + closed: true, + } + } + + /// Whether every field carries an ordinary readable value. + pub fn is_all_plain(&self) -> bool { + self.fields.iter().all(Field::is_plain) + } + + /// Look up a field by name (first match). `None` if not present. pub fn column_id(&self, name: &str) -> Option { - self.columns.iter().position(|c| c.name == name) + self.fields.iter().position(|c| c.name == name) } - /// Look up a column by `(table, name)` qualifier — disambiguates columns - /// that share a `name` across a join (`a.k` vs `b.k`). `None` if no column + /// Look up a field by `(table, name)` qualifier — disambiguates columns + /// that share a `name` across a join (`a.k` vs `b.k`). `None` if no field /// has both that qualifier and name. pub fn column_id_qualified(&self, table: &str, name: &str) -> Option { - self.columns + self.fields .iter() .position(|c| c.name == name && c.table.as_deref() == Some(table)) } /// Whether this schema has *any* provable unique key — the signal a - /// future reuse-aware planning pass would need to decide whether a - /// producer's output can be safely shared across consumers. + /// reuse-aware planning pass needs to decide whether a producer's output + /// can be safely shared across consumers. pub fn has_unique_key(&self) -> bool { !self.unique_keys.is_empty() } /// Append `cols` as an additional unique-key set if not already present. - /// Used by `Dedup { cols }` per design.md §6 schema-flow table: - /// "the input schema with `unique_keys` tightened to include `cols`". + /// Used by `Dedup { cols }`: "the input schema with `unique_keys` + /// tightened to include `cols`". pub fn add_unique_key(&mut self, cols: Vec) { if !self.unique_keys.contains(&cols) { self.unique_keys.push(cols); @@ -309,8 +440,8 @@ impl Schema { mod tests { use super::*; - fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) + fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } #[test] @@ -375,30 +506,33 @@ mod tests { } #[test] - fn column_table_defaults_to_none_when_absent() { - // `Column.table` is `#[serde(default)]` so schemas serialized before the + fn field_table_defaults_to_none_when_absent() { + // `Field.table` is `#[serde(default)]` so schemas serialized before the // qualifier field existed still deserialize (to `table: None`) instead - // of erroring. Drop the key from a serialized column to simulate that. + // of erroring. Drop the key from a serialized field to simulate that. let mut v = serde_json::to_value(col("svc", DataType::Utf8)).unwrap(); assert!(v.as_object_mut().unwrap().remove("table").is_some()); - let back: Column = serde_json::from_value(v).unwrap(); + let back: Field = serde_json::from_value(v).unwrap(); assert_eq!(back, col("svc", DataType::Utf8)); assert!(back.table.is_none()); } #[test] - fn qualified_column_serde_roundtrip() { + fn qualified_field_serde_roundtrip() { let c = col("service", DataType::Utf8).with_table("hosts"); - let back: Column = serde_json::from_str(&serde_json::to_string(&c).unwrap()).unwrap(); + let back: Field = serde_json::from_str(&serde_json::to_string(&c).unwrap()).unwrap(); assert_eq!(back, c); assert_eq!(back.table.as_deref(), Some("hosts")); } - // Direct scalar literals remain valid vector inputs when series typing runs. + + /// A plain field compares equal to its value type; state never does. #[test] - fn series_identity_accepts_direct_vector_literal() { - let root = super::super::QueryExpr::PromqlVectorFromScalar(std::rc::Rc::new( - super::super::QueryExpr::Literal(super::super::ScalarValue::Float64(1.0)), - )); - assert!(with_promql_series_identity(&root).is_ok()); + fn plain_field_type_compares_with_data_type() { + assert!(FieldDataType::Plain(DataType::Utf8) == DataType::Utf8); + assert!(FieldDataType::Plain(DataType::Utf8) != DataType::Int64); + assert_eq!( + col("a", DataType::Int64).plain_dtype(), + Some(&DataType::Int64) + ); } } diff --git a/crates/types/src/pre_asap/schema_resolver.rs b/crates/types/src/pre_asap/schema_resolver.rs deleted file mode 100644 index b8d23b6a3..000000000 --- a/crates/types/src/pre_asap/schema_resolver.rs +++ /dev/null @@ -1,492 +0,0 @@ -//! The **SchemaResolver** — name resolution as an explicit pass. -//! -//! [`SchemaResolver::resolve_schema`] produces the complete, self-contained [`Schema`] every -//! `ColumnId` in the canonical tree indexes into. [`resolve`](super::resolve) -//! then becomes purely structural: it threads the SchemaResolver's schema and -//! positional resolution downstream is **total**. -//! -//! The default [`UsageDerivedCatalog`] knows nothing — every schema is derived -//! purely from the query's own usage. That is the honest state for the -//! observability domain (metric label sets are open-ended). A registry-backed -//! `SchemaCatalog` is future work; the `SchemaResolver` pass does not change when it -//! lands, only the catalog impl swaps. - -use super::expr_ir::ColumnRef; -use super::query_expr::UnresolvedQueryExpr; -use super::schema::{Column, DataType, Schema}; - -/// The DB / source-schema metadata source — resolves a source (metric / -/// table) name to its known columns. -/// Source of truth for a source's columns — the "catalog". `SqlCatalog` backs -/// it for SQL; PromQL uses [`UsageDerivedCatalog`] (returns `None`) until a -/// registry-backed impl (returning a metric's known label set) drops in here. -/// Distinct from `Scan.schema`, which is the *resolved* binding schema this -/// feeds — the catalog is the input, the schema is the result. Even a -/// registry-backed PromQL catalog yields an **open** schema -/// ([`Schema::closed`] `= false`): a metric's -/// labels are per-series and time-varying, so the registry is a superset hint, -/// not a per-row contract. -pub trait SchemaCatalog { - /// Columns known for `source`. `None` when unknown — the [`SchemaResolver`] then - /// falls back to a usage-derived column set. - fn columns_for(&self, source: &str) -> Option>; -} - -/// The default catalog: knows nothing. Every schema the [`SchemaResolver`] produces -/// is derived purely from the query's own usage. -pub struct UsageDerivedCatalog; - -impl SchemaCatalog for UsageDerivedCatalog { - fn columns_for(&self, _source: &str) -> Option> { - None - } -} - -/// The explicit name-resolution pass. -pub struct SchemaResolver { - catalog: C, -} - -impl Default for SchemaResolver { - fn default() -> Self { - Self::new() - } -} - -impl SchemaResolver { - pub fn new() -> Self { - Self { - catalog: UsageDerivedCatalog, - } - } -} - -impl SchemaResolver { - pub fn with_catalog(catalog: C) -> Self { - Self { catalog } - } - - /// Resolve the complete [`Schema`] in scope for a query rooted at `tree`. - /// - /// Contains the time axis, the synthetic `value` column, and one column - /// per distinct name referenced anywhere in the tree — so positional - /// `ColumnId` resolution downstream is total. - pub fn resolve_schema(&self, tree: &UnresolvedQueryExpr) -> Schema { - self.resolve_schema_with_inherited(tree, &[]) - } - - /// Like [`resolve_schema`](Self::resolve_schema), but also seeds `inherited` label names that are - /// referenced by an **enclosing** scope rather than by `tree` itself. This is - /// how an independently-bound `BinaryOp` side (each side re-binds against its - /// own sub-tree) still sees an outer aggregate's group keys — e.g. the - /// `__name__` / `job` in `sum by (__name__)(a or b)`, which appear in neither - /// side's own matchers (issue #52). - pub fn resolve_schema_with_inherited( - &self, - tree: &UnresolvedQueryExpr, - inherited: &[String], - ) -> Schema { - let mut columns: Vec = leftmost_scan_name(tree) - .and_then(|name| self.catalog.columns_for(name)) - .unwrap_or_else(default_leaf_columns); - - // Ensure the (ts, value) floor is present. - for floor in default_leaf_columns() { - if !columns.iter().any(|c| c.name == floor.name) { - columns.push(floor); - } - } - - // Append one column per referenced-but-unknown name (group keys etc.), - // plus any inherited-from-enclosing-scope names. - let referenced = collect_referenced_columns(tree); - for name in referenced.iter().chain(inherited) { - if !columns.iter().any(|c| c.name == *name) { - columns.push(Column::new(name.clone(), DataType::Utf8, true)); - } - } - - let time_index = columns.iter().position(|c| c.name == "ts"); - Schema { - columns, - time_index, - unique_keys: Vec::new(), - // Usage-derived (schemaless PromQL): the metric's full label set is - // open and runtime-only, so this lists only what the query references. - closed: false, - } - } -} - -/// The conventional PromQL leaf shape: `(ts: Timestamp, value: Float64)`. -fn default_leaf_columns() -> Vec { - vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - ] -} - -/// Push a `ColumnRef`'s bare name (the schema-seedable identifier). `Qualified` -/// collapses to its `name`; `SampleValue`/`Wildcard` carry no name. -fn push_ref_name(c: &ColumnRef, out: &mut Vec) { - match c { - ColumnRef::Named(n) => out.push(n.clone()), - ColumnRef::Qualified { name, .. } => out.push(name.clone()), - ColumnRef::SampleValue | ColumnRef::Wildcard => {} - } -} - -/// The leftmost `Scan`'s source name in a canonical (`UnresolvedQueryExpr`) tree — -/// the [`collect_referenced_columns`] counterpart to what a dedicated -/// `Source` leaf type would carry as a method; the canonical tree's `Scan` -/// leaf needs this walk written out instead. -fn leftmost_scan_name(tree: &UnresolvedQueryExpr) -> Option<&str> { - use UnresolvedQueryExpr as QE; - match tree { - QE::Scan { source, .. } => Some(match source { - super::query_expr::Source::TimeSeries { metric } => metric.as_str(), - super::query_expr::Source::Table { table_ref } => table_ref.as_str(), - }), - // A scalar bridge's child is a scalar-sub-language leaf (in practice - // always a `Literal`, issue #220) — never a `Scan`, same as - // `EvalTimestamp`. - QE::PromqlScalarBridge(_) | QE::EvalTimestamp | QE::CurrentTimestamp => None, - QE::PromqlVectorFromScalar(child) | QE::PromqlScalarFromVector(child) => { - leftmost_scan_name(child) - } - QE::PromqlRelabel { child, .. } - | QE::PromqlInfoEnrich { child, .. } - | QE::PromqlSeriesSample { child, .. } - | QE::Filter { child, .. } - | QE::Project { child, .. } - | QE::Aggregate { child, .. } - | QE::Dedup { child, .. } - | QE::Sort { child, .. } - | QE::Limit { child, .. } - | QE::PromqlSubquery { child, .. } - | QE::TimeRange { child, .. } - | QE::TimeShift { child, .. } - | QE::SQLWindowFunc { child, .. } => leftmost_scan_name(child), - QE::Concat { children, .. } => children.first().and_then(leftmost_scan_name), - QE::Join { left, .. } | QE::SetOp { left, .. } | QE::BinaryOp { lhs: left, .. } => { - leftmost_scan_name(left) - } - // The scalar variants (issue #205) never appear as a direct - // `leftmost_scan_name` target — every reachable one sits behind a - // wrapper field (`Predicate`, `ProjectItem`, …) this walk never - // descends into; it only follows the relational skeleton. - QE::Column(_) - | QE::Literal(_) - | QE::Compare { .. } - | QE::BoolAnd(_) - | QE::BoolOr(_) - | QE::Not(_) - | QE::IsNull(_) - | QE::IsNotNull(_) - | QE::Cast { .. } - | QE::InList { .. } - | QE::FunctionCall { .. } - | QE::Arithmetic { .. } - | QE::Case { .. } => None, - } -} - -/// Collect every distinct column name referenced anywhere in `tree` that -/// resolves positionally — every place a front end constructing -/// [`QueryExpr`](super::query_expr::QueryExpr) directly (issue -/// #179) puts a name-based reference: `Scan.predicates`, `Aggregate`'s -/// `reduction`/`having`/per-measure `col`, `Dedup.cols`, `PromqlSeriesSample.by`, -/// `Filter.pred`, `Project.cols`, `Sort.keys`/`partition_by`, -/// `SQLWindowFunc.args`/`partition_by`/`order_by`, `Join.pred`, `PromqlRelabel.value`. -/// The SchemaResolver seeds these into the usage-derived leaf so positional -/// resolution downstream is total. -pub(crate) fn collect_referenced_columns(tree: &UnresolvedQueryExpr) -> Vec { - use UnresolvedQueryExpr as QE; - fn named(expr: &UnresolvedQueryExpr, out: &mut Vec) { - for c in expr.columns_referenced() { - push_ref_name(c, out); - } - } - fn group_keys(g: &super::query_expr::GroupKeys, out: &mut Vec) { - g.keys().iter().for_each(|k| push_ref_name(k, out)); - } - fn measure_cols(measures: &[super::agg_intent::AggIntent], out: &mut Vec) { - for m in measures { - for c in m.input_cols() { - push_ref_name(&c, out); - } - } - } - fn walk(node: &UnresolvedQueryExpr, out: &mut Vec) { - match node { - QE::Scan { predicates, .. } => { - for super::query_expr::Predicate(p) in predicates { - named(p, out); - } - } - QE::Aggregate { - reduction, - measures, - filters, - having, - child, - .. - } => { - if let super::query_expr::Reduction::Reduce(by) = reduction { - group_keys(by, out); - } - measure_cols(measures, out); - for super::query_expr::Predicate(f) in filters.iter().flatten() { - named(f, out); - } - if let Some(super::query_expr::Predicate(h)) = having { - named(h, out); - } - walk(child, out); - } - QE::Dedup { cols, child } => { - cols.iter().for_each(|c| push_ref_name(c, out)); - walk(child, out); - } - QE::PromqlSeriesSample { by, child, .. } => { - group_keys(by, out); - walk(child, out); - } - QE::Filter { pred, child } => { - named(&pred.0, out); - walk(child, out); - } - QE::Project { cols, child, .. } => { - for item in cols { - named(&item.expr, out); - } - walk(child, out); - } - QE::Sort { - keys, - partition_by, - child, - } => { - for k in keys { - named(&k.expr, out); - } - group_keys(partition_by, out); - walk(child, out); - } - QE::SQLWindowFunc { - args, - partition_by, - order_by, - child, - .. - } => { - for a in args { - named(a, out); - } - group_keys(partition_by, out); - for k in order_by { - named(&k.expr, out); - } - walk(child, out); - } - QE::PromqlRelabel { value, child, .. } => { - named(value, out); - walk(child, out); - } - QE::Join { - pred, left, right, .. - } => { - named(&pred.0, out); - walk(left, out); - walk(right, out); - } - QE::EvalTimestamp | QE::CurrentTimestamp => {} - // The bridged child is a genuine scalar-sub-language position now - // (issue #220) — peel its column refs off with `named`, same as - // every other scalar-typed field (`Scan.predicates`, - // `Filter.pred`, …). In practice it's always a `Literal`, which - // references no columns, so this is a no-op today. - QE::PromqlScalarBridge(inner) => named(inner, out), - QE::PromqlVectorFromScalar(child) | QE::PromqlScalarFromVector(child) => { - walk(child, out) - } - QE::PromqlInfoEnrich { child, .. } - | QE::Limit { child, .. } - | QE::PromqlSubquery { child, .. } - | QE::TimeRange { child, .. } - | QE::TimeShift { child, .. } => walk(child, out), - QE::Concat { - children, - discriminator_unique_key, - } => { - // Same treatment as `Dedup.cols` above: an own-field - // `ColumnRef` must be seeded here too, or a discriminator - // column that isn't otherwise referenced anywhere else in - // the tree (plausible — a raw usage-derived label, not one a - // `Project`/relabel freshly created) is absent from the - // SchemaResolver's usage-derived fallback schema, and - // `resolve.rs`'s later `resolve_column_ref` call fails with - // `NotFound` for a column the caller correctly named. - if let Some(key) = discriminator_unique_key { - push_ref_name(key.discriminator(), out); - key.inner_key().iter().for_each(|c| push_ref_name(c, out)); - } - children.iter().for_each(|c| walk(c, out)); - } - QE::SetOp { left, right, .. } => { - walk(left, out); - walk(right, out); - } - QE::BinaryOp { lhs, rhs, .. } => { - walk(lhs, out); - walk(rhs, out); - } - // The scalar variants (issue #205) never appear as a direct - // `walk` target — every reachable one is peeled off first by - // `named` at whichever operator field holds it (`Scan.predicates`, - // `Filter.pred`, `Project.cols`, …). - QE::Column(_) - | QE::Literal(_) - | QE::Compare { .. } - | QE::BoolAnd(_) - | QE::BoolOr(_) - | QE::Not(_) - | QE::IsNull(_) - | QE::IsNotNull(_) - | QE::Cast { .. } - | QE::InList { .. } - | QE::FunctionCall { .. } - | QE::Arithmetic { .. } - | QE::Case { .. } => { - unreachable!("walk reached a scalar QueryExpr variant directly: {node:?}") - } - } - } - let mut out: Vec = Vec::new(); - walk(tree, &mut out); - out.sort(); - out.dedup(); - out -} - -#[cfg(test)] -mod tests { - use std::rc::Rc; - - use super::super::query_expr::{GroupKeys, Source}; - use super::*; - - fn src(name: &str) -> UnresolvedQueryExpr { - UnresolvedQueryExpr::Scan { - source: Source::TimeSeries { - metric: name.into(), - }, - predicates: vec![], - schema: None, - } - } - - // Both correlation inputs must seed a usage-derived schema before positional resolution. - #[test] - fn pearson_corr_inputs_seed_usage_derived_schema() { - use crate::pre_asap::{AggIntent, Reduction}; - let tree = UnresolvedQueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::PearsonCorr { - left: ColumnRef::Named("x".into()), - right: ColumnRef::Named("y".into()), - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(src("m")), - }; - assert_eq!(collect_referenced_columns(&tree), vec!["x", "y"]); - let schema = SchemaResolver::new().resolve_schema(&tree); - assert!(schema.column_id("x").is_some()); - assert!(schema.column_id("y").is_some()); - } - - #[test] - fn bare_source_yields_ts_value_floor() { - let schema = SchemaResolver::new().resolve_schema(&src("m")); - assert_eq!(schema.columns.len(), 2); - assert_eq!(schema.columns[0].name, "ts"); - assert_eq!(schema.columns[1].name, "value"); - assert_eq!(schema.time_index, Some(0)); - } - - #[test] - fn sort_partition_keys_land_in_schema() { - // Per-group ranking keys (`topk by (host)` → `Sort.partition_by`) must be - // seeded into the usage-derived leaf so they resolve positionally. - let tree = UnresolvedQueryExpr::Sort { - keys: vec![super::super::query_expr::SortKey { - expr: UnresolvedQueryExpr::Column(ColumnRef::SampleValue), - ascending: false, - nulls_first: false, - }], - partition_by: GroupKeys::by(vec![ColumnRef::Named("host".into())]), - child: Rc::new(src("hits")), - }; - let schema = SchemaResolver::new().resolve_schema(&tree); - assert!(schema.column_id("host").is_some()); - } - - /// Issue #228 review: a `Concat`'s `discriminator_unique_key` columns — - /// even one referenced nowhere else in the tree — must be seeded into - /// the usage-derived fallback schema, exactly like `Dedup.cols`, or - /// `resolve.rs`'s later `resolve_column_ref` fails `NotFound` for a - /// column the caller correctly named. - #[test] - fn concat_discriminator_key_is_seeded_into_the_resolver_schema() { - let tree = UnresolvedQueryExpr::concat_with_discriminator( - vec![src("m")], - ColumnRef::Named("phi".into()), - vec![ColumnRef::Named("host".into())], - ); - let schema = SchemaResolver::new().resolve_schema(&tree); - assert!( - schema.column_id("phi").is_some(), - "discriminator column must be seeded" - ); - assert!( - schema.column_id("host").is_some(), - "inner_key column must be seeded" - ); - } - - #[test] - fn inherited_names_are_seeded_alongside_referenced() { - // A `BinaryOp` side re-binds against its own sub-tree, but must still see - // an enclosing aggregate's group key (`__name__` / `job`) that appears in - // neither side's own matchers (issue #52). `resolve_schema_with_inherited` seeds it. - let schema = - SchemaResolver::new().resolve_schema_with_inherited(&src("m"), &["__name__".into()]); - assert!(schema.column_id("__name__").is_some()); - // `resolve_schema` (no inheritance) does not conjure it. - let plain = SchemaResolver::new().resolve_schema(&src("m")); - assert!(plain.column_id("__name__").is_none()); - } - - #[test] - fn custom_catalog_supplies_base_columns() { - struct FixedCatalog; - impl SchemaCatalog for FixedCatalog { - fn columns_for(&self, source: &str) -> Option> { - (source == "known").then(|| { - vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - Column::new("datacenter", DataType::Utf8, false), - ] - }) - } - } - let schema = SchemaResolver::with_catalog(FixedCatalog).resolve_schema(&src("known")); - let dc = schema - .column_id("datacenter") - .and_then(|id| schema.columns.get(id)); - assert!(matches!(dc, Some(c) if !c.nullable)); - } -} diff --git a/crates/types/tests/planner_vocabulary.rs b/crates/types/tests/planner_vocabulary.rs index 567e2c284..2ba817861 100644 --- a/crates/types/tests/planner_vocabulary.rs +++ b/crates/types/tests/planner_vocabulary.rs @@ -1,7 +1,5 @@ -use asap_types::post_asap::{ - validate_pane_coverage, PaneLayout, WindowEdgeCompatibility, WindowEdgeCoverage, -}; -use asap_types::pre_asap::{SchemaResolver, Source, UnresolvedQueryExpr}; +use asap_types::ir::export::WindowEdgeCompatibility; +use asap_types::post_asap::{validate_pane_coverage, PaneLayout, WindowEdgeCoverage}; use asap_types::resources::{PhysicalHandoffBytes, PhysicalHandoffKind}; // Renamed pane APIs still read and emit the deployed wire contract. @@ -30,18 +28,11 @@ fn window_edge_names_preserve_wire_values() { ); } -// External consumers can use the new resolver and resource names without changing behavior. +// External consumers can use the new resource names without changing behavior. +// (The schema-resolver half moved with the resolver to `asap-frontend-common`; +// `schema_resolver::tests::bare_source_yields_ts_value_floor` covers it.) #[test] -fn renamed_schema_and_handoff_apis_are_public() { - let tree = UnresolvedQueryExpr::Scan { - source: Source::TimeSeries { - metric: "requests".into(), - }, - predicates: vec![], - schema: None, - }; - let schema = SchemaResolver::new().resolve_schema(&tree); - assert!(schema.column_id("value").is_some()); +fn renamed_handoff_apis_are_public() { let bytes = PhysicalHandoffBytes { network_bytes: 12, materialization_bytes: 4, diff --git a/crates/types/tests/structure_contract.rs b/crates/types/tests/structure_contract.rs new file mode 100644 index 000000000..a7171ed14 --- /dev/null +++ b/crates/types/tests/structure_contract.rs @@ -0,0 +1,166 @@ +use asap_types::ir::{ + ExprSemantics, NonASAPOp, OperatorNode, OperatorResultKind, Predicate, ScalarExpr, +}; +use asap_types::pre_asap::{DataType, Field, ScalarValue, Schema, Source}; +use std::rc::Rc; +fn scan() -> Rc { + OperatorNode::non_asap_node(NonASAPOp::Scan { + source: Source::Table { + table_ref: "t".into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("x", DataType::Float64, false)]), + }) + .unwrap() +} +/// Resolved filters cannot hide invalid scalar types or out-of-scope columns. +#[test] +fn invalid_predicates_are_rejected() { + for expr in [ + ScalarExpr::Column(7), + ScalarExpr::literal_f64(1.0), + ScalarExpr::Not(Box::new(ScalarExpr::literal_f64(1.0))), + ] { + let op = NonASAPOp::Filter { + child: scan(), + pred: Predicate(expr), + }; + assert!(op.validate_inputs().is_err()); + } +} +/// Common metadata must agree with the actual operation, including result kind. +#[test] +fn retained_result_kind_is_checked() { + let mut node = (*scan()).clone(); + node.result_kind = OperatorResultKind::RangeVector; + assert!(Rc::new(node).validate_structure().is_err()); +} +/// Values validates arity and declared nullability without inventing columns. +#[test] +fn values_contract_is_checked() { + for row in [ + vec![], + vec![ScalarExpr::Literal(ScalarValue::Null)], + vec![ScalarExpr::Literal(ScalarValue::Utf8("x".into()))], + ] { + let op = NonASAPOp::Values { + rows: vec![row], + schema: scan().schema.clone(), + }; + assert!(op.validate_inputs().is_err()); + } +} +/// Scalar typing validates every branch and never assigns placeholder types. +#[test] +fn scalar_signatures_fail_closed() { + for expr in [ + ScalarExpr::Column(99), + ScalarExpr::FunctionCall { + name: "not_registered".into(), + args: vec![], + }, + ScalarExpr::Negative { + expr: Box::new(ScalarExpr::Literal(ScalarValue::Utf8("x".into()))), + semantics: ExprSemantics::Sql, + }, + ScalarExpr::Case { + operand: None, + branches: vec![(ScalarExpr::literal_f64(1.0), ScalarExpr::literal_f64(2.0))], + else_expr: None, + }, + ScalarExpr::ScalarSubquery( + OperatorNode::non_asap_node(NonASAPOp::Values { + rows: vec![], + schema: Schema::default(), + }) + .unwrap(), + ), + ScalarExpr::PromqlScalarFromVector(scan()), + ] { + assert!(expr.scalar_type(&scan().schema).is_err(), "{expr:?}"); + } +} + +/// A state family is not interchangeable with another sketch or a scalar field. +#[test] +fn state_evaluations_and_passthrough_keep_their_contracts() { + use asap_types::ir::{ASAPOp, Operator, ProjectItem}; + use asap_types::post_asap::{ + FieldDataType, GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, + SketchStatistic, SummaryUpdate, + }; + use asap_types::pre_asap::{ColumnRef, Reduction}; + let family = FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 100 }), + GroupingStrategy::default(), + ); + let state = Rc::new( + OperatorNode::new(Operator::ASAP(ASAPOp::SummaryAgg { + child: scan(), + family, + input: SummaryUpdate::column(ColumnRef::Named("x".into())), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + })) + .unwrap(), + ); + state.validate_structure().unwrap(); + let pass = OperatorNode::non_asap_node(NonASAPOp::Project { + child: state.clone(), + cols: vec![ProjectItem { + expr: ScalarExpr::Column(0), + alias: None, + }], + qualifier: None, + }) + .unwrap(); + pass.validate_structure().unwrap(); + assert_eq!(pass.result_kind, OperatorResultKind::State); + assert!(ScalarExpr::Column(0).scalar_type(&pass.schema).is_err()); + assert!(ASAPOp::SummaryEstimate { + summary_input: state.clone(), + query: SketchStatistic::Cardinality + } + .validate_inputs() + .is_err()); + assert!(ASAPOp::SummaryEstimate { + summary_input: state.clone(), + query: SketchStatistic::Quantile { q: 0.99 } + } + .validate_inputs() + .is_ok()); + assert!(ASAPOp::FinalizeExactAccumulator { child: state } + .validate_inputs() + .is_err()); +} + +/// Phase validation checks dependencies, without declaring a computation query-only. +#[test] +fn execution_timing_checks_edges_not_function_names() { + use asap_types::ir::ProjectItem; + use asap_types::post_asap::ExecutionTiming::{IngestionTime, QueryTime}; + let input = Rc::new((*scan()).clone().with_timing(Some(IngestionTime))); + let mut project = OperatorNode::new(asap_types::ir::Operator::NonASAP(NonASAPOp::Project { + child: input, + cols: vec![ProjectItem { + expr: ScalarExpr::FunctionCall { + name: "promql_abs".into(), + args: vec![ScalarExpr::Column(0)], + }, + alias: None, + }], + qualifier: None, + })) + .unwrap() + .with_timing(Some(IngestionTime)); + Rc::new(project.clone()) + .validate_execution_timing() + .unwrap(); + if let asap_types::ir::Operator::NonASAP(NonASAPOp::Project { child, .. }) = + &mut project.operator + { + *child = Rc::new(child.as_ref().clone().with_timing(Some(QueryTime))); + } + assert!(Rc::new(project).validate_execution_timing().is_err()); +} diff --git a/docs/design_docs/architecture/README.md b/docs/design_docs/architecture/README.md index 8b936eafe..e5f8ff2e7 100644 --- a/docs/design_docs/architecture/README.md +++ b/docs/design_docs/architecture/README.md @@ -17,7 +17,7 @@ flowchart TD W["PlanningWorkload: query demand + optional data facts"] F["Frontend dependencies: SQL catalog or PromQL time"] E["Strategy, accuracy model, and applicable evidence"] - PRE["Frontend lowering → canonical Pre-ASAP QueryExpr roots"] + PRE["Frontend lowering → canonical Pre-ASAP OperatorNode roots"] SEARCH["Whole-workload candidate search: sharing, legality, accuracy"] SPACE["CandidateLogicalASAPDAGs: compact logical candidate DAG space"] RANK["Optional cost_sorted: ranked inspection view"] @@ -58,7 +58,8 @@ an unsupported physical alternative into a deployable plan. | Area | Main crate or module | Responsibility | |---|---|---| -| Shared IR | `asap-types` | Pre-ASAP and Post-ASAP expressions, schemas, workloads, guarantees, and exported plan data | +| Shared IR | `asap-types` | The unified operator IR (`ir`: one `OperatorNode` before and after ASAP optimization), schemas, workloads, guarantees, and exported plan data | +| Front-end common | `frontend-common` | Name-based `UnresolvedOp` tree shared by the front ends, and `resolve_root` into the operator IR | | Query frontends | `frontend-sql`, `frontend-promql`, `frontend-metricsql` | Parse source languages and produce canonical Pre-ASAP queries | | ASAP-aware mapping | `asap-aware-mapping` | Candidate generation, CSE, legality, accuracy propagation, lifecycle expansion, costing, and ranking | | Developer inspection | `devtools` | Expose planner DAGs, alternatives, decisions, and explanations for inspection | diff --git a/docs/design_docs/architecture/asap-aware-mapping.md b/docs/design_docs/architecture/asap-aware-mapping.md index 2600fc42d..78d3e0f9d 100644 --- a/docs/design_docs/architecture/asap-aware-mapping.md +++ b/docs/design_docs/architecture/asap-aware-mapping.md @@ -44,7 +44,7 @@ budgets; deployment belongs to a later stage. - **Replacement Sub-DAG**: A candidate post-ASAP sub-DAG to replace a target sub-DAG. For example, a quantile aggregation may have KLL, DDSketch, and exact aggregation as alternatives. - **ReplacementStrategy**: A rule to recognize a target Sub-DAG and produces one or more valid replacement Sub-DAGs. - **Candidate Plan**: A complete post-ASAP plan formed by choosing compatible replacement alternatives across the plan. -- **Maintained population**: A multiset of qualifying records retained across evaluations and updated as members enter, change, leave or expire; multiple readouts can share this state. +- **Maintained population**: A multiset of qualifying records retained across evaluations and updated as members enter, change, leave or expire; multiple evaluations can share this state. - **Cost Model**: A model used to compare valid candidate plans according to criteria such as storage, update cost, query latency, and accuracy. The distinction between **ReplacementStrategy** and **Candidate Plan** is important. A ReplacementStrategy generates alternatives at a decision point, while a candidate plan is a complete plan that combines choices across all relevant decision points. diff --git a/docs/design_docs/architecture/evidence-dependent-candidates.md b/docs/design_docs/architecture/evidence-dependent-candidates.md index abdfbdb90..9dcbba3fd 100644 --- a/docs/design_docs/architecture/evidence-dependent-candidates.md +++ b/docs/design_docs/architecture/evidence-dependent-candidates.md @@ -15,7 +15,7 @@ target)` returns `true`. **Uncertified** means Planner cannot make that claim: the guarantee is absent, contains unknown terms, or is known not to meet the target. An uncertified summary may still be a well-formed logical candidate; this label says nothing about whether the backend can physically execute it. -The exact `KeepPreAsap` path has an exact guarantee. +The exact path (the pre-ASAP sub-DAG kept by `retain_exact`) has an exact guarantee. | State | Planner representation | Consequence / next step | |---|---|---| @@ -88,7 +88,7 @@ The default `global_selection()` skips summaries that `has_missing_accuracy_evidence()` identifies as uncertified. Its `GlobalSelection::assemble_selected_dag()` result is a selected logical plan, not an instruction to deploy every candidate in `CandidateLogicalASAPDAGs`. If no alternative -is chosen at a site, DAG assembly retains the exact `KeepPreAsap` path. The +is chosen at a site, DAG assembly retains the exact pre-ASAP sub-DAG. The backend can inspect alternatives, apply its own evidence and policy, then choose a physically supported one; it must not equate candidate presence with approval. Models may explicitly opt into qualitative candidate ranking when no @@ -109,7 +109,7 @@ backend. | PromQL input | Before this PR | After this PR | |---|---|---| | `count by(job)(up)` with an ε/δ target | Hydra's shared CMS/CountSketch alternatives are absent: missing shared-grid bounds make the strategy decline the target. | Both Hydra alternatives remain in `CandidateLogicalASAPDAGs` with symbolic unknown bound/probability terms. `has_missing_accuracy_evidence()` is true; default `global_selection()` does not choose either as a certified answer. | -| `entropy_over_time(m[5m])` with an ε target | The uncalibrated frequency readout has no `SummaryEstimate` candidate. | Its `SummaryEstimate` remains inspectable with `guarantee: None`. Default selection still skips it, so candidate visibility is not an accuracy certificate. | +| `entropy_over_time(m[5m])` with an ε target | The uncalibrated frequency evaluation has no `SummaryEstimate` candidate. | Its `SummaryEstimate` remains inspectable with `guarantee: None`. Default selection still skips it, so candidate visibility is not an accuracy certificate. | | `quantile_over_time(0.9,data[5m]) / quantile_over_time(0.5,data[5m])` with an ε target | The uncertified direct DDSketch ratio is **already** visible because of #449. | Still visible with `guarantee: None`, and still skipped by default selection. This is a regression/control example, not a new candidate introduced by this PR. | For the first two rows, the observable change is the alternative set delivered @@ -121,7 +121,7 @@ evidence (for example a failure probability of `1.5`) instead produces a The corresponding reproducible checks are `cargo test -p asap-frontend-promql grouped_count_keeps_uncertified_hydra_candidates_for_backend_review`, -`cargo test -p asap-frontend-promql uncalibrated_frequency_readouts_do_not_bypass_accuracy_targets`, +`cargo test -p asap-frontend-promql uncalibrated_frequency_evaluations_do_not_bypass_accuracy_targets`, and `cargo test -p asap-integration-tests ddsketch_ratio_without_domain_proof_is_uncertified`. All three start from PromQL text and exercise frontend lowering and planning. None runs a deployed query. diff --git a/docs/design_docs/architecture/input-output-workflow.md b/docs/design_docs/architecture/input-output-workflow.md index 65ee43141..4120811ca 100644 --- a/docs/design_docs/architecture/input-output-workflow.md +++ b/docs/design_docs/architecture/input-output-workflow.md @@ -66,12 +66,12 @@ flowchart TD D["data_workload: continuous arrival; declared ingestion interval 15 s"] T["Frontend argument: now_ms"] F["PromQL lowering"] - R["One canonical QueryExpr root"] + R["One canonical OperatorNode root"] S["Candidate search"] P["CandidateLogicalASAPDAGs: logical choices for this root"] I["cost_sorted: inspect choices"] G["global_selection + assemble_selected_dag(root)"] - L["One selected Post-ASAP DAG; exact KeepPreAsap if no optimization is selected"] + L["One selected Post-ASAP DAG; the exact pre-ASAP sub-DAG if no optimization is selected"] X["Extra lifecycle inputs: horizon; update rate; capabilities; comparable summary/raw costs"] H["Summary-maintenance-lifecycle-aware selection"] HM["Assemble one selected DAG and decide summary maintenance"] @@ -102,7 +102,7 @@ flowchart LR Q["query_batch: SELECT COUNT(*) FROM metrics; invocations 1; AdHoc"] C["SqlCatalog: resolves metrics and its columns"] F["SQL lowering"] - R["One QueryExpr root"] + R["One OperatorNode root"] P["Candidate search → CandidateLogicalASAPDAGs"] Q --> F C --> F @@ -214,7 +214,7 @@ fields expand as follows: | `TimeSelection` | `lookback` | Optional event-time duration selected before the upper bound. | | `TimeSelection` | `as_of` | Optional fixed upper-bound timestamp; `None` means planning/evaluation time. | -Frontend lowering produces one Pre-ASAP `QueryExpr` root for each normalized +Frontend lowering produces one Pre-ASAP `Rc` root for each normalized query entry. The caller must retain each root's association with its workload entry for later recurrence and lifecycle planning. @@ -334,7 +334,7 @@ below. A future higher-level API could hide `CandidateLogicalASAPDAGs` behind th the current interface lets an integrator own them. DAG assembly connects choices after selection and does not replace this candidate interface. -Here, a **root** is the top-level `Rc` for a workload query. A +Here, a **root** is the top-level `Rc` for a workload query. A **target** is any discovered sub-DAG that may be replaced, including roots. For `count(up) + 1`, the addition is a root and `count(up)` can be an inner target. `TargetSubDAGCandidates` holds the alternatives for one such target. @@ -361,7 +361,7 @@ All paths start by lowering the workload and searching for candidates: ```text PlanningWorkload + frontend dependencies + planning models/evidence - -> frontend lowering: one QueryExpr root per normalized query entry + -> frontend lowering: one OperatorNode root per normalized query entry -> search_workload_with_targets -> CandidateLogicalASAPDAGs ``` @@ -390,7 +390,7 @@ The return type is `Vec>`; each element has thi ```rust struct RankedTargetSubDAGCandidates<'a> { - target: &'a Rc, + target: &'a Rc, consumer_count: usize, candidates: Vec<&'a ReplacementSubDAG>, costs: Vec, // costs[i] describes candidates[i] @@ -430,7 +430,8 @@ the result for one query root. | **Output:** one selected logical [Post-ASAP DAG](../concepts/post-asap-ir.md) per query root | Each output DAG specifies the chosen operators, parameters, and accuracy -guarantees. Its root is represented by `Rc`; the +guarantees. Its root is an `Rc` (the same IR as the input, +with some nodes now ASAP operators) and carries no execution timing yet; the [API reference](../../develop_docs/library-api.md#api-definition-and-example) describes the function signatures and return handling. @@ -456,7 +457,7 @@ there is no need to run the ordinary selection/assembly workflow first: `Result, SummaryMaintenanceLifecycleAssemblyError>`. When a summary does not beat a known raw cost, or a required comparable cost is unavailable, the result - retains the exact `KeepPreAsap` root and no summary deployments. + retains the exact pre-ASAP root (`retain_exact`) and no summary deployments. As in ordinary selection, one selection call serves the workload and assembly is per root. The second helper calls `assemble_selected_dag` internally; callers @@ -489,7 +490,7 @@ facts remain unknown rather than being treated as zero. The per-query output, `SummaryMaintenanceLifecyclePlan`, **contains** the Post-ASAP DAG rather than being a parallel representation. It records: -* the assembled Post-ASAP DAG root (`Rc`); +* the assembled Post-ASAP DAG root (`Rc`, with execution timing written); * lifecycle choices for summary state; * planning horizon and expected reads/updates; * selected window implementation and guarantees; diff --git a/docs/design_docs/architecture/metricsql-frontend.md b/docs/design_docs/architecture/metricsql-frontend.md index ddc463239..ffb45b728 100644 --- a/docs/design_docs/architecture/metricsql-frontend.md +++ b/docs/design_docs/architecture/metricsql-frontend.md @@ -9,16 +9,17 @@ on VictoriaMetrics and models MetricsQL syntax directly, including `WITH`, rollup expressions, step-relative durations, MetricsQL binary operators, aggregate limits, or-delimited matchers, and `keep_metric_names`. -The frontend walks that AST directly and emits the existing canonical -`QueryExpr`. It does not add MetricsQL fields to `QueryExpr`, SDS descriptors, -or the physical summary DAG. +The frontend walks that AST directly into the shared name-based `UnresolvedOp` +tree (`asap-frontend-common`) and calls `resolve_root`, which returns the +canonical `Rc` DAG. It does not add MetricsQL fields to the +operator IR, SDS descriptors, or the physical summary DAG. ```text MetricsQL source | MetricsqlExpr (extension semantics retained) | -canonical QueryExpr +UnresolvedOp tree --resolve_root--> canonical OperatorNode DAG | existing ASAP-aware mapping and physical Summary DAG ``` @@ -33,8 +34,8 @@ existing ASAP-aware mapping and physical Summary DAG | Common rollups: rate/increase/derivatives and statistical `*_over_time` | Existing per-entity canonical intents over the lowered range. | | PromQL arithmetic, comparison, and set binary operators without modifiers | Existing canonical `BinaryOp`. | | `default_rollup(selector[range])` | Lower to `Aggregate(LastOverTime)` over the explicit `TimeRange`. | -| `default_rollup(selector)` | Reject for exact fallback because the implicit lookbehind window depends on the runtime evaluation step, which is not a property of canonical `QueryExpr`. | -| `expr keep_metric_names` | Parsed natively, then rejected for exact fallback because canonical `QueryExpr` does not carry metric-name lineage. | +| `default_rollup(selector)` | Reject for exact fallback because the implicit lookbehind window depends on the runtime evaluation step, which is not a property of the canonical operator IR. | +| `expr keep_metric_names` | Parsed natively, then rejected for exact fallback because the canonical operator IR does not carry metric-name lineage. | | `if`, `ifnot`, `default`, aggregate `limit`, or-delimited matchers, binary match modifiers | Parsed natively and rejected until the canonical executor has the exact semantics. | | `WITH` | Expanded by the native parser; the expanded expression lowers when every resulting node is supported. | diff --git a/docs/design_docs/architecture/physical-plan-integration.md b/docs/design_docs/architecture/physical-plan-integration.md index 47fc3a1b6..5176dbc30 100644 --- a/docs/design_docs/architecture/physical-plan-integration.md +++ b/docs/design_docs/architecture/physical-plan-integration.md @@ -5,14 +5,13 @@ This document defines the boundary between ASAPPlanner's logical plans, physical lowering, statistics resolution, and analytical resource estimation. It answers which representation is authoritative at each stage and prevents -the cost model from being coupled directly to either logical IR. +the cost model from being coupled directly to the logical IR. The integration pipeline is: ```text -pre-ASAP QueryExpr ─┐ - ├─ physical lowering ─> PhysicalOperator DAG -post-ASAP SummaryExpr┘ │ +logical OperatorNode DAG ─ physical lowering ─> PhysicalOperator DAG +(NonASAPOp + ASAPOp nodes) │ v OperatorStatistics │ @@ -30,8 +29,8 @@ Each representation is authoritative for a different concern: | Representation | Authoritative concern | |---|---| -| `QueryExpr` | Original exact query semantics: sources, predicates, relational and PromQL operations, and output shape. | -| `SummaryExpr` | Logical summary semantics: selected family, grouping strategy, summary composition, and summary readout. | +| `NonASAPOp` nodes | Exact query semantics: sources, predicates, relational and PromQL operations, and output shape. | +| `ASAPOp` nodes | Logical summary semantics: selected family, grouping strategy, summary composition, and summary evaluation. | | `PhysicalOperator` DAG | Selected executable algorithms, their configuration, physical identity, edges, and execution multiplicity. | | `OperatorStatistics` | Workload-dependent evidence required by each selected physical operator's resource formula. | | `ResourceEstimate` | Estimated CPU operations, peak live memory, and physical source/disk reads over one comparison scope. | @@ -39,7 +38,7 @@ Each representation is authoritative for a different concern: `PhysicalOperator` is therefore the source of truth for the operator vocabulary consumed by analytical costing. `OperatorStatistics` corresponds one-to-one with that vocabulary. It must not independently invent operator kinds or copy -all variants from either logical IR. +all variants from the logical IR. The canonical physical-plan types should live at a neutral boundary shared by lowering, costing, explanation, and downstream compilation. Their conceptual @@ -51,7 +50,7 @@ being established. One logical operation may choose between algorithms or expand into a physical sub-DAG. Conversely, one physical operator may implement nodes originating -from either logical IR. +from either operator category (`NonASAPOp` or `ASAPOp`). Examples include: @@ -61,14 +60,14 @@ Examples include: supported join algorithm. - `SummaryAgg` may lower to an exact accumulator build, CMS build, KLL build, or another physical summary algorithm selected by the candidate. -- `SummaryEstimate` must lower to a readout operator compatible with the +- `SummaryEstimate` must lower to a evaluation operator compatible with the concrete summary state it consumes. - shared logical sub-DAGs become shared physical nodes only when they refer to the same physical identity and compatible evidence. -For this reason, aligning `OperatorStatistics` directly with `QueryExpr` would -lose post-ASAP summary implementations, while aligning it directly with -`SummaryExpr` would lose raw query operators and physical algorithm choices. +For this reason, aligning `OperatorStatistics` directly with the logical +operators would lose physical algorithm choices, and with only one category +would lose either summary implementations or raw query operators. ## Lowering obligations @@ -91,14 +90,16 @@ its modeled descendants is invalid because it undercounts the candidate. ### Pre-ASAP lowering -`KeepPreAsap` recursively lowers its contained `QueryExpr`. Typical physical +Every `NonASAPOp` node lowers recursively, whether it is in a raw query or +kept exact inside a post-ASAP plan. Typical physical operators include scans, filters, projections, hash aggregates, joins, ordering, bounded Top-K, limits, and PromQL-specific operators. The selected physical algorithm, rather than the logical spelling, determines the formula. ### Post-ASAP lowering -Every `SummaryExpr` operation also needs explicit physical realization: +Every `ASAPOp` node, and every exact operator composed with one, also needs +explicit physical realization: | Logical summary operation | Required physical realization | |---|---| @@ -107,12 +108,14 @@ Every `SummaryExpr` operation also needs explicit physical realization: | `SummaryMerge` | merge operator over compatible concrete summary states | | `SummarySubtract` | subtract operator supported by the selected state representation | | `SummaryDelete` | physical deletion/update operator supported by the selected representation | -| `SummaryEstimate` | family- and query-specific readout operator | -| `KeepPreAsap` | recursive lowering of the contained `QueryExpr` | -| `BinaryOp` | binary evaluation preserving operand order, execution timing and any typed finite/relative-division guard | -| `ValueOperation` | concrete realization of the value operation with its required execution timing and data state | -| `RelationalJoin` | concrete row-join algorithm preserving join kind and predicate | -| `RelationalJoin` with `JoinKind::Semi` | retain left rows matching explicit right-side keys; candidate pruning carries completeness evidence and ordinary TopK ranks the result | +| `SummaryEstimate` | family- and query-specific evaluation operator | +| `FinalizeExactAccumulator` | exact-state finalization before value consumers | +| `MaintainPopulation` / `EvaluatePopulation` | maintained-population update and its aggregate or TopK-prefix evaluation | +| retained `NonASAPOp` sub-DAG | recursive lowering of the exact operators (see above) | +| `BinaryOp` | binary evaluation preserving operand order, the node's execution timing and any typed finite/relative-division guard | +| `Project` / `Filter` / `Sort` / `Limit` / `Aggregate` over a evaluation | concrete realization at the node's execution timing and data state | +| `Join` | concrete row-join algorithm preserving join kind and predicate | +| `Join` with `JoinKind::Semi` | retain left rows matching explicit right-side keys; candidate pruning carries completeness evidence and ordinary TopK ranks the result | This table is a completeness requirement, not a claim that every realization already exists. Until lowering introduces an explicit physical operator, @@ -120,13 +123,13 @@ statistics contract, validation rule, and resource formula for an operation, a candidate containing it is unavailable. The streaming integration can consume a complete binding through -`SummaryNodeEvidence`. That binding is keyed to exact `SummaryNode` +`SummaryNodeEvidence`. That binding is keyed to exact `OperatorNode` identities and uses structured evidence for aggregate state, join, merge, -subtract, delete, readout, and retained pre-ASAP work. It is a physical +subtract, delete, evaluation, and retained pre-ASAP work. It is a physical evidence boundary, not automatic physical lowering: a deployment must still select each concrete implementation and provide all edges, resource facts, multiplicities, source ownership, and stable physical identities. The planner -fails closed when any reachable `SummaryExpr` node lacks that binding. +fails closed when any reachable ASAP node lacks that binding. The raw/query portion of a streaming comparison remains a `PhysicalDag` using the canonical `PhysicalOperator` and `OperatorStatistics` pairing. Summary @@ -137,7 +140,7 @@ summary-family semantics. Lifecycle choice affects the physical DAG but does not replace it. Ephemeral, prepared, shared, and continuously maintained alternatives determine when -build, update, readout, merge, subtract, or delete nodes execute. The physical +build, update, evaluation, merge, subtract, or delete nodes execute. The physical operators still determine how each execution consumes CPU, memory, and I/O. ## Statistics contract @@ -427,7 +430,7 @@ recovering average semantics from query text. ### Candidate pruning is a subgraph -Candidate-based TopK uses a summary key readout, a general semi-join over +Candidate-based TopK uses a summary key evaluation, a general semi-join over explicit matching key columns, grouped Sort by the authoritative score, and grouped Limit. Sort and Limit carry the same partition keys. The join preserves authoritative left-side values and does not rank or limit @@ -440,10 +443,12 @@ fields. The phase assignment API updates producer edge states and rejects an ingestion computation that depends on query-time work. Deployment capability, storage readiness, schemas and approximation guarantees remain separate checks. -Post-ASAP DAG wire version 4 removes the special membership operator, its edge +Post-ASAP DAG wire version 4 removed the special membership operator, its edge roles and the duplicate operator phase fields without compatibility aliases. +Version 6 (current) exports one node per operator: retained exact operators are +`Relational` nodes, not embedded sub-DAGs. -Post-ASAP DAG wire version 6 adds a per-measure row predicate to the aggregate +Post-ASAP DAG wire version 7 adds a per-measure row predicate to the aggregate operators (#466): `filters` on the exact aggregate value operation, parallel to its measures, and `filter` on `SummaryAgg`, gating which rows update the summary state. The version bump makes an older reader fail loudly instead of @@ -465,8 +470,9 @@ a numeric entity key is not a score. Exact accumulator inputs are explicitly finalized before row operators consume them. None of these operations proves candidate completeness; that evidence belongs to the semi-join's pruning step. -The semantic `SummaryExpr` constructors still propose an initial execution -layout. Uniform phase assignment applies to the exported post-ASAP DAG; +The logical DAG carries no execution layout: `apply_lifecycle_timings` writes +each node's timing from the lifecycle assignment before export. Uniform phase +assignment applies to the exported post-ASAP DAG; it is not a claim that every deployment has implemented every placement. diff --git a/docs/design_docs/architecture/planner-runtime-contract.md b/docs/design_docs/architecture/planner-runtime-contract.md index 2d071e654..c19c68aa1 100644 --- a/docs/design_docs/architecture/planner-runtime-contract.md +++ b/docs/design_docs/architecture/planner-runtime-contract.md @@ -53,7 +53,7 @@ backend still owns how the selected algorithms are physically realized. The same contract applies when ASAPPlanner selects a summary algorithm. Planner can choose KLL rather than DDSketch, while downstream chooses the concrete KLL implementation and runtime configuration that satisfies the selected parameter -and accuracy contract. Empirical KLL error, update work, state size, and readout +and accuracy contract. Empirical KLL error, update work, state size, and evaluation work observed on a particular workload can be fed back as evidence for later Planner comparisons. @@ -124,7 +124,7 @@ The ASAPQuery configuration and MIP formulations can supply physical alternatives and coefficients. Their general principles also inform Planner costing: arrival rate scales ingestion work, overlapping active windows multiply update work and live state, retained windows consume memory, and -merge/subtract/readout work scales with query recurrence. Disagreement between +merge/subtract/evaluation work scales with query recurrence. Disagreement between formulations must become distinct explicit alternatives, not hidden assumptions in one cost formula. diff --git a/docs/design_docs/architecture/updated_interface_with_pluggable_optimization.md b/docs/design_docs/architecture/updated_interface_with_pluggable_optimization.md index 72e626a2e..4820b6226 100644 --- a/docs/design_docs/architecture/updated_interface_with_pluggable_optimization.md +++ b/docs/design_docs/architecture/updated_interface_with_pluggable_optimization.md @@ -185,7 +185,7 @@ for (index, entry) in workload.query_workload.entries().enumerate() { let accuracy = entry.requirements.accuracy.target(); let expr = lower_sql_dialect(&entry.query.0, &catalog, dialect.clone(), accuracy.clone()) .await?; - roots.push((index, Rc::new(expr), Some(accuracy))); + roots.push((index, expr, Some(accuracy))); entry_indices.push(index); } diff --git a/docs/design_docs/concepts/accuracy-models.md b/docs/design_docs/concepts/accuracy-models.md index 5353ef9ef..baeadbe9e 100644 --- a/docs/design_docs/concepts/accuracy-models.md +++ b/docs/design_docs/concepts/accuracy-models.md @@ -69,7 +69,7 @@ query text or cost estimates. flowchart TD Request[Query semantics and accuracy target] --> Generate[Generate candidates and size parameters] Evidence[Scoped source contracts and evidence] --> Generate - Generate --> Local[Derive local readout guarantees] + Generate --> Local[Derive local evaluation guarantees] Evidence --> Local Local --> Compose[Propagate guarantees through the DAG] Evidence --> Compose @@ -100,7 +100,7 @@ is ready, or that a complete deployment cost is available. ## Local estimator models and parameter sizing -A local model describes a specific readout of a specific estimator with +A local model describes a specific evaluation of a specific estimator with committed parameters and applicable assumptions. A family name or a parameter such as HLL precision is not, by itself, a confidence certificate. @@ -121,8 +121,8 @@ The built-in models currently include: | CMS | L1-normalized frequency bound from width and depth; does not by itself certify TopK membership | | CountSketch | L2-normalized frequency bound and median concentration bound, requiring valid odd depth | | KMV / Theta | Parameter-derived cardinality bounds using the registered variance/Chebyshev model at 99% confidence | -| UnivMon | Exact unit-update total for the supported readout; no universal guarantee for all its statistics | -| Other families/readouts | No default certificate where no accuracy model is registered | +| UnivMon | Exact unit-update total for the supported evaluation; no universal guarantee for all its statistics | +| Other families/evaluations | No default certificate where no accuracy model is registered | This table describes Planner's registered contracts, not independent mathematical verification of every estimator or permission to substitute @@ -214,7 +214,7 @@ an observation into a guarantee. A deployment supplies `EstimatorContract::ClassicHll` for the complete aggregate expression. It asserts the classic estimator, independent uniform bucket -hashing and an enforced maximum distinct population per readout, including +hashing and an enforced maximum distinct population per evaluation, including all merged panes. Planner combines this contract with the query or allocated local target, selects a supported precision, derives the guarantee and uses the normal propagation and selection checks. @@ -290,7 +290,7 @@ accuracy/ ├── allocation.rs # End-to-end budget allocation ├── reconciliation.rs # Accuracy coordination across consumers └── estimators/ - ├── mod.rs # Family/readout dispatch and source-contract integration + ├── mod.rs # Family/evaluation dispatch and source-contract integration ├── kll.rs ├── ddsketch.rs ├── hll.rs # Generic HLL and bounded Classic HLL @@ -308,7 +308,7 @@ share the same contract. Adding an estimator or composition requires: -1. A precisely defined error metric, estimator/readout semantics and assumptions. +1. A precisely defined error metric, estimator/evaluation semantics and assumptions. 2. Sizing behavior and a guarantee derived from the committed parameters, including unsupported parameter domains. 3. Explicit evidence requirements, population scope and provenance. 4. Propagation rules where supported; rejection or retained unknowns elsewhere. diff --git a/docs/design_docs/concepts/post-asap-ir.md b/docs/design_docs/concepts/post-asap-ir.md index 67b5535a1..786d83389 100644 --- a/docs/design_docs/concepts/post-asap-ir.md +++ b/docs/design_docs/concepts/post-asap-ir.md @@ -1,22 +1,53 @@ # Post-ASAP IR The goal of the post-ASAP IR is to represent operations using ASAP primitives -such as sketches, exact summaries, samples and wavelets. Post-ASAP IR also -retains exact Pre-ASAP subtrees and supports operations over summary readouts, -since only some query operations can be satisfied using summaries. - -The lists below cover every current variant of -[`SummaryExpr`](../../../crates/types/src/post_asap/expr.rs). A node's presence -in the IR does not imply that every summary family, cost model or downstream -runtime supports it. - -## ASAP-specific nodes operated over a summary structure, not raw data - -- `SummaryAgg`: produce summary state from input data using the selected family, - parameters, update input, reduction and grouping layout. -- `SummaryEstimate`: read the requested statistic from summary state and return - query values. Exact accumulators can expose results without a separate sketch - readout. +such as sketches, exact summaries, samples and wavelets, while retaining the +exact query operators that no summary replaces, and supporting operations over +summary evaluations. + +ASAPPlanner has one operator IR before and after ASAP optimization +([`crates/types/src/ir/`](../../../crates/types/src/ir/)). A post-ASAP plan is +the same `Rc` DAG a front end produced, in which some nodes now +carry `Operator::ASAP(ASAPOp)` instead of `Operator::NonASAP(NonASAPOp)`. There +is no wrapper around retained exact work: an unreplaced `Filter`, `Join` or +`Aggregate` is the same node it was before, and either category can consume +the other's output. The node structure, the `Schema`, scalar expressions and +the catalog of non-ASAP operators are described once in the +[Pre-ASAP IR reference](../../develop_docs/pre-asap-ir.md); this document covers +what optimization adds: the ASAP operators, the accuracy guarantee, execution +timing, and the exported DAG. + +A node's presence in the IR does not imply that every summary family, cost +model or downstream runtime supports it. + +## ASAP operators + +Every variant of [`ASAPOp`](../../../crates/types/src/ir/asap.rs) operates +over summary state rather than raw data. The summary family, kind/algorithm and +parameters are committed in the node; the state itself is typed by the +`FieldDataType` of the output field that carries it (`ExactAggregate`, +`Sketch`, `Sample`, `Wavelet`, `StatModel`). + +Implemented: + +- `SummaryAgg { child, family, input, reduction, grouping }`: produce summary + state from input rows using the selected family, parameters, update input, + reduction and grouping layout. Output: the grouping columns plus one `state` + field typed `family`; result kind `State`. +- `SummaryEstimate { summary_input, query }`: read the requested statistic + (`SketchStatistic`) from summary state and return query values in a row-shaped + schema. +- `FinalizeExactAccumulator { child }`: read an exact accumulator's state as + its finalized value — the maintenance-to-read boundary before query-time + operators consume it. +- `MaintainPopulation { child, population }`: maintain the full declared + population, including membership changes. +- `EvaluatePopulation { child, evaluation }`: read an aggregate or TopK prefix from a + maintained population. + +Reserved (migrated but unimplemented; schema derivation, timing and export +reject them with `UNIMPLEMENTED_ASAP_OP`): + - `SummaryMerge`: merge compatible summary states when the family supports merging. - `SummarySubtract`: subtract one summary state from another when supported by the selected representation. @@ -24,62 +55,114 @@ runtime supports it. deletion. - `SummaryJoin`: combine summary states for join estimation; this is distinct from joining ordinary rows. +- `Extension`: a deployment-defined operator. The earlier draft listed `SummaryCreate` and `SummaryInsert`. These are not -separate variants in the current IR. `SummaryAgg` describes the state-producing +separate variants. `SummaryAgg` describes the state-producing computation and its update input. The [summary-maintenance lifecycle](../proposals/asap-aware-mapping/workload-demand-and-summary-lifecycle.md) separately describes when state is created, retained, shared, updated and retired. Physical binding and runtime execution implement the actual build and update -operations. This is not a one-to-one rename of the old nodes, and not every -summary family supports incremental maintenance. - -## Exact work and composition nodes - -- `KeepPreAsap`: retain an exact Pre-ASAP subtree when it is not rewritten. -- `BinaryOp`: combine independently planned operands with the specified binary - semantics and execution timing. -- `ValueOperation`: apply aggregate, exact-function, population, projection, - filter, sort, limit or extension semantics with explicit execution timing. -- `RelationalJoin`: join row-producing children using the specified join kind - and predicate. -- Candidate pruning uses `RelationalJoin` with `JoinKind::Semi` and an explicit +operations. Not every summary family supports incremental maintenance. + +## Exact work and composition + +Exact work is represented by the ordinary operators, unchanged: + +- A sub-DAG the planner does not rewrite keeps its `NonASAPOp` nodes. Plan + assembly marks such a sub-DAG with an exact `ResultGuarantee` + (`asap_aware_mapping::replacement::retain_exact`); a sub-DAG with no ASAP + operator and no guarantee is a logical rewrite candidate that has not been + assessed yet (`is_logical_rewrite`). +- `BinaryOp` combines independently planned operands. Summary planning may set + its typed division guards (`checked_finite_division`, + `checked_relative_division`); the operator's timing comes from the lifecycle + assignment, not from the operator. +- Aggregate, projection, filter, sort and limit over a evaluation are the ordinary + `Aggregate`, `Project`, `Filter`, `Sort` and `Limit` operators reading an ASAP + node. Exact-accumulator state may pass through the projection-like + operators unchanged; a value consumer needs a `FinalizeExactAccumulator` + boundary first. +- Candidate pruning uses `Join` with `JoinKind::Semi` and an explicit equality predicate on key columns. The left input supplies authoritative - values; the right input supplies keys. Grouped Sort followed by grouped Limit ranks - and selects the joined rows. Completeness evidence belongs to pruning, not ranking. + values; the right input supplies keys. Grouped `Sort` followed by grouped + `Limit` (both with the same `partition_by`) ranks and selects the joined + rows. Completeness evidence belongs to pruning, not ranking. -A `SummaryNode` carries its expression, schema and optional result guarantee. +Every `OperatorNode` carries its schema and an optional `ResultGuarantee`. State and query values have different contracts. Exact operations over -approximate readouts still require composed accuracy guarantees. See the +approximate evaluations still require composed accuracy guarantees. See the [accuracy implementation companion](../../develop_docs/end-to-end-accuracy-guarantees.md) and [physical-plan integration](../architecture/physical-plan-integration.md) for the corresponding correctness and realization requirements. -## Tree and exported DAG forms - -The Pre-ASAP DAG and the Post-ASAP DAG are both logical: they describe what is -computed, not which physical operators execute it. The Post-ASAP DAG has two -forms of the same content. Planning builds and shares `SummaryNode` trees. -`compile_post_asap_dag` converts a selected tree into a -[`PostAsapDag`](../../../crates/types/src/post_asap/post_asap_dag.rs) with -stable node IDs and typed edges; `PostAsapDagDocument` is its versioned wire -envelope. Physical compilation consumes `PostAsapDag` and produces a separate -physical DAG. - -## Execution phase +## Execution timing An operator defines what computation happens. The plan decides when it happens: **ingestion time** or **query time**. Operator identity must not imply one of these phases. Backend capability restrictions are implementation gaps, not definitions of the operator. -Every post-ASAP operator payload supports both phase assignments. Phase is -stored on the `PostAsapDag` node, independently of its operator payload. -`PostAsapDag::with_execution_phases` assigns a phase to every node and updates -its edges. Ingestion work cannot depend on a future query result. Default -semantic realization still proposes an initial layout; it does not restrict -which phase an operator may use. Deployments must separately check that -they have an implementation and a valid data source for the chosen placement. +The logical DAG carries no timing: `OperatorNode::timing` is `None` on every +front-end node and every candidate, and `map_children` clears it. Summary +materialization chooses a lifecycle per summary state and records it in a +[`LifecycleAssignment`](../../../crates/types/src/ir/timing.rs) (ingestion-time +maintenance or query-time recomputation per `SummaryAgg`; a state absent from +the assignment defaults to ingestion-time maintenance). +`apply_lifecycle_timings(root, &assignment, &mut TimingMemo)` then writes a +timing into every node, top-down: + +- a node of fixed kind takes its kind's timing — `SummaryEstimate` and + `EvaluatePopulation` run at query time, `MaintainPopulation` at ingestion time; +- a `SummaryAgg` takes the assignment's timing, unless something below it can + only exist at query time (a evaluation); +- every other node runs when its consumer runs: everything that feeds a + maintained state runs at ingestion time, everything above a evaluation at + query time. + +The pass then validates every edge (rows or exact-accumulator state into a +`SummaryAgg`, state into a evaluation, an ingestion-time `MaintainPopulation` under +a `EvaluatePopulation`, no ingestion work reading a query-time value) and rejects a +node reached from two consumers that need different timings; +`split_shared_by_phase` copies such a sub-DAG for one side before the +assignment is applied. `validate_default` and `planned_data_state` answer the +same questions for a candidate at planning time without keeping anything. + +## Exported DAG + +The pre-ASAP DAG and the post-ASAP DAG are both logical: they describe what is +computed, not which physical operators execute it. Planning builds and shares +`OperatorNode` trees; +[`asap_types::ir::export::compile_post_asap_dag`](../../../crates/types/src/ir/export.rs) +converts a selected, timed tree into a `PostAsapDag` with stable node IDs and +typed edges, and `PostAsapDagDocument` is its versioned wire envelope +(`schema_version` = `POST_ASAP_DAG_WIRE_VERSION`, currently 6). Physical +compilation consumes `PostAsapDag` and produces a separate physical DAG. + +Wire version 7 emits **one node per operator** — relational operators +included — with children as edges and no embedded sub-DAGs: + +- A non-ASAP node is a `Relational { operator: NonASAPOpKind }` payload: + the operator's own fields with scalar expressions mirrored as + `WireScalarExpr`, children removed. An ASAP node's payload is its variant + (`SummaryAgg`, `SummaryEstimate`, `FinalizeExactAccumulator`, + `MaintainPopulation`, `EvaluatePopulation`, …). +- Edges carry a role: `Input`, `Left`/`Right` for the two sides of a `Join`, + `SetOp`, `BinaryOp`, `SummarySubtract` or `SummaryJoin`, and `ScalarRef` + when the consumer reads the producer from inside one of its scalar + expressions (`scalar(v)`). Every edge records the intermediate schema, the + producer's data state and grouping/window compatibility. +- Each node records `output_state` (timing plus `Raw` or `SummaryState`), + `output_schema` and `guarantee`. Export reads the timing written by + `apply_lifecycle_timings` and rejects an untimed node + (`ExecutionDataStateError::UntimedNode`); it does not re-run data-state + validation. + +Phase is stored on the `PostAsapDag` node, independently of its payload. +`PostAsapDag::with_execution_phases` reassigns a phase to every node and +updates its edges; ingestion work cannot depend on a future query result. +Deployments must separately check that they have an implementation and a +valid data source for the chosen placement. ## Weighted grouped TopK @@ -90,19 +173,19 @@ the update weight is the series rate. Summing updates for one item implements the logical grouped sum without first constructing all exact grouped sums. The DAG is per-series rate → finalized values → partitioned summary construction -→ typed candidate/score readout → output projection → grouped Sort → grouped +→ typed candidate/score evaluation → output projection → grouped Sort → grouped Limit. The output count is two per job. The candidate capacity is a separate parameter, provisionally `max(k, ceil(1 / epsilon))`; this sizing choice is not a membership theorem. Missing evidence retains a logical candidate with symbolic unknown guarantees; default selection does not certify or choose it. -The row readout restores job and service identities and returns estimated sums. +The row evaluation restores job and service identities and returns estimated sums. There is no mandatory exact scoring branch or candidate semi-join in this path. The old raw counter-delta update expression is removed rather than retained as a compatibility option: counter increments are not complete windowed rate results. -The direct readout represents both score error and membership. A source provider +The direct evaluation represents both score error and membership. A source provider supplies an enforced upper bound on distinct partition/item identities for the -complete readout. Planner uses this bound to size confidence and union-bound +complete evaluation. Planner uses this bound to size confidence and union-bound score errors over adaptively selected items. Membership evidence is evaluated for the query's output count, not the candidate capacity. Score and membership failure probabilities are combined, and the score guarantee remains in the @@ -110,8 +193,8 @@ membership guarantee's child provenance. An exact request does not accept this approximate output path merely because its selected identities are certified. Deployment chooses ingestion time or query time for these operators. The -semantic constructor proposes a layout; `with_execution_phases` assigns the -placement. Either deployment must give each evaluation a complete +lifecycle assignment writes the placement; `with_execution_phases` can +reassign it on the exported DAG. Either deployment must give each evaluation a complete rate window and an isolated summary state, or maintain an equivalent replacement strategy. Appending successive rate snapshots to one cumulative state is invalid. An ingestion execution can compute a window before the query and store its state; diff --git a/docs/design_docs/concepts/pre-asap-ir.md b/docs/design_docs/concepts/pre-asap-ir.md index 227e61e39..0b12766b1 100644 --- a/docs/design_docs/concepts/pre-asap-ir.md +++ b/docs/design_docs/concepts/pre-asap-ir.md @@ -12,16 +12,18 @@ Only semantics that affect correctness, summary applicability, or cost become fi ### Time -- TimeRange — a PromQL range-vector lookback such as [5m]. +- TimeRange — PromQL sample selection: an instant selector's lookback, or a range selector such as [5m]. - TimeShift — moves when a selector is evaluated (offset or @). - PromqlSubquery — re-evaluates an instant-vector expression over a range. ### Relational - Scan — identifies a logical data source. +- Values — literal rows; one empty row is the input of a `SELECT` without `FROM`. +- ScalarBridge — a scalar expression at an operator position: a bare scalar query, or the scalar operand of ` op `. - Filter — restricts rows using a predicate. - Project — selects or derives output columns. -- BinaryOp — composes two inputs with arithmetic, comparison, or boolean logic. +- BinaryOp — composes two inputs with arithmetic, comparison, or boolean logic. A PromQL `bool` comparison returns 0/1 instead of filtering. - Sort — orders rows without expressing a heavy-hitter intent. - Limit — caps a row count, optionally after an offset. - Dedup — removes duplicate rows. @@ -31,10 +33,7 @@ Only semantics that affect correctness, summary applicability, or cost become fi ### PromQL-specific -- PromqlScalarBridge — holds a scalar sub-expression at an operator-tree position. -- EvalTimestamp — provides the evaluation timestamp as a scalar. -- PromqlVectorFromScalar — promotes a scalar to a label-less instant vector. -- PromqlScalarFromVector — collapses a single-series vector to a scalar. +- PromqlVectorFromScalar — promotes a scalar to a label-less instant vector. Its inverse, PromQL `scalar(v)`, is a scalar expression that reads `v`. - PromqlRelabel — rewrites labels on each series. - PromqlInfoEnrich — enriches labels from an info metric. - PromqlSeriesSample — selects whole series without reducing them. diff --git a/docs/design_docs/decisions/cse-cost-model.md b/docs/design_docs/decisions/cse-cost-model.md index 5689390aa..a3ddaf467 100644 --- a/docs/design_docs/decisions/cse-cost-model.md +++ b/docs/design_docs/decisions/cse-cost-model.md @@ -80,7 +80,7 @@ was retired along with `bind.rs` — this crate no longer commits to one physically-materialized answer at all; picking and building one final `SummaryNode` per shared subtree is a downstream deployment's job, not this crate's). For a `TargetSubDAGCandidates` whose candidates are a -[`SharedSubtreeStrategy`](../../../crates/asap-aware-mapping/src/replacement.rs) +[`SharedSubDagStrategy`](../../../crates/asap-aware-mapping/src/replacement.rs) share-vs-recompute pair, `cost_sorted`'s ranking step (`rank_group`/ `cse_preference`) asks `CostModel::cse_share_decision` once per group — using one representative bound `SummaryNode` built just for that comparison, not diff --git a/docs/design_docs/proposals/asap-aware-mapping/ddsketch-quantile-ratios.md b/docs/design_docs/proposals/asap-aware-mapping/ddsketch-quantile-ratios.md index f71dde2ea..8bb452783 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/ddsketch-quantile-ratios.md +++ b/docs/design_docs/proposals/asap-aware-mapping/ddsketch-quantile-ratios.md @@ -14,7 +14,7 @@ The final guarantee records both input ranges and their contract identifiers. Th ## Candidate generation without evidence -The default `SketchAlgorithmStrategy` permits a direct DDSketch quantile-ratio +The default `ASAPStrategies` permits a direct DDSketch quantile-ratio candidate when domain evidence is absent, but leaves the root guarantee unset. This is useful for the v1 integration path; it does not turn missing evidence into evidence. Other approximate divisions still require their own composition diff --git a/docs/design_docs/proposals/asapquery-rule-coverage.md b/docs/design_docs/proposals/asapquery-rule-coverage.md index dd6b9a924..7d1acd919 100644 --- a/docs/design_docs/proposals/asapquery-rule-coverage.md +++ b/docs/design_docs/proposals/asapquery-rule-coverage.md @@ -21,7 +21,7 @@ cost, and selection rules under `optimizer/`. The reviewed source is | Temporal aggregate functions | Lowering covered; realization varies | `Aggregate(PerEntity)` over `TimeRange` represents the full family. Sum, count, min, max, quantile, rate, and increase have summary realizations; `avg_over_time` is currently exact `PassThrough`, matching ASAPQuery's exact-only multi-stat fallback rather than claiming a maintained summary. | | Spatial aggregate functions | Lowering covered; realization varies | `Aggregate(Reduce(GroupKeys))` is shared by SQL and PromQL. Supported single accumulators and ordinary `by(...)` avg rewrites generate candidates; shapes such as `avg without(...)` retain the same exact raw fallback that ASAPQuery uses for multi-stat AQEs. | | Collapsible temporal + spatial aggregates | Semantic-equivalent rewriting | The existing rewrite strategy uses accumulator algebra: sum∘sum, sum∘count, min∘min, and max∘max. It rejects all other pairs and requires identical output schemas. | -| Sketch alternatives and exact fallback | Covered more generally | `SketchAlgorithmStrategy` enumerates legal summary realizations. The enclosing memo group always retains the original raw expression as the exact fallback; the strategy does not falsely label an approximate sketch as exact. | +| Sketch alternatives and exact fallback | Covered more generally | `ASAPStrategies` enumerates legal summary realizations. The enclosing memo group always retains the original raw expression as the exact fallback; the strategy does not falsely label an approximate sketch as exact. | | Subpopulation label placement | Covered more generally | `HydraGroupingStrategy` and `GroupingStrategy` express per-subpopulation and shared multi-subpopulation realizations. | | Shared computation | Covered more generally | workload-wide CSE and `SharedSubtreeStrategy` operate on physical DAG identity rather than AQE names. | | Average decomposition | Semantic-equivalent rewriting | The same rewrite strategy exposes independently optimizable sum/count accumulators when null semantics and schema permit it. | @@ -38,7 +38,7 @@ does not create a new strategy category. | Decision | Existing owner | |---|---| -| Which summary algorithm can implement one aggregate intent | `SketchAlgorithmStrategy` | +| Which summary algorithm can implement one aggregate intent | `ASAPStrategies` | | How grouping/subpopulation state is laid out | `HydraGroupingStrategy` | | Whether an equivalent logical expression exposes better accumulators | `SemanticEquivalentRewriteStrategy` (the broadened existing avg rewrite; `AvgToSumOverCountStrategy` remains a compatibility name) | | Whether identical physical work is shared | `SharedSubtreeStrategy` | @@ -52,7 +52,7 @@ does not create a new strategy category. Accordingly, ASAPQuery's four collapsible temporal/spatial patterns extend the existing semantic-rewrite owner. Temporal and spatial function recognition is already front-end lowering into `AggIntent`; sketch compatibility remains in -`SketchAlgorithmStrategy`; labels remain in `HydraGroupingStrategy`; and +`ASAPStrategies`; labels remain in `HydraGroupingStrategy`; and maintenance lifecycle legality remains in the lifecycle planner. Window framework selection is separate physical-planning work. None of these become a parallel syntax-oriented `PatternStrategy`. diff --git a/docs/design_docs/proposals/decoupling_op_and_expr.md b/docs/design_docs/proposals/decoupling_op_and_expr.md index 2e9a9342b..58859db41 100644 --- a/docs/design_docs/proposals/decoupling_op_and_expr.md +++ b/docs/design_docs/proposals/decoupling_op_and_expr.md @@ -1,113 +1,342 @@ # Decoupling Operators From Scalar Expressions -> Status: proposed, not implemented. Companion to [Operator sharing](operator-sharing.md) -> (same PR): this document splits `QueryExpr`; that one builds the shared operator -> language on the result. Code is referenced by file and function; counts are -> approximate, measured on `main` at `8acb472`. +> Status: proposal, not implemented. Audience: planner designers and architects. +> Companion to [Operator sharing](operator-sharing.md). -**The idea.** `QueryExpr` holds two different kinds of node in one enum. This proposal -splits it into `NonASAPOp` (operators) and `ScalarExpr` (scalar expressions). +## 1. Problem and goal -``` -Today Proposed -Filter { pred: Rc, Filter { pred: Predicate(Rc), - child: Rc } child: Rc } -``` +The current query representation mixes query-plan operators and scalar +expressions. Their roles are distinguished by where they occur, so an expression +can be placed where a table input is expected and fail only when the plan is checked. -## 1. Problem +Separate these concepts so the plan model expresses which combinations are valid. +Consider: + +```sql +SELECT l_quantity * 2 AS q2 +FROM lineitem +WHERE l_quantity > 10 ``` -SELECT l_quantity * 2 AS q2 FROM lineitem WHERE l_quantity > 10 -Project ← operator: outputs a table - cols: [Column(4) * Literal(2)] ← scalar expression: outputs one value per input row - child: Filter ← operator - pred: Column(4) > Literal(10) ← scalar expression - child: Scan lineitem ← operator +```text +Scan lineitem → Filter → Project + │ │ + predicate expression + quantity quantity * 2 + > 10 ``` -- An **operator** outputs a table. It is a node of the plan: the planner can replace it, - share it, or put a summary under it. -- A **scalar expression** has no table of its own. `Column(4)` means "column 4 of the - input of the operator I sit in"; outside that operator it means nothing. +The scan, filter and projection produce tables. The predicate and multiplication +compute values within the schema selected by their owning operators. + +## 2. Proposed data structures + +`NonASAPOp` describes a relation or vector computation: reading data, selecting +rows or samples, combining inputs, or reducing them. `ScalarExpr` computes one +value in a column and evaluation context. An operator supplies that context; +a standalone scalar query has no row columns. The distinction is about semantic role, +not whether the source language calls something an “expression”. + +Keep existing names where the semantics match. The proposal changes the boundary +where needed, rather than copying every current `QueryExpr` variant unchanged: -Today both are `QueryExpr` variants, told apart only by field position. The code already -separates them, but only by convention: +| Current representation | Proposed representation | Reason | +|---|---|---| +| `PromqlScalarBridge(Literal(...))` | `ScalarExpr::Literal` | Constants need no operator node. | +| Operator `EvalTimestamp` | Scalar `EvalTimestamp` | `time()` reads evaluation context and returns one number. | +| Operator `PromqlScalarFromVector` | Scalar `PromqlScalarFromVector` referencing its input plan | `scalar(v)` has real cardinality/conversion semantics. | +| Scalar operands wrapped as `BinaryOp` inputs | `Arithmetic` / `Compare` inside `Project` or `Filter` | Vector/scalar operations need no constant-producing input node. | +| `BinaryOp` without comparison mode | Vector/vector `BinaryOp` with `return_bool` | Filtering and numeric comparison results differ. | +| Undifferentiated `TimeRange` | `TimeRange` with instant/range kind | Selecting the latest sample differs from selecting all samples in an interval. | -- `QueryExpr::output_schema` returns `ScalarHasNoRowSchema` for all 13 scalar variants - (`query_expr.rs`), so `Filter { child: Literal(2) }` compiles and fails at run time. -- `pre_asap/cse.rs` never descends into a scalar, and repeats a "scalar: nothing to do" - arm in each of its three traversals; `canonicalize` likewise never rewrites one. +### 2.1 Operator nodes -## 2. Types +Use the canonical [`NonASAPOp` and `OperatorNode` definitions](operator-sharing.md#11-unified-operator-type) +from the sharing proposal. Both documents describe the same resolved model: +operator inputs and scalar query-result references use `Rc`. +`NonASAPOp` is the payload of an ordinary operator, not a second graph-node type. +`BinaryOp` likewise uses the single `BinaryOperator` payload specified there. + +Names are resolved to `ColumnId` before constructing these nodes. Parsing and +unresolved `ColumnRef` handling remain frontend concerns; no alternative generic +operator definition is proposed here. These wrappers belong to operator fields +and use the `ScalarExpr` defined in §2.2: ```rust -pub enum NonASAPOp { - Scan { .. }, Filter { pred: Predicate, child: Rc> }, Project { cols: Vec>, child }, - Aggregate { .. }, Join { .. }, SetOp { .. }, Concat { .. }, Dedup { .. }, Sort { .. }, Limit { .. }, BinaryOp { .. }, - SQLWindowFunc { .. }, TimeRange { .. }, TimeShift { .. }, Promql* { .. }, - ScalarBridge(Rc>), // formerly PromqlScalarBridge (the `2` in PromQL `v * 2`) - EvalTimestamp, // PromQL time() +struct Predicate(ScalarExpr); + +struct ProjectItem { + alias: Option, + expr: ScalarExpr, } -pub enum ScalarExpr { - Column(C), Literal(ScalarValue), Compare { .. }, BoolAnd(..), BoolOr(..), Not(..), IsNull(..), IsNotNull(..), - Cast { .. }, InList { .. }, FunctionCall { .. }, Arithmetic { .. }, Case { .. }, CurrentTimestamp, + +struct SortKey { + expr: ScalarExpr, + ascending: bool, + nulls_first: bool, } -pub struct Predicate(pub Rc>); -pub struct ProjectItem { pub alias: Option, pub expr: ScalarExpr } ``` -```text -NonASAPOp -├─ children: Rc> (Rc> after operator sharing) -└─ scalar fields: Predicate / ProjectItem / SortKey / ScalarBridge / ... - └─ ScalarExpr - └─ children: ScalarExpr only, never an operator +`Predicate`, `ProjectItem` and `SortKey` remain separate: they describe a condition, +a named output and an ordering requirement. `Values` is a relation constructor +for SQL `VALUES` and `SELECT` without `FROM`, not a wrapper for PromQL scalars. +One empty row provides the input for `SELECT 1`; zero rows represent an empty +relation. Its expressions have no input-column scope. + +`Limit.n = None` permits offset without a limit. `partition_by` adopts the existing +post-ASAP field for limits within groups, including constant-parameter PromQL +top/bottom selection. `BinaryOp` retains vector matching and gains `return_bool`, +valid only for comparisons. Scalar/vector cases lower as described in §3.3. + +`TimeRangeKind::Instant` uses `range` as the lookback horizon and selects the latest +eligible sample per series, respecting staleness. `Range` selects a sample window. +This fixes the current ambiguity between selectors; lookback is not the ingestion +interval. `TimeShift` around a selector or `PromqlSubquery` changes the evaluation +time of that whole input. Existing signed offsets and `AtModifier` anchors remain. + +### 2.2 Scalar expressions + +Scalar recursion uses owned `Box` and `Vec` children. The only plan references are +explicit operations that consume a query result to compute a value. Those edges +remain visible to plan traversal and costing; they cannot hide a separate plan. +The definitions below use resolved `ColumnId`s and the common `OperatorNode`; +there is no separate pre-ASAP scalar representation. + +```rust +enum ScalarExpr { + Column(ColumnId), + Literal(ScalarValue), + Negative { expr: Box, semantics: ExprSemantics }, + Compare { + left: Box, op: CompareOpKind, right: Box, + semantics: ExprSemantics, + }, + BoolAnd(Vec), + BoolOr(Vec), + Not(Box), + IsNull(Box), + IsNotNull(Box), + Cast { expr: Box, to: DataType, try_cast: bool }, + InList { expr: Box, list: Vec, negated: bool }, + FunctionCall { name: String, args: Vec }, + Arithmetic { + op: ArithmeticOpKind, left: Box, right: Box, + semantics: ExprSemantics, + }, + Case { + operand: Option>, + branches: Vec<(ScalarExpr, ScalarExpr)>, + else_expr: Option>, + }, + CurrentTimestamp, + EvalTimestamp, + PromqlScalarFromVector(Rc), + ScalarSubquery(Rc), + Exists { subquery: Rc, negated: bool }, + InSubquery { + expr: Box, subquery: Rc, negated: bool, + }, +} + +enum ExprSemantics { Sql, Promql } ``` -- **Naming**: `NonASAPOp` is named for [Operator sharing](operator-sharing.md), where it - becomes the non-ASAP category of `Operator`. `QueryExpr` goes away. -- **Scalar fields**: `Filter.pred`, `Join.pred`, `Aggregate.having` (`Predicate`); - `Project.cols` (`ProjectItem`); `Sort` / `SQLWindowFunc` sort keys (`SortKey`); - `SQLWindowFunc.args`; `PromqlRelabel.value`. -- **Borderline variants** go by position, not by look. `PromqlScalarBridge`, - `EvalTimestamp`, `PromqlScalarFromVector` and `PromqlVectorFromScalar` sit in operator - position with a row schema (e.g. a `BinaryOp` operand) → `NonASAPOp`. `CurrentTimestamp` - (SQL `NOW()`) is produced only by scalar lowering (`df_expr_to_unresolved`); its only - operator-position use is in a unit test → `ScalarExpr`. -- **Gain**: neither a scalar in operator position nor an operator in scalar position is - expressible, and `ScalarHasNoRowSchema` is deleted. - -## 3. Changes - -| Location | Change | +`ExprSemantics` distinguishes numeric/comparison rules even when both languages +use `Float64`; a result type alone does not preserve NaN, ordering or error rules. +`Negative` preserves unary negation directly. `Compare` returns Boolean internally; +PromQL numeric comparison results use `Case` to produce `1.0` or `0.0`. +`FunctionCall.name` must resolve to an unambiguous function contract, including +argument/result types, null behavior and volatility. An arbitrary name is not +proof that a function is supported. + +The plan-reading variants have different contracts: + +| Scalar variant | Required input and result | |---|---| -| frontend expression lowering (`df_expr_to_unresolved`, PromQL `walk`) | scalar positions build `ScalarExpr`, operator positions `NonASAPOp` | -| `resolve`, `column_resolution.rs` | already separate: `resolve` walks operators and calls `resolve_expr` for scalars. Each scalar resolves against one schema its operator picks (usually the child's output; the `Aggregate`'s output for `HAVING`, left + right for a `Join` predicate, the `Scan`'s own schema for `Scan` predicates). The split only changes their signatures: `resolve` takes `NonASAPOp`, `resolve_expr` takes `ScalarExpr`. Leaf schemas are still inferred from scalar column references across the whole tree | -| `canonicalize`, `pre_asap/cse.rs` | the "scalar: nothing to do" arms go; scalars are hashed as plain data | -| `scalar_signature.rs`, `infer_expr_type` | take `ScalarExpr` | -| `QueryExpr::output_schema` | becomes `NonASAPOp::output_schema`; the scalar arms and `ScalarHasNoRowSchema` go | - -## 4. Implementation and tests - -This is stage 1 of the joint plan ([Operator sharing §8](operator-sharing.md#8-stages-and-tests)): -children stay `Rc`; operator sharing widens them to `Rc` in its -stage 2. - -**No wire change.** The `fallback` payload serializes a `QueryExpr`, externally tagged. -Variant names are kept, so a tree serializes the same; `ScalarBridge` keeps the name -`PromqlScalarBridge` with `#[serde(rename)]`. - -- Existing tests pass unchanged apart from construction syntax. -- Tests that place a scalar in operator position no longer compile and are rewritten or - deleted: the `CurrentTimestamp` unit test, the `ScalarHasNoRowSchema` tests, and the - `post_asap_dag.rs` tests using `QueryExpr::Literal` as a `fallback` expression. - -## 5. Limits - -The split relies on no scalar containing an operator. That holds today: SQL -`IN (SELECT …)` / `EXISTS` in a filter lower to a semi-join (`lower_filter`), and every -other subquery-valued expression is rejected (`frontend-sql/src/sql/expr.rs`). Supporting -a scalar subquery (`WHERE x > (SELECT avg(x) …)`) would add `ScalarExpr::Subquery(Rc<..>)`, -make the two types mutually recursive, and require CSE and the planner to look inside -scalars. +| `PromqlScalarFromVector` | An instant vector; one float sample becomes its value, otherwise NaN. | +| `ScalarSubquery` | A one-column SQL relation; zero rows gives typed NULL, one row gives its value, multiple rows is an error. | +| `Exists` | A SQL relation; returns a non-null Boolean based on whether it has any rows. | +| `InSubquery` | A one-column SQL relation; applies SQL membership and NULL rules, including for `NOT IN`. | + +These consume query results; they are not interchangeable bridge nodes. The SQL +variants cover uncorrelated subqueries here. Correlation needs outer-scope bindings +that this proposal does not define (§3.4). + +### 2.3 Composition without a bridge + +Standalone scalar expressions have no input-column scope and cannot contain free +column references. This proposal defines their semantics without introducing a +separate query-entry data structure. +`EvalTimestamp` reads the current PromQL evaluation time; `CurrentTimestamp` reads +SQL's statement time. Moving `EvalTimestamp` into `ScalarExpr` must not make it +constant across evaluation steps or nested subquery times. + +For float samples, the following plans need no `PromqlScalarBridge`: + +```text +2 → ScalarExpr: Literal(2.0) +time() → ScalarExpr: EvalTimestamp +up * 2 → Project(sample * 2.0, child = instant selection of up) +vector(time()) → PromqlVectorFromScalar(EvalTimestamp) +scalar(sum(up)) + 1 → ScalarExpr: Arithmetic( + PromqlScalarFromVector(Aggregate(...)), Literal(1.0)) +``` + +Here, `scalar()` and `vector()` are Prometheus PromQL built-in conversion +functions explicitly present in the query, not wrappers inserted by this proposal. +`sum(up)` alone remains a valid instant-vector query. + +A PromQL projection retains the time and label fields required by the operation +and applies its metric-name rules; it does not project only the numeric sample. +For an open label schema, lowering must retain the complete series identity, +including unreferenced labels. If the input provides neither a complete label +schema nor a full identity value, this lowering is not valid. +`vector(s)` remains a real conversion to a one-element, label-free vector. +Scalar expression trees are owned, while their operator references preserve graph +identity. The companion's [complete DAG example](operator-sharing.md#13-example-composing-a-logical-dag) +shows these expressions inside ordinary operators before and after an ASAP rewrite. + +## 3. Semantic requirements + +### 3.1 Version and coverage contract + +This section maps source semantics to §2. **Direct** means the structure records +the operation; **Lowered** means an equivalent composition is specified. Neither +means implemented or runtime-verified. **Partial** limits coverage to individually +registered function contracts. **Gap** means §2 lacks a necessary payload, +type or binding rule; the frontend must reject that case until it is supplied. + +| Language | Version used for this proposal | Repository relationship | +|---|---|---| +| DataFusion SQL | **DataFusion 55.1.0**, its SQL query dialect and logical expressions | Target semantic baseline. The repository still uses 43.0.0; dependency migration is separate. This is not a claim about every SQL standard or dialect. | +| PromQL | **Prometheus 3.15.0**, with experimental features identified separately | Target semantic baseline. The repository pins ProjectASAP parser revision `9fede7eecca923c9882fe256484d00d37f8706cb`, declaring 3.8 compatibility; upgrading that parser is separate. | + +References are pinned to those versions: +[DataFusion SELECT][df-select], [logical plans][df-plan], [expressions][df-expr], +[aggregates][df-agg], [windows][df-window], [function volatility][df-volatility]; +Prometheus [query basics][prom-basics], [operators][prom-operators], +[functions][prom-functions], [AST][prom-ast] and [function signatures][prom-signatures]. +The [DataFusion release][df-release] and [Prometheus release][prom-release] identify +the targets; the parser's [compatibility declaration][parser-version] describes the +older repository dependency. This documentation change upgrades neither dependency. + +### 3.2 SQL semantic mapping + +| SQL construct / example | Representation in §2 | Coverage and semantic condition | +|---|---|---| +| `FROM t`, `WHERE x > 1`, `SELECT x * 2` | `Scan`, `Filter`, `Project`; scalar `Compare`, `Arithmetic` | **Direct.** Resolve columns against the input; predicates must be Boolean and only TRUE passes. | +| `VALUES (1), (2)`; `SELECT 1` | `Values`; `Project` over one empty row | **Direct.** SQL still returns a relation, unlike a PromQL scalar query. | +| Literals, columns, `CASE`, `CAST`, `TRY_CAST`, `IN (...)`, `IS NULL`, Boolean logic | Corresponding `ScalarExpr` variants | **Direct** within supported types. Preserve coercion, NULL propagation and conditional evaluation. Type gaps are listed below. | +| Unary minus, `BETWEEN`, `IS TRUE`, null-safe equality | `Negative`; compositions of `Compare`, `Case`, `IsNull`, Boolean expressions | **Direct/Lowered.** Evaluate reused nontrivial operands once, using a projected column where needed; do not duplicate volatile calls. | +| Scalar functions and `NOW()` | Resolved `FunctionCall`; `CurrentTimestamp` | **Direct** for registered contracts. Preserve stable/volatile evaluation behavior; unresolved names are unsupported. | +| Joins, cross joins, semi/anti joins | `Join` and its predicate | **Direct.** Predicate scope includes both inputs. Preserve outer-join null extension; ordinary anti-join is not nullable `NOT IN`. | +| `GROUP BY`, `SUM(x)`, `HAVING` | `Aggregate`, `Reduction`, existing `AggIntent`; scalar predicate on aggregate outputs | **Direct** for represented intents. `COUNT(*)` counts rows; `COUNT(x)` requires counting non-null values, not reusing row count unchanged. | +| `SUM(price * quantity)`; expression grouping keys | Input `Project`, then `Aggregate` referencing its output columns | **Lowered.** Current conversion rejects some expression arguments; the proposed representation can retain them. | +| `COUNT(x)` alongside other measures | Project a 0/1 non-null indicator, sum it, return 0 for an empty global group | **Lowered.** Filtering the whole input would incorrectly change the other measures. | +| `QUALIFY`, `DISTINCT ON` | `Filter` after window evaluation; ordered partitioned `Limit(n = Some(1))` | **Lowered.** Resolve aliases first; retain window/filter order and the selected row. | +| `ROLLUP`, `CUBE`, `GROUPING SETS` | One `Aggregate` per grouping set, projected missing keys and grouping discriminator, then `Concat` | **Lowered.** Retain duplicate grouping sets and distinguish omitted keys from input NULLs. | +| Wildcards, `UNION BY NAME`, pipe syntax for supported operations | Resolve columns, align with `Project`, compose the corresponding operators | **Lowered.** These syntax forms need no additional computational category. | +| `ROW_NUMBER()`, `LAG(x)`, `SUM(x) OVER (...)` | `SQLWindowFunc`, scalar args and sort keys; `Project` for expression partition keys | **Direct/Lowered** for existing `WindowFuncKind` and ROWS/RANGE frames. Window output is a column, not a scalar aggregate call. Missing features are listed below. | +| `DISTINCT`, `UNION`, `INTERSECT`, `EXCEPT`, their supported `ALL` forms | `Dedup`, `Concat` / `SetOp` | **Direct.** Preserve bag multiplicity and SQL duplicate/NULL equality rules. | +| `ORDER BY`, `LIMIT`, offset-only queries | `Sort`, `Limit` | **Direct.** Preserve direction and NULL placement; `n = None` means no fetch limit. | +| Uncorrelated scalar subquery, `EXISTS`, `IN` / `NOT IN (SELECT ...)` | Explicit scalar plan-reading variants | **Direct.** Preserve the cardinality and NULL contracts in §2.2, including when used in a SELECT list. | +| Derived tables and nonrecursive CTEs | Existing operator subgraphs; aliases resolved to output columns | **Lowered.** Naming alone needs no computation node. Reuse must not alter volatile evaluation. | + +### 3.3 PromQL semantic mapping + +Operator results must distinguish relations, instant vectors and range vectors; +`ScalarExpr` has a value type. Infer and validate these kinds from the source and +operation, rather than treating identical column schemas as interchangeable. +A range-query request evaluates its root at successive timestamps; it is not a +range-vector expression. The request permits scalar or instant-vector roots. + +| PromQL construct / example | Representation in §2 | Coverage and semantic condition | +|---|---|---| +| Numeric/string literals; parentheses; `time()` | Scalar `Literal`, nested expression, `EvalTimestamp` | **Direct.** No bridge node; standalone scalar queries cannot reference free columns. | +| Scalar arithmetic, unary minus, `1 < bool 2` | `Arithmetic`, `Negative`, `Case(Compare(...), 1.0, 0.0)` | **Direct/Lowered** with PromQL numeric semantics. Scalar comparison without `bool` is invalid. | +| `up{job="api"}` | Time-series `Scan` with predicates, `TimeRange(Instant)` | **Direct.** Label matching treats absent labels as empty and regexes as anchored. Selection uses lookback and staleness, not SQL row filtering alone. | +| `up[5m]` | `TimeRange(Range)` over a time-series scan | **Direct.** Retain samples and their timestamps in the left-open, right-closed window. | +| `-up` | `Project` with `Negative(sample)` | **Lowered** for float samples; retain series identity and unary-operation naming rules. | +| `up * 2`, `2 / up`, `up * scalar(sum(other))` | `Project` with scalar arithmetic and, where needed, `PromqlScalarFromVector` | **Lowered** for float samples. Preserve operand order, time/label fields and metric-name removal. Scalar input is evaluated in the same query-time context. | +| `up > 0`; `up > bool 0` | `Filter`; or `Project` with `Case(Compare(...), 1.0, 0.0)` | **Lowered** for float samples. Filtering preserves surviving sample values; bool mode produces numbers and removes the metric name. Scalar-on-left comparisons still retain the vector's sample value when filtering. | +| `a / on(job) group_left b`; `a > bool b` | `BinaryOp` with `operator.vector_match`, `return_bool` | **Direct.** Retain matching cardinality, labels and metric-name rules; unmatched elements disappear, not become false rows. | +| `a and b`, `a or b`, `a unless b` | `BinaryOp.operator.kind = BinaryOpKind::Set` | **Direct.** Label-set matching, not SQL Boolean evaluation or SQL bag set operations. | +| `sum by(job)(up)`, `avg without(instance)(up)` | `Aggregate(Reduction::Reduce(...), AggIntent)` | **Direct** for represented intents; preserve PromQL label grouping and empty-input behavior. | +| `topk(3, up)`, `bottomk(3, up)` with optional grouping | `Sort` + partitioned `Limit` | **Lowered** for constant parameters and float samples, retaining selected series labels and specified NaN ordering. Existing heavy-hitter `AggIntent::TopK` is not a substitute for sample-value ranking. | +| `rate(x[5m])`, `sum_over_time(x[5m])` | Range input + `Aggregate(Reduction::PerEntity, corresponding AggIntent)` | **Direct** for existing contracts: counter resets, extrapolation and range reduction belong to the named intent, not ordinary SQL SUM. | +| `vector(s)`, `scalar(v)` | `PromqlVectorFromScalar`; scalar `PromqlScalarFromVector` | **Direct.** These change result kind/cardinality and cannot be removed as representation wrappers. | +| `expr[30m:1m] offset 5m`, selectors with `@` | `PromqlSubquery`; surrounding `TimeShift` | **Direct.** Apply the anchor/offset to the whole child evaluation; preserve nested grids, default resolution and query-level start/end anchors. | +| Per-sample math/date functions, including scalar parameters | `Project` with a resolved scalar `FunctionCall` | **Lowered** for registered float-sample contracts. Scalar parameters can be expressions; do not require constant-only `AggIntent::Math` payloads. | +| Request-context functions and duration expressions, e.g. `x[max_of(step(), 5s)]` | Context-reading `FunctionCall`; resolve duration arithmetic before constructing `TimeRange` / `TimeShift` | **Lowered** when request parameters and the relevant subquery context determine the duration. A stored duration is valid only for that binding. | +| Relabeling, absence, `info`, series sampling | `PromqlRelabel`, relevant `AggIntent`, `PromqlInfoEnrich`, `PromqlSeriesSample` | **Partial.** Registered contracts must preserve label construction, missing-series behavior and feature gates. Parameter/type gaps below remain. | + +### 3.4 Remaining semantic gaps + +These are limits of the proposed payloads or an as-yet unspecified lowering, not +reasons to mix all operators and expressions back into one enum. The tables above +cover the core structures; they do not assert full language conformance. + +| Semantics not fully covered | Why §2 cannot currently express it faithfully | Required extension or decision | +|---|---|---| +| General SQL aggregate `FILTER`, `DISTINCT`, internal ordering, null treatment and arbitrary aggregate UDFs | `AggIntent` is a fixed intent vocabulary with no general per-measure modifier payload. `COUNT(DISTINCT ...)` has `Cardinality`, but this does not cover every aggregate. | Define per-measure semantics or an equivalent lowering for each case. A filter on the whole aggregate input is not a general replacement. | +| All DataFusion window functions, GROUPS frames, explicit null treatment, window `FILTER` / `DISTINCT` | `WindowFuncKind` is a subset; `WindowFrameUnits` only has Rows/Range; the window payload lacks these modifier fields. | Extend these existing payloads for the requested features; ordinary scalar `FunctionCall` cannot supply window context. | +| Correlated SQL subqueries and recursive CTEs | Scalar plan references have no outer-scope binding, and an ordinary DAG has no recursive/fixpoint contract. | Define correlation scopes and recursive evaluation separately; only proven equivalent decorrelation is usable today. | +| SQL higher-order functions and lambda expressions | `FunctionCall.args` has no lambda parameter bindings or body scope; ordinary column references cannot stand for lambda variables. | Add explicit scalar lambda/binding structures before claiming coverage. | +| General SQL `ANY` / `ALL` subquery comparisons | `InSubquery` only records membership, not a comparison operator and quantifier. | Add a scalar `SetComparison` payload or a proven lowering retaining empty-set and NULL behavior. | +| Unbound SQL parameters / scalar variables | No placeholder or variable binding is represented. | Bind them to typed values before this IR, or define a binding payload; a free column is not a parameter. | +| Full DataFusion value/type fidelity | Existing `DataType`/`ScalarValue` cannot retain every decimal, unsigned width, timestamp unit/timezone or typed nested literal. Widening values can change result types and errors. | Extend the value/type model; casts cannot recover information already lost. The SQL mapping is restricted to faithfully represented types. | +| SQL pattern matching with explicit escape rules and all dialect-specific scalar operators | Current `CompareOpKind` pattern variants do not carry `ESCAPE`; function names alone do not define missing semantics. | Add the missing payload or a registered equivalent scalar contract. Simple LIKE does not prove ESCAPE support. | +| SQL `UNNEST` / general table functions | No operator expands a collection or invokes a table-valued function with its output cardinality/schema contract. | Add a relational operation, not a scalar function pretending to produce rows. | +| PromQL native-histogram samples and mixed-sample behavior | Existing value types have no native-histogram sample representation or annotation contract. Named histogram intents alone do not preserve those samples. | Extend sample types and define invalid-operation/annotation behavior; the float mappings above do not cover this case. | +| Dynamic PromQL parameters, e.g. `quantile(scalar(q), up)` | `AggIntent.q`, sampling parameters and `Limit.n` store constants rather than expression dependencies. Per-sample math can instead use `FunctionCall` as mapped above. | Permit scalar expressions in the relevant parameter positions. Constant folding only covers genuinely constant inputs. | +| PromQL experimental fill modifiers | `VectorMatch` has no left/right fill values, so it cannot distinguish dropping an unmatched series from supplying a numeric default. | Extend that payload for `fill`, `fill_left`, `fill_right`; preserve the `promql-binop-fill-modifiers` feature gate. | +| Extended PromQL range selectors and start-timestamp functions | `TimeRangeKind` has no anchored/smoothed selection mode, and the sample schema has no distinct start-timestamp metadata contract. | Define the temporal/sample metadata and feature gates before mapping these forms. A normal sample timestamp is not a start timestamp. | + +### 3.5 Context and evaluation invariants + +Scalar columns resolve against the owning input, both inputs for a join predicate, +and aggregate outputs for `HAVING`. Scan predicates use the source schema. SQL +subqueries in this proposal have their own scope and no implicit outer references. +Predicate wrapping does not bypass Boolean typing or SQL three-valued logic. + +Function resolution preserves volatility: SQL `NOW()` is stable within a statement; +repeated volatile calls need not agree. Expression ownership does not authorize +copying, sharing or moving evaluations. PromQL scalar plan reads and +`EvalTimestamp` use the active evaluation instant, including within subqueries. +A reused operator must not be evaluated once and then incorrectly reused across +different time contexts. + +## 4. Acceptance and scope + +This is a representation proposal, not a runtime implementation or a declaration +of complete SQL/PromQL support. Acceptance requires: + +- The SQL example in §1 retains its values, schema and predicate context. +- Every Direct/Lowered mapping in §3 has a valid typed representation; each Gap is + explicitly rejected until its payload or equivalent lowering is defined. +- The scalar queries and mixed scalar/vector examples in §2.3 need no + `PromqlScalarBridge` or equivalent constant-wrapper node. +- Operator dependencies inside scalar conversions/subqueries remain visible and + shared; scalar trees remain owned. Invalid result-kind combinations are rejected. +- SQL NULL/cardinality rules, PromQL labels and evaluation times survive conversion. + +Implementation will require frontend, validation and plan-format migration for +these explicit structural changes. DDL/DML, session commands, physical execution, +new optimization algorithms, accuracy and execution-timing policy are outside this +proposal. The companion document defines the common pre-/post-ASAP operator graph. + +[df-release]: https://github.com/apache/datafusion/releases/tag/55.1.0 +[prom-release]: https://github.com/prometheus/prometheus/releases/tag/v3.15.0 +[df-select]: https://github.com/apache/datafusion/blob/55.1.0/docs/source/user-guide/sql/select.md +[df-plan]: https://github.com/apache/datafusion/blob/55.1.0/datafusion/expr/src/logical_plan/plan.rs +[df-expr]: https://github.com/apache/datafusion/blob/55.1.0/datafusion/expr/src/expr.rs +[df-agg]: https://github.com/apache/datafusion/blob/55.1.0/docs/source/user-guide/sql/aggregate_functions.md +[df-window]: https://github.com/apache/datafusion/blob/55.1.0/docs/source/user-guide/sql/window_functions.md +[df-volatility]: https://github.com/apache/datafusion/blob/55.1.0/datafusion/expr-common/src/signature.rs +[prom-basics]: https://github.com/prometheus/prometheus/blob/v3.15.0/docs/querying/basics.md +[prom-operators]: https://github.com/prometheus/prometheus/blob/v3.15.0/docs/querying/operators.md +[prom-functions]: https://github.com/prometheus/prometheus/blob/v3.15.0/docs/querying/functions.md +[prom-ast]: https://github.com/prometheus/prometheus/blob/v3.15.0/promql/parser/ast.go +[prom-signatures]: https://github.com/prometheus/prometheus/blob/v3.15.0/promql/parser/functions.go +[parser-version]: https://github.com/ProjectASAP/promql-parser/blob/9fede7eecca923c9882fe256484d00d37f8706cb/README.md#promql-compliance diff --git a/docs/design_docs/proposals/operator-sharing.md b/docs/design_docs/proposals/operator-sharing.md index 0256b4ba1..8be98a503 100644 --- a/docs/design_docs/proposals/operator-sharing.md +++ b/docs/design_docs/proposals/operator-sharing.md @@ -1,535 +1,544 @@ # Sharing Operators Between Pre-ASAP IR and Post-ASAP IR -> - Status: proposed, not implemented. -> - Problem statement: [#468](https://github.com/ProjectASAP/ASAPPlanner/issues/468). -> - Builds on [Decoupling operators from scalar expressions](decoupling_op_and_expr.md) (same PR), which splits `QueryExpr` into `NonASAPOp` and `ScalarExpr`. +> Status: proposal, not implemented. Audience: planner designers and architects. +> Addresses [#468](https://github.com/ProjectASAP/ASAPPlanner/issues/468). +> Companion: [Decoupling operators from scalar expressions](decoupling_op_and_expr.md). -**The idea.** Today a post-ASAP plan is glued together from two sets of operator types. -This proposal keeps one operator language and makes summary operators extra node kinds in it: any relational operator can sit above a summary, and a summary can read any relational subtree. -Nothing is wrapped and nothing is duplicated. +## Goal and problem -``` -Today Proposed -ValueOperation(Project) ← a copy NonASAP(Project) - SummaryEstimate ASAP(SummaryEstimate) - SummaryAgg(Kll) ASAP(SummaryAgg(Kll)) - KeepPreAsap(Scan lineitem) ← a black box NonASAP(Scan lineitem) -``` +Use one operator model before and after ASAP optimization, so ordinary query +operations and summary operations can form one visible computation graph. -| Part | Sections | -|---|---| -| I. New IR | §1 Types, §2 Schema, guarantee, and timing | -| II. Changes, in data-flow order | §3 Entry → §4 Planner → §5 Timing → §6 Export → §7 Other consumers | -| III. Implementation | §8 Stages and tests, §9 Out of scope, §10 Open questions | +Today, the post-ASAP representation wraps relational subplans and duplicates some +relational operators outside those wrappers. This causes three problems: + +- A projection above a summary needs a different representation from a projection + below it, although both perform the same operation. +- An exact aggregate cannot directly share a scan hidden inside a summary's input. +- An operator without a post-ASAP counterpart cannot naturally contain summary-based + children. + +For example, consider a p99 latency query that projects its input columns, builds a +KLL summary, and projects the estimated result. The trees below read from the result +at the top to the data source at the bottom: ---- +```text +Today Proposed +Post-ASAP projection Project +└─ Summary estimation └─ Summary estimation + └─ KLL summary build └─ KLL summary build + └─ Wrapped relational subplan └─ Project + └─ Ordinary projection └─ Scan latency + └─ Scan latency +``` + +Today the two projections need separate representations, and the scan is hidden +inside the wrapped subplan. In the proposed graph, both projections use the same +operator definition and the scan is directly visible. A union can likewise consume +summary estimates without needing a separate post-ASAP union definition. -# I. New IR +The design removes these representation barriers. It makes composition and sharing +possible; whether a particular rewrite or shared computation is valid still depends +on query semantics, accuracy and execution timing. -## 1. Types +## 1. Operator model ### 1.1 Unified `Operator` type -Operator attributes differ in how widely they apply. -We define the `Operator` type structure based on the breadth of its attributes. +`Operator` describes an ordinary or ASAP operation; `OperatorNode` combines it +with common planning properties. `ScalarExpr` describes value computation. The +following overview and payload definitions are the canonical resolved interfaces +used by both proposals. The companion document defines `ScalarExpr` and its +operator-field wrappers; it does not define a second operator model. -| Applies to | Examples | Defined as | -|---|---|---| -| every operator | children, schema, timing, guarantee | methods implemented for `Operator` | -| one category | for all `NonASAP` operators, timing is derived from the consuming edge, and guarantee from the children | implementation specified to one enum branch of `Operator` | -| one operator | `Aggregate.measures`, `SummaryAgg.family` | fields of that variant | +**Proposed data structures — overview.** The complete outer structure is below; +operation variants and schema internals are expanded afterward. These declarations +are shared by the detailed sections, not separate abbreviated types. ```rust -pub enum Operator { - NonASAP(NonASAPOp), // today's relational and timeseries operators in `QueryExpr` (§1.2) - ASAP(ASAPOp), // summary operators (§1.3) +// A graph node combines its operation with common planning properties (§2). +struct OperatorNode { + operator: Operator, + result_kind: OperatorResultKind, + schema: Schema, + guarantee: Option, + timing: Option, } -impl Operator { // implemented for every operator - pub fn children(&self) -> Vec<&Rc>>; - pub fn map_children(&self, f: impl FnMut(&Rc>) -> Rc>) -> Self; - pub fn output_schema(&self) -> Result; // schema: §2.1 - pub fn guarantee(&self) -> &Slot>; // accuracy guarantee: §2.2 - pub fn timing(&self) -> &Slot; // execution timing: §2.3 - pub fn with_guarantee(&self, guarantee: Option) -> Self; // Setter of accuracy guarantee - pub fn with_timing(&self, timing: ExecutionTiming) -> Self; // Setter of execution timing +// Operation payloads: each variant below defines its own inputs and parameters. +enum Operator { + NonASAP(NonASAPOp), + ASAP(ASAPOp), } -/// `Slot` represents a value that may be unset or set. -/// In the current design, it will be used to wrap the `timing` and `guarantee` values, -/// whose values will only be determined after derivation. -pub enum Slot { Unset, Set(T) } - -/// Memo of one derivation, keyed by (node pointer, incoming timing). -/// One is shared by every root of a workload, so a node shared by two roots stays one `Rc`. -pub struct DerivationMemo { .. } - -/// Build the accuracy guarantee of one DAG root by derivation -pub fn derive_guarantees( - root: &Rc, - model: &dyn AccuracyModel, // accuracy model used for derivation - evidence: &dyn AccuracyEvidenceProvider, // evidence provider used for derivation - memo: &mut DerivationMemo, -) -> Result, AccuracyError>; -/// Build the execution timing of one DAG root by derivation; the root runs at query time -pub fn derive_timings( - root: &Rc, - memo: &mut DerivationMemo, -) -> Result, ExecutionDataStateError>; -/// Same, with the root's timing given, e.g. `IngestionTime` for a maintenance candidate -/// (like today's `validate_execution_data_states_at`) -pub fn derive_timings_at( - root: &Rc, - root_timing: ExecutionTiming, - memo: &mut DerivationMemo, -) -> Result, ExecutionDataStateError>; +// NonASAPOp / ASAPOp: detailed below; their inputs are Rc. +// ScalarExpr: an owned value-expression tree, defined in the companion proposal. +// Schema / OperatorResultKind: defined in §2.1. ``` -- **Derivation recomputes**: `derive_*` keep the values set at construction (§2.2, §2.3) - and recompute every other slot, so calling them again after a rewrite is safe. -- **Equality**: both slots take part in `PartialEq` and hashing, so CSE never merges two - nodes that differ in timing or guarantee. +An operator owns its scalar expressions and references input nodes through +`Rc`. Either operation category can consume the other's outputs when +the input contract permits it. `NonASAP` describes one operation, not its entire +subgraph. Frontend graphs contain only NonASAP operations; ASAP optimization may +introduce state construction and readout. -Following diagram conceptually displays the structure of `Operator`: -```text -Operator -├─ NonASAP(NonASAPOp) -│ ├─ children: Rc> → back to Operator: NonASAP or ASAP -│ ├─ timing / guarantee -│ └─ scalar expressions: Predicate / ProjectItem / SortKey / ScalarBridge / ... -│ └─ ScalarExpr: never contains an Operator -└─ ASAP(ASAPOp) - ├─ children: Rc> → back to Operator: NonASAP or ASAP - └─ timing / guarantee -``` +| Category | Meaning | All operations | +|---|---|---| +| `Operator::NonASAP(NonASAPOp)` | Ordinary query operations that transform, combine or aggregate data | `Scan`, `Values`, `Filter`, `Project`, `Aggregate`, `Join`, `SetOp`, `Concat`, `Dedup`, `Sort`, `Limit`, `BinaryOp`, `SQLWindowFunc`, `TimeRange`, `TimeShift`, `PromqlVectorFromScalar`, `PromqlRelabel`, `PromqlInfoEnrich`, `PromqlSeriesSample`, `PromqlSubquery` | +| `Operator::ASAP(ASAPOp)` | Operations on summary state and its results, including reserved operations | `SummaryAgg`, `SummaryEstimate`, `SummaryMerge`, `SummarySubtract`, `SummaryDelete`, `SummaryJoin`, `FinalizeExactAccumulator`, `MaintainPopulation`, `ReadPopulation`, `Extension` | -### 1.2 `NonASAPOp` +`CurrentTimestamp`, `EvalTimestamp` and `PromqlScalarFromVector` belong to +`ScalarExpr`, defined in the [companion proposal](decoupling_op_and_expr.md#22-scalar-expressions). +A constant needs no bridge operator. The sketches use resolved `ColumnId`s and +`Schema`; name resolution precedes construction of these nodes. -`NonASAPOp` is the non-ASAP category of `Operator`. -It comes from splitting `QueryExpr` into "operator" and "scalar expression" parts ([decoupling doc](decoupling_op_and_expr.md#2-types)). +`NonASAPOp` retains the query semantics needed before and after optimization: ```rust -pub enum NonASAPOp { - Scan { .. }, - Filter { pred: Predicate, child: Rc> }, - Project { cols: Vec>, child: Rc> }, - Aggregate { reduction, measures, having: Option>, child: Rc> }, - Join { kind, pred: Predicate, left: Rc>, right: Rc> }, - SetOp { kind, all, left: Rc>, right: Rc> }, - Concat { children: Vec>> }, - Sort { keys: Vec>, child: Rc> }, - Limit { n, offset, child: Rc> }, - BinaryOp { op, lhs, rhs }, - SQLWindowFunc { args: Vec>, order_by: Vec>, child: Rc>, .. }, - Dedup { .. }, TimeRange { .. }, TimeShift { .. }, Promql* { .. }, - ScalarBridge(Rc>), // the `2` in PromQL `v * 2` - EvalTimestamp, // PromQL time() +enum NonASAPOp { + Scan { + source: Source, predicates: Vec, schema: Schema, + }, + Values { rows: Vec>, schema: Schema }, + Filter { child: Rc, pred: Predicate }, + Project { + child: Rc, cols: Vec, qualifier: Option, + }, + Aggregate { + child: Rc, reduction: Reduction, measures: Vec, + output_names: Vec, having: Option, + }, + Join { left: Rc, right: Rc, kind: JoinKind, pred: Predicate }, + SetOp { left: Rc, right: Rc, kind: RelationalSetOpKind, all: bool }, + Concat { + children: Vec>, discriminator_unique_key: Option, + }, + Dedup { child: Rc, cols: Vec }, + Sort { child: Rc, keys: Vec, partition_by: GroupKeys }, + Limit { child: Rc, n: Option, offset: usize, partition_by: GroupKeys }, + BinaryOp { + lhs: Rc, rhs: Rc, operator: BinaryOperator, return_bool: bool, + }, + SQLWindowFunc { + child: Rc, func: WindowFuncKind, args: Vec, + partition_by: GroupKeys, order_by: Vec, + frame: Option, output_name: String, + }, + TimeRange { child: Rc, range: Duration, kind: TimeRangeKind }, + TimeShift { child: Rc, shift: TimeShift }, + PromqlVectorFromScalar(ScalarExpr), + PromqlRelabel { child: Rc, dst: String, value: ScalarExpr }, + PromqlInfoEnrich { child: Rc, selector: Vec }, + PromqlSeriesSample { child: Rc, by: GroupKeys, kind: SampleKind }, + PromqlSubquery { child: Rc, range: Duration, resolution: Option }, } + +enum TimeRangeKind { Instant, Range } ``` -Every variant also carries the `timing` and `guarantee` slots (§1.1), omitted above. +Ordinary payload fields have these roles: -### 1.3 `ASAPOp` +- `Predicate` describes a row-level condition; `ProjectItem` contains a scalar + expression and its optional output alias. +- `reduction` describes whether aggregation combines groups or operates per entity; + `measures` describes the requested aggregates. Grouping is distinct from ordering + or limiting within groups, represented by `partition_by`. +- Join/set kinds, vector matching, window frames and time selections preserve + source-language semantics. Output names, qualifiers and proven uniqueness also + survive optimization. The optional concatenation key records a discriminator + that distinguishes branches together with their within-branch key. -`ASAPOp` is the ASAP category of `Operator`. -`ASAPOp` comes from today's `SummaryExpr`: its summary variants, and the summary-specific `ValueOperation` variants. +`ASAPOp` describes state construction, state operations and readout separately. +`FieldDataType` (§2.1) types every output field; state-producing operations use its +summary or exact-accumulator cases, never its `Plain` case. ```rust -pub enum ASAPOp { - SummaryAgg { child: Rc>, family: ASAPType, input, reduction, grouping, - exact_rule: Option }, - SummaryEstimate { child: Rc>, query: SketchQuery, - local_guarantee: Option }, - SummaryMerge { children: Vec>> }, - SummarySubtract { left: Rc>, right: Rc> }, - SummaryDelete { child: Rc>, key: C }, - SummaryJoin { outer: Rc>, inner: Rc>, key: C, family: ASAPType }, - FinalizeExactAccumulator { child: Rc> }, - MaintainPopulation { child: Rc>, population }, - ReadPopulation { child: Rc>, readout }, - Extension { child: Rc>, name: String }, +enum ASAPOp { + SummaryAgg { + child: Rc, family: FieldDataType, input: SummaryUpdate, + reduction: Reduction, grouping: GroupingStrategy, + }, + SummaryEstimate { + summary_input: Rc, query: SketchQuery, + }, + FinalizeExactAccumulator { child: Rc }, + MaintainPopulation { child: Rc, population: MaintainedPopulation }, + ReadPopulation { child: Rc, readout: PopulationReadout }, + + // Reserved operations; semantics and support require further design. + SummaryMerge { children: Vec> }, + SummarySubtract { left: Rc, right: Rc }, + SummaryDelete { summary_input: Rc, key: ColumnId }, + SummaryJoin { + outer: Rc, inner: Rc, key: ColumnId, family: FieldDataType, + }, + Extension { child: Rc, name: String }, } ``` -Every variant also carries the `timing` and `guarantee` slots (§1.1), omitted above. - -**Unused branches**: `SummaryMerge`, `SummarySubtract`, `SummaryDelete`, `SummaryJoin` and `Extension` are built only in tests today. They are migrated, but for safety, we have all their methods return `Unimplemented`. - -Following table shows how some legacy types get expressed in the new framework. +The summary fields distinguish state construction and readout: -| Legacy types | Expressed as | +| Field | Design meaning | |---|---| -| `SummaryExpr::KeepPreAsap(q)` | `q` itself, an `NonASAP(..)` subtree | -| `ValueOperation::{Project, Filter, Sort, Limit}` | `NonASAPOp::{Project, Filter, Sort, Limit}` | -| `SummaryExpr::{BinaryOp, RelationalJoin}` | `NonASAPOp::{BinaryOp, Join}` | -| `ValueOperation::Exact(Aggregate)`, `ExactOperation` | `NonASAPOp::Aggregate` | -| `SummaryNode` | `Operator` itself: `schema` is computed, `timing` / `guarantee` are slots on every variant (§2) | - -### 1.4 Child field - -Non-ASAP operators now sit on the same level as ASAP operators, so their children must -be `Rc` to allow free placement: - -```rust -// After the decoupling doc // After this proposal -Filter { pred: Predicate(Rc), Filter { pred: Predicate(Rc), - child: Rc } child: Rc } // NonASAP(..) or ASAP(SummaryEstimate ..) +| `family` | The summary or exact accumulator chosen, including its family-specific parameters | +| `input` | The item identity and observation or weight supplied to a state update | +| `reduction` | Which input entities contribute to each logical result | +| `grouping` | Whether those groups use separate state instances or a supported shared structure | +| `query` / `readout` | The result requested from summary or maintained-population state | +| `population` | The population whose membership and values are maintained | + +The common `OperatorNode` fields are declared in the overview above and explained +in §2. The operation variants do not repeat them. Reserved ASAP variants require +further semantic and capability design before use. + +### 1.2 Operators and scalar expressions + +A filter is an operator because it transforms a table. Its predicate, such as +`latency > 100`, is a scalar expression evaluated in that table's schema. + +Scalar expressions belong to an operator field or a scalar query. Predicates, +projection expressions and sort keys describe value computation in that context. +Explicit scalar conversions and subqueries may reference operators; those are +visible graph dependencies with defined cardinality rules. This prevents an +arbitrary expression from being mistaken for a table-producing plan. The +[companion proposal](decoupling_op_and_expr.md) defines this distinction. + +The companion's `ScalarExpr` uses `Rc` for `PromqlScalarFromVector`, +`ScalarSubquery`, `Exists` and `InSubquery`, so those expressions already reference +this common graph before and after optimization. + +In `scalar(sum(up))`, `scalar()` is Prometheus PromQL's built-in vector-to-scalar +function, explicitly written by the query author. This proposal does not insert +it automatically: `sum(up)` alone is a valid query returning an instant vector. +The scalar expression `PromqlScalarFromVector` represents that function and references its +result to obtain one number. A valid ASAP rewrite may replace that producer with +a summary readout, preserving the required vector and accuracy semantics; it cannot +substitute raw summary state. Ordinary expressions such as `price * 2` reference +columns and literals, not a query subgraph. + +These are **query subgraphs referenced by scalar expressions**, with the same +producer identity as any other operator dependency. + +### 1.3 Example: composing a logical DAG + +Consider this SQL query, with integer `bytes` and `status` columns: + +```sql +SELECT SUM(bytes) + 1 AS total_bytes +FROM requests +WHERE status = 200; ``` -Now an original operator can also sit on ASAP operators, e.g. a `SetOp` sitting on two `SummaryEstimate` operators. - -`Concat.children` is `Vec` today: branches are stored by value and have no `Rc` identity, so the planner (§4), which identifies targets by pointer, -cannot replace a branch — e.g. the branches of SQL `ROLLUP` or PromQL `histogram_quantiles`. It becomes `Vec>` (§8 stage 0). +Before ASAP optimization, its logical DAG is composed as follows. Each named node is an +`OperatorNode`; arrows point from a consumer to its input producer. The scalar +expressions shown beside nodes are owned fields, not additional DAG nodes. -## 2. Schema, Guarantee, and Timing - -This section discusses three key per-node attributes, `schema`, `guarantee`, and `timing`, as well as how they are stored and derived in the new framework. - -| Field | Meaning | Today | After | -|---|---|---|---| -| `schema` | output columns and their types | pre-ASAP: computed by `QueryExpr::output_schema()`
post-ASAP: a `SummarySchema` stored on every `SummaryNode` | can be obtained by `output_schema()` | -| `guarantee` | accuracy bound | pre-ASAP: none
post-ASAP: stored on every `SummaryNode` | can be obtained by `guarantee()`
binding stores only each operator's own error
complete error bound need to be derived by `derive_guarantees()` | -| `timing` | execution time | pre-ASAP: none
post-ASAP, stored: a field on `BinaryOp` / `ValueOperation` / `SummaryMerge`
post-ASAP, not stored: `KeepPreAsap` from the consuming edge, `SummaryAgg` from the child. | can be obtained by `timing()`
set by binding (`SummaryAgg`) or the planner (`FinalizeExactAccumulator`)
timing of the rest of operators need to be derived by `derive_timings()` | - -### 2.1 Schema: fused into one type - -Today schemas of pre-ASAP operators and post-ASAP operators are different: -- pre-ASAP uses `Schema { columns: Vec, time_index, unique_keys, closed }` with `Column.dtype: DataType` (plain values only), -- post-ASAP stores a `SummarySchema { fields: Vec, time_index }` on every node, with `SummaryField.dtype: SummaryFamilyType` (`Plain(DataType)` or summary state). -Now since the two operators types are unified into one, we need a unified schema type as well. - -We implement the new schema type based on the original `Schema` type used in pre-ASAP operators, with two changes: - -- `Column` is renamed `Field`, and `Schema.columns` `Schema.fields`: the struct describes - a column and holds none of its data. (Arrow and DataFusion use the same names.) -- `Field.dtype` widens from `DataType` to an enum `FieldType`, which covers both plain data types and ASAP summary types. - -`SummarySchema` / `SummaryField` are then redundant and deleted. - -Detailed code design is shown below. -```rust -pub enum FieldType { DataType(DataType), ASAPType(ASAPType) } -pub enum ASAPType { // SummaryFamilyType without Plain - ExactAggregate(ExactKind, ExactParams), Sketch(SketchKind, GroupingStrategy), - Sample(SamplingKind, SamplingParams), Wavelet(WaveletKind, WaveletParams), StatModel(StatModelKind, StatModelParams), -} -pub struct Schema { pub fields: Vec, pub time_index, pub unique_keys, pub closed } -pub struct Field { pub name, pub dtype: FieldType, pub nullable, pub table: Option } -impl Field { - pub fn plain(name, DataType) -> Self; - pub fn plain_dtype(&self) -> Option<&DataType>; // None for a state column - pub fn expect_plain_dtype(&self) -> &DataType; // frontends, scalar type inference; panics on state -} +```text +Project node: OperatorNode + operator = Operator::NonASAP(NonASAPOp::Project) + cols[0].expr = ScalarExpr::Arithmetic(Column(sum_bytes), Add, Literal(1)) + │ child: Rc + ▼ +Aggregate node: OperatorNode + operator = Operator::NonASAP(NonASAPOp::Aggregate) + measures = [AggIntent::Sum(bytes)] + │ child: Rc + ▼ +Filter node: OperatorNode + operator = Operator::NonASAP(NonASAPOp::Filter) + pred = Predicate(ScalarExpr::Compare(Column(status), Eq, Literal(200))) + │ child: Rc + ▼ +Scan node: OperatorNode + operator = Operator::NonASAP(NonASAPOp::Scan) + source = requests ``` -| Node | Today | After | -|---|---|---| -| `NonASAPOp` | post-ASAP `KeepPreAsap`: `QueryExpr` schema lifted to `SummarySchema` and stored
post-ASAP `ValueOperation` / `BinaryOp` / `RelationalJoin` copies: stored at construction | using the same logic as `QueryExpr::output_schema()` | -| `SummaryAgg` | the replaced `Aggregate`'s output with the measure column retyped to `family` | grouping columns + one `ASAPType(family)` column | -| `SummaryEstimate` | the replaced operator's output schema | the child's grouping columns + the value columns of the `SketchQuery` | -| `FinalizeExactAccumulator` | the logical operator's output, lifted | the child's schema, `ASAPType(ExactAggregate ..)` columns changed into `DataType(..)` | -| `MaintainPopulation` / `ReadPopulation` | the source's schema / the replaced aggregate's output | the same rules, computed from the child and the `readout` | -| unused variants | one field typed `family` | unimplemented | - -### 2.2 Guarantee: always derived - -A guarantee is filled in two steps: +This is abbreviated structural notation: `Column` and `Literal` above are +`ScalarExpr` variants; column names stand for resolved `ColumnId`s. The arithmetic +and comparison use `ExprSemantics::Sql`. The aggregate has no grouping keys and +names its output `sum_bytes`; the projection names its output `total_bytes`. -1. **Binding** records local accuracy guarantee: a `SummaryEstimate`'s `local_guarantee` (the sketch's error over an exact input) and an exact `SummaryAgg`'s `exact_rule`. No `guarantee` slot is set yet. To size a sketch and check its target, binding still needs the child's error, as today: it runs `derive_guarantees` on the child with a fresh memo, reads the result, and drops it. -2. **`derive_guarantees`** fills every slot bottom-up: a node without an `ASAP` descendant is exact, and every other node composes its children's guarantees by its own rule. +An eligible ASAP rewrite can implement the sum using an exact accumulator. The +resulting logical DAG contains both operation categories: -``` -Project p99 ±1% ← the child's - SummaryEstimate p99 ±1% ← local ±1%, composed with the child's - SummaryAgg(Kll) None ← state has no guarantee - Scan t exact ← no ASAP descendant +```text +Project node: NonASAP(Project) + expression: sum_bytes + 1 + │ child + ▼ +Finalize node: ASAP(FinalizeExactAccumulator) + output: ordinary sum_bytes value + │ child + ▼ +Summary build node: ASAP(SummaryAgg) + output: exact SUM accumulator state + │ child + ▼ +Filter node: NonASAP(Filter) + predicate: status = 200 + │ child + ▼ +Scan node: NonASAP(Scan) + source: requests ``` -Per node kind: - -| Node | Today | After | -|---|---|---| -| `SummaryEstimate` | stored at binding: the sketch's own error composed with the child's (`compose_guarantee`) | **derived**: `local_guarantee` composed with the child's. `local_guarantee` is set at binding: the sketch's error over an exact input, `None` when the model has no error model for the family | -| `SummaryAgg` | stored: ExactAggregate family composed with the child's; sketch families `None` | **derived**: ExactAggregate family: exact, composed with the child's under `exact_rule`, except `ExactKind::Count`, exact whatever the child (as today); sketch families `Set(None)`, state has no guarantee | -| `NonASAPOp` | pre-ASAP `QueryExpr`: none
post-ASAP `KeepPreAsap`: exact
post-ASAP `ValueOperation` / `BinaryOp` / `RelationalJoin` copies: composed at construction | **derived**: composed from the children; exact if no `ASAP` descendant | -| `FinalizeExactAccumulator` | copies the child's | **derived**: the child's | -| `MaintainPopulation` / `ReadPopulation` | stored: exact | **derived**: exact | -| unused variants | `None`: state has no guarantee of its own | unimplemented (§1.3) | - -### 2.3 Timing: set where position does not decide it - -A timing is filled in two steps: - -1. **Binding** sets every `SummaryAgg` to the timing today's fallback derives: - `IngestionTime`, or `QueryTime` when binding built its child at query time (a - query-time `FinalizeExactAccumulator`). The planner sets every - `FinalizeExactAccumulator`. -2. **`derive_timings`** runs on each assembled root, sharing one memo, top-down: a root is query - time, a set node keeps its value, a node of fixed kind takes that kind's time, and - every other node takes its parent's. A node reached at two timings is copied (§4). +The second diagram abbreviates the same nesting: `ASAP(SummaryAgg)` means an +`OperatorNode` whose `operator` is `Operator::ASAP(ASAPOp::SummaryAgg { ... })`. +Its family is `FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum)`; +its update reads `bytes`, and it uses the same ungrouped reduction. Finalization +must preserve SQL SUM's NULL and empty-input behavior. This example assumes the +existing capability and rewrite checks permit that exact implementation. -``` - binding derive_timings -Project Unset QueryTime ← root - SummaryEstimate Unset QueryTime ← fixed by kind - SummaryAgg IngestionTime IngestionTime ← kept - Scan t Unset IngestionTime ← from parent -``` - -Exported timings are unchanged. Moving the `SummaryAgg` default to `QueryTime`, and -letting the lifecycle step choose ingestion time, is a separate PR. +| Part of the design | Role in this example | +|---|---| +| `OperatorNode` | Every graph node, holding its operation and common result/schema, guarantee and timing properties. | +| `Operator` | Selects the `NonASAP` or `ASAP` operation category in each node. | +| `NonASAPOp` | Scan, filter, aggregate and projection before optimization; scan, filter and projection still use these definitions afterward. | +| `ASAPOp` | Builds accumulator state and finalizes it after the rewrite. | +| `ScalarExpr` | Computes `status = 200` and `sum_bytes + 1` within the filter and projection; neither computation needs a bridge node. | +| `Rc` | Connects each consumer to its producer, including `Project.child` pointing to an ASAP finalization node. | -Per node kind: +For this example, assume `bytes` is nullable `Int64`. The output metadata is: -| Node | Today | After | +| Node | `result_kind` | Output columns (`name: dtype`, nullability) | |---|---|---| -| `NonASAPOp` | pre-ASAP `QueryExpr`: none
post-ASAP `KeepPreAsap`: from the consuming edge
post-ASAP `ValueOperation` / `BinaryOp` copies: a stored field | **derived** from the consuming edge (§5) | -| `SummaryAgg` | from the child; ingestion time under `KeepPreAsap` | **set** by binding, as today's fallback: `IngestionTime`, or `QueryTime` over a query-time child | -| `FinalizeExactAccumulator` | a stored field, set by the planner | **set** by the planner: the same position allows either time | -| `SummaryEstimate` | query time, fixed by the kind | **derived** from the kind: query time | -| `MaintainPopulation` / `ReadPopulation` | a stored field: population timing set by its lifecycle; readout always query time | population: **set** by its lifecycle; readout: **derived**, query time | -| unused variants | `SummaryMerge`: a stored field; `Join` / `Subtract` / `Delete`: ingestion time | unimplemented (§1.3) | - -Unlike a guarantee, a timing depends on the parents, so `derive_timings` needs the whole -DAG and runs only after assembly. +| Aggregate before optimization | `Relation` | `sum_bytes: Plain(Int64)`, nullable | +| Summary build after optimization | `State` | `sum_state: ExactAggregate(Sum, Sum)`, non-null accumulator state | +| Finalize after optimization | `Relation` | `sum_bytes: Plain(Int64)`, nullable | +| Project in either graph | `Relation` | `total_bytes: Plain(Int64)`, nullable | -### 2.4 Workflow of setting up `guarantee` and `timing`: today vs. after +The empty accumulator finalizes to SQL NULL; the accumulator itself is state, not +a nullable numeric value. The projection consumes the finalized column. Guarantees +follow the existing assessment rules, while `timing` may remain `None` until +physical planning. The topmost Project node produces the query result. -Today: +This illustrates the connection between the two proposals: scalar separation +makes predicates and value expressions explicit; operator unification lets those +same ordinary operations consume ASAP results through normal graph edges. -``` -search / binding each SummaryNode's guarantee is composed when the node is built; - BinaryOp / ValueOperation store their timing -selection reads each candidate's stored guarantee against its target -assembly assemble_residual builds kept nodes and composes their guarantee; - relink_summary copies the old guarantee onto a relinked SummaryAgg -lifecycle reads the root's guarantee -export validate_execution_data_states, per root, derives the remaining - timings into a side table and writes them onto the edges -``` - -After: +### 1.4 Scope of operator sharing -``` -search / binding sets only what cannot be derived: SummaryAgg.timing, - local_guarantee, exact_rule; checks accuracy on a derived copy, - then drops the copy -assembly builds each root; kept NonASAP nodes stay as they are -derive_timings per root, one shared memo: fills timings top-down, copies a node - read at two timings -derive_guarantees per root, one shared memo: fills guarantees bottom-up -lifecycle reads the derived guarantee -export reads the slots; rejects an Unset one -``` +Here, sharing means pre-ASAP and post-ASAP use the same operator definitions. +A `Project`, for example, has one representation whether its input is an ordinary +aggregate or a summary estimate. This proposal removes the representation boundary; +it does not introduce rules for sharing computations across queries. ---- +## 2. Node properties and why they differ -# II. Changes, in data-flow order - -## 3. Optimizer entry - -Frontends and `resolve` build `NonASAP` trees only and access children with -`expect_non_asap()`. `search_cse_workload_with`, which every `search_workload*` entry -reaches, panics on a root that `contains_asap()`: an ASAP node there is a caller bug. +Both operation categories use the `OperatorNode` declared in the §1.1 overview. +That resolved node follows the current +`SummaryNode` separation between an operation and its metadata, generalized to +all operators. The field is named `operator` because it holds `Operator` (§1.1), +not a scalar expression. The table below explains those common fields; individual +operation variants do not repeat them. ```rust -impl Operator { - pub fn contains_asap(&self) -> bool; - pub fn expect_non_asap(&self) -> &NonASAPOp; // an ASAP node here is a bug: panic +// Existing enum; the node's Option represents an unassigned phase. +enum ExecutionTiming { + IngestionTime, + QueryTime, } ``` -A compile-time alternative — an associated type on `ColState` with -`ColumnRef::ASAP = Never` — only protects frontend code before `resolve`: frontends -already return `ColumnId` trees, where `ASAP` is allowed. The entry check covers every -input (frontends after `resolve`, deserialized plans, test IR) with simpler types. +| Field | Meaning | How it is determined | +|---|---|---| +| `operator` | Operation category, parameters and dependencies | `Operator`, `NonASAPOp` and `ASAPOp` in §1 | +| `result_kind`, `schema` | The output category and fields, including identity/time metadata | Derived from `operator` and its actual inputs, then retained on the resolved node (§2.1) | +| `guarantee` | An established result-accuracy guarantee, when available | Existing `ResultGuarantee` and composition rules (§2.2); `None` never means exact | +| `timing` | The assigned ingestion/query execution phase | Physical planning under #509 (§2.3); `None` means not assigned | -## 4. Planner: search and assembly +`ResultGuarantee` retains its existing definition. `Operator`, `OperatorNode`, +`OperatorResultKind` and the common node layout are proposed; `Schema` is unified +as specified below. This is a resolved-plan interface: name resolution must finish +before producing these concrete `ColumnId`/`Schema` nodes. -```rust -pub enum Replacement { - Subtree(Rc), // formerly Summary(Rc) and Rewrite(Rc) - ExactComposition { .. }, // its plan becomes Rc -} -``` +| Plan stage | Required property state | +|---|---| +| Resolved frontend / logical candidate | Valid `result_kind` and `schema`; `guarantee` only where established; `timing` may be `None`. | +| Executable physical candidate | Valid output metadata, accuracy acceptable under the existing requirements, and `Some(timing)` for every executable operator. | -**Candidates stay bottom-up, as today**: a candidate is built on a concrete child plan -(`realize_child_with`, or each child candidate in `prepare_compositions`), so a chosen -plan is complete. Where `realize_child_with` falls back to `keep_pre_asap` today, it -returns the child's original subtree, and assembly keeps it as is. +Changing an operation or dependency requires re-deriving its output metadata and +revalidating dependent guarantees and timing assignments. Derived fields must not +retain facts from the plan that was replaced. This defines consistency, not a new +caching or mutation mechanism. -**Accuracy check during search**: binding sets no `guarantee` slot (§2.2), so the -candidate filter in `search_workload_with_targets` and `prepare_compositions` run -`derive_guarantees` on the candidate alone, with a fresh memo, then check its accuracy target. The derived -copy is only read, then dropped: CandidateLogicalASAPDAGs keeps the original candidate, whose nodes are -shared with other queries. +### 2.1 One schema model for values and state -**Assembly** — one rule replaces `assemble_residual`: +Use one `Schema` for operator outputs before and after optimization. Rename today's +`SummaryFamilyType` to `FieldDataType`: it types every field, and `Plain` is not a summary +family. Rename `Column` to `Field` and `Schema.columns` to `Schema.fields`: the struct +describes a column and holds none of its data. Retain the current `Schema` metadata. +The following is the proposed resolved interface; it is not the current Rust definition. ```rust -fn assemble(&self, t: &Rc) -> Rc { - memo by ptr; // shared children stay one Rc - let chosen = if query_time_nested_sum(t) { None } // as today: keep the outer SUM so the - else { self.chosen(t) }; // inner target's own choice is assembled - match chosen { - Some(Subtree(r)) => r, // a complete plan, used as is - Some(ExactComposition{..}) => composition.plan, - None => t.map_children(|c| if is_target(c) { self.assemble(c) } else { c }), - // keep the node, assemble its children — assemble_residual does this for four operators only - } +struct Field { + name: String, + dtype: FieldDataType, + nullable: bool, + table: Option, } -``` - -Then `GlobalSelection::assemble_selected_dag` runs `derive_timings` (§5) → -`derive_guarantees` (§2.2) on each root it assembles. The `DerivationMemo` lives on -`GlobalSelection` next to `assembled_nodes`, so roots assembled one call at a time still -share nodes. `assemble_selected_dag_with_summary_maintenance_lifecycles` plans lifecycles -on that result, as today: lifecycle planning holds `Rc`s into the plan and reads the -root's guarantee, so it must see the derived tree. - -- **Illegal child** (e.g. a query-time `SummaryEstimate` under a `SummaryAgg`): candidates - are checked when built with `derive_timings_at(candidate, placement's timing)`, as - `relink_summary` does today with `validate_execution_data_states_at`. An error from `derive_timings` after assembly is - a bug, and planning fails with that error. -- **A shared subtree read at two timings**, e.g. a query-time `Aggregate` and an - ingestion-time `SummaryAgg` reading one `Scan`: both choices are legal, only the sharing - is not. `derive_timings` memoizes by (pointer, timing), so it builds one copy per - timing; a subtree read at one timing stays one `Rc`, within a root or across roots. - -`map_children` is `rebuild_children` from -`pre_asap/cse.rs`, dispatching to `NonASAPOp::map_children` / `ASAPOp::map_children`. -Deleted: `assemble_residual`, `keep_pre_asap` / `keep_pre_asap_rc`, and the -`KeepPreAsap` branch of `finalize_exact_accumulator`. Kept: `relink_summary` and the -`query_time_nested_sum` special case, which pick a child after selection today. - -| #468 problem | Resolution | -|---|---| -| 1. A `Project` is a `QueryExpr` inside `KeepPreAsap` and a `ValueOperation` outside | one set of types | -| 2. Nothing outside `KeepPreAsap` can reference the `Scan` inside, so an exact aggregate and a sketch cannot share a scan | `Aggregate` and `SummaryAgg` can point to the same `Scan`. This holds when both run at the same time; otherwise the scan is copied (above). With today's defaults the sketch runs at ingestion time and the exact `Aggregate` at query time, so they share only after the `QueryTime` default (separate PR, §2.3). Splitting a multi-measure `Aggregate` into exact + sketch is a binding rule, out of scope (§9) | -| 3. `SetOp` and similar have no post-ASAP copy, so no summary below them | `SetOp` takes `None => t`; both children are assembled | - -## 5. Timing: execution data states - -`validate_execution_data_states` becomes `derive_timings`: instead of returning the -`ExecutionDataStateAssignment` side table (deleted), it writes each node's `timing` slot. -`produced_data_state(KeepPreAsap) = None` (set by the consuming edge) extends to all -`NonASAPOp`s: - -| Node | Produced state | -|---|---| -| `NonASAPOp` | set by the consuming edge (`QUERY_ROWS` at the root), passed to its children | -| `SummaryAgg` | `{timing, SummaryState}` from its `timing` slot (§2.3); a `NonASAPOp` child takes the same timing | -| other `ASAPOp` | unchanged | -A data state is the `timing` slot plus a primitive (`Raw` / `SummaryState` / …) fixed by -the kind; only the timing is stored. +struct Schema { + fields: Vec, + time_index: Option, + unique_keys: Vec>, + closed: bool, +} -The `KeepPreAsap` / `BinaryOp` / `ValueOperation` / `RelationalJoin` arms of today's -`validate_execution_data_states` merge into one `NonASAP` arm of `derive_timings`: +// Today's `SummaryFamilyType`, renamed; variants and payloads unchanged. +enum FieldDataType { + Plain(DataType), + ExactAggregate(ExactKind, ExactParams), + Sketch(SketchKind, GroupingStrategy), + Sample(SamplingKind, SamplingParams), + Wavelet(WaveletKind, WaveletParams), + StatModel(StatModelKind, StatModelParams), +} -- Pass the state to each child; an `ASAP` child is checked by the `ASAP` edge rules. -- `check_plain_operands` stays: referenced columns must be `FieldType::DataType` - (`Project` / `Filter` / `Sort` / `Limit` may pass `ExactAggregate` columns through). - This rejects `Project(ASAP(SummaryAgg))`. -- `BinaryOp`'s ingestion-side constraints move into this arm. -- `AmbiguousKeepPreAsap` is deleted: a subtree read at two timings is copied (§4). +// Proposed derived output classification, separate from column types. +enum OperatorResultKind { + Relation, + InstantVector, + RangeVector, + State, +} -## 6. Export: fragments in the post-ASAP DAG +impl Operator { + fn output_schema(&self) -> Result; + fn output_kind(&self) -> Result; + fn validate_inputs(&self) -> Result<(), SchemaDerivationError>; +} -The four original-operator payloads (`fallback{expression: QueryExpr}`, `binary`, -`value`, `relational_join`) become one: +impl OperatorNode { + fn validate_structure(&self) -> Result<(), SchemaDerivationError>; + fn validate_execution_timing(&self) -> Result<(), SchemaDerivationError>; +} -```rust -PostAsapOperatorPayload::Relational { - /// No ASAP node inside. Leaves are Scans, or Scan { source: Source::DagInput { role } } - /// for an incoming edge whose schema is the edge's intermediate_schema. - expression: Operator, +impl ScalarExpr { + fn scalar_type(&self, input: &Schema) -> Result<(DataType, bool), SchemaDerivationError>; } ``` -`compile_post_asap_dag` takes each **largest connected subtree without `ASAP`** as one -fragment, cutting an edge with a `DagInput` leaf wherever it meets an `ASAP` node. -`ASAP` nodes map one-to-one onto the existing summary payloads; -`FinalizeExactAccumulator` / `MaintainPopulation` / `ReadPopulation` stay -`value{operation}`. A backend lowers every fragment with its existing `QueryExpr` -lowering plus a `DagInput` arm (an incoming edge as a materialized table); the -`binary` / `value::Project` / `relational_join` lowerings go. - -- **Wire 5 → 6**: three fewer payloads; `fallback` becomes `relational` with `DagInput` - leaves; `output_schema` / `intermediate_schema` become `Schema`. One cutover (§8 - stage 4), together with the downstream readers. -- **Timing and guarantee** are read from the node slots; an `Unset` slot is rejected. - An edge's `data_state` is its producer's timing plus the primitive of its kind (§5). - `compile_post_asap_dag` no longer re-runs data-state validation. -- **`SummaryMerge`** stays a wire payload, although its planner-side variant is - unimplemented (§1.3, §10). -- **Phases** become per fragment. Switching phase inside a fragment would need a - materialization point, and those are `ASAP` nodes, so nothing is lost. - -## 7. Other consumers - -| Location | Change | -|---|---| -| `post_asap/cse.rs` | delete; `share_common_subtrees` covers `ASAPOp` (derives `PartialEq` + serde) | -| `dag_export.rs` | delete `build_summary` / `build_summary_hybrid` / `summary_kind_tag`; one exporter with an `ASAP` arm; update the viewer's `node-style.js` and the pin test `viewer_categorizes_exactly_the_exported_node_kinds` | -| `summary_maintenance_cost/estimator.rs` (80 sites) | `KeepPreAsap` branches (`query_source_selections`, `retained_queries`) use the §6 fragment; `exact_binary` / `value_operation` costs fold into it. Also fixes the missing `RelationalJoin` arm in `summary_operation_evidence` | -| `physical_plan_cost_model.rs::estimate_candidate` | every fragment goes through `lower_query_physical_dag` | -| `summary_maintenance_lifecycle.rs` | `selected_raw_recompute` becomes `!contains_asap(root)`; the `keep_pre_asap(target)` fallback in `assemble_selected_dag_with_summary_maintenance_lifecycles` becomes `target` | -| `maintained_population.rs` | `KeepPreAsap(source)` becomes `source`; `population.matches_input` reads an `NonASAP` child directly | -| `exact_composition.rs` | `ExactOperation::Aggregate` becomes a `NonASAP(Aggregate)`, built over each child candidate as `prepare_compositions` does today | -| `RelationalJoin.pruning` | never set to `Some` in production; delete. Candidate pruning can return as an `ASAPOp` variant | - ---- - -# III. Implementation - -## 8. Stages and tests +**Relationship to current types.** `Field` is today's pre-ASAP `Column` with `dtype` +widened from `DataType` to `FieldDataType`. `FieldDataType` is today's `SummaryFamilyType` +under a name that also fits its `Plain` case. The proposed common `Schema` replaces +the separate operator-edge roles of pre-ASAP `Schema` and post-ASAP `SummarySchema` / +`SummaryField`; it does not rename `DataType`. A pre-ASAP value column becomes +`Plain(dtype)`. +Frontend validation permits only ordinary value columns, preserving the current +pre-ASAP restriction even though the common schema can also express state. -`main` builds and passes all tests after every stage. - -| Stage | Content | Touches | -|---|---|---| -| 0 Preparation | `Rc` for `Concat.children`; `rebuild_children` → `map_children`; `Column::plain` | `asap-types` | -| 1 Split | [decoupling doc](decoupling_op_and_expr.md): `NonASAPOp` + `ScalarExpr`; children stay `Rc` | scalar code ([decoupling doc §3](decoupling_op_and_expr.md#3-changes)) | -| 2 Two levels | §1.1, §1.4: `Operator`, an empty `ASAPOp`, `contains_asap()`, `expect_non_asap()`; child slots become `Rc>`; every variant gets `timing` / `guarantee` slots, and nodes are built through constructors that leave both `Unset` | every crate; the same mechanical change everywhere | -| 3 One schema | §2.1: `Column` → `Field` and `Schema.columns` → `fields` (serde keeps the name `columns` until stage 4); `FieldType`, `ASAPType`, `PlainField`, `Schema` everywhere except the `post_asap_dag.rs` wire types, which keep `SummarySchema` until stage 4. **No wire change** | `asap-types` + schema construction in every crate | -| 4 New types | fill `ASAPOp`; `ASAP` arms of `output_schema`; `derive_guarantees` and `derive_timings` (§2.2, §5); the entry check (§3); `flatten(&SummaryNode) -> Rc` so export runs on the new types, copying each node's guarantee and today's derived timing into the slots, so the export is unchanged; wire types become `Schema`, and `Schema.fields` serializes as `fields`. Wire → 6. **The only wire-breaking stage**; merged together with ASAPQuery-backend and ASAPCollector | `asap-types`, `devtools`, viewer | -| 5 Planner | §4: candidates and assembly on `Rc`; §7 moves to the new types; delete `flatten` | `asap-aware-mapping` | -| 6 Cleanup | delete `SummaryExpr`, `SummaryNode`, extra `ValueOperation` variants, `ExactOperation`, `post_asap/cse.rs`; update `post-asap-ir.md`, `physical-plan-integration.md`, developer and viewer docs | docs | - -Wrapping pre-ASAP operators in `ValueOperation` first is not planned: stage 4 gives -the same early flat export, on the final types. - -**Tests**: - -- One integration test per #468 problem: - 1. `WITH metric AS (SELECT avg(CASE WHEN l_quantity BETWEEN 1 AND 50 THEN 1.0 ELSE 0.0 END) AS in_range FROM lineitem) SELECT in_range, in_range = 1.0 AS ok FROM metric` — no post-ASAP-only node besides `ASAP`; all `Project`s are one variant. - 2. `SELECT avg(l_extendedprice), approx_percentile_cont(l_discount, 0.99) FROM lineitem` — the `avg` `Aggregate` and the KLL `SummaryAgg` share one `Scan` by `Rc::ptr_eq` when both run at the same time (once a binding rule splits measures); with today's ingestion-time default the `Scan` is copied. - 3. `SELECT approx_distinct(l_partkey) FROM lineitem UNION ALL SELECT approx_distinct(l_suppkey) FROM lineitem` — each side of the `SetOp` has a `SummaryEstimate`. -- A shared `Scan` read at two timings is copied once per timing; read at one timing, it stays one `Rc`. -- A node shared by two roots, assembled in two calls, is still one `Rc` after `derive_*`. -- After `derive_*`, no slot is `Unset`; export rejects a tree with one. -- Exported timings of today's plans are unchanged. -- A kept `NonASAP` node (e.g. a `SetOp`) reports the guarantee composed from its assembled children. -- Each unused branch (§1.3) returns `Unimplemented` from `output_schema`, `derive_*` and export. -- `search_workload*` panics on a root containing `ASAP`. -- Rewrite the 117 `SummaryExpr::` assertions in `sql_to_post_asap.rs` / `promql_to_post_asap.rs` / `exact_composition.rs`. -- Wire 6 round trip with a `DagInput` fragment; a version-5 document is rejected. -- The 52 `execution_data_state.rs` tests keep their shapes; assertions read the `timing` slot instead of `ExecutionDataStateAssignment`. - -## 9. Out of scope - -- The binding rule splitting a multi-measure `Aggregate` into exact + summary over one child. -- Candidate pruning as an `ASAPOp` variant. -- Accuracy through `SummaryMerge` / `Subtract` / `Delete` / `Join` (§2.2): an accuracy - descriptor on state, or composing error along the state chain at readout. -- Holes: letting a chosen plan's child be filled by that child target's own choice at - assembly, instead of fixing it when the candidate is built. A search-strategy change, - independent of the types here. -- Folding `ExactComposition` into `Subtree` (both of its forms become expressible); - deferred until stage 5 is stable. - -## 10. Open questions - -- Does ASAPQuery insert `SummaryMerge` only on the exported post-ASAP DAG, or through ASAPPlanner's - post-ASAP types? The planner-side variant is unimplemented (§1.3). +| Field | Meaning and requirement | +|---|---| +| `fields` | Ordered named fields. `Plain(DataType)` is a readable value; other variants retain the identity and parameters of summary or exact-accumulator state. | +| `Field.nullable`, `Field.table` | Preserve SQL nullability and qualified column resolution. | +| `time_index` | Identifies the time column when present; it does not by itself distinguish an instant vector from a range vector. | +| `unique_keys` | Proven column combinations identifying rows; an empty list asserts no known key. Recompute these proofs when a rewrite changes identity. | +| `closed` | Whether `fields` completely describes the output. An open PromQL schema must retain unlisted labels through the existing complete-series-identity contract. | + +`OperatorResultKind` is derived from the operation and its inputs and retained as +`OperatorNode.result_kind`. `State` describes an output carrying unfinalized state; its +schema may also contain ordinary grouping keys. `SummaryEstimate`, +`FinalizeExactAccumulator` and other readouts derive the appropriate relation or +vector kind from their operation and input context. Matching numeric columns do +not make those kinds interchangeable. + +**Interface contracts.** `Operator::output_schema` and `output_kind` derive output +metadata from the payload and validated inputs. `validate_inputs` checks local +producer/consumer compatibility, such as vector inputs for `BinaryOp` or the +required state family for a summary readout. Scalar typing checks the input-kind +contract of `PromqlScalarFromVector` and other scalar plan reads. + +| Validation entry | Scope and stage | +|---|---| +| `OperatorNode::validate_structure()` | Walks the reachable operator graph, including scalar plan references; checks input contracts, scalar typing and agreement between retained and derived output metadata. Valid for logical and physical plans; permits `timing = None`. | +| `OperatorNode::validate_execution_timing()` | Includes structural validation, then requires assigned timing on every executable operator and checks phase dependencies. Used for executable physical candidates. | +| Existing planner assessment and selection (#509) | Establishes guarantees using the existing accuracy models and checks them against request requirements and deployment capabilities. Neither node method re-proves a guarantee or decides request feasibility. | + +The two node methods need only the graph and its annotations. Request requirements +and deployment models remain inputs to the existing planning/selection workflow, +not implicit globals of `validate_structure`. Passing the timing check alone does +not establish that a physical candidate satisfies the query's accuracy requirement. + +`Scan.schema` declares the source columns; `Values.schema` declares the constructed +row shape. `OperatorNode.schema` is the derived output for any operation. A scan's +predicates cannot change its declared output columns; a Values row must match the +declared arity, types and nullability. These leaf outputs retain the declaration's +column layout and time/identity information, with only justified metadata changes. +The declaration and derived output therefore have distinct roles, and structural +validation rejects disagreement rather than trusting two independent schemas. + +`scalar_type` keeps the existing method name and `(DataType, nullable)` result. +Its `input` is the applicable column scope: the child schema for a projection, +both input schemas for a join predicate, or aggregate outputs for `HAVING`. +Explicit subquery/conversion expressions validate their referenced producer using +the contracts above. Numeric expressions cannot consume state columns as numbers. +A standalone scalar expression is checked with an empty column scope and needs no fabricated +relation output schema. `SchemaDerivationError` retains the existing error-type name; +result-kind, state-family, schema and execution-phase mismatches require +corresponding validation errors. + +For example, a KLL build outputs `State` with a +`Sketch(SketchKind, GroupingStrategy)` column identifying KLL and its parameters. +Its p99 readout outputs an ordinary `Plain(Float64)` column in the appropriate +relation/vector schema. A numeric predicate can use that readout, but not the KLL +state. Exact accumulator state similarly requires `FinalizeExactAccumulator`. +An ordinary operator may pass state through only where its input/output contract +permits it. A bare-column projection can preserve the field's `FieldDataType` +directly during `output_schema` derivation; `scalar_type` applies when that column +is used as a scalar value and rejects state. Copying a state column does not turn +it into a readable scalar. + +### 2.2 Preserve existing accuracy semantics + +The unified representation must preserve the existing accuracy model, composition +rules and result guarantees. An operation's guarantee must still account for its +actual inputs, including producers referenced by scalar expressions. Reading an +approximate result through `PromqlScalarFromVector` or a SQL scalar subquery does +not make it exact; unknown accuracy must not be treated as exactness. + +The common node reuses `guarantee: Option` from `SummaryNode`. +`Some` records an established guarantee; `None` covers an unassessed or unknown +result, or state whose accuracy is only established at readout. Exactness must be +explicitly established using the existing model. This proposal introduces no new +accuracy metric or guarantee-calculation workflow. + +### 2.3 Timing follows the planning-stage design + +The [planning-stage design in #509](https://github.com/ProjectASAP/ASAPPlanner/pull/509) +separates logical decisions about what to compute from physical decisions about how +and when to compute it. This proposal follows that division. + +For example, a KLL summary build may execute at ingestion time or query time, +depending on the materialization choice. `OperatorNode.timing` records that assignment +as `Some(ExecutionTiming::IngestionTime)` or `Some(ExecutionTiming::QueryTime)`. +Logical nodes may retain `None`; `validate_execution_timing` rejects unassigned +executable nodes. Timing is common node metadata rather than a separate payload +field on selected `ASAPOp` variants. + +The representation must preserve the resulting execution constraints: ingestion-time +work cannot depend on query-time results, and consumers must receive values or state +that are available when needed. Materialization choices, retention and plan selection +remain governed by #509; this document does not define another lifecycle policy. +These constraints also apply to query subgraphs referenced by scalar expressions. + +PromQL evaluation timestamps and SQL statement time are separate from these +execution phases. `TimeShift`, subquery grids and `EvalTimestamp` retain their +source-language evaluation context. A shared node identity alone does not permit +reusing a result across different evaluation times. + +## 3. Acceptance criteria + +The design is successful when: + +- A projection uses the same semantics above and below summary computations. +- A union or another ordinary operator can consume summary estimates on its inputs. +- Unifying the representation preserves existing graph dependencies, including + any shared inputs; it does not introduce new sharing rules. +- Existing value/state, accuracy and execution constraints remain enforceable on + the unified representation. +- Scalar expressions and conversions use the same representation before and after + optimization, with no bridge nodes or hidden subplans. +- Structural and timing validation include query subgraphs referenced by scalar + expressions; planner assessment includes their accuracy dependencies. diff --git a/docs/design_docs/proposals/planner-layering.md b/docs/design_docs/proposals/planner-layering.md index b4b021608..0056c3507 100644 --- a/docs/design_docs/proposals/planner-layering.md +++ b/docs/design_docs/proposals/planner-layering.md @@ -178,7 +178,7 @@ A summary-based candidate uses three kinds of summary nodes: summaries into a coarser one. * A **summary estimation node** computes an answer from a summary, for example the p99 estimate from a KLL, or the entropy estimate from a UnivMon. -* **summary subtract node** and **summary delete node** design is TODO. +* **summary subtract node** and **summary delete node** design is TODO. One summary build node can feed several estimation nodes, which is what Pass 2 exploits. diff --git a/docs/develop_docs/asap-aware-mapping-architecture.md b/docs/develop_docs/asap-aware-mapping-architecture.md index f7e2a9a7b..a631e7f65 100644 --- a/docs/develop_docs/asap-aware-mapping-architecture.md +++ b/docs/develop_docs/asap-aware-mapping-architecture.md @@ -55,18 +55,20 @@ The diagram below follows a workload of one or more query roots through target d Terminology used in the diagram: - A **workload** is the set of named queries planned together. A **query root** - is the top-level `QueryExpr` (the logical query-expression type) for one of - those queries. **Pre-ASAP** means this logical input form, before the planner - realizes an operation as a concrete ASAP realization; **post-ASAP** means - the resulting realization form. -- A **DAG** (directed acyclic graph) represents query operators whose subtrees + is the top-level `Rc` (the unified operator IR) for one of + those queries. **Pre-ASAP** means a DAG that contains only ordinary + `NonASAPOp` operators, before the planner realizes an operation with ASAP + primitives; **post-ASAP** means the same IR after some nodes became `ASAPOp` + summary operators. +- A **DAG** (directed acyclic graph) represents query operators whose sub-DAGs may be shared. **CSE** (common subexpression elimination) finds equivalent - subtrees and represents legal reuse by making them the same shared node. + sub-DAGs and represents legal reuse by making them the same shared node. Rust's `Rc` (reference-counted pointer) records that shared node identity. - A **target** is one replaceable site. A **candidate** is one valid alternative - for it. `Replacement::Summary` is a constructed post-ASAP summary—maintained state - such as an exact accumulator or an approximate sketch—while - `Replacement::Rewrite` is another pre-ASAP logical expression. + for it. `Replacement::SubDag` is a replacement sub-DAG: either a constructed + post-ASAP summary (it contains an `ASAPOp`, e.g. an exact accumulator or an + approximate sketch) or a logical rewrite with no ASAP operator + (`is_logical_rewrite` tells them apart). `Replacement::ExactComposition` refers to a child target whose realization must remain undecided until compatible selection. A **sketch** is a compact data structure that trades exactness for bounded error. A @@ -86,15 +88,15 @@ flowchart TB classDef report fill:#f2eafe,stroke:#7950b3,color:#34204f subgraph DISCOVERY[1. Discover every replaceable site] - WL["Input workload
one or more named pre-ASAP QueryExpr roots"]:::input + WL["Input workload
one or more named pre-ASAP OperatorNode roots"]:::input SEARCH["search_workload_with
run CSE once, then visit every node in every root DAG"]:::generate - TARGET["TargetSubDAG
one candidate site plus the number of workload locations
that reference the same Rc<QueryExpr>"]:::generate + TARGET["TargetSubDAG
one candidate site plus the number of workload locations
that reference the same Rc<OperatorNode>"]:::generate WL -->|"roots"| SEARCH -->|"one target per distinct node"| TARGET end subgraph GENERATION[2. Generate all legal alternatives at each site] STRATEGY["ReplacementStrategy
when a target matches, enumerate every legal replacement;
implementations generate but do not choose"]:::generate - CAND["ReplacementSubDAG candidates
each contains a Summary, Rewrite or ExactComposition
plus typed provenance and rationale;
no alternative is removed solely on cost"]:::store + CAND["ReplacementSubDAG candidates
each contains a Subtree (summary or logical rewrite) or ExactComposition
plus typed provenance and rationale;
no alternative is removed solely on cost"]:::store TARGET -->|"try every registered strategy"| STRATEGY --> CAND CM(["CostModel
orders candidates and supplies
deployment-specific parameters"]):::choose CM -. "rank and parameterize; accuracy checks remain required" .-> STRATEGY @@ -120,7 +122,7 @@ flowchart TB ``` The generic `ReplacementStrategy` box is the extension point. The default -registry supplies summary realization, Hydra grouping, shared-subtree, +registry supplies summary realization, Hydra grouping, shared-sub-DAG, average-rewrite and exact-composition strategies. Section 3.3 describes the registries and the workload-derived roll-up rule. @@ -138,7 +140,7 @@ complete alternative set has been built. Use `search_workload` or `search_workload_with` for normal planner search. The search performs these steps: -1. Run CSE once to merge structurally identical subtrees that may legally be +1. Run CSE once to merge structurally identical sub-DAGs that may legally be shared. 2. Walk the complete DAG beneath every query root, including nodes below unshared parents. @@ -157,10 +159,10 @@ flowchart LR classDef workload fill:#e7f7ef,stroke:#31835e,color:#173f2d classDef common fill:#fff6dd,stroke:#b78922,color:#513d0c - ROOTS["Input
one or more named QueryExpr roots"]:::workload - ROOTS --> CSE["Canonicalize sharing
merge structurally identical, legally shareable subtrees"]:::workload + ROOTS["Input
one or more named OperatorNode roots"]:::workload + ROOTS --> CSE["Canonicalize sharing
merge structurally identical, legally shareable sub-DAGs"]:::workload CSE --> WALK["Discover sites
walk the complete DAG, including nodes below unshared parents"]:::workload - WALK --> T["Build TargetSubDAG
retain the subtree's Rc identity and measured consumer_count"]:::workload + WALK --> T["Build TargetSubDAG
retain the sub-DAG's Rc identity and measured consumer_count"]:::workload T --> MATCH MATCH["matches(target)
cheaply decide whether this strategy has alternatives"]:::common MATCH -->|"true"| REPLACE["propose(target)
construct supported legal alternatives;
retain structured accuracy rejections"]:::common @@ -169,7 +171,7 @@ flowchart LR ``` `consumer_count` is workload information, not an estimate of runtime -executions. It matters to strategies such as `SharedSubtreeStrategy`, which +executions. It matters to strategies such as `SharedSubDagStrategy`, which only has a share-versus-recompute choice when a target has multiple consumers. ### 3.2 Generate candidates through `ReplacementStrategy` @@ -199,13 +201,13 @@ cost. The default context-free registry contains five `ReplacementStrategy` implementations: -- `SketchAlgorithmStrategy` matches supported aggregate and binary shapes. Its - `replacements(target)` method constructs every legal post-ASAP `SummaryNode`, +- `ASAPStrategies` matches supported aggregate and binary shapes. Its + `replacements(target)` method constructs every legal post-ASAP summary sub-DAG, including applicable sketch, exact-accumulator, and pass-through realizations. Candidates are sized and ordered for the target's accuracy requirement; candidates without a sufficient guarantee are rejected before costing. -- `SharedSubtreeStrategy` uses `consumer_count` to identify shared targets. It +- `SharedSubDagStrategy` uses `consumer_count` to identify shared targets. It emits both build-once-and-share and recompute-independently rewrites when a target has multiple consumers. - `HydraGroupingStrategy` proposes eligible shared multi-subpopulation layouts. diff --git a/docs/develop_docs/asap-aware-mapping-contracts.md b/docs/develop_docs/asap-aware-mapping-contracts.md index d1245366f..51a90b2a7 100644 --- a/docs/develop_docs/asap-aware-mapping-contracts.md +++ b/docs/develop_docs/asap-aware-mapping-contracts.md @@ -10,25 +10,25 @@ first; use the [extension guide](extend-asap-aware-mapping.md) when changing one ### `TargetSubDAG` -A pre-ASAP `QueryExpr` node that a strategy may replace. +A pre-ASAP `OperatorNode` that a strategy may replace. ```rust pub struct TargetSubDAG<'a> { - pub root: &'a Rc, + pub root: &'a Rc, pub consumer_count: usize, } ``` -`root` is the actual `Rc` from the workload. +`root` is the actual `Rc` from the workload. -`consumer_count` counts structural references, not runtime executions. It is the number of places in the workload DAG that point to this exact `Rc` node. +`consumer_count` counts structural references, not runtime executions. It is the number of places in the workload DAG that point to this exact `Rc` node. For example, consider two top-level queries: - `sum by (service) (rate(m[5m]))` - `avg by (service) (rate(m[5m]))` -After `share_common_subtrees` merges their identical `rate(m[5m])` subtrees, both query trees point to the same `Rc`. That node's `consumer_count` is `2`, regardless of how often either query executes. +After `share_common_subdags` merges their identical `rate(m[5m])` sub-DAGs, both query trees point to the same `Rc`. That node's `consumer_count` is `2`, regardless of how often either query executes. Use: @@ -54,19 +54,24 @@ when the caller already knows the real number of consumers. The actual object that substitutes the target. -There are currently three forms: +There are currently two forms: ```rust pub enum Replacement { - Summary(Rc), - Rewrite(Rc), + SubDag(Rc), ExactComposition(ExactComposition), } ``` -Use `Replacement::Summary` when the alternative is a constructed post-ASAP summary plan. +Use `Replacement::SubDag` for a replacement sub-DAG. It is one of: -Use `Replacement::Rewrite` when the alternative is still a logical pre-ASAP `QueryExpr`. +- a constructed post-ASAP summary plan: the sub-DAG contains an `ASAPOp` + (`SummaryAgg`, `SummaryEstimate`, ...); +- a logical rewrite: only `NonASAPOp` nodes and no guarantee yet. + +`is_logical_rewrite(&node)` tells the two apart. A kept pre-ASAP sub-DAG +(`retain_exact`) has no ASAP operator but carries an exact guarantee, so it +counts as a bound decision, not a rewrite. Use `Replacement::ExactComposition` when an exact operation refers to a child target whose realization must remain undecided. Selection coordinates the @@ -77,18 +82,18 @@ Examples: ```text Quantile(...) - -> KLL SummaryNode + -> SummaryEstimate(SummaryAgg(KLL)) ``` -is a `Summary`; KLL (Karnin–Lang–Liberty) is a quantile-sketch algorithm. +is a summary `Subtree`; KLL (Karnin–Lang–Liberty) is a quantile-sketch algorithm. ```text compute independently vs. -reuse an already shared logical subtree +reuse an already shared logical sub-DAG ``` -is represented as a `Rewrite`. +is represented as two logical-rewrite `Subtree`s. --- @@ -157,7 +162,7 @@ aggregation must compute without committing to a physical summary algorithm. A realization may be an approximate sketch, an exact mergeable accumulator, or a pass-through that keeps the original operation instead of building a summary. `realizations_for_intent` enumerates these concrete -realizations; `SketchAlgorithmStrategy::replacements()` constructs each one as +realizations; `ASAPStrategies::replacements()` constructs each one as a `ReplacementSubDAG`. It returns all candidates in preferred order without selecting a winner. At workload scale, `search_workload`/`search_workload_with` preserve all supported legal alternatives @@ -170,8 +175,8 @@ This guide uses the Cascades/Volcano terminology: realization. For example, a quantile `AggIntent` may have KLL and DDSketch `Realization` values. - A **transformation rule** maps a logical operation to another logical - operation. In this crate, that kind of candidate is represented by - `Replacement::Rewrite`. + operation. In this crate, that kind of candidate is a logical-rewrite + `Replacement::SubDag`. - A **replacement candidate** packages either kind of result as a `ReplacementSubDAG` for search. `CandidateLogicalASAPDAGs` stores and ranks these candidates. - **Physical commitment and placement** happen downstream. An `Realization` @@ -183,7 +188,7 @@ The concrete flow is: ```text AggIntent -> realizations_for_intent(): enumerate Realization values - -> SketchAlgorithmStrategy: construct ReplacementSubDAG candidates + -> ASAPStrategies: construct ReplacementSubDAG candidates -> CandidateLogicalASAPDAGs: store and rank candidates -> downstream deployment: select and place a final choice ``` @@ -206,7 +211,7 @@ bounds, but does not execute workloads or own deployment measurements. Most hook | `rank_candidates` | Order valid sketch algorithms | No | | `size_params` | Convert an accuracy target into sketch parameters | Yes | | `realize_extension` | Map a custom intent to a realization | Yes | -| `readout_extension` | Query a custom extension summary | Panics until paired with a custom realization | +| `evaluation_extension` | Query a custom extension summary | Panics until paired with a custom realization | | `cse_recompute_cost` | Estimate independent recomputation | Yes | | `cse_shared_maintenance_cost` | Estimate shared maintenance | Yes | | `cse_share_decision` | Choose sharing or recomputation | Yes | @@ -247,13 +252,13 @@ bounds, but does not execute workloads or own deployment measurements. Most hook fn realize_extension(&self, ext_kind: &str, payload: &serde_json::Value) -> Realization; ``` -- **`readout_extension`** — define how queries read an extension summary that `realize_extension` mapped to a `Sketch`. The two hooks are a pair: realization defines what is maintained; readout defines how it is queried. Override both for the same `ext_kind`. The default readout panics to prevent a silent wrong answer. +- **`evaluation_extension`** — define how queries read an extension summary that `realize_extension` mapped to a `Sketch`. The two hooks are a pair: realization defines what is maintained; evaluation defines how it is queried. Override both for the same `ext_kind`. The default evaluation panics to prevent a silent wrong answer. ```rust - fn readout_extension(&self, ext_kind: &str, payload: &serde_json::Value, col: &ColumnRef) -> SketchQuery; + fn evaluation_extension(&self, ext_kind: &str, payload: &serde_json::Value, col: &ColumnRef) -> SketchStatistic; ``` -- **`cse_recompute_cost`** — estimate the one-time cost of recomputing a CSE candidate's subtree independently at a single consumer. Default: `default_cse_recompute_cost`, a structural-size proxy. +- **`cse_recompute_cost`** — estimate the one-time cost of recomputing a CSE candidate's sub-DAG independently at a single consumer. Default: `default_cse_recompute_cost`, a structural-size proxy. ```rust fn cse_recompute_cost(&self, candidate: &CseCandidate) -> Cost; @@ -297,23 +302,23 @@ A custom cost model does not necessarily need to override every hook. The curren // One TargetSubDAGCandidates per distinct TargetSubDAG in the whole workload — // never a flat list of fully assembled plans. pub struct TargetSubDAGCandidates { - pub target: Rc, + pub target: Rc, pub consumer_count: usize, pub candidates: Vec, // accepted alternatives, unranked pub rejected: Vec, // failed accuracy checks } pub struct RankedTargetSubDAGCandidates<'a> { - pub target: &'a Rc, + pub target: &'a Rc, pub consumer_count: usize, pub candidates: Vec<&'a ReplacementSubDAG>, // same candidates, ranked pub costs: Vec, // costs[i] <-> candidates[i] } ``` -`search_workload(roots)` runs the shared-subtree pass once, discovers every target across every root's whole DAG (not just root-level sharing — a `SharedSubtreeStrategy` candidate three levels under an unshared `Filter` is exactly as real a site as a shared whole root), and asks every registered strategy to a fixpoint. Two logically different candidates at two different targets are never copied into two separate plans — they're two entries in two different `TargetSubDAGCandidates`s, sharing every other node in the workload by construction. +`search_workload(roots)` runs the shared-sub-DAG pass once, discovers every target across every root's whole DAG (not just root-level sharing — a `SharedSubDagStrategy` candidate three levels under an unshared `Filter` is exactly as real a site as a shared whole root), and asks every registered strategy to a fixpoint. Two logically different candidates at two different targets are never copied into two separate plans — they're two entries in two different `TargetSubDAGCandidates`s, sharing every other node in the workload by construction. -`CandidateLogicalASAPDAGs::cost_sorted(cost_model)` is the one ranking step: for each candidate set, it dispatches by candidate shape — a same-shape `Rewrite` pair (a `SharedSubtreeStrategy` share/recompute choice) goes through `CostModel::cse_share_decision`; a same-shape run of `Summary` candidates realizing sketches (a `SketchAlgorithmStrategy` choice) goes through `CostModel::rank_candidates`; and a mixed candidate set is ordered by each candidate's `CostModel::estimate_cost`. Every candidate gets a numeric cost aligned index-for-index in `costs`. Count in, count out—nothing is dropped to produce a ranking. Legality checks +`CandidateLogicalASAPDAGs::cost_sorted(cost_model)` is the one ranking step: for each candidate set, it dispatches by candidate shape — the `SharedSubDagStrategy` share/recompute pair (recognized by `ReplacementProvenance::CseShare`/`CseRecompute`) goes through `CostModel::cse_share_decision`; a set with a Hydra shared-grid alternative goes through `CostModel::grouping_state_cost`; a set whose candidates all realize sketches (a `ASAPStrategies` choice) goes through `CostModel::rank_candidates`; and any other mixed set is ordered by `CostModel::candidate_cost`. Every candidate gets a numeric cost aligned index-for-index in `costs`. Count in, count out—nothing is dropped to produce a ranking. Legality checks may already have removed proposals before this boundary. In particular, `search_workload_with_targets` checks explicit per-root targets, while retaining direct DDSketch ratios with missing domain evidence and no root guarantee for @@ -329,7 +334,7 @@ Sketches separate their query category from the concrete algorithm and its param | Level | Type | Example | | --- | --- | --- | -| **family** | `SummaryFamilyType` | `Sketch`, `Sample`, `Wavelet`, `StatModel`, `ExactAggregate` | +| **family** | `FieldDataType` (non-`Plain` variants) | `Sketch`, `Sample`, `Wavelet`, `StatModel`, `ExactAggregate` | | **category** | `SketchCategory` | `Quantile`, `Cardinality`, `Frequency`, `TopK` | | **algorithm** | `SketchAlgorithm` | `Kll` / `DDSketch` (both quantile); `Hll` (HyperLogLog) / `Theta` / `Kmv` (K-Minimum Values), all cardinality | | **committed choice** | `SketchKind` | one validated category + algorithm + parameter combination | @@ -340,7 +345,7 @@ to the selected algorithm and classifies the pair into its category. The public `.category()`, `.algorithm()`, and `.params()` accessors expose the committed values without permitting an invalid combination. -Where this matters in practice: `CostModel::rank_candidates`, `CostModel::size_params`, and `SketchAlgorithmStrategy::replacements` operate at the **algorithm** level. `summary_candidates(intent)` returns a list of `SketchAlgorithm`s (`[Kll, DDSketch]` for a `Quantile` intent), never a bare `SketchKind` with nothing chosen underneath it. `SketchKind` appears after an algorithm has been selected and sized—on `Realization::Sketch(SketchKind)` and `SummaryFamilyType::Sketch(SketchKind)`. +Where this matters in practice: `CostModel::rank_candidates`, `CostModel::size_params`, and `ASAPStrategies::replacements` operate at the **algorithm** level. `summary_candidates(intent)` returns a list of `SketchAlgorithm`s (`[Kll, DDSketch]` for a `Quantile` intent), never a bare `SketchKind` with nothing chosen underneath it. `SketchKind` appears after an algorithm has been selected and sized—on `Realization::Sketch(SketchKind)` and `FieldDataType::Sketch(SketchKind, GroupingStrategy)`. `Sample`, `Wavelet`, and `StatModel` each use a flat `(Kind, Params)` pair. `Sketch` needs the additional algorithm level because multiple algorithms can serve the same purpose—for example, KLL and DDSketch both answer quantile queries. @@ -378,15 +383,15 @@ The crate provides no default `Matcher` implementation because the answer depend Concretely, `explanation.rs` reports three candidate kinds from each `TargetSubDAGCandidates`: -- `ExplanationKind::SketchApproximation` — the set contains a `Replacement::Summary` that realizes `SummaryFamilyType::Sketch(..)`, not just an exact/pass-through candidate. -- `ExplanationKind::CommonSubexpressionReuse` — `consumer_count >= 2` and the set contains `SharedSubtreeStrategy`'s "build once and share" candidate (the `Replacement::Rewrite` whose `Rc` is the set's `target`). +- `ExplanationKind::SketchApproximation` — the set contains a summary `Replacement::SubDag` that realizes `FieldDataType::Sketch(..)`, not just an exact/pass-through candidate. +- `ExplanationKind::CommonSubexpressionReuse` — `consumer_count >= 2` and the set contains `SharedSubDagStrategy`'s "build once and share" candidate (the `Replacement::SubDag` whose `Rc` is the set's `target`). - `ExplanationKind::ExactComposition` — the candidate set contains an exact operation composed with a child target whose realization remains a coordinated choice. Each `ReplacementExplanation::reason` is copied verbatim from the matching candidate's own `ReplacementSubDAG::rationale`. Nothing in `explanation.rs` re-explains why a candidate is valid; that explanation already exists exactly once, on the candidate itself. -`ReplacementExplanation` carries both `node_hash` and `target`. A downstream consumer first compares `node_hash` with an exported `DagNode::hash` to narrow the search, then compares the exact target expression with the node's in-process source expression. This preserves the hash's role as a fast filter while making the final association collision-safe; `location` remains human-readable presentation text rather than a machine identifier. +`ReplacementExplanation` carries both `node_hash` and `target`. A downstream consumer first compares `node_hash` with an exported `DagNode::hash` to narrow the search, then compares the exact `target` node with the exported node's in-process `DagNode::source_node`. This preserves the hash's role as a fast filter while making the final association collision-safe; `location` remains human-readable presentation text rather than a machine identifier. ### Why there is no `ExplanationRule` trait @@ -394,6 +399,6 @@ Explanations are derived from candidates already present in `CandidateLogicalASA ### How it derives `location` text -`CandidateLogicalASAPDAGs`/`TargetSubDAGCandidates` track `Rc` pointer identity, not human-readable breadcrumbs. `ReplacementExplanation::location` provides prose such as `root "dash_a" > lhs` so reporting consumers can identify the relevant part of the query without interpreting pointer identity. Location derivation does not make replacement or costing decisions. +`CandidateLogicalASAPDAGs`/`TargetSubDAGCandidates` track `Rc` pointer identity, not human-readable breadcrumbs. `ReplacementExplanation::location` provides prose such as `root "dash_a" > lhs` so reporting consumers can identify the relevant part of the query without interpreting pointer identity. Location derivation does not make replacement or costing decisions. --- diff --git a/docs/develop_docs/extend-asap-aware-mapping.md b/docs/develop_docs/extend-asap-aware-mapping.md index 235da5460..8af150dba 100644 --- a/docs/develop_docs/extend-asap-aware-mapping.md +++ b/docs/develop_docs/extend-asap-aware-mapping.md @@ -44,7 +44,7 @@ There are four decisions to make. `matches` should contain the minimum structural and semantic checks needed to determine whether the strategy applies. -For example, the aggregate path in `SketchAlgorithmStrategy` requires a +For example, the aggregate path in `ASAPStrategies` requires a supported shape: - the node is an `Aggregate`, @@ -65,7 +65,7 @@ fn matches(&self, target: &TargetSubDAG<'_>) -> bool { } ``` -is enough for the current shared-subtree strategy. +is enough for the current shared-sub-DAG strategy. #### Guideline @@ -112,23 +112,20 @@ and let costing decide later. --- -### Choose `Summary` vs. `Rewrite` +### Summary sub-DAG vs. logical rewrite -Return: +Both are returned as: ```rust -Replacement::Summary(...) +Replacement::SubDag(node) ``` -when the candidate is a fully constructed post-ASAP summary. +- A fully constructed post-ASAP summary: `node` contains an `ASAPOp`. +- A logical pre-ASAP rewrite: `node` has only `NonASAPOp` nodes and no + guarantee. `is_logical_rewrite(&node)` checks this. -Return: - -```rust -Replacement::Rewrite(...) -``` - -when the candidate is a logical pre-ASAP rewrite. +Set `provenance` to say which one it is (`ReplacementProvenance::SummaryRealization`, +`LogicalRewrite`, ...); selection reads provenance, not the sub-DAG's shape. Use `Replacement::ExactComposition` when a candidate depends on a child target whose implementation must be selected compatibly later. Do not bind it to the @@ -204,7 +201,7 @@ how to realize it, wrap that logic. Do not create a second implementation of the same semantics inside the strategy. -The existing `SketchAlgorithmStrategy` is the model to follow: it reuses +The existing `ASAPStrategies` is the model to follow: it reuses `replacement.rs`'s existing candidate list and summary-construction path. --- @@ -229,23 +226,23 @@ If your transformation requires context not currently represented in `TargetSubD --- -### Example: current `SketchAlgorithmStrategy` +### Example: current `ASAPStrategies` -`SketchAlgorithmStrategy` is the reference implementation for a strategy that +`ASAPStrategies` is the reference implementation for a strategy that produces constructed post-ASAP summaries. Construction: ```rust let strategy = - SketchAlgorithmStrategy::default_cost_model(); + ASAPStrategies::default_cost_model(); ``` or with a custom cost model: ```rust let model = MyCostModel; // illustrative -let strategy = SketchAlgorithmStrategy::new(&model); +let strategy = ASAPStrategies::new(&model); ``` The strategy matches supported aggregate nodes. @@ -254,10 +251,10 @@ At a high level: ```mermaid flowchart LR - A["Input TargetSubDAG
root is a supported Aggregate"] --> B["SketchAlgorithmStrategy::matches
check whether the target shape can produce summaries"] - B -->|"true"| C["SketchAlgorithmStrategy::replacements
use CostModel preferences and sizing while preserving
every semantically valid realization"] + A["Input TargetSubDAG
root is a supported Aggregate"] --> B["ASAPStrategies::matches
check whether the target shape can produce summaries"] + B -->|"true"| C["ASAPStrategies::replacements
use CostModel preferences and sizing while preserving
every semantically valid realization"] B -->|"false"| NONE["Empty candidate list"] - C --> F["Output Vec<ReplacementSubDAG>
each entry contains a constructed SummaryNode and rationale;
all candidates retained in preferred order"] + C --> F["Output Vec<ReplacementSubDAG>
each entry contains a constructed summary sub-DAG and rationale;
all candidates retained in preferred order"] ``` For an approximate quantile, both KLL and DDSketch remain candidates when @@ -271,16 +268,17 @@ even if the cost model prefers one. When only one realization is legal, such as Call the public strategy interface and inspect every returned candidate: ```rust -let strategy = SketchAlgorithmStrategy::new(&cost_model); +let strategy = ASAPStrategies::new(&cost_model); let candidates = strategy.replacements(&target); for candidate in candidates { match candidate.replacement { - Replacement::Summary(summary) => { - // Inspect or execute this constructed SummaryNode. + Replacement::SubDag(node) => { + // A constructed summary sub-DAG (`node.is_asap()`), or a kept + // pre-ASAP sub-DAG with an exact guarantee for pass-through. } - Replacement::Rewrite(_) => unreachable!( - "SketchAlgorithmStrategy produces summary candidates" + Replacement::ExactComposition(_) => unreachable!( + "ASAPStrategies produces sub-DAG candidates" ), } } @@ -293,9 +291,9 @@ aggregate choices remain independent. --- -### Example: current `SharedSubtreeStrategy` +### Example: current `SharedSubDagStrategy` -`SharedSubtreeStrategy` is the reference implementation for a logical rewrite strategy. +`SharedSubDagStrategy` is the reference implementation for a logical rewrite strategy. It applies when: @@ -310,16 +308,16 @@ and returns two alternatives: 2. Build independently for each consumer. ``` -The shared candidate reuses the same `Rc`: +The shared candidate reuses the same `Rc`: ```rust -Replacement::Rewrite(Rc::clone(target.root)) +Replacement::SubDag(Rc::clone(target.root)) ``` The independent candidate creates a structurally equal but separately allocated node: ```rust -Replacement::Rewrite( +Replacement::SubDag( Rc::new((**target.root).clone()) ) ``` @@ -331,9 +329,9 @@ That preference belongs to the cost model. share-versus-recompute candidate pair. The strategy still returns both alternatives because enumeration and ranking are separate steps: -- `consumer_count >= 2` means `share_common_subtrees` has already merged the expression into one shared `Rc`. The shared alternative is therefore an `Rc::clone`; the independent alternative requires a deep clone. +- `consumer_count >= 2` means `share_common_subdags` has already merged the expression into one shared `Rc`. The shared alternative is therefore an `Rc::clone`; the independent alternative requires a deep clone. - `cse_share_decision` is used by the ranking path, not by - `SharedSubtreeStrategy`. + `SharedSubDagStrategy`. - The strategy must return both valid alternatives even if the current cost model strongly prefers one. A future whole-plan search may choose differently from today's local comparison. This example is useful when implementing transformations such as: @@ -353,7 +351,7 @@ The basic calling pattern is: ```rust let target = TargetSubDAG::new(&root); let strategy = - SketchAlgorithmStrategy::default_cost_model(); + ASAPStrategies::default_cost_model(); if strategy.matches(&target) { let candidates = @@ -462,7 +460,7 @@ assert!( For a strategy whose explanation includes important context, also test that context. -For example, the shared-subtree tests verify that the consumer count appears in the rationale. +For example, the shared-sub-DAG tests verify that the consumer count appears in the rationale. --- @@ -470,7 +468,7 @@ For example, the shared-subtree tests verify that the consumer count appears in For logical rewrites, test the structural property that distinguishes the alternatives. -For example, the current shared-subtree tests verify: +For example, the current shared-sub-DAG tests verify: ```rust Rc::ptr_eq(shared, &q) @@ -546,13 +544,13 @@ Then inject it into code that accepts a `&dyn CostModel`: let model = PreferDDSketch; let strategy = - SketchAlgorithmStrategy::new(&model); + ASAPStrategies::new(&model); let replacements = strategy.replacements(&target); ``` -Important: changing `rank_candidates` changes the preferred ordering, but `SketchAlgorithmStrategy` still enumerates every valid sketch candidate. +Important: changing `rank_candidates` changes the preferred ordering, but `ASAPStrategies` still enumerates every valid sketch candidate. A custom cost model should not change which alternatives are semantically legal. @@ -641,26 +639,26 @@ Use it for implementation families that are intentionally outside the built-in e --- -#### `readout_extension` +#### `evaluation_extension` -Use when an extension-defined summary also needs custom query/readout behavior. +Use when an extension-defined summary also needs custom query/evaluation behavior. ```rust -fn readout_extension( +fn evaluation_extension( &self, ext_kind: &str, payload: &serde_json::Value, col: &ColumnRef, -) -> SketchQuery; +) -> SketchStatistic; ``` -This complements `realize_extension`: realization defines what gets maintained; readout defines how it is queried (see the [CostModel reference](asap-aware-mapping-contracts.md#costmodel)). +This complements `realize_extension`: realization defines what gets maintained; evaluation defines how it is queried (see the [CostModel reference](asap-aware-mapping-contracts.md#costmodel)). --- #### `cse_recompute_cost` -Use to estimate the cost of computing a common subtree independently at each consumer. +Use to estimate the cost of computing a common sub-DAG independently at each consumer. ```rust fn cse_recompute_cost( @@ -673,7 +671,7 @@ fn cse_recompute_cost( #### `cse_shared_maintenance_cost` -Use to estimate the cost of computing and maintaining a shared subtree. +Use to estimate the cost of computing and maintaining a shared sub-DAG. ```rust fn cse_shared_maintenance_cost( @@ -741,7 +739,7 @@ For example: ```rust let strategy = - SketchAlgorithmStrategy::new(&model); + ASAPStrategies::new(&model); let replacements = strategy.replacements(&target); @@ -770,7 +768,7 @@ Declare built-in sketch applicability through the public candidate registry: summary_candidates(intent) ``` -`SketchAlgorithmStrategy` consumes this registry through its public `replacements` method. +`ASAPStrategies` consumes this registry through its public `replacements` method. Therefore, when adding a new built-in sketch algorithm, the intended flow is: @@ -778,9 +776,9 @@ Therefore, when adding a new built-in sketch algorithm, the intended flow is: flowchart LR MAP["1. Declare legality
add the algorithm to summary_candidates
for each AggIntent it can answer"] MAP --> MODEL["2. Define costing
rank it, derive its SketchParams,
and provide a comparable numeric cost"] - MODEL --> BUILD["3. Define realization behavior
ensure the public strategy output contains a valid SummaryNode
with the correct maintained state and readout"] + MODEL --> BUILD["3. Define realization behavior
ensure the public strategy output contains a valid summary sub-DAG
with the correct maintained state and evaluation"] BUILD --> ACC["4. Certify accuracy
derive from committed parameters;
propagate and check the final target"] - ACC --> ENUM["5. Verify integration
SketchAlgorithmStrategy includes it automatically;
tests confirm enumeration, ordering, sizing, and cost"] + ACC --> ENUM["5. Verify integration
ASAPStrategies includes it automatically;
tests confirm enumeration, ordering, sizing, and cost"] ``` This keeps one source of truth for sketch applicability. Applicability alone @@ -793,9 +791,9 @@ ranking; preserve exact fallback and structured rejection information. See the [accuracy implementation companion](end-to-end-accuracy-guarantees.md) for formulas and evidence requirements. For a new algorithm, also update its -parameter, readout, schema and serialization definitions in `asap-types`. +parameter, evaluation, schema and serialization definitions in `asap-types`. -Do not special-case the new sketch inside `SketchAlgorithmStrategy` unless the strategy itself needs fundamentally new behavior. +Do not special-case the new sketch inside `ASAPStrategies` unless the strategy itself needs fundamentally new behavior. ### Verifying a new sketch algorithm @@ -804,7 +802,7 @@ or malformed evidence, incompatible metrics and unsupported composition. Test root-target checking before cost ranking, exact fallback, and exported rejection or guarantee data. A cheaper estimate must never admit an accuracy-illegal plan. -After wiring the new algorithm into `summary_candidates` and giving the cost model a real `rank_candidates`/`size_params` opinion about it, check two things. First, that `SketchAlgorithmStrategy::replacements()` for a matching `TargetSubDAG` actually includes a candidate realizing the new algorithm — extend a test shaped like `replacement.rs`'s own test-module coverage-matrix tests (e.g. `agg_intent_to_summary_kind_coverage_matrix`) to cover the new algorithm's `AggIntent`. Second, that `cost_sorted`/`estimate_cost` produce sane, comparable numbers for the new candidate rather than a `NaN` placeholder or an outlier that swamps every other candidate. +After wiring the new algorithm into `summary_candidates` and giving the cost model a real `rank_candidates`/`size_params` opinion about it, check two things. First, that `ASAPStrategies::replacements()` for a matching `TargetSubDAG` actually includes a candidate realizing the new algorithm — extend a test shaped like `replacement.rs`'s own test-module coverage-matrix tests (e.g. `agg_intent_to_summary_kind_coverage_matrix`) to cover the new algorithm's `AggIntent`. Second, that `cost_sorted`/`estimate_cost` produce sane, comparable numbers for the new candidate rather than a `NaN` placeholder or an outlier that swamps every other candidate. --- @@ -895,10 +893,11 @@ silently disagree. ### Mistake: reimplementing summary construction inside a strategy -If the candidate should produce a normal `SummaryNode`, use the existing +If the candidate should produce a normal summary sub-DAG (`SummaryAgg` / +`SummaryEstimate`), use the existing summary-construction path. -A strategy should steer or wrap that path when necessary, not recreate schema derivation, column resolution, readout construction, or parameter sizing. +A strategy should steer or wrap that path when necessary, not recreate schema derivation, column resolution, evaluation construction, or parameter sizing. --- @@ -922,7 +921,7 @@ Workload-wide target discovery, deduplication, and consumer counting are separat For CSE-style decisions, pointer identity can encode actual sharing. -Two `Rc` values can be structurally equal but deliberately represent independent computation. +Two `Rc` values can be structurally equal but deliberately represent independent computation. Use the distinction intentionally. @@ -937,8 +936,8 @@ When adding a new strategy: - [ ] Implement `ReplacementStrategy::replacements`. - [ ] Return every semantically valid replacement. - [ ] Return an empty vector for non-matching targets. -- [ ] Use `Replacement::Summary` for constructed post-ASAP output. -- [ ] Use `Replacement::Rewrite` for logical pre-ASAP alternatives. +- [ ] Return `Replacement::SubDag` for both constructed post-ASAP output and + logical pre-ASAP alternatives, with the matching `provenance`. - [ ] Add a useful rationale to every candidate. - [ ] Reuse existing legality and implementation logic instead of duplicating it. - [ ] Keep ranking and cost-based pruning out of the strategy. @@ -953,11 +952,11 @@ When adding a new cost model: - [ ] Keep semantic applicability outside the cost model. - [ ] Use `rank_candidates` for algorithm preference; return every input candidate exactly once. - [ ] Use `size_params` for accuracy-to-parameter mapping. -- [ ] Use extension hooks for extension-defined implementations/readouts. +- [ ] Use extension hooks for extension-defined implementations/evaluations. - [ ] Use CSE hooks for recompute-vs.-sharing costs. - [ ] Override `estimate_cost` if consumers require numeric costs instead of `NaN`. - [ ] Test the hook directly. -- [ ] Test integration through a consumer such as `SketchAlgorithmStrategy`. +- [ ] Test integration through a consumer such as `ASAPStrategies`. - [ ] Verify that changing cost preferences does not silently remove valid replacement candidates. --- @@ -975,12 +974,12 @@ Use this table to find the right place for a change. | Prefer one sketch algorithm over another | `CostModel::rank_candidates` | | Change sketch sizing for an accuracy target | `CostModel::size_params` | | Add extension-defined implementation behavior | `CostModel::realize_extension` | -| Add extension-defined readout behavior | `CostModel::readout_extension` | +| Add extension-defined evaluation behavior | `CostModel::evaluation_extension` | | Change CSE recomputation cost | `CostModel::cse_recompute_cost` | | Change shared-maintenance cost | `CostModel::cse_shared_maintenance_cost` | | Change current share/recompute choice | `CostModel::cse_share_decision` | | Decide whether an available implementation satisfies a required one | `impl Matcher` | -| Produce a normal (ranked-first) post-ASAP summary for one target | `SketchAlgorithmStrategy::replacements(...).into_iter().next()` | +| Produce a normal (ranked-first) post-ASAP summary for one target | `ASAPStrategies::replacements(...).into_iter().next()` | | Search a whole workload for supported legal candidates | `search_workload`/`search_workload_with` | | Enforce per-root result accuracy requirements | `search_workload_with_targets` | | Coordinate compatible choices across groups | `CandidateLogicalASAPDAGs::global_selection` | diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index b1d1226b0..05557a33e 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -38,14 +38,16 @@ asap-types = { git = "https://github.com/ProjectASAP/ASAPPlanner", rev = "e7fdb2 | Public function | Required input | Output | | --- | --- | --- | -| `asap_frontend_promql::lower_promql_workload` | PromQL `PlanningWorkload` with a nonzero `data_ingestion_interval` | All-or-nothing `Result, PromqlError>` for normalized batch and repeating entries | -| `asap_frontend_metricsql::lower_metricsql` | Query string, `AccuracyTarget` | `Result` | -| `asap_frontend_sql::lower_sql` | Query string, `SqlCatalog`, accuracy | Async `Result`; default SQL dialect is DataFusionSQL | +| `asap_frontend_promql::lower_promql_workload` | PromQL `PlanningWorkload` with a nonzero `data_ingestion_interval` | All-or-nothing `Result>, PromqlError>` for normalized batch and repeating entries | +| `asap_frontend_metricsql::lower_metricsql` | Query string, `AccuracyTarget` | `Result, MetricsqlError>` | +| `asap_frontend_sql::lower_sql` | Query string, `SqlCatalog`, accuracy | Async `Result, SqlError>`; default SQL dialect is DataFusionSQL | | `asap_frontend_sql::lower_sql_dialect` | Same inputs plus `SqlDialect` | Async resolved Pre-ASAP query or error | | `asap_frontend_sql::lower_sql_batch` | `QueryWorkload` and catalog | Per-query results for `query_batch`; does not iterate `repeating_queries` | Lowering resolves the supported source language into the canonical query -representation. It does not enumerate Post-ASAP alternatives. A frontend may +representation: an `asap_types::ir::OperatorNode` DAG containing only +`NonASAPOp` operators, with no timing (see the +[Pre-ASAP IR reference](pre-asap-ir.md)). It does not enumerate Post-ASAP alternatives. A frontend may reject unsupported syntax or semantics; a declared language/dialect enum does not imply complete support. PromQL workload lowering uses normalized `PlanningWorkload::query_workload.entries()` order, preserving entry-to-root associations for later @@ -58,7 +60,7 @@ PromQL's public signature (types are imported from their respective crates): ```text lower_promql_workload(workload: &PlanningWorkload, now_ms: u64) - -> Result, PromqlError> + -> Result>, PromqlError> ``` `DataWorkload.data_ingestion_interval` must contain a nonzero `Evidence`. @@ -125,9 +127,9 @@ For SQL, the corresponding signatures are: ```text async lower_sql(query: &str, catalog: &SqlCatalog, accuracy: AccuracyTarget) - -> Result + -> Result, SqlError> async lower_sql_dialect(query: &str, catalog: &SqlCatalog, - dialect: SqlDialect, accuracy: AccuracyTarget) -> Result + dialect: SqlDialect, accuracy: AccuracyTarget) -> Result, SqlError> ``` | `SqlDialect` value | Current behavior | @@ -162,7 +164,7 @@ It keeps the alternatives available; it does not select an entire workload plan. ```text search_workload_with_targets<'s, Id>( - roots: Vec<(Id, Rc, Option)>, + roots: Vec<(Id, Rc, Option)>, strategies: &[Box], accuracy_model: &dyn AccuracyModel, ) -> CandidateLogicalASAPDAGs @@ -197,7 +199,6 @@ accuracy target, and prints every ranked candidate instead of selecting a winner The default cost model is suitable for inspection, not deployment calibration. ```rust -use std::rc::Rc; use asap_frontend_promql::lower_promql_workload; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, Query, @@ -235,7 +236,7 @@ fn main() -> Result<(), Box> { ..Default::default() }), }; - let root = Rc::new(lower_promql_workload(&workload, 0)?.remove(0)); + let root = lower_promql_workload(&workload, 0)?.remove(0); let cost_model = DefaultCostModel; let strategies = default_strategies_with(&cost_model); let space = search_workload_with_targets( @@ -254,12 +255,12 @@ fn main() -> Result<(), Box> { | API (`asap_aware_mapping`, unless qualified) | Inputs | Output and limits | | --- | --- | --- | -| `search_workload` | `(query_id, Rc)` roots | `CandidateLogicalASAPDAGs` with built-in strategies/model; no explicit per-root target argument | +| `search_workload` | `(query_id, Rc)` roots | `CandidateLogicalASAPDAGs` with built-in strategies/model; no explicit per-root target argument | | `search_workload_with` | Roots, strategy slice | `CandidateLogicalASAPDAGs`; callers choose context-free replacement strategies | | `search_workload_with_targets` | Roots with optional end-to-end targets, strategies, accuracy model | Candidate space with supplied root-target checks; `None` does not supply a root-level requirement; uncertified direct DDSketch ratios remain available for backend selection | | `CandidateLogicalASAPDAGs::cost_sorted` | Cost model | `Vec`; retains alternatives and pairs `candidates[i]` with `costs[i]` | | `CandidateLogicalASAPDAGs::cost_sorted_with_recurrence` | Cost model, recurrence profiles, optional horizon | Ranked per-target candidate sets or `RecurrenceError`; uses recurrence for applicable share/recompute comparisons | -| `SketchAlgorithmStrategy::replacements` through `ReplacementStrategy` | One `TargetSubDAG` | Alternatives at that target; not whole-workload search | +| `ASAPStrategies::replacements` through `ReplacementStrategy` | One `TargetSubDAG` | Alternatives at that target; not whole-workload search | `cost_sorted` is a ranking view, not a request to discard all but the first candidate. Display costs follow model hooks and may be unavailable/non-finite; @@ -279,12 +280,12 @@ choices are not multiplied in. Exceeding `expansion_limit` is an error, never a partial inventory. For PromQL roots that carry a target, `search_workload_with_targets` also asks -each strategy's `ReplacementStrategy::propose_for_root`. `SketchAlgorithmStrategy` +each strategy's `ReplacementStrategy::propose_for_root`. `ASAPStrategies` answers an instant-vector TopK with current-series heap realizations over rows carrying the complete series identity (`$promql_series_identity`). They are finalized, deduplicated, and marked `ReplacementProvenance::RootPhysicalRealization`. Callers do not apply `with_series_identity` themselves. Compile each with -`promql_rows::compile_current_series_readout`; other queries keep their previous +`promql_rows::compile_current_series_evaluation`; other queries keep their previous inventory. `global_selection` never commits these candidates; the backend compiles and prices them. CandidateLogicalASAPDAGs lists no placement variants: node timing comes from the summary maintenance lifecycle. @@ -300,11 +301,11 @@ pass. An omitted strategy contributes no proposals of its own. | Value to put inside `Box::new(...)` | Meaning | In default factories? | | --- | --- | --- | -| `SketchAlgorithmStrategy::new(&model)` | Enumerates supported exact/sketch implementations and parameter choices for aggregate targets | Yes | +| `ASAPStrategies::new(&model)` | Enumerates supported exact/sketch implementations and parameter choices for aggregate targets | Yes | | `HydraGroupingStrategy::new(&model)` | Considers a shared multi-subpopulation structure for supported grouped sketch families, subject to accuracy evidence | Yes | -| `SharedSubtreeStrategy` | Proposes sharing versus independent recomputation at reused subtrees | Yes | +| `SharedSubDagStrategy` | Proposes sharing versus independent recomputation at reused sub-DAGs | Yes | | `SemanticEquivalentRewriteStrategy` | Proposes supported equivalent aggregate rewrites, including decomposing average into sum/count | Yes | -| `ExactCompositionStrategy::new(&model)` | Proposes supported exact operations around summary readouts or in maintenance | Yes | +| `ExactCompositionStrategy::new(&model)` | Proposes supported exact operations around summary evaluations or in maintenance | Yes | | Your `ReplacementStrategy` implementation | Adds domain-specific legal replacement proposals | No | `AvgToSumOverCountStrategy` is an alias for `SemanticEquivalentRewriteStrategy` @@ -345,7 +346,6 @@ replacement::default_strategies_with_evidence<'a>( ### Example: supply two strategies and run search ```rust -use std::rc::Rc; use asap_frontend_promql::lower_promql_workload; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, Query, @@ -353,7 +353,7 @@ use asap_types::workload::{ }; use asap_aware_mapping::{ search_workload_with_targets, DefaultAccuracyModel, DefaultCostModel, - ReplacementStrategy, SketchAlgorithmStrategy, SharedSubtreeStrategy, + ReplacementStrategy, ASAPStrategies, SharedSubDagStrategy, }; use asap_types::types::AccuracyTarget; @@ -383,11 +383,11 @@ fn main() -> Result<(), Box> { ..Default::default() }), }; - let root = Rc::new(lower_promql_workload(&workload, 0)?.remove(0)); + let root = lower_promql_workload(&workload, 0)?.remove(0); let model = DefaultCostModel; let strategies: Vec> = vec![ - Box::new(SketchAlgorithmStrategy::new(&model)), - Box::new(SharedSubtreeStrategy), + Box::new(ASAPStrategies::new(&model)), + Box::new(SharedSubDagStrategy), ]; let space = search_workload_with_targets( vec![("q1", root, Some(accuracy))], &strategies, &DefaultAccuracyModel, @@ -442,7 +442,7 @@ accuracy guarantees. ```rust use asap_aware_mapping::{ DefaultAccuracyModel, DefaultCostModel, EqualSplitAllocator, - NoAccuracyEvidence, ReplacementStrategy, SketchAlgorithmStrategy, + NoAccuracyEvidence, ReplacementStrategy, ASAPStrategies, }; fn main() { @@ -451,7 +451,7 @@ fn main() { let allocation = EqualSplitAllocator; let evidence = NoAccuracyEvidence; let strategies: Vec> = vec![Box::new( - SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ASAPStrategies::new_with_planning_inputs_and_evidence( &cost, &accuracy, &allocation, &evidence, ), )]; @@ -463,16 +463,16 @@ fn main() { Constructor definition: ```text -SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( +ASAPStrategies::new_with_planning_inputs_and_evidence( cost_model: &dyn CostModel, accuracy_model: &dyn AccuracyModel, allocator: &dyn AccuracyBudgetAllocator, evidence: &dyn AccuracyEvidenceProvider, -) -> SketchAlgorithmStrategy +) -> ASAPStrategies ``` All provider arguments are required for this constructor. They must outlive the -strategy vector. `SketchAlgorithmStrategy::new(&cost_model)` is the shorter +strategy vector. `ASAPStrategies::new(&cost_model)` is the shorter constructor using default accuracy/allocation and no extra evidence. | Extension point | What it controls | What it cannot establish alone | @@ -488,7 +488,7 @@ with the intended model/evidence; replacing only the final sorting model does no regenerate parameter choices. For evidence-aware defaults, use `asap_aware_mapping::replacement::default_strategies_with_evidence`. For custom accuracy/allocation/evidence on sketches, -`SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence` exposes these providers. +`ASAPStrategies::new_with_planning_inputs_and_evidence` exposes these providers. Keep each provider's evidence scope and freshness valid for the query population. ## Workload inputs and defaults @@ -525,7 +525,7 @@ Use this workflow when Planner owns summary-maintenance lifecycle decisions; otherwise the backend may make them from logical candidates. It includes both selection and DAG assembly, so callers do not first run the ordinary workflow. The first helper returns one `GlobalSelection`; the second is called per root -and returns a plan containing `root: Rc` plus maintenance decisions. +and returns a plan containing `root: Rc` (already timed) plus maintenance decisions. See the [workflow design](../design_docs/architecture/input-output-workflow.md#summary-maintenance-lifecycle-aware-helper). Two capabilities are distinct: the runtime can orchestrate a lifecycle, and the @@ -542,7 +542,7 @@ global_selection_with_summary_maintenance_lifecycles<'a, Id>( ) -> Result, SummaryMaintenanceLifecycleSelectionError> assemble_selected_dag_with_summary_maintenance_lifecycles( - selection: &GlobalSelection<'_>, target: &Rc, + selection: &GlobalSelection<'_>, target: &Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &dyn CostModel, ) -> Result, SummaryMaintenanceLifecycleAssemblyError> @@ -735,7 +735,7 @@ workflow for those decisions. Downstream still owns physical commitment. | --- | --- | | `CandidateLogicalASAPDAGs::global_selection(&model)` | Compatible structural selection across targets; no recurrence or lifecycle planning implied | | `CandidateLogicalASAPDAGs::global_selection_with_recurrence(...)` | Compatible selection using supplied recurrence profiles/horizon; no lifecycle commitments implied | -| `GlobalSelection::assemble_selected_dag(&target)` | `Result>, RealizationError>`; constructs semantic IR, not stored summary data | +| `GlobalSelection::assemble_selected_dag(&target)` | `Result>, RealizationError>`; constructs untimed semantic IR, not stored summary data | Use a target associated with the searched space; DAG assembly can return `None` when that target is absent. A downstream integration can use these convenience @@ -747,8 +747,8 @@ for checking complete physical alternatives and deployment constraints. ```text CandidateLogicalASAPDAGs::global_selection(&self, cost_model: &dyn CostModel) -> GlobalSelection<'_> -GlobalSelection::assemble_selected_dag(&self, target: &Rc) - -> Result>, RealizationError> +GlobalSelection::assemble_selected_dag(&self, target: &Rc) + -> Result>, RealizationError> ``` For structural inspection only, this complete example selects a semantic root @@ -756,7 +756,6 @@ and exports its inspection graph. It performs no lifecycle or deployment plannin Use lifecycle-aware selection above when the comparison needs those decisions. ```rust -use std::rc::Rc; use asap_frontend_promql::lower_promql_workload; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, Query, @@ -790,7 +789,7 @@ fn main() -> Result<(), Box> { ..Default::default() }), }; - let root = Rc::new(lower_promql_workload(&workload, 0)?.remove(0)); + let root = lower_promql_workload(&workload, 0)?.remove(0); let space = search_workload(vec![("q1", root)]); let selection = space.global_selection(&DefaultCostModel); // Search may canonicalize roots; use the root returned by CandidateLogicalASAPDAGs. @@ -808,7 +807,8 @@ fn main() -> Result<(), Box> { | --- | --- | | `asap_types::dag_export::export(&query)` | Pre-ASAP inspection graph | | `asap_types::dag_export::export_summary(&summary)` | Post-ASAP inspection graph | -| `asap_types::post_asap::compile_post_asap_dag(&root)` | Compile a semantic DAG with execution-data-state validation; not a physical plan | +| `asap_types::ir::apply_lifecycle_timings(&root, &assignment, &mut TimingMemo::new())` | Write execution timing into every node from a `LifecycleAssignment` and validate the data-state edges; a lifecycle plan's `root` is already timed | +| `asap_types::ir::export::compile_post_asap_dag(&timed_root)` | Export a timed DAG as a `PostAsapDag` (wire version 7); rejects an untimed node; not a physical plan | | `PostAsapDagDocument::new(dag)` and `.validate()` | Versioned semantic envelope and explicit validation; constructing it alone does not validate | | `asap_aware_mapping::export_summary_maintenance_plan(&plan)` | Graph plus lifecycle deployments, alternatives and available cost/guarantee information | | `explain_replacements` / `explain_replacements_with` | Findings from default/custom-strategy search; not a complete physical feasibility report | diff --git a/docs/develop_docs/metrics-observability-corpora.md b/docs/develop_docs/metrics-observability-corpora.md index 3dd1d2b8f..44c70a1eb 100644 --- a/docs/develop_docs/metrics-observability-corpora.md +++ b/docs/develop_docs/metrics-observability-corpora.md @@ -49,20 +49,20 @@ o11y-bench, and awesome-prometheus-alerts. They are not duplicated here. The test prints totals, parse errors, lowering errors, pre-ASAP successes, post-ASAP candidates, unchanged queries, and post-ASAP errors. `Pre-ASAP` means -that parsing and lowering produced a `QueryExpr`. `Post-ASAP candidate` means -the isolated `SketchAlgorithmStrategy` produced a non-`KeepPreAsap` summary -candidate. `Unchanged` is a successful pre-ASAP query for which that strategy -returned only the pre-ASAP fallback. +that parsing and lowering produced an `OperatorNode` DAG. `Post-ASAP candidate` +means the isolated `ASAPStrategies` produced a candidate that contains +an ASAP operator (`contains_asap()`). `Unchanged` is a successful pre-ASAP query +for which that strategy returned only the kept pre-ASAP sub-DAG (`retain_exact`). ## Strategies The corpus measurement deliberately uses only -`SketchAlgorithmStrategy::default_cost_model().replacements(...)` on each +`ASAPStrategies::default_cost_model().replacements(...)` on each query root. It does not measure workload-wide search or the other default strategies. -The default workload search currently registers `SketchAlgorithmStrategy`, -`HydraGroupingStrategy`, `SharedSubtreeStrategy`, and +The default workload search currently registers `ASAPStrategies`, +`HydraGroupingStrategy`, `SharedSubDagStrategy`, and `AvgToSumOverCountStrategy`. Workload context can additionally contribute `RollupStrategy` and `AccuracyReconciliationStrategy`. This baseline is therefore a sketch-only comparison point. diff --git a/docs/develop_docs/native-promql-inputs.md b/docs/develop_docs/native-promql-inputs.md index aa1a52b63..f5a260f82 100644 --- a/docs/develop_docs/native-promql-inputs.md +++ b/docs/develop_docs/native-promql-inputs.md @@ -23,7 +23,7 @@ operator's metric-name/result-label rules. Source selection, complete window coverage and revision admission remain deployment responsibilities. Planner's maintained-population candidate recognizes this explicit identity -representation. Its TopK readout compiles automatically to `CurrentSeries`, +representation. Its TopK evaluation compiles automatically to `CurrentSeries`, `Sort`, and `Limit`; deployment supplies the raw boundary or an already maintained population boundary. Compilation does not open either source. diff --git a/docs/develop_docs/offline-sketch-evidence.md b/docs/develop_docs/offline-sketch-evidence.md index cd331e57d..56c507de1 100644 --- a/docs/develop_docs/offline-sketch-evidence.md +++ b/docs/develop_docs/offline-sketch-evidence.md @@ -91,7 +91,7 @@ hooks. The provider's lifecycle helper returns available build/update CPU costs for a single independently instantiated state. It deliberately leaves retention, retirement and read costs unknown. In particular, a point-frequency benchmark read does not price a total-count read, even when both use CMS. A deployment must -match readout semantics and supply the missing lifecycle and raw-query evidence +match evaluation semantics and supply the missing lifecycle and raw-query evidence before selecting and pricing a complete physical plan. Never combine these nanosecond costs with CPU operation counts without explicit calibration. @@ -116,7 +116,7 @@ not be passed as these disjoint phase measurements. `MeasurementQueryBinding` is the producer's explicit assertion identifying the read/error probe population. The consumer checks that binding and the error -record's readout kind/value type; it cannot recover or certify the original +record's evaluation kind/value type; it cannot recover or certify the original probe set from an aggregate error number alone. The supported workload is an immutable i64 point-frequency snapshot, fully @@ -131,7 +131,7 @@ post-merge error and an exact merge baseline exist. The caller supplies an `EmpiricalAccuracyRequirement`: the exact observed error metric, maximum accepted mean, and minimum number of offline trials. This is -separate from `AccuracyTarget`. Every candidate must match the readout descriptor, +separate from `AccuracyTarget`. Every candidate must match the evaluation descriptor, error metric, trial count and all ordinary distribution/configuration/environment checks. A zero observed error is neither proof of exactness nor a per-key bound. diff --git a/docs/develop_docs/operator-design-acceptance.md b/docs/develop_docs/operator-design-acceptance.md new file mode 100644 index 000000000..2596bb94c --- /dev/null +++ b/docs/develop_docs/operator-design-acceptance.md @@ -0,0 +1,121 @@ +# Operator/scalar design acceptance for #528 + +Audience: maintainers reviewing the implementation of [#511](https://github.com/ProjectASAP/ASAPPlanner/pull/511). +Reviewed against the current proposal and upstream documentation on 2026-10-02. +This is a representation and semantic-contract review, not a claim that the native +executor implements every SQL or PromQL feature. + +## Unified graph and migration + +`LogicalDAG` and `LogicalASAPDAG` name planning stages, not different Rust graphs. +Both use `Rc` with `Operator::NonASAP(NonASAPOp)` or +`Operator::ASAP(ASAPOp)`, the same `Schema`/`Field`/`FieldDataType`, and shared +operator edges, including producers referenced by scalar expressions. +`QueryRoot::Scalar` owns a scalar expression without fabricating an operator. +There is no `KeepPreAsap` or `ScalarBridge` operator. `retain_exact` annotates an +unchanged ordinary sub-DAG; it does not wrap it. The viewer uses ordinary node +kinds in both stages. Historical `PostAsapDag` names denote the flat executable +wire document (version 7), not another logical IR. + +`validate_structure` permits unassigned timing, checks scalar scopes, predicates, +result kinds, leaf declarations, state families, and retained field types. +`validate_execution_timing` additionally checks assigned phases and dependencies; +it does not decide whether a backend supports an implementation. A regression +allows pointwise arithmetic at ingestion time. Existing lifecycle/deployment +policy remains separate work in [#520](https://github.com/ProjectASAP/ASAPPlanner/issues/520) +and [#530](https://github.com/ProjectASAP/ASAPPlanner/issues/530). + +`PlanOutput` is one multi-root workload DAG: `operator_roots()` exposes operator +roots, and `operators()` inventories shared nodes once across operator and scalar +roots. Replacement regions and CSE use `SubDag` and `share_common_subdags`. +Bulk retained-sub-DAG cost evidence may cover only ordinary operators; it is +rejected if any descendant is an ASAP operator, so summary work cannot be hidden. +Supporting operator parameters live in `ir::operator_properties`; schema derivation +and errors have dedicated modules: + +| Module | Responsibility | Example | +|---|---|---| +| `ir::operator_properties` | Parameter types stored in operator payloads, rather than derived node metadata | `GroupKeys` for aggregation, `JoinKind` for joins, `WindowFrame` for SQL windows | +| `ir::aggregate_schema` | Compute output columns and types from input schema and aggregate reduction | Preserve grouping columns and derive the aggregate result column | +| `ir::error` | `SchemaDerivationError` from schema/type derivation; other validation errors remain separate | Invalid grouping-column index or scalar-function signature | + +Summary operations use “evaluation”; +`SketchStatistic` specifies the statistic to compute, rather than another query. +Wire version 7 reflects these renamed serialized variants and fields. Regenerate +older exported graphs and native programs; no legacy-name aliases are provided. + +`ASAPStrategies` proposes supported ASAP realizations, including exact +accumulators and sketches; its name does not restrict candidates to sketches. +The API, diagnostics, caller imports, and current documentation use this name. + +## Document examples + +| Example | Evidence | +|---|---| +| Batch `SUM(bytes) + 1`, `SUM(bytes) * 2` | `batch_planning_replaces_and_shares_summary_operators`: invokes the actual planner, selects one shared summary state across two roots, validates and natively executes both results (31 and 60 for inputs 10 and 20). | +| `SELECT l_quantity * 2 AS q2 FROM lineitem WHERE l_quantity > 10` | `integration-tests/tests/operator_design_examples.rs`: parse, resolve, validate and flat export; Int64 projection and Boolean predicate. | +| `SELECT SUM(bytes) + 1 AS total_bytes FROM requests WHERE status = 200` | Same suite: explicit summary build/finalize rewrite, identical nullable Int64 result schema, one exported node per operator. Native execution of the ordinary SQL plan covers filtered rows, empty input and all-NULL input. The logical rewrite does not imply native Int64 summary-kernel support. | +| `2`, `time()`, `up * 2`, `vector(time())`, `scalar(sum(up)) + 1` | `asap-physical-operators/tests/promql_fallback.rs::scalar_design_document_examples_execute`: parse through native compilation and execution with hand-computed values. `frontend-promql/tests/scalar_design.rs` checks scalar roots, expression ownership and producer sharing. | +| Unary/math expressions and nonconstant scalar parameters | Native `pointwise_projection_names_and_dynamic_parameters`: unary name retention, arithmetic/math name removal, `round(m, scalar(vector(2)))`, `clamp(m, time()-301, time())`, reversed clamp bounds and calendar functions. | + +## Comparison-table review + +The following groups cover the rows in the proposal's SQL and PromQL comparison +tables. “Represented” describes the typed IR, not blanket parser/runtime coverage. + +| Table rows | Review result and regression evidence | +|---|---| +| SQL scan/filter/project, VALUES, constants, Boolean/NULL/CASE/casts | Represented. Predicates require Boolean types, columns must resolve, and Values rows match declared arity/nullability. DataFusion coercions are retained explicitly. `types/tests/structure_contract.rs`, `frontend-sql/tests/sql_lowering.rs`. | +| SQL unary minus, BETWEEN, IS TRUE, null-safe comparisons, scalar functions/NOW | Explicit scalar expressions; only registered function contracts receive types. Unknown functions no longer receive a placeholder Float64. NOW remains a timestamp context read. Existing SQL lowering tests cover these forms. | +| SQL joins and set/bag operations | Relation inputs and concatenated predicate scopes checked; existing SQL lowering tests cover outer null extension, semi/anti joins and ALL flags. Nullable NOT IN is retained explicitly, not rewritten to ordinary anti-join. | +| SQL grouping, computed aggregate arguments, COUNT(x), HAVING | Scalar input projection and per-measure filters are retained. Global and filtered SUM/AVG are nullable; SUM preserves integer type. Window SUM preserves its argument type, and MIN/MAX window outputs are nullable because frames may be empty. Existing count-filter/grouping tests and the SQL document-example suite cover these contracts. | +| SQL QUALIFY/DISTINCT ON, grouping sets, wildcard/name alignment, window functions, sorting/limits | Existing frontend lowerings and payload tests remain. ROW_NUMBER stays a window column followed by its filter; the old TopK rewrite discarded that column and broke outer positional references. ORDER BY/LIMIT promotion remains only for its existing additive-ranking contract. Parser syntax coverage remains that of the installed DataFusion version. | +| SQL scalar/EXISTS/IN subqueries, derived tables/CTEs | Scalar plan references participate in traversal/canonicalization/export. Scalar subqueries retain zero-row NULL and multi-row error requirements. Positive EXISTS/IN filter conjuncts may still use proven semi joins. Select-list subquery execution is not added to the native backend by this PR. | +| PromQL literals/time/scalar arithmetic/comparisons | Scalar roots and nested expressions; bool comparisons return numeric 0/1. No scalar operator result kind. Scalar conversions validate producer kinds. Scalar and native fallback tests. | +| PromQL selectors/ranges, vector arithmetic/comparison/set matching | Typed temporal and binary operators preserve lookback, staleness, matching, bool mode and operand order. Existing conformance, numeric, label-matching and native fallback suites remain. Mixed scalar/vector operations use Project/Filter and visible scalar dependencies. | +| PromQL grouping/ranking/range reducers | Existing intent, grouping and partitioned Sort/Limit contracts remain; exact named range functions are not replaced by ordinary SQL SUM. Constant-parameter limitation is explicit. | +| PromQL vector/scalar conversions and nested subqueries | Real kind/cardinality conversions retained. Nested grid/offset/anchor tests remain. A SQL relation cannot silently enter PromqlSubquery without a vector conversion. | +| PromQL math/date functions | Scalar FunctionCall within Project, including expression parameters. Unary Negative retains metric name. `timestamp(v)` remains a temporal sample-timestamp operation, unlike calendar functions over sample values. | +| Request-duration expressions; relabel/absence/info/sampling | Existing registered/parser-supported forms remain; request-context syntax unavailable in the installed parser is a frontend gap, not an invented scalar contract. Existing partial-support and rejection tests remain. | + +## Explicit gaps and compatibility changes + +The proposal's gap rows remain gaps: unsupported aggregate/window modifiers, +correlation/recursion, lambdas, general ANY/ALL, unbound parameters, unrepresented +value types, explicit pattern escapes, general table functions, dynamic aggregate +parameters, fill modifiers, and extended range/start-timestamp metadata. Rejection +may occur in the parser or the frontend; this PR does not upgrade either language +parser to the current upstream release. + +Native histogram samples have no IR sample type. Declared native samples and +native-histogram functions are rejected rather than treated as ordinary floats. +Classic bucket interpolation remains supported. Generic histogram-quantile sketches +require the explicitly declared, nonstandard `HistogramKind::RawSamples` extension. +`histogram_metadata`, `promql_conformance` and `promql_lowering` test both paths. +The corpus now records 1121 lowered / 469 rejected / 233 parser gaps; the previous +floor counted unsupported histogram operations as successful float lowerings. + +Registered ClickHouse parser stubs are not automatically typed scalar contracts. +The 200-query January corpus now records 105 lowered, 53 invalid representations, +41 planning errors and 1 unsupported feature. This is an intentional compatibility +change: unsupported signatures and invalid scopes fail instead of receiving dummy +types. The corpus remains fully exercised. Exact SUM state now records whether any +non-NULL value was observed; persisted older SUM payloads lacking that field must +be rebuilt (decoding fails closed). + +## Upstream references and validation + +Semantic references: [Prometheus operators](https://prometheus.io/docs/prometheus/latest/querying/operators/), +[Prometheus functions](https://prometheus.io/docs/prometheus/latest/querying/functions/), +[DataFusion scalar functions](https://datafusion.apache.org/user-guide/sql/scalar_functions.html), +and [DataFusion subqueries](https://datafusion.apache.org/user-guide/sql/subqueries.html). +The comparison is against these current contracts; dependencies remain pinned by +Cargo.lock. Tests use explicit expected results, not live upstream differential +execution. Review and implementation were performed by the same agent. + +Validation commands: workspace tests, fmt, clippy with warnings denied, the external +MetricsQL consumer, and DAG viewer Python tests (24 passed; 6 requiring Node.js +skipped because Node.js is unavailable in this environment). The vendored MetricsQL baseline also passes on Rust 1.99 (the CI toolchain), +verifying its existing 21 library and 3 doctest failures. Rust 1.98 changes one +compiler-diagnostic fingerprint; no baseline hashes or vendored sources were +changed to accommodate that older toolchain. Formatting and clippy pass on 1.99. diff --git a/docs/develop_docs/physical-compile-coverage.md b/docs/develop_docs/physical-compile-coverage.md index 3fa54fb4e..99f2bff40 100644 --- a/docs/develop_docs/physical-compile-coverage.md +++ b/docs/develop_docs/physical-compile-coverage.md @@ -8,7 +8,7 @@ Audience: developers moving computation from ASAPQuery-backend into Logical selection decides what to compute. The maintenance lifecycle sets node timing. `physical_planner::compile` turns a timed `PostAsapDag` into physical operator DAGs. The backend owns ingestion, panes, storage, stored-state -readout, external exact engines, pricing/selection, and execution scheduling. +evaluation, external exact engines, pricing/selection, and execution scheduling. A backend lowering is *covered* when `compile` accepts the corresponding `PostAsapDag` node and produces operators with the same result. The backend @@ -29,7 +29,7 @@ Status values: | # | Backend site | Computation | Planner node | Status at #475 | Notes | |---|---|---|---|---|---| -| 1 | `query_time.rs` `Lower::lower`, `compile_logical` | PromQL AST → `QueryTimeOperator` graph for a native query | `Fallback { QueryExpr }` subtrees plus value payloads | Missing | `compile` lowers `Fallback` only as a raw `Scan` source. | +| 1 | `query_time.rs` `Lower::lower`, `compile_logical` | PromQL AST → `QueryTimeOperator` graph for a native query | `Fallback { QueryExpr }` sub-DAGs plus value payloads | Missing | `compile` lowers `Fallback` only as a raw `Scan` source. | | 2 | `QueryTimeOperator::Aggregate` (sum/min/max/avg/count) | Grouped value aggregation | `Value::Exact(Aggregate)`; `SummaryAgg{ExactAggregate, Reduce}` over finalized values | Supported | Also `promql_values::compile_aggregate`. | | 3 | `QueryTimeOperator::Sort`, `Limit` (topk, sort, sort_desc) | Ordering and per-group limits | `Value::Sort`, `Value::Limit` | Supported | | | 4 | `QueryTimeOperator::Binary`, `QueryPlanNode::Binary` (vector ⊗ scalar) | Arithmetic with a scalar operand | `Binary` whose operand is `Fallback{PromqlScalarBridge(Literal)}` | Missing | Query-time `Binary` accepts only label-map vector schemas. The literal node has no native binding. | @@ -43,15 +43,15 @@ Status values: | 12 | `logical_dag.rs` `Subquery`, `subquery_grid`, `expanded_inputs` | Re-evaluate the child on a step grid and assemble a matrix | `Fallback{PromqlSubquery}` | Missing | No Planner operator. | | 13 | `QueryPlanNode::Scalar`, `DagCompiler::lower` scalar literal | Scalar constant | `Fallback{PromqlScalarBridge(Literal)}` | Missing | Only `promql_values::compile_scalar`. | | 14 | `DagCompiler::lower` `ReduceSum`; `physical_values.rs` PerEntity projection | Sum over finalized values; per-entity identity | `SummaryAgg{ExactAggregate(Sum)}` | Supported | The backend builds an identity `Operator::project` itself for PerEntity. | -| 15 | `DagCompiler::lower` `ExactReadout`; `post_asap_readout.rs` ExactReadout | Finalize exact state (sum/count/min/max/rate/increase) | `Value::FinalizeExactAccumulator` | Partial | Count yields Int64 against a declared Float64 PromQL value. `compile` rejects it. | -| 16 | `post_asap_readout.rs` SummaryEstimate (`readout_bound`, `expand_item_rows`) | Sketch estimate per group; TopK item expansion | `SummaryEstimate` | Partial | The backend's label-map state layout and MetricsQL `__name__` rules have no Planner equivalent. `compile_exact_readout` has no sketch counterpart. | -| 17 | `post_asap_readout.rs` SummaryMerge (`merge_bound_states`) | Merge states by group | `SummaryMerge` | Supported | Union plus `summary_merge`. | -| 18 | `post_asap_readout.rs` counter range parameters | Counter lookback for rate/increase | `TimeRange` ancestor of finalization | Supported | Applied through `with_counter_lookback`. | -| 19 | `post_asap_readout.rs` `execute_value_fragment` | Per-timestamp binding of a value fragment | n/a | Backend | Evaluation scheduling. | +| 15 | `DagCompiler::lower` `ExactEvaluation`; `post_asap_evaluation.rs` ExactEvaluation | Finalize exact state (sum/count/min/max/rate/increase) | `Value::FinalizeExactAccumulator` | Partial | Count yields Int64 against a declared Float64 PromQL value. `compile` rejects it. | +| 16 | `post_asap_evaluation.rs` SummaryEstimate (`evaluation_bound`, `expand_item_rows`) | Sketch estimate per group; TopK item expansion | `SummaryEstimate` | Partial | The backend's label-map state layout and MetricsQL `__name__` rules have no Planner equivalent. `compile_exact_evaluation` has no sketch counterpart. | +| 17 | `post_asap_evaluation.rs` SummaryMerge (`merge_bound_states`) | Merge states by group | `SummaryMerge` | Supported | Union plus `summary_merge`. | +| 18 | `post_asap_evaluation.rs` counter range parameters | Counter lookback for rate/increase | `TimeRange` ancestor of finalization | Supported | Applied through `with_counter_lookback`. | +| 19 | `post_asap_evaluation.rs` `execute_value_fragment` | Per-timestamp binding of a value fragment | n/a | Backend | Evaluation scheduling. | | 20 | `DagCompiler::lower` SummaryJoin / Subtract / Delete | Summary algebra | `SummaryJoin`, `SummarySubtract`, `SummaryDelete` | Missing | The backend also rejects these (`ExactFallback`). | -| 21 | `current_series.rs` Snapshot + TopK | Current-series ranking | `ReadPopulation{TopK}` | Supported | | -| 22 | `current_series.rs` Sum / Count / Average | Current-series aggregates | `ReadPopulation{Sum,Count,Average}` | Missing | `compile` accepts only TopK. | -| 23 | `current_series.rs` Quantile | Current-series quantile | `ReadPopulation{Quantile}` | Missing | No exact quantile reduction. | +| 21 | `current_series.rs` Snapshot + TopK | Current-series ranking | `EvaluatePopulation{TopK}` | Supported | | +| 22 | `current_series.rs` Sum / Count / Average | Current-series aggregates | `EvaluatePopulation{Sum,Count,Average}` | Missing | `compile` accepts only TopK. | +| 23 | `current_series.rs` Quantile | Current-series quantile | `EvaluatePopulation{Quantile}` | Missing | No exact quantile reduction. | | 24 | `raw_dag.rs` weight `Column` | Summary update from a sample/projected value | `SummaryAgg` | Supported | | | 25 | `raw_dag.rs` weight `Constant` | Unit/constant-weight update | `SummaryAgg` | Missing | `compile_node` requires a column weight. | | 26 | `raw_dag.rs` item `Column` / `Tuple` | Keyed update item | `SummaryAgg{item}` | Supported | `keyed_summary_build`. | @@ -70,7 +70,7 @@ Totals at #475: 11 Supported, 4 Partial, 14 Missing, 2 Backend. | 4, 8, 13 | Query-time `Binary` folds a scalar-literal operand into a projection over grouped value rows. | | 5 | Query-time `Binary` over grouped value rows performs an inner equi-join on equal label columns, then applies the operator. Per-series rows remain Partial. | | 15 | Count finalization converts exactly to the declared Float64 value. | -| 22, 23 | `ReadPopulation` Sum/Count/Average/Quantile compile to grouped aggregation. `Reduction::Quantile` implements PromQL interpolation. | +| 22, 23 | `EvaluatePopulation` Sum/Count/Average/Quantile compile to grouped aggregation. `Reduction::Quantile` implements PromQL interpolation. | Totals after this change: 17 Supported, 4 Partial, 8 Missing, 2 Backend. @@ -124,10 +124,10 @@ Totals are unchanged: 19 Supported, 5 Partial, 5 Missing, 2 Backend. | Row | Change | |---|---| -| 5 | Query-time `Binary` over rows with a series identity, such as per-series readouts of stored state, uses the Fallback's `series_labels` and `series_binary`. Examples: `avg_over_time` as stored sum/count, and `rate(a) / rate(b)`. Matching drops `__name__` and honors `on`/`ignoring` when the payload carries them. Only one-to-one arithmetic is covered; `group_left`/`group_right` stay rejected and comparisons are row 7. Now Supported. | +| 5 | Query-time `Binary` over rows with a series identity, such as per-series evaluations of stored state, uses the Fallback's `series_labels` and `series_binary`. Examples: `avg_over_time` as stored sum/count, and `rate(a) / rate(b)`. Matching drops `__name__` and honors `on`/`ignoring` when the payload carries them. Only one-to-one arithmetic is covered; `group_left`/`group_right` stay rejected and comparisons are row 7. Now Supported. | | 4, 8 | A literal operand also applies to per-series rows and drops `__name__`, in the Fallback too. Series whose label sets become equal are an error, as in Prometheus. | -Grouped `sum`/`avg`, current-series `Sum`/`Average` readouts, and +Grouped `sum`/`avg`, current-series `Sum`/`Average` evaluations, and `sum_over_time`/`avg_over_time` use Prometheus' Kahan-Neumaier summation. An average switches to an incremental mean once the running sum would overflow. The grouped path also serves SQL `SUM`/`AVG` over Float64, which are now @@ -175,7 +175,7 @@ and `group_left`, including a right-side series identity when needed. Thus |---|---| | 1 | Comparisons, `bool`, set operators, `group_left`/`group_right`, `scalar()` operands, and literals over aggregates whose value has another name, such as `sum by (job) (a) * 2`. Still Partial. | | 5 | Grouped `Binary` rows use the same operator instead of a relational join. A duplicate match group is now an error instead of a cross product. | -| 7 | Fallback, grouped `Binary`, and per-series comparisons and sets. Temporal stored readouts drop `__name__` before matching, including exact Count conversion, and reject duplicate output identities. | +| 7 | Fallback, grouped `Binary`, and per-series comparisons and sets. Temporal stored evaluations drop `__name__` before matching, including exact Count conversion, and reject duplicate output identities. | Totals after this change: 20 Supported, 5 Partial, 4 Missing, 2 Backend. @@ -188,7 +188,7 @@ Totals after this change: 20 Supported, 5 Partial, 4 Missing, 2 Backend. An argument whose output provably lacks `le`, such as `sum by (job) (rate(x_bucket[5m]))`, is rejected at lowering. Prometheus returns an empty vector for it. Candidate search keeps the classic form as one -exact `KeepPreAsap` subtree for every accuracy target; it has no sketch +retained exact ordinary sub-DAG for every accuracy target; it has no sketch candidate. `histogram_quantiles` lowers each branch the same way; the Fallback compiler accepts its `Concat` of relabeled branches and rejects duplicate output label sets. Nested aggregation, such as @@ -208,8 +208,8 @@ In order of backend usage: After these shapes are covered, the backend can delete rows 28 and 30. 2. Rows 25 and 27: constant weights and `EntityIdentity` items for precompute `SummaryAgg`. -3. Row 16: a label-map sketch-state readout, the counterpart of - `compile_exact_readout`, and MetricsQL `__name__` retention rules. +3. Row 16: a label-map sketch-state evaluation, the counterpart of + `compile_exact_evaluation`, and MetricsQL `__name__` retention rules. 4. Row 20: summary join, subtract, and delete. `fill`, `fill_left`, and `fill_right` matching modifiers are rejected by the diff --git a/docs/develop_docs/planner-vocabulary-migration.md b/docs/develop_docs/planner-vocabulary-migration.md index f5f7f70ae..f0ad2e5cc 100644 --- a/docs/develop_docs/planner-vocabulary-migration.md +++ b/docs/develop_docs/planner-vocabulary-migration.md @@ -32,8 +32,8 @@ names. | Physical evidence/comparison `boundaries` fields | `handoffs` | | `BoundaryEstimate::per_boundary` | `PhysicalHandoffEstimate::per_handoff` | | Internal `Models` | `CandidatePlanningInputs` | -| `SketchAlgorithmStrategy::with_models` | `SketchAlgorithmStrategy::new_with_planning_inputs` | -| `SketchAlgorithmStrategy::with_models_and_evidence` | `SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence` | +| `ASAPStrategies::with_models` | `ASAPStrategies::new_with_planning_inputs` | +| `ASAPStrategies::with_models_and_evidence` | `ASAPStrategies::new_with_planning_inputs_and_evidence` | | `HydraGroupingStrategy::with_models_and_evidence` | `HydraGroupingStrategy::new_with_planning_inputs_and_evidence` | For example, `Binder::new().bind(&tree)` becomes diff --git a/docs/develop_docs/pre-asap-ir.md b/docs/develop_docs/pre-asap-ir.md index 749f97cf9..16901760e 100644 --- a/docs/develop_docs/pre-asap-ir.md +++ b/docs/develop_docs/pre-asap-ir.md @@ -2,7 +2,15 @@ This is the detailed node reference. Start with the [Pre-ASAP IR concept](../design_docs/concepts/pre-asap-ir.md) for purpose and the compact catalog. -The goal of the pre-ASAP IR is represent operations from different query languages in a single representation, and make it easier to analyze how/where ASAP primitives can be used. +ASAPPlanner has **one operator IR before and after ASAP optimization**, defined in +`crates/types/src/ir/`. "Pre-ASAP" is not a separate type: it is this IR as a front end +emits it, before any ASAP operator has been introduced. This document covers what every +plan shares — the node, the schema, scalar expressions, how front ends produce the DAG, and +the catalog of ordinary (`NonASAPOp`) operators. The ASAP operators, execution timing and +the exported wire form are described in the [Post-ASAP IR](../design_docs/concepts/post-asap-ir.md) +document; the two do not repeat each other. + +The goal of the pre-ASAP form is to represent operations from different query languages in a single representation, and make it easier to analyze how/where ASAP primitives can be used. Only operations that are semantically relevant to answering the query and selecting an ASAP primitive need to become first-class nodes here. ## Design principles @@ -13,7 +21,135 @@ Only operations that are semantically relevant to answering the query and select > Notes: **SQL and PromQL use different schema models**. SQL typically uses a closed schema, where tables, columns, and types are predefined, while PromQL uses an open (schemaless) schema, where metrics and labels can evolve without a fixed table schema. Closed schemas provide stronger structure and validation; open schemas provide greater flexibility and makes it easier to evolve or ingest diverse data, but can require more care around naming conventions, label cardinality, and query consistency. -The pre-ASAP IR is defined using the `QueryExpr` enum. We discuss some of important enum types below. +## The node + +A plan is a DAG of `Rc` (`crates/types/src/ir/node.rs`). Nodes are immutable +and shared through `Rc`: a structurally identical sub-DAG referenced from several parents is +one node, and that pointer identity is what CSE, target discovery and plan assembly key on. + +```rust +pub struct OperatorNode { + pub operator: Operator, // NonASAP(NonASAPOp) | ASAP(ASAPOp) + pub result_kind: OperatorResultKind, // Relation | InstantVector | RangeVector | State | Scalar + pub schema: Schema, // output schema, derived at construction + pub guarantee: Option, // None until accuracy assessment establishes one + pub timing: Option, // None until a lifecycle assignment is applied +} +``` + +- `operator` is the operation. A front-end DAG contains only `Operator::NonASAP` nodes; + `OperatorNode::expect_non_asap()` relies on that. +- `result_kind` is the output category, derived from the operator and its inputs. Matching + column schemas do not make categories interchangeable (a range vector is not an instant + vector). +- `schema` is derived by `OperatorNode::new(operator)`; it fails when the schema cannot be + derived (a column reference out of range, a reserved ASAP operator). ASAP planning may + retain a more specific schema through `OperatorNode::with_schema`. +- `guarantee` is `None` until accuracy assessment establishes one; `None` never means exact. +- `timing` is `None` in every front-end DAG and every candidate. It is written by + `ir::timing::apply_lifecycle_timings` (see the Post-ASAP IR document); export rejects an + untimed node. + +`OperatorNode::children()` returns the operator's inputs in field order followed by the +operator nodes its scalar expressions read (see "Scalar expressions"). Every DAG traversal — +`map_children`, `reachable`, `contains_asap`, CSE, export — follows that same list. +`OperatorNode::validate_structure()` checks every operator's input contract, scalar typing +against the owning operator's input schema, and that each retained schema agrees with the +derived one. + +## Schema + +One `Schema` type (`crates/types/src/pre_asap/schema.rs`) describes every edge, whether it +carries rows or summary state: + +```rust +pub struct Schema { + pub fields: Vec, // positional; every ColumnId indexes into this + pub time_index: Option, // the time axis, if any (PromQL leaves always have one) + pub unique_keys: Vec>, + pub closed: bool, // true: these are all the columns; false: open (schemaless) superset +} + +pub struct Field { + pub name: String, + pub dtype: FieldDataType, // Plain(DataType) | ExactAggregate(..) | Sketch(..) | Sample(..) | Wavelet(..) | StatModel(..) + pub nullable: bool, + pub table: Option, // SQL table/alias qualifier; None for PromQL labels +} +``` + +A pre-ASAP field is always `FieldDataType::Plain(DataType)`. The other variants carry summary +state and only appear below an ASAP operator; a scalar expression that reads such a field is a +typing error (`ScalarExpr::scalar_type`), because state must be read out before a value can use +it. Column references are positional `ColumnId`s (indexes into the input schema), never names. + +`Schema::has_unique_key()` is the legality gate CSE uses: a non-ASAP producer is only shared +across consumers when its row identity is provable. + +## Scalar expressions + +Value computation lives in `ScalarExpr` (`crates/types/src/ir/scalar.rs`), owned **by value** +by an operator field: `Scan.predicates`, `Filter.pred`, `Join.pred`, `Project.cols[i].expr`, +`Aggregate.having`, `Sort.keys[i].expr`, `SQLWindowFunc.args`/`order_by`, `PromqlRelabel.value`, +`Values.rows`, and `QueryRoot::Scalar` and `PromqlVectorFromScalar`. A scalar expression never +produces a table and is never a node of the DAG; it is evaluated against the input schema of +the operator that owns it. + +Variants: `Column(ColumnId)`, `Literal(ScalarValue)`, `Negative` (unary minus), `Compare`, +`BoolAnd` / `BoolOr` (flat conjunction/disjunction), `Not`, `IsNull` / `IsNotNull`, `Cast` +(with `try_cast`), `InList`, `FunctionCall { name, args }`, `Arithmetic`, `Case`, +`CurrentTimestamp` (SQL `NOW()`), `EvalTimestamp` (PromQL `time()`), and four +**plan-reading** variants that reference an operator node: + +| Variant | Meaning | +|---|---| +| `PromqlScalarFromVector(Rc)` | PromQL `scalar(v)`: the single sample of an instant vector, NaN otherwise | +| `ScalarSubquery(Rc)` | Uncorrelated SQL scalar subquery: one column; zero rows is NULL, more than one row is an error | +| `Exists { subquery, negated }` | SQL `[NOT] EXISTS (subquery)` | +| `InSubquery { expr, subquery, negated }` | SQL `expr [NOT] IN (subquery)` over a one-column relation | + +These are the **only** operator references inside a scalar tree. `ScalarExpr::operator_refs()` +lists them, `NonASAPOp::children()` appends them after the operator's own inputs, and +canonicalization lowers the three SQL subquery forms to joins (see below), so a canonical SQL +DAG contains none of them. `PromqlScalarFromVector` survives canonicalization: its referenced +vector is a real plan dependency, exported as a `ScalarRef` edge. + +`Compare`, `Arithmetic` and `Negative` carry an `ExprSemantics` (`Sql` or `Promql`): both +languages use `Float64`, so the result type alone does not preserve NaN, ordering or error +rules, and the executing engine needs to know which language's rules apply. + +Wrapper types: `Predicate(ScalarExpr)`, `ProjectItem { alias, expr }`, +`SortKey { expr, ascending, nulls_first }`. + +## How a front end produces the DAG + +A front end never constructs `OperatorNode`s directly. It builds a name-based tree in +`crates/frontend-common` — `UnresolvedOp` / `UnresolvedScalar`, a mirror of `NonASAPOp` / +`ScalarExpr` in which every column reference is a `ColumnRef` and a PromQL `Scan` has no schema +yet — and calls `asap_frontend_common::resolve_root`, which does three things in order: + +1. **Resolution** — a bottom-up walk that binds every `ColumnRef` to a positional `ColumnId` + against the derived schema of the already-resolved child. A schemaless (PromQL) leaf gets + its binding schema from `SchemaResolver`, built from the names the query references. + `Join` / `SetOp` sides and the operators referenced from scalar positions are each bound as + a root in their own scope; a `BinaryOp` side additionally inherits the label names its + enclosing scope references. +2. **Schema derivation** — each `OperatorNode::new` derives the node's output schema and + result kind from the operator and its children. +3. **Canonicalization** — `asap_types::ir::canonicalize::canonicalize` erases structural + differences between semantically identical queries: it promotes an additive + `Limit { Sort { Aggregate } }` ranking to the `AggIntent::TopK` heavy-hitter shape, and + lowers `EXISTS` / `NOT EXISTS` / `IN (subquery)` predicates to `Join { Semi | Anti }` and a + scalar subquery to a `Join { Cross }` plus column reference. The pass is idempotent and + keeps the pointer identity of every untouched sub-DAG. + +The result is `Rc`. `lower_promql_workload`, `lower_sql` / `lower_sql_dialect` / +`lower_sql_batch` and `lower_metricsql` all return it. + +Workload search then runs structural CSE (`asap_types::ir::cse::share_common_subdags`) once +across every root: bottom-up hash-consing where the structural hash is only a filter and the +typed `PartialEq` decides sharing, following scalar references like any other input, and +gated by `Schema::has_unique_key()` for non-ASAP producers. ## Node index @@ -24,27 +160,28 @@ to one source language. - [`Aggregate`](#aggregate) — collapses input rows into fewer output rows via a reduction and aggregate intents. **[Time-related nodes](#time-related-nodes)** -- [`TimeRange`](#timerange) — a range-vector lookback over the time axis (PromQL `[5m]`). +- [`TimeRange`](#timerange) — temporal selection over a time-series input (PromQL instant lookback or `[5m]` range selector). - [`TimeShift`](#timeshift) — shifts *when* a selector is evaluated (PromQL `offset`/`@`). - [`PromqlSubquery`](#promqlsubquery) — re-evaluates an instant-vector expression over a range at a given step. **[Relational nodes](#relational-nodes)** — common to both SQL and PromQL - [`Scan`](#scan) — identifies the logical data source. +- [`Values`](#values) — SQL `VALUES` rows, or the one empty row of a `SELECT` without `FROM`. - [`Filter`](#filter) — restricts rows using a predicate. - [`Project`](#project) — column projection (SQL `SELECT` list). -- [`BinaryOp`](#binaryop) — arithmetic / comparison / boolean composition of two inputs. +- [`BinaryOp`](#binaryop) — arithmetic / comparison / set composition of two inputs. - [`Sort`](#sort) — generic (non-heavy-hitter) order-by, optionally per-group. -- [`Limit`](#limit) — caps the row count, with an offset. +- [`Limit`](#limit) — caps the row count, with an offset, optionally per-group. - [`Dedup`](#dedup) — row-level deduplication. - [`Join`](#join) — logical join of two inputs. - [`SetOp`](#setop) — SQL's typed set operations (`UNION`/`INTERSECT`/`EXCEPT`). - [`Concat`](#concat) — exact, untyped `UNION ALL` of union-compatible branches. -**[PromQL-specific nodes](#promql-specific-nodes)** -- [`PromqlScalarBridge`](#promqlscalarbridge) — a scalar sub-expression at an operator-tree position. -- [`EvalTimestamp`](#evaltimestamp) — the query evaluation time as a scalar (PromQL `time()`). +**[Scalar-position nodes](#scalar-position-nodes)** +- `QueryRoot::Scalar` — a standalone scalar expression, without an operator node. - [`PromqlVectorFromScalar`](#promqlvectorfromscalar) — promotes a scalar to a label-less instant vector. -- [`PromqlScalarFromVector`](#promqlscalarfromvector) — collapses a single-series vector to a scalar. + +**[PromQL-specific nodes](#promql-specific-nodes)** - [`PromqlRelabel`](#promqlrelabel) — per-series label rewrite (PromQL `label_replace`/`label_join`). - [`PromqlInfoEnrich`](#promqlinfoenrich) — left-join label enrichment from an info metric. - [`PromqlSeriesSample`](#promqlseriessample) — keeps a subset of whole series, not a reduction. @@ -52,6 +189,9 @@ to one source language. **[SQL-specific nodes](#sql-specific-nodes)** - [`SQLWindowFunc`](#sqlwindowfunc) — SQL analytic window function (`OVER (...)`). +PromQL `time()` and `scalar(v)` are scalar expressions (`ScalarExpr::EvalTimestamp`, +`ScalarExpr::PromqlScalarFromVector`), not nodes. + ## Aggregation-related nodes ### Aggregate @@ -86,7 +226,7 @@ list of aggregate intents (`measures`). value is still recomputed by the agg intent, e.g. `Rate`), for a computation with no `by(...)` clause to attach to. `PerEntity` is different from `by` for all columns, because in PromQL, it is schemaless and you don't know all columns beforehand. E.g. PromQL `rate(http_requests_total[5m])`, which has one rate value - per input series: + per input series: ```text Aggregate( @@ -94,7 +234,7 @@ list of aggregate intents (`measures`). measures = [Rate], output_names = [], having = None, - child = TimeRange(range = 5m, child = Scan("http_requests_total")) + child = TimeRange(range = 5m, kind = Range, child = Scan("http_requests_total")) ) ``` @@ -201,7 +341,7 @@ Example for `filters`: `count(CASE WHEN p THEN x END)` (`p`, plus `x IS NOT NULL` when `x` is nullable), and from `count(expr)` over any other nullable `expr` (`expr IS NOT NULL`), because canonical `Count` counts rows and never consults its argument. A filtered measure has no summary binding yet: - `asap-aware-mapping` keeps such an `Aggregate` as `KeepPreAsap`, and canonicalization does + `asap-aware-mapping` retains such an `Aggregate` as an ordinary exact sub-DAG, and canonicalization does not promote a filtered count ranking to a heavy-hitter `TopK`. Example for `having`: @@ -225,8 +365,8 @@ Example for `having`: **Rules/Invariants**: A filtering predicate will be passed to at the lowest node (closer to the leaves) in the AST/DAG that can express it — `Scan.predicates`, then `Aggregate.having`, then `Filter` as the fallback — so its constraint is visible at - the node it actually applies to, not behind an opaque wrapper, once pre-ASAP IR translates - to post-ASAP IR with summary binding. The upper nodes (closer to the root) in the AST/DAG can still have a `Filter` node with the same condition. This intentional duplication is for Summary related translation and optimizations. + the node it actually applies to, not behind an opaque wrapper, once summary binding reads it. + The upper nodes (closer to the root) in the AST/DAG can still have a `Filter` node with the same condition. This intentional duplication is for Summary related translation and optimizations. For example, `SELECT srcip, COUNT(*) AS cnt FROM packets GROUP BY srcip HAVING COUNT(*) > 10` pins `cnt > 10` to the lowest node that can express it, `Aggregate.having`: @@ -259,7 +399,7 @@ Example for `having`: Both are valid at once, and neither is derived from the other: `having` is the canonical spot a summary-aware pass reads to decide whether `Aggregate` can bind to a summary, while the outer `Filter` is what a plain logical evaluator runs without knowing `having` exists. The duplication is forward-looking groundwork for - once HAVING-aware summary binding (pre-ASAP-IR to post-ASAP-IR translation) lands. + once HAVING-aware summary binding lands. Neither direction of that push-down is enforced yet: the SQL front end doesn't populate `having` from a real `HAVING` clause (#201), and canonicalization doesn't fold an existing @@ -271,14 +411,21 @@ Example for `having`: ### TimeRange -Represents a range of time. Kept different from `Filter` to treat time as an explicit concern. +Temporal selection over a time-series input. Kept different from `Filter` to treat time as an +explicit concern. `kind` records which samples a PromQL selector reads: + +- `TimeRangeKind::Instant` — an instant selector: `range` is the lookback horizon and the + latest eligible sample per series is selected (the planner injects the declared + `data_ingestion_interval` around a bare selector). +- `TimeRangeKind::Range` — a range selector (`m[5m]`): every sample in the window. ```promql rate(http_requests_total[5m]) ``` **Fields:** -- `range` — how far back to look (the PromQL `[5m]` duration). +- `range` — how far back to look (the PromQL `[5m]` duration, or the instant lookback). +- `kind` — `Instant` or `Range`. - `child` — the input the range applies to. ### TimeShift @@ -327,9 +474,19 @@ the same logical data domain. **Fields:** - `source` — the logical data source (a table name or PromQL metric selector). - `predicates` — row-level filters pushed all the way down to this scan (Rules/Invariants - rule 1); enforced structurally at lowering time — a `Filter` directly over a `Scan` never - survives. + rule 1): PromQL label matchers and pushed-down `WHERE` conjuncts. - `schema` — the binding schema every positional column reference in the tree resolves against. + A catalog-backed SQL leaf carries its catalog schema; a PromQL leaf carries the usage-derived + schema `SchemaResolver` built from the labels the query references. + +### Values + +SQL `VALUES` rows, or the one empty row of a `SELECT` without `FROM` +(`SELECT 1 + 1`). Row expressions have no input-column scope. + +**Fields:** +- `rows` — one `Vec` per row. +- `schema` — the output schema of the rows. ### Filter @@ -363,6 +520,10 @@ that's neither a base scan column nor an aggregate output: SELECT * FROM (SELECT srcip, bytes_in + bytes_out AS total FROM packets) t WHERE total > 500 ``` +A `Filter` whose predicate contains `EXISTS` / `NOT EXISTS` / `IN (subquery)` does not +survive canonicalization: the conjunct becomes a `Join { Semi | Anti }` under the remaining +predicate. + **Fields:** - `pred` — the row-level predicate to apply. - `child` — the input being filtered. @@ -383,18 +544,21 @@ SELECT srcip, dstip FROM packets ### BinaryOp -Arithmetic / comparison / boolean composition. PromQL binary operators between two vectors, -a vector and a scalar, or two scalars. +Arithmetic / comparison / set composition of two operands. PromQL binary operators between two vectors, +two vectors. Mixed vector/scalar arithmetic uses `Project`; non-bool comparison uses `Filter`. Standalone scalar expressions are `QueryRoot::Scalar`. ```promql up > 1 ``` **Fields:** -- `op` — the arithmetic/comparison/boolean operator. +- `operator` — a `BinaryOperator { kind, vector_match, checked_relative_division, checked_finite_division }`: + - `kind` — `BinaryOpKind::Arithmetic(..)`, `Compare(..)` or `Set(..)` (PromQL `and`/`or`/`unless`). + - `vector_match` — PromQL vector-matching modifiers (`on`/`ignoring`, `group_left`/`group_right`); `None` outside PromQL and the only supported value today. + - `checked_relative_division` / `checked_finite_division` — typed division guards set by summary planning, never by a front end (see [physical-plan integration](../design_docs/architecture/physical-plan-integration.md#conditional-temporal-average-lowering)). +- `return_bool` — the PromQL `bool` modifier: a comparison returns `0`/`1` instead of filtering. Valid only for comparison operators. - `lhs` — the left operand. - `rhs` — the right operand. -- `vector_match` — PromQL vector-matching modifiers (`on`/`ignoring`, `group_left`/`group_right`); `None` outside PromQL. ### Sort @@ -406,7 +570,7 @@ sort_desc(up) ``` **Fields:** -- `keys` — the ordering columns/expressions and direction. +- `keys` — the ordering expressions and direction (`SortKey`). - `partition_by` — grouping keys that make the ordering per-group instead of global; empty = a single global order. - `child` — the input being ordered. @@ -420,8 +584,9 @@ topk(3, up) ``` **Fields:** -- `n` — the maximum number of rows to keep. +- `n` — the maximum number of rows to keep; `None` is offset-only. - `offset` — how many leading rows to skip first. +- `partition_by` — applies the limit per group (PromQL `topk by (..)`); empty = global. - `child` — the input being capped. ### Dedup @@ -440,14 +605,16 @@ SELECT DISTINCT srcip, dstip FROM packets ### Join -Logical join; the physical strategy (hash/merge/broadcast) is picked in the post-ASAP IR. SQL `JOIN`. +Logical join; the physical strategy (hash/merge/broadcast) is picked downstream of the planner. SQL `JOIN`, +and the shape canonicalization lowers subqueries to. ```sql SELECT u.prefix FROM bgp_updates u JOIN bgp_rib_state r ON u.prefix = r.prefix ``` **Fields:** -- `kind` — the join type (inner/left/right/full/semi/anti). +- `kind` — the join type (`Inner`/`Left`/`Right`/`Full`/`Cross`/`Semi`/`Anti`). A semi/anti join + outputs the left input's columns alone, but its predicate resolves against `left ++ right`. - `pred` — the join condition. - `left` — the left input. - `right` — the right input. @@ -473,7 +640,7 @@ SELECT srcip FROM packets UNION ALL SELECT dstip FROM packets never dedup. Used when a single `Aggregate` can't express the shape — the canonical case is PromQL `histogram_quantiles` (one branch per φ, each its own `HistogramQuantile` reduction relabeled with its `le` value) — and SQL `ROLLUP`/`CUBE`/`GROUPING SETS` (one branch per -grouping level). +grouping level). The output schema is the first child's. ```promql histogram_quantiles(rate(http_request_duration_seconds_bucket[5m]), "le", 0.5, 0.9) @@ -481,39 +648,32 @@ histogram_quantiles(rate(http_request_duration_seconds_bucket[5m]), "le", 0.5, 0 **Fields:** - `children` — the union-compatible branches to concatenate; must be non-empty. +- `discriminator_unique_key` — an optional caller-proven compound unique key + `(discriminator, inner_key)` over the output; nothing verifies the claim. -## PromQL-specific nodes - -### PromqlScalarBridge +## Scalar-position nodes -A scalar sub-expression (issue #220: in practice always `Literal(ScalarValue::Float64(_))` — -a PromQL number literal, or a folded constant scalar expression) sitting at an **operator-tree -position** — a `BinaryOp` operand for ` op ` thresholds and unit conversions, -a `PromqlVectorFromScalar` child, or a whole query's root. This wrapper is what marks the -position; it no longer duplicates `Literal`'s value the way the old `PromqlScalar(f64)` variant -did. - -```promql -up > 1 -``` +### Scalar query roots -**Fields:** a single unnamed child `QueryExpr` — the wrapped scalar sub-expression. +`QueryRoot` distinguishes an operator result from an owned `ScalarExpr`. It is +an API root discriminator, not an operator. `2`, `time()`, and +`scalar(sum(up)) + 1` therefore introduce no constant-wrapper nodes. -### EvalTimestamp +Use `lower_promql_query_workload` for mixed scalar/vector workloads. The +operator-only convenience API rejects standalone scalar roots. `ParsedWorkload` +retains each scalar's workload index; `PlanOutput::roots()` returns all results +in workload order. Scalar plan reads remain exact and retain their operator +references; summary selection currently operates on operator roots. -The query **evaluation timestamp** as Unix seconds — PromQL `time()` — and the implicit -input of the no-argument calendar functions (`hour()`, `day_of_week()`, ...). It is the -instant or range-step at which the expression is evaluated, not inherently the current -wall-clock time. The Prometheus instant-query HTTP API separately defaults an omitted -`time` request parameter to the server's current time. - -```promql -time() -``` +`up * 2` projects the sample expression while retaining time and full series +identity, removing the metric name. `up > 0` and `0 < up` filter the vector and +retain its sample and name. `up > bool 0` projects a zero-or-one `Case`. +Open label schemas acquire a full runtime series-identity field before this +lowering. The runtime must populate that field with all labels. ### PromqlVectorFromScalar -The scalar→instant-vector bridge — PromQL `vector(s)`. Promotes a scalar-typed child to a +The scalar→instant-vector bridge — PromQL `vector(s)`. Promotes a scalar expression to a single label-less series carrying that value at every step, e.g. for dead-man's-switch patterns (`up or vector(0)`). @@ -521,18 +681,9 @@ patterns (`up or vector(0)`). vector(1) ``` -**Fields:** a single unnamed child `QueryExpr` — the scalar-typed expression being promoted to a vector. - -### PromqlScalarFromVector +**Fields:** a single unnamed `ScalarExpr` — the scalar being promoted to a vector. -The instant-vector→scalar bridge — PromQL `scalar(v)`. Collapses a single-element vector to -its value (NaN at runtime if the input isn't exactly one series). - -```promql -scalar(up) -``` - -**Fields:** a single unnamed child `QueryExpr` — the single-series vector being collapsed to a scalar. +## PromQL-specific nodes ### PromqlRelabel @@ -593,5 +744,6 @@ SELECT srcip, LAG(time) OVER (PARTITION BY srcip ORDER BY time) FROM packets - `args` — the function's operand expressions; empty for rank-only functions. - `partition_by` — grouping keys the window is computed within. - `order_by` — the ordering the window function reads. +- `frame` — the optional window frame. - `output_name` — the name of the new output column. - `child` — the input the window function is computed over. diff --git a/docs/develop_docs/target-candidate-api-migration.md b/docs/develop_docs/target-candidate-api-migration.md index 6139f53dc..a05cc6d0c 100644 --- a/docs/develop_docs/target-candidate-api-migration.md +++ b/docs/develop_docs/target-candidate-api-migration.md @@ -13,7 +13,7 @@ are unchanged. #453 separately defines the integration API surface. | `MaterializeSummaryMaintenanceLifecycleError` | `SummaryMaintenanceLifecycleAssemblyError` | Failure assembling a DAG or deriving maintenance decisions | | Error variant `Materialize` | `AssembleDag` | Wrap an underlying `RealizationError` from DAG assembly | | Internal `materialize_inner` / `materialize_residual` | `assemble_target` / `assemble_residual` | Assemble selected nodes, not runtime materialized views | -| Internal assembly cache `materialized` | `assembled_nodes` | Preserve shared `Rc` identity | +| Internal assembly cache `materialized` | `assembled_nodes` | Preserve shared node identity (now `Rc`, see below) | Update imports and calls together; old public names are not retained as aliases. Downstream Rust integrations using these symbols must migrate. No serialized @@ -26,3 +26,26 @@ counterpart) are prerequisites, not additional changes here. The workflow remains one selection call per workload followed by one assembly call per query root. `SummaryMaintenanceLifecyclePlan` contains the assembled Post-ASAP DAG root plus maintenance decisions; it is not an executable plan. + +## Later: unified operator IR (operator flattening) + +The pre-ASAP and post-ASAP trees became one IR in `asap_types::ir`. Every +node is an `Rc` whose `operator` is `Operator::NonASAP(NonASAPOp)` +or `Operator::ASAP(ASAPOp)`. Old public names are not kept as aliases. + +| Old | New | +|---|---| +| `Rc` (pre-ASAP) | `Rc` holding `Operator::NonASAP(NonASAPOp)` | +| `Rc` / `SummaryExpr` (post-ASAP) | The same `Rc`; summary steps are `Operator::ASAP(ASAPOp)` | +| `SummaryExpr::KeepPreAsap(q)` | The non-ASAP sub-DAG itself; `retain_exact` only adds an exact `guarantee` | +| `SummaryExpr::ValueOperation { .. }` over a evaluation | An ordinary `NonASAPOp` (`Project`, `Filter`, `Sort`, `Limit`, `Aggregate`) reading an ASAP node; `FinalizeExactAccumulator`, `MaintainPopulation`, `EvaluatePopulation` are `ASAPOp` variants | +| `Replacement::Summary(..)` / `Replacement::Rewrite(..)` | `Replacement::SubDag(Rc)`; `is_logical_rewrite` tells them apart | +| `SummaryFamilyType` | `FieldDataType` (its non-`Plain` variants) | +| Timing stored on post-ASAP nodes | `OperatorNode::timing`, `None` until `ir::timing::apply_lifecycle_timings` writes it from a `LifecycleAssignment` | +| `UnresolvedQueryExpr` + `asap_types::pre_asap::resolve_root` | `UnresolvedOp` / `UnresolvedScalar` + `asap_frontend_common::resolve_root` | +| `pre_asap::canonicalize`, `pre_asap::cse::share_common_subdags` | `ir::canonicalize::canonicalize`, `ir::cse::share_common_subdags` | +| `asap_types::post_asap::compile_post_asap_dag` (wire version 5, `Fallback`/`Binary`/`Value` payloads) | `asap_types::ir::export::compile_post_asap_dag` (wire version 7: one node per operator, `Relational` payloads, `ScalarRef` edges); input must be timed | +| Exported schema JSON `columns` | `fields` | + +Field and schema details: [Pre-ASAP IR](pre-asap-ir.md) and +[Post-ASAP IR](../design_docs/concepts/post-asap-ir.md). diff --git a/docs/user_guide_docs/run-a-query.md b/docs/user_guide_docs/run-a-query.md index af2cfc07e..4621ac41f 100644 --- a/docs/user_guide_docs/run-a-query.md +++ b/docs/user_guide_docs/run-a-query.md @@ -109,8 +109,10 @@ whether to select them using its own evidence. Planner's automatic `global_selection` skips them; their presence alone does not show that they meet the requested target. -Each input line is followed by its debug IR or an `ERR:` message. Post-ASAP -output may contain summary state, readouts or exact `KeepPreAsap` work. An +Each input line is followed by its debug IR or an `ERR:` message. Pre-ASAP and +Post-ASAP output use the same node format: Post-ASAP output adds summary nodes +(state, readouts) and keeps the original exact operators wherever no summary +replaces them. An approximate target permits approximation; it does not guarantee a legal or certified sketch. The tool prints plans, not query results. diff --git a/tools/dag-viewer/README.md b/tools/dag-viewer/README.md index d2c1a65ff..178414020 100644 --- a/tools/dag-viewer/README.md +++ b/tools/dag-viewer/README.md @@ -69,7 +69,8 @@ cargo run -p asap-devtools --bin dag_export -- \ Load the JSON with the page's file picker. `--planner-cost-json` is a complete physical-evidence document: an immutable `evidence_version`, calibration, and -target records containing the exact target `QueryExpr` and comparison scope. +target records containing the exact target node (a serialized pre-ASAP +`OperatorNode`) and comparison scope. Each exact replacement candidate owns its complete logical-node `PhysicalNodeEvidence`; summary candidates additionally own their bound `PhysicalDag`. Candidate-local evidence prevents statistics for one physical @@ -129,7 +130,7 @@ a selected replacement directly contains: { "decision": { "id": 7, - "strategy": "SketchAlgorithmStrategy", + "strategy": "ASAPStrategies", "rationale": "count realizes as a Cms sketch", "rank": 0, "cost": 1.14001088, @@ -147,8 +148,13 @@ The exporter assigns `workload_node_id`; union rendering reads that mapping directly. Node boxes use concrete IR fields: aggregate measures/grouping, sort keys, -filter predicates, projections, sources, summary families, and readout -queries. Category icons are deliberately omitted so they cannot be confused +filter predicates, projections, sources, summary families, and evaluation +queries. A node's `kind` is the operator variant name (`Operator::kind_name`): +a `NonASAPOp` such as `Aggregate` or `Values`, or an `ASAPOp` such as +`SummaryAgg` or `EvaluatePopulation`. `node-style.js` maps each kind to a color +category. Scalar expressions are not nodes; an operator a scalar expression +reads (`scalar(v)`, `EXISTS (subquery)`) is a child node, shown in `detail` +as `{"scalar_ref": }`. Schemas list their entries under `fields`. Category icons are deliberately omitted so they cannot be confused with IR text. ### Cost/benefit annotations (issue #286) diff --git a/tools/dag-viewer/dag.example.json b/tools/dag-viewer/dag.example.json index da3e88708..f9b34aa95 100644 --- a/tools/dag-viewer/dag.example.json +++ b/tools/dag-viewer/dag.example.json @@ -210,7 +210,7 @@ { "decision_id": 0, "target_pre_id": 1, - "strategy": "SketchAlgorithmStrategy", + "strategy": "ASAPStrategies", "rationale": "count realizes as a Cms sketch", "rank": 0, "cost": 1.14001088, @@ -363,108 +363,7 @@ "kind": "Summary", "graph": { "nodes": [ - { - "id": 0, - "kind": "KeepPreAsap", - "label": "KeepPreAsap(Scan)", - "detail": { - "pre_asap_subgraph": { - "nodes": [ - { - "children": [], - "detail": { - "predicates": [], - "schema": { - "closed": true, - "columns": [ - { - "dtype": "timestamp", - "name": "ts", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "utf8", - "name": "service", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "utf8", - "name": "region", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "float64", - "name": "latency", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "int64", - "name": "bytes", - "nullable": false, - "table": "metrics" - } - ], - "time_index": 0, - "unique_keys": [] - }, - "source": { - "Table": { - "table_ref": "metrics" - } - } - }, - "hash": 2606922452740434172, - "id": 0, - "kind": "Scan", - "label": "Scan(metrics)", - "schema": { - "closed": true, - "columns": [ - { - "dtype": "timestamp", - "name": "ts", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "utf8", - "name": "service", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "utf8", - "name": "region", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "float64", - "name": "latency", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "int64", - "name": "bytes", - "nullable": false, - "table": "metrics" - } - ], - "time_index": 0, - "unique_keys": [] - } - } - ], - "root": 0 - } - }, - "children": [] - }, + {"id": 0, "kind": "Scan", "label": "Scan(metrics)", "detail": {"predicates": [], "schema": {"closed": true, "columns": [{"dtype": "timestamp", "name": "ts", "nullable": false, "table": "metrics"}, {"dtype": "utf8", "name": "service", "nullable": false, "table": "metrics"}, {"dtype": "utf8", "name": "region", "nullable": false, "table": "metrics"}, {"dtype": "float64", "name": "latency", "nullable": false, "table": "metrics"}, {"dtype": "int64", "name": "bytes", "nullable": false, "table": "metrics"}], "time_index": 0, "unique_keys": []}, "source": {"Table": {"table_ref": "metrics"}}}, "children": []}, { "id": 1, "kind": "SummaryAgg", @@ -657,7 +556,7 @@ "hash": 2606922452740434172, "decision": { "id": 0, - "strategy": "SketchAlgorithmStrategy", + "strategy": "ASAPStrategies", "rationale": "count realizes as a Cms sketch", "rank": 0, "cost": 1.14001088, @@ -763,7 +662,7 @@ "workload_node_id": 1, "decision": { "id": 0, - "strategy": "SketchAlgorithmStrategy", + "strategy": "ASAPStrategies", "rationale": "count realizes as a Cms sketch", "rank": 0, "cost": 1.14001088, @@ -862,7 +761,7 @@ "workload_node_id": 2, "decision": { "id": 0, - "strategy": "SketchAlgorithmStrategy", + "strategy": "ASAPStrategies", "rationale": "count realizes as a Cms sketch", "rank": 0, "cost": 1.14001088, diff --git a/tools/dag-viewer/generate-sample.sh b/tools/dag-viewer/generate-sample.sh index e916cba5f..66921d8c7 100755 --- a/tools/dag-viewer/generate-sample.sh +++ b/tools/dag-viewer/generate-sample.sh @@ -6,7 +6,7 @@ set -euo pipefail cd "$(dirname "${BASH_SOURCE[0]}")/../.." # --epsilon asks for an approximate accuracy target instead of the default -# Exact, so SketchAlgorithmStrategy actually has a sketch alternative to +# Exact, so ASAPStrategies actually has a sketch alternative to # report — without it, no query below would ever pick up a `notes` badge # (see crates/devtools/src/bin/dag_export.rs's own `--epsilon` doc comment). cargo run -p asap-devtools --bin dag_export -- \ diff --git a/tools/dag-viewer/node-style.js b/tools/dag-viewer/node-style.js index 447f08801..f68be81b9 100644 --- a/tools/dag-viewer/node-style.js +++ b/tools/dag-viewer/node-style.js @@ -1,20 +1,17 @@ -// Logical QueryExpr/SummaryExpr kinds exported by +// Operator kinds (`Operator::kind_name`) exported by // crates/types/src/dag_export.rs. Categories describe the visible logical DAG // shape. They do not model hidden physical inputs: for example, // PromqlInfoEnrich is a one-child enrichment here even if physical costing // later accounts for an auxiliary source scan. const KIND_CATEGORY_JSON = `{ "Scan": "data", - "PromqlScalarBridge": "data", - "EvalTimestamp": "data", - "CurrentTimestamp": "data", + "Values": "data", "Filter": "filter", "PromqlSeriesSample": "sample", "Project": "derive", "PromqlRelabel": "derive", "PromqlInfoEnrich": "derive", "PromqlVectorFromScalar": "derive", - "PromqlScalarFromVector": "derive", "BinaryOp": "derive", "Aggregate": "aggregate", "TimeRange": "window", @@ -22,21 +19,21 @@ const KIND_CATEGORY_JSON = `{ "TimeShift": "window", "SQLWindowFunc": "window", "Join": "join", - "RelationalJoin": "join", "Dedup": "set", "SetOp": "set", "Concat": "combine", "Sort": "sort", "Limit": "sort", - "KeepPreAsap": "summary", "SummaryAgg": "summary", "SummaryJoin": "summary", "SummarySubtract": "summary", - "SummaryBinaryOp": "summary", - "ValueOperation": "summary", "SummaryDelete": "summary", "SummaryEstimate": "summary", - "SummaryMerge": "summary" + "SummaryMerge": "summary", + "FinalizeExactAccumulator": "summary", + "MaintainPopulation": "summary", + "EvaluatePopulation": "summary", + "Extension": "summary" }`; const KIND_CATEGORY = Object.freeze(JSON.parse(KIND_CATEGORY_JSON)); @@ -44,7 +41,7 @@ const KIND_CATEGORY = Object.freeze(JSON.parse(KIND_CATEGORY_JSON)); const CATEGORIES = { data: { label: 'Data', - description: 'Scan, PromqlScalarBridge, EvalTimestamp, CurrentTimestamp — leaves that introduce a value', + description: 'Scan and Values — data sources', light: { bg: '#eef5fd', border: '#0369a1' }, dark: { bg: '#0c2438', border: '#38bdf8' }, }, @@ -62,7 +59,7 @@ const CATEGORIES = { }, derive: { label: 'Derive', - description: 'Project, PromqlRelabel, PromqlInfoEnrich, PromqlVectorFromScalar, PromqlScalarFromVector, BinaryOp — transforms or enriches columns on otherwise-unchanged rows', + description: 'Project, PromqlRelabel, PromqlInfoEnrich, PromqlVectorFromScalar, BinaryOp — transforms or enriches columns on otherwise-unchanged rows', light: { bg: '#f5f0fd', border: '#6d28d9' }, dark: { bg: '#241a3d', border: '#a78bfa' }, }, @@ -102,10 +99,10 @@ const CATEGORIES = { light: { bg: '#eef4fd', border: '#1d4ed8' }, dark: { bg: '#12233d', border: '#60a5fa' }, }, - // Post-ASAP nodes use a neutral palette; KeepPreAsap has a muted override. + // ASAP operators use a neutral palette. summary: { label: 'Summary', - description: 'KeepPreAsap, SummaryBinaryOp, ValueOperation, SummaryAgg, SummaryJoin, SummarySubtract, SummaryDelete, SummaryEstimate, SummaryMerge — post-ASAP materialized structures', + description: 'SummaryAgg, SummaryEstimate, FinalizeExactAccumulator, MaintainPopulation, EvaluatePopulation, SummaryJoin, SummarySubtract, SummaryDelete, SummaryMerge, Extension — summary state and its evaluations', light: { bg: '#f1f2f4', border: '#4b5563' }, dark: { bg: '#20242b', border: '#9ca3af' }, }, diff --git a/tools/dag-viewer/post_asap_fixture.json b/tools/dag-viewer/post_asap_fixture.json index a680d28f4..cc613f9e1 100644 --- a/tools/dag-viewer/post_asap_fixture.json +++ b/tools/dag-viewer/post_asap_fixture.json @@ -14,9 +14,9 @@ }, "post_graph": { "nodes": [ - { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {"pre_asap_subgraph": {"nodes": [{"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": []}], "root": 0}}, "children": [] }, + {"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": []}, { "id": 1, "kind": "SummaryAgg", "label": "SummaryAgg(Kll)", "detail": {"family": {"Sketch": ["Kll", {"k": 200}]}, "col": {"Column": 6}, "reduction": {"Reduce": [1]}, "grouping": "PerSubpopulationInstance"}, "children": [0], "origin_pre_id": 1 }, - { "id": 2, "kind": "KeepPreAsap", "label": "KeepPreAsap(Project)", "detail": {"pre_asap_subgraph": {"nodes": [{"id": 0, "kind": "Project", "label": "Project(2 cols)", "detail": {}, "children": []}], "root": 0}}, "children": [1] } + {"id": 2, "kind": "Project", "label": "Project(2 cols)", "detail": {}, "children": [1]} ], "root": 2 }, @@ -39,7 +39,7 @@ "kind": "Summary", "graph": { "nodes": [ - { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {"pre_asap_subgraph": {"nodes": [{"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": [], "hash": 111}], "root": 0}}, "children": [] }, + {"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": []}, { "id": 1, "kind": "SummaryAgg", "label": "SummaryAgg(Kll)", "detail": {"family": {"Sketch": ["Kll", {"k": 200}]}, "col": {"Column": 6}, "reduction": {"Reduce": [1]}, "grouping": "PerSubpopulationInstance"}, "children": [0], "origin_pre_id": 1 } ], "root": 1 @@ -64,7 +64,7 @@ "kind": "Summary", "graph": { "nodes": [ - { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {}, "children": [] }, + {"id": 0, "kind": "Scan", "label": "Scan", "detail": {}, "children": []}, { "id": 1, "kind": "SummaryAgg", "label": "SummaryAgg(HydraKll)", "detail": {"family": {"Sketch": ["Kll", {"k": 200}]}, "grouping": {"SharedMultiSubpopulation": {"params": {}}}}, "children": [0], "origin_pre_id": 1 } ], "root": 1 diff --git a/tools/dag-viewer/render.py b/tools/dag-viewer/render.py index 6659561ef..4d5e82dfb 100755 --- a/tools/dag-viewer/render.py +++ b/tools/dag-viewer/render.py @@ -90,7 +90,7 @@ def _compact(value: object) -> str: def _column(value: object, input_schema: object = None) -> str: if isinstance(value, int): if isinstance(input_schema, dict): - columns = input_schema.get("columns") + columns = input_schema.get("fields", input_schema.get("columns")) if isinstance(columns, list) and value < len(columns): column = columns[value] if isinstance(column, dict) and column.get("name"): @@ -174,7 +174,11 @@ def _semantic_label(node: dict, input_schema: object = None) -> str: lines.append(f"within: {_compact(detail['partition_by'])}") elif kind == "Project": output_schema = node.get("schema") - output_columns = output_schema.get("columns") if isinstance(output_schema, dict) else None + output_columns = ( + output_schema.get("fields", output_schema.get("columns")) + if isinstance(output_schema, dict) + else None + ) if isinstance(output_columns, list) and output_columns: names = [str(column.get("name", "?")) for column in output_columns if isinstance(column, dict)] lines.append("columns: " + ", ".join(names)) @@ -212,12 +216,6 @@ def _semantic_label(node: dict, input_schema: object = None) -> str: lines.append(f"query: {_compact(detail.get('query'))}") elif kind == "SummaryDelete": lines.append(f"key: {_compact(detail.get('key'))}") - elif kind == "KeepPreAsap": - nested = detail.get("pre_asap_subgraph") - nested_nodes = nested.get("nodes", []) if isinstance(nested, dict) else [] - nested_root = nested.get("root") if isinstance(nested, dict) else None - root = next((item for item in nested_nodes if item.get("id") == nested_root), None) - lines.append(f"unchanged: {root.get('kind', 'pre-ASAP subtree') if root else 'pre-ASAP subtree'}") else: # Less common variants still show their own scalar IR fields. Avoid # schema/subgraph blobs, which belong in the click-to-inspect panel. diff --git a/tools/dag-viewer/viewer.js b/tools/dag-viewer/viewer.js index 62b2e84ca..c792e9d0a 100644 --- a/tools/dag-viewer/viewer.js +++ b/tools/dag-viewer/viewer.js @@ -290,22 +290,6 @@ function buildCyStyle() { selector: 'node[category = "unknown"]', style: { 'border-style': 'dashed', 'border-width': 3 }, }, - { - // KeepPreAsap (post-ASAP lane only) is post-ASAP-only - // as a *kind*, but represents literally unchanged pre-ASAP content — - // override the 'summary' category's color/icon with the same neutral - // panel/muted/dashed treatment the rest of the chrome uses for "nothing - // to see here", so a glance at the After lane separates "the planner - // did something" (solid, colored) from "left alone" (dashed, muted). - // See node-style.js's CATEGORIES.summary comment for the category-level - // color choice this overrides. - selector: 'node[kind = "KeepPreAsap"]', - style: { - 'background-color': panelColor, - 'border-color': borderColor, - 'border-style': 'dashed', - }, - }, { selector: 'node.root', style: { 'border-width': 2.5 }, @@ -663,10 +647,7 @@ function laneElements(laneId, laneLabel, graph, query, stage, laneCost) { // exactly the plain IR label. label: node.label + nodeCostBadgeSuffix(node), node, - // Flat (not nested under `node`) so buildCyStyle's - // `node[kind = "KeepPreAsap"]` selector can actually match it — - // cytoscape selectors can't reach into a data field that's itself an - // object. + // Cytoscape selectors read flat data fields. kind: node.kind, category: categoryOf(node.kind), root: node.id === graph.root, @@ -1171,10 +1152,6 @@ function renderLegend() { Shared workload nodeExplicitly identified by the exporter as shared across selected queries`); rows.push(`
Query root${escapeHtml(ROOT_BADGE.description)}
`); - const panelBg = getComputedStyle(document.documentElement).getPropertyValue('--panel2').trim() || '#f0f2f5'; - const mutedColor = getComputedStyle(document.documentElement).getPropertyValue('--muted').trim() || '#6b7280'; - rows.push(`
- Pass-through (KeepPreAsap)Unchanged pre-ASAP subtree carried into the Summary graph as-is
`); legendList.innerHTML = rows.join(''); }