From 050b98399c4ef721cbc0667c18918c4af0d9fe9b Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Fri, 2 Oct 2026 17:46:00 +0000 Subject: [PATCH 1/7] refactor: unify value and summary-state schemas --- .../src/accuracy/estimators/cms.rs | 4 +- .../src/accuracy/estimators/count_sketch.rs | 2 +- .../src/accuracy/estimators/hll.rs | 2 +- .../src/accuracy/estimators/kll.rs | 2 +- .../src/accuracy/estimators/mod.rs | 20 +- .../src/accuracy/evidence.rs | 8 +- crates/asap-aware-mapping/src/accuracy/mod.rs | 9 +- .../src/accuracy/reconciliation.rs | 8 +- crates/asap-aware-mapping/src/cost_model.rs | 73 +++-- .../asap-aware-mapping/src/empirical_cost.rs | 4 +- .../src/exact_composition.rs | 22 +- crates/asap-aware-mapping/src/explanation.rs | 22 +- crates/asap-aware-mapping/src/grouping.rs | 34 ++- .../src/maintained_population.rs | 27 +- .../src/physical_plan_cost_model.rs | 4 +- .../src/query_physical_lowering.rs | 58 ++-- crates/asap-aware-mapping/src/recurrence.rs | 50 ++-- crates/asap-aware-mapping/src/replacement.rs | 255 ++++++++--------- crates/asap-aware-mapping/src/rewrite.rs | 40 +-- crates/asap-aware-mapping/src/rollup.rs | 10 +- .../src/summary_maintenance_cost/mod.rs | 4 +- .../src/summary_maintenance_cost/model.rs | 47 ++-- .../src/summary_maintenance_cost/window.rs | 4 +- .../src/summary_maintenance_lifecycle.rs | 48 ++-- .../asap-physical-operators/src/capability.rs | 42 ++- .../src/expressions/planner.rs | 24 +- .../src/operators/aggregate/temporal.rs | 16 +- .../src/operators/aligned_binary.rs | 8 +- .../src/operators/common.rs | 14 +- .../src/operators/mod.rs | 8 +- .../src/operators/scope_timestamp.rs | 2 +- .../src/operators/summary/mod.rs | 44 +-- .../src/operators/vector_window.rs | 5 +- .../src/physical_planner/mod.rs | 24 +- .../src/physical_planner/precompute.rs | 82 +++--- .../src/physical_planner/promql_fallback.rs | 2 +- .../src/physical_planner/promql_rows.rs | 24 +- .../src/physical_planner/promql_values.rs | 8 +- crates/asap-physical-operators/src/readout.rs | 4 +- .../src/runtime/batch_execution.rs | 23 +- .../src/sources/memory.rs | 2 +- .../src/sources/mod.rs | 18 +- .../src/summary_kernels/exact.rs | 44 ++- .../src/summary_kernels/factory.rs | 10 +- crates/asap-physical-operators/src/values.rs | 32 +-- .../tests/blocking_resources.rs | 27 +- .../tests/current_series_heap.rs | 19 +- .../tests/deployment.rs | 9 +- .../tests/deployment_computation.rs | 13 +- .../tests/physical_dag.rs | 70 +++-- .../tests/physical_plan_recovery.rs | 11 +- .../tests/physical_semantics.rs | 27 +- .../tests/plan_properties.rs | 19 +- .../tests/planspace_series_identity_heap.rs | 2 +- .../tests/precompute_candidates.rs | 46 ++-- .../tests/precompute_population.rs | 53 ++-- .../tests/promql_binary.rs | 34 ++- .../tests/promql_values.rs | 4 +- .../asap-physical-operators/tests/raw_scan.rs | 14 +- .../tests/summary_projection.rs | 21 +- .../tests/weighted_topk_binding.rs | 28 +- .../devtools/examples/canonical_examples.rs | 6 +- crates/devtools/examples/topk_ir.rs | 6 +- crates/devtools/src/bin/analyze_corpora.rs | 6 +- crates/devtools/src/bin/dag_export.rs | 30 +- crates/devtools/src/bin/show_post_asap_ir.rs | 6 +- crates/devtools/src/bin/show_pre_asap_ir.rs | 6 +- crates/devtools/src/bin/sketch_coverage.rs | 6 +- crates/devtools/src/bin/variant_coverage.rs | 6 +- crates/devtools/tests/cross_language.rs | 6 +- .../frontend-promql/tests/count_planning.rs | 10 +- .../tests/promql_conformance.rs | 53 ++-- .../frontend-promql/tests/promql_lowering.rs | 16 +- .../tests/univmon_candidates.rs | 10 +- .../src/sql/collection_planning.rs | 6 +- crates/frontend-sql/src/sql/mod.rs | 10 +- crates/frontend-sql/src/sql/types.rs | 42 +-- .../tests/bgp_analytics/bgp_analytics.rs | 6 +- .../bgp_jan2024_workload.rs | 6 +- .../synthetic_packet_trace.rs | 6 +- .../tests/data_quality_check/tpch_deequ.rs | 6 +- .../tests/maintained_population.rs | 6 +- crates/frontend-sql/tests/netflow/netflow.rs | 6 +- crates/frontend-sql/tests/pearson_corr.rs | 18 +- crates/frontend-sql/tests/sql_lowering.rs | 142 +++++----- crates/frontend-sql/tests/temporal_types.rs | 14 +- crates/integration-tests/src/lib.rs | 16 +- .../tests/exact_composition.rs | 31 ++- .../tests/frontend_timestamps.rs | 8 +- .../tests/kll_pane_execution.rs | 17 +- .../tests/precompute_raw_samples.rs | 25 +- .../tests/promql_numeric_regressions.rs | 25 +- .../tests/promql_to_post_asap.rs | 42 +-- crates/integration-tests/tests/schema.rs | 2 +- .../tests/sql_to_physical.rs | 10 +- .../tests/sql_to_post_asap.rs | 39 ++- .../summary_maintenance_lifecycle_e2e.rs | 24 +- crates/planner/tests/e2e_plan.rs | 6 +- crates/planner/tests/summary_sharing.rs | 16 +- crates/types/src/dag_export.rs | 57 ++-- crates/types/src/post_asap/cse.rs | 33 +-- .../src/post_asap/execution_data_state.rs | 145 +++++----- crates/types/src/post_asap/expr.rs | 20 +- .../src/post_asap/maintained_population.rs | 8 +- crates/types/src/post_asap/mod.rs | 5 +- crates/types/src/post_asap/post_asap_dag.rs | 77 +++--- crates/types/src/post_asap/schema.rs | 64 ----- crates/types/src/post_asap/sketch.rs | 6 +- crates/types/src/pre_asap/agg_intent.rs | 18 +- crates/types/src/pre_asap/canonicalize.rs | 18 +- .../types/src/pre_asap/column_resolution.rs | 55 ++-- crates/types/src/pre_asap/cse.rs | 10 +- crates/types/src/pre_asap/expr_ir.rs | 8 +- crates/types/src/pre_asap/mod.rs | 2 +- crates/types/src/pre_asap/query_expr.rs | 206 +++++++------- crates/types/src/pre_asap/resolve.rs | 24 +- crates/types/src/pre_asap/scalar_signature.rs | 46 ++-- crates/types/src/pre_asap/schema.rs | 258 +++++++++++++----- crates/types/src/pre_asap/schema_resolver.rs | 34 +-- 119 files changed, 1703 insertions(+), 1677 deletions(-) delete mode 100644 crates/types/src/post_asap/schema.rs diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs b/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs index 0f2f37c63..692e418ad 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs @@ -47,7 +47,7 @@ mod tests { let params = default_size_params(SketchAlgorithm::Cms, &c, 0.01, 0.001); let g = DefaultAccuracyModel .local_guarantee( - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Cms, params), GroupingStrategy::default(), ), @@ -74,7 +74,7 @@ mod tests { }; let topk_frequency = DefaultAccuracyModel .local_guarantee( - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::CmsWithHeap, cms_heap), GroupingStrategy::default(), ), diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/count_sketch.rs b/crates/asap-aware-mapping/src/accuracy/estimators/count_sketch.rs index 73af6ffc9..bba46f5d7 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/count_sketch.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/count_sketch.rs @@ -62,7 +62,7 @@ mod tests { let count_sketch = default_size_params(SketchAlgorithm::CountSketch, &intent, 0.01, 0.01); let guarantee = DefaultAccuracyModel .local_guarantee( - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::CountSketch, count_sketch), GroupingStrategy::default(), ), diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs b/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs index 6a17e1594..ec5913a32 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs @@ -237,7 +237,7 @@ mod tests { let params = default_size_params(SketchAlgorithm::Hll, &c, 0.01, 0.01); let g = DefaultAccuracyModel .local_guarantee( - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Hll, params), GroupingStrategy::default(), ), diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs b/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs index 005d34ad2..f9932526a 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs @@ -49,7 +49,7 @@ mod tests { let params = default_size_params(SketchAlgorithm::Kll, &q, 0.01, 0.01); let g = DefaultAccuracyModel .local_guarantee( - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, params), GroupingStrategy::default(), ), diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs b/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs index df9ef4d9f..2731e0097 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs @@ -56,21 +56,19 @@ fn bounded_guarantee( } pub(super) fn local_guarantee( - family: &SummaryFamilyType, + family: &FieldDataType, query: &SketchQuery, ) -> Option { match family { - SummaryFamilyType::Plain(_) => Some(ResultGuarantee::exact("Plain value")), - SummaryFamilyType::ExactAggregate(kind, _) => { + FieldDataType::Plain(_) => Some(ResultGuarantee::exact("Plain value")), + FieldDataType::ExactAggregate(kind, _) => { Some(ResultGuarantee::exact(format!("ExactAggregate({kind:?})"))) } - SummaryFamilyType::Sketch(kind, _) => { - sketch_guarantee(kind.algorithm(), kind.params(), query) - } + FieldDataType::Sketch(kind, _) => sketch_guarantee(kind.algorithm(), kind.params(), query), // No error model is registered for these families. - SummaryFamilyType::Sample(..) - | SummaryFamilyType::Wavelet(..) - | SummaryFamilyType::StatModel(..) => None, + FieldDataType::Sample(..) | FieldDataType::Wavelet(..) | FieldDataType::StatModel(..) => { + None + } } } pub(crate) fn size_params( @@ -197,10 +195,10 @@ impl AccuracyModel for EstimatorAccuracy<'_> { } fn local_guarantee( &self, - family: &SummaryFamilyType, + family: &FieldDataType, query: &SketchQuery, ) -> Option { - if let (Some(_), SummaryFamilyType::Sketch(kind, grouping), SketchQuery::Cardinality) = + if let (Some(_), FieldDataType::Sketch(kind, grouping), SketchQuery::Cardinality) = (self.contract, family, query) { if let (SketchAlgorithm::Hll, SketchParams::Hll { precision }) = diff --git a/crates/asap-aware-mapping/src/accuracy/evidence.rs b/crates/asap-aware-mapping/src/accuracy/evidence.rs index e73a1fc84..6ca296cfd 100644 --- a/crates/asap-aware-mapping/src/accuracy/evidence.rs +++ b/crates/asap-aware-mapping/src/accuracy/evidence.rs @@ -118,7 +118,7 @@ pub trait AccuracyEvidenceProvider { fn propagation_stats( &self, _op: &CompositionOperator, - _family: &SummaryFamilyType, + _family: &FieldDataType, _query: Option<&SketchQuery>, ) -> PropagationStats { PropagationStats::default() @@ -142,7 +142,7 @@ impl AccuracyEvidenceProvider for WorkloadAccuracyEvidence<'_> { fn propagation_stats( &self, _op: &CompositionOperator, - _family: &SummaryFamilyType, + _family: &FieldDataType, _query: Option<&SketchQuery>, ) -> PropagationStats { PropagationStats { @@ -180,7 +180,7 @@ mod tests { }; let fresh = provider.propagation_stats( &CompositionOperator::ExactSum, - &SummaryFamilyType::ExactAggregate( + &FieldDataType::ExactAggregate( asap_types::post_asap::ExactKind::Sum, asap_types::post_asap::ExactParams::Sum, ), @@ -195,7 +195,7 @@ mod tests { } .propagation_stats( &CompositionOperator::ExactSum, - &SummaryFamilyType::ExactAggregate( + &FieldDataType::ExactAggregate( asap_types::post_asap::ExactKind::Sum, asap_types::post_asap::ExactParams::Sum, ), diff --git a/crates/asap-aware-mapping/src/accuracy/mod.rs b/crates/asap-aware-mapping/src/accuracy/mod.rs index 9eef3315d..1451dd382 100644 --- a/crates/asap-aware-mapping/src/accuracy/mod.rs +++ b/crates/asap-aware-mapping/src/accuracy/mod.rs @@ -21,9 +21,8 @@ pub use evidence::{ }; use asap_types::post_asap::{ - AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ExactOperation, GuaranteeSource, - ProbabilityExpr, ResultGuarantee, SketchAlgorithm, SketchParams, SketchQuery, - SummaryFamilyType, + AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ExactOperation, FieldDataType, + GuaranteeSource, ProbabilityExpr, ResultGuarantee, SketchAlgorithm, SketchParams, SketchQuery, }; use asap_types::types::AccuracyTarget; @@ -48,7 +47,7 @@ pub trait AccuracyModel { /// `Sample`/`Wavelet`/`StatModel`). fn local_guarantee( &self, - family: &SummaryFamilyType, + family: &FieldDataType, query: &SketchQuery, ) -> Option; @@ -94,7 +93,7 @@ impl AccuracyModel for DefaultAccuracyModel { } fn local_guarantee( &self, - family: &SummaryFamilyType, + family: &FieldDataType, query: &SketchQuery, ) -> Option { estimators::local_guarantee(family, query) diff --git a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs index 15f8f27df..6014d3ba3 100644 --- a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs +++ b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs @@ -394,7 +394,7 @@ mod tests { use asap_types::post_asap::SketchAlgorithm; use asap_types::pre_asap::cse::share_common_sub_dags; use asap_types::pre_asap::query_expr::{GroupKeys, Source}; - use asap_types::pre_asap::schema::{Column, ColumnId, DataType, Schema}; + use asap_types::pre_asap::schema::{ColumnId, DataType, Field, Schema}; /// `[ts(0), value(1), job(2)]`. /// A unique-keyed scan (`[ts]`) so `share_common_sub_dags` is actually @@ -407,9 +407,9 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - Column::new("job", DataType::Utf8, true), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("job", DataType::Utf8, true), ], 0, vec![vec![0]], diff --git a/crates/asap-aware-mapping/src/cost_model.rs b/crates/asap-aware-mapping/src/cost_model.rs index 6b3c32e1e..8a51aa790 100644 --- a/crates/asap-aware-mapping/src/cost_model.rs +++ b/crates/asap-aware-mapping/src/cost_model.rs @@ -49,8 +49,8 @@ use std::rc::Rc; use asap_types::post_asap::{ - ExactOperation, GroupingStrategy, HydraParams, ResultGuarantee, SketchAlgorithm, SketchParams, - SketchQuery, SummaryExpr, SummaryFamilyType, SummaryMaintenanceLifecycleGuarantee, SummaryNode, + ExactOperation, FieldDataType, GroupingStrategy, HydraParams, ResultGuarantee, SketchAlgorithm, + SketchParams, SketchQuery, SummaryExpr, SummaryMaintenanceLifecycleGuarantee, SummaryNode, SummaryWindowFramework, }; use asap_types::pre_asap::agg_intent::AggIntent; @@ -275,10 +275,10 @@ fn finite_rate(units_per_second: f64) -> Option { /// onto one `Rc` for two or more workload roots. See /// `docs/design_docs/cse-cost-model-decision.md`. pub struct CseCandidate<'a> { - /// The shared pre-ASAP sub-DAG itself. + /// The shared pre-ASAP sub_dag itself. pub sub_dag: &'a QueryExpr, - /// The `SummaryNode` this sub-DAG bound to — gives the cost model the - /// concrete `SummaryFamilyType`/`(kind, params)` actually at stake, not + /// The `SummaryNode` this sub_dag bound to — gives the cost model the + /// concrete `FieldDataType`/`(kind, params)` actually at stake, not /// just the pre-ASAP shape. pub bound_summary: &'a SummaryNode, /// How many workload roots reference this exact shared sub-DAG, counted @@ -387,7 +387,7 @@ pub fn default_cse_recompute_cost(sub_dag: &QueryExpr) -> Cost { } /// Default [`CostModel::cse_shared_maintenance_cost`]: a small -/// per-[`SummaryFamilyType`] weight, scaled to the same order of magnitude +/// per-[`FieldDataType`] weight, scaled to the same order of magnitude /// as [`default_cse_recompute_cost`]'s typical output (a small node /// count, not a byte length), reflecting that families differ in how /// expensive they are to keep *continuously updated* for the life of a @@ -398,15 +398,15 @@ pub fn default_cse_recompute_cost(sub_dag: &QueryExpr) -> Cost { /// deployment with real memory/update-cost numbers should override /// [`CostModel::cse_shared_maintenance_cost`] instead of relying on this /// table. -pub fn default_cse_shared_maintenance_cost(family: &SummaryFamilyType) -> Cost { +pub fn default_cse_shared_maintenance_cost(family: &FieldDataType) -> Cost { const UNIT: f64 = 1.0; let weight = match family { - SummaryFamilyType::Plain(_) => 1.0, - SummaryFamilyType::ExactAggregate(..) => 1.0, - SummaryFamilyType::Sketch(..) => 3.0, - SummaryFamilyType::Sample(..) => 3.0, - SummaryFamilyType::Wavelet(..) => 5.0, - SummaryFamilyType::StatModel(..) => 6.0, + FieldDataType::Plain(_) => 1.0, + FieldDataType::ExactAggregate(..) => 1.0, + FieldDataType::Sketch(..) => 3.0, + FieldDataType::Sample(..) => 3.0, + FieldDataType::Wavelet(..) => 5.0, + FieldDataType::StatModel(..) => 6.0, }; Cost(weight * UNIT) } @@ -590,9 +590,9 @@ pub trait CostModel { .fields .iter() .map(|f| &f.dtype) - .find(|dtype| !matches!(dtype, SummaryFamilyType::Plain(_))) + .find(|dtype| !matches!(dtype, FieldDataType::Plain(_))) .cloned() - .unwrap_or(SummaryFamilyType::Plain( + .unwrap_or(FieldDataType::Plain( asap_types::pre_asap::DataType::Float64, )); default_cse_shared_maintenance_cost(&family) @@ -953,7 +953,7 @@ fn sketch_state( match &node.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_state(summary_input), SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, grouping), + family: FieldDataType::Sketch(kind, grouping), .. } => Some((kind, grouping)), _ => None, @@ -1347,11 +1347,10 @@ mod tests { // ── CSE sharing (issue #237, #223 stage 4) ────────────────────────── use asap_types::post_asap::{ - ExactKind, ExactParams, GroupingStrategy, SketchKind, SummaryExpr, SummaryField, - SummarySchema, + ExactKind, ExactParams, Field, GroupingStrategy, Schema, SketchKind, SummaryExpr, }; use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{Column, DataType, Schema}; + use asap_types::pre_asap::schema::DataType; fn scan() -> QueryExpr { QueryExpr::Scan { @@ -1359,8 +1358,8 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], 0, vec![], @@ -1368,15 +1367,12 @@ mod tests { } } - fn summary_node(family: SummaryFamilyType) -> SummaryNode { + fn summary_node(family: FieldDataType) -> SummaryNode { SummaryNode { expr: SummaryExpr::SummaryAgg { child: std::rc::Rc::new(SummaryNode { expr: SummaryExpr::KeepPreAsap(Rc::new(scan())), - schema: SummarySchema { - fields: vec![], - time_index: None, - }, + schema: Schema::lifted(vec![], None), guarantee: None, }), family: family.clone(), @@ -1387,14 +1383,7 @@ mod tests { grouping: GroupingStrategy::default(), filter: None, }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "state".into(), - dtype: family, - nullable: false, - }], - time_index: None, - }, + schema: Schema::lifted(vec![Field::new("state", family, false)], None), guarantee: None, } } @@ -1455,11 +1444,11 @@ mod tests { #[test] fn default_shared_maintenance_cost_orders_families_cheapest_to_priciest() { - let exact = default_cse_shared_maintenance_cost(&SummaryFamilyType::ExactAggregate( + let exact = default_cse_shared_maintenance_cost(&FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); - let sketch = default_cse_shared_maintenance_cost(&SummaryFamilyType::Sketch( + let sketch = default_cse_shared_maintenance_cost(&FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Hll, SketchParams::Hll { precision: 12 }), GroupingStrategy::default(), )); @@ -1474,7 +1463,7 @@ mod tests { fn cse_share_decision_shares_when_recompute_dominates_maintenance() { let candidate = CseCandidate { sub_dag: &scan(), - bound_summary: &summary_node(SummaryFamilyType::ExactAggregate( + bound_summary: &summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )), @@ -1492,7 +1481,7 @@ mod tests { fn cse_share_decision_recomputes_when_maintenance_dominates_recompute() { let candidate = CseCandidate { sub_dag: &scan(), - bound_summary: &summary_node(SummaryFamilyType::StatModel( + bound_summary: &summary_node(FieldDataType::StatModel( asap_types::post_asap::StatModelKind::Parametric, asap_types::post_asap::StatModelParams::Parametric { family: "gaussian_mixture".into(), @@ -1532,7 +1521,7 @@ mod tests { // hardcoding a comparison against its own defaults. let candidate = CseCandidate { sub_dag: &scan(), - bound_summary: &summary_node(SummaryFamilyType::StatModel( + bound_summary: &summary_node(FieldDataType::StatModel( asap_types::post_asap::StatModelKind::Parametric, asap_types::post_asap::StatModelParams::Parametric { family: "gaussian_mixture".into(), @@ -1570,7 +1559,7 @@ mod tests { let target = TargetSubDAG::new(&root); let candidate = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Summary(Rc::new(summary_node(SummaryFamilyType::Plain( + replacement: Replacement::Summary(Rc::new(summary_node(FieldDataType::Plain( asap_types::pre_asap::DataType::Float64, )))), provenance: crate::replacement::ReplacementProvenance::SummaryRealization, @@ -1595,14 +1584,14 @@ mod tests { let cheap = ReplacementSubDAG { strategy: "TestStrategy", replacement: Replacement::Summary(Rc::new(summary_node( - SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), ))), provenance: crate::replacement::ReplacementProvenance::SummaryRealization, rationale: "exact accumulator".into(), }; let pricey = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Summary(Rc::new(summary_node(SummaryFamilyType::StatModel( + replacement: Replacement::Summary(Rc::new(summary_node(FieldDataType::StatModel( asap_types::post_asap::StatModelKind::Parametric, asap_types::post_asap::StatModelParams::Parametric { family: "gaussian_mixture".into(), diff --git a/crates/asap-aware-mapping/src/empirical_cost.rs b/crates/asap-aware-mapping/src/empirical_cost.rs index ebfbc773c..b6d4f9a08 100644 --- a/crates/asap-aware-mapping/src/empirical_cost.rs +++ b/crates/asap-aware-mapping/src/empirical_cost.rs @@ -3,7 +3,7 @@ //! of an accuracy guarantee. CPU quantities are nanoseconds, never CPU operations. use asap_types::post_asap::{ - GroupingStrategy, SketchAlgorithm, SketchParams, SummaryExpr, SummaryFamilyType, SummaryNode, + FieldDataType, GroupingStrategy, SketchAlgorithm, SketchParams, SummaryExpr, SummaryNode, }; use asap_types::pre_asap::AggIntent; use serde::{Deserialize, Serialize}; @@ -217,7 +217,7 @@ impl EmpiricalEvidenceProvider { summary: &SummaryNode, ) -> SummaryMaintenanceLifecycleCostInputs { let SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, GroupingStrategy::PerSubpopulationInstance), + family: FieldDataType::Sketch(kind, GroupingStrategy::PerSubpopulationInstance), grouping: GroupingStrategy::PerSubpopulationInstance, .. } = &summary.expr diff --git a/crates/asap-aware-mapping/src/exact_composition.rs b/crates/asap-aware-mapping/src/exact_composition.rs index 0a683550c..54bce26f6 100644 --- a/crates/asap-aware-mapping/src/exact_composition.rs +++ b/crates/asap-aware-mapping/src/exact_composition.rs @@ -65,8 +65,8 @@ use std::rc::Rc; use asap_types::post_asap::execution_data_state::validate_execution_data_states_at; use asap_types::post_asap::{ exact_operation_output_schema, produced_data_state, AccuracyError, ExactOperation, - ExecutionDataState, ExecutionDataStateError, ResultGuarantee, SummaryExpr, SummaryNode, - SummarySchema, ValueOperation, + ExecutionDataState, ExecutionDataStateError, ResultGuarantee, Schema, SummaryExpr, SummaryNode, + ValueOperation, }; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::pre_asap::query_expr::{any_measure_filtered, QueryExpr, Reduction}; @@ -124,7 +124,7 @@ pub struct ExactComposition { /// The composed node's output schema — the target's own pre-ASAP /// output schema, lifted with every column `Plain` (an exact operator /// only ever produces plain values). - pub schema: SummarySchema, + pub schema: Schema, } impl ExactComposition { @@ -466,17 +466,21 @@ mod tests { use super::*; use crate::cost_model::{DefaultCostModel, ValueOperationCapabilities}; use crate::replacement::keep_pre_asap; - use asap_types::post_asap::{ExecutionDataStateError, SketchAlgorithm, SummaryFamilyType}; + use asap_types::post_asap::{ExecutionDataStateError, FieldDataType, SketchAlgorithm}; use asap_types::pre_asap::agg_intent::default_quantile; use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{Column, DataType, Schema}; + use asap_types::pre_asap::schema::{DataType, Field, Schema}; fn metric_scan(labels: &[&str]) -> QueryExpr { let mut columns = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ]; - columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); + columns.extend( + labels + .iter() + .map(|n| Field::plain(*n, DataType::Utf8, true)), + ); QueryExpr::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], @@ -658,7 +662,7 @@ mod tests { .schema .fields .iter() - .all(|f| matches!(f.dtype, SummaryFamilyType::Plain(_)))); + .all(|f| matches!(f.dtype, FieldDataType::Plain(_)))); } #[test] diff --git a/crates/asap-aware-mapping/src/explanation.rs b/crates/asap-aware-mapping/src/explanation.rs index 633e30d34..e7afff2ba 100644 --- a/crates/asap-aware-mapping/src/explanation.rs +++ b/crates/asap-aware-mapping/src/explanation.rs @@ -44,7 +44,7 @@ //! //! - [`ExplanationKind::SketchApproximation`] — the `TargetSubDAG`'s //! candidate list contains at least one [`Replacement::Summary`] that -//! actually realizes a sketch family (`SummaryFamilyType::Sketch`), i.e. +//! actually realizes a sketch family (`FieldDataType::Sketch`), i.e. //! [`SketchAlgorithmStrategy`] found something to offer beyond whatever //! exact/pass-through candidate [`crate::replacement`]'s own //! `realizations_for_intent` would have committed to on its own. @@ -165,7 +165,7 @@ //! | Roll-ups (fine-to-coarse group-by reuse) | [`RollupStrategy`](crate::rollup::RollupStrategy), derived from workload siblings after CSE/target discovery (issue #254) | Any `Replacement::Rewrite` candidate that rolls a coarse aggregate up from a compatible finer aggregate | //! | Wavelets/OMP | Params type exists (`WaveletKind`/`WaveletParams`), reachable only via a deployment `CostModel::realize_extension` (no core `AggIntent` dispatch picks it) | A `ReplacementStrategy` that inspects a deployment's own `CostModel`, once some intent shape actually maps to `Realization::Wavelet` | //! | Sampling | Same story as Wavelets: `SamplingKind`/`SamplingParams` exist, unreachable from core dispatch | Same hook as Wavelets, for `Realization::Sample` | -//! | Deep generative compression | No representation at all — no `Realization`/`SummaryFamilyType` variant | Needs a new summary family added to `asap_types::post_asap` first | +//! | Deep generative compression | No representation at all — no `Realization`/`FieldDataType` variant | Needs a new summary family added to `asap_types::post_asap` first | //! | Approximation frameworks for windows | No representation — `TimeRange`/`PromqlSubquery` windows are always evaluated exactly | Would key off those node types once an approximate-window operator exists | //! | Function decomposition | No representation anywhere | No hook point identified yet | //! | Continuous distributed monitoring | No representation — `RepeatingEntry`/`RepetitionInterval` in `asap_types::workload` describe *that* a query repeats, not any monitoring-specific decomposition | Would likely key off `RepeatingEntry` once such logic exists | @@ -186,7 +186,7 @@ use std::collections::HashMap; use std::fmt::Display; use std::rc::Rc; -use asap_types::post_asap::{SummaryExpr, SummaryFamilyType, SummaryNode}; +use asap_types::post_asap::{FieldDataType, SummaryExpr, SummaryNode}; use asap_types::pre_asap::cse::{structural_hash, HashCache}; use asap_types::pre_asap::query_expr::QueryExpr; @@ -405,7 +405,7 @@ fn shared_subexpr_finding_reason(group: &TargetSubDAGCandidates) -> Option bool { } match &node.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => is_sketch_realization(summary_input), - SummaryExpr::SummaryAgg { family, .. } => matches!(family, SummaryFamilyType::Sketch(..)), + SummaryExpr::SummaryAgg { family, .. } => matches!(family, FieldDataType::Sketch(..)), _ => false, } } @@ -516,15 +516,19 @@ mod tests { use super::*; use asap_types::pre_asap::agg_intent::{default_quantile, AggIntent}; use asap_types::pre_asap::query_expr::{Reduction, Source}; - use asap_types::pre_asap::schema::{Column, DataType, Schema}; + use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; fn metric_scan(labels: &[&str]) -> QueryExpr { let mut columns = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ]; - columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); + columns.extend( + labels + .iter() + .map(|n| Field::plain(*n, DataType::Utf8, true)), + ); QueryExpr::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], diff --git a/crates/asap-aware-mapping/src/grouping.rs b/crates/asap-aware-mapping/src/grouping.rs index f74205205..247472ed9 100644 --- a/crates/asap-aware-mapping/src/grouping.rs +++ b/crates/asap-aware-mapping/src/grouping.rs @@ -9,7 +9,7 @@ //! //! `SummaryExpr::SummaryAgg` carries the grouping choice next to the //! `Reduction` whose `by` keys determine legality. The same choice is also -//! committed to `SummaryFamilyType::Sketch` on the aggregate's output edge. +//! committed to `FieldDataType::Sketch` on the aggregate's output edge. //! That duplication is intentional: the node field makes the choice easy to //! inspect during planning, while the edge type ensures an independent KLL/ //! CMS state and a Hydra-backed state cannot be accepted as compatible inputs @@ -73,8 +73,8 @@ use std::rc::Rc; use asap_types::post_asap::{ default_hydra_params, hydra_kind_for, AccuracyError, BoundExpr, CompositionOperator, - GroupingStrategy, GuaranteeSource, HydraKind, ProbabilityExpr, ResultGuarantee, - SketchAlgorithm, SketchParams, SummaryExpr, SummaryFamilyType, SummaryNode, + FieldDataType, GroupingStrategy, GuaranteeSource, HydraKind, ProbabilityExpr, ResultGuarantee, + SketchAlgorithm, SketchParams, SummaryExpr, SummaryNode, }; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; @@ -363,7 +363,7 @@ fn per_subpopulation_sketch_params(node: &SummaryNode) -> Option { per_subpopulation_sketch_params(summary_input) } SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + family: FieldDataType::Sketch(kind, _), .. } => Some(kind.params().clone()), _ => None, @@ -401,15 +401,15 @@ fn with_grouping( .. } => { let grouped_family = match family { - SummaryFamilyType::Sketch(kind, _) => { - SummaryFamilyType::Sketch(kind.clone(), grouping.clone()) + FieldDataType::Sketch(kind, _) => { + FieldDataType::Sketch(kind.clone(), grouping.clone()) } _ => family.clone(), }; let mut grouped_schema = node.schema.clone(); for field in &mut grouped_schema.fields { - if let SummaryFamilyType::Sketch(kind, _) = &field.dtype { - field.dtype = SummaryFamilyType::Sketch(kind.clone(), grouping.clone()); + if let FieldDataType::Sketch(kind, _) = &field.dtype { + field.dtype = FieldDataType::Sketch(kind.clone(), grouping.clone()); } } Rc::new(SummaryNode { @@ -495,15 +495,19 @@ mod tests { use asap_types::post_asap::ErrorMetric; use asap_types::pre_asap::agg_intent::{default_cardinality, default_quantile}; use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{Column, DataType, Schema}; + use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; fn metric_scan(labels: &[&str]) -> QueryExpr { let mut columns = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ]; - columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); + columns.extend( + labels + .iter() + .map(|n| Field::plain(*n, DataType::Utf8, true)), + ); QueryExpr::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], @@ -668,7 +672,7 @@ mod tests { fn propagation_stats( &self, _op: &CompositionOperator, - _family: &SummaryFamilyType, + _family: &FieldDataType, _query: Option<&asap_types::post_asap::SketchQuery>, ) -> PropagationStats { PropagationStats { @@ -712,7 +716,7 @@ mod tests { fn propagation_stats( &self, _op: &CompositionOperator, - _family: &SummaryFamilyType, + _family: &FieldDataType, _query: Option<&asap_types::post_asap::SketchQuery>, ) -> PropagationStats { PropagationStats { @@ -758,7 +762,7 @@ mod tests { fn propagation_stats( &self, _op: &CompositionOperator, - _family: &SummaryFamilyType, + _family: &FieldDataType, _query: Option<&asap_types::post_asap::SketchQuery>, ) -> PropagationStats { PropagationStats { diff --git a/crates/asap-aware-mapping/src/maintained_population.rs b/crates/asap-aware-mapping/src/maintained_population.rs index 61c88b0ad..99960dd08 100644 --- a/crates/asap-aware-mapping/src/maintained_population.rs +++ b/crates/asap-aware-mapping/src/maintained_population.rs @@ -3,8 +3,8 @@ use crate::replacement::{ Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_types::post_asap::{ - maintained_population::*, ExecutionTiming, ResultGuarantee, SummaryExpr, SummaryFamilyType, - SummaryField, SummaryNode, SummarySchema, ValueOperation, + maintained_population::*, ExecutionTiming, ResultGuarantee, SummaryExpr, SummaryNode, + ValueOperation, }; use asap_types::pre_asap::{ any_measure_filtered, AggIntent, CompareOpKind, DataType, QueryExpr, Reduction, ScalarValue, @@ -12,19 +12,8 @@ use asap_types::pre_asap::{ }; use std::rc::Rc; -fn plain(schema: Schema) -> SummarySchema { - SummarySchema { - time_index: schema.time_index, - fields: schema - .columns - .into_iter() - .map(|c| SummaryField { - name: c.name, - dtype: SummaryFamilyType::Plain(c.dtype), - nullable: c.nullable, - }) - .collect(), - } +fn plain(schema: Schema) -> Schema { + Schema::lifted(schema.fields, schema.time_index) } fn strip_projection(mut root: &QueryExpr) -> &QueryExpr { @@ -62,7 +51,7 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou _ => return None, }; let schema = child.output_schema().ok()?; - if col.is_some_and(|c| schema.columns.get(c).is_none()) { + if col.is_some_and(|c| schema.fields.get(c).is_none()) { return None; } (child, grouping, readout, col) @@ -106,7 +95,7 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou { let value_column = value_column.or_else(|| { schema - .columns + .fields .iter() .position(|c| c.dtype == DataType::Float64 && !c.nullable) })?; @@ -145,7 +134,7 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou else { return None; }; - if value_column.is_some_and(|c| schema.columns.get(c).is_none_or(|c| c.name != "value")) { + if value_column.is_some_and(|c| schema.fields.get(c).is_none_or(|c| c.name != "value")) { return None; } // PromQL can retain open labels or resolve them into a complete identity column. @@ -156,7 +145,7 @@ fn recognize(root: &QueryExpr) -> Option<(MaintainedPopulation, PopulationReadou return None; } let label = |col: usize| -> Option { - let c = schema.columns.get(col)?; + let c = schema.fields.get(col)?; (c.dtype == DataType::Utf8).then(|| c.name.clone()) }; let mut matchers = Vec::new(); diff --git a/crates/asap-aware-mapping/src/physical_plan_cost_model.rs b/crates/asap-aware-mapping/src/physical_plan_cost_model.rs index cd6bf7cb6..307fb9a61 100644 --- a/crates/asap-aware-mapping/src/physical_plan_cost_model.rs +++ b/crates/asap-aware-mapping/src/physical_plan_cost_model.rs @@ -359,7 +359,7 @@ mod tests { use std::cell::Cell; use std::collections::HashMap; - use asap_types::pre_asap::{Column, DataType, QueryExpr, Reduction, Schema, Source}; + use asap_types::pre_asap::{DataType, Field, QueryExpr, Reduction, Schema, Source}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, @@ -419,7 +419,7 @@ mod tests { table_ref: "events".into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), }), }) } diff --git a/crates/asap-aware-mapping/src/query_physical_lowering.rs b/crates/asap-aware-mapping/src/query_physical_lowering.rs index 0aa526059..1b8e5a751 100644 --- a/crates/asap-aware-mapping/src/query_physical_lowering.rs +++ b/crates/asap-aware-mapping/src/query_physical_lowering.rs @@ -343,7 +343,7 @@ pub fn lower_query_physical_dag( child .output_schema() .map_err(|_| AnalyticalCostError::UnsupportedQueryOperator)? - .columns + .fields .len() } else { cols.len() @@ -1060,8 +1060,8 @@ fn hash_join_key_count( let (Ok(left_schema), Ok(right_schema)) = (left.output_schema(), right.output_schema()) else { return None; }; - let left_width = left_schema.columns.len(); - let total_width = left_width.saturating_add(right_schema.columns.len()); + let left_width = left_schema.fields.len(); + let total_width = left_width.saturating_add(right_schema.fields.len()); fn column_side(column: usize, left_width: usize, total_width: usize) -> Option { if column < left_width { @@ -1452,7 +1452,7 @@ mod tests { #[test] fn correlation_lowers_to_physical_hash_aggregate() { use asap_types::pre_asap::{ - AggIntent, Column, DataType, QueryExpr, Reduction, Schema, Source, + AggIntent, DataType, Field, QueryExpr, Reduction, Schema, Source, }; let source = Source::Table { table_ref: "pairs".into(), @@ -1467,8 +1467,8 @@ mod tests { source: source.clone(), predicates: vec![], schema: Schema::new(vec![ - Column::new("x", DataType::Float64, true), - Column::new("y", DataType::Float64, true), + Field::plain("x", DataType::Float64, true), + Field::plain("y", DataType::Float64, true), ]), }), }); @@ -1502,7 +1502,7 @@ mod tests { #[test] fn query_lowering_recurses_and_fuses_global_sort_limit() { use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr, Reduction, SortKey, Source}; - use asap_types::pre_asap::{Column, DataType, Schema}; + use asap_types::pre_asap::{DataType, Field, Schema}; use std::rc::Rc; let scan = Rc::new(QueryExpr::Scan { @@ -1513,8 +1513,8 @@ mod tests { QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Boolean(true)), ))], schema: Schema::new(vec![ - Column::new("service", DataType::Utf8, false), - Column::new("value", DataType::Float64, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("value", DataType::Float64, false), ]), }); let aggregate = Rc::new(QueryExpr::Aggregate { @@ -1640,7 +1640,7 @@ mod tests { #[test] fn query_lowering_shares_only_provider_identified_physical_nodes() { - use asap_types::pre_asap::{Column, CompareOpKind, DataType, Schema}; + use asap_types::pre_asap::{CompareOpKind, DataType, Field, Schema}; use asap_types::pre_asap::{JoinKind, Predicate, QueryExpr, Source}; use std::rc::Rc; @@ -1649,7 +1649,7 @@ mod tests { table_ref: "dimensions".into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), + schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), }); let root = Rc::new(QueryExpr::Join { kind: JoinKind::Inner, @@ -1802,7 +1802,7 @@ mod tests { #[test] fn query_lowering_covers_relational_unary_operators() { - use asap_types::pre_asap::{Column, DataType, ScalarValue, Schema}; + use asap_types::pre_asap::{DataType, Field, ScalarValue, Schema}; use asap_types::pre_asap::{ GroupKeys, Predicate, QueryExpr, SortKey, Source, TimeShift, WindowFuncKind, }; @@ -1813,7 +1813,7 @@ mod tests { table_ref: "events".into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), + schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), }); let filter = Rc::new(QueryExpr::Filter { pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), @@ -1976,7 +1976,7 @@ mod tests { #[test] fn query_lowering_maps_concat_and_union_all_but_rejects_distinct_set_ops() { - use asap_types::pre_asap::{Column, DataType, Schema}; + use asap_types::pre_asap::{DataType, Field, Schema}; use asap_types::pre_asap::{QueryExpr, RelationalSetOpKind, Source}; use std::rc::Rc; @@ -1985,7 +1985,7 @@ mod tests { table_ref: name.into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), + schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), }; let union = Rc::new(QueryExpr::SetOp { kind: RelationalSetOpKind::Union, @@ -2057,7 +2057,7 @@ mod tests { #[test] fn query_lowering_fails_closed_for_missing_or_inconsistent_statistics() { - use asap_types::pre_asap::{Column, DataType, Schema}; + use asap_types::pre_asap::{DataType, Field, Schema}; use asap_types::pre_asap::{QueryExpr, Source}; use std::rc::Rc; @@ -2066,7 +2066,7 @@ mod tests { table_ref: "events".into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), + schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), }); let root = Rc::new(QueryExpr::Project { cols: vec![], @@ -2173,7 +2173,7 @@ mod tests { #[test] fn query_lowering_accepts_a_consistently_empty_edge() { - use asap_types::pre_asap::{Column, DataType, ScalarValue, Schema}; + use asap_types::pre_asap::{DataType, Field, ScalarValue, Schema}; use asap_types::pre_asap::{Predicate, QueryExpr, Source}; use std::rc::Rc; @@ -2182,7 +2182,7 @@ mod tests { table_ref: "events".into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("id", DataType::Int64, false)]), + schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), }); let filter = Rc::new(QueryExpr::Filter { pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(false)))), @@ -2236,7 +2236,7 @@ mod tests { use asap_types::pre_asap::{ AggIntent, GroupKeys, QueryExpr, Reduction, Source, WindowFuncKind, }; - use asap_types::pre_asap::{Column, DataType, Schema}; + use asap_types::pre_asap::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use std::rc::Rc; @@ -2246,7 +2246,7 @@ mod tests { table_ref: "events".into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), }) }; let exact_quantile = Rc::new(QueryExpr::Aggregate { @@ -2323,7 +2323,7 @@ mod tests { #[test] fn promql_presence_is_lowered_with_a_per_step_output_bound() { use asap_types::pre_asap::{ - AggIntent, Column, DataType, QueryExpr, Reduction, Schema, Source, + AggIntent, DataType, Field, QueryExpr, Reduction, Schema, Source, }; let source = Source::TimeSeries { @@ -2332,7 +2332,7 @@ mod tests { let scan = Rc::new(QueryExpr::Scan { source: source.clone(), predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), }); let root = Rc::new(QueryExpr::Aggregate { reduction: Reduction::PerEntity, @@ -2390,14 +2390,14 @@ mod tests { #[test] fn promql_range_and_subquery_preserve_internal_steps() { - use asap_types::pre_asap::{Column, DataType, QueryExpr, Schema, Source}; + use asap_types::pre_asap::{DataType, Field, QueryExpr, Schema, Source}; use std::time::Duration; let source = Source::TimeSeries { metric: "m".into() }; let scan = Rc::new(QueryExpr::Scan { source: source.clone(), predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), }); let range = Rc::new(QueryExpr::TimeRange { range: Duration::from_secs(300), @@ -2465,7 +2465,7 @@ mod tests { #[test] fn promql_binary_lowering_keeps_operation_and_matching_cardinality() { use asap_types::pre_asap::{ - ArithmeticOpKind, BinaryOpKind, Column, DataType, GroupSide, QueryExpr, Schema, Source, + ArithmeticOpKind, BinaryOpKind, DataType, Field, GroupSide, QueryExpr, Schema, Source, VectorGrouping, VectorMatch, VectorMatchKind, }; @@ -2475,7 +2475,7 @@ mod tests { Rc::new(QueryExpr::Scan { source, predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), }) }; let root = Rc::new(QueryExpr::BinaryOp { @@ -2548,7 +2548,7 @@ mod tests { #[test] fn promql_relabel_sample_and_per_series_lower_as_a_complete_chain() { use asap_types::pre_asap::{ - AggIntent, Column, DataType, GroupKeys, QueryExpr, Reduction, SampleKind, ScalarValue, + AggIntent, DataType, Field, GroupKeys, QueryExpr, Reduction, SampleKind, ScalarValue, Schema, Source, }; @@ -2558,7 +2558,7 @@ mod tests { let scan = Rc::new(QueryExpr::Scan { source: source.clone(), predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), }); let relabel = Rc::new(QueryExpr::PromqlRelabel { dst: "service".into(), diff --git a/crates/asap-aware-mapping/src/recurrence.rs b/crates/asap-aware-mapping/src/recurrence.rs index 56cc67b44..660787cdb 100644 --- a/crates/asap-aware-mapping/src/recurrence.rs +++ b/crates/asap-aware-mapping/src/recurrence.rs @@ -780,12 +780,12 @@ mod tests { use crate::cost_model::CseCandidate; use asap_types::post_asap::{ - ExactKind, ExactParams, GroupingStrategy, ResultGuarantee, SummaryExpr, SummaryFamilyType, - SummaryField, SummaryNode, SummarySchema, + ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, ResultGuarantee, Schema, + SummaryExpr, SummaryNode, }; use asap_types::pre_asap::expr_ir::ColumnRef; use asap_types::pre_asap::query_expr::{QueryExpr, Reduction, Source}; - use asap_types::pre_asap::schema::{Column, DataType, Schema}; + use asap_types::pre_asap::schema::DataType; use std::rc::Rc; fn scan() -> QueryExpr { @@ -794,8 +794,8 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], 0, vec![], @@ -803,15 +803,12 @@ mod tests { } } - fn summary_node(family: SummaryFamilyType) -> SummaryNode { + fn summary_node(family: FieldDataType) -> SummaryNode { SummaryNode { expr: SummaryExpr::SummaryAgg { child: Rc::new(SummaryNode { expr: SummaryExpr::KeepPreAsap(Rc::new(scan())), - schema: SummarySchema { - fields: vec![], - time_index: None, - }, + schema: Schema::lifted(vec![], None), guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), }), family: family.clone(), @@ -822,14 +819,7 @@ mod tests { grouping: GroupingStrategy::default(), filter: None, }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "state".into(), - dtype: family, - nullable: false, - }], - time_index: None, - }, + schema: Schema::lifted(vec![Field::new("state", family, false)], None), guarantee: None, } } @@ -837,7 +827,7 @@ mod tests { #[test] fn decide_falls_back_to_structural_decision_when_profile_is_empty() { let sub_dag = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -863,7 +853,7 @@ mod tests { #[test] fn decide_rejects_mixed_one_shot_and_repeating_without_horizon() { let sub_dag = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -882,7 +872,7 @@ mod tests { #[test] fn decide_accepts_mixed_one_shot_and_repeating_with_an_explicit_horizon() { let sub_dag = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -943,7 +933,7 @@ mod tests { #[test] fn high_frequency_selects_maintained_low_frequency_selects_recompute() { let sub_dag = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -999,7 +989,7 @@ mod tests { #[test] fn update_rate_only_affects_maintained_cost_evaluation_rate_affects_both() { let sub_dag = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -1039,7 +1029,7 @@ mod tests { #[test] fn one_shot_only_consumer_decides_without_an_explicit_horizon() { let sub_dag = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -1072,7 +1062,7 @@ mod tests { #[test] fn one_shot_only_single_consumer_does_not_unconditionally_prefer_share() { let sub_dag = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -1099,7 +1089,7 @@ mod tests { #[test] fn batch_only_workload_does_not_unconditionally_prefer_share_under_default_cost_model() { let sub_dag = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); @@ -1148,9 +1138,9 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - Column::new("job", DataType::Utf8, true), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("job", DataType::Utf8, true), ], 0, vec![], @@ -1490,7 +1480,7 @@ mod tests { #[test] fn decide_rejects_a_zero_or_negative_horizon() { let sub_dag = scan(); - let bound = summary_node(SummaryFamilyType::ExactAggregate( + let bound = summary_node(FieldDataType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index aecf65f89..4b766579a 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -351,10 +351,11 @@ use std::collections::{HashMap, HashSet, VecDeque}; use asap_types::post_asap::{ validate_execution_data_states_at, EntityIdentity, ExactKind, ExactOperation, ExactOperationSchemaError, ExactParams, ExecutionDataState, ExecutionDataStateError, - ExecutionTiming, GroupingStrategy, NonNegativeWeightProof, SamplingKind, SamplingParams, - SketchAlgorithm, SketchKind, SketchParams, SketchQuery as PostAsapSketchQuery, StatModelKind, - StatModelParams, SummaryExpr, SummaryFamilyType, SummaryField, SummaryInputExpr, SummaryNode, - SummarySchema, SummaryUpdate, ValueOperation, WaveletKind, WaveletParams, WeightDomain, + ExecutionTiming, Field, FieldDataType, GroupingStrategy, NonNegativeWeightProof, SamplingKind, + SamplingParams, Schema, SketchAlgorithm, SketchKind, SketchParams, + SketchQuery as PostAsapSketchQuery, StatModelKind, StatModelParams, SummaryExpr, + SummaryInputExpr, SummaryNode, SummaryUpdate, ValueOperation, WaveletKind, WaveletParams, + WeightDomain, }; use asap_types::post_asap::{AccuracyError, CompositionOperator, GuaranteeSource, ResultGuarantee}; use asap_types::pre_asap::agg_intent::{agg_is_mergeable, AggIntent}; @@ -365,7 +366,7 @@ use asap_types::pre_asap::query_expr::any_measure_filtered; use asap_types::pre_asap::query_expr::{ BinaryOpKind, Predicate, QueryExpr, QueryExprError, Reduction, }; -use asap_types::pre_asap::schema::{ColumnId, Schema}; +use asap_types::pre_asap::schema::ColumnId; use asap_types::types::AccuracyTarget; use asap_types::workload::{DataWorkload, QueryRecurrence, QueryWorkload, RepeatedDemand}; use std::rc::Rc; @@ -397,7 +398,7 @@ use crate::topk_reuse::TopKLimitReuseStrategy; /// to workload-wide orchestration. #[derive(Debug, Error)] pub enum RealizationError { - /// Schema derivation failed while lifting an edge to `SummarySchema`. + /// Schema derivation failed while lifting an edge to `Schema`. #[error("schema derivation failed during pre-ASAP → post-ASAP binding: {0}")] Schema(#[from] QueryExprError), /// The candidate is accuracy-illegal (issue #172): its composed @@ -1346,7 +1347,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { || partition_by.is_without() || !schema.has_promql_series_identity() || !schema - .columns + .fields .get(value) .is_some_and(|column| column.name == "value") || !is_current_series_source(child) @@ -1511,7 +1512,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { else { continue; }; - let family = SummaryFamilyType::Sketch(kind.clone(), GroupingStrategy::default()); + let family = FieldDataType::Sketch(kind.clone(), GroupingStrategy::default()); let Some(child) = aggregate_child(root) else { continue; }; @@ -2170,7 +2171,7 @@ fn finalize_exact_accumulator_at( let is_exact_state = matches!( node.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(..), + family: FieldDataType::ExactAggregate(..), .. } ); @@ -2262,7 +2263,7 @@ fn ddsketch_quantile_alpha(node: &SummaryNode) -> Option { return None; }; let SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + family: FieldDataType::Sketch(kind, _), .. } = &summary_input.expr else { @@ -2312,10 +2313,7 @@ fn realize_binary_operand( if is_promql_scalar(operand) { return Ok(Rc::new(SummaryNode { expr: SummaryExpr::KeepPreAsap(Rc::clone(operand)), - schema: SummarySchema { - fields: Vec::new(), - time_index: None, - }, + schema: Schema::lifted(Vec::new(), None), guarantee: Some(ResultGuarantee::exact("PromQL scalar")), })); } @@ -2336,8 +2334,8 @@ fn override_accuracy(intent: &AggIntent, target: &AccuracyTarget) -> AggIntent { out } -/// Wrap an unrewritten pre-ASAP sub-DAG, lifting its schema with every column -/// `SummaryFamilyType::Plain`. `pub` so a caller can fall back to this +/// Wrap an unrewritten pre-ASAP sub_dag, lifting its schema with every column +/// `FieldDataType::Plain`. `pub` so a caller can fall back to this /// explicitly — e.g. when `SketchAlgorithmStrategy::replacements()` returns no /// candidate for a target, or a deployment wants to force a node its own /// runtime can't actually implement — through the same fallback this @@ -2584,20 +2582,18 @@ fn is_snapshot_weighted_topk(intent: &AggIntent, child: &QueryExpr) -> bool { /// Every family's partial state needs a readout to recover a value, except /// `ExactAggregate` — its partial state *is* the value already, so no /// estimate step follows it. -fn summary_family(realization: Realization) -> Option<(SummaryFamilyType, bool)> { +fn summary_family(realization: Realization) -> Option<(FieldDataType, bool)> { Some(match realization { Realization::ExactAggregate { kind, params } => { - (SummaryFamilyType::ExactAggregate(kind, params), false) + (FieldDataType::ExactAggregate(kind, params), false) } Realization::Sketch(kind) => ( - SummaryFamilyType::Sketch(kind, GroupingStrategy::default()), + FieldDataType::Sketch(kind, GroupingStrategy::default()), true, ), - Realization::Sample { kind, params } => (SummaryFamilyType::Sample(kind, params), true), - Realization::Wavelet { kind, params } => (SummaryFamilyType::Wavelet(kind, params), true), - Realization::StatModel { kind, params } => { - (SummaryFamilyType::StatModel(kind, params), true) - } + Realization::Sample { kind, params } => (FieldDataType::Sample(kind, params), true), + Realization::Wavelet { kind, params } => (FieldDataType::Wavelet(kind, params), true), + Realization::StatModel { kind, params } => (FieldDataType::StatModel(kind, params), true), Realization::PassThrough => return None, }) } @@ -2617,12 +2613,8 @@ enum PhysicalSummaryInputRuleResult { Unsupported(&'static str), } -type PhysicalSummaryInputRule = fn( - &AggIntent, - &SummaryFamilyType, - &Reduction, - &Rc, -) -> PhysicalSummaryInputRuleResult; +type PhysicalSummaryInputRule = + fn(&AggIntent, &FieldDataType, &Reduction, &Rc) -> PhysicalSummaryInputRuleResult; /// Ordered physical-realization rules for realizations that consume more /// than the immediate logical input. New composite primitives add a rule here @@ -2637,13 +2629,13 @@ const PHYSICAL_SUMMARY_INPUT_RULES: &[PhysicalSummaryInputRule] = &[ fn realize_value_frequency_summary_input( intent: &AggIntent, - family: &SummaryFamilyType, + family: &FieldDataType, _reduction: &Reduction, child: &Rc, ) -> PhysicalSummaryInputRuleResult { // Frequency counts hash sample values as items but add one per observation. // Using the sample as a weight would turn counts into sums and admit signed CMS updates. - if !matches!(family, SummaryFamilyType::Sketch(kind, _) + if !matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon || (matches!(intent, AggIntent::Count { .. }) && matches!(kind.algorithm(), SketchAlgorithm::Cms | SketchAlgorithm::CountSketch))) @@ -2677,7 +2669,7 @@ fn realize_value_frequency_summary_input( fn realize_physical_summary_input( intent: &AggIntent, - family: &SummaryFamilyType, + family: &FieldDataType, reduction: &Reduction, child: &Rc, ) -> Result { @@ -2748,7 +2740,7 @@ fn maintenance_exact_values(node: Rc) -> Option> { } if matches!( child.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(..), + family: FieldDataType::ExactAggregate(..), .. } ) => @@ -2774,7 +2766,7 @@ fn construct_summary_agg( reduction: &Reduction, intent: &AggIntent, input: PhysicalSummaryInput, - family: SummaryFamilyType, + family: FieldDataType, estimate: bool, planning_inputs: CandidatePlanningInputs<'_>, child_target: Option<&AccuracyTarget>, @@ -2786,7 +2778,7 @@ fn construct_summary_agg( let keyed_heap = input.input.item.is_some() && matches!( &family, - SummaryFamilyType::Sketch(kind, _) + FieldDataType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap) ); let snapshot_weighted = matches!(node, QueryExpr::Aggregate { child, .. } @@ -2799,7 +2791,7 @@ fn construct_summary_agg( "invalid weighted TopK distinct-item bound", )); } - if let (Some(n), SummaryFamilyType::Sketch(kind, grouping)) = (bound, &family) { + if let (Some(n), FieldDataType::Sketch(kind, grouping)) = (bound, &family) { let (eps, delta) = accuracy_budget(accuracy_target(intent).expect("TopK target")); let params = default_size_params( kind.algorithm().clone(), @@ -2807,7 +2799,7 @@ fn construct_summary_agg( eps, delta / (2.0 * n as f64), ); - family = SummaryFamilyType::Sketch( + family = FieldDataType::Sketch( SketchKind::new(kind.algorithm().clone(), params), grouping.clone(), ); @@ -2838,7 +2830,7 @@ fn construct_summary_agg( RealizationError::PhysicalRealization("invalid TopK partition key"), )?; let matches = source - .columns + .fields .iter() .enumerate() .filter(|(_, column)| column_ref(column) == reference) @@ -2876,7 +2868,7 @@ fn construct_summary_agg( let summary_input = input.input; let query = estimate.then(|| { if snapshot_weighted { - if let SummaryFamilyType::Sketch(kind, _) = &family { + if let FieldDataType::Sketch(kind, _) = &family { let capacity = match kind.params() { SketchParams::CmsWithHeap { heap_size, .. } | SketchParams::CountSketchWithHeap { heap_size, .. } => *heap_size, @@ -2900,13 +2892,10 @@ fn construct_summary_agg( Vec::new() }; fields.push(state); - state_schema = SummarySchema { - fields, - time_index: None, - }; + state_schema = Schema::lifted(fields, None); } else if let Some(field) = state_schema.fields.get_mut(state_idx) { field.dtype = family.clone(); - if matches!(&family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) + if matches!(&family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) { // State identity is independent of which statistic reads it. field.name = "univmon".into(); @@ -2917,7 +2906,7 @@ fn construct_summary_agg( // column it summarizes, not after the query's output column. let child_schema = input.child.output_schema()?; if let Ok(i) = resolve_column_ref(col, &child_schema) { - field.name = child_schema.columns[i].name.clone(); + field.name = child_schema.fields[i].name.clone(); } } } @@ -3100,7 +3089,7 @@ fn construct_summary_agg( fn keyed_heap_readout_schema( input: &PhysicalSummaryInput, node: &QueryExpr, -) -> Result { +) -> Result { let source = input.child.output_schema()?; let mut refs = Vec::new(); let QueryExpr::Aggregate { @@ -3141,7 +3130,7 @@ fn keyed_heap_readout_schema( "dynamic label identity requires an explicit row representation", )); } - for (index, column) in schema.columns.iter().enumerate() { + for (index, column) in schema.fields.iter().enumerate() { if Some(index) != schema.time_index && column.name != "value" { let reference = match &column.table { Some(table) => ColumnRef::Qualified { @@ -3175,10 +3164,10 @@ fn keyed_heap_readout_schema( &source, &mut refs, )?; - let mut fields = Vec::::new(); + let mut fields = Vec::::new(); for reference in refs { let matches: Vec<_> = source - .columns + .fields .iter() .filter(|column| match &reference { ColumnRef::Named(name) => &column.name == name, @@ -3200,40 +3189,33 @@ fn keyed_heap_readout_schema( "heap keys must have distinct output names", )); } - fields.push(asap_types::post_asap::SummaryField { - name: column.name.clone(), - dtype: SummaryFamilyType::Plain(column.dtype.clone()), - nullable: column.nullable, - }); + fields.push(Field::new( + column.name.clone(), + column.dtype.clone(), + column.nullable, + )); } if fields.is_empty() { return Err(RealizationError::PhysicalRealization( "heap readout has no identity columns", )); } - fields.push(asap_types::post_asap::SummaryField { - name: "__asap_estimate".into(), - dtype: SummaryFamilyType::Plain(asap_types::pre_asap::DataType::Float64), - nullable: false, - }); - Ok(SummarySchema { - fields, - time_index: None, - }) + fields.push(Field::new( + "__asap_estimate", + FieldDataType::Plain(asap_types::pre_asap::DataType::Float64), + false, + )); + Ok(Schema::lifted(fields, None)) } -fn ranking_score_index( - logical: &QueryExpr, - values: &SummarySchema, -) -> Result { +fn ranking_score_index(logical: &QueryExpr, values: &Schema) -> Result { if is_current_series_source(logical) { return values .fields .iter() .position(|field| { field.name == "value" - && field.dtype - == SummaryFamilyType::Plain(asap_types::pre_asap::DataType::Float64) + && field.dtype == FieldDataType::Plain(asap_types::pre_asap::DataType::Float64) }) .ok_or(RealizationError::PhysicalRealization( "snapshot ranking requires the sample value column", @@ -3273,7 +3255,7 @@ fn ranking_score_index( || !values.fields.get(index).is_some_and(|field| { matches!( field.dtype, - SummaryFamilyType::Plain( + FieldDataType::Plain( asap_types::pre_asap::DataType::Int64 | asap_types::pre_asap::DataType::Float64 ) ) @@ -3290,12 +3272,12 @@ fn ranking_score_index( /// The rate window is preserved; raw counter samples never become CMS weights. fn realize_counter_value_summary_input( intent: &AggIntent, - family: &SummaryFamilyType, + family: &FieldDataType, output_reduction: &Reduction, child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) - || !matches!(family, SummaryFamilyType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap)) + || !matches!(family, FieldDataType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap)) || !matches!(child.as_ref(), QueryExpr::Aggregate { reduction: Reduction::PerEntity, measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])) { return PhysicalSummaryInputRuleResult::NotApplicable; @@ -3323,7 +3305,7 @@ fn realize_counter_value_summary_input( // Retain the evaluation timestamp in each returned row. This sketch is a // snapshot, not an additive history of successive rate evaluations. let items = schema - .columns + .fields .iter() .enumerate() .filter(|(index, column)| column.name != "value" && !groups.contains(index)) @@ -3348,14 +3330,14 @@ fn realize_counter_value_summary_input( /// Rebuild the state for each evaluation; historical samples are not updates. fn realize_current_series_summary_input( intent: &AggIntent, - family: &SummaryFamilyType, + family: &FieldDataType, output_reduction: &Reduction, child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) || !is_current_series_source(child) { return PhysicalSummaryInputRuleResult::NotApplicable; } - let SummaryFamilyType::Sketch(kind, _) = family else { + let FieldDataType::Sketch(kind, _) = family else { return PhysicalSummaryInputRuleResult::NotApplicable; }; match kind.algorithm() { @@ -3383,7 +3365,7 @@ fn realize_current_series_summary_input( ); }; let items = schema - .columns + .fields .iter() .enumerate() .filter(|(index, column)| column.name != "value" && !groups.contains(index)) @@ -3409,14 +3391,14 @@ fn realize_current_series_summary_input( /// it does not consume an independently materialized Count result. fn realize_keyed_additive_summary_input( intent: &AggIntent, - family: &SummaryFamilyType, + family: &FieldDataType, output_reduction: &Reduction, child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) { return PhysicalSummaryInputRuleResult::NotApplicable; } - let SummaryFamilyType::Sketch(kind, _) = family else { + let FieldDataType::Sketch(kind, _) = family else { return PhysicalSummaryInputRuleResult::NotApplicable; }; let heap_algorithm = kind.algorithm(); @@ -3531,7 +3513,7 @@ fn realize_keyed_additive_summary_input( fn schema_column_ref(child: &QueryExpr, index: usize) -> Option { let schema = child.output_schema().ok()?; - let column = schema.columns.get(index)?; + let column = schema.fields.get(index)?; Some(match &column.table { Some(table) => ColumnRef::Qualified { table: table.clone(), @@ -3551,7 +3533,7 @@ fn schema_column_ref(child: &QueryExpr, index: usize) -> Option { /// (an approximate family the model has no local guarantee for, over an /// exact child) — unknown, never exact. fn compose_guarantee( - family: &SummaryFamilyType, + family: &FieldDataType, query: Option<&PostAsapSketchQuery>, child: &SummaryNode, intent: &AggIntent, @@ -3560,7 +3542,7 @@ fn compose_guarantee( allocation: Option, ) -> Result, AccuracyError> { let (op, local) = match (family, query) { - (SummaryFamilyType::ExactAggregate(kind, _), _) => { + (FieldDataType::ExactAggregate(kind, _), _) => { let op = match kind { // A row count does not depend on the rows' values: exact // regardless of the child's own error. @@ -3638,10 +3620,10 @@ fn summary_col_index(out_schema: &Schema, reduction: &Reduction, measures: usize match reduction { Reduction::PerEntity => out_schema .column_id("value") - .or_else(|| (0..out_schema.columns.len()).find(|&i| Some(i) != out_schema.time_index)) + .or_else(|| (0..out_schema.fields.len()).find(|&i| Some(i) != out_schema.time_index)) .unwrap_or(0), Reduction::Reduce(keys) if keys.is_without() => { - out_schema.columns.len().saturating_sub(measures) + out_schema.fields.len().saturating_sub(measures) } Reduction::Reduce(keys) => keys.len(), } @@ -3655,14 +3637,14 @@ fn summarised_column(intent: &AggIntent, child_schema: &Schema) -> ColumnRef { match intent .input_cols() .first() - .and_then(|id| child_schema.columns.get(*id)) + .and_then(|id| child_schema.fields.get(*id)) { Some(c) => column_ref(c), None => ColumnRef::SampleValue, } } -fn column_ref(column: &asap_types::pre_asap::Column) -> ColumnRef { +fn column_ref(column: &Field) -> ColumnRef { match &column.table { Some(t) => ColumnRef::Qualified { table: t.clone(), @@ -3693,7 +3675,7 @@ fn summarised_input( } let legs = cols .iter() - .map(|id| child_schema.columns.get(*id).map(column_ref)) + .map(|id| child_schema.fields.get(*id).map(column_ref)) .collect::>>() .ok_or(RealizationError::PhysicalRealization( "a tuple column is outside the input schema", @@ -3738,22 +3720,11 @@ fn readout( } } -/// Lift a pre-ASAP [`Schema`] to a [`SummarySchema`] with every column -/// `SummaryFamilyType::Plain` — shared by [`construct_summary_agg`] and +/// Lift a pre-ASAP [`Schema`] to a [`Schema`] with every column +/// `FieldDataType::Plain` — shared by [`construct_summary_agg`] and /// [`keep_pre_asap`], both in this module. -fn lift(schema: &Schema) -> SummarySchema { - SummarySchema { - fields: schema - .columns - .iter() - .map(|c| SummaryField { - name: c.name.clone(), - dtype: SummaryFamilyType::Plain(c.dtype.clone()), - nullable: c.nullable, - }) - .collect(), - time_index: schema.time_index, - } +fn lift(schema: &Schema) -> Schema { + Schema::lifted(schema.fields.clone(), schema.time_index) } // ── SharedSubDAGStrategy ──────────────────────────────────────────────── @@ -5026,7 +4997,7 @@ fn sketch_kind_of(node: &SummaryNode) -> Option { match &node.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_kind_of(summary_input), SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + family: FieldDataType::Sketch(kind, _), .. } => Some(kind.algorithm().clone()), _ => None, @@ -5274,8 +5245,8 @@ impl<'a> GlobalSelection<'a> { pred, } = target.as_ref() { - let left_width = left.output_schema()?.columns.len(); - let total_width = left_width + right.output_schema()?.columns.len(); + let left_width = left.output_schema()?.fields.len(); + let total_width = left_width + right.output_schema()?.fields.len(); let normalized_pred = matches!(kind, asap_types::pre_asap::JoinKind::Inner) .then(|| normalize_cross_input_equi_predicate(pred, left_width, total_width)) .flatten(); @@ -6995,7 +6966,7 @@ mod tests { agg_is_exact, default_cardinality, default_quantile, MathFunc, TimeFunc, }; use asap_types::pre_asap::query_expr::{Reduction as ReductionTy, Source}; - use asap_types::pre_asap::schema::{Column, DataType, Schema as SchemaTy}; + use asap_types::pre_asap::schema::{DataType, Field, Schema as SchemaTy}; use asap_types::types::AccuracyTarget; use std::collections::HashMap; @@ -7053,7 +7024,7 @@ mod tests { .unwrap(); let is_exact = |node: &SummaryNode, kind: ExactKind| { matches!(&node.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(k, _), .. + family: FieldDataType::ExactAggregate(k, _), .. } if *k == kind) }; assert!(inventory.candidates.iter().any(|forest| { @@ -7091,7 +7062,7 @@ mod tests { .schema .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_))), + .all(|field| matches!(field.dtype, FieldDataType::Plain(_))), "direct candidate {query} leaks state" ); } @@ -7111,7 +7082,7 @@ mod tests { node.schema .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_))), + .all(|field| matches!(field.dtype, FieldDataType::Plain(_))), "{query}: query root leaks state: {:?}", node.schema ); @@ -7954,10 +7925,14 @@ mod tests { fn metric_scan(labels: &[&str]) -> QueryExpr { let mut columns = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ]; - columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); + columns.extend( + labels + .iter() + .map(|n| Field::plain(*n, DataType::Utf8, true)), + ); QueryExpr::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], @@ -8247,7 +8222,7 @@ mod tests { ); } - /// The `SummaryFamilyType`'s committed `SketchAlgorithm`, from the top + /// The `FieldDataType`'s committed `SketchAlgorithm`, from the top /// `SummaryAgg` reachable under a (possibly `SummaryEstimate`-wrapped) /// bound root. fn summary_family_algorithm(node: &SummaryNode) -> SketchAlgorithm { @@ -8256,9 +8231,7 @@ mod tests { summary_family_algorithm(summary_input) } asap_types::post_asap::SummaryExpr::SummaryAgg { family, .. } => match family { - asap_types::post_asap::SummaryFamilyType::Sketch(kind, _) => { - kind.algorithm().clone() - } + asap_types::post_asap::FieldDataType::Sketch(kind, _) => kind.algorithm().clone(), other => panic!("expected a Sketch family, got {other:?}"), }, other => panic!("expected SummaryAgg/SummaryEstimate, got {other:?}"), @@ -9843,7 +9816,7 @@ mod tests { } } - fn field<'a>(schema: &'a SummarySchema, name: &str) -> &'a SummaryField { + fn field<'a>(schema: &'a Schema, name: &str) -> &'a Field { schema .fields .iter() @@ -9880,11 +9853,11 @@ mod tests { // Estimate edge: plain row shape — group key + Float64 answer. assert_eq!( field(&root.schema, "quantile_0_99").dtype, - SummaryFamilyType::Plain(DataType::Float64) + FieldDataType::Plain(DataType::Float64) ); assert_eq!( field(&root.schema, "job").dtype, - SummaryFamilyType::Plain(DataType::Utf8) + FieldDataType::Plain(DataType::Utf8) ); let SummaryExpr::SummaryAgg { @@ -9899,7 +9872,7 @@ mod tests { }; assert_eq!( family, - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) @@ -9910,7 +9883,7 @@ mod tests { // the committed family. assert_eq!( field(&summary_input.schema, "value").dtype, - SummaryFamilyType::Sketch( + FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) @@ -9954,7 +9927,7 @@ mod tests { }; assert!(matches!( family, - SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll + FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll )); // With `PreferDDSketchViaCostModel`: DDSketch instead, same query. @@ -9967,7 +9940,7 @@ mod tests { }; assert_eq!( family, - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::DDSketch, SketchParams::DDSketch { alpha: 0.01 } @@ -10063,7 +10036,7 @@ mod tests { }; assert_eq!( family, - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::CountSketch, SketchParams::CountSketch { @@ -10088,11 +10061,11 @@ mod tests { }; assert_eq!( family, - &SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + &FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) ); assert_eq!( field(&root.schema, "sum").dtype, - SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) ); } @@ -10114,7 +10087,7 @@ mod tests { }; assert_eq!( family, - &SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) + &FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) ); assert_eq!( root.schema @@ -10126,7 +10099,7 @@ mod tests { ); assert_eq!( field(&root.schema, "value").dtype, - SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) + FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) ); assert_eq!(root.schema.time_index, Some(0)); } @@ -10195,7 +10168,7 @@ mod tests { }; assert!(matches!( family, - SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll + FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll )); let SummaryExpr::ValueOperation { child, @@ -10215,7 +10188,7 @@ mod tests { }; assert_eq!( inner_family, - &SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + &FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) ); assert!(matches!(leaf.expr, SummaryExpr::KeepPreAsap(_))); } @@ -10401,7 +10374,7 @@ mod tests { fn propagation_stats( &self, op: &CompositionOperator, - _family: &SummaryFamilyType, + _family: &FieldDataType, _query: Option<&PostAsapSketchQuery>, ) -> PropagationStats { if matches!(op, CompositionOperator::TopKSelection) { @@ -10520,7 +10493,7 @@ mod tests { ); assert!(matches!( family, - SummaryFamilyType::Sketch(kind, _) + FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::CmsWithHeap )); assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(_))); @@ -10759,9 +10732,9 @@ mod tests { }, predicates: vec![], schema: SchemaTy { - columns: vec![ - Column::new("host", DataType::Utf8, false), - Column::new("bytes", DataType::Int64, false), + fields: vec![ + Field::plain("host", DataType::Utf8, false), + Field::plain("bytes", DataType::Int64, false), ], time_index: None, unique_keys: vec![], @@ -10793,7 +10766,7 @@ mod tests { impl AccuracyModel for RankAdditiveModel { fn local_guarantee( &self, - family: &SummaryFamilyType, + family: &FieldDataType, query: &PostAsapSketchQuery, ) -> Option { DefaultAccuracyModel.local_guarantee(family, query) @@ -11186,7 +11159,7 @@ mod tests { panic!("readout") }; let SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + family: FieldDataType::Sketch(kind, _), .. } = &summary_input.expr else { @@ -11288,14 +11261,14 @@ mod tests { .schema .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); } // A numeric group key must not be mistaken for the ranked aggregate score. #[test] fn ranking_uses_aggregate_output_position_not_first_numeric_column() { let logical = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["id"])); let mut values = lift(&logical.output_schema().unwrap()); - values.fields[0].dtype = SummaryFamilyType::Plain(DataType::Int64); + values.fields[0].dtype = FieldDataType::Plain(DataType::Int64); assert_eq!(ranking_score_index(&logical, &values).unwrap(), 1); } // A heap's key schema is derived from its encoded item, not all label columns. @@ -11305,7 +11278,7 @@ mod tests { let QueryExpr::Scan { schema, .. } = &mut raw else { unreachable!() }; - schema.columns[2].dtype = DataType::Int64; + schema.fields[2].dtype = FieldDataType::Plain(DataType::Int64); let node = agg( vec![], AggIntent::TopK { @@ -11335,7 +11308,7 @@ mod tests { ); assert_eq!( schema.fields[0].dtype, - SummaryFamilyType::Plain(DataType::Int64) + FieldDataType::Plain(DataType::Int64) ); } } diff --git a/crates/asap-aware-mapping/src/rewrite.rs b/crates/asap-aware-mapping/src/rewrite.rs index 83c73156b..94a3d638c 100644 --- a/crates/asap-aware-mapping/src/rewrite.rs +++ b/crates/asap-aware-mapping/src/rewrite.rs @@ -105,8 +105,8 @@ fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { let input_schema = child.output_schema().ok()?; let value_col = col .or_else(|| input_schema.column_id("value")) - .or_else(|| (0..input_schema.columns.len()).find(|i| !by.contains(i)))?; - if input_schema.columns.get(value_col)?.nullable { + .or_else(|| (0..input_schema.fields.len()).find(|i| !by.contains(i)))?; + if input_schema.fields.get(value_col)?.nullable { return None; } Some((by.keys().len(), *col)) @@ -158,7 +158,7 @@ pub(crate) fn temporal_average_components(root: &Rc) -> Option) -> Option QueryExpr { let mut columns = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ]; - columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); + columns.extend( + labels + .iter() + .map(|n| Field::plain(*n, DataType::Utf8, true)), + ); QueryExpr::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], @@ -631,7 +635,7 @@ mod tests { }; let rewritten_schema = rewritten.output_schema().unwrap(); assert_eq!(original_schema, rewritten_schema); - assert_eq!(rewritten_schema.columns[0].name, "avg_latency"); + assert_eq!(rewritten_schema.fields[0].name, "avg_latency"); } /// A grouped rewrite must preserve the aggregate's grouping-key metadata; @@ -702,9 +706,9 @@ mod tests { #[test] fn works_with_a_bound_column_not_just_the_sample_value() { let mut schema_cols = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("job", DataType::Utf8, true), - Column::new("bytes", DataType::Int64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("job", DataType::Utf8, true), + Field::plain("bytes", DataType::Int64, false), ]; let child = QueryExpr::Scan { source: Source::TimeSeries { metric: "m".into() }, @@ -731,9 +735,9 @@ mod tests { // too, and a bare (uncast) `Int64 / Int64` would type the avg // column `Int64` — this assertion is what would catch that // regression. - assert_eq!(rewritten_schema.columns, original_schema.columns); + assert_eq!(rewritten_schema.fields, original_schema.fields); assert_eq!( - rewritten_schema.columns.last().unwrap().dtype, + rewritten_schema.fields.last().unwrap().dtype, DataType::Float64 ); @@ -759,9 +763,9 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - Column::new("latency", DataType::Float64, true), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("latency", DataType::Float64, true), ], 0, vec![], @@ -879,6 +883,6 @@ mod tests { let Replacement::Rewrite(rewritten) = &candidate.replacement else { panic!("expected logical rewrite") }; - assert_eq!(rewritten.output_schema().unwrap().columns[1].name, "sum"); + assert_eq!(rewritten.output_schema().unwrap().fields[1].name, "sum"); } } diff --git a/crates/asap-aware-mapping/src/rollup.rs b/crates/asap-aware-mapping/src/rollup.rs index 3b6ef0f37..ab6eff3d9 100644 --- a/crates/asap-aware-mapping/src/rollup.rs +++ b/crates/asap-aware-mapping/src/rollup.rs @@ -395,7 +395,7 @@ fn build_rollup( mod tests { use super::*; use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{Column, DataType}; + use asap_types::pre_asap::schema::{DataType, Field}; use asap_types::types::AccuracyTarget; /// `[ts(0), value(1), job(2), region(3)]`. @@ -405,10 +405,10 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - Column::new("job", DataType::Utf8, true), - Column::new("region", DataType::Utf8, true), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("job", DataType::Utf8, true), + Field::plain("region", DataType::Utf8, true), ], 0, vec![], diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs index 33a7f85c8..bc1b7ddaf 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs @@ -8,8 +8,8 @@ use std::collections::{HashMap, HashSet}; use std::rc::Rc; use asap_types::post_asap::{ - BoundExpr, ErrorMetric, ExactKind, GuaranteeSource, ProbabilityExpr, ResultGuarantee, - SketchAlgorithm, SummaryExpr, SummaryFamilyType, SummaryMaintenanceLifecycle, + BoundExpr, ErrorMetric, ExactKind, FieldDataType, GuaranteeSource, ProbabilityExpr, + ResultGuarantee, SketchAlgorithm, SummaryExpr, SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryNode, SummaryWindowFramework, }; use asap_types::pre_asap::{ diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs index 99917cf20..00e2ce705 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs @@ -881,12 +881,12 @@ mod tests { use std::rc::Rc; use asap_types::post_asap::{ - EvaluationSchedule, ExactKind, ExactParams, GroupingStrategy, OutputRepresentation, - SummaryExpr, SummaryFamilyType, SummaryField, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, SummarySchema, + EvaluationSchedule, ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, + OutputRepresentation, Schema, SummaryExpr, SummaryMaintenanceLifecycle, + SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, }; use asap_types::pre_asap::{ - agg_intent::AggIntent, Column, ColumnRef, DataType, QueryExpr, Reduction, Schema, Source, + agg_intent::AggIntent, ColumnRef, DataType, QueryExpr, Reduction, Source, }; use asap_types::workload::{ DataWorkload, Evidence, EvidenceSource, Predictability, Query, QueryLanguage, @@ -2672,7 +2672,7 @@ mod tests { let nested = Rc::new(SummaryNode { expr: SummaryExpr::SummaryAgg { child, - family: SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count), + family: FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Wildcard), reduction: Reduction::by(vec![]), grouping: GroupingStrategy::PerSubpopulationInstance, @@ -3010,15 +3010,8 @@ mod tests { } fn summary_with_operations(merge: bool, subtract: bool, delete: bool) -> Rc { - let state_type = SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count); - let schema = SummarySchema { - fields: vec![SummaryField { - name: "count".into(), - dtype: state_type.clone(), - nullable: false, - }], - time_index: None, - }; + let state_type = FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count); + let schema = Schema::lifted(vec![Field::new("count", state_type.clone(), false)], None); let leaf = Rc::new(SummaryNode { expr: SummaryExpr::KeepPreAsap(Rc::new(QueryExpr::Scan { source: Source::TimeSeries { @@ -3027,8 +3020,8 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], 0, vec![], @@ -3116,7 +3109,7 @@ mod tests { outer: Rc::clone(left), inner: Rc::clone(right), key: ColumnRef::Wildcard, - family: SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count), + family: FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), }, schema: schema.clone(), guarantee: None, @@ -3150,14 +3143,14 @@ mod tests { vector_match: None, }, }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), - nullable: false, - }], - time_index: None, - }, + schema: Schema::lifted( + vec![Field::new( + "value", + FieldDataType::Plain(DataType::Float64), + false, + )], + None, + ), guarantee: Some(ResultGuarantee::exact("test binary")), }) } @@ -3201,8 +3194,8 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], 0, vec![], diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs index 867a60e58..c3c12c20b 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs @@ -148,7 +148,7 @@ impl SummaryWindowAccuracyEvidence { matches!( &assignment.summary.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::Kll ) @@ -162,7 +162,7 @@ impl SummaryWindowAccuracyEvidence { matches!( &assignment.summary.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate( + family: FieldDataType::ExactAggregate( ExactKind::Count | ExactKind::Sum, _ ), diff --git a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs index f0569d1ae..cf41efeff 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs @@ -1837,11 +1837,11 @@ mod tests { } use super::*; use asap_types::post_asap::{ - ExactKind, ExactParams, GroupingStrategy, PostAsapOperatorPayload, ResultGuarantee, - SketchAlgorithm, SummaryFamilyType, SummaryField, SummarySchema, + ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, PostAsapOperatorPayload, + ResultGuarantee, Schema, SketchAlgorithm, }; use asap_types::pre_asap::AggIntent; - use asap_types::pre_asap::{Column, ColumnRef, DataType, QueryExpr, Reduction, Schema, Source}; + use asap_types::pre_asap::{ColumnRef, DataType, QueryExpr, Reduction, Source}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ BatchEntry, DataWorkload, DurationMs, Evidence, EvidenceSource, Predictability, Query, @@ -2064,7 +2064,7 @@ mod tests { match &node.expr { SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_algorithm(summary_input), SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + family: FieldDataType::Sketch(kind, _), .. } => Some(kind.algorithm().clone()), _ => None, @@ -2083,8 +2083,8 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], 0, vec![], @@ -2121,13 +2121,10 @@ mod tests { fn summary() -> Rc { let child = Rc::new(SummaryNode { expr: SummaryExpr::KeepPreAsap(query_root()), - schema: SummarySchema { - fields: vec![], - time_index: None, - }, + schema: Schema::lifted(vec![], None), guarantee: Some(ResultGuarantee::exact("raw")), }); - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); Rc::new(SummaryNode { expr: SummaryExpr::SummaryAgg { child, @@ -2139,21 +2136,14 @@ mod tests { grouping: GroupingStrategy::default(), filter: None, }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "state".into(), - dtype: family, - nullable: false, - }], - time_index: None, - }, + schema: Schema::lifted(vec![Field::new("state", family, false)], None), guarantee: Some(ResultGuarantee::exact("sum")), }) } fn nested_summary() -> Rc { let child = summary(); - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); Rc::new(SummaryNode { expr: SummaryExpr::SummaryAgg { child, @@ -2165,14 +2155,7 @@ mod tests { grouping: GroupingStrategy::default(), filter: None, }, - schema: SummarySchema { - fields: vec![SummaryField { - name: "state".into(), - dtype: family, - nullable: false, - }], - time_index: None, - }, + schema: Schema::lifted(vec![Field::new("state", family, false)], None), guarantee: Some(ResultGuarantee::exact("nested sum")), }) } @@ -3274,10 +3257,13 @@ mod tests { operation: ValueOperation::FinalizeExactAccumulator, timing: ExecutionTiming::QueryTime, }, - schema: SummarySchema { - fields: vec![SummaryField { + schema: Schema { + closed: true, + unique_keys: vec![], + fields: vec![Field { + table: None, name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, }], time_index: None, diff --git a/crates/asap-physical-operators/src/capability.rs b/crates/asap-physical-operators/src/capability.rs index e2e69be06..459371dbd 100644 --- a/crates/asap-physical-operators/src/capability.rs +++ b/crates/asap-physical-operators/src/capability.rs @@ -10,14 +10,14 @@ //! owned by `binding`, which also validates schemas, expressions and inputs. use crate::Error; use planner_types::post_asap::{ - ExactKind, ExactParams, GroupingStrategy, SketchAlgorithm, SketchParams, SketchQuery, - SummaryFamilyType, SummaryUpdate, + ExactKind, ExactParams, FieldDataType, GroupingStrategy, SketchAlgorithm, SketchParams, + SketchQuery, SummaryUpdate, }; /// Check the same contract used by `create_planner_accumulator` before a plan /// is accepted. Execution timing is deliberately not a kernel property. pub fn validate_summary_kernel( - family: &SummaryFamilyType, + family: &FieldDataType, input: &SummaryUpdate, grouping: &GroupingStrategy, ) -> Result<(), String> { @@ -25,7 +25,7 @@ pub fn validate_summary_kernel( return Err("shared summary grouping has no registered kernel".into()); } let keyed = match family { - SummaryFamilyType::ExactAggregate(kind, params) => { + FieldDataType::ExactAggregate(kind, params) => { use ExactKind as K; use ExactParams as P; if !matches!( @@ -41,7 +41,7 @@ pub fn validate_summary_kernel( } input.item.is_some() } - SummaryFamilyType::Sketch(kind, layout) => { + FieldDataType::Sketch(kind, layout) => { if layout != grouping { return Err("Planner family and operator grouping disagree".into()); } @@ -138,9 +138,9 @@ pub(crate) fn is_unit_sample_frequency(update: &planner_types::post_asap::Summar ) } -pub fn validate_native_family(family: &SummaryFamilyType) -> Result<(), Error> { +pub fn validate_native_family(family: &FieldDataType) -> Result<(), Error> { use planner_types::post_asap::SketchAlgorithm as A; - if let SummaryFamilyType::Sketch(kind, grouping) = family { + if let FieldDataType::Sketch(kind, grouping) = family { // Plain Count-Min is native as stored state only: it merges and reads // its bare count, but the DAG does not build it from rows. if let (A::Cms, SketchParams::Cms { width, depth }) = (kind.algorithm(), kind.params()) { @@ -165,8 +165,8 @@ pub fn validate_native_family(family: &SummaryFamilyType) -> Result<(), Error> { } } match family { - SummaryFamilyType::ExactAggregate(..) => {} - SummaryFamilyType::Sketch(kind, _) + FieldDataType::ExactAggregate(..) => {} + FieldDataType::Sketch(kind, _) if matches!(kind.algorithm(), A::Kll | A::DDSketch | A::Hll) => {} _ => { return Err(Error::Invalid( @@ -185,16 +185,13 @@ pub fn validate_native_family(family: &SummaryFamilyType) -> Result<(), Error> { } /// A sketch readout is native only for the families Planner can read directly. -pub fn validate_sketch_readout( - family: &SummaryFamilyType, - query: &SketchQuery, -) -> Result<(), Error> { +pub fn validate_sketch_readout(family: &FieldDataType, query: &SketchQuery) -> Result<(), Error> { validate_native_family(family)?; use planner_types::post_asap::SketchAlgorithm as A; // A point count without an item value reads the total count. let bare_count = matches!(query, SketchQuery::PointCount { value: None, .. }); let supported = match family { - SummaryFamilyType::Sketch(kind, _) => match (kind.algorithm(), query) { + FieldDataType::Sketch(kind, _) => match (kind.algorithm(), query) { (A::Kll, SketchQuery::Quantile { q }) | (A::DDSketch, SketchQuery::Quantile { q }) => { if !(0.0..=1.0).contains(q) { return Err(Error::Invalid( @@ -223,7 +220,7 @@ pub fn validate_sketch_readout( /// An exact readout must match the exact family it reads. pub fn validate_exact_readout( - family: &SummaryFamilyType, + family: &FieldDataType, readout: &crate::summary_kernels::exact::ExactReadout, ) -> Result<(), Error> { validate_native_family(family)?; @@ -231,15 +228,12 @@ pub fn validate_exact_readout( use planner_types::post_asap::ExactKind as E; let supported = matches!( (family, readout.statistic), - (SummaryFamilyType::ExactAggregate(E::Sum, _), S::Sum) - | (SummaryFamilyType::ExactAggregate(E::Count, _), S::Count) - | (SummaryFamilyType::ExactAggregate(E::Min, _), S::Min) - | (SummaryFamilyType::ExactAggregate(E::Max, _), S::Max) - | (SummaryFamilyType::ExactAggregate(E::Rate, _), S::Rate) - | ( - SummaryFamilyType::ExactAggregate(E::Increase, _), - S::Increase - ) + (FieldDataType::ExactAggregate(E::Sum, _), S::Sum) + | (FieldDataType::ExactAggregate(E::Count, _), S::Count) + | (FieldDataType::ExactAggregate(E::Min, _), S::Min) + | (FieldDataType::ExactAggregate(E::Max, _), S::Max) + | (FieldDataType::ExactAggregate(E::Rate, _), S::Rate) + | (FieldDataType::ExactAggregate(E::Increase, _), S::Increase) ); if !supported { return Err(Error::Invalid( diff --git a/crates/asap-physical-operators/src/expressions/planner.rs b/crates/asap-physical-operators/src/expressions/planner.rs index 2130a2f70..0c33e5743 100644 --- a/crates/asap-physical-operators/src/expressions/planner.rs +++ b/crates/asap-physical-operators/src/expressions/planner.rs @@ -345,12 +345,12 @@ impl CompiledExpression { .fields .iter() .map(|field| { - let planner_types::post_asap::SummaryFamilyType::Plain(dtype) = &field.dtype else { + let planner_types::post_asap::FieldDataType::Plain(dtype) = &field.dtype else { return Err(Error::Invalid( "scalar expression cannot consume opaque summary state".into(), )); }; - Ok(planner_types::pre_asap::Column::new( + Ok(planner_types::pre_asap::Field::plain( field.name.clone(), dtype.clone(), field.nullable, @@ -378,15 +378,13 @@ impl CompiledExpression { "persisted expression type differs from its semantics".into(), )); } - if input.fields.len() != self.schema.columns.len() + if input.fields.len() != self.schema.fields.len() || input .fields .iter() - .zip(&self.schema.columns) + .zip(&self.schema.fields) .any(|(field, column)| { - field.dtype - != planner_types::post_asap::SummaryFamilyType::Plain(column.dtype.clone()) - || field.nullable != column.nullable + field.dtype != column.dtype.clone() || field.nullable != column.nullable }) { return Err(Error::Invalid( @@ -397,11 +395,13 @@ impl CompiledExpression { } /// Evaluate a row under the same typed schema used when binding the expression. pub fn evaluate(&self, row: &[Value]) -> Result { - if row.len() != self.schema.columns.len() - || row - .iter() - .zip(&self.schema.columns) - .any(|(value, column)| !value.matches(&column.dtype, column.nullable)) + if row.len() != self.schema.fields.len() + || row.iter().zip(&self.schema.fields).any(|(value, column)| { + !column + .dtype + .plain() + .is_some_and(|dtype| value.matches(dtype, column.nullable)) + }) { return Err(Error::Invalid( "expression input differs from its bound schema".into(), diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs index 1d694d9d4..09c0ac1a8 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -308,7 +308,7 @@ mod tests { values::Batch, }; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{Field, FieldDataType, Schema as LogicalSchema}, pre_asap::DataType, types::AccuracyTarget, }; @@ -317,16 +317,20 @@ mod tests { // The same window operator must give the same answer in either engine phase. #[test] fn temporal_windows_execute_in_both_phases_and_count_is_integer() { - let schema = Arc::new(SummarySchema { + let schema = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: vec![ - SummaryField { + Field { + table: None, name: "time".into(), - dtype: SummaryFamilyType::Plain(DataType::Timestamp), + dtype: FieldDataType::Plain(DataType::Timestamp), nullable: false, }, - SummaryField { + Field { + table: None, name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, }, ], diff --git a/crates/asap-physical-operators/src/operators/aligned_binary.rs b/crates/asap-physical-operators/src/operators/aligned_binary.rs index 941a9f61e..4f58886c4 100644 --- a/crates/asap-physical-operators/src/operators/aligned_binary.rs +++ b/crates/asap-physical-operators/src/operators/aligned_binary.rs @@ -22,9 +22,11 @@ impl Operator { )); } for (input, value) in [(&left, values.0), (&right, values.1)] { - if input.fields.get(value).is_none_or(|f| { - f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Float64) - }) { + if input + .fields + .get(value) + .is_none_or(|f| f.nullable || f.dtype != FieldDataType::Plain(DataType::Float64)) + { return Err(invalid( "aligned arithmetic requires non-null Float64 values", )); diff --git a/crates/asap-physical-operators/src/operators/common.rs b/crates/asap-physical-operators/src/operators/common.rs index 9f1709da4..fd237ccc0 100644 --- a/crates/asap-physical-operators/src/operators/common.rs +++ b/crates/asap-physical-operators/src/operators/common.rs @@ -2,16 +2,19 @@ use super::*; pub(super) fn invalid(message: &str) -> Error { Error::Invalid(message.into()) } -pub(super) fn schema(fields: Vec) -> Schema { - Arc::new(SummarySchema { +pub(super) fn schema(fields: Vec) -> Schema { + Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields, time_index: None, }) } -pub(super) fn result_field(name: &str, dtype: DataType, nullable: bool) -> SummaryField { - SummaryField { +pub(super) fn result_field(name: &str, dtype: DataType, nullable: bool) -> Field { + Field { + table: None, name: name.into(), - dtype: SummaryFamilyType::Plain(dtype), + dtype: FieldDataType::Plain(dtype), nullable, } } @@ -73,3 +76,4 @@ pub(super) fn key_bytes(key: &[Vec]) -> usize { .map(|part| std::mem::size_of::>() + part.len()) .sum::() } +use planner_types::pre_asap::Schema as LogicalSchema; diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index df6ae0796..fd5dac32d 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -8,7 +8,7 @@ use crate::{ }; use futures::StreamExt; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema, SummaryUpdate}, + post_asap::{Field, FieldDataType, SummaryUpdate}, pre_asap::{ColumnRef, DataType}, }; use std::{collections::BTreeMap, sync::Arc}; @@ -133,13 +133,13 @@ enum Kind { predicate: Box, }, SummaryBuild { - family: SummaryFamilyType, + family: FieldDataType, value: usize, time: Option, groups: Vec, }, KeyedSummaryBuild { - family: SummaryFamilyType, + family: FieldDataType, value: usize, items: Vec, groups: Vec, @@ -252,7 +252,7 @@ impl Operator { } if output.time_index.is_some_and(|i| { i >= output.fields.len() - || output.fields[i].dtype != SummaryFamilyType::Plain(DataType::Timestamp) + || output.fields[i].dtype != FieldDataType::Plain(DataType::Timestamp) }) { return Err(invalid("invalid output time column")); } diff --git a/crates/asap-physical-operators/src/operators/scope_timestamp.rs b/crates/asap-physical-operators/src/operators/scope_timestamp.rs index 3561d6a69..bdbd7b0d8 100644 --- a/crates/asap-physical-operators/src/operators/scope_timestamp.rs +++ b/crates/asap-physical-operators/src/operators/scope_timestamp.rs @@ -26,7 +26,7 @@ impl Operator { candidate.dtype == field.dtype && candidate.nullable == field.nullable && (candidate.name == field.name - || !matches!(field.dtype, SummaryFamilyType::Plain(_))) + || !matches!(field.dtype, FieldDataType::Plain(_))) }) .map(|(index, _)| index) .collect(); diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs index ef5c49cb5..b6fff4f39 100644 --- a/crates/asap-physical-operators/src/operators/summary/mod.rs +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -9,14 +9,14 @@ pub enum ReadoutQuery { impl Operator { pub fn keyed_summary_build( input: Schema, - family: SummaryFamilyType, + family: FieldDataType, value: usize, items: Vec, groups: Vec, ) -> Result { use crate::summary_kernels::weighted_frequency::WeightedFrequency; crate::values::validate_family(&family)?; - let SummaryFamilyType::Sketch(kind, _) = &family else { + let FieldDataType::Sketch(kind, _) = &family else { return Err(invalid("keyed sketch required")); }; WeightedFrequency::configuration(kind)?; @@ -43,7 +43,8 @@ impl Operator { .iter() .map(|&i| input.fields[i].clone()) .collect::>(); - fields.push(SummaryField { + fields.push(Field { + table: None, name: "state".into(), dtype: family.clone(), nullable: false, @@ -67,7 +68,7 @@ impl Operator { ) -> Result { use crate::summary_kernels::weighted_frequency::WeightedFrequency; crate::values::validate_family(&field(&input, state)?.dtype)?; - let SummaryFamilyType::Sketch(kind, _) = &field(&input, state)?.dtype else { + let FieldDataType::Sketch(kind, _) = &field(&input, state)?.dtype else { return Err(invalid("keyed readout requires summary state")); }; let (_, _, _, capacity) = WeightedFrequency::configuration(kind)?; @@ -76,7 +77,7 @@ impl Operator { } if state + 1 != input.fields.len() || output.fields[..state] != input.fields[..state] - || output.fields.last().unwrap().dtype != SummaryFamilyType::Plain(DataType::Float64) + || output.fields.last().unwrap().dtype != FieldDataType::Plain(DataType::Float64) { return Err(invalid( "keyed readout must preserve partitions and return a Float64 score", @@ -91,7 +92,7 @@ impl Operator { } pub fn summary_build( input: Schema, - family: SummaryFamilyType, + family: FieldDataType, value: usize, time: Option, groups: Vec, @@ -109,7 +110,7 @@ impl Operator { if time.is_none() && matches!( family, - SummaryFamilyType::ExactAggregate( + FieldDataType::ExactAggregate( planner_types::post_asap::ExactKind::Rate | planner_types::post_asap::ExactKind::Increase, _ @@ -128,7 +129,8 @@ impl Operator { .iter() .map(|&i| input.fields[i].clone()) .collect::>(); - fields.push(SummaryField { + fields.push(Field { + table: None, name: "state".into(), dtype: family.clone(), nullable: false, @@ -147,7 +149,7 @@ impl Operator { pub fn summary_merge(input: Schema, state: usize, groups: Vec) -> Result { validate_groups(&input, &groups)?; crate::values::validate_family(&field(&input, state)?.dtype)?; - if matches!(field(&input, state)?.dtype, SummaryFamilyType::Plain(_)) { + if matches!(field(&input, state)?.dtype, FieldDataType::Plain(_)) { return Err(invalid("summary state required")); } let mut fields = groups @@ -175,7 +177,7 @@ impl Operator { let mut fields = input.fields.clone(); let result_type = if matches!( fields[state].dtype, - SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) + FieldDataType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) ) || integral_count(family, &query) { DataType::Int64 @@ -187,7 +189,7 @@ impl Operator { let nullable = fields.len() == 1 && matches!( fields[state].dtype, - SummaryFamilyType::ExactAggregate( + FieldDataType::ExactAggregate( planner_types::post_asap::ExactKind::Min | planner_types::post_asap::ExactKind::Max, _ @@ -204,8 +206,8 @@ impl Operator { /// The Planner reads a Count-Min bare count only for count intents, whose /// output is Int64 and whose updates have unit weight; execution rejects a /// non-integral total rather than rounding it. -fn integral_count(family: &SummaryFamilyType, query: &ReadoutQuery) -> bool { - matches!(family, SummaryFamilyType::Sketch(kind, _) +fn integral_count(family: &FieldDataType, query: &ReadoutQuery) -> bool { + matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::Cms) && matches!( query, @@ -266,7 +268,7 @@ pub(super) fn execute<'a>( // The typed output schema restores epoch-millisecond // timestamp keys from the kernel's Int64 representation. for (value, field) in values.iter_mut().zip(&output.fields) { - if field.dtype == SummaryFamilyType::Plain(DataType::Timestamp) { + if field.dtype == FieldDataType::Plain(DataType::Timestamp) { if let Value::Int64(time) = value { *value = Value::Timestamp(*time); } @@ -296,7 +298,7 @@ pub(super) fn execute<'a>( .estimate(query) .map_err(|e| Error::Operator(e.to_string()))?; if output.fields[*state].dtype - == SummaryFamilyType::Plain(DataType::Int64) + == FieldDataType::Plain(DataType::Int64) { // Below 2^53 an f64 sum of unit updates is exact. if value.fract() != 0.0 || !(0.0..9.007_199_254_740_992e15).contains(&value) { @@ -314,7 +316,7 @@ pub(super) fn execute<'a>( .as_any() .downcast_ref::() .ok_or_else(|| invalid("exact readout requires exact state"))?; - if output.fields[*state].dtype == SummaryFamilyType::Plain(DataType::Int64) { + if output.fields[*state].dtype == FieldDataType::Plain(DataType::Int64) { let count = exact.count().ok_or_else(|| { Error::Operator("exact count state lacks an integer count".into()) })?; @@ -366,7 +368,7 @@ pub(super) fn execute_merge<'a>( async fn build_summary( mut input: Input<'_, Batch>, - family: &SummaryFamilyType, + family: &FieldDataType, value: usize, time: Option, groups: &[usize], @@ -397,7 +399,7 @@ async fn build_summary( } let ordered_time = matches!( family, - SummaryFamilyType::ExactAggregate( + FieldDataType::ExactAggregate( planner_types::post_asap::ExactKind::Rate | planner_types::post_asap::ExactKind::Increase, _ @@ -466,7 +468,7 @@ async fn merge_summary( groups: &[usize], context: &RunContext, ) -> Result>, Error> { - type GroupState = (Vec, SummaryFamilyType, Arc); + type GroupState = (Vec, FieldDataType, Arc); let mut states: BTreeMap>, GroupState> = BTreeMap::new(); let mut work = Cooperative::new(context); let mut memory = context.reserve(0)?; @@ -525,14 +527,14 @@ async fn merge_summary( async fn build_keyed_summary( mut input: Input<'_, Batch>, - family: &SummaryFamilyType, + family: &FieldDataType, value: usize, items: &[usize], groups: &[usize], context: &RunContext, ) -> Result>, Error> { use crate::{summary_kernels::weighted_frequency::WeightedFrequency, AggregateCore}; - let SummaryFamilyType::Sketch(kind, _) = family else { + let FieldDataType::Sketch(kind, _) = family else { unreachable!() }; let (algorithm, width, depth, capacity) = WeightedFrequency::configuration(kind)?; diff --git a/crates/asap-physical-operators/src/operators/vector_window.rs b/crates/asap-physical-operators/src/operators/vector_window.rs index 7592d0b54..1464fc330 100644 --- a/crates/asap-physical-operators/src/operators/vector_window.rs +++ b/crates/asap-physical-operators/src/operators/vector_window.rs @@ -1,13 +1,16 @@ //! Window bounds are typed input data; aggregation and histogram semantics stay native. use super::*; use planner_types::pre_asap::AggIntent; +use planner_types::pre_asap::Schema as LogicalSchema; pub(crate) fn matrix_schema() -> Schema { let mut fields = vector_binary::value_schema(false).fields.clone(); fields.insert(1, result_field("timestamp", DataType::Timestamp, false)); fields.push(result_field("window_start", DataType::Timestamp, false)); fields.push(result_field("window_end", DataType::Timestamp, false)); - Arc::new(SummarySchema { + Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields, time_index: Some(1), }) diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 3121ede05..c6d2cf0f7 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -10,8 +10,8 @@ use crate::{ }; use planner_types::{ post_asap::{ - ExactOperation, PostAsapDAG, PostAsapDAGNode, PostAsapOperatorPayload as Payload, - SketchQuery, SummaryFamilyType, SummaryInputExpr, ValueOperation, + ExactOperation, FieldDataType, PostAsapDAG, PostAsapDAGNode, + PostAsapOperatorPayload as Payload, SketchQuery, SummaryInputExpr, ValueOperation, }, pre_asap::{ AggIntent, ColumnRef, CompareOpKind, DataType, GroupKeys, QueryExpr, @@ -499,7 +499,7 @@ fn compile_internal( schema .fields .iter() - .any(|f| matches!(f.dtype, SummaryFamilyType::Plain(DataType::Map { .. }))) + .any(|f| matches!(f.dtype, FieldDataType::Plain(DataType::Map { .. }))) }; // Grouped rows carry their labels as columns; per-series rows // carry the series identity. @@ -534,8 +534,8 @@ fn compile_internal( .map_err(|error| invalid(format!("node {id}: {error}")))?; let actual = readout.schema(); let converted = actual.fields.iter().zip(&output.fields).position(|(a, d)| { - a.dtype == SummaryFamilyType::Plain(DataType::Int64) - && d.dtype == SummaryFamilyType::Plain(DataType::Float64) + a.dtype == FieldDataType::Plain(DataType::Int64) + && d.dtype == FieldDataType::Plain(DataType::Float64) }); if let Some(column) = converted { let columns = actual @@ -654,7 +654,7 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[Schema]) -> Result>(); @@ -834,7 +834,7 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[Schema]) -> Result match kind { + FieldDataType::ExactAggregate(kind, _) => match kind { E::Sum => S::Sum, E::Count => S::Count, E::Min => S::Min, @@ -877,7 +877,7 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[Schema]) -> Result Result { .fields .iter() .enumerate() - .filter(|(_, f)| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + .filter(|(_, f)| !matches!(f.dtype, FieldDataType::Plain(_))) .map(|(i, _)| i) .collect::>(); match columns.as_slice() { @@ -978,7 +978,7 @@ fn summary_column(input: &Schema) -> Result { } fn named_column(input: &Schema, column: &ColumnRef) -> Result { let name = match column { - // Executable SummarySchema retains column names, not table qualifiers. + // Executable Schema as LogicalSchema retains column names, not table qualifiers. // Frontend binding has resolved the qualifier; still reject ambiguous // names here rather than guessing a join side. ColumnRef::Named(name) | ColumnRef::Qualified { name, .. } => name.as_str(), @@ -1145,8 +1145,8 @@ fn semi_join_keys( /// Deployments may use these positions to bind their source columns. pub fn equijoin_keys( pred: &planner_types::pre_asap::Predicate, - left: &planner_types::post_asap::SummarySchema, - right: &planner_types::post_asap::SummarySchema, + left: &planner_types::post_asap::Schema, + right: &planner_types::post_asap::Schema, ) -> Result, Error> { let mut keys = Vec::new(); semi_join_keys(&pred.0, left.fields.len(), right.fields.len(), &mut keys)?; diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index bc76821d4..03a8047f1 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -2,30 +2,35 @@ use super::promql_rows::SERIES_IDENTITY_COLUMN as SERIES_IDENTITY; use super::*; use planner_types::{ - post_asap::{ExecutionTiming, GroupingStrategy, SummarySchema}, + post_asap::{ExecutionTiming, GroupingStrategy, Schema as LogicalSchema}, pre_asap::DataType, }; /// Physical rows carry the population and pane coordinate alongside the logical value. /// These fields preserve identities which are implicit in a stored summary instance. -pub fn population_schema(family: SummaryFamilyType) -> Schema { - Arc::new(SummarySchema { +pub fn population_schema(family: FieldDataType) -> Schema { + Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: vec![ - planner_types::post_asap::SummaryField { + planner_types::post_asap::Field { + table: None, name: "$population".into(), - dtype: SummaryFamilyType::Plain(DataType::Map { + dtype: FieldDataType::Plain(DataType::Map { key: Box::new(DataType::Utf8), value: Box::new(DataType::Utf8), value_nullable: false, }), nullable: false, }, - planner_types::post_asap::SummaryField { + planner_types::post_asap::Field { + table: None, name: "$window_end".into(), - dtype: SummaryFamilyType::Plain(DataType::Timestamp), + dtype: FieldDataType::Plain(DataType::Timestamp), nullable: false, }, - planner_types::post_asap::SummaryField { + planner_types::post_asap::Field { + table: None, name: "value".into(), dtype: family, nullable: false, @@ -43,7 +48,7 @@ pub fn population_schema(family: SummaryFamilyType) -> Schema { /// must be canonical (sorted, unique, no empty values), since they are the /// population identity: build rows with [`raw_sample_row`]. pub fn raw_sample_schema() -> Schema { - let mut schema = (*population_schema(SummaryFamilyType::Plain(DataType::Float64))).clone(); + let mut schema = (*population_schema(FieldDataType::Plain(DataType::Float64))).clone(); schema.fields[1].name = "$timestamp".into(); Arc::new(schema) } @@ -101,11 +106,11 @@ pub fn boundary_schema(node: &PostAsapDAGNode) -> Result { .iter() .enumerate() .all(|(i, field)| match &field.dtype { - SummaryFamilyType::Plain(DataType::Timestamp) => { + FieldDataType::Plain(DataType::Timestamp) => { Some(i) == logical.time_index && !field.nullable } - SummaryFamilyType::Plain(DataType::Float64) => field.name == "value" && !field.nullable, - SummaryFamilyType::Plain(DataType::Utf8) => true, + FieldDataType::Plain(DataType::Float64) => field.name == "value" && !field.nullable, + FieldDataType::Plain(DataType::Utf8) => true, _ => false, }) && !logical @@ -123,18 +128,18 @@ pub fn boundary_schema(node: &PostAsapDAGNode) -> Result { } /// Validate the adapter layout during installed-plan recovery without lowering operators. -pub fn source_schema(logical: &SummarySchema) -> Result { +pub fn source_schema(logical: &LogicalSchema) -> Result { let states = logical .fields .iter() - .filter(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + .filter(|f| !matches!(f.dtype, FieldDataType::Plain(_))) .collect::>(); let [state] = states.as_slice() else { return Err(invalid( "stored population requires one typed summary state", )); }; - if logical.fields.iter().enumerate().any(|(i, field)| matches!(&field.dtype, SummaryFamilyType::Plain(dtype) + if logical.fields.iter().enumerate().any(|(i, field)| matches!(&field.dtype, FieldDataType::Plain(dtype) if field.nullable || !matches!(dtype, DataType::Utf8) && !(Some(i) == logical.time_index && *dtype == DataType::Timestamp))) { return Err(invalid("stored population metadata cannot reconstruct extra value columns")); } @@ -267,7 +272,7 @@ fn validate_value_output(node: &PostAsapDAGNode) -> Result<(), Error> { if identities.len() > 1 || identities .iter() - .any(|field| field.nullable || field.dtype != SummaryFamilyType::Plain(DataType::Utf8)) + .any(|field| field.nullable || field.dtype != FieldDataType::Plain(DataType::Utf8)) { return Err(invalid( "precompute series identity requires one non-null Utf8 column", @@ -279,11 +284,12 @@ fn validate_value_output(node: &PostAsapDAGNode) -> Result<(), Error> { .enumerate() .filter(|(i, field)| Some(*i) != schema.time_index && field.name != identity) .collect::>(); - if !matches!(values.as_slice(), [(_, field)] if !field.nullable && field.dtype == SummaryFamilyType::Plain(DataType::Float64)) + if !matches!(values.as_slice(), [(_, field)] if !field.nullable && field.dtype == FieldDataType::Plain(DataType::Float64)) || schema.time_index.is_some_and(|i| { - schema.fields.get(i).is_none_or(|f| { - f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Timestamp) - }) + schema + .fields + .get(i) + .is_none_or(|f| f.nullable || f.dtype != FieldDataType::Plain(DataType::Timestamp)) }) { return Err(invalid( @@ -343,13 +349,12 @@ fn fragment( }; validate_value_output(node)?; let statistic = match &input.fields[2].dtype { - SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Sum, _) => { + FieldDataType::ExactAggregate(planner_types::post_asap::ExactKind::Sum, _) => { crate::Statistic::Sum } - SummaryFamilyType::ExactAggregate( - planner_types::post_asap::ExactKind::Count, - _, - ) => crate::Statistic::Count, + FieldDataType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) => { + crate::Statistic::Count + } _ => { return Err(invalid( "precompute finalization requires explicit Sum or Count semantics", @@ -377,9 +382,7 @@ fn fragment( ), ], )? - .with_output_schema(population_schema(SummaryFamilyType::Plain( - DataType::Float64, - )))?; + .with_output_schema(population_schema(FieldDataType::Plain(DataType::Float64)))?; add(vec![read], project)? } Payload::SummaryAgg { @@ -403,7 +406,7 @@ fn fragment( // A unit-frequency summary (HLL) observes each raw sample value. let unit_frequency = raw && crate::capability::is_unit_sample_frequency(update) - && matches!(family, SummaryFamilyType::Sketch(kind, _) if !matches!( + && matches!(family, FieldDataType::Sketch(kind, _) if !matches!( kind.algorithm(), planner_types::post_asap::SketchAlgorithm::Cms | planner_types::post_asap::SketchAlgorithm::CountSketch @@ -431,7 +434,7 @@ fn fragment( )); } if keyed - && matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::CmsWithHeap) + && matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::CmsWithHeap) && !matches!( update.weight_domain, planner_types::post_asap::WeightDomain::NonNegative { .. } @@ -456,7 +459,7 @@ fn fragment( .filter(|field| { (raw || !field.nullable) && field.name != SERIES_IDENTITY - && field.dtype == SummaryFamilyType::Plain(DataType::Utf8) + && field.dtype == FieldDataType::Plain(DataType::Utf8) }) .map(|f| f.name.clone()) .ok_or_else(|| { @@ -476,7 +479,7 @@ fn fragment( SummaryInputExpr::Column(ColumnRef::SampleValue) => Expression::Column(2), SummaryInputExpr::Column(ColumnRef::Named(name)) if parents[0].output_schema.fields.iter().any(|f| { - f.name == *name && f.dtype == SummaryFamilyType::Plain(DataType::Float64) + f.name == *name && f.dtype == FieldDataType::Plain(DataType::Float64) }) => { Expression::Column(2) @@ -492,7 +495,7 @@ fn fragment( ("$window_end".into(), Expression::Column(1)), ("value".into(), Expression::FiniteFloat64(Box::new(weight))), ]; - let mut fields = population_schema(SummaryFamilyType::Plain(DataType::Float64)) + let mut fields = population_schema(FieldDataType::Plain(DataType::Float64)) .fields .clone(); if keyed { @@ -504,9 +507,10 @@ fn fragment( )?; for (index, (expression, dtype)) in items.into_iter().enumerate() { let name = format!("$item{index}"); - fields.push(planner_types::post_asap::SummaryField { + fields.push(planner_types::post_asap::Field { + table: None, name: name.clone(), - dtype: SummaryFamilyType::Plain(dtype), + dtype: FieldDataType::Plain(dtype), nullable: false, }); columns.push((name, expression)); @@ -514,7 +518,9 @@ fn fragment( } let item_columns = (3..fields.len()).collect::>(); let project = Operator::project(input.clone(), columns)?.with_output_schema( - Arc::new(SummarySchema { + Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields, time_index: Some(1), }), @@ -566,7 +572,7 @@ fn fragment( /// of the label set less excluded labels. fn raw_items( expr: &SummaryInputExpr, - scan: &SummarySchema, + scan: &LogicalSchema, items: &mut Vec<(Expression, DataType)>, ) -> Result<(), Error> { // Open PromQL scans need not list every label, so any name that is not @@ -575,7 +581,7 @@ fn raw_items( ColumnRef::Named(name) | ColumnRef::Qualified { name, .. } if !name.starts_with('$') && scan.fields.iter().all(|f| { - &f.name != name || f.dtype == SummaryFamilyType::Plain(DataType::Utf8) + &f.name != name || f.dtype == FieldDataType::Plain(DataType::Utf8) }) => { Some(name.clone()) diff --git a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs index 7f6ac49c1..bbc4d2986 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs @@ -439,7 +439,7 @@ impl Lowering { .fields .iter() .enumerate() - .filter(|(_, f)| f.dtype == SummaryFamilyType::Plain(DataType::Float64)) + .filter(|(_, f)| f.dtype == FieldDataType::Plain(DataType::Float64)) .map(|(i, _)| i) .collect::>(); let [value] = value.as_slice() else { diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index fba6060d4..70cacec3f 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -45,10 +45,7 @@ pub fn series_row( .enumerate() .map(|(index, field)| { if field.name == SERIES_IDENTITY_COLUMN { - if field.dtype != SummaryFamilyType::Plain(DataType::Utf8) - || field.nullable - || found - { + if field.dtype != FieldDataType::Plain(DataType::Utf8) || field.nullable || found { return Err(invalid("invalid series identity column")); } found = true; @@ -56,10 +53,10 @@ pub fn series_row( } else if Some(index) == schema.time_index { Ok(Value::Timestamp(timestamp)) } else if field.name == "value" - && field.dtype == SummaryFamilyType::Plain(DataType::Float64) + && field.dtype == FieldDataType::Plain(DataType::Float64) { Ok(Value::Float64(value)) - } else if field.dtype == SummaryFamilyType::Plain(DataType::Utf8) { + } else if field.dtype == FieldDataType::Plain(DataType::Utf8) { Ok(labels.get(&field.name).map_or_else( || Value::Utf8("".into()), |value| Value::Utf8(value.clone().into()), @@ -82,7 +79,7 @@ pub fn compile_current_series_readout( selected: &Rc, ) -> Result { use planner_types::post_asap::{ - compile_post_asap_dag, maintained_population::PopulationReadout, SummaryField, + compile_post_asap_dag, maintained_population::PopulationReadout, Field, }; let mut dag = compile_post_asap_dag(selected).map_err(|error| invalid(error.to_string()))?; // Typed snapshot candidates already carry full identity throughout the DAG. @@ -154,9 +151,10 @@ pub fn compile_current_series_readout( "current-series input already has a physical identity column", )); } - node.output_schema.fields.push(SummaryField { + node.output_schema.fields.push(Field { + table: None, name: SERIES_IDENTITY_COLUMN.into(), - dtype: SummaryFamilyType::Plain(DataType::Utf8), + dtype: FieldDataType::Plain(DataType::Utf8), nullable: false, }); } @@ -208,7 +206,7 @@ pub fn compile_rate_ranking( operation: ValueOperation::FinalizeExactAccumulator, timing: planner_types::post_asap::ExecutionTiming::QueryTime, } if matches!(&child.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), reduction: planner_types::pre_asap::Reduction::PerEntity, child: raw, .. } if matches!(&raw.expr, SummaryExpr::KeepPreAsap(expr) if matches!(expr.as_ref(), QueryExpr::TimeRange { .. }))) => @@ -263,7 +261,7 @@ pub fn compile_fixed_window_rate_aggregation( matches!( &n.payload, Payload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), reduction: planner_types::pre_asap::Reduction::PerEntity, .. } @@ -277,14 +275,14 @@ pub fn compile_fixed_window_rate_aggregation( n.output_state.timing == ExecutionTiming::IngestionTime && match &n.payload { Payload::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + family: FieldDataType::Sketch(kind, _), .. } => matches!( kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap ), Payload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + family: FieldDataType::ExactAggregate(ExactKind::Sum, _), .. } => true, _ => false, diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/asap-physical-operators/src/physical_planner/promql_values.rs index 0f0bcf0b4..73f86779c 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_values.rs @@ -209,8 +209,8 @@ pub fn compile_vector_to_scalar() -> Result { /// A stored exact-state input retains the complete population identity. The /// deployment supplies eligible panes; merging and finalization are computation. -pub fn exact_state_schema(family: SummaryFamilyType) -> Result { - if !matches!(family, SummaryFamilyType::ExactAggregate(..)) { +pub fn exact_state_schema(family: FieldDataType) -> Result { + if !matches!(family, FieldDataType::ExactAggregate(..)) { return Err(invalid("exact-state input requires an exact family")); } crate::values::validate_family(&family)?; @@ -221,13 +221,13 @@ pub fn exact_state_schema(family: SummaryFamilyType) -> Result { /// Retain exact readout semantics before any deployment state is opened. pub fn compile_exact_readout( - family: SummaryFamilyType, + family: FieldDataType, lookback_ms: u64, preserve_metric_name: bool, ) -> Result { use planner_types::post_asap::ExactKind; let statistic = match &family { - SummaryFamilyType::ExactAggregate(kind, _) => match kind { + FieldDataType::ExactAggregate(kind, _) => match kind { ExactKind::Sum => crate::Statistic::Sum, ExactKind::Count => crate::Statistic::Count, ExactKind::Min => crate::Statistic::Min, diff --git a/crates/asap-physical-operators/src/readout.rs b/crates/asap-physical-operators/src/readout.rs index e884ecc3f..e84d8c828 100644 --- a/crates/asap-physical-operators/src/readout.rs +++ b/crates/asap-physical-operators/src/readout.rs @@ -54,10 +54,10 @@ pub fn exact_readout( #[cfg(test)] mod counter_tests { use super::*; - use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; + use planner_types::post_asap::{ExactKind, ExactParams, FieldDataType}; fn counter(kind: ExactKind, params: ExactParams, keyed: bool) -> ExactAccumulator { - ExactAccumulator::new(SummaryFamilyType::ExactAggregate(kind, params), keyed).unwrap() + ExactAccumulator::new(FieldDataType::ExactAggregate(kind, params), keyed).unwrap() } // A counter population with a single sample is absent, keyed or not. diff --git a/crates/asap-physical-operators/src/runtime/batch_execution.rs b/crates/asap-physical-operators/src/runtime/batch_execution.rs index 3de02492b..d85b56a9c 100644 --- a/crates/asap-physical-operators/src/runtime/batch_execution.rs +++ b/crates/asap-physical-operators/src/runtime/batch_execution.rs @@ -88,7 +88,7 @@ mod tests { values::Value, }; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{Field, FieldDataType, Schema as LogicalSchema}, pre_asap::DataType, }; use std::sync::Arc; @@ -96,10 +96,13 @@ mod tests { // Engine adapters can run the identical native chain from an outer executor. #[test] fn same_native_chain_inside_query_and_ingestion_execution() { - let schema = Arc::new(SummarySchema { - fields: vec![SummaryField { + let schema = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], + fields: vec![Field { + table: None, name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, }], time_index: None, @@ -139,7 +142,9 @@ mod tests { // Native sources may cross the runtime's cooperative batch quantum. #[test] fn in_memory_source_drives_cooperative_yields() { - let schema = Arc::new(SummarySchema { + let schema = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: vec![], time_index: None, }); @@ -159,7 +164,9 @@ mod tests { // An adapter-held output must retain its parent's reservation after execution. #[test] fn returned_batches_keep_their_resource_reservation() { - let schema = Arc::new(SummarySchema { + let schema = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: vec![], time_index: None, }); @@ -188,7 +195,9 @@ mod tests { // A cancelled surrounding execution also prevents its native computation. #[test] fn cancellation_is_not_bypassed_by_in_memory_execution() { - let schema = Arc::new(SummarySchema { + let schema = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: vec![], time_index: None, }); diff --git a/crates/asap-physical-operators/src/sources/memory.rs b/crates/asap-physical-operators/src/sources/memory.rs index 7f77c54f2..52ffbc303 100644 --- a/crates/asap-physical-operators/src/sources/memory.rs +++ b/crates/asap-physical-operators/src/sources/memory.rs @@ -11,7 +11,7 @@ impl MemorySource { if schema .fields .iter() - .any(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + .any(|f| !matches!(f.dtype, FieldDataType::Plain(_))) { return Err(Error::Invalid( "raw source cannot contain summary states".into(), diff --git a/crates/asap-physical-operators/src/sources/mod.rs b/crates/asap-physical-operators/src/sources/mod.rs index 4da778b0c..dfe6b1f0d 100644 --- a/crates/asap-physical-operators/src/sources/mod.rs +++ b/crates/asap-physical-operators/src/sources/mod.rs @@ -8,7 +8,7 @@ use crate::{ }; use futures::{stream, StreamExt}; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{FieldDataType, Schema as LogicalSchema}, pre_asap::{DataType, QueryExpr, Source}, }; use std::sync::Arc; @@ -51,18 +51,10 @@ impl DataSources { "raw Scan requires a Planner Scan leaf".into(), )); }; - let output = Arc::new(SummarySchema { - fields: schema - .columns - .iter() - .map(|column| SummaryField { - name: column.name.clone(), - dtype: SummaryFamilyType::Plain(column.dtype.clone()), - nullable: column.nullable, - }) - .collect(), - time_index: schema.time_index, - }); + let output = Arc::new(LogicalSchema::lifted( + schema.fields.clone(), + schema.time_index, + )); crate::values::validate_schema(&output)?; let reader = self .sources diff --git a/crates/asap-physical-operators/src/summary_kernels/exact.rs b/crates/asap-physical-operators/src/summary_kernels/exact.rs index 84328cfd4..d4754686e 100644 --- a/crates/asap-physical-operators/src/summary_kernels/exact.rs +++ b/crates/asap-physical-operators/src/summary_kernels/exact.rs @@ -2,7 +2,7 @@ use super::increase::IncreaseAccumulator; use crate::Statistic; use crate::{AggregateCore, KeyByLabelValues, Measurement}; -use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; +use planner_types::post_asap::{ExactKind, ExactParams, FieldDataType}; use serde::{Deserialize, Serialize}; use std::collections::HashMap; @@ -26,14 +26,14 @@ enum ScalarState { #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(try_from = "ExactPayload")] pub struct ExactAccumulator { - family: SummaryFamilyType, + family: FieldDataType, scalar: ScalarState, keyed: Option>, } #[derive(Deserialize)] struct ExactPayload { - family: SummaryFamilyType, + family: FieldDataType, scalar: ScalarState, keyed: Option>, } @@ -128,21 +128,19 @@ impl ExactAccumulator { Ok(()) } - pub fn new(family: SummaryFamilyType, keyed: bool) -> Result { + pub fn new(family: FieldDataType, keyed: bool) -> Result { use ExactKind as K; use ExactParams as P; let scalar = match &family { - SummaryFamilyType::ExactAggregate(K::Sum, P::Sum) => ScalarState::Sum { + FieldDataType::ExactAggregate(K::Sum, P::Sum) => ScalarState::Sum { sum: 0.0, compensation: 0.0, }, - SummaryFamilyType::ExactAggregate(K::Count, P::Count) => ScalarState::Count(0), - SummaryFamilyType::ExactAggregate(K::Min, P::Min) => ScalarState::Min(None), - SummaryFamilyType::ExactAggregate(K::Max, P::Max) => ScalarState::Max(None), - SummaryFamilyType::ExactAggregate(K::Rate, P::Rate) - | SummaryFamilyType::ExactAggregate(K::Increase, P::Increase) => { - ScalarState::Counter(None) - } + FieldDataType::ExactAggregate(K::Count, P::Count) => ScalarState::Count(0), + FieldDataType::ExactAggregate(K::Min, P::Min) => ScalarState::Min(None), + FieldDataType::ExactAggregate(K::Max, P::Max) => ScalarState::Max(None), + FieldDataType::ExactAggregate(K::Rate, P::Rate) + | FieldDataType::ExactAggregate(K::Increase, P::Increase) => ScalarState::Counter(None), _ => return Err(format!("unsupported exact Planner family: {family:?}")), }; Ok(Self { @@ -152,7 +150,7 @@ impl ExactAccumulator { }) } - pub fn family(&self) -> &SummaryFamilyType { + pub fn family(&self) -> &FieldDataType { &self.family } pub(crate) fn insufficient_counter_samples( @@ -216,12 +214,12 @@ impl ExactAccumulator { fn statistic(&self) -> Statistic { match self.family { - SummaryFamilyType::ExactAggregate(ExactKind::Sum, _) => Statistic::Sum, - SummaryFamilyType::ExactAggregate(ExactKind::Count, _) => Statistic::Count, - SummaryFamilyType::ExactAggregate(ExactKind::Min, _) => Statistic::Min, - SummaryFamilyType::ExactAggregate(ExactKind::Max, _) => Statistic::Max, - SummaryFamilyType::ExactAggregate(ExactKind::Rate, _) => Statistic::Rate, - SummaryFamilyType::ExactAggregate(ExactKind::Increase, _) => Statistic::Increase, + FieldDataType::ExactAggregate(ExactKind::Sum, _) => Statistic::Sum, + FieldDataType::ExactAggregate(ExactKind::Count, _) => Statistic::Count, + FieldDataType::ExactAggregate(ExactKind::Min, _) => Statistic::Min, + FieldDataType::ExactAggregate(ExactKind::Max, _) => Statistic::Max, + FieldDataType::ExactAggregate(ExactKind::Rate, _) => Statistic::Rate, + FieldDataType::ExactAggregate(ExactKind::Increase, _) => Statistic::Increase, _ => unreachable!("validated exact family"), } } @@ -313,7 +311,7 @@ mod tests { #[derive(Serialize)] struct Payload { - family: SummaryFamilyType, + family: FieldDataType, scalar: ScalarState, keyed: Option>, } @@ -322,8 +320,8 @@ mod tests { rmp_serde::from_slice(&rmp_serde::to_vec_named(payload).unwrap()) } - fn sum() -> SummaryFamilyType { - SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + fn sum() -> FieldDataType { + FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) } // Stored Sum preserves low-order increments across updates, persistence and pane merge. @@ -402,7 +400,7 @@ mod tests { #[test] fn decode_rejects_unsupported_family() { let payload = Payload { - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Count), + family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Count), scalar: ScalarState::Sum { sum: 0.0, compensation: 0.0, diff --git a/crates/asap-physical-operators/src/summary_kernels/factory.rs b/crates/asap-physical-operators/src/summary_kernels/factory.rs index d44e64ab8..4be126d6e 100644 --- a/crates/asap-physical-operators/src/summary_kernels/factory.rs +++ b/crates/asap-physical-operators/src/summary_kernels/factory.rs @@ -6,7 +6,7 @@ use crate::summary_kernels::{ HydraKllSketchAccumulator, }; use crate::{AggregateCore, KeyByLabelValues}; -use planner_types::post_asap::{SketchAlgorithm, SketchParams, SummaryFamilyType}; +use planner_types::post_asap::{FieldDataType, SketchAlgorithm, SketchParams}; /// Generate the clone-based `AccumulatorUpdater` methods for updaters whose /// inner `acc` field implements `Clone + AggregateCore`. @@ -527,7 +527,7 @@ fn cms_heap_dims(params: &SketchParams) -> (usize, usize, usize) { /// Construct the kernel declared by a Planner SummaryAgg. No deployment config /// tags participate in this dispatch and unsupported payloads are errors. pub fn create_planner_accumulator( - family: &SummaryFamilyType, + family: &FieldDataType, input: &planner_types::post_asap::SummaryUpdate, grouping: &planner_types::post_asap::GroupingStrategy, ) -> Result, String> { @@ -548,7 +548,7 @@ pub fn create_planner_accumulator( if grouping != &GroupingStrategy::PerSubpopulationInstance { return Err("shared summary grouping requires a supported Planner Hydra kernel".into()); } - if matches!(family, SummaryFamilyType::ExactAggregate(..)) { + if matches!(family, FieldDataType::ExactAggregate(..)) { return Ok(Box::new(PlannerExactUpdater { acc: crate::summary_kernels::exact::ExactAccumulator::new( family.clone(), @@ -556,7 +556,7 @@ pub fn create_planner_accumulator( )?, })); } - let SummaryFamilyType::Sketch(kind, family_grouping) = family else { + let FieldDataType::Sketch(kind, family_grouping) = family else { return Err(format!("unsupported Planner summary family {family:?}")); }; if family_grouping != grouping { @@ -748,7 +748,7 @@ mod planner_parameter_regression { }, ), ] { - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new(algorithm.clone(), params), Default::default(), ); diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs index 1d36c4557..9476fcd10 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -2,11 +2,11 @@ use crate::AggregateCore; use crate::Error; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{Field, FieldDataType, Schema as LogicalSchema}, pre_asap::DataType, }; use std::{cmp::Ordering, sync::Arc}; -pub type Schema = Arc; +pub type Schema = Arc; #[derive(Clone, serde::Serialize, serde::Deserialize)] pub enum Value { Null, @@ -26,7 +26,7 @@ pub enum Value { Map(Arc<[(Value, Value)]>), #[serde(skip)] Summary { - family: SummaryFamilyType, + family: FieldDataType, state: Arc, }, } @@ -212,9 +212,7 @@ impl Batch { } for (value, field) in row.iter().zip(&schema.fields) { let matches = match (&field.dtype, value) { - (SummaryFamilyType::Plain(dtype), value) => { - value.matches(dtype, field.nullable) - } + (FieldDataType::Plain(dtype), value) => value.matches(dtype, field.nullable), (expected, Value::Summary { family, state }) => { expected == family && validate_state(family, state.as_ref()).is_ok() } @@ -260,7 +258,7 @@ pub(crate) fn group_key(row: &[Value], columns: &[usize]) -> Result> pub(crate) use crate::capability::validate_native_family as validate_family; -fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Result<(), Error> { +fn validate_state(family: &FieldDataType, state: &dyn AggregateCore) -> Result<(), Error> { use crate::summary_kernels::{ count_min_sketch::CountMinSketchAccumulator, datasketches_kll::DatasketchesKLLAccumulator, dd_sketch::DDSketchAccumulator, exact::ExactAccumulator, hll_sketch::HllSketchAccumulator, @@ -268,7 +266,7 @@ fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Resu use planner_types::post_asap::SketchParams; validate_family(family)?; let valid = match family { - SummaryFamilyType::Sketch(kind, _) + FieldDataType::Sketch(kind, _) if matches!( kind.params(), SketchParams::CmsWithHeap { .. } | SketchParams::CountSketchWithHeap { .. } @@ -284,11 +282,11 @@ fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Resu }) } - SummaryFamilyType::ExactAggregate(..) => state + FieldDataType::ExactAggregate(..) => state .as_any() .downcast_ref::() .is_some_and(|s| s.family() == family && !s.is_keyed()), - SummaryFamilyType::Sketch(kind, _) => match kind.params() { + FieldDataType::Sketch(kind, _) => match kind.params() { SketchParams::Kll { k } => state .as_any() .downcast_ref::() @@ -325,14 +323,14 @@ pub(crate) fn validate_schema(schema: &Schema) -> Result<(), Error> { schema .fields .get(index) - .is_none_or(|field| field.dtype != SummaryFamilyType::Plain(DataType::Timestamp)) + .is_none_or(|field| field.dtype != FieldDataType::Plain(DataType::Timestamp)) }) { return Err(Error::Invalid( "time index must name a Timestamp column".into(), )); } for field in &schema.fields { - if !matches!(field.dtype, SummaryFamilyType::Plain(_)) { + if !matches!(field.dtype, FieldDataType::Plain(_)) { validate_family(&field.dtype)?; if field.nullable { return Err(Error::Invalid( @@ -344,7 +342,7 @@ pub(crate) fn validate_schema(schema: &Schema) -> Result<(), Error> { Ok(()) } -pub(crate) fn field(schema: &Schema, column: usize) -> Result<&SummaryField, Error> { +pub(crate) fn field(schema: &Schema, column: usize) -> Result<&Field, Error> { schema .fields .get(column) @@ -352,7 +350,7 @@ pub(crate) fn field(schema: &Schema, column: usize) -> Result<&SummaryField, Err } pub(crate) fn plain(schema: &Schema, column: usize) -> Result<(&DataType, bool), Error> { let f = field(schema, column)?; - let SummaryFamilyType::Plain(dtype) = &f.dtype else { + let FieldDataType::Plain(dtype) = &f.dtype else { return Err(Error::Invalid("plain value required".into())); }; Ok((dtype, f.nullable)) @@ -367,7 +365,7 @@ mod weighted_state_tests { // A state cannot acquire a different family or shape merely by relabeling its batch. #[test] fn weighted_state_family_and_shape_must_match() { - let cms = SummaryFamilyType::Sketch( + let cms = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::CmsWithHeap, SketchParams::CmsWithHeap { @@ -378,7 +376,7 @@ mod weighted_state_tests { ), Default::default(), ); - let cs = SummaryFamilyType::Sketch( + let cs = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::CountSketchWithHeap, SketchParams::CountSketchWithHeap { @@ -395,7 +393,7 @@ mod weighted_state_tests { let wrong_shape = WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 64, 5, 8).unwrap(); assert!(validate_state(&cs, &wrong_shape).is_err()); - let even_depth = SummaryFamilyType::Sketch( + let even_depth = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::CountSketchWithHeap, SketchParams::CountSketchWithHeap { diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/asap-physical-operators/tests/blocking_resources.rs index 27ddd1095..7ba650211 100644 --- a/crates/asap-physical-operators/tests/blocking_resources.rs +++ b/crates/asap-physical-operators/tests/blocking_resources.rs @@ -8,17 +8,20 @@ use asap_physical_operators::{ }; use futures::{executor::block_on, FutureExt, StreamExt}; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{Field, FieldDataType, Schema as LogicalSchema}, pre_asap::{DataType, JoinKind, Predicate, QueryExpr, ScalarValue}, }; use std::sync::Arc; fn schema(width: usize) -> Schema { - Arc::new(SummarySchema { + Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: (0..width) - .map(|i| SummaryField { + .map(|i| Field { + table: None, name: format!("v{i}"), - dtype: SummaryFamilyType::Plain(DataType::Int64), + dtype: FieldDataType::Plain(DataType::Int64), nullable: false, }) .collect(), @@ -186,16 +189,20 @@ fn cooperative_sort_preserves_ties_across_chunks() { #[test] fn weighted_summary_build_yields_within_a_batch() { use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; - let input = Arc::new(SummarySchema { + let input = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: vec![ - SummaryField { + Field { + table: None, name: "item".into(), - dtype: SummaryFamilyType::Plain(DataType::Int64), + dtype: FieldDataType::Plain(DataType::Int64), nullable: false, }, - SummaryField { + Field { + table: None, name: "weight".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, }, ], @@ -216,7 +223,7 @@ fn weighted_summary_build_yields_within_a_batch() { Operator::source(input.clone(), vec![batch]).unwrap(), ) .unwrap(); - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::CmsWithHeap, SketchParams::CmsWithHeap { diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs index df0c17b95..b0b496df9 100644 --- a/crates/asap-physical-operators/tests/current_series_heap.rs +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -9,11 +9,14 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; +use planner_types::pre_asap::Schema as LogicalSchema; use planner_types::{post_asap::*, pre_asap::DataType}; use std::{collections::BTreeMap, sync::Arc}; -fn schema() -> Arc { - Arc::new(SummarySchema { +fn schema() -> Arc { + Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: [ ("ts", DataType::Timestamp), ("value", DataType::Float64), @@ -21,9 +24,10 @@ fn schema() -> Arc { (SERIES_IDENTITY_COLUMN, DataType::Utf8), ] .into_iter() - .map(|(name, dtype)| SummaryField { + .map(|(name, dtype)| Field { + table: None, name: name.into(), - dtype: SummaryFamilyType::Plain(dtype), + dtype: FieldDataType::Plain(dtype), nullable: false, }) .collect(), @@ -163,10 +167,11 @@ fn spatial_heap_ranks_latest_values_in_independent_runs() { heap_size: 100, }, }; - let family = - SummaryFamilyType::Sketch(SketchKind::new(algorithm, params), Default::default()); + let family = FieldDataType::Sketch(SketchKind::new(algorithm, params), Default::default()); let build = Operator::keyed_summary_build(schema(), family, 1, vec![3], vec![2]).unwrap(); - let output = Arc::new(SummarySchema { + let output = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: vec![ schema().fields[2].clone(), schema().fields[3].clone(), diff --git a/crates/asap-physical-operators/tests/deployment.rs b/crates/asap-physical-operators/tests/deployment.rs index 0c261a033..6db0e990d 100644 --- a/crates/asap-physical-operators/tests/deployment.rs +++ b/crates/asap-physical-operators/tests/deployment.rs @@ -1,15 +1,14 @@ //! Exercise the public library without a backend server, store, or scheduler. use asap_physical_operators::planner::{ post_asap::{ - GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryFamilyType, - SummaryUpdate, + FieldDataType, GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate, }, pre_asap::ColumnRef, }; use asap_physical_operators::{factory::create_planner_accumulator, AggregateCore}; -fn family(k: u32) -> SummaryFamilyType { - SummaryFamilyType::Sketch( +fn family(k: u32) -> FieldDataType { + FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), GroupingStrategy::PerSubpopulationInstance, ) @@ -65,7 +64,7 @@ fn invalid_kll_parameters_are_rejected_at_binding() { fn native_count_sketch_dimensions_are_not_packed_wire_dimensions() { use asap_physical_operators::planner::post_asap::SummaryInputExpr; use asap_physical_operators::KeyByLabelValues; - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::CountSketchWithHeap, SketchParams::CountSketchWithHeap { diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index a3738ac53..4fd00c1a3 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -7,6 +7,7 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; +use planner_types::pre_asap::Schema as LogicalSchema; use planner_types::{post_asap::*, pre_asap::QueryExpr, types::AccuracyTarget, workload::*}; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; @@ -57,7 +58,7 @@ fn exact_dag(query: &str) -> PostAsapDAG { let dag = compile_post_asap_dag(&node).ok()?; dag.nodes .iter() - .all(|n| !matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { family, .. } if !matches!(family, SummaryFamilyType::ExactAggregate(..)))) + .all(|n| !matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { family, .. } if !matches!(family, FieldDataType::ExactAggregate(..)))) .then_some(dag) } _ => None, @@ -76,7 +77,7 @@ fn population_dag(query: &str) -> PostAsapDAG { } /// Raw scan nodes are the frontier; everything above them is compiled. -fn raw_inputs(dag: &PostAsapDAG) -> Vec<(u64, Arc, String)> { +fn raw_inputs(dag: &PostAsapDAG) -> Vec<(u64, Arc, String)> { dag.nodes .iter() .filter_map(|node| match &node.payload { @@ -338,7 +339,7 @@ fn exact_count_finalizes_to_declared_float_value() { read.id = PostAsapNodeId(root.id.0 + 1); read.output_schema = root.output_schema.clone(); read.output_schema.fields.last_mut().unwrap().dtype = - SummaryFamilyType::Plain(planner_types::pre_asap::DataType::Float64); + FieldDataType::Plain(planner_types::pre_asap::DataType::Float64); edge.producer = root.id; edge.consumer = read.id; edge.intermediate_schema = root.output_schema.clone(); @@ -644,7 +645,7 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { }); let count_min = dag.nodes.iter().any(|n| { matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), .. + family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::Cms) }); (bare_count && count_min).then_some(dag) @@ -658,7 +659,7 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { .find(|n| matches!(n.payload, PostAsapOperatorPayload::SummaryAgg { .. })) .unwrap(); let PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + family: FieldDataType::Sketch(kind, _), .. } = &state.payload else { @@ -686,7 +687,7 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { .fields .iter() .map(|field| match &field.dtype { - SummaryFamilyType::Plain(_) => panic!("unexpected stored column {field:?}"), + FieldDataType::Plain(_) => panic!("unexpected stored column {field:?}"), family => Value::Summary { family: family.clone(), state: Arc::new(sketch.clone()), diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 8e8602216..8f138ecf4 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -9,17 +9,20 @@ use asap_physical_operators::{ }; use futures::{executor::block_on, StreamExt}; use planner_types::{ - post_asap::{ExactKind, ExactParams, SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{ExactKind, ExactParams, Field, FieldDataType, Schema as LogicalSchema}, pre_asap::DataType, }; use std::sync::Arc; fn schema(fields: &[(&str, DataType, bool)]) -> Schema { - Arc::new(SummarySchema { + Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: fields .iter() - .map(|(name, dtype, nullable)| SummaryField { + .map(|(name, dtype, nullable)| Field { + table: None, name: (*name).into(), - dtype: SummaryFamilyType::Plain(dtype.clone()), + dtype: FieldDataType::Plain(dtype.clone()), nullable: *nullable, }) .collect(), @@ -119,7 +122,7 @@ fn summary_construction_merge_and_readout_at_both_phases() { let batches = (1..=20) .map(|v| Batch::try_new(schema.clone(), vec![vec![Value::Float64(v as f64)]]).unwrap()) .collect(); - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let build = Operator::summary_build(schema.clone(), family, 0, None, vec![]).unwrap(); let state = build.schema(); let mut dag = PhysicalDAG::default(); @@ -278,7 +281,7 @@ fn binding_rejects_unsupported_operations() { let schema = schema(&[("v", DataType::Float64, false)]); assert!(Operator::summary_build( schema.clone(), - SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), + FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), 0, None, vec![] @@ -286,7 +289,7 @@ fn binding_rejects_unsupported_operations() { .is_err()); let sum = Operator::summary_build( schema.clone(), - SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), 0, None, vec![], @@ -308,7 +311,7 @@ fn binding_rejects_unsupported_operations() { fn kll_raw_partial_and_precomputed_are_native_dags() { use planner_types::post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams}; let input = schema(&[("value", DataType::Float64, false)]); - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), GroupingStrategy::PerSubpopulationInstance, ); @@ -417,11 +420,14 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { #[test] fn exact_state_and_family_validation() { use asap_physical_operators::summary_kernels::exact::ExactAccumulator; - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); acc.update(None, 7., 0); - let schema = Arc::new(SummarySchema { - fields: vec![SummaryField { + let schema = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], + fields: vec![Field { + table: None, name: "state".into(), dtype: family.clone(), nullable: false, @@ -461,7 +467,7 @@ fn exact_state_and_family_validation() { .unwrap(); assert_eq!(floats(&run(&dag, 1, query()), 0), vec![7.]); let wrong = ExactAccumulator::new( - SummaryFamilyType::ExactAggregate(ExactKind::Max, ExactParams::Max), + FieldDataType::ExactAggregate(ExactKind::Max, ExactParams::Max), false, ) .unwrap(); @@ -563,7 +569,7 @@ fn empty_exact_count_is_an_integer_state_readout() { let input = schema(&[("value", DataType::Float64, false)]); let build = Operator::summary_build( input.clone(), - SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count), + FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), 0, None, vec![], @@ -712,21 +718,23 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { ("score", DataType::Float64, false), ]); let keys_schema = schema(&[("key", DataType::Utf8, false)]); - let node = |id, payload, schema: &Schema| PostAsapDAGNode { + let node = |id, payload, schema: &asap_physical_operators::values::Schema| PostAsapDAGNode { id: PostAsapNodeId(id), payload, output_schema: (**schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, guarantee: None, }; - let edge = |producer, consumer, role, schema: &Schema| PostAsapDAGEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), - role, - intermediate_schema: (**schema).clone(), - data_state: ExecutionDataState::QUERY_ROWS, - grouping: GroupingEdgeCompatibility::NotApplicable, - window: WindowEdgeCompatibility::NotApplicable, + let edge = |producer, consumer, role, schema: &asap_physical_operators::values::Schema| { + PostAsapDAGEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role, + intermediate_schema: (**schema).clone(), + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + } }; let groups = GroupKeys::by(vec![0]); let dag = PostAsapDAG { @@ -1047,7 +1055,7 @@ fn assert_weighted_rate_topk(count_sketch: bool) { Some((0, 60_000)), ) .unwrap(); - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new( if count_sketch { SketchAlgorithm::CountSketchWithHeap @@ -1138,12 +1146,12 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { ValueOperation, }; use planner_types::pre_asap::{ - aggregate_output_schema, AggIntent, Column, GroupKeys, QueryExpr, Reduction as IrReduction, + aggregate_output_schema, AggIntent, Field, GroupKeys, QueryExpr, Reduction as IrReduction, Schema as IrSchema, }; let grouped = IrSchema::new(vec![ - Column::new("job", DataType::Utf8, false), - Column::new("sum", DataType::Float64, false), + Field::plain("job", DataType::Utf8, false), + Field::plain("sum", DataType::Float64, false), ]); let output = aggregate_output_schema( &grouped, @@ -1154,9 +1162,15 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { .unwrap(); let input = schema( &output - .columns + .fields .iter() - .map(|c| (c.name.as_str(), c.dtype.clone(), c.nullable)) + .map(|c| { + ( + c.name.as_str(), + c.dtype.plain().unwrap().clone(), + c.nullable, + ) + }) .collect::>(), ); let node = |id, operation| PostAsapDAGNode { diff --git a/crates/asap-physical-operators/tests/physical_plan_recovery.rs b/crates/asap-physical-operators/tests/physical_plan_recovery.rs index 1d3625154..3f2956af4 100644 --- a/crates/asap-physical-operators/tests/physical_plan_recovery.rs +++ b/crates/asap-physical-operators/tests/physical_plan_recovery.rs @@ -5,16 +5,19 @@ use asap_physical_operators::{ physical_planner::{CompiledPhysicalDAG, InputContract}, }; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{Field, FieldDataType, Schema as LogicalSchema}, pre_asap::DataType, }; use std::{collections::BTreeMap, sync::Arc}; fn sorted() -> CompiledPhysicalDAG { - let schema = Arc::new(SummarySchema { - fields: vec![SummaryField { + let schema = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], + fields: vec![Field { + table: None, name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, }], time_index: None, diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs index cea0faccf..2813506a6 100644 --- a/crates/asap-physical-operators/tests/physical_semantics.rs +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -10,18 +10,21 @@ use asap_physical_operators::{ }; use futures::{executor::block_on, StreamExt}; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + post_asap::{Field, FieldDataType, Schema as LogicalSchema}, pre_asap::{CompareOpKind, DataType, JoinKind, Predicate, QueryExpr}, }; use std::{rc::Rc, sync::Arc}; fn schema(fields: &[(&str, DataType, bool)]) -> Schema { - Arc::new(SummarySchema { + Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: fields .iter() - .map(|(name, dtype, nullable)| SummaryField { + .map(|(name, dtype, nullable)| Field { + table: None, name: (*name).into(), - dtype: SummaryFamilyType::Plain(dtype.clone()), + dtype: FieldDataType::Plain(dtype.clone()), nullable: *nullable, }) .collect(), @@ -342,7 +345,7 @@ fn global_extrema_bind_with_planner_derived_schema() { use asap_physical_operators::physical_planner::compile_node; use planner_types::{ post_asap::*, - pre_asap::{AggIntent, Column, GroupKeys, Reduction as PlanReduction}, + pre_asap::{AggIntent, Field, GroupKeys, Reduction as PlanReduction}, }; let input = schema(&[("v", DataType::Int64, false)]); for measure in [ @@ -350,7 +353,7 @@ fn global_extrema_bind_with_planner_derived_schema() { AggIntent::Max { col: Some(0) }, ] { let planner_input = - planner_types::pre_asap::Schema::new(vec![Column::new("v", DataType::Int64, false)]); + planner_types::pre_asap::Schema::new(vec![Field::plain("v", DataType::Int64, false)]); let derived = planner_types::pre_asap::query_expr::aggregate_output_schema( &planner_input, &PlanReduction::Reduce(GroupKeys::by(vec![])), @@ -358,8 +361,12 @@ fn global_extrema_bind_with_planner_derived_schema() { &[], ) .unwrap(); - let result = derived.columns[0].clone(); - let output = schema(&[(&result.name, result.dtype, result.nullable)]); + let result = derived.fields[0].clone(); + let output = schema(&[( + &result.name, + result.dtype.plain().unwrap().clone(), + result.nullable, + )]); let node = PostAsapDAGNode { id: PostAsapNodeId(1), payload: PostAsapOperatorPayload::Value { @@ -539,7 +546,7 @@ fn boolean_truth_tables_agree_between_expression_paths() { fn kll_partial_merge_and_multiple_readouts_preserve_population() { use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; let input = schema(&[("v", DataType::Float64, false)]); - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), Default::default(), ); @@ -662,7 +669,7 @@ fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { ] { let build = Operator::summary_build( input.clone(), - SummaryFamilyType::ExactAggregate(kind, params), + FieldDataType::ExactAggregate(kind, params), 0, None, vec![], diff --git a/crates/asap-physical-operators/tests/plan_properties.rs b/crates/asap-physical-operators/tests/plan_properties.rs index 272079a90..41cba5424 100644 --- a/crates/asap-physical-operators/tests/plan_properties.rs +++ b/crates/asap-physical-operators/tests/plan_properties.rs @@ -8,8 +8,8 @@ use asap_physical_operators::{ Error, }; use planner_types::{ - post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, - pre_asap::{Column, DataType, QueryExpr, Schema as LogicalSchema, Source}, + post_asap::{Field, FieldDataType, Schema as LogicalSchema}, + pre_asap::{DataType, QueryExpr, Source}, }; use std::sync::{ atomic::{AtomicUsize, Ordering}, @@ -35,10 +35,13 @@ impl RawSource for DeclaredSource { // A blocking parent must reject unknown and unbounded Scan inputs without opening a reader. #[test] fn blocking_inputs_require_an_explicit_finite_source() { - let schema = Arc::new(SummarySchema { - fields: vec![SummaryField { + let schema = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], + fields: vec![Field { + table: None, name: "v".into(), - dtype: SummaryFamilyType::Plain(DataType::Int64), + dtype: FieldDataType::Plain(DataType::Int64), nullable: false, }], time_index: None, @@ -66,7 +69,7 @@ fn blocking_inputs_require_an_explicit_finite_source() { let scan = registry .bind(&QueryExpr::Scan { source: identity, - schema: LogicalSchema::new(vec![Column::new("v", DataType::Int64, false)]), + schema: LogicalSchema::new(vec![Field::plain("v", DataType::Int64, false)]), predicates: vec![], }) .unwrap(); @@ -121,7 +124,7 @@ fn summary_capability_levels_are_distinct() { pre_asap::ColumnRef, }; let grouping = GroupingStrategy::default(); - let cms = SummaryFamilyType::Sketch( + let cms = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::Cms, SketchParams::Cms { @@ -154,7 +157,7 @@ fn summary_capability_levels_are_distinct() { } ) .is_err()); - let kll = SummaryFamilyType::Sketch( + let kll = FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 128 }), grouping, ); diff --git a/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs b/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs index 367650df2..3ca750ca8 100644 --- a/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs +++ b/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs @@ -31,7 +31,7 @@ impl AccuracyEvidenceProvider for Evidence { fn propagation_stats( &self, op: &CompositionOperator, - _: &SummaryFamilyType, + _: &FieldDataType, _: Option<&SketchQuery>, ) -> PropagationStats { if matches!(op, CompositionOperator::TopKSelection) { diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index 6a6669fc7..955948460 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -93,7 +93,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { matches!( node.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. } ) @@ -150,21 +150,19 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .fields .iter() .map(|field| match &field.dtype { - SummaryFamilyType::ExactAggregate(..) => summary.clone(), - SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(2000), - SummaryFamilyType::Plain(DataType::Utf8) => { - Value::Utf8(if field.name == "job" { - "api".into() - } else { - serde_json::to_string(&BTreeMap::from([ - ("__name__", "m".to_string()), - ("job", "api".to_string()), - ("instance", format!("series-{index}")), - ])) - .unwrap() - .into() - }) - } + FieldDataType::ExactAggregate(..) => summary.clone(), + FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(2000), + FieldDataType::Plain(DataType::Utf8) => Value::Utf8(if field.name == "job" { + "api".into() + } else { + serde_json::to_string(&BTreeMap::from([ + ("__name__", "m".to_string()), + ("job", "api".to_string()), + ("instance", format!("series-{index}")), + ])) + .unwrap() + .into() + }), _ => panic!("unexpected input field {field:?}"), }) .collect() @@ -391,7 +389,7 @@ fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { matches!( &node.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. } ) @@ -429,7 +427,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { matches!( node.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. } ) @@ -443,7 +441,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { matches!( node.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + family: FieldDataType::ExactAggregate(ExactKind::Sum, _), .. } ) @@ -484,9 +482,9 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { .fields .iter() .map(|field| match &field.dtype { - SummaryFamilyType::ExactAggregate(..) => summary.clone(), - SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), - SummaryFamilyType::Plain(DataType::Utf8) + FieldDataType::ExactAggregate(..) => summary.clone(), + FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), + FieldDataType::Plain(DataType::Utf8) if field.name == "$promql_series_identity" => { Value::Utf8( @@ -498,7 +496,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { .into(), ) } - SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8("api".into()), + FieldDataType::Plain(DataType::Utf8) => Value::Utf8("api".into()), _ => panic!("unexpected input field {field:?}"), }) .collect() @@ -540,7 +538,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { .schema() .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); assert!( output[0].rows()[0] .iter() diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs index 2c9e5f764..59ece54e2 100644 --- a/crates/asap-physical-operators/tests/precompute_population.rs +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -8,6 +8,7 @@ use asap_physical_operators::{ Statistic, }; use futures::{executor::block_on, StreamExt}; +use planner_types::pre_asap::Schema as LogicalSchema; use planner_types::{ post_asap::*, pre_asap::{ArithmeticOpKind, BinaryOpKind, ColumnRef, DataType, GroupKeys, Reduction}, @@ -17,9 +18,12 @@ use std::{collections::BTreeMap, sync::Arc}; // Typed series identity survives finalization and derived precompute through population metadata. #[test] fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let schema = |dtype| SummarySchema { - fields: vec![SummaryField { + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let schema = |dtype| LogicalSchema { + closed: true, + unique_keys: vec![], + fields: vec![Field { + table: None, name: "value".into(), dtype, nullable: false, @@ -27,16 +31,18 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { time_index: None, }; let state_schema = schema(family.clone()); - let mut value_schema = schema(SummaryFamilyType::Plain(DataType::Float64)); - value_schema.fields.push(SummaryField { + let mut value_schema = schema(FieldDataType::Plain(DataType::Float64)); + value_schema.fields.push(Field { + table: None, name: "time".into(), - dtype: SummaryFamilyType::Plain(DataType::Timestamp), + dtype: FieldDataType::Plain(DataType::Timestamp), nullable: false, }); value_schema.time_index = Some(1); - value_schema.fields.push(SummaryField { + value_schema.fields.push(Field { + table: None, name: planner_types::pre_asap::schema::PROMQL_SERIES_IDENTITY.into(), - dtype: SummaryFamilyType::Plain(DataType::Utf8), + dtype: FieldDataType::Plain(DataType::Utf8), nullable: false, }); for (weight, expected) in [ @@ -119,7 +125,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { let fields = &mut invalid_identity.nodes[1].output_schema.fields; match mutation { 0 => fields[2].nullable = true, - 1 => fields[2].dtype = SummaryFamilyType::Plain(DataType::Float64), + 1 => fields[2].dtype = FieldDataType::Plain(DataType::Float64), _ => fields.push(fields[2].clone()), } assert!(precompute::compile(&invalid_identity, &[0], &[3]).is_err()); @@ -214,9 +220,12 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { } } -fn logical_schema(family: SummaryFamilyType) -> SummarySchema { - SummarySchema { - fields: vec![SummaryField { +fn logical_schema(family: FieldDataType) -> LogicalSchema { + LogicalSchema { + closed: true, + unique_keys: vec![], + fields: vec![Field { + table: None, name: "value".into(), dtype: family, nullable: false, @@ -225,8 +234,8 @@ fn logical_schema(family: SummaryFamilyType) -> SummarySchema { } } fn state_dag( - family: SummaryFamilyType, - target: Option, + family: FieldDataType, + target: Option, merge: bool, ) -> CompiledPhysicalDAG { let mut nodes = vec![PostAsapDAGNode { @@ -250,7 +259,7 @@ fn state_dag( operation: ValueOperation::FinalizeExactAccumulator, }, output_state: ExecutionDataState::INGESTION_ROWS, - output_schema: logical_schema(SummaryFamilyType::Plain(DataType::Float64)), + output_schema: logical_schema(FieldDataType::Plain(DataType::Float64)), guarantee: None, }); if let Some(target) = target { @@ -289,7 +298,7 @@ fn state_dag( } fn native_run( program: &CompiledPhysicalDAG, - family: SummaryFamilyType, + family: FieldDataType, states: Vec>, context: RunContext, ) -> Result>, asap_physical_operators::Error> { @@ -337,7 +346,7 @@ fn ingestion_context(limits: Limits) -> RunContext { } fn sum_state(value: f64) -> Arc { let mut state = asap_physical_operators::summary_kernels::exact::ExactAccumulator::new( - planner_types::post_asap::SummaryFamilyType::ExactAggregate( + planner_types::post_asap::FieldDataType::ExactAggregate( planner_types::post_asap::ExactKind::Sum, planner_types::post_asap::ExactParams::Sum, ), @@ -351,7 +360,7 @@ fn sum_state(value: f64) -> Arc { // Only an explicit merge may collapse distinct pane updates before finalization. #[test] fn explicit_merge_changes_pane_cardinality() { - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); for (merge, expected) in [(false, vec![2., 7.]), (true, vec![9.])] { let program = state_dag(family.clone(), None, merge); let rows = native_run( @@ -376,8 +385,8 @@ fn explicit_merge_changes_pane_cardinality() { // Typed updates reject invalid domains before publishing any target state. #[test] fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { - let source = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let target = SummaryFamilyType::Sketch( + let source = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let target = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::DDSketch, SketchParams::DDSketch { alpha: 0.01 }, @@ -407,7 +416,7 @@ fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { #[test] fn precompute_count_conversion_checks_precision() { use asap_physical_operators::summary_kernels::exact::ExactAccumulator; - let family = SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count); + let family = FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count); let program = state_dag(family.clone(), None, false); for (count, valid) in [(3u64, true), ((1u64 << 53) + 1, false)] { let mut state = @@ -427,7 +436,7 @@ fn precompute_count_conversion_checks_precision() { // DAG execution retains terminal cancellation and shared workspace limits. #[test] fn precompute_dag_enforces_cancellation_and_budget() { - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let program = state_dag(family.clone(), None, true); let context = ingestion_context(Limits::default()); context.cancel(); diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs index bbfc62c0d..c620b0342 100644 --- a/crates/asap-physical-operators/tests/promql_binary.rs +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -8,28 +8,32 @@ use asap_physical_operators::{ use futures::{executor::block_on, StreamExt}; use planner_types::{ post_asap::{ - BinaryOperator, ExecutionDataState, PostAsapDAGNode, PostAsapNodeId, - PostAsapOperatorPayload, SummaryFamilyType, SummaryField, SummarySchema, + BinaryOperator, ExecutionDataState, Field, FieldDataType, PostAsapDAGNode, PostAsapNodeId, + PostAsapOperatorPayload, Schema as LogicalSchema, }, pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, }; use std::{collections::BTreeMap, sync::Arc}; fn schema() -> Schema { - Arc::new(SummarySchema { + Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: vec![ - SummaryField { + Field { + table: None, name: "labels".into(), - dtype: SummaryFamilyType::Plain(DataType::Map { + dtype: FieldDataType::Plain(DataType::Map { key: Box::new(DataType::Utf8), value: Box::new(DataType::Utf8), value_nullable: false, }), nullable: false, }, - SummaryField { + Field { + table: None, name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, }, ], @@ -319,15 +323,19 @@ fn stored_series_readouts_support_filters_and_sets() { (ExactKind::Sum, ExactParams::Sum), (ExactKind::Count, ExactParams::Count), ] { - let family = SummaryFamilyType::ExactAggregate(exact_kind.clone(), params); - let state_schema = Arc::new(SummarySchema { + let family = FieldDataType::ExactAggregate(exact_kind.clone(), params); + let state_schema = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: vec![ - SummaryField { + Field { + table: None, name: PROMQL_SERIES_IDENTITY.into(), - dtype: SummaryFamilyType::Plain(DataType::Utf8), + dtype: FieldDataType::Plain(DataType::Utf8), nullable: false, }, - SummaryField { + Field { + table: None, name: "value".into(), dtype: family.clone(), nullable: false, @@ -336,7 +344,7 @@ fn stored_series_readouts_support_filters_and_sets() { time_index: None, }); let mut value_schema = (*state_schema).clone(); - value_schema.fields[1].dtype = SummaryFamilyType::Plain(DataType::Float64); + value_schema.fields[1].dtype = FieldDataType::Plain(DataType::Float64); for kind in [ BinaryOpKind::Compare(CompareOpKind::Gt), BinaryOpKind::Set(PromQLVectorSetOpKind::And), diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs index 108d9b19f..ffb726b1b 100644 --- a/crates/asap-physical-operators/tests/promql_values.rs +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -411,7 +411,7 @@ fn exact_state_readouts_recover_and_finalize_panes() { (ExactKind::Min, ExactParams::Min, 1.), (ExactKind::Max, ExactParams::Max, 5.), ] { - let family = SummaryFamilyType::ExactAggregate(kind, params); + let family = FieldDataType::ExactAggregate(kind, params); for preserve in [false, true] { let rows = [[1., 2.], [4., 5.]] .into_iter() @@ -459,7 +459,7 @@ fn recovered_exact_counter_uses_window_and_omits_insufficient_samples() { (ExactKind::Rate, ExactParams::Rate, 1.), (ExactKind::Increase, ExactParams::Increase, 60.), ] { - let family = SummaryFamilyType::ExactAggregate(kind, params); + let family = FieldDataType::ExactAggregate(kind, params); let rows = [1, 2] .into_iter() .map(|count| { diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs index 3e9f4cb10..b95987796 100644 --- a/crates/asap-physical-operators/tests/raw_scan.rs +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -6,9 +6,10 @@ use asap_physical_operators::dag::{ Error, Limits, OutputStream, RunContext, Scope, }; use futures::{executor::block_on, stream, StreamExt}; +use planner_types::pre_asap::Schema as LogicalSchema; use planner_types::{ post_asap::*, - pre_asap::{Column, DataType, GroupKeys, Predicate, QueryExpr, Source}, + pre_asap::{DataType, Field, GroupKeys, Predicate, QueryExpr, Source}, }; use std::{ collections::BTreeMap, @@ -21,11 +22,14 @@ use std::{ fn fixture() -> (QueryExpr, Schema, Vec) { let schema = - planner_types::pre_asap::Schema::new(vec![Column::new("value", DataType::Int64, true)]); - let output = Arc::new(SummarySchema { - fields: vec![SummaryField { + planner_types::pre_asap::Schema::new(vec![Field::plain("value", DataType::Int64, true)]); + let output = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], + fields: vec![Field { + table: None, name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Int64), + dtype: FieldDataType::Plain(DataType::Int64), nullable: true, }], time_index: None, diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs index 63785a1b3..ee54b3d45 100644 --- a/crates/asap-physical-operators/tests/summary_projection.rs +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -8,6 +8,7 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; +use planner_types::pre_asap::Schema as LogicalSchema; use planner_types::{ post_asap::*, pre_asap::{ColumnRef, DataType, ProjectItem, QueryExpr}, @@ -18,26 +19,32 @@ use std::{collections::BTreeMap, sync::Arc}; // the family and pass through the same immutable state, without decoding the payload. #[test] fn post_asap_summary_projection_survives_recovery() { - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let schema = Arc::new(SummarySchema { + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let schema = Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], fields: vec![ - SummaryField { + Field { + table: None, name: "state".into(), dtype: family.clone(), nullable: false, }, - SummaryField { + Field { + table: None, name: "service".into(), - dtype: SummaryFamilyType::Plain(DataType::Utf8), + dtype: FieldDataType::Plain(DataType::Utf8), nullable: false, }, ], time_index: None, }); - let output = SummarySchema { + let output = LogicalSchema { + closed: true, + unique_keys: vec![], fields: vec![ schema.fields[1].clone(), - SummaryField { + Field { name: "renamed".into(), ..schema.fields[0].clone() }, diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index f5e875bab..a5d0d4dd8 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -27,7 +27,7 @@ impl AccuracyEvidenceProvider for Evidence { fn propagation_stats( &self, op: &CompositionOperator, - _: &SummaryFamilyType, + _: &FieldDataType, _: Option<&SketchQuery>, ) -> PropagationStats { if matches!(op, CompositionOperator::TopKSelection) { @@ -89,7 +89,7 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S }) .unwrap(); let dag = compile_post_asap_dag(&plan).unwrap(); - let build=dag.nodes.iter().find(|node|matches!(&node.payload,PostAsapOperatorPayload::SummaryAgg{family:SummaryFamilyType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); + let build=dag.nodes.iter().find(|node|matches!(&node.payload,PostAsapOperatorPayload::SummaryAgg{family:FieldDataType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); let rate_id = dag .edges .iter() @@ -123,7 +123,7 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S "job" => Value::Utf8(job.into()), "value" => Value::Float64(value), _ => match field.dtype { - SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), + FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), _ => panic!("unexpected rate column {field:?}"), }, }) @@ -235,7 +235,7 @@ pub fn lower_promql( // The old untyped heap updater must not silently round a Planner rate update. #[test] fn rate_updates_cannot_enter_integer_heap_factory() { - let family = SummaryFamilyType::Sketch( + let family = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::CmsWithHeap, SketchParams::CmsWithHeap { @@ -292,8 +292,8 @@ fn check_direct_rate_topk(dynamic: bool) { QueryExpr::Scan { schema, .. } => { schema.closed = true; schema - .columns - .push(planner_types::pre_asap::schema::Column::new( + .fields + .push(planner_types::pre_asap::schema::Field::plain( "service", DataType::Utf8, false, @@ -354,9 +354,9 @@ fn check_direct_rate_topk(dynamic: bool) { } let dag = compile_post_asap_dag(candidate).unwrap(); assert!(dag.nodes.iter().any(|node| matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); + PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); let build = dag.nodes.iter().find(|node| matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); + PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); let input_id = dag .edges .iter() @@ -888,7 +888,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { matches!( &node.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. } ) @@ -901,7 +901,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { matches!( &node.payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::Sketch(..), + family: FieldDataType::Sketch(..), .. } ) @@ -990,9 +990,9 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .fields .iter() .map(|field| match &field.dtype { - SummaryFamilyType::ExactAggregate(..) => summary.clone(), - SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(end), - SummaryFamilyType::Plain(DataType::Utf8) + FieldDataType::ExactAggregate(..) => summary.clone(), + FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(end), + FieldDataType::Plain(DataType::Utf8) if field.name == "$promql_series_identity" => { Value::Utf8( @@ -1004,7 +1004,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .into(), ) } - SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8("api".into()), + FieldDataType::Plain(DataType::Utf8) => Value::Utf8("api".into()), _ => panic!("unexpected state field {field:?}"), }) .collect() diff --git a/crates/devtools/examples/canonical_examples.rs b/crates/devtools/examples/canonical_examples.rs index c7d089243..b0df222a3 100644 --- a/crates/devtools/examples/canonical_examples.rs +++ b/crates/devtools/examples/canonical_examples.rs @@ -5,12 +5,12 @@ use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn packets_catalog() -> SqlCatalog { diff --git a/crates/devtools/examples/topk_ir.rs b/crates/devtools/examples/topk_ir.rs index ee467aa1c..21dcefb4b 100644 --- a/crates/devtools/examples/topk_ir.rs +++ b/crates/devtools/examples/topk_ir.rs @@ -4,11 +4,11 @@ // resulting pre-ASAP IR. Used for interactive exploration; not a test. use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn catalog() -> SqlCatalog { diff --git a/crates/devtools/src/bin/analyze_corpora.rs b/crates/devtools/src/bin/analyze_corpora.rs index 2c4bacefb..7aa87c8e7 100644 --- a/crates/devtools/src/bin/analyze_corpora.rs +++ b/crates/devtools/src/bin/analyze_corpora.rs @@ -6,7 +6,7 @@ use asap_devtools::{lower_promql_with_data_ingestion_interval, SqlCatalog}; use asap_frontend_sql::lower_sql_dialect; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use serde::Serialize; @@ -29,8 +29,8 @@ const NETFLOW: &str = include_str!("../../../frontend-sql/tests/netflow/data/net const BGP: &str = include_str!("../../../frontend-sql/tests/bgp_analytics/data/bgp_analytics.sql"); const BGP_WORKLOAD: &str = include_str!("../../../frontend-sql/tests/bgp_jan2024_workload/data/bgp_jan2024_rrc00_200_query_workload.yaml"); -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn dqc_catalog() -> SqlCatalog { diff --git a/crates/devtools/src/bin/dag_export.rs b/crates/devtools/src/bin/dag_export.rs index 9c9829e78..de205c4d4 100644 --- a/crates/devtools/src/bin/dag_export.rs +++ b/crates/devtools/src/bin/dag_export.rs @@ -99,10 +99,10 @@ use asap_types::dag_export::{ }; use asap_types::post_asap::SummaryExpr; use asap_types::post_asap::SummaryNode; -use asap_types::post_asap::{CompositionOperator, SketchQuery, SummaryFamilyType}; +use asap_types::post_asap::{CompositionOperator, FieldDataType, SketchQuery}; use asap_types::pre_asap::cse::{structural_hash, HashCache}; use asap_types::pre_asap::query_expr::QueryExpr; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::resources::CacheProfile; use asap_types::types::AccuracyTarget; @@ -712,11 +712,11 @@ fn default_catalog() -> SqlCatalog { "metrics", Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("service", DataType::Utf8, false), - Column::new("region", DataType::Utf8, false), - Column::new("latency", DataType::Float64, false), - Column::new("bytes", DataType::Int64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("region", DataType::Utf8, false), + Field::plain("latency", DataType::Float64, false), + Field::plain("bytes", DataType::Int64, false), ], 0, vec![], @@ -725,8 +725,8 @@ fn default_catalog() -> SqlCatalog { .with_table( "hosts", Schema::new(vec![ - Column::new("service", DataType::Utf8, false), - Column::new("region", DataType::Utf8, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("region", DataType::Utf8, false), ]), ) } @@ -741,8 +741,8 @@ fn catalog(custom: &[String]) -> SqlCatalog { .expect("--table-schema.name must be a string"); let columns = value["columns"] .as_array() - .expect("--table-schema.columns must be an array"); - let columns: Vec = columns + .expect("--table-schema.fields must be an array"); + let columns: Vec = columns .iter() .map(|column| { let column_name = column["name"] @@ -760,7 +760,7 @@ fn catalog(custom: &[String]) -> SqlCatalog { "int64" | "bigint" => DataType::Int64, other => panic!("unsupported column type {other:?}"), }; - Column::new( + Field::plain( column_name, data_type, column["nullable"].as_bool().unwrap_or(true), @@ -825,7 +825,7 @@ impl AccuracyEvidenceProvider for TopKMarginEvidence { fn propagation_stats( &self, op: &CompositionOperator, - _family: &SummaryFamilyType, + _family: &FieldDataType, _query: Option<&SketchQuery>, ) -> PropagationStats { if matches!(op, CompositionOperator::TopKSelection) { @@ -1704,7 +1704,7 @@ mod tests { }; use asap_aware_mapping::query_physical_lowering::lower_query_physical_dag; use asap_devtools::PromqlError; - use asap_types::pre_asap::{Column, DataType, Reduction, Schema, Source}; + use asap_types::pre_asap::{DataType, Field, Reduction, Schema, Source}; fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Result { lower_promql_with_data_ingestion_interval(query, accuracy, 1_000) @@ -1724,7 +1724,7 @@ mod tests { table_ref: "events".into(), }, predicates: vec![], - schema: Schema::new(vec![Column::new("v", DataType::Int64, false)]), + schema: Schema::new(vec![Field::plain("v", DataType::Int64, false)]), }), } } diff --git a/crates/devtools/src/bin/show_post_asap_ir.rs b/crates/devtools/src/bin/show_post_asap_ir.rs index 18260b42f..4b2cbf917 100644 --- a/crates/devtools/src/bin/show_post_asap_ir.rs +++ b/crates/devtools/src/bin/show_post_asap_ir.rs @@ -26,7 +26,7 @@ use asap_aware_mapping::{ }; use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; use asap_types::pre_asap::query_expr::QueryExpr; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use std::io::Read; use std::rc::Rc; @@ -58,8 +58,8 @@ fn bind_all(expr: &QueryExpr) -> Result Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn catalog() -> SqlCatalog { diff --git a/crates/devtools/src/bin/show_pre_asap_ir.rs b/crates/devtools/src/bin/show_pre_asap_ir.rs index b2f60463b..491b48ff2 100644 --- a/crates/devtools/src/bin/show_pre_asap_ir.rs +++ b/crates/devtools/src/bin/show_pre_asap_ir.rs @@ -17,12 +17,12 @@ // bytes)` catalog — the same table used in cross_language.rs and topk_ir.rs. use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use std::io::Read; -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn catalog() -> SqlCatalog { diff --git a/crates/devtools/src/bin/sketch_coverage.rs b/crates/devtools/src/bin/sketch_coverage.rs index 58fa25348..78bb4cbbd 100644 --- a/crates/devtools/src/bin/sketch_coverage.rs +++ b/crates/devtools/src/bin/sketch_coverage.rs @@ -28,7 +28,7 @@ use asap_aware_mapping::{explain_replacements, ExplanationKind}; use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; @@ -57,8 +57,8 @@ fn promql_lines(corpus: &str) -> impl Iterator { .filter(|l| !l.is_empty() && !l.starts_with('#')) } -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn dqc_catalog() -> SqlCatalog { diff --git a/crates/devtools/src/bin/variant_coverage.rs b/crates/devtools/src/bin/variant_coverage.rs index 290d5584d..fe494a0f6 100644 --- a/crates/devtools/src/bin/variant_coverage.rs +++ b/crates/devtools/src/bin/variant_coverage.rs @@ -6,7 +6,7 @@ use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; @@ -171,8 +171,8 @@ fn promql_lines(corpus: &str) -> impl Iterator { .filter(|l| !l.is_empty() && !l.starts_with('#')) } -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn dqc_catalog() -> SqlCatalog { diff --git a/crates/devtools/tests/cross_language.rs b/crates/devtools/tests/cross_language.rs index 0518f62b7..2e4d6ff4f 100644 --- a/crates/devtools/tests/cross_language.rs +++ b/crates/devtools/tests/cross_language.rs @@ -14,12 +14,12 @@ //! explicit inner `Aggregate([Count])`. use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; use asap_types::types::AccuracyTarget; -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn catalog() -> SqlCatalog { diff --git a/crates/frontend-promql/tests/count_planning.rs b/crates/frontend-promql/tests/count_planning.rs index fded3cebe..3d3b65651 100644 --- a/crates/frontend-promql/tests/count_planning.rs +++ b/crates/frontend-promql/tests/count_planning.rs @@ -9,8 +9,8 @@ use asap_aware_mapping::{ }; mod support; use asap_types::post_asap::{ - compile_post_asap_dag, ExactKind, NonNegativeWeightProof, PostAsapOperatorPayload, - SketchAlgorithm, SummaryExpr, SummaryFamilyType, SummaryInputExpr, WeightDomain, + compile_post_asap_dag, ExactKind, FieldDataType, NonNegativeWeightProof, + PostAsapOperatorPayload, SketchAlgorithm, SummaryExpr, SummaryInputExpr, WeightDomain, }; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -58,7 +58,7 @@ fn exact_counts_select_count_accumulators() { candidates.iter().any(|candidate| { matches!(&candidate.replacement, Replacement::Summary(node) if matches!(&node.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Count, _), .. })) + family: FieldDataType::ExactAggregate(ExactKind::Count, _), .. })) }), "{query}: {candidates:?}" ); @@ -81,7 +81,7 @@ fn frequency_count_candidates_use_unit_weights() { continue; }; let SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + family: FieldDataType::Sketch(kind, _), input, .. } = &summary_input.expr @@ -240,7 +240,7 @@ fn cms_count_updates_total_ten_for_zero_positive_and_negative_samples() { .iter() .any(|node| { matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } + PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::Cms) }) .then_some(dag) diff --git a/crates/frontend-promql/tests/promql_conformance.rs b/crates/frontend-promql/tests/promql_conformance.rs index a7b405f0d..6f932879e 100644 --- a/crates/frontend-promql/tests/promql_conformance.rs +++ b/crates/frontend-promql/tests/promql_conformance.rs @@ -670,7 +670,7 @@ fn unary_negation_lowers_as_multiply_by_minus_one() { let schema = ok("-some_metric").output_schema().unwrap(); assert_eq!( schema - .columns + .fields .iter() .map(|c| c.name.as_str()) .collect::>(), @@ -1186,7 +1186,7 @@ fn nested_subquery_from_prometheus_docs() { let schema = qe.output_schema().expect("schema derivation"); assert_eq!( schema - .columns + .fields .iter() .map(|c| c.name.as_str()) .collect::>(), @@ -1315,7 +1315,7 @@ fn count_over_time_value_column_is_float64() { // derived `value` column must be `Float64` like every other range reducer. let schema = ok("count_over_time(m[5m])").output_schema().unwrap(); let value = schema - .columns + .fields .iter() .find(|c| c.name == "value") .expect("value column"); @@ -1749,9 +1749,9 @@ fn absent_keeps_matcher_labels_for_the_synthesized_output() { let qe = ok(r#"absent(up{job="x"})"#); let cols = qe.output_schema().unwrap(); assert!( - cols.columns.iter().any(|c| c.name == "job"), + cols.fields.iter().any(|c| c.name == "job"), "matcher label `job` kept, got {:?}", - cols.columns.iter().map(|c| &c.name).collect::>() + cols.fields.iter().map(|c| &c.name).collect::>() ); } @@ -1766,8 +1766,8 @@ fn time_lowers_to_the_eval_time_scalar() { assert!(matches!(ok("time()"), QueryExpr::EvalTimestamp)); // …and it is scalar-shaped: a single float `value`, no time index. let sch = ok("time()").output_schema().unwrap(); - assert_eq!(sch.columns.len(), 1); - assert_eq!(sch.columns[0].name, "value"); + assert_eq!(sch.fields.len(), 1); + assert_eq!(sch.fields[0].name, "value"); assert!(sch.time_index.is_none()); } @@ -1855,7 +1855,7 @@ fn vector_promotes_a_scalar_to_a_vector() { // Vector-typed: schema has a time index (a scalar leaf has none). let sch = qe.output_schema().unwrap(); assert!(sch.time_index.is_some()); - assert!(sch.columns.iter().any(|c| c.name == "value")); + assert!(sch.fields.iter().any(|c| c.name == "value")); } #[test] @@ -1870,8 +1870,8 @@ fn scalar_collapses_a_vector_to_a_scalar() { // PromqlScalarBridge-typed: single `value` column, no time index. let sch = qe.output_schema().unwrap(); assert!(sch.time_index.is_none()); - assert_eq!(sch.columns.len(), 1); - assert_eq!(sch.columns[0].name, "value"); + assert_eq!(sch.fields.len(), 1); + assert_eq!(sch.fields[0].name, "value"); } #[test] @@ -1973,7 +1973,7 @@ fn group_lowers_to_a_constant_group_intent() { assert!(matches!(measures.as_slice(), [AggIntent::Group])); // Output column is the constant-1 `group` value. let sch = qe.output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "group")); + assert!(sch.fields.iter().any(|c| c.name == "group")); } #[test] @@ -1981,7 +1981,7 @@ fn group_by_keeps_the_grouping_keys() { // `group by (job) (up)` — the grouping keys ride on `Aggregate.by`. let qe = ok("group by (job) (up)"); let sch = qe.output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "job")); + assert!(sch.fields.iter().any(|c| c.name == "job")); assert!(has(&qe, |i| *i == AggIntent::Group)); } @@ -1999,7 +1999,7 @@ fn count_values_groups_by_value_and_synthesizes_a_label() { ); let sch = qe.output_schema().unwrap(); let version = sch - .columns + .fields .iter() .find(|c| c.name == "version") .expect("synthesized `version` label column"); @@ -2009,7 +2009,7 @@ fn count_values_groups_by_value_and_synthesizes_a_label() { "the value becomes a string label" ); assert!( - sch.columns.iter().any(|c| c.name == "count"), + sch.fields.iter().any(|c| c.name == "count"), "and a count column" ); } @@ -2024,8 +2024,8 @@ fn count_values_accepts_a_parenthesised_label_and_by_grouping() { |i| matches!(i, AggIntent::CountValues { label } if label == "v") )); let sch = qe.output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "job")); - assert!(sch.columns.iter().any(|c| c.name == "v")); + assert!(sch.fields.iter().any(|c| c.name == "job")); + assert!(sch.fields.iter().any(|c| c.name == "v")); } #[test] @@ -2035,9 +2035,9 @@ fn count_values_label_colliding_with_a_group_key_is_not_duplicated() { // output must carry a single `job` column, never two. let qe = ok(r#"count_values by (job) ("job", version)"#); let sch = qe.output_schema().unwrap(); - let jobs = sch.columns.iter().filter(|c| c.name == "job").count(); - assert_eq!(jobs, 1, "collision deduped, got {:?}", sch.columns); - assert!(sch.columns.iter().any(|c| c.name == "count")); + let jobs = sch.fields.iter().filter(|c| c.name == "job").count(); + assert_eq!(jobs, 1, "collision deduped, got {:?}", sch.fields); + assert!(sch.fields.iter().any(|c| c.name == "count")); } #[test] @@ -2058,7 +2058,7 @@ fn limitk_and_limit_ratio_lower_to_series_sampling() { )); // Series-preserving: the output schema equals the input's (ts, value). let sch = ok("limitk(2, http_requests)").output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "value")); + assert!(sch.fields.iter().any(|c| c.name == "value")); assert!(sch.time_index.is_some()); } @@ -2138,8 +2138,8 @@ fn label_replace_is_a_relabel_over_the_vector() { assert!(is_fn_named(value, "label_replace")); // Output: the child's columns + the synthesized `host` label; value & ts kept. let sch = qe.output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "host")); - assert!(sch.columns.iter().any(|c| c.name == "value")); + assert!(sch.fields.iter().any(|c| c.name == "host")); + assert!(sch.fields.iter().any(|c| c.name == "value")); assert!(sch.time_index.is_some(), "the vector's time axis survives"); } @@ -2154,7 +2154,7 @@ fn label_join_concatenates_source_labels() { assert_eq!(dst, "combined"); assert!(is_fn_named(value, "label_join")); let sch = qe.output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "combined")); + assert!(sch.fields.iter().any(|c| c.name == "combined")); } #[test] @@ -2167,7 +2167,7 @@ fn label_replace_composes_under_an_aggregation() { assert!(matches!(relabel, QueryExpr::PromqlRelabel { dst, .. } if dst == "host")); assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); let sch = qe.output_schema().unwrap(); - assert!(sch.columns.iter().any(|c| c.name == "host")); + assert!(sch.fields.iter().any(|c| c.name == "host")); } // ───────────────────────────────────────────────────────────────────────────── @@ -2240,10 +2240,7 @@ fn sort_by_label_orders_on_each_label_in_turn() { assert!(keys.iter().all(|k| k.ascending)); let sch = qe.output_schema().unwrap(); for label in ["group", "instance", "job"] { - assert!( - sch.columns.iter().any(|c| c.name == label), - "{label} seeded" - ); + assert!(sch.fields.iter().any(|c| c.name == label), "{label} seeded"); } } diff --git a/crates/frontend-promql/tests/promql_lowering.rs b/crates/frontend-promql/tests/promql_lowering.rs index 619d066c2..243bf6d18 100644 --- a/crates/frontend-promql/tests/promql_lowering.rs +++ b/crates/frontend-promql/tests/promql_lowering.rs @@ -308,7 +308,7 @@ fn classic_histogram(qe: &QueryExpr) -> (Vec, usize, Vec) { let names = qe .output_schema() .unwrap() - .columns + .fields .iter() .map(|c| c.name.clone()) .collect(); @@ -325,7 +325,7 @@ fn classic_histogram_quantile_groups_without_le() { unreachable!() }; let child = child.output_schema().unwrap(); - assert_eq!(child.columns[le].name, "le"); + assert_eq!(child.fields[le].name, "le"); assert_eq!(keys, vec![le]); assert_eq!(names, vec!["histogram_quantile"]); } @@ -861,10 +861,10 @@ fn has_intent bool>(e: &QueryExpr, pred: F) -> bool { all_intents(e).iter().any(pred) } -/// Column names on the first `Scan` reachable by descending single-child nodes. +/// Field names on the first `Scan` reachable by descending single-child nodes. fn scan_columns(e: &QueryExpr) -> Vec { match e { - QueryExpr::Scan { schema, .. } => schema.columns.iter().map(|c| c.name.clone()).collect(), + QueryExpr::Scan { schema, .. } => schema.fields.iter().map(|c| c.name.clone()).collect(), QueryExpr::Aggregate { child, .. } | QueryExpr::TimeRange { child, .. } | QueryExpr::Filter { child, .. } @@ -1002,7 +1002,7 @@ fn aggregate_output_schema_preserves_time_axis_and_labels() { panic!("expected Aggregate, got {qe:?}"); }; let schema = qe.output_schema().expect("aggregate schema"); - let names: Vec<&str> = schema.columns.iter().map(|c| c.name.as_str()).collect(); + let names: Vec<&str> = schema.fields.iter().map(|c| c.name.as_str()).collect(); assert_eq!(names, vec!["ts", "value", "env"]); assert_eq!( schema.time_index, @@ -1028,7 +1028,7 @@ fn scan_schema_carries_ts_value_and_group_keys() { let QueryExpr::Scan { schema, .. } = find_scan(&qe) else { unreachable!() }; - let mut names: Vec<&str> = schema.columns.iter().map(|c| c.name.as_str()).collect(); + let mut names: Vec<&str> = schema.fields.iter().map(|c| c.name.as_str()).collect(); names.sort(); assert_eq!(names, vec!["service", "ts", "value"]); assert_eq!(schema.time_index, Some(0)); // ts @@ -1279,7 +1279,7 @@ fn histogram_quantiles_branches_are_union_compatible() { .map(|c| { c.output_schema() .expect("branch schema") - .columns + .fields .iter() .map(|c| c.name.clone()) .collect() @@ -1288,7 +1288,7 @@ fn histogram_quantiles_branches_are_union_compatible() { assert_eq!(shapes[0], shapes[1], "branches must be union-compatible"); assert_eq!(shapes[0], vec!["value".to_string(), "q".to_string()]); assert_eq!( - q.output_schema().expect("merged schema").columns.len(), + q.output_schema().expect("merged schema").fields.len(), 2, "the merged schema describes every branch" ); diff --git a/crates/frontend-promql/tests/univmon_candidates.rs b/crates/frontend-promql/tests/univmon_candidates.rs index 1c7083464..e0ce77fe5 100644 --- a/crates/frontend-promql/tests/univmon_candidates.rs +++ b/crates/frontend-promql/tests/univmon_candidates.rs @@ -9,8 +9,8 @@ use asap_aware_mapping::{Replacement, ReplacementStrategy, SketchAlgorithmStrate mod support; use asap_types::post_asap::{ compile_post_asap_dag, cse::share_common_summary_sub_dags, AccuracyError, BoundExpr, - CompositionOperator, ErrorMetric, ProbabilityExpr, ResultGuarantee, SketchAlgorithm, - SketchQuery, SummaryExpr, SummaryFamilyType, SummaryInputExpr, SummaryNode, + CompositionOperator, ErrorMetric, FieldDataType, ProbabilityExpr, ResultGuarantee, + SketchAlgorithm, SketchQuery, SummaryExpr, SummaryInputExpr, SummaryNode, }; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -20,10 +20,10 @@ struct TestEvidence; impl AccuracyModel for TestEvidence { fn local_guarantee( &self, - family: &SummaryFamilyType, + family: &FieldDataType, query: &SketchQuery, ) -> Option { - if matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) + if matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) && !matches!(query, SketchQuery::PointCount { .. }) { let mut guarantee = ResultGuarantee::exact("SYNTHETIC test evidence; not measured"); @@ -57,7 +57,7 @@ fn candidate(query: &str, accuracy: AccuracyTarget) -> Rc { .find_map(|candidate| { let Replacement::Summary(node) = candidate.replacement else { return None }; let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { return None }; - matches!(&summary_input.expr, SummaryExpr::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } + matches!(&summary_input.expr, SummaryExpr::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::UnivMon).then_some(node) }).expect("UnivMon candidate") } diff --git a/crates/frontend-sql/src/sql/collection_planning.rs b/crates/frontend-sql/src/sql/collection_planning.rs index eba207751..05f309470 100644 --- a/crates/frontend-sql/src/sql/collection_planning.rs +++ b/crates/frontend-sql/src/sql/collection_planning.rs @@ -4,7 +4,7 @@ use super::types::{arrow_to_dtype, dtype_to_arrow, scalar_value_to_asap}; use asap_types::pre_asap::scalar_signature::{ element_access_type, struct_field_type, MapScalarFunction, }; -use asap_types::pre_asap::{Column, QueryExpr, Schema}; +use asap_types::pre_asap::{Field, QueryExpr, Schema}; use datafusion::arrow::datatypes::DataType; use datafusion::common::{DataFusionError, ExprSchema, Result}; use datafusion::logical_expr::{ @@ -82,11 +82,11 @@ impl CollectionPlanningFunction { .into_iter() .enumerate() .map(|(index, (dtype, nullable))| { - Column::new(format!("argument_{index}"), dtype, nullable) + Field::plain(format!("argument_{index}"), dtype, nullable) }) .collect(), ); - let args = (0..schema.columns.len()) + let args = (0..schema.fields.len()) .map(|index| { if let Some(Expr::Literal(value)) = expressions.and_then(|args| args.get(index)) { diff --git a/crates/frontend-sql/src/sql/mod.rs b/crates/frontend-sql/src/sql/mod.rs index 52c065e9a..d1401988c 100644 --- a/crates/frontend-sql/src/sql/mod.rs +++ b/crates/frontend-sql/src/sql/mod.rs @@ -58,7 +58,7 @@ use asap_types::pre_asap::query_expr::{ UnresolvedQueryExpr as Unresolved, WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, }; -use asap_types::pre_asap::schema::{DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, FieldDataType, Schema}; use asap_types::pre_asap::{ resolve_column_ref, resolve_root, ColumnRef, CompareOpKind, JoinKind, RelationalSetOpKind, ScalarValue, WindowFuncKind, @@ -558,8 +558,8 @@ impl<'a> SqlLowerer<'a> { .get(table) .ok_or_else(|| LoweringError::TableNotFound(table.to_string()))?; let qualified = Schema { - columns: schema - .columns + fields: schema + .fields .iter() .cloned() .map(|c| c.with_table(qualifier)) @@ -941,8 +941,8 @@ impl<'a> SqlLowerer<'a> { })?; if value_id == timestamp_id || !matches!( - input_schema.columns[value_id].dtype, - DataType::Int64 | DataType::Float64 + input_schema.fields[value_id].dtype, + FieldDataType::Plain(DataType::Int64 | DataType::Float64) ) { return Err(LoweringError::InvalidExpression(format!( diff --git a/crates/frontend-sql/src/sql/types.rs b/crates/frontend-sql/src/sql/types.rs index 22611e6d0..282d07711 100644 --- a/crates/frontend-sql/src/sql/types.rs +++ b/crates/frontend-sql/src/sql/types.rs @@ -5,11 +5,11 @@ use std::collections::HashMap; use datafusion::arrow::datatypes::{ - DataType as ArrowDataType, Field, Fields, Schema as ArrowSchema, + DataType as ArrowDataType, Field as ArrowField, Fields, Schema as ArrowSchema, }; use datafusion::common::ScalarValue as DfScalarValue; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::pre_asap::ScalarValue; use crate::error::SqlError as LoweringError; @@ -93,7 +93,7 @@ pub(super) fn arrow_to_dtype(dt: &ArrowDataType) -> Result Ok(DataType::Date), ArrowDataType::Interval(_) => Ok(DataType::Interval), ArrowDataType::List(element) => Ok(DataType::List { - element: Box::new(Column::new( + element: Box::new(Field::new( element.name(), arrow_to_dtype(element.data_type())?, element.is_nullable(), @@ -103,7 +103,7 @@ pub(super) fn arrow_to_dtype(dt: &ArrowDataType) -> Result ArrowDataType { DataType::Float64 => ArrowDataType::Float64, DataType::Utf8 => ArrowDataType::Utf8, DataType::Bool => ArrowDataType::Boolean, - DataType::List { element } => ArrowDataType::List(std::sync::Arc::new(Field::new( + DataType::List { element } => ArrowDataType::List(std::sync::Arc::new(ArrowField::new( &element.name, dtype_to_arrow(&element.dtype), element.nullable, @@ -150,7 +150,9 @@ pub(super) fn dtype_to_arrow(dt: &DataType) -> ArrowDataType { DataType::Struct { fields } => ArrowDataType::Struct( fields .iter() - .map(|field| Field::new(&field.name, dtype_to_arrow(&field.dtype), field.nullable)) + .map(|field| { + ArrowField::new(&field.name, dtype_to_arrow(&field.dtype), field.nullable) + }) .collect::>() .into(), ), @@ -159,12 +161,12 @@ pub(super) fn dtype_to_arrow(dt: &DataType) -> ArrowDataType { value, value_nullable, } => ArrowDataType::Map( - std::sync::Arc::new(Field::new( + std::sync::Arc::new(ArrowField::new( "entries", ArrowDataType::Struct( vec![ - Field::new("key", dtype_to_arrow(key), false), - Field::new("value", dtype_to_arrow(value), *value_nullable), + ArrowField::new("key", dtype_to_arrow(key), false), + ArrowField::new("value", dtype_to_arrow(value), *value_nullable), ] .into(), ), @@ -193,9 +195,11 @@ pub(super) fn dtype_to_arrow(dt: &DataType) -> ArrowDataType { /// Build an Arrow schema from a canonical [`Schema`] (column name + type + nullability). pub(super) fn schema_to_arrow(schema: &Schema) -> ArrowSchema { let fields: Fields = schema - .columns + .fields .iter() - .map(|c: &Column| Field::new(&c.name, dtype_to_arrow(&c.dtype), c.nullable)) + .map(|c: &Field| { + ArrowField::new(&c.name, dtype_to_arrow(c.expect_plain_dtype()), c.nullable) + }) .collect(); ArrowSchema::new(fields) } @@ -302,21 +306,21 @@ mod collection_tests { fn nested_collections_preserve_field_names_order_and_nullability() { let dtype = DataType::Struct { fields: vec![ - Column::new( + Field::new( "samples", DataType::List { - element: Box::new(Column::new( + element: Box::new(Field::new( "sample", DataType::Struct { fields: vec![ - Column::new("timestamp", DataType::Timestamp, false), - Column::new("value", DataType::Float64, true), - Column::new( + Field::new("timestamp", DataType::Timestamp, false), + Field::new("value", DataType::Float64, true), + Field::new( "labels", DataType::Map { key: Box::new(DataType::Utf8), value: Box::new(DataType::List { - element: Box::new(Column::new( + element: Box::new(Field::new( "label_value", DataType::Utf8, false, @@ -333,7 +337,7 @@ mod collection_tests { }, false, ), - Column::new("optional", DataType::Int64, true), + Field::new("optional", DataType::Int64, true), ], }; let arrow = dtype_to_arrow(&dtype); @@ -345,7 +349,7 @@ mod collection_tests { #[test] fn empty_struct_and_nonnullable_list_element_roundtrip() { let dtype = DataType::List { - element: Box::new(Column::new( + element: Box::new(Field::new( "empty", DataType::Struct { fields: vec![] }, false, diff --git a/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs b/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs index ae27e31e0..23f757985 100644 --- a/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs +++ b/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs @@ -36,7 +36,7 @@ //! coverage so a regression (or a future improvement) is visible, not silent. use asap_frontend_sql::{lower_sql_dialect, SqlCatalog, SqlError as LoweringError}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; @@ -44,8 +44,8 @@ use datafusion::error::DataFusionError; const CORPUS: &str = include_str!("data/bgp_analytics.sql"); -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } /// `bgp_updates(timestamp, collector, peer_ip, peer_asn, prefix, operation, diff --git a/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs b/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs index 302a349ad..6e382bb57 100644 --- a/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs +++ b/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs @@ -14,7 +14,7 @@ //! tally** by outcome category -- see the module doc on [`Category`] for why. use asap_frontend_sql::{lower_sql_dialect, SqlCatalog, SqlError}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use datafusion::error::DataFusionError; @@ -34,8 +34,8 @@ struct QueryCase { sql: String, } -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } /// `bgp.bgp_updates`, widened past the 7-column `bgp_analytics` schema with diff --git a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs index 189ff0ff7..caed56594 100644 --- a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs +++ b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs @@ -20,14 +20,14 @@ //! flow / 5-tuple = `(srcip, dstip, srcport, dstport, proto)`. use asap_frontend_sql::{lower_sql, SqlCatalog, SqlError as LoweringError}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/synthetic_packet_trace_queries.sql"); -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } /// `packets(srcip, dstip, srcport, dstport, proto, time, pkt_len)`. IPs and diff --git a/crates/frontend-sql/tests/data_quality_check/tpch_deequ.rs b/crates/frontend-sql/tests/data_quality_check/tpch_deequ.rs index 679fbc95e..83cff5ec9 100644 --- a/crates/frontend-sql/tests/data_quality_check/tpch_deequ.rs +++ b/crates/frontend-sql/tests/data_quality_check/tpch_deequ.rs @@ -23,13 +23,13 @@ //! which is the same narrowing sidra's own catalog file makes. use asap_frontend_sql::{lower_sql, SqlCatalog, SqlError as LoweringError}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/tpch_deequ_queries.sql"); -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } /// No `time_index` and no `unique_keys`: the checks do not slice by time, and diff --git a/crates/frontend-sql/tests/maintained_population.rs b/crates/frontend-sql/tests/maintained_population.rs index 5d73d7e90..c8269fd2a 100644 --- a/crates/frontend-sql/tests/maintained_population.rs +++ b/crates/frontend-sql/tests/maintained_population.rs @@ -7,7 +7,7 @@ use asap_types::{ maintained_population::{MaintainedPopulation, PopulationInput}, share_common_summary_sub_dags, SummaryExpr, ValueOperation, }, - pre_asap::{Column, DataType, QueryExpr, Schema}, + pre_asap::{DataType, Field, QueryExpr, Schema}, types::AccuracyTarget, }; use std::rc::Rc; @@ -16,8 +16,8 @@ async fn aggregate(q: &str) -> Rc { let catalog = SqlCatalog::new().with_table( "samples", Schema::new(vec![ - Column::new("latency", DataType::Float64, false), - Column::new("job", DataType::Utf8, false), + Field::plain("latency", DataType::Float64, false), + Field::plain("job", DataType::Utf8, false), ]), ); let root = lower_sql(q, &catalog, AccuracyTarget::Exact).await.unwrap(); diff --git a/crates/frontend-sql/tests/netflow/netflow.rs b/crates/frontend-sql/tests/netflow/netflow.rs index dbe315381..22236b850 100644 --- a/crates/frontend-sql/tests/netflow/netflow.rs +++ b/crates/frontend-sql/tests/netflow/netflow.rs @@ -5,14 +5,14 @@ //! optional `ORDER BY`/`LIMIT`, plus the nested aggregate shape. use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/netflow.sql"); -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } fn catalog() -> SqlCatalog { diff --git a/crates/frontend-sql/tests/pearson_corr.rs b/crates/frontend-sql/tests/pearson_corr.rs index ae959560d..618f6c38f 100644 --- a/crates/frontend-sql/tests/pearson_corr.rs +++ b/crates/frontend-sql/tests/pearson_corr.rs @@ -2,14 +2,14 @@ use std::rc::Rc; use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_types::pre_asap::{AggIntent, Column, DataType, QueryExpr, Schema}; +use asap_types::pre_asap::{AggIntent, DataType, Field, QueryExpr, Schema}; use asap_types::types::AccuracyTarget; fn catalog() -> SqlCatalog { let schema = Schema::new(vec![ - Column::new("x", DataType::Float64, true), - Column::new("y", DataType::Float64, true), - Column::new("g", DataType::Int64, false), + Field::plain("x", DataType::Float64, true), + Field::plain("y", DataType::Float64, true), + Field::plain("g", DataType::Int64, false), ]); SqlCatalog::new() .with_table("a", schema.clone()) @@ -54,7 +54,7 @@ async fn corr_materializes_both_arguments() { .iter() .any(|col| !matches!(col.expr, QueryExpr::Column(_)))); assert_eq!( - query.output_schema().unwrap().columns[0].dtype, + query.output_schema().unwrap().fields[0].dtype, DataType::Float64 ); } @@ -84,13 +84,13 @@ async fn corr_coexists_with_grouping_having_and_other_measures() { .unwrap(); let schema = child.output_schema().unwrap(); for id in pair.input_cols() { - assert!(id < schema.columns.len()); + assert!(id < schema.fields.len()); } assert!(measures.iter().any(|m| matches!(m, AggIntent::Sum { .. }))); let output = query.output_schema().unwrap(); - assert_eq!(output.columns[1].name, "r"); - assert_eq!(output.columns[1].dtype, DataType::Float64); - assert!(output.columns[1].nullable); + assert_eq!(output.fields[1].name, "r"); + assert_eq!(output.fields[1].dtype, DataType::Float64); + assert!(output.fields[1].nullable); } // Repeated inputs reuse their value while retaining two argument positions. diff --git a/crates/frontend-sql/tests/sql_lowering.rs b/crates/frontend-sql/tests/sql_lowering.rs index e3e065e23..883966db9 100644 --- a/crates/frontend-sql/tests/sql_lowering.rs +++ b/crates/frontend-sql/tests/sql_lowering.rs @@ -6,7 +6,7 @@ //! DAG (the same resolver the PromQL path uses). use asap_frontend_sql::{lower_sql, lower_sql_dialect, SqlCatalog, SqlError as LoweringError}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, FieldDataType, Schema}; use asap_types::pre_asap::{ AggIntent, CompareOpKind, GroupKeys, JoinKind, Predicate, QueryExpr, Reduction, ScalarValue, Source, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, @@ -14,8 +14,8 @@ use asap_types::pre_asap::{ use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } /// `metrics(ts, service, latency, bytes)` + `hosts(service, region)`. @@ -177,7 +177,7 @@ fn reducer_input_names(qe: &QueryExpr) -> (Vec, bool) { let names = measures .iter() .flat_map(|a| a.input_cols()) - .map(|id| schema.columns[id].name.clone()) + .map(|id| schema.fields[id].name.clone()) .collect(); (names, matches!(**child, QueryExpr::Project { .. })) } @@ -257,14 +257,14 @@ async fn projection_over_aggregate_resolves_output_types_via_output_names() { let schema = qe .output_schema() .expect("root projection schema derivation"); - assert_eq!(schema.columns.len(), 2); + assert_eq!(schema.fields.len(), 2); assert_eq!( - schema.columns[0].dtype, + schema.fields[0].dtype, DataType::Int64, "SUM(bytes:Int64) resolves to Int64, not the Utf8 fallback" ); assert_eq!( - schema.columns[1].dtype, + schema.fields[1].dtype, DataType::Float64, "AVG(latency) resolves to Float64" ); @@ -285,13 +285,13 @@ async fn single_agg_group_by_keeps_key_in_output_schema() { // Both the group key and the aggregate resolve in the root projection schema. let schema = qe.output_schema().expect("root projection schema"); - assert_eq!(schema.columns.len(), 2); + assert_eq!(schema.fields.len(), 2); assert_eq!( - schema.columns[0].dtype, + schema.fields[0].dtype, DataType::Utf8, "service is in the output" ); - assert_eq!(schema.columns[1].dtype, DataType::Int64, "SUM(bytes)"); + assert_eq!(schema.fields[1].dtype, DataType::Int64, "SUM(bytes)"); } #[tokio::test] @@ -473,7 +473,7 @@ fn join_eq_columns(join: &QueryExpr) -> [usize; 2] { cols.sort_unstable(); cols } - other => panic!("expected Column = Column, got {other:?}"), + other => panic!("expected Field = Field, got {other:?}"), } } @@ -648,7 +648,7 @@ fn join_parts(qe: &QueryExpr) -> (&JoinKind, &QueryExpr, usize) { else { unreachable!() }; - let left_len = left.output_schema().expect("left schema").columns.len(); + let left_len = left.output_schema().expect("left schema").fields.len(); (kind, pred.0.as_ref(), left_len) } @@ -685,7 +685,7 @@ async fn a_semi_join_outputs_only_the_left_schema() { let names: Vec<_> = join .output_schema() .expect("join schema") - .columns + .fields .iter() .map(|c| c.name.clone()) .collect(); @@ -789,9 +789,9 @@ async fn window_function_lowers_to_positional_windowfunc() { // and the enclosing projection resolves it (output_name threading). let schema = qe.output_schema().expect("root schema"); assert!( - schema.columns.iter().any(|c| c.dtype == DataType::Int64), + schema.fields.iter().any(|c| c.dtype == DataType::Int64), "row_number output column present, got {:?}", - schema.columns + schema.fields ); } @@ -999,7 +999,7 @@ async fn derived_table_aggregate_over_aggregate_nests() { ); // The whole nested DAG's output schema derives without error (positional // resolution is total across the derived-table boundary). - assert_eq!(qe.output_schema().unwrap().columns.len(), 1); + assert_eq!(qe.output_schema().unwrap().fields.len(), 1); } #[tokio::test] @@ -1297,9 +1297,9 @@ async fn time_bucketing_group_by_lowers_to_a_derived_key() { let schema = child.output_schema().expect("child schema"); assert_eq!(reduction, &Reduction::by(vec![0])); assert!( - schema.columns[0].name.contains("date_trunc"), + schema.fields[0].name.contains("date_trunc"), "group key should be the projected bucket, got {:?}", - schema.columns[0].name + schema.fields[0].name ); // The reducer still binds its own column, not the bucket. assert!(matches!( @@ -1363,7 +1363,7 @@ async fn a_shared_expression_is_materialized_once() { unreachable!() }; assert_eq!( - child.output_schema().expect("child schema").columns.len(), + child.output_schema().expect("child schema").fields.len(), 1, "the two reducers should share one derived column" ); @@ -1401,7 +1401,7 @@ fn grouping_levels(qe: &QueryExpr) -> Vec<(GroupKeys, Vec)> { let names = b .output_schema() .expect("level schema") - .columns + .fields .iter() .map(|c| c.name.clone()) .collect(); @@ -1466,9 +1466,9 @@ async fn omitted_grouping_keys_become_typed_nulls() { let schema = merge_branches(&qe)[1] .output_schema() .expect("level schema"); - assert_eq!(schema.columns[0].name, "service"); + assert_eq!(schema.fields[0].name, "service"); assert_eq!( - schema.columns[0].dtype, + schema.fields[0].dtype, DataType::Utf8, "the omitted key must keep its declared type" ); @@ -1485,7 +1485,7 @@ async fn grouping_levels_are_union_compatible() { .map(|b| { b.output_schema() .expect("level schema") - .columns + .fields .iter() .map(|c| (c.name.clone(), c.dtype.clone())) .collect::>() @@ -2071,7 +2071,7 @@ async fn current_timestamp_lowers_to_typed_current_timestamp_leaf() { }; assert!(matches!(&cols[0].expr, QueryExpr::CurrentTimestamp)); let schema = cols[0].expr.output_schema().expect("timestamp schema"); - assert_eq!(schema.columns[0].dtype, DataType::Timestamp); + assert_eq!(schema.fields[0].dtype, DataType::Timestamp); } // A `count` over a non-null input is a plain row count; over a nullable @@ -2082,8 +2082,8 @@ async fn count_null_semantics_become_a_measure_filter() { let catalog = SqlCatalog::new().with_table( "samples", Schema::new(vec![ - Column::new("nullable_value", DataType::Float64, true), - Column::new("value", DataType::Float64, false), + Field::plain("nullable_value", DataType::Float64, true), + Field::plain("value", DataType::Float64, false), ]), ); for sql in [ @@ -2164,7 +2164,7 @@ async fn grouped_map_column_preserves_map_type() { ) .await .unwrap(); - assert_eq!(query.output_schema().unwrap().columns[0].dtype, map); + assert_eq!(query.output_schema().unwrap().fields[0].dtype, map); } #[tokio::test] @@ -2172,9 +2172,9 @@ async fn clickhouse_modulo_uses_native_arithmetic_types_and_nullability() { let catalog = SqlCatalog::new().with_table( "numbers", Schema::new(vec![ - Column::new("i", DataType::Int64, false), - Column::new("n", DataType::Int64, true), - Column::new("f", DataType::Float64, false), + Field::plain("i", DataType::Int64, false), + Field::plain("n", DataType::Int64, true), + Field::plain("f", DataType::Float64, false), ]), ); for (call, native) in [ @@ -2217,8 +2217,8 @@ async fn clickhouse_modulo_uses_native_arithmetic_types_and_nullability() { .unwrap() .output_schema() .unwrap(); - assert_eq!(nullable.columns[0].dtype, DataType::Int64); - assert!(nullable.columns[0].nullable); + assert_eq!(nullable.fields[0].dtype, DataType::Int64); + assert!(nullable.fields[0].nullable); } #[tokio::test] @@ -2226,10 +2226,10 @@ async fn original_o11y_map_queries_lower_with_typed_results() { let catalog = SqlCatalog::new().with_table( "raw_samples", Schema::new(vec![ - Column::new("metric", DataType::Utf8, false), - Column::new("ts_ms", DataType::Int64, false), - Column::new("value", DataType::Float64, false), - Column::new( + Field::plain("metric", DataType::Utf8, false), + Field::plain("ts_ms", DataType::Int64, false), + Field::plain("value", DataType::Float64, false), + Field::plain( "labels", DataType::Map { key: Box::new(DataType::Utf8), @@ -2258,9 +2258,9 @@ async fn original_o11y_map_queries_lower_with_typed_results() { let schema = query.output_schema().unwrap(); assert!( schema - .columns + .fields .iter() - .any(|column| matches!(column.dtype, DataType::Map { .. })), + .any(|column| matches!(column.dtype, FieldDataType::Plain(DataType::Map { .. }))), "{schema:?}" ); } @@ -2287,7 +2287,7 @@ async fn clickhouse_modulo_preserves_projection_names_and_outer_references() { ) .await .unwrap(); - assert_eq!(query.output_schema().unwrap().columns[0].name, name); + assert_eq!(query.output_schema().unwrap().fields[0].name, name); } } @@ -2296,7 +2296,7 @@ async fn clickhouse_map_access_keeps_generated_names_and_rejects_variant_coercio let catalog = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new( + Field::plain( "labels", DataType::Map { key: Box::new(DataType::Utf8), @@ -2305,8 +2305,8 @@ async fn clickhouse_map_access_keeps_generated_names_and_rejects_variant_coercio }, false, ), - Column::new("integer", DataType::Int64, false), - Column::new("floating", DataType::Float64, false), + Field::plain("integer", DataType::Int64, false), + Field::plain("floating", DataType::Float64, false), ]), ); let query = lower_sql_dialect( @@ -2318,9 +2318,9 @@ async fn clickhouse_map_access_keeps_generated_names_and_rejects_variant_coercio .await .unwrap(); let output = query.output_schema().unwrap(); - assert_eq!(output.columns[0].name, "arrayElement(labels, 'job')"); - assert_eq!(output.columns[0].dtype, DataType::Utf8); - assert!(!output.columns[0].nullable); + assert_eq!(output.fields[0].name, "arrayElement(labels, 'job')"); + assert_eq!(output.fields[0].dtype, DataType::Utf8); + assert!(!output.fields[0].nullable); assert!(lower_sql_dialect( "SELECT map()['a'] FROM t", &catalog, @@ -2344,9 +2344,9 @@ async fn arg_selector_result_schema_tracks_selected_argument() { let catalog = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new("v", DataType::Float64, false), - Column::new("text", DataType::Utf8, true), - Column::new("ts", DataType::Int64, true), + Field::plain("v", DataType::Float64, false), + Field::plain("text", DataType::Utf8, true), + Field::plain("ts", DataType::Int64, true), ]), ); for (sql, dtype, nullable) in [ @@ -2370,8 +2370,8 @@ async fn arg_selector_result_schema_tracks_selected_argument() { .await .unwrap(); let schema = query.output_schema().unwrap(); - assert_eq!(schema.columns[0].dtype, dtype); - assert_eq!(schema.columns[0].nullable, nullable); + assert_eq!(schema.fields[0].dtype, dtype); + assert_eq!(schema.fields[0].nullable, nullable); } } @@ -2380,14 +2380,14 @@ async fn clickhouse_list_element_uses_canonical_typed_access() { let catalog = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new( + Field::plain( "samples", DataType::List { - element: Box::new(Column::new("item", DataType::Int64, false)), + element: Box::new(Field::new("item", DataType::Int64, false)), }, false, ), - Column::new("index", DataType::Int64, true), + Field::plain("index", DataType::Int64, true), ]), ); for (sql, nullable) in [ @@ -2404,8 +2404,8 @@ async fn clickhouse_list_element_uses_canonical_typed_access() { .await .unwrap(); let output = query.output_schema().unwrap(); - assert_eq!(output.columns[0].dtype, DataType::Int64); - assert_eq!(output.columns[0].nullable, nullable); + assert_eq!(output.fields[0].dtype, DataType::Int64); + assert_eq!(output.fields[0].nullable, nullable); let serialized = serde_json::to_string(&query).unwrap(); assert!(serialized.contains("asap_element_access"), "{serialized}"); } @@ -2429,17 +2429,17 @@ async fn clickhouse_tuple_element_preserves_declared_field_metadata() { let catalog = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new( + Field::plain( "sample", DataType::Struct { fields: vec![ - Column::new("time", DataType::Int64, false), - Column::new("value", DataType::Float64, true), + Field::new("time", DataType::Int64, false), + Field::new("value", DataType::Float64, true), ], }, false, ), - Column::new("index", DataType::Int64, false), + Field::plain("index", DataType::Int64, false), ]), ); for (sql, dtype, nullable) in [ @@ -2463,8 +2463,8 @@ async fn clickhouse_tuple_element_preserves_declared_field_metadata() { .await .unwrap(); let output = query.output_schema().unwrap(); - assert_eq!(output.columns[0].dtype, dtype); - assert_eq!(output.columns[0].nullable, nullable); + assert_eq!(output.fields[0].dtype, dtype); + assert_eq!(output.fields[0].nullable, nullable); assert!(serde_json::to_string(&query) .unwrap() .contains("asap_struct_field")); @@ -2490,9 +2490,9 @@ async fn clickhouse_tuple_element_preserves_declared_field_metadata() { async fn corr_result_is_nullable_float() { let query = lower("SELECT corr(latency, bytes) AS correlation FROM metrics").await; let schema = query.output_schema().unwrap(); - assert_eq!(schema.columns[0].name, "correlation"); - assert_eq!(schema.columns[0].dtype, DataType::Float64); - assert!(schema.columns[0].nullable); + assert_eq!(schema.fields[0].name, "correlation"); + assert_eq!(schema.fields[0].dtype, DataType::Float64); + assert!(schema.fields[0].nullable); } // A multi-column DISTINCT counts tuples; one column stays the single-column @@ -2502,8 +2502,8 @@ async fn composite_distinct_counts_tuples() { let cat = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new("a", DataType::Int64, false), - Column::new("b", DataType::Int64, false), + Field::plain("a", DataType::Int64, false), + Field::plain("b", DataType::Int64, false), ]), ); let composite = lower_sql( @@ -2548,8 +2548,8 @@ async fn composite_distinct_rejects_expression_arguments() { let cat = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new("a", DataType::Int64, false), - Column::new("b", DataType::Int64, false), + Field::plain("a", DataType::Int64, false), + Field::plain("b", DataType::Int64, false), ]), ); let error = lower_sql( @@ -2572,8 +2572,8 @@ async fn distinct_with_derived_sibling() { let catalog = SqlCatalog::new().with_table( "t", Schema::new(vec![ - Column::new("a", DataType::Int64, false), - Column::new("b", DataType::Int64, false), + Field::plain("a", DataType::Int64, false), + Field::plain("b", DataType::Int64, false), ]), ); for sql in [ @@ -2689,7 +2689,7 @@ async fn measure_filter_columns_survive_a_derived_column_projection() { let QueryExpr::Column(id) = left.as_ref() else { panic!("{left:?}"); }; - assert_eq!(child.output_schema().unwrap().columns[*id].name, "latency"); + assert_eq!(child.output_schema().unwrap().fields[*id].name, "latency"); } // `GROUP BY ROLLUP` fans one measure list out into one `Aggregate` per level; diff --git a/crates/frontend-sql/tests/temporal_types.rs b/crates/frontend-sql/tests/temporal_types.rs index bdb53d246..a6b4f1187 100644 --- a/crates/frontend-sql/tests/temporal_types.rs +++ b/crates/frontend-sql/tests/temporal_types.rs @@ -1,12 +1,12 @@ use asap_frontend_sql::{lower_sql, SqlCatalog}; use asap_types::{ - pre_asap::schema::{Column, DataType, Schema}, + pre_asap::schema::{DataType, Field, Schema}, types::AccuracyTarget, }; fn catalog() -> SqlCatalog { SqlCatalog::new().with_table( "t", - Schema::new(vec![Column::new("d", DataType::Date, false)]), + Schema::new(vec![Field::plain("d", DataType::Date, false)]), ) } // Unsupported fixed-duration results fail lowering instead of acquiring a float schema. @@ -14,7 +14,7 @@ fn catalog() -> SqlCatalog { async fn temporal_subtraction_rejects_unrepresentable_duration() { for dtype in [DataType::Date, DataType::Timestamp] { let catalog = - SqlCatalog::new().with_table("t", Schema::new(vec![Column::new("d", dtype, false)])); + SqlCatalog::new().with_table("t", Schema::new(vec![Field::plain("d", dtype, false)])); let error = lower_sql( "SELECT d - d AS elapsed FROM t", &catalog, @@ -39,7 +39,7 @@ async fn date_shifts_keep_their_type() { .await .unwrap(); assert_eq!( - node.output_schema().unwrap().columns[0].dtype, + node.output_schema().unwrap().fields[0].dtype, DataType::Date ); } @@ -55,7 +55,7 @@ async fn interval_cast_lowers_like_interval_literal() { .await .unwrap(); assert_eq!( - node.output_schema().unwrap().columns[0].dtype, + node.output_schema().unwrap().fields[0].dtype, DataType::Interval ); } @@ -91,7 +91,7 @@ async fn negative_intervals_keep_their_type() { .await .unwrap(); assert_eq!( - node.output_schema().unwrap().columns[0].dtype, + node.output_schema().unwrap().fields[0].dtype, DataType::Interval, "{query}" ); @@ -109,7 +109,7 @@ async fn sql_date_literals_keep_their_type() { .await .unwrap(); assert_eq!( - node.output_schema().unwrap().columns[0].dtype, + node.output_schema().unwrap().fields[0].dtype, DataType::Date ); } diff --git a/crates/integration-tests/src/lib.rs b/crates/integration-tests/src/lib.rs index 59602ac07..be8e259bc 100644 --- a/crates/integration-tests/src/lib.rs +++ b/crates/integration-tests/src/lib.rs @@ -13,7 +13,7 @@ pub mod fixtures { use asap_frontend_promql::lower_promql_workload; - use asap_types::pre_asap::schema::{Column, DataType, Schema}; + use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; use asap_types::workload::{ @@ -55,16 +55,16 @@ pub mod fixtures { Ok(lowered.remove(0)) } - pub fn ts_col() -> Column { - Column::new("ts", DataType::Timestamp, false) + pub fn ts_col() -> Field { + Field::plain("ts", DataType::Timestamp, false) } - pub fn value_col() -> Column { - Column::new("value", DataType::Float64, false) + pub fn value_col() -> Field { + Field::plain("value", DataType::Float64, false) } - pub fn label_col(name: &str) -> Column { - Column::new(name, DataType::Utf8, true) + pub fn label_col(name: &str) -> Field { + Field::plain(name, DataType::Utf8, true) } /// Canonical PromQL leaf schema: `(ts: Timestamp, value: Float64)` plus @@ -74,7 +74,7 @@ pub mod fixtures { let mut cols = vec![ts_col(), value_col()]; cols.extend(labels.iter().map(|n| label_col(n))); Schema { - columns: cols, + fields: cols, time_index: Some(0), unique_keys: vec![], // Schemaless PromQL leaf: open (the metric's full label set is diff --git a/crates/integration-tests/tests/exact_composition.rs b/crates/integration-tests/tests/exact_composition.rs index 7e2e7c12a..5067a6112 100644 --- a/crates/integration-tests/tests/exact_composition.rs +++ b/crates/integration-tests/tests/exact_composition.rs @@ -27,21 +27,25 @@ use asap_integration_tests::fixtures::lower_promql; use asap_types::dag_export; use asap_types::post_asap::{ validate_execution_data_states, ExactKind, ExactOperation, ExecutionDataState, ExecutionTiming, - SketchAlgorithm, SummaryExpr, SummaryFamilyType, SummaryNode, SummaryUpdate, + FieldDataType, SketchAlgorithm, SummaryExpr, SummaryNode, SummaryUpdate, }; use asap_types::pre_asap::agg_intent::{default_quantile, AggIntent}; use asap_types::pre_asap::query_expr::{QueryExpr, Reduction, Source}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; // ── fixtures ──────────────────────────────────────────────────────────── fn metric_scan(labels: &[&str]) -> QueryExpr { let mut columns = vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ]; - columns.extend(labels.iter().map(|n| Column::new(*n, DataType::Utf8, true))); + columns.extend( + labels + .iter() + .map(|n| Field::plain(*n, DataType::Utf8, true)), + ); QueryExpr::Scan { source: Source::TimeSeries { metric: "latency".into(), @@ -101,7 +105,7 @@ fn custom_accuracy_rule_survives_root_target_and_materialization() { } fn local_guarantee( &self, - family: &SummaryFamilyType, + family: &FieldDataType, query: &SketchQuery, ) -> Option { DefaultAccuracyModel.local_guarantee(family, query) @@ -284,7 +288,7 @@ fn is_plain(node: &SummaryNode) -> bool { node.schema .fields .iter() - .all(|f| matches!(f.dtype, SummaryFamilyType::Plain(_))) + .all(|f| matches!(f.dtype, FieldDataType::Plain(_))) } fn names(node: &SummaryNode) -> Vec<&str> { @@ -376,7 +380,7 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { assert!( matches!( &child.expr, - SummaryExpr::SummaryAgg { family: SummaryFamilyType::ExactAggregate(k, _), .. } if *k == kind + SummaryExpr::SummaryAgg { family: FieldDataType::ExactAggregate(k, _), .. } if *k == kind ), "{kind:?}: expected the exact accumulator under its finalization, got {:?}", child.expr @@ -470,7 +474,7 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { names(&composed), root.output_schema() .unwrap() - .columns + .fields .iter() .map(|c| c.name.as_str()) .collect::>(), @@ -684,7 +688,7 @@ fn summary_construction_follows_its_value_input_phase() { let illegal = Rc::new(SummaryNode { expr: SummaryExpr::SummaryAgg { child: post, - family: SummaryFamilyType::ExactAggregate( + family: FieldDataType::ExactAggregate( ExactKind::Max, asap_types::post_asap::ExactParams::Max, ), @@ -693,10 +697,7 @@ fn summary_construction_follows_its_value_input_phase() { grouping: Default::default(), filter: None, }, - schema: asap_types::post_asap::SummarySchema { - fields: vec![], - time_index: None, - }, + schema: asap_types::post_asap::Schema::lifted(vec![], None), guarantee: None, }); let state = asap_types::post_asap::produced_data_state(&illegal.expr).unwrap(); @@ -780,7 +781,7 @@ fn dag_export_carries_explicit_stage_and_plain_schema_for_a_composed_plan() { // Pre-ASAP export of the same target still describes the same columns. let pre = dag_export::export(root); let pre_root = &pre.nodes[pre.root as usize]; - let pre_cols: Vec = pre_root.schema.as_ref().unwrap()["columns"] + let pre_cols: Vec = pre_root.schema.as_ref().unwrap()["fields"] .as_array() .unwrap() .iter() diff --git a/crates/integration-tests/tests/frontend_timestamps.rs b/crates/integration-tests/tests/frontend_timestamps.rs index 67c5418cc..388261da6 100644 --- a/crates/integration-tests/tests/frontend_timestamps.rs +++ b/crates/integration-tests/tests/frontend_timestamps.rs @@ -2,7 +2,7 @@ use asap_frontend_sql::{lower_sql, SqlCatalog}; use asap_integration_tests::fixtures::lower_promql; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; @@ -14,11 +14,11 @@ async fn promql_eval_time_and_sql_current_timestamp_remain_distinct() { let promql = lower_promql("time()", AccuracyTarget::Exact).expect("lower PromQL time()"); assert!(matches!(promql, QueryExpr::EvalTimestamp)); let promql_schema = promql.output_schema().expect("PromQL time() schema"); - assert_eq!(promql_schema.columns[0].dtype, DataType::Float64); + assert_eq!(promql_schema.fields[0].dtype, DataType::Float64); let catalog = SqlCatalog::new().with_table( "metrics", - Schema::new(vec![Column::new("value", DataType::Float64, false)]), + Schema::new(vec![Field::plain("value", DataType::Float64, false)]), ); let sql = lower_sql( "SELECT CURRENT_TIMESTAMP FROM metrics", @@ -35,5 +35,5 @@ async fn promql_eval_time_and_sql_current_timestamp_remain_distinct() { .expr .output_schema() .expect("SQL CURRENT_TIMESTAMP schema"); - assert_eq!(sql_schema.columns[0].dtype, DataType::Timestamp); + assert_eq!(sql_schema.fields[0].dtype, DataType::Timestamp); } diff --git a/crates/integration-tests/tests/kll_pane_execution.rs b/crates/integration-tests/tests/kll_pane_execution.rs index 52eea7e63..b7f51711c 100644 --- a/crates/integration-tests/tests/kll_pane_execution.rs +++ b/crates/integration-tests/tests/kll_pane_execution.rs @@ -11,8 +11,8 @@ use asap_physical_operators::{ }; use asap_types::{ post_asap::{ - SketchAlgorithm, SketchKind, SketchParams, SketchQuery, SummaryFamilyType, SummaryField, - SummarySchema, + Field, FieldDataType, Schema as LogicalSchema, SketchAlgorithm, SketchKind, SketchParams, + SketchQuery, }, pre_asap::DataType, }; @@ -25,17 +25,20 @@ use std::{ }, }; -fn family(k: u32) -> SummaryFamilyType { - SummaryFamilyType::Sketch( +fn family(k: u32) -> FieldDataType { + FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), Default::default(), ) } fn raw_schema() -> Schema { - Arc::new(SummarySchema { - fields: vec![SummaryField { + Arc::new(LogicalSchema { + closed: true, + unique_keys: vec![], + fields: vec![Field { + table: None, name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, }], time_index: None, diff --git a/crates/integration-tests/tests/precompute_raw_samples.rs b/crates/integration-tests/tests/precompute_raw_samples.rs index c348eed9f..4ce172709 100644 --- a/crates/integration-tests/tests/precompute_raw_samples.rs +++ b/crates/integration-tests/tests/precompute_raw_samples.rs @@ -18,8 +18,9 @@ use asap_physical_operators::{ AggregateCore, KeyByLabelValues, Statistic, }; use asap_types::post_asap::{ - compile_post_asap_dag, EntityIdentity, ExactKind, PostAsapDAG, PostAsapOperatorPayload, - SketchAlgorithm, SketchQuery, SummaryFamilyType, SummaryInputExpr, SummaryNode, SummaryUpdate, + compile_post_asap_dag, EntityIdentity, ExactKind, FieldDataType, PostAsapDAG, + PostAsapOperatorPayload, SketchAlgorithm, SketchQuery, SummaryInputExpr, SummaryNode, + SummaryUpdate, }; use asap_types::pre_asap::{expr_ir::ColumnRef, query_expr::Reduction}; use asap_types::types::AccuracyTarget; @@ -239,9 +240,9 @@ fn weight(update: &SummaryUpdate, value: f64) -> f64 { } /// Estimates that identify a state's content for comparison. -fn readouts(state: &dyn AggregateCore, family: &SummaryFamilyType) -> Vec { +fn readouts(state: &dyn AggregateCore, family: &FieldDataType) -> Vec { if let Some(exact) = state.as_any().downcast_ref::() { - let SummaryFamilyType::ExactAggregate(kind, _) = family else { + let FieldDataType::ExactAggregate(kind, _) = family else { unreachable!() }; let statistic = match kind { @@ -258,7 +259,7 @@ fn readouts(state: &dyn AggregateCore, family: &SummaryFamilyType) -> Vec { .unwrap() .unwrap()]; } - let SummaryFamilyType::Sketch(kind, _) = family else { + let FieldDataType::Sketch(kind, _) = family else { panic!("sketch state for exact family") }; match kind.algorithm() { @@ -309,7 +310,7 @@ fn check( .collect(), Reduction::PerEntity => vec![], }; - let stored_only = matches!(family, SummaryFamilyType::Sketch(kind, _) + let stored_only = matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &asap_types::post_asap::SketchAlgorithm::Cms); if stored_only || asap_physical_operators::capability::validate_native_family(family).is_err() { // Families without a native state (e.g. UnivMon), or with native @@ -320,11 +321,11 @@ fn check( } let actual = execute(dag, source, root, rows); let label = match family { - SummaryFamilyType::ExactAggregate(kind, _) => format!("{kind:?}"), - SummaryFamilyType::Sketch(kind, _) => format!("{:?}", kind.algorithm()), + FieldDataType::ExactAggregate(kind, _) => format!("{kind:?}"), + FieldDataType::Sketch(kind, _) => format!("{:?}", kind.algorithm()), other => format!("{other:?}"), }; - if let SummaryFamilyType::Sketch(kind, _) = family { + if let FieldDataType::Sketch(kind, _) = family { if let (Some(keyed), false) = (&input.item, kind.algorithm() == &SketchAlgorithm::Hll) { // Keyed heaps: every item's estimated weight is its exact // total at this scale (no collisions in the fixture). @@ -458,7 +459,7 @@ fn raw_sample_summaries_compile_and_match_their_kernels() { /// Replace the raw summary of `sum by (service) (sum_over_time(m[5m]))` with /// another update, keeping its raw input and reduction. -fn grouped_raw_summary(family: SummaryFamilyType, input: SummaryUpdate) -> (PostAsapDAG, u64, u64) { +fn grouped_raw_summary(family: FieldDataType, input: SummaryUpdate) -> (PostAsapDAG, u64, u64) { let candidate = candidates( "sum by (service) (sum_over_time(m[5m]))", AccuracyTarget::Exact, @@ -507,7 +508,7 @@ fn raw_sample_heaps_resolve_items_from_labels() { SketchParams, WeightDomain, }; let heap = |algorithm, params| { - SummaryFamilyType::Sketch(SketchKind::new(algorithm, params), PerSubpopulationInstance) + FieldDataType::Sketch(SketchKind::new(algorithm, params), PerSubpopulationInstance) }; let cms = heap( SketchAlgorithm::CmsWithHeap, @@ -588,7 +589,7 @@ fn raw_sample_heaps_resolve_items_from_labels() { fn raw_sample_without_grouping_drops_labels_and_name() { use asap_types::pre_asap::query_expr::GroupKeys; let family = - SummaryFamilyType::ExactAggregate(ExactKind::Sum, asap_types::post_asap::ExactParams::Sum); + FieldDataType::ExactAggregate(ExactKind::Sum, asap_types::post_asap::ExactParams::Sum); let (mut dag, source, root) = grouped_raw_summary(family, SummaryUpdate::column(ColumnRef::SampleValue)); let service = dag diff --git a/crates/integration-tests/tests/promql_numeric_regressions.rs b/crates/integration-tests/tests/promql_numeric_regressions.rs index bec86289d..9fbb2a9bb 100644 --- a/crates/integration-tests/tests/promql_numeric_regressions.rs +++ b/crates/integration-tests/tests/promql_numeric_regressions.rs @@ -3,8 +3,8 @@ use asap_aware_mapping::{Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG}; use asap_integration_tests::fixtures::lower_promql; use asap_types::post_asap::{ - compile_post_asap_dag, ExactKind, SummaryExpr, SummaryFamilyType, SummaryInputExpr, - SummaryNode, SummaryUpdate, + compile_post_asap_dag, ExactKind, FieldDataType, SummaryExpr, SummaryInputExpr, SummaryNode, + SummaryUpdate, }; use asap_types::pre_asap::{ColumnRef, Reduction}; use asap_types::types::AccuracyTarget; @@ -21,7 +21,7 @@ fn plan(query: &str, accuracy: AccuracyTarget) -> Rc { }) .unwrap_or_else(|| asap_aware_mapping::replacement::keep_pre_asap(&pre).unwrap()) } -fn aggregate(node: &SummaryNode) -> (&SummaryFamilyType, &SummaryUpdate, &Reduction) { +fn aggregate(node: &SummaryNode) -> (&FieldDataType, &SummaryUpdate, &Reduction) { match &node.expr { SummaryExpr::SummaryAgg { family, @@ -34,11 +34,8 @@ fn aggregate(node: &SummaryNode) -> (&SummaryFamilyType, &SummaryUpdate, &Reduct other => panic!("not a maintained accumulator: {other:?}"), } } -fn contribution(family: &SummaryFamilyType, update: &SummaryUpdate, value: f64) -> f64 { - if matches!( - family, - SummaryFamilyType::ExactAggregate(ExactKind::Count, _) - ) { +fn contribution(family: &FieldDataType, update: &SummaryUpdate, value: f64) -> f64 { + if matches!(family, FieldDataType::ExactAggregate(ExactKind::Count, _)) { return 1.; } match update.weight { @@ -55,7 +52,7 @@ fn count_up_counts_targets_even_when_values_repeat_or_change_sign() { let (family, update, _) = aggregate(&node); assert!(matches!( family, - SummaryFamilyType::ExactAggregate(ExactKind::Count, _) + FieldDataType::ExactAggregate(ExactKind::Count, _) )); for values in [[1., 1., 1.], [1., 1., 0.], [0., 0., 0.], [-1., -1., -1.]] { assert_eq!( @@ -78,7 +75,7 @@ fn window_counts_and_sums_distinguish_one_zero_three_and_negative_values() { ] { let node = plan(query, AccuracyTarget::Exact); let (family, update, reduction) = aggregate(&node); - assert!(matches!(family, SummaryFamilyType::ExactAggregate(k, _) if *k == kind)); + assert!(matches!(family, FieldDataType::ExactAggregate(k, _) if *k == kind)); assert_eq!(*reduction, Reduction::PerEntity); for value in [1., 0., 3., -3.] { let got: f64 = (0..10).map(|_| contribution(family, update, value)).sum(); @@ -98,7 +95,7 @@ fn sum_rate_and_increase_have_real_exact_accumulator_nodes() { ] { let node = plan(query, AccuracyTarget::Exact); let (family, _, _) = aggregate(&node); - assert!(matches!(family, SummaryFamilyType::ExactAggregate(k, _) if *k == kind)); + assert!(matches!(family, FieldDataType::ExactAggregate(k, _) if *k == kind)); assert!(node.guarantee.as_ref().unwrap().is_exact()); compile_post_asap_dag(&node).unwrap(); } @@ -173,7 +170,7 @@ impl asap_aware_mapping::accuracy::AccuracyEvidenceProvider for OneKeyTopKEviden fn propagation_stats( &self, op: &asap_types::post_asap::CompositionOperator, - _family: &SummaryFamilyType, + _family: &FieldDataType, _query: Option<&asap_types::post_asap::SketchQuery>, ) -> asap_aware_mapping::accuracy::PropagationStats { // Single-key fixture: no excluded keys; bounds cover every value below. @@ -224,7 +221,7 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { return None; }; let (family, _, _) = aggregate(node); - matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &wanted) + matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &wanted) .then_some(node) }) .expect("weighted sketch candidate"); @@ -245,7 +242,7 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { for c in &candidates { if let Replacement::Summary(n) = &c.replacement { assert!( - !matches!(aggregate(n).0, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::CmsWithHeap) + !matches!(aggregate(n).0, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::CmsWithHeap) ); } } diff --git a/crates/integration-tests/tests/promql_to_post_asap.rs b/crates/integration-tests/tests/promql_to_post_asap.rs index 73f781443..d48ef6078 100644 --- a/crates/integration-tests/tests/promql_to_post_asap.rs +++ b/crates/integration-tests/tests/promql_to_post_asap.rs @@ -21,8 +21,8 @@ use asap_aware_mapping::{ use asap_integration_tests::fixtures::lower_promql; use asap_types::post_asap::{ compile_post_asap_dag, CompositionOperator, EntityIdentity, ExactKind, ExactParams, - GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, SummaryExpr, - SummaryFamilyType, SummaryInputExpr, SummaryNode, SummarySchema, SummaryUpdate, ValueOperation, + FieldDataType, GroupingStrategy, Schema, SketchAlgorithm, SketchKind, SketchParams, + SketchQuery, SummaryExpr, SummaryInputExpr, SummaryNode, SummaryUpdate, ValueOperation, }; use asap_types::pre_asap::expr_ir::ColumnRef; use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; @@ -67,7 +67,7 @@ fn distinct_over_time_offers_hll_cardinality_readout() { let Replacement::Summary(node) = &candidate.replacement else { return false }; let SummaryExpr::SummaryEstimate { summary_input, query, .. } = &node.expr else { return false }; matches!(query, SketchQuery::Cardinality) - && matches!(&summary_input.expr, SummaryExpr::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } + && matches!(&summary_input.expr, SummaryExpr::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::Hll) }), "no HLL cardinality candidate: {candidates:?}"); } @@ -122,7 +122,7 @@ fn value_ranked_topk_preserves_summary_children_in_post_asap_dag() { .schema .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); } } @@ -168,7 +168,7 @@ fn instant_topk_and_unsupported_child_remain_local_residuals() { } } -fn dtype<'a>(schema: &'a SummarySchema, name: &str) -> &'a SummaryFamilyType { +fn dtype<'a>(schema: &'a Schema, name: &str) -> &'a FieldDataType { &schema .fields .iter() @@ -254,7 +254,7 @@ impl AccuracyEvidenceProvider for SeparatedTopK { fn propagation_stats( &self, op: &CompositionOperator, - _family: &SummaryFamilyType, + _family: &FieldDataType, _query: Option<&SketchQuery>, ) -> PropagationStats { matches!(op, CompositionOperator::TopKSelection) @@ -298,7 +298,7 @@ fn grouped_rate_topk_consumes_finalized_rate_values() { asap_types::post_asap::PostAsapOperatorPayload::RelationalJoin { .. } ))); let node = dag.nodes.iter().find(|node| matches!(&node.payload, - asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } + asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::CmsWithHeap)).unwrap(); assert_eq!( node.output_state.timing, @@ -326,7 +326,7 @@ fn weighted_topk_keeps_candidates_with_missing_population_evidence() { fn propagation_stats( &self, op: &CompositionOperator, - family: &SummaryFamilyType, + family: &FieldDataType, query: Option<&SketchQuery>, ) -> PropagationStats { SeparatedTopK.propagation_stats(op, family, query) @@ -723,7 +723,7 @@ fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() else { panic!("expected structured Top-K state input") }; - let SummaryFamilyType::Sketch(kind, _) = family else { + let FieldDataType::Sketch(kind, _) = family else { panic!("expected a heap-backed sketch family, got {family:?}") }; assert_eq!(format!("{:?}", kind.algorithm()), expected_family); @@ -915,7 +915,7 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { assert!(matches!(query, SketchQuery::Quantile { q } if *q == 0.99)); assert_eq!( dtype(&root.schema, "quantile_0_99"), - &SummaryFamilyType::Plain(DataType::Float64), + &FieldDataType::Plain(DataType::Float64), "the summary-state type must not propagate past the estimate" ); @@ -936,7 +936,7 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { }; assert_eq!( family, - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) @@ -949,7 +949,7 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { ); assert_eq!( dtype(&summary_input.schema, "value"), - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) @@ -978,12 +978,12 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { }; assert_eq!( family, - &SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) + &FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) ); assert_eq!(reduction, &Reduction::PerEntity); assert_eq!( dtype(&child.schema, "value"), - &SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) + &FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate) ); assert_eq!( child.schema.time_index, @@ -1004,7 +1004,7 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { leaf.schema .fields .iter() - .all(|f| matches!(f.dtype, SummaryFamilyType::Plain(_))), + .all(|f| matches!(f.dtype, FieldDataType::Plain(_))), "logical edges carry only plain columns" ); } @@ -1025,7 +1025,7 @@ fn promql_exact_workload_binds_accumulators_not_sketches() { }; assert_eq!( family, - &SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + &FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) ); assert_eq!( reduction, @@ -1034,7 +1034,7 @@ fn promql_exact_workload_binds_accumulators_not_sketches() { ); assert_eq!( dtype(&root.schema, "job"), - &SummaryFamilyType::Plain(DataType::Utf8), + &FieldDataType::Plain(DataType::Utf8), "group keys pass through verbatim" ); @@ -1141,7 +1141,7 @@ fn nested_summary_explicitly_finalizes_exact_child_at_ingestion_time() { assert!(matches!( source.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + family: FieldDataType::ExactAggregate(ExactKind::Sum, _), .. } )); @@ -1149,12 +1149,12 @@ fn nested_summary_explicitly_finalizes_exact_child_at_ingestion_time() { .schema .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); assert!(child .schema .fields .iter() - .any(|field| matches!(field.dtype, SummaryFamilyType::Plain(DataType::Float64)))); + .any(|field| matches!(field.dtype, FieldDataType::Plain(DataType::Float64)))); compile_post_asap_dag(&plan).expect("explicit boundary is a valid post-ASAP DAG"); } @@ -1370,7 +1370,7 @@ fn ddsketch_ratio_bound_holds_for_signed_pinned_sketch_readouts() { panic!("readout") }; let SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + family: FieldDataType::Sketch(kind, _), .. } = &summary_input.expr else { diff --git a/crates/integration-tests/tests/schema.rs b/crates/integration-tests/tests/schema.rs index 7d7e2570a..08b925dd1 100644 --- a/crates/integration-tests/tests/schema.rs +++ b/crates/integration-tests/tests/schema.rs @@ -30,7 +30,7 @@ fn schema_filtered_scan_is_open() { .output_schema() .unwrap(); assert!(!s.closed, "PromQL scan with predicates must remain open"); - assert_eq!(s.columns.len(), 3, "[ts, value, job]"); + assert_eq!(s.fields.len(), 3, "[ts, value, job]"); } // per-series rate is label-preserving → output stays open diff --git a/crates/integration-tests/tests/sql_to_physical.rs b/crates/integration-tests/tests/sql_to_physical.rs index be96107ea..dd4426c22 100644 --- a/crates/integration-tests/tests/sql_to_physical.rs +++ b/crates/integration-tests/tests/sql_to_physical.rs @@ -8,8 +8,8 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use asap_types::{ - post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, - pre_asap::{Column, DataType, QueryExpr, Schema}, + post_asap::{compile_post_asap_dag, FieldDataType, PostAsapOperatorPayload}, + pre_asap::{DataType, Field, QueryExpr, Schema}, types::AccuracyTarget, }; use futures::StreamExt; @@ -22,8 +22,8 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { let catalog = SqlCatalog::new().with_table( "metrics", Schema::new(vec![ - Column::new("service", DataType::Utf8, false), - Column::new("value", DataType::Float64, true), + Field::plain("service", DataType::Utf8, false), + Field::plain("value", DataType::Float64, true), ]), ); for query in [ @@ -58,7 +58,7 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { assert!(schema .fields .iter() - .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); let plan = compile( &dag, BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), diff --git a/crates/integration-tests/tests/sql_to_post_asap.rs b/crates/integration-tests/tests/sql_to_post_asap.rs index 5d10fb2f5..abd6c90c5 100644 --- a/crates/integration-tests/tests/sql_to_post_asap.rs +++ b/crates/integration-tests/tests/sql_to_post_asap.rs @@ -27,13 +27,13 @@ use asap_aware_mapping::{ }; use asap_frontend_sql::{lower_sql, lower_sql_dialect, SqlCatalog}; use asap_types::post_asap::{ - compile_post_asap_dag, EdgeRole, ExactKind, ExactParams, GroupingStrategy, + compile_post_asap_dag, EdgeRole, ExactKind, ExactParams, FieldDataType, GroupingStrategy, PostAsapOperatorPayload, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, SummaryExpr, - SummaryFamilyType, SummaryNode, SummarySchema, SummaryUpdate, ValueOperation, + SummaryNode, SummaryUpdate, ValueOperation, }; use asap_types::pre_asap::expr_ir::ColumnRef; use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; @@ -58,7 +58,7 @@ fn realize(expr: &QueryExpr) -> Result, RealizationError> { } } -fn dtype<'a>(schema: &'a SummarySchema, name: &str) -> &'a SummaryFamilyType { +fn dtype<'a>(schema: &'a Schema, name: &str) -> &'a FieldDataType { &schema .fields .iter() @@ -67,8 +67,8 @@ fn dtype<'a>(schema: &'a SummarySchema, name: &str) -> &'a SummaryFamilyType { .dtype } -fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } /// `metrics(ts, service, latency, bytes)` — mirrors @@ -100,11 +100,11 @@ async fn clickhouse_temporal_sql_reuses_rate_and_increase_physical_summaries() { for (function, expected) in [ ( "asap_rate", - SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), + FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), ), ( "asap_increase", - SummaryFamilyType::ExactAggregate(ExactKind::Increase, ExactParams::Increase), + FieldDataType::ExactAggregate(ExactKind::Increase, ExactParams::Increase), ), ] { let sql = format!( @@ -176,8 +176,7 @@ async fn clickhouse_outer_sum_recursively_binds_inner_temporal_aggregate() { fn has_temporal_summary(node: &SummaryNode) -> bool { match &node.expr { SummaryExpr::SummaryAgg { - family: - SummaryFamilyType::ExactAggregate(ExactKind::Rate | ExactKind::Increase, _), + family: FieldDataType::ExactAggregate(ExactKind::Rate | ExactKind::Increase, _), .. } => true, SummaryExpr::ValueOperation { child, .. } @@ -254,7 +253,7 @@ async fn sql_full_query_retains_project_and_binds_inner_aggregate() { assert_eq!(root.schema.fields[0].name, "p99", "project output schema"); assert_eq!( root.schema.fields[0].dtype, - SummaryFamilyType::Plain(DataType::Float64) + FieldDataType::Plain(DataType::Float64) ); assert!( matches!(child.expr, SummaryExpr::SummaryEstimate { .. }), @@ -352,7 +351,7 @@ async fn sql_join_recursively_binds_both_temporal_aggregate_children() { assert!(matches!( aggregate.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), + family: FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), .. } )); @@ -638,7 +637,7 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { ); assert_eq!( root.schema.fields[0].dtype, - SummaryFamilyType::Plain(DataType::Float64), + FieldDataType::Plain(DataType::Float64), "the summary-state type must not propagate past the estimate" ); @@ -654,7 +653,7 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { }; assert_eq!( family, - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) @@ -674,7 +673,7 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { ); assert_eq!( summary_input.schema.fields[0].dtype, - SummaryFamilyType::Sketch( + FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) @@ -689,7 +688,7 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { .schema .fields .iter() - .all(|f| matches!(f.dtype, SummaryFamilyType::Plain(_))), + .all(|f| matches!(f.dtype, FieldDataType::Plain(_))), "logical edges carry only plain columns" ); } @@ -720,7 +719,7 @@ async fn sql_count_distinct_with_epsilon_binds_hll_rse_over_named_column() { assert!(matches!(query, SketchQuery::Cardinality)); assert_eq!( root.schema.fields[0].dtype, - SummaryFamilyType::Plain(DataType::Int64), + FieldDataType::Plain(DataType::Int64), "COUNT(DISTINCT …) reads back out as an integer count" ); @@ -735,7 +734,7 @@ async fn sql_count_distinct_with_epsilon_binds_hll_rse_over_named_column() { }; assert_eq!( family, - &SummaryFamilyType::Sketch( + &FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Hll, SketchParams::Hll { precision: 14 }), GroupingStrategy::default() ) @@ -772,7 +771,7 @@ async fn sql_exact_workload_binds_accumulators_not_sketches() { }; assert_eq!( family, - &SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + &FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) ); assert_eq!( reduction, @@ -781,7 +780,7 @@ async fn sql_exact_workload_binds_accumulators_not_sketches() { ); assert_eq!( dtype(&root.schema, "service"), - &SummaryFamilyType::Plain(DataType::Utf8), + &FieldDataType::Plain(DataType::Utf8), "group keys pass through verbatim" ); diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index ed961cc85..5a28a13fc 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -274,7 +274,7 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { values::{Batch, Value}, }; use asap_types::{ - post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, + post_asap::{compile_post_asap_dag, FieldDataType, PostAsapOperatorPayload}, pre_asap::DataType, }; use std::{collections::BTreeMap, sync::Arc}; @@ -365,10 +365,8 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { .fields .iter() .map(|field| match field.dtype { - SummaryFamilyType::Plain(DataType::Float64) => { - Value::Float64(f64::from(value)) - } - SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), + FieldDataType::Plain(DataType::Float64) => Value::Float64(f64::from(value)), + FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), _ => panic!("unexpected field {field:?}"), }) .collect() @@ -544,7 +542,7 @@ fn chosen_lifecycle_timing_decides_precompute_contents() { runtime::Scope, values::{Batch, Value}, }; - use asap_types::{post_asap::SummaryFamilyType, pre_asap::DataType}; + use asap_types::{post_asap::FieldDataType, pre_asap::DataType}; use std::collections::BTreeMap; let mut answers = Vec::new(); @@ -568,10 +566,8 @@ fn chosen_lifecycle_timing_decides_precompute_contents() { .fields .iter() .map(|field| match field.dtype { - SummaryFamilyType::Plain(DataType::Float64) => { - Value::Float64(f64::from(value)) - } - SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), + FieldDataType::Plain(DataType::Float64) => Value::Float64(f64::from(value)), + FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), _ => panic!("unexpected field {field:?}"), }) .collect() @@ -840,9 +836,7 @@ fn chosen_population_lifecycle_decides_precompute_contents() { fn grouped_rate_sum_placement_is_a_lifecycle_choice() { use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; use asap_physical_operators::physical_planner::{compile_candidate, InputContract}; - use asap_types::post_asap::{ - ExactKind, PostAsapOperatorPayload, SummaryExpr, SummaryFamilyType, - }; + use asap_types::post_asap::{ExactKind, FieldDataType, PostAsapOperatorPayload, SummaryExpr}; use std::{collections::BTreeMap, sync::Arc}; let workload = quantile_workload("sum by(job)(rate(m[1m]))"); @@ -854,7 +848,7 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { ); let is_exact = |node: &SummaryNode, kind: ExactKind| { matches!(&node.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(k, _), .. + family: FieldDataType::ExactAggregate(k, _), .. } if *k == kind) }; let inventory = asap_aware_mapping::search_workload(vec![("q", root)]) @@ -952,7 +946,7 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { }; let state = |payload: &PostAsapOperatorPayload, kind: ExactKind| { matches!(payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(k, _), .. + family: FieldDataType::ExactAggregate(k, _), .. } if *k == kind) }; assert!(state(retained, ExactKind::Sum)); diff --git a/crates/planner/tests/e2e_plan.rs b/crates/planner/tests/e2e_plan.rs index c3ff85d88..ff8a8dde0 100644 --- a/crates/planner/tests/e2e_plan.rs +++ b/crates/planner/tests/e2e_plan.rs @@ -13,7 +13,7 @@ use asap_aware_mapping::{ use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; use asap_planner::{e2e_plan, FrontendInput, PlanError, UserInput, UserInputError}; use asap_types::post_asap::SummaryExpr; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataArrival, DataWorkload, DurationMs, Evidence, @@ -51,8 +51,8 @@ fn lineitem_catalog() -> SqlCatalog { SqlCatalog::new().with_table( "lineitem", Schema::new(vec![ - Column::new("l_orderkey", DataType::Int64, false), - Column::new("l_extendedprice", DataType::Float64, false), + Field::plain("l_orderkey", DataType::Int64, false), + Field::plain("l_extendedprice", DataType::Float64, false), ]), ) } diff --git a/crates/planner/tests/summary_sharing.rs b/crates/planner/tests/summary_sharing.rs index 069bda8ff..a7c1f4ea0 100644 --- a/crates/planner/tests/summary_sharing.rs +++ b/crates/planner/tests/summary_sharing.rs @@ -25,10 +25,10 @@ use asap_types::post_asap::{ ProbabilityExpr, ResultGuarantee, SketchQuery, }; use asap_types::post_asap::{ - SketchAlgorithm, SketchParams, SummaryExpr, SummaryFamilyType, SummaryNode, + FieldDataType, SketchAlgorithm, SketchParams, SummaryExpr, SummaryNode, }; use asap_types::pre_asap::agg_intent::default_quantile; -use asap_types::pre_asap::schema::{Column, DataType, Schema}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::pre_asap::{AggIntent, QueryExpr}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ @@ -171,8 +171,8 @@ async fn plan_sql(queries: &[&str], costs: &FixedCosts) -> PlanOutput { let catalog = SqlCatalog::new().with_table( "lineitem", Schema::new(vec![ - Column::new("l_orderkey", DataType::Int64, false), - Column::new("l_extendedprice", DataType::Float64, false), + Field::plain("l_orderkey", DataType::Int64, false), + Field::plain("l_extendedprice", DataType::Float64, false), ]), ); let input = UserInput::new( @@ -298,7 +298,7 @@ fn kll_k(plan: &asap_aware_mapping::pass::QueryLifecyclePlan) -> u32 { panic!("one state: {:?}", plan.plan.deployments.len()); }; let SummaryExpr::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), + family: FieldDataType::Sketch(kind, _), .. } = &deployment.summary.expr else { @@ -458,10 +458,10 @@ struct UnivMonEvidence; impl AccuracyModel for UnivMonEvidence { fn local_guarantee( &self, - family: &SummaryFamilyType, + family: &FieldDataType, query: &SketchQuery, ) -> Option { - if matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) + if matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) { let mut guarantee = ResultGuarantee::exact("SYNTHETIC test evidence; not measured"); guarantee.metric = ErrorMetric::RelativeValue; @@ -549,7 +549,7 @@ fn certified_frequency_readouts_share_one_univmon_state() { }; assert!(matches!( &summary_input.expr, - SummaryExpr::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } + SummaryExpr::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::UnivMon )); states.push(Rc::clone(summary_input)); diff --git a/crates/types/src/dag_export.rs b/crates/types/src/dag_export.rs index e5689de3c..43f248054 100644 --- a/crates/types/src/dag_export.rs +++ b/crates/types/src/dag_export.rs @@ -325,7 +325,7 @@ pub struct WorkloadDAG { /// [`TargetReplacementAfter::Summary`] site, never matched back against a /// separately-exported DAG the way pre-ASAP notes are). /// -/// Several of `SummaryExpr`'s own fields (`SummaryFamilyType`, +/// Several of `SummaryExpr`'s own fields (`FieldDataType`, /// `GroupingStrategy`, `SketchQuery`) derive neither `Serialize` nor /// `Deserialize` in `asap_types::post_asap` — they carry no reporting /// obligation there, since nothing before this module ever needed to @@ -428,22 +428,22 @@ fn push_summary_node( id } -/// A short, human-readable label for a [`crate::post_asap::SummaryFamilyType`] +/// A short, human-readable label for a [`crate::post_asap::FieldDataType`] /// (e.g. `"Sketch(Kll)"`, `"ExactAggregate(Sum)"`) — for /// [`SummaryDAGNode::label`] text on a `SummaryAgg`/`SummaryJoin` node. Not /// exhaustive prose (mirrors `asap_aware_mapping::replacement::describe_intent`'s /// own "this is a label, not a decision" stance) — every variant is covered, /// but via `Debug` for the inner kind rather than hand-written prose per /// algorithm. -fn family_label(family: &crate::post_asap::SummaryFamilyType) -> String { - use crate::post_asap::SummaryFamilyType; +fn family_label(family: &crate::post_asap::FieldDataType) -> String { + use crate::post_asap::FieldDataType; match family { - SummaryFamilyType::Plain(dtype) => format!("Plain({dtype:?})"), - SummaryFamilyType::ExactAggregate(kind, _) => format!("ExactAggregate({kind:?})"), - SummaryFamilyType::Sketch(kind, _grouping) => format!("Sketch({:?})", kind.algorithm()), - SummaryFamilyType::Sample(kind, _) => format!("Sample({kind:?})"), - SummaryFamilyType::Wavelet(kind, _) => format!("Wavelet({kind:?})"), - SummaryFamilyType::StatModel(kind, _) => format!("StatModel({kind:?})"), + FieldDataType::Plain(dtype) => format!("Plain({dtype:?})"), + FieldDataType::ExactAggregate(kind, _) => format!("ExactAggregate({kind:?})"), + FieldDataType::Sketch(kind, _grouping) => format!("Sketch({:?})", kind.algorithm()), + FieldDataType::Sample(kind, _) => format!("Sample({kind:?})"), + FieldDataType::Wavelet(kind, _) => format!("Wavelet({kind:?})"), + FieldDataType::StatModel(kind, _) => format!("StatModel({kind:?})"), } } @@ -980,7 +980,7 @@ fn build_summary_hybrid( id } -fn summary_schema_json(schema: &crate::post_asap::SummarySchema) -> serde_json::Value { +fn summary_schema_json(schema: &crate::post_asap::Schema) -> serde_json::Value { serde_json::json!({ "fields": schema.fields.iter().map(|field| serde_json::json!({ "name": field.name, @@ -1410,17 +1410,17 @@ mod tests { use crate::pre_asap::agg_intent::AggIntent; use crate::pre_asap::expr_ir::ScalarValue; use crate::pre_asap::query_expr::{GroupKeys, Predicate, Reduction}; - use crate::pre_asap::schema::{Column, DataType, Schema}; + use crate::pre_asap::schema::{DataType, Field, Schema}; use crate::types::AccuracyTarget; - fn scan(table: &str, columns: Vec) -> QueryExpr { + fn scan(table: &str, columns: Vec) -> QueryExpr { QueryExpr::Scan { source: Source::Table { table_ref: table.into(), }, predicates: vec![], schema: Schema { - columns, + fields: columns, time_index: None, unique_keys: vec![], closed: true, @@ -1428,8 +1428,8 @@ mod tests { } } - fn value_col() -> Vec { - vec![Column::new("value", DataType::Float64, false)] + fn value_col() -> Vec { + vec![Field::plain("value", DataType::Float64, false)] } #[test] @@ -1703,23 +1703,20 @@ mod tests { #[test] fn export_carries_guarantee_allocation_and_rejection_reason() { use crate::post_asap::{ - BoundExpr, CompositionOperator, ErrorMetric, GroupingStrategy, GuaranteeSource, - ProbabilityExpr, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, - SummaryFamilyType, SummarySchema, + BoundExpr, CompositionOperator, ErrorMetric, FieldDataType, GroupingStrategy, + GuaranteeSource, ProbabilityExpr, Schema, SketchAlgorithm, SketchKind, SketchParams, + SketchQuery, }; - let leaf = Rc::new(scan("t", vec![Column::new("v", DataType::Float64, false)])); + let leaf = Rc::new(scan("t", vec![Field::plain("v", DataType::Float64, false)])); let kept = Rc::new(SummaryNode { expr: SummaryExpr::KeepPreAsap(Rc::clone(&leaf)), - schema: SummarySchema { - fields: vec![], - time_index: None, - }, + schema: Schema::lifted(vec![], None), guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), }); let agg = Rc::new(SummaryNode { expr: SummaryExpr::SummaryAgg { child: kept, - family: SummaryFamilyType::Sketch( + family: FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 40 }), GroupingStrategy::default(), ), @@ -1730,10 +1727,7 @@ mod tests { grouping: GroupingStrategy::default(), filter: None, }, - schema: SummarySchema { - fields: vec![], - time_index: None, - }, + schema: Schema::lifted(vec![], None), guarantee: None, }); let guarantee = ResultGuarantee { @@ -1766,10 +1760,7 @@ mod tests { summary_input: agg, query: SketchQuery::Quantile { q: 0.99 }, }, - schema: SummarySchema { - fields: vec![], - time_index: None, - }, + schema: Schema::lifted(vec![], None), guarantee: Some(guarantee), }; let dag = export_summary(&root); diff --git a/crates/types/src/post_asap/cse.rs b/crates/types/src/post_asap/cse.rs index 745758674..549c48e15 100644 --- a/crates/types/src/post_asap/cse.rs +++ b/crates/types/src/post_asap/cse.rs @@ -250,7 +250,7 @@ pub fn share_common_summary_sub_dags( #[cfg(test)] mod tests { use super::*; - use crate::post_asap::{ResultGuarantee, SummarySchema}; + use crate::post_asap::{ResultGuarantee, Schema}; use crate::pre_asap::{QueryExpr, ScalarValue}; fn leaf(value: f64) -> Rc { @@ -258,10 +258,7 @@ mod tests { expr: SummaryExpr::KeepPreAsap(Rc::new(QueryExpr::Literal(ScalarValue::Float64( value, )))), - schema: SummarySchema { - fields: vec![], - time_index: None, - }, + schema: Schema::lifted(vec![], None), guarantee: Some(ResultGuarantee::exact("fixture")), }) } @@ -283,10 +280,7 @@ mod tests { timing: crate::post_asap::ExecutionTiming::IngestionTime, children: vec![leaf(1.0), leaf(2.0)], }, - schema: SummarySchema { - fields: vec![], - time_index: None, - }, + schema: Schema::lifted(vec![], None), guarantee: None, }); let roots = share_common_summary_sub_dags(vec![(0, leaf(1.0)), (1, merge)]); @@ -361,15 +355,15 @@ mod tests { #[test] fn quantile_roots_share_producer_but_not_readout_or_parameters() { use crate::post_asap::{ - GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, - SummaryFamilyType, SummaryUpdate, + FieldDataType, GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, + SketchQuery, SummaryUpdate, }; use crate::pre_asap::{ColumnRef, Reduction}; fn readout(q: f64, alpha: f64) -> Rc { let producer = Rc::new(SummaryNode { expr: SummaryExpr::SummaryAgg { child: leaf(1.0), - family: SummaryFamilyType::Sketch( + family: FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::DDSketch, SketchParams::DDSketch { alpha }, @@ -381,10 +375,7 @@ mod tests { grouping: GroupingStrategy::default(), filter: None, }, - schema: SummarySchema { - fields: vec![], - time_index: None, - }, + schema: Schema::lifted(vec![], None), guarantee: None, }); Rc::new(SummaryNode { @@ -392,10 +383,7 @@ mod tests { summary_input: producer, query: SketchQuery::Quantile { q }, }, - schema: SummarySchema { - fields: vec![], - time_index: None, - }, + schema: Schema::lifted(vec![], None), guarantee: None, }) } @@ -436,10 +424,7 @@ mod tests { vector_match: None, }, }, - schema: super::super::SummarySchema { - fields: vec![], - time_index: None, - }, + schema: super::super::Schema::lifted(vec![], None), guarantee: None, }); } diff --git a/crates/types/src/post_asap/execution_data_state.rs b/crates/types/src/post_asap/execution_data_state.rs index a9368377c..d70467ef8 100644 --- a/crates/types/src/post_asap/execution_data_state.rs +++ b/crates/types/src/post_asap/execution_data_state.rs @@ -48,9 +48,10 @@ use std::rc::Rc; use thiserror::Error; use super::expr::{ExactOperation, SummaryExpr, SummaryNode, ValueOperation}; -use super::schema::{SummaryFamilyType, SummaryField, SummarySchema}; +use crate::pre_asap::schema::FieldDataType; + use crate::pre_asap::query_expr::{aggregate_output_schema, Predicate, QueryExprError}; -use crate::pre_asap::schema::{Column, Schema}; +use crate::pre_asap::schema::Schema; /// When a post-ASAP value is produced. #[derive( @@ -206,6 +207,10 @@ pub enum ExecutionDataStateError { /// declared data_state. #[error("exact operator consumes non-plain column {column:?} ({dtype})")] NonPlainOperand { column: String, dtype: String }, + /// A reserved ASAP operator (`SummaryMerge`, `SummarySubtract`, + /// `SummaryDelete`, `SummaryJoin`, `Extension`) in an executable plan. + #[error("{operator} is a reserved operator with no execution contract yet")] + UnimplementedOperator { operator: &'static str }, } /// The data_state assigned to every node of a validated plan, keyed by @@ -266,10 +271,10 @@ pub fn produced_data_state(expr: &SummaryExpr) -> Option { /// Is `family` the exact-accumulator family whose partial state *is* the /// value — the one summary state a `SummaryAgg` may re-accumulate? -fn is_exact_accumulator_state(schema: &SummarySchema) -> Result<(), ExecutionDataStateError> { +fn is_exact_accumulator_state(schema: &Schema) -> Result<(), ExecutionDataStateError> { for field in &schema.fields { match &field.dtype { - SummaryFamilyType::Plain(_) | SummaryFamilyType::ExactAggregate(..) => {} + FieldDataType::Plain(_) | FieldDataType::ExactAggregate(..) => {} other => { return Err(ExecutionDataStateError::UnsupportedStateComposition { family: format!("{other:?}"), @@ -328,7 +333,7 @@ fn per_series_rows(node: &SummaryNode) -> Option<&crate::pre_asap::QueryExpr> { } => match &child.expr { SummaryExpr::SummaryAgg { child, - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum | ExactKind::Count, _), + family: FieldDataType::ExactAggregate(ExactKind::Sum | ExactKind::Count, _), reduction: crate::pre_asap::query_expr::Reduction::PerEntity, .. } => match &child.expr { @@ -409,11 +414,11 @@ fn visit( || !node.schema.fields.iter().all(|field| { !field.nullable && if field.name == crate::pre_asap::schema::PROMQL_SERIES_IDENTITY { - field.dtype == SummaryFamilyType::Plain(DataType::Utf8) + field.dtype == FieldDataType::Plain(DataType::Utf8) } else { matches!( field.dtype, - SummaryFamilyType::Plain(DataType::Float64 | DataType::Timestamp) + FieldDataType::Plain(DataType::Float64 | DataType::Timestamp) ) } }) @@ -422,7 +427,7 @@ fn visit( .fields .iter() .filter(|field| { - matches!(field.dtype, SummaryFamilyType::Plain(DataType::Float64)) + matches!(field.dtype, FieldDataType::Plain(DataType::Float64)) }) .count() != 1 @@ -683,7 +688,7 @@ fn state_only( /// column. fn check_plain_operands( op: &ValueOperation, - input: &SummarySchema, + input: &Schema, ) -> Result<(), ExecutionDataStateError> { if matches!( op, @@ -721,7 +726,7 @@ fn check_plain_operands( if !(implicit || referenced.contains(&i)) { continue; } - if !matches!(field.dtype, SummaryFamilyType::Plain(_)) { + if !matches!(field.dtype, FieldDataType::Plain(_)) { return Err(ExecutionDataStateError::NonPlainOperand { column: field.name.clone(), dtype: format!("{:?}", field.dtype), @@ -731,11 +736,11 @@ fn check_plain_operands( Ok(()) } -fn check_plain_or_exact_values(input: &SummarySchema) -> Result<(), ExecutionDataStateError> { +fn check_plain_or_exact_values(input: &Schema) -> Result<(), ExecutionDataStateError> { for field in &input.fields { if !matches!( field.dtype, - SummaryFamilyType::Plain(_) | SummaryFamilyType::ExactAggregate(..) + FieldDataType::Plain(_) | FieldDataType::ExactAggregate(..) ) { return Err(ExecutionDataStateError::NonPlainOperand { column: field.name.clone(), @@ -746,9 +751,9 @@ fn check_plain_or_exact_values(input: &SummarySchema) -> Result<(), ExecutionDat Ok(()) } -fn check_all_plain(input: &SummarySchema) -> Result<(), ExecutionDataStateError> { +fn check_all_plain(input: &Schema) -> Result<(), ExecutionDataStateError> { for field in &input.fields { - if !matches!(field.dtype, SummaryFamilyType::Plain(_)) { + if !matches!(field.dtype, FieldDataType::Plain(_)) { return Err(ExecutionDataStateError::NonPlainOperand { column: field.name.clone(), dtype: format!("{:?}", field.dtype), @@ -758,39 +763,18 @@ fn check_all_plain(input: &SummarySchema) -> Result<(), ExecutionDataStateError> Ok(()) } -/// The plain pre-ASAP `Schema` underlying an all-`Plain` `SummarySchema`, or -/// `None` if any column carries summary state. -pub fn plain_schema(schema: &SummarySchema) -> Option { - let mut columns = Vec::with_capacity(schema.fields.len()); - for field in &schema.fields { - let SummaryFamilyType::Plain(dtype) = &field.dtype else { - return None; - }; - columns.push(Column::new(&field.name, dtype.clone(), field.nullable)); - } - Some(Schema { - columns, - time_index: schema.time_index, - unique_keys: Vec::new(), - closed: true, - }) +/// `schema` with its reuse metadata dropped, or `None` if any field carries +/// summary state — the shape an exact operator reads. +pub fn plain_schema(schema: &Schema) -> Option { + schema + .is_all_plain() + .then(|| Schema::lifted(schema.fields.clone(), schema.time_index)) } -/// Lift a plain pre-ASAP schema to a `SummarySchema` with every column -/// `Plain` — the output of every exact operator. -pub fn lift_plain(schema: &Schema) -> SummarySchema { - SummarySchema { - fields: schema - .columns - .iter() - .map(|c| SummaryField { - name: c.name.clone(), - dtype: SummaryFamilyType::Plain(c.dtype.clone()), - nullable: c.nullable, - }) - .collect(), - time_index: schema.time_index, - } +/// `schema` as a summary-planning node output: fields and time axis kept, +/// unique keys dropped, closed. +pub fn lift_plain(schema: &Schema) -> Schema { + Schema::lifted(schema.fields.clone(), schema.time_index) } /// Output schema of `op` applied to a child whose edge carries `input` — @@ -800,8 +784,8 @@ pub fn lift_plain(schema: &Schema) -> SummarySchema { /// state the operator cannot read. pub fn exact_operation_output_schema( op: &ExactOperation, - input: &SummarySchema, -) -> Result { + input: &Schema, +) -> Result { let plain = plain_schema(input).ok_or(ExactOperationSchemaError::NonPlainInput)?; let ExactOperation::Aggregate { reduction, @@ -829,7 +813,7 @@ mod tests { use crate::pre_asap::agg_intent::AggIntent; use crate::pre_asap::expr_ir::ColumnRef; use crate::pre_asap::query_expr::{QueryExpr, Reduction, Source}; - use crate::pre_asap::schema::DataType; + use crate::pre_asap::schema::{DataType, Field}; /// Both execution phases use raw values, distinct from maintained state. #[test] @@ -853,9 +837,9 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - Column::new("zone", DataType::Utf8, true), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("zone", DataType::Utf8, true), ], 0, vec![], @@ -873,21 +857,22 @@ mod tests { }) } - fn plain(names: &[&str]) -> SummarySchema { - SummarySchema { - fields: names + fn plain(names: &[&str]) -> Schema { + Schema::lifted( + names .iter() - .map(|n| SummaryField { + .map(|n| Field { name: (*n).into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, + table: None, }) .collect(), - time_index: None, - } + None, + ) } - fn agg(child: Rc, family: SummaryFamilyType) -> Rc { + fn agg(child: Rc, family: FieldDataType) -> Rc { Rc::new(SummaryNode { expr: SummaryExpr::SummaryAgg { child, @@ -897,21 +882,22 @@ mod tests { grouping: GroupingStrategy::default(), filter: None, }, - schema: SummarySchema { - fields: vec![SummaryField { + schema: Schema::lifted( + vec![Field { name: "state".into(), dtype: family, nullable: false, + table: None, }], - time_index: None, - }, + None, + ), guarantee: None, }) } - fn kll() -> SummaryFamilyType { + fn kll() -> FieldDataType { use crate::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; - SummaryFamilyType::Sketch( + FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 200 }), GroupingStrategy::default(), ) @@ -960,7 +946,7 @@ mod tests { use crate::pre_asap::{ArithmeticOpKind, BinaryOpKind}; let identity = crate::pre_asap::schema::PROMQL_SERIES_IDENTITY; // A finalized per-series Sum of `metric`'s rows. - let operand = |metric: &str, schema: &SummarySchema| { + let operand = |metric: &str, schema: &Schema| { let rows = Rc::new(QueryExpr::Scan { source: Source::TimeSeries { metric: metric.into(), @@ -968,16 +954,15 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], 0, vec![], ), }); let mut state = schema.clone(); - state.fields[0].dtype = - SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + state.fields[0].dtype = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let sum = Rc::new(SummaryNode { expr: SummaryExpr::SummaryAgg { child: Rc::new(SummaryNode { @@ -985,7 +970,7 @@ mod tests { schema: schema.clone(), guarantee: None, }), - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), input: crate::post_asap::SummaryUpdate::column(ColumnRef::SampleValue), reduction: Reduction::PerEntity, grouping: GroupingStrategy::default(), @@ -1004,7 +989,7 @@ mod tests { guarantee: None, }) }; - let validate = |schema: SummarySchema, rhs: &str| { + let validate = |schema: Schema, rhs: &str| { let binary = Rc::new(SummaryNode { expr: SummaryExpr::BinaryOp { lhs: operand("m", &schema), @@ -1023,9 +1008,10 @@ mod tests { validate_execution_data_states(&estimate(agg(binary, kll()))).map(|_| ()) }; let mut schema = plain(&["value"]); - schema.fields.push(SummaryField { + schema.fields.push(Field { + table: None, name: "ts".into(), - dtype: SummaryFamilyType::Plain(DataType::Timestamp), + dtype: FieldDataType::Plain(DataType::Timestamp), nullable: false, }); schema.time_index = Some(1); @@ -1034,9 +1020,10 @@ mod tests { validate(schema.clone(), "n").is_ok(), "no identity to align" ); - schema.fields.push(SummaryField { + schema.fields.push(Field { + table: None, name: identity.into(), - dtype: SummaryFamilyType::Plain(DataType::Utf8), + dtype: FieldDataType::Plain(DataType::Utf8), nullable: false, }); assert!(validate(schema.clone(), "m").is_ok()); @@ -1049,7 +1036,7 @@ mod tests { let mut invalid = schema.clone(); match mutation { 0 => invalid.fields[2].nullable = true, - 1 => invalid.fields[2].dtype = SummaryFamilyType::Plain(DataType::Timestamp), + 1 => invalid.fields[2].dtype = FieldDataType::Plain(DataType::Timestamp), 2 => invalid.fields.push(invalid.fields[2].clone()), _ => invalid.fields[2].name = "label".into(), } @@ -1064,7 +1051,7 @@ mod tests { fn exact_accumulator_state_may_feed_another_summary_agg() { let inner = agg( keep(), - SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), ); let root = estimate(agg(inner, kll())); assert!(validate_execution_data_states(&root).is_ok()); @@ -1340,7 +1327,7 @@ mod tests { assert!(out .fields .iter() - .all(|f| matches!(f.dtype, SummaryFamilyType::Plain(_)))); + .all(|f| matches!(f.dtype, FieldDataType::Plain(_)))); } #[test] diff --git a/crates/types/src/post_asap/expr.rs b/crates/types/src/post_asap/expr.rs index 8373f1843..9312d95c6 100644 --- a/crates/types/src/post_asap/expr.rs +++ b/crates/types/src/post_asap/expr.rs @@ -2,10 +2,10 @@ use super::ExecutionTiming; use std::rc::Rc; use super::guarantee::ResultGuarantee; -use super::schema::{SummaryFamilyType, SummarySchema}; use super::sketch::{GroupingStrategy, SketchQuery, SummaryUpdate}; use crate::pre_asap::agg_intent::AggIntent; use crate::pre_asap::query_expr::Predicate; +use crate::pre_asap::schema::{FieldDataType, Schema}; use crate::pre_asap::{ BinaryOpKind, ColumnRef, GroupKeys, JoinKind, ProjectItem, QueryExpr, Reduction, SortKey, VectorMatch, @@ -92,15 +92,15 @@ pub enum CandidateCompleteness { // ── Post-ASAP DAG node ─────────────────────────────────────────────────────── /// A node in the post-ASAP DAG: wraps the expression and its derived output -/// schema so every edge carries a typed schema. `SummarySchema` may contain -/// summary-state-typed columns (`SummaryFamilyType`'s non-`Plain` variants); +/// schema so every edge carries a typed schema. `Schema` may contain +/// summary-state-typed columns (`FieldDataType`'s non-`Plain` variants); /// the pre-ASAP `Schema` cannot. #[derive(Debug, Clone, PartialEq)] pub struct SummaryNode { pub expr: SummaryExpr, /// Output schema of `expr` — the schema of the data flowing on the edge /// leading *from* this node to its parent(s). - pub schema: SummarySchema, + pub schema: Schema, /// The machine-readable accuracy guarantee of the *value* this node /// produces (issue #172) — `Some` on every finalized, caller-visible /// value: a `SummaryEstimate` readout, an `ExactAggregate`-family @@ -129,8 +129,8 @@ pub struct SummaryNode { pub enum SummaryExpr { /// A pre-ASAP sub-DAG kept as-is because it has no selected implementation /// or supported residual decomposition. Output schema is the inner node's - /// schema, lifted to `SummarySchema` with all fields as - /// `SummaryFamilyType::Plain`. + /// schema, lifted to `Schema` with all fields as + /// `FieldDataType::Plain`. KeepPreAsap(Rc), /// A PromQL binary operation whose operands were planned independently. @@ -172,9 +172,9 @@ pub enum SummaryExpr { SummaryAgg { child: Rc, /// Which summary family realizes this aggregation, and that - /// family's own `(kind, params)`. Never `SummaryFamilyType::Plain` + /// family's own `(kind, params)`. Never `FieldDataType::Plain` /// — this node always produces summary state, not a plain value. - family: SummaryFamilyType, + family: FieldDataType, /// Optional multidimensional item identity and the observation/update /// weight fed into each state update. Subpopulation semantics remain /// on `reduction`; physical sharing remains on `grouping`. @@ -222,8 +222,8 @@ pub enum SummaryExpr { outer: Rc, inner: Rc, key: ColumnRef, - /// Never `SummaryFamilyType::Plain` — see [`SummaryAgg::family`](SummaryExpr::SummaryAgg). - family: SummaryFamilyType, + /// Never `FieldDataType::Plain` — see [`SummaryAgg::family`](SummaryExpr::SummaryAgg). + family: FieldDataType, }, /// Subtract one summary from another. Valid only for families with a diff --git a/crates/types/src/post_asap/maintained_population.rs b/crates/types/src/post_asap/maintained_population.rs index 3939c7c13..28aec0e54 100644 --- a/crates/types/src/post_asap/maintained_population.rs +++ b/crates/types/src/post_asap/maintained_population.rs @@ -70,7 +70,7 @@ impl CurrentSeriesInput { } if self.grouping.iter().any(|label| { !schema - .columns + .fields .iter() .any(|c| c.name == *label && c.dtype == DataType::Utf8) }) { @@ -86,7 +86,7 @@ impl CurrentSeriesInput { else { return false; }; - let Some(column) = schema.columns.get(*col) else { + let Some(column) = schema.fields.get(*col) else { return false; }; if column.dtype != DataType::Utf8 { @@ -142,8 +142,8 @@ impl MaintainedPopulation { use crate::pre_asap::{DataType, QueryExpr, Source}; expected.as_ref() == input && matches!(input, QueryExpr::Scan { source: Source::Table { .. }, schema, .. } - if schema.closed && schema.columns.get(*value_column).is_some_and(|c| c.dtype == DataType::Float64 && !c.nullable) - && !grouping.is_without() && grouping.keys().iter().all(|k| *k < schema.columns.len())) + if schema.closed && schema.fields.get(*value_column).is_some_and(|c| c.dtype == DataType::Float64 && !c.nullable) + && !grouping.is_without() && grouping.keys().iter().all(|k| *k < schema.fields.len())) } } } diff --git a/crates/types/src/post_asap/mod.rs b/crates/types/src/post_asap/mod.rs index 4c35559b4..d4715ca79 100644 --- a/crates/types/src/post_asap/mod.rs +++ b/crates/types/src/post_asap/mod.rs @@ -13,7 +13,7 @@ //! one-pair-per-family shape: it nests a third level, [`sketch::SketchKind`] //! (quantile/cardinality/frequency/top-k), which itself carries the //! committed [`sketch::SketchAlgorithm`] and [`sketch::SketchParams`] — -//! `SummaryFamilyType::Sketch(SketchKind, GroupingStrategy)`, not a flat +//! `FieldDataType::Sketch(SketchKind, GroupingStrategy)`, not a flat //! `(kind, params)` pair //! — because `Sketch` is the one family with more than one algorithm per //! purpose today; no other family needs that extra level yet. @@ -34,12 +34,12 @@ pub mod guarantee; pub mod maintained_population; pub mod post_asap_dag; pub mod query_time; -pub mod schema; pub mod sketch; pub mod summary_maintenance; pub mod summary_maintenance_lifecycle; pub mod summary_window; +pub use crate::pre_asap::schema::{Field, FieldDataType, Schema}; pub use cse::share_common_summary_sub_dags; pub use execution_data_state::{ assigned_child_data_state, exact_operation_output_schema, produced_data_state, @@ -65,7 +65,6 @@ pub use query_time::{ classic_cms_sizing, cms_posterior_error_bound, count_sketch_posterior_error_bound, cu_sketch_posterior_error_bound, traditional_a_priori_bound, }; -pub use schema::{SummaryFamilyType, SummaryField, SummarySchema}; pub use sketch::{ default_hydra_params, hydra_kind_for, EntityIdentity, ExactKind, ExactParams, GroupingStrategy, HydraKind, HydraParams, NonNegativeWeightProof, SamplingKind, SamplingParams, SketchAlgorithm, diff --git a/crates/types/src/post_asap/post_asap_dag.rs b/crates/types/src/post_asap/post_asap_dag.rs index ebbe51554..5cdbb3d81 100644 --- a/crates/types/src/post_asap/post_asap_dag.rs +++ b/crates/types/src/post_asap/post_asap_dag.rs @@ -5,11 +5,11 @@ use std::rc::Rc; use super::{ validate_execution_data_states, ExecutionDataState, ExecutionDataStateError, ResultGuarantee, - SummaryExpr, SummaryNode, SummarySchema, + Schema, SummaryExpr, SummaryNode, }; use super::{ - BinaryOperator, CandidateCompleteness, ExecutionTiming, GroupingStrategy, SketchQuery, - SummaryFamilyType, SummaryUpdate, ValueOperation, + BinaryOperator, CandidateCompleteness, ExecutionTiming, FieldDataType, GroupingStrategy, + SketchQuery, SummaryUpdate, ValueOperation, }; use crate::pre_asap::{ColumnRef, JoinKind, Predicate, QueryExpr, Reduction}; use thiserror::Error; @@ -65,7 +65,7 @@ pub enum PostAsapOperatorPayload { pruning: Option, }, SummaryAgg { - family: SummaryFamilyType, + family: FieldDataType, input: SummaryUpdate, reduction: Reduction, grouping: GroupingStrategy, @@ -76,7 +76,7 @@ pub enum PostAsapOperatorPayload { }, SummaryJoin { key: ColumnRef, - family: SummaryFamilyType, + family: FieldDataType, }, SummarySubtract, SummaryDelete { @@ -96,7 +96,7 @@ pub struct PostAsapDAGNode { pub payload: PostAsapOperatorPayload, /// Phase is a placement choice for every operator, independent of payload kind. pub output_state: ExecutionDataState, - pub output_schema: SummarySchema, + pub output_schema: Schema, pub guarantee: Option, } @@ -106,7 +106,7 @@ pub struct PostAsapDAGEdge { pub producer: PostAsapNodeId, pub consumer: PostAsapNodeId, pub role: EdgeRole, - pub intermediate_schema: SummarySchema, + pub intermediate_schema: Schema, pub data_state: ExecutionDataState, pub grouping: GroupingEdgeCompatibility, pub window: WindowEdgeCompatibility, @@ -231,7 +231,7 @@ impl PostAsapDAG { if &field.dtype == family { found_family = true; } - if let SummaryFamilyType::Sketch(_, schema_grouping) = &field.dtype { + if let FieldDataType::Sketch(_, schema_grouping) = &field.dtype { if schema_grouping != grouping { return Err(PostAsapDAGValidationError::SummaryGroupingMismatch { node: node.id, @@ -572,10 +572,10 @@ pub fn compile_post_asap_dag_with_node_ids( mod tests { use super::*; use crate::post_asap::{ - ExactKind, ExactParams, ExecutionTiming, GroupingStrategy, SummaryFamilyType, SummaryField, - SummaryUpdate, ValueOperation, + ExactKind, ExactParams, ExecutionTiming, FieldDataType, GroupingStrategy, SummaryUpdate, + ValueOperation, }; - use crate::pre_asap::schema::{Column, Schema}; + use crate::pre_asap::schema::{Field, Schema}; use crate::pre_asap::{ColumnRef, DataType, QueryExpr, Reduction, Source}; use std::collections::BTreeMap; @@ -583,7 +583,7 @@ mod tests { fn every_physical_payload_can_be_assigned_either_phase() { use crate::post_asap::DataPrimitive; use crate::pre_asap::{ArithmeticOpKind, BinaryOpKind, JoinKind, Predicate, ScalarValue}; - let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let predicate = Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))); let payloads = vec![ PostAsapOperatorPayload::Fallback { @@ -653,14 +653,15 @@ mod tests { timing: ExecutionTiming::QueryTime, primitive, }, - output_schema: SummarySchema { - fields: vec![SummaryField { + output_schema: Schema::lifted( + vec![Field { name: "value".into(), dtype: family.clone(), nullable: false, + table: None, }], - time_index: None, - }, + None, + ), guarantee: None, }], }; @@ -681,10 +682,7 @@ mod tests { #[test] fn phase_assignment_updates_edges_and_rejects_query_dependencies_in_ingestion() { use crate::pre_asap::ScalarValue; - let schema = SummarySchema { - fields: vec![], - time_index: None, - }; + let schema = Schema::lifted(vec![], None); let nodes = [0, 1] .into_iter() .map(|id| PostAsapDAGNode { @@ -735,22 +733,23 @@ mod tests { let scan = Rc::new(QueryExpr::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], - schema: Schema::new(vec![Column::new("value", DataType::Float64, false)]), + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), }); let raw = Rc::new(SummaryNode { expr: SummaryExpr::KeepPreAsap(scan), - schema: SummarySchema { - fields: vec![SummaryField { + schema: Schema::lifted( + vec![Field { name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, + table: None, }], - time_index: None, - }, + None, + ), guarantee: None, }); let make_agg = |child: Rc, kind, params| { - let family = SummaryFamilyType::ExactAggregate(kind, params); + let family = FieldDataType::ExactAggregate(kind, params); Rc::new(SummaryNode { expr: SummaryExpr::SummaryAgg { child, @@ -760,14 +759,15 @@ mod tests { grouping: GroupingStrategy::default(), filter: None, }, - schema: SummarySchema { - fields: vec![SummaryField { + schema: Schema::lifted( + vec![Field { name: "value".into(), dtype: family, nullable: false, + table: None, }], - time_index: None, - }, + None, + ), guarantee: None, }) }; @@ -779,14 +779,15 @@ mod tests { operation: ValueOperation::FinalizeExactAccumulator, timing: ExecutionTiming::QueryTime, }, - schema: SummarySchema { - fields: vec![SummaryField { + schema: Schema::lifted( + vec![Field { name: "value".into(), - dtype: SummaryFamilyType::Plain(DataType::Float64), + dtype: FieldDataType::Plain(DataType::Float64), nullable: false, + table: None, }], - time_index: None, - }, + None, + ), guarantee: None, }); @@ -819,7 +820,7 @@ mod tests { ); assert!(matches!( dependency.intermediate_schema.fields[0].dtype, - SummaryFamilyType::ExactAggregate(ExactKind::Sum, _) + FieldDataType::ExactAggregate(ExactKind::Sum, _) )); let encoded = serde_json::to_string(&dag).expect("serialize post-ASAP DAG"); let decoded: PostAsapDAG = @@ -846,7 +847,7 @@ mod tests { assert!(matches!( dag.nodes[2].payload, PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), reduction: Reduction::Reduce(_), .. } diff --git a/crates/types/src/post_asap/schema.rs b/crates/types/src/post_asap/schema.rs deleted file mode 100644 index d67e249df..000000000 --- a/crates/types/src/post_asap/schema.rs +++ /dev/null @@ -1,64 +0,0 @@ -use super::sketch::{ - ExactKind, ExactParams, GroupingStrategy, SamplingKind, SamplingParams, SketchKind, - StatModelKind, StatModelParams, WaveletKind, WaveletParams, -}; -use crate::pre_asap::DataType; - -// ── Post-ASAP data types ──────────────────────────────────────────────────── - -/// Column types that may appear on a post-ASAP DAG edge. A strict superset of -/// the pre-ASAP [`DataType`]: adds one variant per summary *family* for -/// edges that carry partial summary state between a `SummaryAgg` and a -/// downstream `SummaryEstimate` or `SummaryMerge`. -/// -/// Every non-`Plain` variant carries the physical state identity required by -/// that family (`Sketch` additionally carries its grouping layout), so the -/// type system can reject merges of incompatible -/// summaries at plan construction time — a `SummaryMerge` over -/// `Sketch(Kll, …)` and `Sketch(Cms, …)` inputs is a plan-time error, and a -/// `Sketch(…)` can never be confused for a `Sample(…)` even though both are -/// "opaque summary state" at a glance. -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -pub enum SummaryFamilyType { - /// An ordinary, readable value — the same closed vocabulary as the - /// pre-ASAP `DataType` (`Int64`/`Float64`/`Utf8`/`Bool`/`Timestamp`), - /// passed through unchanged from a pre-ASAP edge. - Plain(DataType), - /// Exact, mergeable accumulator state (`Sum`/`Count`/`Min`/`Max`/`Rate`/ - /// `Increase`). Value consumers require an explicit finalization boundary. - ExactAggregate(ExactKind, ExactParams), - /// Approximate sketch state (KLL/CMS/HLL/…), read out via a - /// `SummaryEstimate`. A [`SketchKind`] already carries the concrete - /// algorithm, params, and grouping layout committed to, not just its - /// category — a bound node needs to know it's specifically independent - /// KLL or shared Hydra-backed CMS, not merely "some sketch". - Sketch(SketchKind, GroupingStrategy), - /// Sampling-based summary state (a retained row subset). - Sample(SamplingKind, SamplingParams), - /// Wavelet-transform summary state (a coefficient vector). - Wavelet(WaveletKind, WaveletParams), - /// Fitted statistical/parametric-model summary state. - StatModel(StatModelKind, StatModelParams), -} - -// ── Post-ASAP schema ───────────────────────────────────────────────────────── - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -pub struct SummaryField { - pub name: String, - pub dtype: SummaryFamilyType, - pub nullable: bool, -} - -/// Schema carried on every edge of the post-ASAP DAG. Extends the pre-ASAP -/// `Schema` with the ability to express summary-state columns. The two are -/// separate types so a pre-ASAP node structurally cannot carry a -/// summary-state-typed column — any attempt to do so is a compile-time type -/// error. -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -pub struct SummarySchema { - pub fields: Vec, - /// Index into `fields` for the time axis, if any (same semantics as the - /// pre-ASAP `Schema::time_index`). - pub time_index: Option, -} diff --git a/crates/types/src/post_asap/sketch.rs b/crates/types/src/post_asap/sketch.rs index 416341e73..cddb4903a 100644 --- a/crates/types/src/post_asap/sketch.rs +++ b/crates/types/src/post_asap/sketch.rs @@ -134,7 +134,7 @@ pub enum SketchCategory { /// quantile-style, cardinality-style, frequency-style, or heavy-hitter/ /// top-k-style estimation — together with the concrete [`SketchAlgorithm`] /// and [`SketchParams`] realizing it. Sits between -/// [`SummaryFamilyType::Sketch`](super::schema::SummaryFamilyType::Sketch) +/// [`FieldDataType::Sketch`](super::schema::FieldDataType::Sketch) /// (the `Sketch` family as a whole, sibling to `Sample`/`Wavelet`/ /// `StatModel`) and the bare algorithm: `Kll` vs. `DDSketch` is a choice /// *within* `Quantile`, not a choice *of* `SketchKind` — every `Quantile` @@ -495,13 +495,13 @@ pub fn default_hydra_params( /// How a grouped aggregate's summary state is physically instantiated /// across its `by` subpopulations — orthogonal to *which* /// `SketchKind`/`SamplingKind`/`WaveletKind`/`StatModelKind` answers the -/// intent (that choice lives alongside it on `SummaryFamilyType`). Lives here, +/// intent (that choice lives alongside it on `FieldDataType`). Lives here, /// alongside `SketchKind`/`SketchParams` etc., rather than on any of those /// enums themselves, for exactly the reason explained in this section's /// module docs above. /// /// Carried both on `SummaryExpr::SummaryAgg` (where planning consults it) -/// and on sketch-valued `SummaryFamilyType` edges (where it prevents +/// and on sketch-valued `FieldDataType` edges (where it prevents /// incompatible shared and independent physical states from type-checking /// as merge-compatible). #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] diff --git a/crates/types/src/pre_asap/agg_intent.rs b/crates/types/src/pre_asap/agg_intent.rs index 5e4079b2e..c60dd55e3 100644 --- a/crates/types/src/pre_asap/agg_intent.rs +++ b/crates/types/src/pre_asap/agg_intent.rs @@ -16,7 +16,7 @@ use serde::{Deserialize, Serialize}; use crate::pre_asap::query_expr::DataModel; -use crate::pre_asap::schema::{Column, ColumnId, DataType}; +use crate::pre_asap::schema::{ColumnId, DataType, Field, FieldDataType}; use crate::types::AccuracyTarget; /// "What to compute" — the vocabulary the planner pivots on. @@ -532,7 +532,7 @@ impl AggIntent { /// Used by `QueryExpr::Aggregate`'s schema-derivation rule. The PromQL /// convention names the column after the intent kind so consumers can /// locate it without an alias lookup. - pub fn output_column(&self, input: &Column) -> Column { + pub fn output_column(&self, input: &Field) -> Field { match self { AggIntent::Count { .. } => col("count", DataType::Int64, false), AggIntent::Sum { .. } => col("sum", input.dtype.clone(), false), @@ -613,8 +613,8 @@ impl AggIntent { } } -fn col(name: &str, dtype: DataType, nullable: bool) -> Column { - Column::new(name, dtype, nullable) +fn col(name: &str, dtype: impl Into, nullable: bool) -> Field { + Field::new(name, dtype.into(), nullable) } /// `0.99` → `"0_99"`, `0.5` → `"0_5"`. Used by `Quantile` output naming so @@ -739,10 +739,10 @@ pub fn default_quantile(q: f64) -> AggIntent { #[cfg(test)] mod tests { use super::*; - use crate::pre_asap::schema::{Column, DataType}; + use crate::pre_asap::schema::{DataType, Field}; - fn c(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) + fn c(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } // Correlation exposes both dependencies but cannot merge final scalar results. @@ -813,7 +813,7 @@ mod tests { AggIntent::::Sum { col: None } .output_column(&c("c", DataType::Int64)) .dtype, - DataType::Int64 + FieldDataType::Plain(DataType::Int64) )); } @@ -970,7 +970,7 @@ mod arg_selector_contract_tests { use crate::pre_asap::{ColumnRef, Schema}; #[test] fn arg_selector_rejects_missing_or_unresolved_arguments() { - let schema = Schema::new(vec![Column::new("value", DataType::Float64, false)]); + let schema = Schema::new(vec![Field::plain("value", DataType::Float64, false)]); for payload in [ serde_json::json!({"arg_col": ColumnRef::Named("value".into())}), serde_json::json!({"arg_col": ColumnRef::Named("value".into()), "val_col": ColumnRef::Named("missing".into())}), diff --git a/crates/types/src/pre_asap/canonicalize.rs b/crates/types/src/pre_asap/canonicalize.rs index 4f15dcebb..b9e6a653a 100644 --- a/crates/types/src/pre_asap/canonicalize.rs +++ b/crates/types/src/pre_asap/canonicalize.rs @@ -328,7 +328,7 @@ fn try_rewrite_rownumber_topk(expr: &QueryExpr) -> Option { if order_by.is_empty() { return None; } - let inner_cols = inner.output_schema().ok()?.columns.len(); + let inner_cols = inner.output_schema().ok()?.fields.len(); if rn_in_wf != inner_cols { return None; // the predicate ranks some other column, not the row number } @@ -353,7 +353,7 @@ mod tests { GroupKeys, ProjectItem, Source, WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, }; - use crate::pre_asap::schema::{Column, DataType, Schema}; + use crate::pre_asap::schema::{DataType, Field, Schema}; use crate::types::AccuracyTarget; fn scan() -> QueryExpr { @@ -362,9 +362,9 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("service", DataType::Utf8, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("value", DataType::Float64, false), ], 0, vec![], @@ -631,10 +631,10 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("service", DataType::Utf8, false), - Column::new("region", DataType::Utf8, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("region", DataType::Utf8, false), + Field::plain("value", DataType::Float64, false), ], 0, vec![], diff --git a/crates/types/src/pre_asap/column_resolution.rs b/crates/types/src/pre_asap/column_resolution.rs index 3176730a9..cfafbe9e0 100644 --- a/crates/types/src/pre_asap/column_resolution.rs +++ b/crates/types/src/pre_asap/column_resolution.rs @@ -17,7 +17,7 @@ use super::query_expr::{ aggregate_output_schema, GroupKeys, QueryExpr, QueryExprError, Reduction, ResolvedQueryExpr, UnresolvedQueryExpr, }; -use super::schema::{ColumnId, DataType, Schema}; +use super::schema::{ColumnId, DataType, FieldDataType, Schema}; /// Errors returned by the resolution helpers. #[derive(Debug, Error, PartialEq, Eq)] @@ -40,7 +40,7 @@ pub fn resolve_column_ref(col: &ColumnRef, schema: &Schema) -> Result Result schema .column_id("value") @@ -62,16 +62,19 @@ pub fn resolve_column_ref(col: &ColumnRef, schema: &Schema) -> Result = (0..schema.columns.len()) + let numeric: Vec = (0..schema.fields.len()) .filter(|&i| Some(i) != schema.time_index) .filter(|&i| { - matches!(schema.columns[i].dtype, DataType::Float64 | DataType::Int64) + matches!( + schema.fields[i].dtype, + FieldDataType::Plain(DataType::Float64 | DataType::Int64) + ) }) .collect(); (numeric.len() == 1).then(|| numeric[0]) }) .ok_or_else(|| ResolveError::NoSampleValue { - available: schema.columns.iter().map(|c| c.name.clone()).collect(), + available: schema.fields.iter().map(|c| c.name.clone()).collect(), }), ColumnRef::Wildcard => Err(ResolveError::WildcardNotPositional), } @@ -211,14 +214,14 @@ pub fn output_schema_for_aggregate( #[cfg(test)] mod tests { use super::*; - use crate::pre_asap::schema::Column; + use crate::pre_asap::schema::Field; /// The conventional PromQL leaf shape: `(ts: Timestamp, value: Float64)`. fn ts_value_schema() -> Schema { Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], 0, Vec::new(), @@ -237,8 +240,8 @@ mod tests { // column and two non-ts columns (ambiguous), but exactly one numeric // column — the sample value an outer `topk` ranks by. let s = Schema::new(vec![ - Column::new("job", DataType::Utf8, true), - Column::new("sum", DataType::Float64, false), + Field::plain("job", DataType::Utf8, true), + Field::plain("sum", DataType::Float64, false), ]); assert_eq!(resolve_column_ref(&ColumnRef::SampleValue, &s), Ok(1)); } @@ -250,8 +253,8 @@ mod tests { // bind to it (#70) — resolution fails cleanly instead of picking a label. let s = Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("host", DataType::Utf8, true), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("host", DataType::Utf8, true), ], 0, vec![], @@ -266,8 +269,8 @@ mod tests { fn sample_value_ambiguous_when_two_numeric_columns() { // Two numeric non-ts columns → genuinely ambiguous → NoSampleValue. let s = Schema::new(vec![ - Column::new("a", DataType::Float64, false), - Column::new("b", DataType::Int64, false), + Field::plain("a", DataType::Float64, false), + Field::plain("b", DataType::Int64, false), ]); assert!(matches!( resolve_column_ref(&ColumnRef::SampleValue, &s), @@ -287,9 +290,9 @@ mod tests { // The output of a nested cross-series aggregate: closed `[group, sum]`. // `by (job)` — `job` is provably absent → dropped, not rejected (#53). let s = Schema { - columns: vec![ - Column::new("group", DataType::Utf8, true), - Column::new("sum", DataType::Float64, false), + fields: vec![ + Field::plain("group", DataType::Utf8, true), + Field::plain("sum", DataType::Float64, false), ], time_index: None, unique_keys: vec![], @@ -328,8 +331,8 @@ mod tests { fn aggregate_strips_time_and_keeps_unique_keys() { let mut input = ts_value_schema(); input - .columns - .push(Column::new("host", DataType::Utf8, false)); + .fields + .push(Field::plain("host", DataType::Utf8, false)); let out = output_schema_for_aggregate( &input, &GroupKeys::by(vec![2]), @@ -337,9 +340,9 @@ mod tests { &[], ) .expect("valid group-by column"); - assert_eq!(out.columns.len(), 2); // host, sum - assert_eq!(out.columns[0].name, "host"); - assert_eq!(out.columns[1].name, "sum"); + assert_eq!(out.fields.len(), 2); // host, sum + assert_eq!(out.fields[0].name, "host"); + assert_eq!(out.fields[1].name, "sum"); assert!(out.time_index.is_none()); assert_eq!(out.unique_keys, vec![vec![0]]); } @@ -356,8 +359,8 @@ mod tests { let leaf_schema = Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], 0, vec![], @@ -393,7 +396,7 @@ mod tests { "the two aggregate-schema derivations must agree (issue #41)" ); // Sanity: it really is the label-preserving per-series shape, not `[rate]`. - assert!(having_side.columns.iter().any(|c| c.name == "value")); + assert!(having_side.fields.iter().any(|c| c.name == "value")); assert!(having_side.time_index.is_some()); } } diff --git a/crates/types/src/pre_asap/cse.rs b/crates/types/src/pre_asap/cse.rs index 45842bf1d..0a00475e6 100644 --- a/crates/types/src/pre_asap/cse.rs +++ b/crates/types/src/pre_asap/cse.rs @@ -774,7 +774,7 @@ mod tests { use crate::pre_asap::agg_intent::AggIntent; use crate::pre_asap::expr_ir::{CompareOpKind, ScalarValue}; use crate::pre_asap::query_expr::{BinaryOpKind, GroupKeys, Predicate, Reduction, Source}; - use crate::pre_asap::schema::{Column, DataType, Schema}; + use crate::pre_asap::schema::{DataType, Field, Schema}; use crate::types::AccuracyTarget; /// `[ts, service, value, latency]`. @@ -784,10 +784,10 @@ mod tests { predicates: vec![], schema: Schema::with_time_index( vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("service", DataType::Utf8, false), - Column::new("value", DataType::Float64, false), - Column::new("latency", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("value", DataType::Float64, false), + Field::plain("latency", DataType::Float64, false), ], 0, vec![], diff --git a/crates/types/src/pre_asap/expr_ir.rs b/crates/types/src/pre_asap/expr_ir.rs index cc201a617..88c309903 100644 --- a/crates/types/src/pre_asap/expr_ir.rs +++ b/crates/types/src/pre_asap/expr_ir.rs @@ -1,8 +1,8 @@ -//! Column-reference and scalar-operator vocabulary shared by the whole -//! canonical [`QueryExpr`](super::query_expr::QueryExpr) DAG. +//! Field-reference and scalar-operator vocabulary shared by the whole +//! canonical [`QueryExpr`](super::query_expr::QueryExpr) tree. //! -//! Issue #205: the scalar expression shapes (`Column`/`Literal`/`Compare`/…) -//! used to live in a separate, self-recursive `Expr` DAG here, reachable +//! Issue #205: the scalar expression shapes (`Field`/`Literal`/`Compare`/…) +//! used to live in a separate, self-recursive `Expr` tree here, reachable //! from `QueryExpr` only through wrapper fields (`Predicate`, `ProjectItem`, //! `SortKey`). They're variants of `QueryExpr` itself now — one recursive //! DAG, not two type families joined by wrappers — generic over the same diff --git a/crates/types/src/pre_asap/mod.rs b/crates/types/src/pre_asap/mod.rs index bb9b8e309..4cfcfbd96 100644 --- a/crates/types/src/pre_asap/mod.rs +++ b/crates/types/src/pre_asap/mod.rs @@ -60,5 +60,5 @@ pub use query_expr::{ WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, }; pub use resolve::{resolve_root, ResolveDAGError}; -pub use schema::{Column, ColumnId, DataType, Schema}; +pub use schema::{ColumnId, DataType, Field, FieldDataType, Schema}; pub use schema_resolver::{SchemaCatalog, SchemaResolver, UsageDerivedCatalog}; diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs index 60f0e4ec0..1f6021ba4 100644 --- a/crates/types/src/pre_asap/query_expr.rs +++ b/crates/types/src/pre_asap/query_expr.rs @@ -6,9 +6,9 @@ //! one parent, within one query or across a `QueryWorkload` batch, instead of //! being duplicated. Nothing in this module produces that sharing on its //! own — construction still allocates a fresh `Rc` per node, the same shape -//! as the old `Box` DAG — a separate CSE pass is what turns two -//! independently constructed, structurally-equal sub-DAGs into two -//! references to one `Rc` (issue #212, #222). Column identity is +//! as the old `Box` tree — a separate CSE pass is what turns two +//! independently constructed, structurally-equal subtrees into two +//! references to one `Rc` (issue #212, #222). Field identity is //! **positional** (`Aggregate.reduction: Reduction`, wrapping `GroupKeys` //! for the grouped case), resolved by the [`SchemaResolver`](super::schema_resolver) against //! the self-contained [`Schema`] carried on each `Scan`. @@ -21,7 +21,7 @@ use thiserror::Error; use super::agg_intent::AggIntent; use super::expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; -use super::schema::{Column, ColumnId, DataType, Schema}; +use super::schema::{ColumnId, DataType, Field, FieldDataType, Schema}; /// The column-reference resolution state a [`QueryExpr`] DAG carries — /// [`ColumnId`] (the default, and what the bare `QueryExpr` name has always @@ -1210,11 +1210,11 @@ impl QueryExpr { // is no longer provable — drop unique_keys. QueryExpr::PromqlRelabel { dst, child, .. } => { let mut out = child.output_schema()?; - if let Some(existing) = out.columns.iter_mut().find(|c| c.name == *dst) { - existing.dtype = DataType::Utf8; + if let Some(existing) = out.fields.iter_mut().find(|c| c.name == *dst) { + existing.dtype = FieldDataType::Plain(DataType::Utf8); existing.nullable = true; } else { - out.columns.push(Column::new(dst.clone(), DataType::Utf8, true)); + out.fields.push(Field::plain(dst.clone(), DataType::Utf8, true)); } out.unique_keys.clear(); Ok(out) @@ -1224,12 +1224,12 @@ impl QueryExpr { // inferred from its expression against the child schema; the name // is the explicit alias or a derived default. A child unique key // survives exactly when every one of its columns is passed through - // as a bare `Column` item (possibly reordered or aliased). Derived + // as a bare `Field` item (possibly reordered or aliased). Derived // expressions cannot carry key identity. `time_index` is re-found // by name. QueryExpr::Project { cols, qualifier, child } => { let in_schema = child.output_schema()?; - let columns: Vec = cols + let columns: Vec = cols .iter() .enumerate() .map(|(i, item)| { @@ -1238,7 +1238,7 @@ impl QueryExpr { .alias .clone() .unwrap_or_else(|| default_proj_name(&item.expr, i, &in_schema)); - let c = Column::new(name, dtype, nullable); + let c = Field::plain(name, dtype, nullable); // A derived table re-qualifies its output columns with // its alias, so `t.col` (and a join over two derived // tables) resolves to the right relation. @@ -1263,7 +1263,7 @@ impl QueryExpr { }) .collect(); Ok(Schema { - columns, + fields: columns, time_index, unique_keys, // Projection enumerates exactly its items → closed. @@ -1338,19 +1338,19 @@ impl QueryExpr { JoinKind::Inner | JoinKind::Cross => (false, false), JoinKind::Semi | JoinKind::Anti => unreachable!("handled above"), }; - let l_len = l.columns.len(); - let mut columns = Vec::with_capacity(l_len + r.columns.len()); - columns.extend(l.columns.iter().cloned().map(|mut c| { + let l_len = l.fields.len(); + let mut columns = Vec::with_capacity(l_len + r.fields.len()); + columns.extend(l.fields.iter().cloned().map(|mut c| { c.nullable |= left_null; c })); - columns.extend(r.columns.iter().cloned().map(|mut c| { + columns.extend(r.fields.iter().cloned().map(|mut c| { c.nullable |= right_null; c })); let time_index = l.time_index.or(r.time_index.map(|i| i + l_len)); Ok(Schema { - columns, + fields: columns, time_index, unique_keys: Vec::new(), // The concatenation is complete only if both sides are. @@ -1369,10 +1369,13 @@ impl QueryExpr { // First operand's (dtype, nullable) from the child schema, owned // so the borrow ends before we append. let arg = args.first().and_then(|a| match a { - QueryExpr::Column(id) => out.columns.get(*id), + QueryExpr::Column(id) => out.fields.get(*id), _ => None, }); - let arg_dtype = || arg.map_or(DataType::Float64, |c| c.dtype.clone()); + let arg_dtype = || { + arg.and_then(|c| c.plain_dtype().cloned()) + .unwrap_or(DataType::Float64) + }; let (dtype, nullable) = match func { WindowFuncKind::RowNumber | WindowFuncKind::Rank @@ -1391,8 +1394,8 @@ impl QueryExpr { (arg_dtype(), arg.is_none_or(|c| c.nullable)) } }; - out.columns - .push(Column::new(output_name.clone(), dtype, nullable)); + out.fields + .push(Field::plain(output_name.clone(), dtype, nullable)); Ok(out) } @@ -1404,14 +1407,14 @@ impl QueryExpr { // `Literal(Float64)` (issue #220), so the schema doesn't need to // inspect the inner node. QueryExpr::PromqlScalarBridge(_) | QueryExpr::EvalTimestamp => Ok(Schema { - columns: vec![Column::new("value", DataType::Float64, false)], + fields: vec![Field::plain("value", DataType::Float64, false)], time_index: None, unique_keys: Vec::new(), closed: true, }), QueryExpr::CurrentTimestamp => Ok(Schema { - columns: vec![Column::new("value", DataType::Timestamp, false)], + fields: vec![Field::plain("value", DataType::Timestamp, false)], time_index: None, unique_keys: Vec::new(), closed: true, @@ -1421,9 +1424,9 @@ impl QueryExpr { // floor and nothing else. `closed` — its full label set (empty) is // known statically (#48). QueryExpr::PromqlVectorFromScalar(_) => Ok(Schema { - columns: vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + fields: vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ], time_index: Some(0), unique_keys: Vec::new(), @@ -1433,7 +1436,7 @@ impl QueryExpr { // `scalar(v)` collapses to a single `value`, no time index — the same // scalar shape as a constant or `time()` (#48). QueryExpr::PromqlScalarFromVector(_) => Ok(Schema { - columns: vec![Column::new("value", DataType::Float64, false)], + fields: vec![Field::plain("value", DataType::Float64, false)], time_index: None, unique_keys: Vec::new(), closed: true, @@ -1460,14 +1463,14 @@ impl QueryExpr { || matches!(grouping, Some(g) if g.side == GroupSide::Right); let mut additions = Vec::new(); if right_rows { - additions.extend(right.columns.iter().filter(|c| c.dtype == DataType::Utf8).cloned()); + additions.extend(right.fields.iter().filter(|c| c.dtype == DataType::Utf8).cloned()); } if let Some(grouping) = grouping { - additions.extend(grouping.labels.iter().map(|name| Column::new(name.clone(), DataType::Utf8, true))); + additions.extend(grouping.labels.iter().map(|name| Field::plain(name.clone(), DataType::Utf8, true))); } for column in additions { - if !output.columns.iter().any(|c| c.name == column.name) { - output.columns.push(column); + if !output.fields.iter().any(|c| c.name == column.name) { + output.fields.push(column); } } Ok(output) @@ -1504,14 +1507,14 @@ fn per_series_reduction_schema(input: &Schema, agg: &AggIntent) -> Result Result = Vec::with_capacity(by.len() + measures.len()); + let mut out_cols: Vec = Vec::with_capacity(by.len() + measures.len()); for &id in by.keys() { let c = in_schema - .columns + .fields .get(id) .ok_or(QueryExprError::InvalidGroupByColumn( id, - in_schema.columns.len(), + in_schema.fields.len(), ))?; out_cols.push(c.clone()); } let value_col_idx = super::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, in_schema) .ok() - .or_else(|| (0..in_schema.columns.len()).find(|i| !by.contains(i))); + .or_else(|| (0..in_schema.fields.len()).find(|i| !by.contains(i))); let probe = value_col_idx - .and_then(|i| in_schema.columns.get(i)) + .and_then(|i| in_schema.fields.get(i)) .cloned() - .unwrap_or_else(|| Column::new("value", DataType::Float64, false)); + .unwrap_or_else(|| Field::plain("value", DataType::Float64, false)); // Each reducer types off its own input column (`SUM(bytes)` vs `AVG(latency)` // in one node); `None` falls back to the sample-value probe (PromQL's // single-column convention). A non-empty `output_names[i]` overrides the @@ -1601,7 +1604,7 @@ pub fn aggregate_output_schema( // duplicate. if let AggIntent::CountValues { label } = intent { if !out_cols.iter().any(|c| c.name == *label) { - out_cols.push(Column::new(label.clone(), DataType::Utf8, false)); + out_cols.push(Field::plain(label.clone(), DataType::Utf8, false)); } let mut cnt = intent.output_column(&probe); if let Some(name) = output_names.get(i).filter(|s| !s.is_empty()) { @@ -1616,7 +1619,7 @@ pub fn aggregate_output_schema( let in_col = intent .input_cols() .first() - .and_then(|id| in_schema.columns.get(*id)) + .and_then(|id| in_schema.fields.get(*id)) .unwrap_or(&probe); let mut out = intent.output_column(in_col); // A global extremum emits NULL for an empty input, even if its input @@ -1628,8 +1631,8 @@ pub fn aggregate_output_schema( .arg_selector_columns(in_schema) .map_err(QueryExprError::InvalidScalarSignature)? { - out.dtype = in_schema.columns[arg].dtype.clone(); - out.nullable = in_schema.columns[arg].nullable; + out.dtype = in_schema.fields[arg].dtype.clone(); + out.nullable = in_schema.fields[arg].nullable; } if let Some(name) = output_names.get(i).filter(|s| !s.is_empty()) { out.name = name.clone(); @@ -1647,7 +1650,7 @@ pub fn aggregate_output_schema( vec![(0..by.len()).collect()] }; Ok(Schema { - columns: out_cols, + fields: out_cols, time_index: None, unique_keys, // A cross-series aggregate enumerates exactly `by ++ measures`, so its output @@ -1670,10 +1673,10 @@ fn without_output_schema( output_names: &[String], ) -> Result { for &id in excluded { - if id >= in_schema.columns.len() { + if id >= in_schema.fields.len() { return Err(QueryExprError::InvalidGroupByColumn( id, - in_schema.columns.len(), + in_schema.fields.len(), )); } } @@ -1681,17 +1684,17 @@ fn without_output_schema( // it is still the value, not a kept label. let value = super::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, in_schema).ok(); - let mut out_cols: Vec = Vec::new(); - for (i, col) in in_schema.columns.iter().enumerate() { + let mut out_cols: Vec = Vec::new(); + for (i, col) in in_schema.fields.iter().enumerate() { let is_time = in_schema.time_index == Some(i); if !is_time && value != Some(i) && !excluded.contains(&i) { out_cols.push(col.clone()); } } let probe = value - .and_then(|i| in_schema.columns.get(i)) + .and_then(|i| in_schema.fields.get(i)) .cloned() - .unwrap_or_else(|| Column::new("value", DataType::Float64, false)); + .unwrap_or_else(|| Field::plain("value", DataType::Float64, false)); for (i, intent) in measures.iter().enumerate() { // Only the output *type* is read from here, so the leading column is // enough for the multi-column intents: `Cardinality` and `PearsonCorr` @@ -1699,15 +1702,15 @@ fn without_output_schema( let in_col = intent .input_cols() .first() - .and_then(|id| in_schema.columns.get(*id)) + .and_then(|id| in_schema.fields.get(*id)) .unwrap_or(&probe); let mut out = intent.output_column(in_col); if let Some((arg, _)) = intent .arg_selector_columns(in_schema) .map_err(QueryExprError::InvalidScalarSignature)? { - out.dtype = in_schema.columns[arg].dtype.clone(); - out.nullable = in_schema.columns[arg].nullable; + out.dtype = in_schema.fields[arg].dtype.clone(); + out.nullable = in_schema.fields[arg].nullable; } if let Some(name) = output_names.get(i).filter(|s| !s.is_empty()) { out.name = name.clone(); @@ -1715,7 +1718,7 @@ fn without_output_schema( out_cols.push(out); } Ok(Schema { - columns: out_cols, + fields: out_cols, time_index: None, unique_keys: Vec::new(), // The kept label set is runtime-only, so — unlike `by` — this does not @@ -1736,11 +1739,20 @@ fn infer_expr_type( ) -> Result<(DataType, bool), QueryExprError> { Ok(match expr { QueryExpr::CurrentTimestamp => (DataType::Timestamp, false), - QueryExpr::Column(id) => schema - .columns - .get(*id) - .map(|c| (c.dtype.clone(), c.nullable)) - .unwrap_or((DataType::Float64, true)), + QueryExpr::Column(id) => match schema.fields.get(*id) { + Some(c) => match c.plain_dtype() { + Some(dtype) => (dtype.clone(), c.nullable), + // Summary state is not a scalar value: it has to be read + // out (estimated / finalized) before an expression can use it. + None => { + return Err(QueryExprError::InvalidScalarSignature(format!( + "column `{}` carries summary state and cannot be read as a value", + c.name + ))) + } + }, + None => (DataType::Float64, true), + }, QueryExpr::Literal(s) => match s { ScalarValue::Int64(_) => (DataType::Int64, false), ScalarValue::Float64(_) => (DataType::Float64, false), @@ -1843,7 +1855,7 @@ fn infer_expr_type( fn default_proj_name(expr: &QueryExpr, idx: usize, schema: &Schema) -> String { match expr { QueryExpr::Column(id) => schema - .columns + .fields .get(*id) .map(|c| c.name.clone()) .unwrap_or_else(|| format!("col_{idx}")), @@ -1857,8 +1869,8 @@ mod tests { use crate::pre_asap::expr_ir::{ArithmeticOpKind, CompareOpKind}; use crate::types::AccuracyTarget; - fn col(name: &str, dtype: DataType, nullable: bool) -> Column { - Column::new(name, dtype, nullable) + fn col(name: &str, dtype: DataType, nullable: bool) -> Field { + Field::plain(name, dtype, nullable) } /// Shifting an instant by a duration stays an instant, and shifting a date @@ -1911,7 +1923,7 @@ mod tests { } fn scan( - columns: Vec, + columns: Vec, time_index: Option, uk: Vec>, ) -> QueryExpr { @@ -1921,7 +1933,7 @@ mod tests { }, predicates: vec![], schema: Schema { - columns, + fields: columns, time_index, unique_keys: uk, closed: true, @@ -2051,7 +2063,7 @@ mod tests { "the union of two deduplicated branches is not deduplicated" ); // The column shape is still the first branch's. - assert_eq!(schema.columns.len(), 2); + assert_eq!(schema.fields.len(), 2); } /// Same rule as `SetOp`, which already dropped them. @@ -2091,7 +2103,7 @@ mod tests { fn discriminator_override_produces_a_compound_unique_key() { // Two branches, each individually deduplicated on column 0 (`k`) — // but, per `merge_drops_the_branches_unique_keys`, that alone proves - // nothing about the union. Column 1 (`branch_id`) stands in for a + // nothing about the union. Field 1 (`branch_id`) stands in for a // discriminator the constructor has separately proven distinct per // branch (PromQL φ, a synthetic `GROUPING()` id, ...) — this // schema-level test only checks the shape `output_schema` derives @@ -2119,7 +2131,7 @@ mod tests { "(discriminator, inner_key) is the sole asserted unique key" ); assert_eq!( - schema.columns.len(), + schema.fields.len(), 2, "column shape is still the first branch's" ); @@ -2224,10 +2236,10 @@ mod tests { child: Rc::new(child), }; let s = q.output_schema().unwrap(); - assert_eq!(s.columns.len(), 3); - assert_eq!(s.columns[0], col("host", DataType::Utf8, false)); - assert_eq!(s.columns[1], col("dbl", DataType::Float64, false)); - assert_eq!(s.columns[2], col("flag", DataType::Bool, true)); + assert_eq!(s.fields.len(), 3); + assert_eq!(s.fields[0], col("host", DataType::Utf8, false)); + assert_eq!(s.fields[1], col("dbl", DataType::Float64, false)); + assert_eq!(s.fields[2], col("flag", DataType::Bool, true)); // projection drops the time axis + unique keys (ts not retained) assert!(s.time_index.is_none()); assert!(s.unique_keys.is_empty()); @@ -2292,7 +2304,7 @@ mod tests { child: Rc::new(scan_node), }; let s = agg.output_schema().unwrap(); - let names: Vec<_> = s.columns.iter().map(|c| c.name.as_str()).collect(); + let names: Vec<_> = s.fields.iter().map(|c| c.name.as_str()).collect(); assert_eq!(names, vec!["job", "sum"], "kept `job`, dropped `instance`"); assert!(!s.closed, "a `without` result stays open"); assert!(s.time_index.is_none()); @@ -2321,7 +2333,7 @@ mod tests { child: Rc::new(inner), }; let s = agg.output_schema().unwrap(); - let names: Vec<_> = s.columns.iter().map(|c| c.name.as_str()).collect(); + let names: Vec<_> = s.fields.iter().map(|c| c.name.as_str()).collect(); assert_eq!(names, vec!["job", "sum"]); } @@ -2387,9 +2399,9 @@ mod tests { ] { let output = aggregate_output_schema(&input, &Reduction::PerEntity, &[aggregate], &[]).unwrap(); - assert_eq!(output.columns[0], input.columns[0]); - assert_eq!(output.columns[1].name, "value"); - assert_eq!(output.columns[1].dtype, DataType::Float64); + assert_eq!(output.fields[0], input.fields[0]); + assert_eq!(output.fields[1].name, "value"); + assert_eq!(output.fields[1].dtype, DataType::Float64); } } @@ -2421,10 +2433,7 @@ mod tests { }; let s = rate.output_schema().unwrap(); assert_eq!( - s.columns - .iter() - .map(|c| c.name.as_str()) - .collect::>(), + s.fields.iter().map(|c| c.name.as_str()).collect::>(), vec!["ts", "value", "job"], "rate preserves all labels; only the sample value is replaced" ); @@ -2460,10 +2469,7 @@ mod tests { }; let s = avg_over_time.output_schema().unwrap(); assert_eq!( - s.columns - .iter() - .map(|c| c.name.as_str()) - .collect::>(), + s.fields.iter().map(|c| c.name.as_str()).collect::>(), vec!["ts", "value", "job"], "TimeRange-child marks per-series: labels preserved, value renamed" ); @@ -2550,8 +2556,8 @@ mod tests { child: Rc::new(child), }; let s = q.output_schema().unwrap(); - assert_eq!(s.columns[0].name, "value"); - assert_eq!(s.columns[1].name, "ts"); + assert_eq!(s.fields[0].name, "value"); + assert_eq!(s.fields[1].name, "ts"); assert_eq!(s.time_index, Some(1)); } @@ -2569,9 +2575,9 @@ mod tests { #[test] fn inner_join_concatenates_both_sides() { let s = join(JoinKind::Inner).output_schema().unwrap(); - assert_eq!(s.columns.len(), 2); - assert_eq!(s.columns[0], col("a", DataType::Int64, false)); - assert_eq!(s.columns[1], col("b", DataType::Utf8, false)); + assert_eq!(s.fields.len(), 2); + assert_eq!(s.fields[0], col("a", DataType::Int64, false)); + assert_eq!(s.fields[1], col("b", DataType::Utf8, false)); // post-join row identity not provable → no unique keys assert!(s.unique_keys.is_empty()); } @@ -2579,15 +2585,15 @@ mod tests { #[test] fn left_join_makes_right_side_nullable() { let s = join(JoinKind::Left).output_schema().unwrap(); - assert!(!s.columns[0].nullable, "preserved left side stays non-null"); - assert!(s.columns[1].nullable, "right side nullable under LEFT JOIN"); + assert!(!s.fields[0].nullable, "preserved left side stays non-null"); + assert!(s.fields[1].nullable, "right side nullable under LEFT JOIN"); } #[test] fn full_join_makes_both_sides_nullable() { let s = join(JoinKind::Full).output_schema().unwrap(); - assert!(s.columns[0].nullable); - assert!(s.columns[1].nullable); + assert!(s.fields[0].nullable); + assert!(s.fields[1].nullable); } #[test] @@ -2615,8 +2621,8 @@ mod tests { right: Rc::new(right), }; let s = q.output_schema().unwrap(); - assert_eq!(s.columns.len(), 2); - assert_eq!(s.columns[0].name, "k"); + assert_eq!(s.fields.len(), 2); + assert_eq!(s.fields[0].name, "k"); assert!( s.unique_keys.is_empty(), "UNION does not preserve row identity" @@ -2666,9 +2672,9 @@ mod tests { fn row_schema_rides_on_the_bridge_wrapper_not_the_literal_variant() { let bridged = QueryExpr::::promql_scalar(42.0); let schema = bridged.output_schema().expect("bridge has a row schema"); - assert_eq!(schema.columns.len(), 1); - assert_eq!(schema.columns[0].name, "value"); - assert_eq!(schema.columns[0].dtype, DataType::Float64); + assert_eq!(schema.fields.len(), 1); + assert_eq!(schema.fields[0].name, "value"); + assert_eq!(schema.fields[0].dtype, DataType::Float64); assert!(schema.time_index.is_none()); // The identical value, unwrapped (the scalar-sub-language position a diff --git a/crates/types/src/pre_asap/resolve.rs b/crates/types/src/pre_asap/resolve.rs index 104c51a68..b4a5c87a6 100644 --- a/crates/types/src/pre_asap/resolve.rs +++ b/crates/types/src/pre_asap/resolve.rs @@ -307,7 +307,7 @@ fn resolve( let left = resolve_root_with_inherited(left, &[])?; let right = resolve_root_with_inherited(right, &[])?; let mut concat = left.output_schema()?; - concat.columns.extend(right.output_schema()?.columns); + concat.fields.extend(right.output_schema()?.fields); let pred = Predicate(Rc::new(resolve_expr(&pred.0, &concat)?)); QE::Join { kind: kind.clone(), @@ -466,7 +466,7 @@ fn resolve( /// value)` floor. fn inherited_names(schema: &Schema) -> Vec { schema - .columns + .fields .iter() .filter(|c| c.name != "ts" && c.name != "value") .map(|c| c.name.clone()) @@ -644,7 +644,7 @@ mod tests { fn resolve_measure_filters_against_the_child_schema() { use crate::pre_asap::expr_ir::ScalarValue; use crate::pre_asap::query_expr::Predicate; - use crate::pre_asap::{Column, DataType, GroupKeys}; + use crate::pre_asap::{DataType, Field, GroupKeys}; use crate::types::AccuracyTarget; let scan = || UnresolvedQueryExpr::Scan { source: Source::Table { @@ -652,9 +652,9 @@ mod tests { }, predicates: vec![], schema: Some(Schema::new(vec![ - Column::new("service", DataType::Utf8, false), - Column::new("latency", DataType::Float64, false), - Column::new("bytes", DataType::Int64, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("latency", DataType::Float64, false), + Field::plain("bytes", DataType::Int64, false), ])), }; let aggregate = |filters| UnresolvedQueryExpr::Aggregate { @@ -703,10 +703,10 @@ mod tests { // Both sides resolve with qualifiers; an unknown right input is an error. #[test] fn resolve_pearson_corr_inputs() { - use crate::pre_asap::{Column, DataType}; + use crate::pre_asap::{DataType, Field}; let schema = Schema::new(vec![ - Column::new("x", DataType::Float64, true).with_table("a"), - Column::new("x", DataType::Float64, true).with_table("b"), + Field::plain("x", DataType::Float64, true).with_table("a"), + Field::plain("x", DataType::Float64, true).with_table("b"), ]); let intent = AggIntent::PearsonCorr { left: ColumnRef::Qualified { @@ -736,11 +736,11 @@ mod tests { // fails rather than silently shortening the tuple. #[test] fn resolve_distinct_tuple_columns() { - use crate::pre_asap::{Column, DataType}; + use crate::pre_asap::{DataType, Field}; use crate::types::AccuracyTarget; let schema = Schema::new(vec![ - Column::new("k", DataType::Int64, true).with_table("a"), - Column::new("k", DataType::Int64, true).with_table("b"), + Field::plain("k", DataType::Int64, true).with_table("a"), + Field::plain("k", DataType::Int64, true).with_table("b"), ]); let qualified = |table: &str| ColumnRef::Qualified { table: table.into(), diff --git a/crates/types/src/pre_asap/scalar_signature.rs b/crates/types/src/pre_asap/scalar_signature.rs index 7d61f6247..6e316169c 100644 --- a/crates/types/src/pre_asap/scalar_signature.rs +++ b/crates/types/src/pre_asap/scalar_signature.rs @@ -200,7 +200,7 @@ mod tests { #[cfg(test)] mod projection_tests { use super::*; - use crate::pre_asap::{Column, ProjectItem, QueryExpr, ScalarValue, Schema, Source}; + use crate::pre_asap::{Field, ProjectItem, QueryExpr, ScalarValue, Schema, Source}; use std::rc::Rc; fn project(expr: QueryExpr) -> QueryExpr { QueryExpr::Project { @@ -215,8 +215,8 @@ mod projection_tests { }, predicates: vec![], schema: Schema::new(vec![ - Column::new("k", DataType::Utf8, false), - Column::new("v", DataType::Int64, true), + Field::plain("k", DataType::Utf8, false), + Field::plain("v", DataType::Int64, true), ]), }), } @@ -229,21 +229,21 @@ mod projection_tests { }; let schema = project(map.clone()).output_schema().unwrap(); assert_eq!( - schema.columns[0].dtype, + schema.fields[0].dtype, DataType::Map { key: Box::new(DataType::Utf8), value: Box::new(DataType::Int64), value_nullable: true } ); - assert!(!schema.columns[0].nullable); + assert!(!schema.fields[0].nullable); let lookup = QueryExpr::FunctionCall { name: "asap_map_access".into(), args: vec![map, QueryExpr::Literal(ScalarValue::Utf8("missing".into()))], }; assert_eq!( - project(lookup).output_schema().unwrap().columns[0], - Column::new("result", DataType::Int64, true) + project(lookup).output_schema().unwrap().fields[0], + Field::plain("result", DataType::Int64, true) ); assert!(project(QueryExpr::FunctionCall { name: "map".into(), @@ -301,17 +301,17 @@ pub fn struct_field_type( #[cfg(test)] mod struct_field_tests { use super::*; - use crate::pre_asap::{Column, QueryExpr, ScalarValue, Schema}; + use crate::pre_asap::{Field, FieldDataType, QueryExpr, ScalarValue, Schema}; fn schema() -> Schema { - Schema::new(vec![Column::new( + Schema::new(vec![Field::plain( "record", DataType::Struct { fields: vec![ - Column::new("ts", DataType::Int64, false), - Column::new( + Field::new("ts", DataType::Int64, false), + Field::new( "values", DataType::List { - element: Box::new(Column::new("item", DataType::Float64, true)), + element: Box::new(Field::new("item", DataType::Float64, true)), }, true, ), @@ -345,7 +345,7 @@ mod struct_field_tests { named.scalar_type(&schema).unwrap(), ( DataType::List { - element: Box::new(Column::new("item", DataType::Float64, true)) + element: Box::new(Field::new("item", DataType::Float64, true)) }, true ) @@ -366,14 +366,14 @@ mod struct_field_tests { assert!(access(selector).scalar_type(&schema()).is_err()); } let mut ambiguous = schema(); - if let DataType::Struct { fields } = &mut ambiguous.columns[0].dtype { - fields.push(Column::new("ts", DataType::Utf8, false)); + if let FieldDataType::Plain(DataType::Struct { fields }) = &mut ambiguous.fields[0].dtype { + fields.push(Field::new("ts", DataType::Utf8, false)); } assert!(access(QueryExpr::Literal(ScalarValue::Utf8("ts".into()))) .scalar_type(&ambiguous) .is_err()); let mut nullable = schema(); - nullable.columns[0].nullable = true; + nullable.fields[0].nullable = true; assert!(access(QueryExpr::Literal(ScalarValue::Int64(1))) .scalar_type(&nullable) .is_err()); @@ -422,7 +422,7 @@ pub fn element_access_type( #[cfg(test)] mod element_access_tests { use super::*; - use crate::pre_asap::{Column, QueryExpr, ScalarValue, Schema}; + use crate::pre_asap::{Field, QueryExpr, ScalarValue, Schema}; fn access(index: QueryExpr) -> QueryExpr { QueryExpr::FunctionCall { name: "asap_element_access".into(), @@ -433,19 +433,19 @@ mod element_access_tests { fn list_index_preserves_nested_element_metadata() { let element = DataType::Struct { fields: vec![ - Column::new("ts", DataType::Int64, false), - Column::new("value", DataType::Float64, true), + Field::new("ts", DataType::Int64, false), + Field::new("value", DataType::Float64, true), ], }; let schema = Schema::new(vec![ - Column::new( + Field::plain( "samples", DataType::List { - element: Box::new(Column::new("item", element.clone(), false)), + element: Box::new(Field::new("item", element.clone(), false)), }, false, ), - Column::new("i", DataType::Int64, true), + Field::plain("i", DataType::Int64, true), ]); for index in [1, -1, 100] { assert_eq!( @@ -482,7 +482,7 @@ mod element_access_tests { } #[test] fn generic_map_lookup_reuses_legacy_signature() { - let schema = Schema::new(vec![Column::new( + let schema = Schema::new(vec![Field::plain( "m", DataType::Map { key: Box::new(DataType::Utf8), diff --git a/crates/types/src/pre_asap/schema.rs b/crates/types/src/pre_asap/schema.rs index 9d5b2e226..8e17d611c 100644 --- a/crates/types/src/pre_asap/schema.rs +++ b/crates/types/src/pre_asap/schema.rs @@ -1,56 +1,60 @@ -//! Pre-ASAP IR schema flow — every edge carries a typed `Schema`. +//! Per-edge schema of the operator IR — every edge carries a typed `Schema`. //! -//! Per `control_plane/docs/design.md` §6 "Schema flow — every L3 edge carries -//! a typed schema" (that doc's own layer numbering; this crate no longer uses -//! it). The DAG is type-checked: a node's output schema is a function of its -//! inputs and parameters and is verifiable independently of the surrounding -//! context. +//! One `Schema` type serves every operator, before and after ASAP +//! optimization: a field's [`FieldDataType`] is either a plain readable +//! [`DataType`] or the summary / exact-accumulator state a `SummaryAgg` +//! produces. The DAG is type-checked: a node's output schema is a function of +//! its inputs and parameters and is verifiable independently of the +//! surrounding context. //! //! `Schema::unique_keys` is metadata for reuse-aware planning: a producer's //! output can only be safely shared across consumers when its row identity //! is provably stable across reads, which is what this field records. -//! -//! Single-query plans don't read this field; it lives here so the metadata -//! is available the moment workload-aware planning lands without requiring -//! a pre-ASAP-IR-wide schema change. #![allow(dead_code)] use serde::{Deserialize, Serialize}; -/// Index into [`Schema::columns`] used everywhere a column position is +use crate::post_asap::sketch::{ + ExactKind, ExactParams, GroupingStrategy, SamplingKind, SamplingParams, SketchKind, + StatModelKind, StatModelParams, WaveletKind, WaveletParams, +}; + +/// Index into [`Schema::fields`] used everywhere a column position is /// referenced (group-by keys, unique-key sets, the time axis index). /// -/// Aliased to `usize` to match `design.md`'s `Vec>` for -/// `unique_keys`. Kept as a named type so downstream code can pattern on -/// the intent ("this is a column position, not just any number"). +/// Kept as a named type so downstream code can pattern on the intent ("this +/// is a column position, not just any number"). pub type ColumnId = usize; -/// One column in a [`Schema`]. Mirrors `design.md` §6 `Field` — -/// `name + dtype + nullable`. +/// One field of a [`Schema`]: `name + dtype + nullable`, plus an optional +/// table qualifier. The struct describes a column and holds none of its data. +/// +/// `T` is the type vocabulary: [`FieldDataType`] on an operator edge (the +/// default, and what [`Schema::fields`] holds), plain [`DataType`] for the +/// nested element fields of [`DataType::List`] / [`DataType::Struct`], which +/// can never carry summary state. #[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] -pub struct Column { - /// Column name as it appears in the producer's output. PromQL leaves +pub struct Field { + /// Field name as it appears in the producer's output. PromQL leaves /// produce label-name + the synthetic `value` / `timestamp` columns; /// SQL leaves carry their `information_schema` names. pub name: String, - /// Logical scalar or collection type. Sketch state remains a post-ASAP - /// concern and is intentionally absent here. - pub dtype: DataType, - /// Whether NULL values are allowed in this column. PromQL value + pub dtype: T, + /// Whether NULL values are allowed in this field. PromQL value /// columns are non-nullable; SQL columns inherit their DDL nullability. pub nullable: bool, /// Optional table/alias qualifier (SQL `t.col` / `t AS a` → `a`). Travels - /// with the column through joins so a `ColumnRef::Qualified` can pick the + /// with the field through joins so a `ColumnRef::Qualified` can pick the /// right side when both carry the same `name`. `None` for PromQL labels and /// unqualified columns. #[serde(default)] pub table: Option, } -impl Column { - /// An unqualified column (`table = None`). - pub fn new(name: impl Into, dtype: DataType, nullable: bool) -> Self { +impl Field { + /// An unqualified field (`table = None`). + pub fn new(name: impl Into, dtype: T, nullable: bool) -> Self { Self { name: name.into(), dtype, @@ -59,16 +63,113 @@ impl Column { } } - /// This column re-qualified under `table` (e.g. by a `SubqueryAlias`). + /// This field re-qualified under `table` (e.g. by a `SubqueryAlias`). pub fn with_table(mut self, table: impl Into) -> Self { self.table = Some(table.into()); self } } -/// Pre-ASAP IR column data types. Deliberately narrow: no sketch state at -/// this layer (see `design.md` §6.4 for the post-ASAP `DataType::Sketch(...)` -/// extension). +impl Field { + /// An unqualified field carrying an ordinary readable value. + pub fn plain(name: impl Into, dtype: DataType, nullable: bool) -> Self { + Self::new(name, FieldDataType::Plain(dtype), nullable) + } + + /// The value type of a plain field; `None` for summary / accumulator state. + pub fn plain_dtype(&self) -> Option<&DataType> { + match &self.dtype { + FieldDataType::Plain(dtype) => Some(dtype), + _ => None, + } + } + + /// Whether this field carries an ordinary readable value. + pub fn is_plain(&self) -> bool { + matches!(self.dtype, FieldDataType::Plain(_)) + } + + /// The value type of a plain field; panics on summary / accumulator + /// state. For code that has already established the field is plain + /// (front ends, scalar type inference over value columns). + pub fn expect_plain_dtype(&self) -> &DataType { + self.plain_dtype().unwrap_or_else(|| { + panic!( + "field `{}` carries summary state ({:?}), not a plain value", + self.name, self.dtype + ) + }) + } +} + +impl From> for Field { + fn from(field: Field) -> Self { + Self { + name: field.name, + dtype: FieldDataType::Plain(field.dtype), + nullable: field.nullable, + table: field.table, + } + } +} + +/// What a schema field carries: an ordinary readable value, or the summary / +/// exact-accumulator state produced by a `SummaryAgg`. +/// +/// Every non-`Plain` variant carries the physical state identity required by +/// that family (`Sketch` additionally carries its grouping layout), so the +/// type system can reject merges of incompatible summaries at plan +/// construction time — a `SummaryMerge` over `Sketch(Kll, …)` and +/// `Sketch(Cms, …)` inputs is a plan-time error, and a `Sketch(…)` can never +/// be confused for a `Sample(…)` even though both are "opaque summary state" +/// at a glance. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum FieldDataType { + /// An ordinary, readable value — the closed vocabulary of [`DataType`]. + Plain(DataType), + /// Exact, mergeable accumulator state (`Sum`/`Count`/`Min`/`Max`/`Rate`/ + /// `Increase`). Value consumers require an explicit finalization boundary. + ExactAggregate(ExactKind, ExactParams), + /// Approximate sketch state (KLL/CMS/HLL/…), read out via a + /// `SummaryEstimate`. A [`SketchKind`] already carries the concrete + /// algorithm, params, and grouping layout committed to, not just its + /// category — a bound node needs to know it's specifically independent + /// KLL or shared Hydra-backed CMS, not merely "some sketch". + Sketch(SketchKind, GroupingStrategy), + /// Sampling-based summary state (a retained row subset). + Sample(SamplingKind, SamplingParams), + /// Wavelet-transform summary state (a coefficient vector). + Wavelet(WaveletKind, WaveletParams), + /// Fitted statistical/parametric-model summary state. + StatModel(StatModelKind, StatModelParams), +} + +impl FieldDataType { + pub fn is_plain(&self) -> bool { + matches!(self, FieldDataType::Plain(_)) + } + + pub fn plain(&self) -> Option<&DataType> { + match self { + FieldDataType::Plain(dtype) => Some(dtype), + _ => None, + } + } +} + +impl From for FieldDataType { + fn from(dtype: DataType) -> Self { + FieldDataType::Plain(dtype) + } +} + +impl PartialEq for FieldDataType { + fn eq(&self, other: &DataType) -> bool { + matches!(self, FieldDataType::Plain(dtype) if dtype == other) + } +} + +/// Plain value types. Summary state is a [`FieldDataType`] concern. #[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] #[serde(rename_all = "snake_case")] pub enum DataType { @@ -97,9 +198,9 @@ pub enum DataType { Date, /// Variable-length sequence. The existing column contract preserves the /// element field name, type, and nullability. Nested fields are unqualified. - List { element: Box }, + List { element: Box> }, /// Ordered named fields, including each field's independent nullability. - Struct { fields: Vec }, + Struct { fields: Vec> }, /// SQL map entries with non-null keys and explicitly nullable values. Map { key: Box, @@ -108,8 +209,8 @@ pub enum DataType { }, } -/// Per-edge pre-ASAP IR schema. Flowing between any two operators, on every -/// node's input and output. +/// Per-edge schema. Flowing between any two operators, on every node's +/// input and output. /// /// `unique_keys` is metadata for reuse-aware planning: each inner `Vec` /// is a set of column indices that together uniquely identify rows. The @@ -117,15 +218,11 @@ pub enum DataType { /// unique constraint). Populated by per-node input/output spec — /// `Aggregate { by, .. }` emits `unique_keys = [by]`; `Dedup { cols }` /// adds `cols`; most other nodes pass through. -/// -/// **Consumed by**: a future workload-level reuse pass (not yet shipped). -/// The single-query path, the `Bind*` rules, push-down, and a deployment's -/// own physical emitters do not read this field. -#[derive(Debug, Clone, PartialEq, Eq, Default, Serialize, Deserialize)] +#[derive(Debug, Clone, PartialEq, Default, Serialize, Deserialize)] pub struct Schema { - /// Columns flowing on this edge, in positional order. - pub columns: Vec, - /// Index into `columns` for the time axis, if any. PromQL leaves + /// Fields flowing on this edge, in positional order. + pub fields: Vec, + /// Index into `fields` for the time axis, if any. PromQL leaves /// always carry one; SQL leaves may or may not. #[serde(default)] pub time_index: Option, @@ -179,7 +276,7 @@ pub fn with_promql_series_identity(root: &super::QueryExpr) -> Result { if schema - .columns + .fields .iter() .any(|column| column.name == PROMQL_SERIES_IDENTITY) { @@ -189,8 +286,8 @@ pub fn with_promql_series_identity(root: &super::QueryExpr) -> Result Result bool { self.closed - && self.columns.iter().any(|column| { - column.name == PROMQL_SERIES_IDENTITY - && column.dtype == DataType::Utf8 - && !column.nullable - && column.table.is_none() + && self.fields.iter().any(|field| { + field.name == PROMQL_SERIES_IDENTITY + && field.dtype == DataType::Utf8 + && !field.nullable + && field.table.is_none() }) } - /// Construct a `Schema` from columns alone — no time index, no + /// Construct a `Schema` from fields alone — no time index, no /// unique-key constraint. Used by `Scan` over a tabular source /// when the catalog supplies no primary-key metadata. - pub fn new(columns: Vec) -> Self { + pub fn new(fields: Vec) -> Self { Self { - columns, + fields, time_index: None, unique_keys: Vec::new(), closed: false, @@ -260,42 +357,59 @@ impl Schema { /// Construct a `Scan`-style schema with explicit `time_index` + /// inferred unique keys (e.g. PromQL leaves: `[time_index, label_set]`). pub fn with_time_index( - columns: Vec, + fields: Vec, time_index: ColumnId, unique_keys: Vec>, ) -> Self { Self { - columns, + fields, time_index: Some(time_index), unique_keys, closed: false, } } - /// Look up a column by name (first match). `None` if not present. + /// The schema of a summary-planning node: `fields` and a time axis, no + /// unique-key claim, closed. The shape every post-ASAP operator output + /// carried before pre- and post-ASAP schemas were one type. + pub fn lifted(fields: Vec, time_index: Option) -> Self { + Self { + fields, + time_index, + unique_keys: Vec::new(), + closed: true, + } + } + + /// Whether every field carries an ordinary readable value. + pub fn is_all_plain(&self) -> bool { + self.fields.iter().all(Field::is_plain) + } + + /// Look up a field by name (first match). `None` if not present. pub fn column_id(&self, name: &str) -> Option { - self.columns.iter().position(|c| c.name == name) + self.fields.iter().position(|c| c.name == name) } - /// Look up a column by `(table, name)` qualifier — disambiguates columns - /// that share a `name` across a join (`a.k` vs `b.k`). `None` if no column + /// Look up a field by `(table, name)` qualifier — disambiguates columns + /// that share a `name` across a join (`a.k` vs `b.k`). `None` if no field /// has both that qualifier and name. pub fn column_id_qualified(&self, table: &str, name: &str) -> Option { - self.columns + self.fields .iter() .position(|c| c.name == name && c.table.as_deref() == Some(table)) } /// Whether this schema has *any* provable unique key — the signal a - /// future reuse-aware planning pass would need to decide whether a - /// producer's output can be safely shared across consumers. + /// reuse-aware planning pass needs to decide whether a producer's output + /// can be safely shared across consumers. pub fn has_unique_key(&self) -> bool { !self.unique_keys.is_empty() } /// Append `cols` as an additional unique-key set if not already present. - /// Used by `Dedup { cols }` per design.md §6 schema-flow table: - /// "the input schema with `unique_keys` tightened to include `cols`". + /// Used by `Dedup { cols }`: "the input schema with `unique_keys` + /// tightened to include `cols`". pub fn add_unique_key(&mut self, cols: Vec) { if !self.unique_keys.contains(&cols) { self.unique_keys.push(cols); @@ -309,8 +423,8 @@ impl Schema { mod tests { use super::*; - fn col(name: &str, dtype: DataType) -> Column { - Column::new(name, dtype, false) + fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) } #[test] @@ -375,21 +489,21 @@ mod tests { } #[test] - fn column_table_defaults_to_none_when_absent() { - // `Column.table` is `#[serde(default)]` so schemas serialized before the + fn field_table_defaults_to_none_when_absent() { + // `Field.table` is `#[serde(default)]` so schemas serialized before the // qualifier field existed still deserialize (to `table: None`) instead - // of erroring. Drop the key from a serialized column to simulate that. + // of erroring. Drop the key from a serialized field to simulate that. let mut v = serde_json::to_value(col("svc", DataType::Utf8)).unwrap(); assert!(v.as_object_mut().unwrap().remove("table").is_some()); - let back: Column = serde_json::from_value(v).unwrap(); + let back: Field = serde_json::from_value(v).unwrap(); assert_eq!(back, col("svc", DataType::Utf8)); assert!(back.table.is_none()); } #[test] - fn qualified_column_serde_roundtrip() { + fn qualified_field_serde_roundtrip() { let c = col("service", DataType::Utf8).with_table("hosts"); - let back: Column = serde_json::from_str(&serde_json::to_string(&c).unwrap()).unwrap(); + let back: Field = serde_json::from_str(&serde_json::to_string(&c).unwrap()).unwrap(); assert_eq!(back, c); assert_eq!(back.table.as_deref(), Some("hosts")); } diff --git a/crates/types/src/pre_asap/schema_resolver.rs b/crates/types/src/pre_asap/schema_resolver.rs index d6bbd1610..b63b70afd 100644 --- a/crates/types/src/pre_asap/schema_resolver.rs +++ b/crates/types/src/pre_asap/schema_resolver.rs @@ -13,7 +13,7 @@ use super::expr_ir::ColumnRef; use super::query_expr::UnresolvedQueryExpr; -use super::schema::{Column, DataType, Schema}; +use super::schema::{DataType, Field, Schema}; /// The DB / source-schema metadata source — resolves a source (metric / /// table) name to its known columns. @@ -29,7 +29,7 @@ use super::schema::{Column, DataType, Schema}; pub trait SchemaCatalog { /// Columns known for `source`. `None` when unknown — the [`SchemaResolver`] then /// falls back to a usage-derived column set. - fn columns_for(&self, source: &str) -> Option>; + fn columns_for(&self, source: &str) -> Option>; } /// The default catalog: knows nothing. Every schema the [`SchemaResolver`] produces @@ -37,7 +37,7 @@ pub trait SchemaCatalog { pub struct UsageDerivedCatalog; impl SchemaCatalog for UsageDerivedCatalog { - fn columns_for(&self, _source: &str) -> Option> { + fn columns_for(&self, _source: &str) -> Option> { None } } @@ -86,7 +86,7 @@ impl SchemaResolver { dag: &UnresolvedQueryExpr, inherited: &[String], ) -> Schema { - let mut columns: Vec = leftmost_scan_name(dag) + let mut columns: Vec = leftmost_scan_name(tree) .and_then(|name| self.catalog.columns_for(name)) .unwrap_or_else(default_leaf_columns); @@ -102,13 +102,13 @@ impl SchemaResolver { let referenced = collect_referenced_columns(dag); for name in referenced.iter().chain(inherited) { if !columns.iter().any(|c| c.name == *name) { - columns.push(Column::new(name.clone(), DataType::Utf8, true)); + columns.push(Field::plain(name.clone(), DataType::Utf8, true)); } } let time_index = columns.iter().position(|c| c.name == "ts"); Schema { - columns, + fields: columns, time_index, unique_keys: Vec::new(), // Usage-derived (schemaless PromQL): the metric's full label set is @@ -119,10 +119,10 @@ impl SchemaResolver { } /// The conventional PromQL leaf shape: `(ts: Timestamp, value: Float64)`. -fn default_leaf_columns() -> Vec { +fn default_leaf_columns() -> Vec { vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), ] } @@ -410,9 +410,9 @@ mod tests { #[test] fn bare_source_yields_ts_value_floor() { let schema = SchemaResolver::new().resolve_schema(&src("m")); - assert_eq!(schema.columns.len(), 2); - assert_eq!(schema.columns[0].name, "ts"); - assert_eq!(schema.columns[1].name, "value"); + assert_eq!(schema.fields.len(), 2); + assert_eq!(schema.fields[0].name, "ts"); + assert_eq!(schema.fields[1].name, "value"); assert_eq!(schema.time_index, Some(0)); } @@ -473,12 +473,12 @@ mod tests { fn custom_catalog_supplies_base_columns() { struct FixedCatalog; impl SchemaCatalog for FixedCatalog { - fn columns_for(&self, source: &str) -> Option> { + fn columns_for(&self, source: &str) -> Option> { (source == "known").then(|| { vec![ - Column::new("ts", DataType::Timestamp, false), - Column::new("value", DataType::Float64, false), - Column::new("datacenter", DataType::Utf8, false), + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("datacenter", DataType::Utf8, false), ] }) } @@ -486,7 +486,7 @@ mod tests { let schema = SchemaResolver::with_catalog(FixedCatalog).resolve_schema(&src("known")); let dc = schema .column_id("datacenter") - .and_then(|id| schema.columns.get(id)); + .and_then(|id| schema.fields.get(id)); assert!(matches!(dc, Some(c) if !c.nullable)); } } From d1927ac1a56bc850bef1916f8da5c262e1457fe9 Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Fri, 2 Oct 2026 18:16:58 +0000 Subject: [PATCH 2/7] fix: keep pre-unification schema JSON loadable; viewer reads `fields` - Schema deserializes the old pre-ASAP layout (`columns`, bare dtypes) and the old post-ASAP `SummarySchema` (no `closed`, read as the closed `Schema::lifted` shape), so saved plans survive the upgrade. - dag-viewer render.py reads `fields` (falling back to `columns`); viewer.js unwraps `{"Plain": ...}` dtypes for display. - dag_export: `--table-schema` error names the `columns` key it reads. Co-Authored-By: Claude Opus 5.5 --- crates/devtools/src/bin/dag_export.rs | 2 +- crates/types/src/pre_asap/schema.rs | 128 ++++++++++++++++++++++---- tools/dag-viewer/render.py | 11 ++- tools/dag-viewer/test_render.py | 10 ++ tools/dag-viewer/viewer.js | 9 +- 5 files changed, 137 insertions(+), 23 deletions(-) diff --git a/crates/devtools/src/bin/dag_export.rs b/crates/devtools/src/bin/dag_export.rs index de205c4d4..f4c8928de 100644 --- a/crates/devtools/src/bin/dag_export.rs +++ b/crates/devtools/src/bin/dag_export.rs @@ -741,7 +741,7 @@ fn catalog(custom: &[String]) -> SqlCatalog { .expect("--table-schema.name must be a string"); let columns = value["columns"] .as_array() - .expect("--table-schema.fields must be an array"); + .expect("--table-schema.columns must be an array"); let columns: Vec = columns .iter() .map(|column| { diff --git a/crates/types/src/pre_asap/schema.rs b/crates/types/src/pre_asap/schema.rs index 8e17d611c..fd0a69f6c 100644 --- a/crates/types/src/pre_asap/schema.rs +++ b/crates/types/src/pre_asap/schema.rs @@ -219,6 +219,7 @@ pub enum DataType { /// `Aggregate { by, .. }` emits `unique_keys = [by]`; `Dedup { cols }` /// adds `cols`; most other nodes pass through. #[derive(Debug, Clone, PartialEq, Default, Serialize, Deserialize)] +#[serde(try_from = "SchemaWire")] pub struct Schema { /// Fields flowing on this edge, in positional order. pub fields: Vec, @@ -251,6 +252,73 @@ pub struct Schema { pub closed: bool, } +/// Deserialization form of [`Schema`]. Also reads the two layouts that +/// predate the unified schema, so plans saved by older builds still load: +/// +/// - pre-ASAP `Schema`: `columns` (not `fields`), each `dtype` a bare +/// [`DataType`] (`"float64"`), with `closed` / `unique_keys` present; +/// - post-ASAP `SummarySchema`: `fields` with tagged dtypes +/// (`{"Plain":"float64"}`), but no `closed` / `unique_keys`. Those were the +/// [`Schema::lifted`] shape, so a missing `closed` next to `fields` is closed. +#[derive(Deserialize)] +struct SchemaWire { + fields: Option>, + columns: Option>, + #[serde(default)] + time_index: Option, + #[serde(default)] + unique_keys: Vec>, + closed: Option, +} + +#[derive(Deserialize)] +struct FieldWire { + name: String, + dtype: FieldDataTypeWire, + nullable: bool, + #[serde(default)] + table: Option, +} + +/// A tagged [`FieldDataType`], or a bare [`DataType`] from a pre-ASAP column. +#[derive(Deserialize)] +#[serde(untagged)] +enum FieldDataTypeWire { + Current(FieldDataType), + Legacy(DataType), +} + +impl TryFrom for Schema { + type Error = String; + + fn try_from(wire: SchemaWire) -> Result { + let legacy_summary = wire.fields.is_some(); + let fields = match (wire.fields, wire.columns) { + (Some(fields), None) | (None, Some(fields)) => fields, + (Some(_), Some(_)) => return Err("schema has both `fields` and `columns`".into()), + (None, None) => return Err("missing field `fields`".into()), + }; + let fields = fields + .into_iter() + .map(|field| Field { + name: field.name, + dtype: match field.dtype { + FieldDataTypeWire::Current(dtype) => dtype, + FieldDataTypeWire::Legacy(dtype) => FieldDataType::Plain(dtype), + }, + nullable: field.nullable, + table: field.table, + }) + .collect(); + Ok(Self { + fields, + time_index: wire.time_index, + unique_keys: wire.unique_keys, + closed: wire.closed.unwrap_or(legacy_summary), + }) + } +} + /// Reserved physical row column carrying canonical JSON of a complete PromQL /// label map. `$` cannot occur in a user PromQL label name. pub const PROMQL_SERIES_IDENTITY: &str = "$promql_series_identity"; @@ -478,26 +546,54 @@ mod tests { } #[test] - fn schema_closed_defaults_to_open_when_absent() { - // `closed` is `#[serde(default)]` so schemas serialized before the field - // existed deserialize to `closed: false` (open) — the conservative - // default (don't claim completeness you can't prove). - let mut v = serde_json::to_value(Schema::new(vec![col("a", DataType::Utf8)])).unwrap(); - assert!(v.as_object_mut().unwrap().remove("closed").is_some()); - let back: Schema = serde_json::from_value(v).unwrap(); + fn legacy_pre_asap_schema_deserializes() { + // Pre-unification `Schema`: `columns`, bare dtypes, explicit `closed`. + let json = r#"{"columns":[ + {"name":"ts","dtype":"timestamp","nullable":false,"table":null}, + {"name":"value","dtype":"float64","nullable":true,"table":"t"} + ],"time_index":0,"unique_keys":[[0]],"closed":true}"#; + let back: Schema = serde_json::from_str(json).unwrap(); + let mut expected = Schema::with_time_index( + vec![ + col("ts", DataType::Timestamp), + Field::plain("value", DataType::Float64, true).with_table("t"), + ], + 0, + vec![vec![0]], + ); + expected.closed = true; + assert_eq!(back, expected); + } + + #[test] + fn legacy_pre_asap_schema_without_closed_or_table_is_open() { + // Older still: written before `closed` and `Field.table` existed. + let json = r#"{"columns":[{"name":"a","dtype":"utf8","nullable":false}]}"#; + let back: Schema = serde_json::from_str(json).unwrap(); + assert_eq!(back, Schema::new(vec![col("a", DataType::Utf8)])); assert!(!back.closed, "absent `closed` ⇒ open"); } #[test] - fn field_table_defaults_to_none_when_absent() { - // `Field.table` is `#[serde(default)]` so schemas serialized before the - // qualifier field existed still deserialize (to `table: None`) instead - // of erroring. Drop the key from a serialized field to simulate that. - let mut v = serde_json::to_value(col("svc", DataType::Utf8)).unwrap(); - assert!(v.as_object_mut().unwrap().remove("table").is_some()); - let back: Field = serde_json::from_value(v).unwrap(); - assert_eq!(back, col("svc", DataType::Utf8)); - assert!(back.table.is_none()); + fn legacy_summary_schema_deserializes_as_lifted() { + // Pre-unification post-ASAP `SummarySchema`: no `closed`/`unique_keys`. + let json = r#"{"fields":[ + {"name":"ts","dtype":{"Plain":"timestamp"},"nullable":false}, + {"name":"v","dtype":{"Plain":"float64"},"nullable":false} + ],"time_index":0}"#; + let back: Schema = serde_json::from_str(json).unwrap(); + assert_eq!( + back, + Schema::lifted( + vec![col("ts", DataType::Timestamp), col("v", DataType::Float64)], + Some(0) + ) + ); + } + + #[test] + fn schema_without_fields_is_rejected() { + assert!(serde_json::from_str::(r#"{"closed":true}"#).is_err()); } #[test] diff --git a/tools/dag-viewer/render.py b/tools/dag-viewer/render.py index b131bbc9f..fbdaca97a 100755 --- a/tools/dag-viewer/render.py +++ b/tools/dag-viewer/render.py @@ -87,10 +87,17 @@ def _compact(value: object) -> str: return ", ".join(f"{key}={_compact(item)}" for key, item in value.items()) +def _schema_fields(schema: object) -> object: + """A schema's field list: `fields` today, `columns` in older exports.""" + if not isinstance(schema, dict): + return None + return schema.get("fields", schema.get("columns")) + + def _column(value: object, input_schema: object = None) -> str: if isinstance(value, int): if isinstance(input_schema, dict): - columns = input_schema.get("columns") + columns = _schema_fields(input_schema) if isinstance(columns, list) and value < len(columns): column = columns[value] if isinstance(column, dict) and column.get("name"): @@ -174,7 +181,7 @@ def _semantic_label(node: dict, input_schema: object = None) -> str: lines.append(f"within: {_compact(detail['partition_by'])}") elif kind == "Project": output_schema = node.get("schema") - output_columns = output_schema.get("columns") if isinstance(output_schema, dict) else None + output_columns = _schema_fields(output_schema) if isinstance(output_columns, list) and output_columns: names = [str(column.get("name", "?")) for column in output_columns if isinstance(column, dict)] lines.append("columns: " + ", ".join(names)) diff --git a/tools/dag-viewer/test_render.py b/tools/dag-viewer/test_render.py index d19fc187c..20ccb64cd 100644 --- a/tools/dag-viewer/test_render.py +++ b/tools/dag-viewer/test_render.py @@ -293,6 +293,16 @@ def test_sort_names_expression_direction_and_null_order(self): } self.assertEqual(_semantic_label(node), "Sort\nsort: col[2] descending, nulls first") + def test_project_names_columns_from_fields_and_legacy_columns(self): + """Exports write `fields`; older ones wrote `columns`. Both name columns.""" + for key, dtype in (("fields", {"Plain": "utf8"}), ("columns", "utf8")): + node = { + "kind": "Project", + "detail": {"cols": [0]}, + "schema": {key: [{"name": "service", "dtype": dtype, "nullable": False}]}, + } + self.assertEqual(_semantic_label(node), "Project\ncolumns: service", key) + def test_prepares_before_after_and_whole_post_asap_dags(self): node = { "id": 0, diff --git a/tools/dag-viewer/viewer.js b/tools/dag-viewer/viewer.js index 1db3e835d..6e4c696ee 100644 --- a/tools/dag-viewer/viewer.js +++ b/tools/dag-viewer/viewer.js @@ -896,13 +896,14 @@ function computeSelectionWorkloadCost(selected) { function formatSchema(schema) { if (!schema || typeof schema !== 'object') return 'schema unavailable'; - const fields = Array.isArray(schema.columns) ? schema.columns : schema.fields; + const fields = Array.isArray(schema.fields) ? schema.fields : schema.columns; if (!Array.isArray(fields) || fields.length === 0) return 'empty schema'; const rows = fields.map((field, index) => { const name = field && field.name !== undefined ? field.name : '?'; - const dtype = field && field.dtype !== undefined - ? (typeof field.dtype === 'string' ? field.dtype : JSON.stringify(field.dtype)) - : '?'; + // A plain value is tagged `{"Plain": }`; older exports wrote it bare. + const raw = field && field.dtype !== undefined ? field.dtype : '?'; + const plain = raw && typeof raw === 'object' && 'Plain' in raw ? raw.Plain : raw; + const dtype = typeof plain === 'string' ? plain : JSON.stringify(plain); return { name: String(name), dtype: String(dtype).toUpperCase(), From cd69f1caf999204817bdc38f1b41f2358e7992d3 Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Fri, 2 Oct 2026 18:47:14 +0000 Subject: [PATCH 3/7] refactor(schema): rename ColumnId to FieldId --- .../src/accuracy/reconciliation.rs | 6 +- crates/asap-aware-mapping/src/replacement.rs | 14 ++--- crates/asap-aware-mapping/src/rewrite.rs | 6 +- crates/asap-aware-mapping/src/rollup.rs | 42 +++++++------- crates/frontend-promql/src/promql.rs | 4 +- .../frontend-promql/tests/promql_lowering.rs | 2 +- crates/frontend-sql/src/sql/mod.rs | 10 ++-- crates/frontend-sql/tests/sql_lowering.rs | 6 +- crates/integration-tests/tests/aggregate.rs | 2 +- crates/types/src/post_asap/expr.rs | 2 +- crates/types/src/pre_asap/agg_intent.rs | 26 ++++----- crates/types/src/pre_asap/canonicalize.rs | 8 +-- .../types/src/pre_asap/column_resolution.rs | 16 ++--- crates/types/src/pre_asap/expr_ir.rs | 4 +- crates/types/src/pre_asap/mod.rs | 8 +-- crates/types/src/pre_asap/query_expr.rs | 58 +++++++++---------- crates/types/src/pre_asap/resolve.rs | 34 +++++------ crates/types/src/pre_asap/schema.rs | 24 ++++---- crates/types/src/pre_asap/schema_resolver.rs | 10 ++-- .../decisions/concat-unique-keys.md | 10 ++-- .../analytical-resource-cost.md | 2 +- .../proposals/decoupling_op_and_expr.md | 6 +- .../design_docs/proposals/operator-sharing.md | 16 ++--- docs/develop_docs/pre-asap-ir.md | 2 +- 24 files changed, 157 insertions(+), 161 deletions(-) diff --git a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs index 6014d3ba3..76088e227 100644 --- a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs +++ b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs @@ -394,7 +394,7 @@ mod tests { use asap_types::post_asap::SketchAlgorithm; use asap_types::pre_asap::cse::share_common_sub_dags; use asap_types::pre_asap::query_expr::{GroupKeys, Source}; - use asap_types::pre_asap::schema::{ColumnId, DataType, Field, Schema}; + use asap_types::pre_asap::schema::{DataType, Field, FieldId, Schema}; /// `[ts(0), value(1), job(2)]`. /// A unique-keyed scan (`[ts]`) so `share_common_sub_dags` is actually @@ -417,7 +417,7 @@ mod tests { }) } - fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { + fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { Rc::new(QueryExpr::Aggregate { reduction: Reduction::by(by), measures: vec![intent], @@ -462,7 +462,7 @@ mod tests { fn without_quantile( q: f64, accuracy: AccuracyTarget, - excluded: Vec, + excluded: Vec, child: &Rc, ) -> Rc { Rc::new(QueryExpr::Aggregate { diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 4b766579a..0e947c691 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -366,7 +366,7 @@ use asap_types::pre_asap::query_expr::any_measure_filtered; use asap_types::pre_asap::query_expr::{ BinaryOpKind, Predicate, QueryExpr, QueryExprError, Reduction, }; -use asap_types::pre_asap::schema::ColumnId; +use asap_types::pre_asap::schema::FieldId; use asap_types::types::AccuracyTarget; use asap_types::workload::{DataWorkload, QueryRecurrence, QueryWorkload, RepeatedDemand}; use std::rc::Rc; @@ -3050,7 +3050,7 @@ fn construct_summary_agg( } // `reduction` is carried onto `SummaryAgg` verbatim — not flattened to a - // bare `Vec` — so `SummaryExecutor::find_candidates` can tell + // bare `Vec` — so `SummaryExecutor::find_candidates` can tell // a genuine empty-`by` reduction apart from a per-entity shape with no // grouping concept at all (issue #163). `construct_summary_agg` is the // single place that decides this; nothing downstream re-derives it. @@ -5112,8 +5112,8 @@ fn normalize_cross_input_equi_predicate( else { return None; }; - let is_left = |id: ColumnId| id < left_width; - let is_right = |id: ColumnId| left_width <= id && id < total_width; + let is_left = |id: FieldId| id < left_width; + let is_right = |id: FieldId| left_width <= id && id < total_width; let (left_id, right_id) = if is_left(*left_id) && is_right(*right_id) { (*left_id, *right_id) } else if is_right(*left_id) && is_left(*right_id) { @@ -7185,7 +7185,7 @@ mod tests { assert!(space.enumerate_candidate_dags(0).is_err()); } - fn equi_pred(left: ColumnId, right: ColumnId) -> Predicate { + fn equi_pred(left: FieldId, right: FieldId) -> Predicate { Predicate(Rc::new(QueryExpr::Compare { left: Rc::new(QueryExpr::Column(left)), op: asap_types::pre_asap::CompareOpKind::Eq, @@ -10108,7 +10108,7 @@ mod tests { /// `quantile_over_time(...)`) realizes to `SummaryAgg { reduction: /// PerEntity, .. }` — proving the pre-ASAP `Reduction` this crate /// already computes (issue #165) is carried onto the post-ASAP node - /// verbatim, not flattened back into an ambiguous bare `Vec`. + /// verbatim, not flattened back into an ambiguous bare `Vec`. #[test] fn bare_per_series_aggregate_realizes_summary_agg_with_per_entity_reduction() { use std::time::Duration; @@ -10132,7 +10132,7 @@ mod tests { /// Issue #163, case 2: an aggregation operator explicitly invoked with /// no grouping keys realizes to `SummaryAgg { /// reduction: Reduce(vec![]), .. }` — byte-identical `by: []` to the - /// previous test at the old `Vec` shape; `reduction` is what + /// previous test at the old `Vec` shape; `reduction` is what /// tells them apart now. #[test] fn explicit_empty_by_aggregate_realizes_summary_agg_with_reduce_reduction() { diff --git a/crates/asap-aware-mapping/src/rewrite.rs b/crates/asap-aware-mapping/src/rewrite.rs index 94a3d638c..f51839555 100644 --- a/crates/asap-aware-mapping/src/rewrite.rs +++ b/crates/asap-aware-mapping/src/rewrite.rs @@ -64,7 +64,7 @@ use asap_types::pre_asap::expr_ir::ArithmeticOpKind; use asap_types::pre_asap::query_expr::{ any_measure_filtered, BinaryOpKind, ProjectItem, QueryExpr, Reduction, }; -use asap_types::pre_asap::schema::{ColumnId, DataType}; +use asap_types::pre_asap::schema::{DataType, FieldId}; use asap_types::types::AccuracyTarget; use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG}; @@ -74,7 +74,7 @@ use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, Ta /// the module docs' "Scope" for why `without(...)`/`PerEntity` are /// excluded). Returns the grouping key count and the summed column so /// [`build_rewrite`] doesn't have to re-match. -fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { +fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { let QueryExpr::Aggregate { reduction, measures, @@ -431,7 +431,7 @@ mod tests { } } - fn avg_agg(by: Vec, col: Option, child: QueryExpr) -> QueryExpr { + fn avg_agg(by: Vec, col: Option, child: QueryExpr) -> QueryExpr { QueryExpr::Aggregate { reduction: Reduction::by(by), measures: vec![AggIntent::Avg { col }], diff --git a/crates/asap-aware-mapping/src/rollup.rs b/crates/asap-aware-mapping/src/rollup.rs index ab6eff3d9..bf3c001c0 100644 --- a/crates/asap-aware-mapping/src/rollup.rs +++ b/crates/asap-aware-mapping/src/rollup.rs @@ -64,9 +64,9 @@ //! asks for) consults it, and so does this module's `replacements`, so the //! two can never disagree about which intents are eligible. //! -//! ## `ColumnId` comparability — only sound for identical child IR +//! ## `FieldId` comparability — only sound for identical child IR //! -//! A `ColumnId` is a *position* into a specific `Schema` (`crates/types/src/pre_asap/schema.rs`'s +//! A `FieldId` is a *position* into a specific `Schema` (`crates/types/src/pre_asap/schema.rs`'s //! own doc: "the same edge, the same schema, the same positional numbering"). //! Comparing the coarser aggregate's `by` positions against the finer //! aggregate's `by` positions is only meaningful when both aggregates have @@ -74,7 +74,7 @@ //! Thus both `by` lists index the same shape. The equality fallback matters //! for scans that CSE conservatively declines to alias because they have no //! declared unique key. Structurally different sources remain out of scope: -//! this module never reconciles `ColumnId`s across distinct schemas. +//! this module never reconciles `FieldId`s across distinct schemas. //! //! ## Non-goals (tracked separately, not attempted here — same split //! `replacement.rs`'s own module docs draw for `SharedSubDAGStrategy`'s @@ -90,10 +90,10 @@ //! `asap-aware-mapping`'s scope (see issue #254's own "Non-goal" section) //! — this module only constructs the pre-ASAP [`QueryExpr::Aggregate`] //! rewrite; a `CostModel`/search engine decides whether to prefer it. -//! - **No cross-schema reconciliation** (see "`ColumnId` comparability" +//! - **No cross-schema reconciliation** (see "`FieldId` comparability" //! above) and **no `without(...)` grouping support** — `without`'s kept //! set is runtime-open (never enumerable at plan time, per -//! `GroupKeys`'s own doc), so there is no fixed `ColumnId` set to compare +//! `GroupKeys`'s own doc), so there is no fixed `FieldId` set to compare //! against a superset/subset relationship at all; [`is_legal_rollup_source`] //! declines both directions. @@ -102,7 +102,7 @@ use std::rc::Rc; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::pre_asap::query_expr::{any_measure_filtered, GroupKeys, QueryExpr, Reduction}; -use asap_types::pre_asap::schema::{ColumnId, Schema}; +use asap_types::pre_asap::schema::{FieldId, Schema}; use asap_types::types::AccuracyTarget; use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG}; @@ -153,7 +153,7 @@ fn bindable_grouped_aggregate( /// deliberately: `agg_is_mergeable` answers "does *some* partial-state merge /// exist", not "is self- or sum-recombination the right one," and this /// module only ever proposes a rewrite it can construct correctly. -fn rollup_combinator(intent: &AggIntent, finer_measure_col: ColumnId) -> Option { +fn rollup_combinator(intent: &AggIntent, finer_measure_col: FieldId) -> Option { match intent { // Self-combining: reapplying the identical operator over the finer // side's own output column is correct unchanged. @@ -199,7 +199,7 @@ fn rollup_combinator(intent: &AggIntent, finer_measure_col: ColumnId) -> Option< /// never `agg_is_mergeable` — and also excludes every `agg_is_mergeable` /// intent this module doesn't specifically handle, e.g. `Rate`). /// 3. Neither grouping is a `without(...)` exclusion grouping — `without`'s -/// kept set is runtime-open, so there is no fixed `ColumnId` set to +/// kept set is runtime-open, so there is no fixed `FieldId` set to /// compare a superset/subset relationship against (see the module docs). /// 4. `finer_output_schema` (the finer aggregate's own *output* schema, not /// the shared child's) carries a provable unique key @@ -212,7 +212,7 @@ fn rollup_combinator(intent: &AggIntent, finer_measure_col: ColumnId) -> Option< /// exactly the property re-aggregating over `finer` as if it were a /// fresh source requires. /// 5. `coarser_by` is a **strict, proper** subset of `finer_by` (same -/// `ColumnId`s, finer strictly more of them) — an *equal* `by` is +/// `FieldId`s, finer strictly more of them) — an *equal* `by` is /// `SharedSubDAGStrategy`'s CSE-sharing question, not a roll-up, so /// equality is deliberately excluded here, not treated as a degenerate /// roll-up. @@ -239,13 +239,13 @@ pub fn is_legal_rollup_source( } /// Whether `finer` is a strict, proper superset of `coarser` — every -/// `ColumnId` in `coarser` also appears in `finer`, and `finer` has more of +/// `FieldId` in `coarser` also appears in `finer`, and `finer` has more of /// them (an equal-length or shorter `finer` can never be a proper /// superset, so the length check alone rules out equality without a set /// comparison). -fn is_strict_column_superset(finer: &[ColumnId], coarser: &[ColumnId]) -> bool { - let finer_set: HashSet<&ColumnId> = finer.iter().collect(); - let coarser_set: HashSet<&ColumnId> = coarser.iter().collect(); +fn is_strict_column_superset(finer: &[FieldId], coarser: &[FieldId]) -> bool { + let finer_set: HashSet<&FieldId> = finer.iter().collect(); + let coarser_set: HashSet<&FieldId> = coarser.iter().collect(); if finer_set.len() <= coarser_set.len() { return false; } @@ -341,14 +341,14 @@ impl ReplacementStrategy for RollupStrategy { /// measure column, with `child = finer` instead of the original shared /// source. /// -/// `coarser_by`'s `ColumnId`s are positions into the *shared child's* -/// schema (the same schema `finer_by`'s `ColumnId`s index into — see the -/// module docs' "`ColumnId` comparability" section). `finer`'s own output +/// `coarser_by`'s `FieldId`s are positions into the *shared child's* +/// schema (the same schema `finer_by`'s `FieldId`s index into — see the +/// module docs' "`FieldId` comparability" section). `finer`'s own output /// schema is a *different* schema (`finer_by`'s columns, in order, followed /// by its one measure column — `aggregate_output_schema`'s `by ++ measures` /// shape), so each of `coarser_by`'s columns must be translated from its /// position in the shared child to its position in `finer`'s output: the -/// index its `ColumnId` occupies within `finer_by`'s own ordered list. +/// index its `FieldId` occupies within `finer_by`'s own ordered list. fn build_rollup( finer: &Rc, coarser_by: &GroupKeys, @@ -358,10 +358,10 @@ fn build_rollup( let (finer_by, _, _) = bindable_grouped_aggregate(finer)?; // `finer`'s own single measure sits right after its `by` columns in its // output schema (`aggregate_output_schema`'s `by ++ measures` layout). - let finer_measure_col: ColumnId = finer_by.len(); + let finer_measure_col: FieldId = finer_by.len(); let combinator = rollup_combinator(intent, finer_measure_col)?; - let remapped_by: Vec = coarser_by + let remapped_by: Vec = coarser_by .keys() .iter() .map(|id| finer_by.keys().iter().position(|f| f == id)) @@ -416,7 +416,7 @@ mod tests { } } - fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { + fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { Rc::new(QueryExpr::Aggregate { reduction: Reduction::by(by), measures: vec![intent], @@ -428,7 +428,7 @@ mod tests { } fn without_agg( - excluded: Vec, + excluded: Vec, intent: AggIntent, child: &Rc, ) -> Rc { diff --git a/crates/frontend-promql/src/promql.rs b/crates/frontend-promql/src/promql.rs index 83251053e..f0c615a7b 100644 --- a/crates/frontend-promql/src/promql.rs +++ b/crates/frontend-promql/src/promql.rs @@ -8,14 +8,14 @@ //! label matchers) and emits `UnresolvedQueryExpr` nodes with unresolved //! `ColumnRef`s — the same DAG shape //! [`resolve_root`](asap_types::pre_asap::resolve_root) later binds to -//! canonical, positional `QueryExpr`. The structural decisions a +//! canonical, positional `QueryExpr`. The structural decisions a //! separate converter stage would otherwise have to make (heavy-hitter //! `topk` recognition, the `PerEntity`/`Reduce` reduction choice, //! `without(...)` grouping) are made right here, since a front end //! building this shape already knows the answer at parse time — see //! `reduction_for` and `mark_without`. `resolve_root` is left with exactly //! the schema-*dependent* work: binding every `ColumnRef` to its -//! positional `ColumnId`. +//! positional `FieldId`. //! //! # PromQL → canonical unresolved-DAG mapping (summary) //! diff --git a/crates/frontend-promql/tests/promql_lowering.rs b/crates/frontend-promql/tests/promql_lowering.rs index 243bf6d18..25e21ebfc 100644 --- a/crates/frontend-promql/tests/promql_lowering.rs +++ b/crates/frontend-promql/tests/promql_lowering.rs @@ -605,7 +605,7 @@ fn topk_over_count_is_heavy_hitter_topk() { else { panic!("expected Aggregate with TopK, got {qe:?}"); }; - // `service` is the only group key → resolved to a positional ColumnId. + // `service` is the only group key → resolved to a positional FieldId. assert_eq!(reduction.expect_reduce().len(), 1); assert!(matches!( measures.as_slice(), diff --git a/crates/frontend-sql/src/sql/mod.rs b/crates/frontend-sql/src/sql/mod.rs index d1401988c..2fd17f3b0 100644 --- a/crates/frontend-sql/src/sql/mod.rs +++ b/crates/frontend-sql/src/sql/mod.rs @@ -6,7 +6,7 @@ //! walks the unoptimized `LogicalPlan` and emits `UnresolvedQueryExpr` nodes with //! unresolved `ColumnRef`s directly (issue #179) — the same DAG shape //! [`resolve_root`](asap_types::pre_asap::resolve_root) binds to canonical, -//! positional `QueryExpr`. Unlike PromQL's front end, SQL's +//! positional `QueryExpr`. Unlike PromQL's front end, SQL's //! Ordinary SQL `Aggregate` nodes are `Reduction::Reduce`. The explicit //! `asap_rate`/`asap_increase` bridge is the narrow exception: it //! spells a time-series range reducer with an explicit value, time-index, and @@ -1704,7 +1704,7 @@ fn conditional_count_arm(expr: &Expr) -> Option<(&Expr, &Expr)> { /// modifier rule, the "reducer argument must be a bare column" rule /// (`reducer_col`), φ extraction from a literal argument, and the ambient /// `AccuracyTarget`. `resolve_root` resolves `col` to a positional -/// `ColumnId`; the output name (DataFusion's own, e.g. +/// `FieldId`; the output name (DataFusion's own, e.g. /// `"sum(metrics.bytes)"`) is threaded separately as `Aggregate.output_names`, /// not carried here. fn lower_agg_intent(expr: &Expr) -> Result, LoweringError> { @@ -1734,7 +1734,7 @@ fn lower_agg_intent(expr: &Expr) -> Result, LoweringError> // Value reducers (`reducer_col`) require a real column — `SUM(a*b)` // is rejected, not silently reduced over a probe column. Quantile // and CountDistinct reduce a column too, so they take the same path: - // `col` is `Option` once resolved, where `None` means "the + // `col` is `Option` once resolved, where `None` means "the // PromQL sample value", which a SQL query never has. Taking an // expression here would set `col: None` and silently drop it (#115). let col = |args: &[Expr]| -> Result, LoweringError> { @@ -1890,7 +1890,7 @@ fn scalar_positive_u64(value: &DfScalarValue) -> Option { /// core variant (issue #232). Core treats `Extension` opaquely: both columns /// are kept only as validated bare-column names in `payload` (`reducer_col`'s /// same "no expression arguments" rule, issue #115) — they are **not** run -/// through `resolve_agg_intent`'s positional `ColumnRef` -> `ColumnId` +/// through `resolve_agg_intent`'s positional `ColumnRef` -> `FieldId` /// binding the way a real reducer's `col` is, since `Extension` carries no /// typed column field for core to resolve. Shared `arg_selector_columns` validates /// and resolves those names during aggregate schema derivation, preserving the @@ -2104,7 +2104,7 @@ fn expand_grouping_set(gs: &logical_expr::GroupingSet) -> Vec> { /// Derived columns materialized in a `Project` beneath an `Aggregate` (#110). /// -/// `Aggregate.by` holds positional `ColumnId`s and each reducer holds one input +/// `Aggregate.by` holds positional `FieldId`s and each reducer holds one input /// column, so neither can hold an expression. `GROUP BY date_trunc('minute', t)` /// and `SUM(bytes * 8)` are therefore rewritten to group/reduce over a projected /// column that carries the expression's value. diff --git a/crates/frontend-sql/tests/sql_lowering.rs b/crates/frontend-sql/tests/sql_lowering.rs index 883966db9..2f0e6727e 100644 --- a/crates/frontend-sql/tests/sql_lowering.rs +++ b/crates/frontend-sql/tests/sql_lowering.rs @@ -424,7 +424,7 @@ async fn count_distinct_is_cardinality() { async fn select_distinct_lowers_to_distinct_with_positional_cols() { // SELECT DISTINCT → a `Dedup` node whose `cols` are positional ColumnIds // (not name-based ColumnRefs). DataFusion's `Distinct::All` dedups on every - // column, so `cols` is empty here — but the field type is now `Vec`. + // column, so `cols` is empty here — but the field type is now `Vec`. let qe = lower("SELECT DISTINCT service FROM metrics").await; let QueryExpr::Dedup { cols, .. } = &qe else { panic!("expected a Dedup at the root, got {qe:?}"); @@ -453,7 +453,7 @@ async fn inner_join_lowers_to_join_over_two_scans() { assert!(matches!(right.as_ref(), QueryExpr::Scan { .. })); } -/// The two `ColumnId`s an equijoin predicate `Column(l) = Column(r)` binds to, +/// The two `FieldId`s an equijoin predicate `Column(l) = Column(r)` binds to, /// returned sorted so the assertion is independent of left/right ordering. fn join_eq_columns(join: &QueryExpr) -> [usize; 2] { let QueryExpr::Join { pred, .. } = join else { @@ -1955,7 +1955,7 @@ async fn arg_min_lowers_to_its_own_extension_kind() { async fn arg_max_payload_preserves_both_column_names() { // Core never resolves an `Extension`'s payload, so both columns are kept // as validated bare-column `ColumnRef`s in `payload`, not run through - // positional `ColumnId` binding -- see `lower_arg_selector`'s doc. + // positional `FieldId` binding -- see `lower_arg_selector`'s doc. let qe = lower_clickhouse("SELECT argMax(service, latency) AS m FROM metrics").await; let (_, measures) = find_aggregate(&qe).expect("expected an Aggregate"); let AggIntent::Extension { payload, .. } = &measures[0] else { diff --git a/crates/integration-tests/tests/aggregate.rs b/crates/integration-tests/tests/aggregate.rs index 051eeb6b7..2d93d5774 100644 --- a/crates/integration-tests/tests/aggregate.rs +++ b/crates/integration-tests/tests/aggregate.rs @@ -4,7 +4,7 @@ //! //! Cross-series aggregates lower to a single `Aggregate` node with no //! `TimeRange` child (range functions use `TimeRange` — see `time_range.rs`). -//! Group keys land on `Aggregate.by` as positional `ColumnId`s. +//! Group keys land on `Aggregate.by` as positional `FieldId`s. //! Single-stat PromQL aggregates always get `output_names: [""]` (no alias) //! and `having: None`. diff --git a/crates/types/src/post_asap/expr.rs b/crates/types/src/post_asap/expr.rs index 9312d95c6..bac6a5079 100644 --- a/crates/types/src/post_asap/expr.rs +++ b/crates/types/src/post_asap/expr.rs @@ -182,7 +182,7 @@ pub enum SummaryExpr { /// How this aggregation's output rows relate to `child`'s — the /// same [`Reduction`] the pre-ASAP `Aggregate` node it was bound /// from carried (issue #165), reused verbatim rather than - /// flattened to a bare `Vec`. `Reduction::Reduce(by)` + /// flattened to a bare `Vec`. `Reduction::Reduce(by)` /// with an empty `by` is a genuine full reduction (merge every /// candidate into one group); `Reduction::PerEntity` has no /// grouping concept at all (never merge across entities) — the diff --git a/crates/types/src/pre_asap/agg_intent.rs b/crates/types/src/pre_asap/agg_intent.rs index c60dd55e3..e1f5b4e35 100644 --- a/crates/types/src/pre_asap/agg_intent.rs +++ b/crates/types/src/pre_asap/agg_intent.rs @@ -16,19 +16,19 @@ use serde::{Deserialize, Serialize}; use crate::pre_asap::query_expr::DataModel; -use crate::pre_asap::schema::{ColumnId, DataType, Field, FieldDataType}; +use crate::pre_asap::schema::{DataType, Field, FieldDataType, FieldId}; use crate::types::AccuracyTarget; /// "What to compute" — the vocabulary the planner pivots on. /// /// Grouping for `TopK` rides on the enclosing `QueryExpr::Aggregate.by` -/// (positional `ColumnId`s), like every other aggregate; the intent itself +/// (positional `FieldId`s), like every other aggregate; the intent itself /// carries only `k` + the accuracy target. /// /// The single-column reducers (`Sum` / `Min` / `Max` / `Avg` / `StdDev` / /// `Variance` / `Quantile`) carry `col: Option` — the input /// column they reduce, generic over the column-reference state the same way -/// [`QueryExpr`](super::query_expr::QueryExpr) is: positional `ColumnId` once +/// [`QueryExpr`](super::query_expr::QueryExpr) is: positional `FieldId` once /// bound (the default, and every existing use of the bare `AggIntent` name), /// or an unresolved name-based `ColumnRef` for a front end constructing this /// intent directly, before the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has run. @@ -43,7 +43,7 @@ use crate::types::AccuracyTarget; #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(tag = "kind", rename_all = "snake_case", deny_unknown_fields)] #[serde(bound(serialize = "C: Serialize", deserialize = "C: Deserialize<'de>"))] -pub enum AggIntent { +pub enum AggIntent { // ── Data-model-agnostic ────────────────────────────────────────────── Count { accuracy: AccuracyTarget, @@ -386,7 +386,7 @@ pub enum MathFunc { // the `PerEntity`/`Reduce` reduction shape right at construction time (see // `asap_frontend_promql::promql::reduction_for`) — so they stay generic // alongside `input_col`, in one `impl` block. -impl AggIntent { +impl AggIntent { /// Resolve the existing SQL arg-selector extension using its child schema. /// The tuple is (selected value column, ordering column). /// Unknown extensions remain owned by their deployment model. Recognized @@ -394,7 +394,7 @@ impl AggIntent { pub fn arg_selector_columns( &self, schema: &super::schema::Schema, - ) -> Result, String> { + ) -> Result, String> { let Self::Extension { ext_kind, payload } = self else { return Ok(None); }; @@ -407,7 +407,7 @@ impl AggIntent { if fields.len() != 2 { return Err("arg selector requires arg_col and val_col only".into()); } - let resolve = |field: &str| -> Result { + let resolve = |field: &str| -> Result { let reference: super::expr_ir::ColumnRef = serde_json::from_value( fields .get(field) @@ -782,7 +782,7 @@ mod tests { fn output_column_names_are_intent_keyed() { let v = c("value", DataType::Float64); assert_eq!( - AggIntent::::Count { + AggIntent::::Count { accuracy: AccuracyTarget::Exact } .output_column(&v) @@ -790,13 +790,13 @@ mod tests { "count" ); assert_eq!( - AggIntent::::Sum { col: None } + AggIntent::::Sum { col: None } .output_column(&v) .name, "sum" ); assert_eq!( - AggIntent::::Quantile { + AggIntent::::Quantile { col: None, q: 0.99, accuracy: AccuracyTarget::Epsilon(0.01) @@ -810,7 +810,7 @@ mod tests { #[test] fn sum_preserves_input_dtype() { assert!(matches!( - AggIntent::::Sum { col: None } + AggIntent::::Sum { col: None } .output_column(&c("c", DataType::Int64)) .dtype, FieldDataType::Plain(DataType::Int64) @@ -871,12 +871,12 @@ mod tests { fn input_cols_tracks_only_reducers() { assert_eq!(AggIntent::Sum { col: Some(3) }.input_cols(), vec![3]); assert!( - AggIntent::::Avg { col: None } + AggIntent::::Avg { col: None } .input_cols() .is_empty(), "empty = PromQL sample value" ); - assert!(AggIntent::::Count { + assert!(AggIntent::::Count { accuracy: AccuracyTarget::Exact } .input_cols() diff --git a/crates/types/src/pre_asap/canonicalize.rs b/crates/types/src/pre_asap/canonicalize.rs index b9e6a653a..ddab95df7 100644 --- a/crates/types/src/pre_asap/canonicalize.rs +++ b/crates/types/src/pre_asap/canonicalize.rs @@ -38,13 +38,13 @@ pub fn canonicalize(mut expr: QueryExpr) -> QueryExpr { fn canon(expr: &mut QueryExpr) { // A `Concat` asserting a caller-proven `discriminator_unique_key` (issue - // #228) had that key's `ColumnId`s resolved, in `resolve.rs`, against + // #228) had that key's `FieldId`s resolved, in `resolve.rs`, against // exactly the first branch's output schema *as it stood before this // pass ran*. `try_promote_additive_top_ranking`/`try_rewrite_rownumber_topk` // below can restructure that branch (anywhere within it — not only at // its own top level, since the same recursive walk can rewrite a node // nested under a pass-through wrapper too) into a shape with a - // different output schema, which would leave those `ColumnId`s + // different output schema, which would leave those `FieldId`s // pointing at the wrong column, or out of bounds, of the // post-canonicalize schema. Snapshot the schema the discriminator key // was actually resolved against, right here, before recursing into the @@ -475,10 +475,10 @@ mod tests { // ── Concat's discriminator_unique_key vs. canonicalize (issue #228 review) ── // - // `resolve.rs` resolves `discriminator_unique_key`'s `ColumnId`s against + // `resolve.rs` resolves `discriminator_unique_key`'s `FieldId`s against // the first branch's *pre-canonicalize* output schema. If canonicalize // then restructures that branch (heavy-hitter promotion, the - // `ROW_NUMBER()` top-k rewrite), those `ColumnId`s can end up pointing at + // `ROW_NUMBER()` top-k rewrite), those `FieldId`s can end up pointing at // the wrong column — or out of bounds — of the new schema. The two tests // below pin the fix: the key is dropped whenever the branch's schema // actually changed, and survives untouched otherwise. Never guessed at. diff --git a/crates/types/src/pre_asap/column_resolution.rs b/crates/types/src/pre_asap/column_resolution.rs index cfafbe9e0..0cab6362d 100644 --- a/crates/types/src/pre_asap/column_resolution.rs +++ b/crates/types/src/pre_asap/column_resolution.rs @@ -1,7 +1,7 @@ //! Schema-driven column resolution. //! //! Front ends (issue #179) emit `ColumnRef` (name-based, optionally -//! table-qualified); the canonical DAG uses positional [`ColumnId`] resolved +//! table-qualified); the canonical tree uses positional [`FieldId`] resolved //! against a per-node [`Schema`]. These helpers bridge the two — the //! [`SchemaResolver`](super::schema_resolver) builds the schema, and [`resolve_column_refs`] //! turns name-based refs (group keys, dedup columns) into positional ids, @@ -17,7 +17,7 @@ use super::query_expr::{ aggregate_output_schema, GroupKeys, QueryExpr, QueryExprError, Reduction, ResolvedQueryExpr, UnresolvedQueryExpr, }; -use super::schema::{ColumnId, DataType, FieldDataType, Schema}; +use super::schema::{DataType, FieldDataType, FieldId, Schema}; /// Errors returned by the resolution helpers. #[derive(Debug, Error, PartialEq, Eq)] @@ -29,12 +29,12 @@ pub enum ResolveError { }, #[error("ColumnRef::SampleValue has no `value` column in schema (have: {available:?})")] NoSampleValue { available: Vec }, - #[error("ColumnRef::Wildcard cannot be resolved to a single ColumnId")] + #[error("ColumnRef::Wildcard cannot be resolved to a single FieldId")] WildcardNotPositional, } -/// Resolve a single [`ColumnRef`] to a positional [`ColumnId`]. -pub fn resolve_column_ref(col: &ColumnRef, schema: &Schema) -> Result { +/// Resolve a single [`ColumnRef`] to a positional [`FieldId`]. +pub fn resolve_column_ref(col: &ColumnRef, schema: &Schema) -> Result { match col { ColumnRef::Named(name) => schema .column_id(name) @@ -62,7 +62,7 @@ pub fn resolve_column_ref(col: &ColumnRef, schema: &Schema) -> Result = (0..schema.fields.len()) + let numeric: Vec = (0..schema.fields.len()) .filter(|&i| Some(i) != schema.time_index) .filter(|&i| { matches!( @@ -84,7 +84,7 @@ pub fn resolve_column_ref(col: &ColumnRef, schema: &Schema) -> Result Result, ResolveError> { +) -> Result, ResolveError> { cols.iter().map(|c| resolve_column_ref(c, schema)).collect() } @@ -106,7 +106,7 @@ pub fn resolve_column_refs( pub fn resolve_group_keys_promql( cols: &[ColumnRef], schema: &Schema, -) -> Result, ResolveError> { +) -> Result, ResolveError> { cols.iter() .filter_map(|c| match resolve_column_ref(c, schema) { Err(ResolveError::NotFound { .. }) if schema.closed => None, diff --git a/crates/types/src/pre_asap/expr_ir.rs b/crates/types/src/pre_asap/expr_ir.rs index 88c309903..27809c8fe 100644 --- a/crates/types/src/pre_asap/expr_ir.rs +++ b/crates/types/src/pre_asap/expr_ir.rs @@ -8,7 +8,7 @@ //! DAG, not two type families joined by wrappers — generic over the same //! column-reference state `C` the rest of `QueryExpr` already carries //! (issue #179): [`ColumnRef`] (name-based, front-end-emitted) or -//! [`ColumnId`](super::schema::ColumnId) (positional, once bound). +//! [`FieldId`](super::schema::FieldId) (positional, once bound). //! //! What's left here is the vocabulary those scalar variants are built from — //! [`ScalarValue`], [`CompareOpKind`], [`ArithmeticOpKind`] — the **union** of what the two @@ -21,7 +21,7 @@ use serde::{Deserialize, Serialize}; /// A name-based column reference — the front-end-emitted, unresolved state of /// [`QueryExpr::Column`](super::query_expr::QueryExpr::Column) (`C = /// ColumnRef`); the [`SchemaResolver`](super::schema_resolver::SchemaResolver) resolves it to a -/// positional [`ColumnId`](super::schema::ColumnId). Includes the two +/// positional [`FieldId`](super::schema::FieldId). Includes the two /// PromQL-conventional synthetic columns. #[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] pub enum ColumnRef { diff --git a/crates/types/src/pre_asap/mod.rs b/crates/types/src/pre_asap/mod.rs index 4cfcfbd96..8e06fbaba 100644 --- a/crates/types/src/pre_asap/mod.rs +++ b/crates/types/src/pre_asap/mod.rs @@ -3,7 +3,7 @@ //! - [`query_expr`] — the canonical, language- and deployment-independent //! intent algebra: one recursive [`QueryExpr`] DAG (relational operators //! *and* scalar expression shapes both, since issue #205) + [`AggIntent`], -//! generic over the column-reference state (positional [`ColumnId`] once +//! generic over the column-reference state (positional [`FieldId`] once //! bound, name-based [`ColumnRef`] before). //! - [`agg_intent`] — the aggregation-intent vocabulary. //! - [`expr_ir`] — the [`ColumnRef`] column-reference type and the scalar @@ -11,8 +11,8 @@ //! [`QueryExpr`]'s scalar variants are built from. //! - [`schema`] — the per-edge [`Schema`] every node carries. //! - [`schema_resolver`] / [`column_resolution`] — name resolution: turn a `ColumnRef` -//! into a positional `ColumnId` against an in-scope [`Schema`]. -//! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedQueryExpr`] DAG to +//! into a positional `FieldId` against an in-scope [`Schema`]. +//! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedQueryExpr`] tree to //! canonical [`ResolvedQueryExpr`] (issue #179): both front ends //! (`asap-frontend-promql`, `asap-frontend-sql`) construct `UnresolvedQueryExpr` //! directly during their own `interpret` step and call @@ -60,5 +60,5 @@ pub use query_expr::{ WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, }; pub use resolve::{resolve_root, ResolveDAGError}; -pub use schema::{ColumnId, DataType, Field, FieldDataType, Schema}; +pub use schema::{DataType, Field, FieldDataType, FieldId, Schema}; pub use schema_resolver::{SchemaCatalog, SchemaResolver, UsageDerivedCatalog}; diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs index 1f6021ba4..ba88103f0 100644 --- a/crates/types/src/pre_asap/query_expr.rs +++ b/crates/types/src/pre_asap/query_expr.rs @@ -21,10 +21,10 @@ use thiserror::Error; use super::agg_intent::AggIntent; use super::expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; -use super::schema::{ColumnId, DataType, Field, FieldDataType, Schema}; +use super::schema::{DataType, Field, FieldDataType, FieldId, Schema}; -/// The column-reference resolution state a [`QueryExpr`] DAG carries — -/// [`ColumnId`] (the default, and what the bare `QueryExpr` name has always +/// The column-reference resolution state a [`QueryExpr`] tree carries — +/// [`FieldId`] (the default, and what the bare `QueryExpr` name has always /// meant) once the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has resolved every /// reference positionally, or the front-end-emitted, name-based [`ColumnRef`] /// before binding. The only place the two states differ in *shape* rather @@ -41,7 +41,7 @@ pub trait ColState: type ScanSchema: Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de>; } -impl ColState for ColumnId { +impl ColState for FieldId { type ScanSchema = Schema; } @@ -55,7 +55,7 @@ pub enum QueryExprError { #[error("invalid scalar function signature: {0}")] InvalidScalarSignature(String), #[error("by-column id {0} out of range (input has {1} columns)")] - InvalidGroupByColumn(ColumnId, usize), + InvalidGroupByColumn(FieldId, usize), #[error("Concat requires at least one child")] EmptyConcat, /// [`QueryExpr::output_schema`] called on (or reached, while recursing, a @@ -92,10 +92,10 @@ pub enum QueryExprError { /// `PromqlSeriesSample` groupings are always `by`. /// /// Serialises as a bare array for the (overwhelmingly common) `by` case — -/// wire-compatible with the `Vec` this field held before — and as +/// wire-compatible with the `Vec` this field held before — and as /// `{"without": [...]}` for the exclusion case. #[derive(Debug, Clone, PartialEq, Eq, Hash)] -pub struct GroupKeys { +pub struct GroupKeys { keys: Vec, without: bool, } @@ -430,7 +430,7 @@ pub enum SampleKind { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct SortKey { +pub struct SortKey { pub expr: QueryExpr, pub ascending: bool, pub nulls_first: bool, @@ -505,7 +505,7 @@ pub enum GroupSide { /// part of it — the box is what makes the recursive type's size finite there. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct Predicate(pub Rc>); +pub struct Predicate(pub Rc>); /// Whether any entry of an `Aggregate.filters` vector is set — the shape /// no binding rule accepts yet (issue #466): a filtered measure stays @@ -517,7 +517,7 @@ pub fn any_measure_filtered(filters: &[Option>]) -> bo /// One item in a SELECT projection list. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct ProjectItem { +pub struct ProjectItem { pub alias: Option, pub expr: QueryExpr, } @@ -531,7 +531,7 @@ pub struct ProjectItem { /// whether a grouping-key list happens to be empty or from a neighboring /// node's shape. See design proposal #165. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub enum Reduction { +pub enum Reduction { /// Collapses input rows via `by` — `by`/`without` semantics are exactly /// [`GroupKeys`]'s. May still collapse every row into one (an empty, /// non-`without` `by`) — that's a genuine reduction with zero grouping @@ -623,7 +623,7 @@ impl Reduction { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(deny_unknown_fields)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct ConcatDiscriminatorKey { +pub struct ConcatDiscriminatorKey { discriminator: C, inner_key: Vec, } @@ -650,9 +650,9 @@ impl ConcatDiscriminatorKey { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub enum QueryExpr { +pub enum QueryExpr { /// Outermost leaf. `schema` is the **binding schema** — the resolved column - /// set every positional `ColumnId` in the DAG indexes into, *not* a full + /// set every positional `FieldId` in the tree indexes into, *not* a full /// description of the runtime row — once bound (`schema: Schema`, always /// present: the [`SchemaResolver`](super::schema_resolver) is total). Before binding, a /// front-end-emitted `Scan` (`C = ColumnRef`) knows it only when the front @@ -959,7 +959,7 @@ pub enum QueryExpr { // has to get right structurally anyway (a `Filter` is never built with an // operator sub-DAG as its `pred`). /// A column reference — unresolved [`ColumnRef`] (front-end-emitted, `C = - /// ColumnRef`) or positional [`ColumnId`] (once bound, `C = ColumnId`). + /// ColumnRef`) or positional [`FieldId`] (once bound, `C = FieldId`). Column(C), /// A constant literal value. Literal(ScalarValue), @@ -1146,12 +1146,12 @@ impl QueryExpr { } } -/// The canonical, positional, resolved DAG — what the bare `QueryExpr` name -/// has always meant (the default `C = ColumnId`). Every existing consumer +/// The canonical, positional, resolved tree — what the bare `QueryExpr` name +/// has always meant (the default `C = FieldId`). Every existing consumer /// keeps using `QueryExpr` unparameterized; this alias exists only to name /// the resolved state explicitly at a use site that also wants to name /// [`UnresolvedQueryExpr`] nearby. -pub type ResolvedQueryExpr = QueryExpr; +pub type ResolvedQueryExpr = QueryExpr; /// The front-end-emitted, name-based, unresolved DAG — /// `QueryExpr`: front ends construct this directly during their @@ -1164,7 +1164,7 @@ pub type UnresolvedQueryExpr = QueryExpr; // only on the resolved instantiation, not `impl QueryExpr`. // Same reasoning as `AggIntent`'s `output_column`/`requires`/`is_per_series` // (#205): a schema-shaped property that is only meaningful post-binding. -impl QueryExpr { +impl QueryExpr { /// Infer a scalar expression against its input relation using the same /// canonical rules as projection schema derivation. pub fn scalar_type(&self, input: &Schema) -> Result<(DataType, bool), QueryExprError> { @@ -1668,7 +1668,7 @@ pub fn aggregate_output_schema( /// so the schema can't freeze to closed and claims no unique key (issue #39). fn without_output_schema( in_schema: &Schema, - excluded: &[ColumnId], + excluded: &[FieldId], measures: &[AggIntent], output_names: &[String], ) -> Result { @@ -1734,7 +1734,7 @@ fn without_output_schema( /// must be one of the scalar variants (issue #205) — an operator variant here /// is a construction bug, not a shape this needs to handle silently. fn infer_expr_type( - expr: &QueryExpr, + expr: &QueryExpr, schema: &Schema, ) -> Result<(DataType, bool), QueryExprError> { Ok(match expr { @@ -1852,7 +1852,7 @@ fn infer_expr_type( /// Default output-column name for a projection item with no explicit alias: /// a bare column keeps its (schema) name; anything else gets `col_{i}`. -fn default_proj_name(expr: &QueryExpr, idx: usize, schema: &Schema) -> String { +fn default_proj_name(expr: &QueryExpr, idx: usize, schema: &Schema) -> String { match expr { QueryExpr::Column(id) => schema .fields @@ -1922,11 +1922,7 @@ mod tests { ); } - fn scan( - columns: Vec, - time_index: Option, - uk: Vec>, - ) -> QueryExpr { + fn scan(columns: Vec, time_index: Option, uk: Vec>) -> QueryExpr { QueryExpr::Scan { source: Source::Table { table_ref: "t".into(), @@ -2638,7 +2634,7 @@ mod tests { /// different DAG position. `as_promql_scalar` is the round-trip inverse. #[test] fn promql_scalar_bridges_a_literal_float_at_an_operator_position() { - let bridge = QueryExpr::::promql_scalar(2.5); + let bridge = QueryExpr::::promql_scalar(2.5); assert_eq!( bridge, QueryExpr::PromqlScalarBridge(Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.5)))) @@ -2649,7 +2645,7 @@ mod tests { // its native (unwrapped, no row schema) scalar-sub-language position — // no longer a different variant, just not bridged to this DAG // position. - let sql_literal = QueryExpr::::Literal(ScalarValue::Float64(2.5)); + let sql_literal = QueryExpr::::Literal(ScalarValue::Float64(2.5)); assert_eq!(bridge.as_promql_scalar(), Some(2.5)); assert_ne!( bridge, sql_literal, @@ -2670,7 +2666,7 @@ mod tests { /// duplicate variants was used, only by whether the wrapper is present. #[test] fn row_schema_rides_on_the_bridge_wrapper_not_the_literal_variant() { - let bridged = QueryExpr::::promql_scalar(42.0); + let bridged = QueryExpr::::promql_scalar(42.0); let schema = bridged.output_schema().expect("bridge has a row schema"); assert_eq!(schema.fields.len(), 1); assert_eq!(schema.fields[0].name, "value"); @@ -2681,7 +2677,7 @@ mod tests { // `Compare`/`Arithmetic` operand would occupy) has no row schema of // its own — it's a construction bug to call `output_schema` on it // directly, caught as `ScalarHasNoRowSchema` rather than panicking. - let bare = QueryExpr::::Literal(ScalarValue::Float64(42.0)); + let bare = QueryExpr::::Literal(ScalarValue::Float64(42.0)); assert!(matches!( bare.output_schema(), Err(QueryExprError::ScalarHasNoRowSchema) diff --git a/crates/types/src/pre_asap/resolve.rs b/crates/types/src/pre_asap/resolve.rs index b4a5c87a6..3297767a1 100644 --- a/crates/types/src/pre_asap/resolve.rs +++ b/crates/types/src/pre_asap/resolve.rs @@ -1,5 +1,5 @@ //! Resolve a front-end-emitted, unresolved [`UnresolvedQueryExpr`] (`QueryExpr`) -//! into the canonical, positional [`ResolvedQueryExpr`] (`QueryExpr`). +//! into the canonical, positional [`ResolvedQueryExpr`] (`QueryExpr`). //! //! Both front ends (`asap-frontend-promql`, `asap-frontend-sql`) construct //! canonical `QueryExpr` shapes directly during their own `interpret` step @@ -10,9 +10,9 @@ //! "mechanical, schema-dependent substitution" #179 describes: a single //! generic, shape-preserving walk — every [`UnresolvedQueryExpr`] variant maps to the //! identical [`ResolvedQueryExpr`] variant — that resolves every [`ColumnRef`] to -//! the [`SchemaResolver`](super::schema_resolver::SchemaResolver)-computed positional [`ColumnId`]. +//! the [`SchemaResolver`](super::schema_resolver::SchemaResolver)-computed positional [`FieldId`]. //! -//! ## Why positional `ColumnId`, not just carrying names all the way through (issue #216) +//! ## Why positional `FieldId`, not just carrying names all the way through (issue #216) //! //! A mature query engine can legitimately choose either design — DataFusion's //! own logical plan (what `asap-frontend-sql` walks to build its `QueryExpr`) @@ -28,7 +28,7 @@ //! (`crates/frontend-sql/tests/sql_lowering.rs`) exists specifically because //! `metrics.service` and `hosts.service` are both just `"service"` once their //! schemas are concatenated. A bare name is ambiguous the moment two sources -//! share one; `ColumnId` is what makes "the second `service`, position 4, not +//! share one; `FieldId` is what makes "the second `service`, position 4, not //! the first" a fact recorded once, instead of a lookup redone at every use site. //! 2. **A name's meaning changes going up the DAG.** `Project` renames/aliases, //! `Aggregate` collapses columns and introduces synthetic ones, `Join` @@ -38,7 +38,7 @@ //! pins each reference to "this exact column of this exact node's //! already-derived output schema," so nothing downstream re-derives that scope. //! 3. **It concentrates scoping logic in one place instead of ~6.** Every -//! downstream pass just compares/indexes `ColumnId`s — O(1), unambiguous. If +//! downstream pass just compares/indexes `FieldId`s — O(1), unambiguous. If //! they worked on names instead, each would need its own qualifier-aware, //! join-collision-aware name resolver, or risk silently binding to the wrong //! `"service"`. @@ -61,7 +61,7 @@ use super::query_expr::{ aggregate_output_schema, any_measure_filtered, ConcatDiscriminatorKey, GroupKeys, Predicate, ProjectItem, QueryExprError, Reduction, ResolvedQueryExpr, SortKey, UnresolvedQueryExpr, }; -use super::schema::{ColumnId, Schema}; +use super::schema::{FieldId, Schema}; use super::schema_resolver::SchemaResolver; /// Errors from resolving a canonical, unresolved [`UnresolvedQueryExpr`] DAG. @@ -76,8 +76,8 @@ pub enum ResolveDAGError { Schema(#[from] QueryExprError), } -/// Resolve a whole [`UnresolvedQueryExpr`] DAG rooted at `dag` into canonical -/// [`ResolvedQueryExpr`]: binds every `ColumnRef` to a `ColumnId` via the +/// Resolve a whole [`UnresolvedQueryExpr`] tree rooted at `tree` into canonical +/// [`ResolvedQueryExpr`]: binds every `ColumnRef` to a `FieldId` via the /// [`SchemaResolver`], then [`canonicalize`](super::canonicalize::canonicalize)s the /// result. pub fn resolve_root(dag: &UnresolvedQueryExpr) -> Result { @@ -474,11 +474,11 @@ fn inherited_names(schema: &Schema) -> Vec { } /// Resolve a name-based [`GroupKeys`] into positional -/// [`GroupKeys`], preserving its `by`/`without` mode. +/// [`GroupKeys`], preserving its `by`/`without` mode. fn resolve_group_keys( keys: &GroupKeys, schema: &Schema, -) -> Result, ResolveError> { +) -> Result, ResolveError> { let ids = resolve_column_refs(keys.keys(), schema)?; Ok(if keys.is_without() { GroupKeys::without(ids) @@ -488,7 +488,7 @@ fn resolve_group_keys( } /// Resolve a name-based [`Reduction`] into positional -/// [`Reduction`]. +/// [`Reduction`]. /// /// Uses [`resolve_group_keys_promql`] rather than the strict /// [`resolve_group_keys`], unlike every other group-key site in `resolve` @@ -506,7 +506,7 @@ fn resolve_group_keys( fn resolve_reduction( reduction: &Reduction, schema: &Schema, -) -> Result, ResolveError> { +) -> Result, ResolveError> { Ok(match reduction { Reduction::Reduce(by) => { let ids = resolve_group_keys_promql(by.keys(), schema)?; @@ -521,14 +521,14 @@ fn resolve_reduction( } /// Resolve a name-based [`AggIntent`] into positional -/// [`AggIntent`] — every `col: Option` resolves to -/// `Option` (`None` stays `None`, the sample-value convention); +/// [`AggIntent`] — every `col: Option` resolves to +/// `Option` (`None` stays `None`, the sample-value convention); /// every other field carries straight through unchanged. fn resolve_agg_intent( intent: &AggIntent, schema: &Schema, -) -> Result, ResolveError> { - let col = |c: &Option| -> Result, ResolveError> { +) -> Result, ResolveError> { + let col = |c: &Option| -> Result, ResolveError> { c.as_ref() .map(|r| resolve_column_ref(r, schema)) .transpose() @@ -821,7 +821,7 @@ mod tests { /// SchemaResolver's fallback schema wouldn't contain `phi` at all, and this /// `resolve_column_ref` call would fail `NotFound` for a column the /// caller correctly named. It must resolve cleanly, and the resolved - /// `ConcatDiscriminatorKey` must carry the *positional* `ColumnId`s of + /// `ConcatDiscriminatorKey` must carry the *positional* `FieldId`s of /// the branch's own (usage-derived) schema. #[test] fn resolve_root_seeds_and_resolves_an_otherwise_unreferenced_discriminator_column() { diff --git a/crates/types/src/pre_asap/schema.rs b/crates/types/src/pre_asap/schema.rs index fd0a69f6c..55220dcb4 100644 --- a/crates/types/src/pre_asap/schema.rs +++ b/crates/types/src/pre_asap/schema.rs @@ -25,7 +25,7 @@ use crate::post_asap::sketch::{ /// /// Kept as a named type so downstream code can pattern on the intent ("this /// is a column position, not just any number"). -pub type ColumnId = usize; +pub type FieldId = usize; /// One field of a [`Schema`]: `name + dtype + nullable`, plus an optional /// table qualifier. The struct describes a column and holds none of its data. @@ -212,7 +212,7 @@ pub enum DataType { /// Per-edge schema. Flowing between any two operators, on every node's /// input and output. /// -/// `unique_keys` is metadata for reuse-aware planning: each inner `Vec` +/// `unique_keys` is metadata for reuse-aware planning: each inner `Vec` /// is a set of column indices that together uniquely identify rows. The /// outer `Vec` allows multiple unique-key sets (primary key + another /// unique constraint). Populated by per-node input/output spec — @@ -226,12 +226,12 @@ pub struct Schema { /// Index into `fields` for the time axis, if any. PromQL leaves /// always carry one; SQL leaves may or may not. #[serde(default)] - pub time_index: Option, + pub time_index: Option, /// Unique-key sets — each inner vec is a tuple of column indices /// that together uniquely identifies a row. Empty `Vec` means /// "no provable unique constraint" (the conservative default). #[serde(default)] - pub unique_keys: Vec>, + pub unique_keys: Vec>, /// Whether this schema **completely enumerates** the columns at this point. /// /// - `true` (**closed**): there are no columns beyond these — a catalog-backed @@ -265,9 +265,9 @@ struct SchemaWire { fields: Option>, columns: Option>, #[serde(default)] - time_index: Option, + time_index: Option, #[serde(default)] - unique_keys: Vec>, + unique_keys: Vec>, closed: Option, } @@ -426,8 +426,8 @@ impl Schema { /// inferred unique keys (e.g. PromQL leaves: `[time_index, label_set]`). pub fn with_time_index( fields: Vec, - time_index: ColumnId, - unique_keys: Vec>, + time_index: FieldId, + unique_keys: Vec>, ) -> Self { Self { fields, @@ -440,7 +440,7 @@ impl Schema { /// The schema of a summary-planning node: `fields` and a time axis, no /// unique-key claim, closed. The shape every post-ASAP operator output /// carried before pre- and post-ASAP schemas were one type. - pub fn lifted(fields: Vec, time_index: Option) -> Self { + pub fn lifted(fields: Vec, time_index: Option) -> Self { Self { fields, time_index, @@ -455,14 +455,14 @@ impl Schema { } /// Look up a field by name (first match). `None` if not present. - pub fn column_id(&self, name: &str) -> Option { + pub fn column_id(&self, name: &str) -> Option { self.fields.iter().position(|c| c.name == name) } /// Look up a field by `(table, name)` qualifier — disambiguates columns /// that share a `name` across a join (`a.k` vs `b.k`). `None` if no field /// has both that qualifier and name. - pub fn column_id_qualified(&self, table: &str, name: &str) -> Option { + pub fn column_id_qualified(&self, table: &str, name: &str) -> Option { self.fields .iter() .position(|c| c.name == name && c.table.as_deref() == Some(table)) @@ -478,7 +478,7 @@ impl Schema { /// Append `cols` as an additional unique-key set if not already present. /// Used by `Dedup { cols }`: "the input schema with `unique_keys` /// tightened to include `cols`". - pub fn add_unique_key(&mut self, cols: Vec) { + pub fn add_unique_key(&mut self, cols: Vec) { if !self.unique_keys.contains(&cols) { self.unique_keys.push(cols); } diff --git a/crates/types/src/pre_asap/schema_resolver.rs b/crates/types/src/pre_asap/schema_resolver.rs index b63b70afd..6e84f954d 100644 --- a/crates/types/src/pre_asap/schema_resolver.rs +++ b/crates/types/src/pre_asap/schema_resolver.rs @@ -1,7 +1,7 @@ //! The **SchemaResolver** — name resolution as an explicit pass. //! //! [`SchemaResolver::resolve_schema`] produces the complete, self-contained [`Schema`] every -//! `ColumnId` in the canonical DAG indexes into. [`resolve`](super::resolve) +//! `FieldId` in the canonical tree indexes into. [`resolve`](super::resolve) //! then becomes purely structural: it threads the SchemaResolver's schema and //! positional resolution downstream is **total**. //! @@ -69,10 +69,10 @@ impl SchemaResolver { /// Resolve the complete [`Schema`] in scope for a query rooted at `dag`. /// /// Contains the time axis, the synthetic `value` column, and one column - /// per distinct name referenced anywhere in the DAG — so positional - /// `ColumnId` resolution downstream is total. - pub fn resolve_schema(&self, dag: &UnresolvedQueryExpr) -> Schema { - self.resolve_schema_with_inherited(dag, &[]) + /// per distinct name referenced anywhere in the tree — so positional + /// `FieldId` resolution downstream is total. + pub fn resolve_schema(&self, tree: &UnresolvedQueryExpr) -> Schema { + self.resolve_schema_with_inherited(tree, &[]) } /// Like [`resolve_schema`](Self::resolve_schema), but also seeds `inherited` label names that are diff --git a/docs/design_docs/decisions/concat-unique-keys.md b/docs/design_docs/decisions/concat-unique-keys.md index 9c0148abf..93164e8d1 100644 --- a/docs/design_docs/decisions/concat-unique-keys.md +++ b/docs/design_docs/decisions/concat-unique-keys.md @@ -123,7 +123,7 @@ for why it's fine to ship unused. adds `(discriminator, inner_key)` as the sole unique key, trusting the caller's claim without checking it. - `resolve.rs`'s `Concat` arm resolves a pre-bind (`ColumnRef`) discriminator - key into its post-bind (`ColumnId`) equivalent against the first resolved + key into its post-bind (`FieldId`) equivalent against the first resolved branch's own output schema — the same schema `output_schema()` derives the merged shape from — so the feature works correctly end-to-end for a future caller upstream of `resolve_root`, even though no such caller exists yet. @@ -243,9 +243,9 @@ accuracy issue. All three are fixed on the same PR: the `Concat` arm now pushes `key.discriminator()` and every `key.inner_key()` column into the walk, mirroring `Dedup.cols` exactly. -2. **Resolved `ColumnId`s in `discriminator_unique_key` could go stale after +2. **Resolved `FieldId`s in `discriminator_unique_key` could go stale after `canonicalize()` runs.** `resolve_root_with_inherited` calls `resolve()` - first — which resolves the key's `ColumnRef`s into `ColumnId`s against + first — which resolves the key's `ColumnRef`s into `FieldId`s against `children.first()`'s output schema *as it stood at that point* — then `canonicalize()` runs afterward and can restructure that same first branch: `try_promote_heavy_hitter` and `try_rewrite_rownumber_topk` both @@ -253,7 +253,7 @@ accuracy issue. All three are fixed on the same PR: differently-shaped `Aggregate`, anywhere within the branch (not only at its own top level — the walk is recursive), potentially changing its column count/order. `output_schema()` read the previously-resolved - `ColumnId`s with no consistency check, so a future branch matching one of + `FieldId`s with no consistency check, so a future branch matching one of these rewrite triggers could silently produce a wrong `unique_keys` claim — a wrong query answer, not a missed optimization (per `cse.rs`'s own module doc). Fixed in `canon()` (`canonicalize.rs`): before recursing @@ -264,7 +264,7 @@ accuracy issue. All three are fixed on the same PR: (`Schema` is `PartialEq`/`Eq`); any difference at all — not just a column-count/type change, since a same-shaped-but-different schema is just as unsafe to trust positionally — drops the key (`None`) rather - than risk keeping a `ColumnId` that now points at the wrong column or is + than risk keeping a `FieldId` that now points at the wrong column or is out of bounds. The key is never *re-derived* by guessing at name or position: the two rewrites don't preserve column identity in a way that's safe to infer, so dropping is the only sound outcome once the diff --git a/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md b/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md index 520501731..613acf8e9 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md +++ b/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md @@ -665,7 +665,7 @@ per-series intents such as exact quantile, cardinality, and Top-K aggregate intents remain unavailable until they have an explicit physical algorithm. Hash-join lowering also uses the bound left and right output schemas to prove that every equality -compares one column from each side; same-side or out-of-range `ColumnId`s fail +compares one column from each side; same-side or out-of-range `FieldId`s fail closed. An `Rc` address is not physical identity. Every logical occurrence diff --git a/docs/design_docs/proposals/decoupling_op_and_expr.md b/docs/design_docs/proposals/decoupling_op_and_expr.md index fe458bb80..66c3a69e1 100644 --- a/docs/design_docs/proposals/decoupling_op_and_expr.md +++ b/docs/design_docs/proposals/decoupling_op_and_expr.md @@ -58,7 +58,7 @@ operator inputs and scalar query-result references use `Rc`. `NonASAPOp` is the payload of an ordinary operator, not a second DAG-node type. `BinaryOp` likewise uses the single `BinaryOperator` payload specified there. -Names are resolved to `ColumnId` before constructing these nodes. Parsing and +Names are resolved to `FieldId` before constructing these nodes. Parsing and unresolved `ColumnRef` handling remain frontend concerns; no alternative generic operator definition is proposed here. These wrappers belong to operator fields and use the `ScalarExpr` defined in §2.2: @@ -100,12 +100,12 @@ time of that whole input. Existing signed offsets and `AtModifier` anchors remai Scalar recursion uses owned `Box` and `Vec` children. The only plan references are explicit operations that consume a query result to compute a value. Those edges remain visible to plan traversal and costing; they cannot hide a separate plan. -The definitions below use resolved `ColumnId`s and the common `OperatorNode`; +The definitions below use resolved `FieldId`s and the common `OperatorNode`; there is no separate pre-ASAP scalar representation. ```rust enum ScalarExpr { - Column(ColumnId), + Column(FieldId), Literal(ScalarValue), Negative { expr: Box, semantics: ExprSemantics }, Compare { diff --git a/docs/design_docs/proposals/operator-sharing.md b/docs/design_docs/proposals/operator-sharing.md index 884226238..6e1df70bb 100644 --- a/docs/design_docs/proposals/operator-sharing.md +++ b/docs/design_docs/proposals/operator-sharing.md @@ -89,7 +89,7 @@ introduce state construction and readout. `CurrentTimestamp`, `EvalTimestamp` and `PromqlScalarFromVector` belong to `ScalarExpr`, defined in the [companion proposal](decoupling_op_and_expr.md#22-scalar-expressions). -A constant needs no bridge operator. The sketches use resolved `ColumnId`s and +A constant needs no bridge operator. The sketches use resolved `FieldId`s and `Schema`; name resolution precedes construction of these nodes. `NonASAPOp` retains the query semantics needed before and after optimization: @@ -113,7 +113,7 @@ enum NonASAPOp { Concat { children: Vec>, discriminator_unique_key: Option, }, - Dedup { child: Rc, cols: Vec }, + Dedup { child: Rc, cols: Vec }, Sort { child: Rc, keys: Vec, partition_by: GroupKeys }, Limit { child: Rc, n: Option, offset: usize, partition_by: GroupKeys }, BinaryOp { @@ -168,9 +168,9 @@ enum ASAPOp { // Reserved operations; semantics and support require further design. SummaryMerge { children: Vec> }, SummarySubtract { left: Rc, right: Rc }, - SummaryDelete { summary_input: Rc, key: ColumnId }, + SummaryDelete { summary_input: Rc, key: FieldId }, SummaryJoin { - outer: Rc, inner: Rc, key: ColumnId, family: FieldDataType, + outer: Rc, inner: Rc, key: FieldId, family: FieldDataType, }, Extension { child: Rc, name: String }, } @@ -255,7 +255,7 @@ Scan node: OperatorNode ``` This is abbreviated structural notation: `Column` and `Literal` above are -`ScalarExpr` variants; column names stand for resolved `ColumnId`s. The arithmetic +`ScalarExpr` variants; column names stand for resolved `FieldId`s. The arithmetic and comparison use `ExprSemantics::Sql`. The aggregate has no grouping keys and names its output `sum_bytes`; the projection names its output `total_bytes`. @@ -351,7 +351,7 @@ enum ExecutionTiming { `ResultGuarantee` retains its existing definition. `Operator`, `OperatorNode`, `OperatorResultKind` and the common node layout are proposed; `Schema` is unified as specified below. This is a resolved-plan interface: name resolution must finish -before producing these concrete `ColumnId`/`Schema` nodes. +before producing these concrete `FieldId`/`Schema` nodes. | Plan stage | Required property state | |---|---| @@ -381,8 +381,8 @@ struct Field { struct Schema { fields: Vec, - time_index: Option, - unique_keys: Vec>, + time_index: Option, + unique_keys: Vec>, closed: bool, } diff --git a/docs/develop_docs/pre-asap-ir.md b/docs/develop_docs/pre-asap-ir.md index dcb3597e2..2d1e08b7a 100644 --- a/docs/develop_docs/pre-asap-ir.md +++ b/docs/develop_docs/pre-asap-ir.md @@ -65,7 +65,7 @@ list of aggregate intents (`measures`). - **`Reduce(GroupKeys)`** — a cross-row reduction: group by some columns, or group by every column *except* some listed ones. `GroupKeys` holds positional column references - (`ColumnId`s — indexes into the input schema, not column names) and carries a `by`/`without` + (`FieldId`s — indexes into the input schema, not column names) and carries a `by`/`without` flag, not just a plain list: - `by(keys)` — group by exactly these columns (SQL `GROUP BY`, PromQL `by(...)`). - `without(keys)` — group by every column *except* these (PromQL `without(...)`); the From 4bf4fe95eda1f40a755ecf359b68c3acf487b92b Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Fri, 2 Oct 2026 19:22:36 +0000 Subject: [PATCH 4/7] fix(schema): retain ColumnId for bound column references --- .../src/accuracy/reconciliation.rs | 6 +-- crates/asap-aware-mapping/src/replacement.rs | 14 ++--- crates/asap-aware-mapping/src/rewrite.rs | 6 +-- crates/asap-aware-mapping/src/rollup.rs | 42 +++++++-------- crates/frontend-promql/src/promql.rs | 4 +- .../frontend-promql/tests/promql_lowering.rs | 2 +- crates/frontend-sql/src/sql/mod.rs | 10 ++-- crates/frontend-sql/tests/sql_lowering.rs | 6 +-- crates/integration-tests/tests/aggregate.rs | 2 +- crates/types/src/post_asap/expr.rs | 2 +- crates/types/src/pre_asap/agg_intent.rs | 26 ++++----- crates/types/src/pre_asap/canonicalize.rs | 8 +-- .../types/src/pre_asap/column_resolution.rs | 16 +++--- crates/types/src/pre_asap/expr_ir.rs | 7 +-- crates/types/src/pre_asap/mod.rs | 6 +-- crates/types/src/pre_asap/query_expr.rs | 54 ++++++++++--------- crates/types/src/pre_asap/resolve.rs | 32 +++++------ crates/types/src/pre_asap/schema.rs | 34 ++++++------ crates/types/src/pre_asap/schema_resolver.rs | 4 +- .../decisions/concat-unique-keys.md | 10 ++-- .../analytical-resource-cost.md | 2 +- .../proposals/decoupling_op_and_expr.md | 6 +-- .../design_docs/proposals/operator-sharing.md | 16 +++--- docs/develop_docs/pre-asap-ir.md | 23 +++++++- 24 files changed, 183 insertions(+), 155 deletions(-) diff --git a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs index 76088e227..6014d3ba3 100644 --- a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs +++ b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs @@ -394,7 +394,7 @@ mod tests { use asap_types::post_asap::SketchAlgorithm; use asap_types::pre_asap::cse::share_common_sub_dags; use asap_types::pre_asap::query_expr::{GroupKeys, Source}; - use asap_types::pre_asap::schema::{DataType, Field, FieldId, Schema}; + use asap_types::pre_asap::schema::{ColumnId, DataType, Field, Schema}; /// `[ts(0), value(1), job(2)]`. /// A unique-keyed scan (`[ts]`) so `share_common_sub_dags` is actually @@ -417,7 +417,7 @@ mod tests { }) } - fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { + fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { Rc::new(QueryExpr::Aggregate { reduction: Reduction::by(by), measures: vec![intent], @@ -462,7 +462,7 @@ mod tests { fn without_quantile( q: f64, accuracy: AccuracyTarget, - excluded: Vec, + excluded: Vec, child: &Rc, ) -> Rc { Rc::new(QueryExpr::Aggregate { diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 0e947c691..4b766579a 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -366,7 +366,7 @@ use asap_types::pre_asap::query_expr::any_measure_filtered; use asap_types::pre_asap::query_expr::{ BinaryOpKind, Predicate, QueryExpr, QueryExprError, Reduction, }; -use asap_types::pre_asap::schema::FieldId; +use asap_types::pre_asap::schema::ColumnId; use asap_types::types::AccuracyTarget; use asap_types::workload::{DataWorkload, QueryRecurrence, QueryWorkload, RepeatedDemand}; use std::rc::Rc; @@ -3050,7 +3050,7 @@ fn construct_summary_agg( } // `reduction` is carried onto `SummaryAgg` verbatim — not flattened to a - // bare `Vec` — so `SummaryExecutor::find_candidates` can tell + // bare `Vec` — so `SummaryExecutor::find_candidates` can tell // a genuine empty-`by` reduction apart from a per-entity shape with no // grouping concept at all (issue #163). `construct_summary_agg` is the // single place that decides this; nothing downstream re-derives it. @@ -5112,8 +5112,8 @@ fn normalize_cross_input_equi_predicate( else { return None; }; - let is_left = |id: FieldId| id < left_width; - let is_right = |id: FieldId| left_width <= id && id < total_width; + let is_left = |id: ColumnId| id < left_width; + let is_right = |id: ColumnId| left_width <= id && id < total_width; let (left_id, right_id) = if is_left(*left_id) && is_right(*right_id) { (*left_id, *right_id) } else if is_right(*left_id) && is_left(*right_id) { @@ -7185,7 +7185,7 @@ mod tests { assert!(space.enumerate_candidate_dags(0).is_err()); } - fn equi_pred(left: FieldId, right: FieldId) -> Predicate { + fn equi_pred(left: ColumnId, right: ColumnId) -> Predicate { Predicate(Rc::new(QueryExpr::Compare { left: Rc::new(QueryExpr::Column(left)), op: asap_types::pre_asap::CompareOpKind::Eq, @@ -10108,7 +10108,7 @@ mod tests { /// `quantile_over_time(...)`) realizes to `SummaryAgg { reduction: /// PerEntity, .. }` — proving the pre-ASAP `Reduction` this crate /// already computes (issue #165) is carried onto the post-ASAP node - /// verbatim, not flattened back into an ambiguous bare `Vec`. + /// verbatim, not flattened back into an ambiguous bare `Vec`. #[test] fn bare_per_series_aggregate_realizes_summary_agg_with_per_entity_reduction() { use std::time::Duration; @@ -10132,7 +10132,7 @@ mod tests { /// Issue #163, case 2: an aggregation operator explicitly invoked with /// no grouping keys realizes to `SummaryAgg { /// reduction: Reduce(vec![]), .. }` — byte-identical `by: []` to the - /// previous test at the old `Vec` shape; `reduction` is what + /// previous test at the old `Vec` shape; `reduction` is what /// tells them apart now. #[test] fn explicit_empty_by_aggregate_realizes_summary_agg_with_reduce_reduction() { diff --git a/crates/asap-aware-mapping/src/rewrite.rs b/crates/asap-aware-mapping/src/rewrite.rs index f51839555..94a3d638c 100644 --- a/crates/asap-aware-mapping/src/rewrite.rs +++ b/crates/asap-aware-mapping/src/rewrite.rs @@ -64,7 +64,7 @@ use asap_types::pre_asap::expr_ir::ArithmeticOpKind; use asap_types::pre_asap::query_expr::{ any_measure_filtered, BinaryOpKind, ProjectItem, QueryExpr, Reduction, }; -use asap_types::pre_asap::schema::{DataType, FieldId}; +use asap_types::pre_asap::schema::{ColumnId, DataType}; use asap_types::types::AccuracyTarget; use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG}; @@ -74,7 +74,7 @@ use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, Ta /// the module docs' "Scope" for why `without(...)`/`PerEntity` are /// excluded). Returns the grouping key count and the summed column so /// [`build_rewrite`] doesn't have to re-match. -fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { +fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { let QueryExpr::Aggregate { reduction, measures, @@ -431,7 +431,7 @@ mod tests { } } - fn avg_agg(by: Vec, col: Option, child: QueryExpr) -> QueryExpr { + fn avg_agg(by: Vec, col: Option, child: QueryExpr) -> QueryExpr { QueryExpr::Aggregate { reduction: Reduction::by(by), measures: vec![AggIntent::Avg { col }], diff --git a/crates/asap-aware-mapping/src/rollup.rs b/crates/asap-aware-mapping/src/rollup.rs index bf3c001c0..ab6eff3d9 100644 --- a/crates/asap-aware-mapping/src/rollup.rs +++ b/crates/asap-aware-mapping/src/rollup.rs @@ -64,9 +64,9 @@ //! asks for) consults it, and so does this module's `replacements`, so the //! two can never disagree about which intents are eligible. //! -//! ## `FieldId` comparability — only sound for identical child IR +//! ## `ColumnId` comparability — only sound for identical child IR //! -//! A `FieldId` is a *position* into a specific `Schema` (`crates/types/src/pre_asap/schema.rs`'s +//! A `ColumnId` is a *position* into a specific `Schema` (`crates/types/src/pre_asap/schema.rs`'s //! own doc: "the same edge, the same schema, the same positional numbering"). //! Comparing the coarser aggregate's `by` positions against the finer //! aggregate's `by` positions is only meaningful when both aggregates have @@ -74,7 +74,7 @@ //! Thus both `by` lists index the same shape. The equality fallback matters //! for scans that CSE conservatively declines to alias because they have no //! declared unique key. Structurally different sources remain out of scope: -//! this module never reconciles `FieldId`s across distinct schemas. +//! this module never reconciles `ColumnId`s across distinct schemas. //! //! ## Non-goals (tracked separately, not attempted here — same split //! `replacement.rs`'s own module docs draw for `SharedSubDAGStrategy`'s @@ -90,10 +90,10 @@ //! `asap-aware-mapping`'s scope (see issue #254's own "Non-goal" section) //! — this module only constructs the pre-ASAP [`QueryExpr::Aggregate`] //! rewrite; a `CostModel`/search engine decides whether to prefer it. -//! - **No cross-schema reconciliation** (see "`FieldId` comparability" +//! - **No cross-schema reconciliation** (see "`ColumnId` comparability" //! above) and **no `without(...)` grouping support** — `without`'s kept //! set is runtime-open (never enumerable at plan time, per -//! `GroupKeys`'s own doc), so there is no fixed `FieldId` set to compare +//! `GroupKeys`'s own doc), so there is no fixed `ColumnId` set to compare //! against a superset/subset relationship at all; [`is_legal_rollup_source`] //! declines both directions. @@ -102,7 +102,7 @@ use std::rc::Rc; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::pre_asap::query_expr::{any_measure_filtered, GroupKeys, QueryExpr, Reduction}; -use asap_types::pre_asap::schema::{FieldId, Schema}; +use asap_types::pre_asap::schema::{ColumnId, Schema}; use asap_types::types::AccuracyTarget; use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG}; @@ -153,7 +153,7 @@ fn bindable_grouped_aggregate( /// deliberately: `agg_is_mergeable` answers "does *some* partial-state merge /// exist", not "is self- or sum-recombination the right one," and this /// module only ever proposes a rewrite it can construct correctly. -fn rollup_combinator(intent: &AggIntent, finer_measure_col: FieldId) -> Option { +fn rollup_combinator(intent: &AggIntent, finer_measure_col: ColumnId) -> Option { match intent { // Self-combining: reapplying the identical operator over the finer // side's own output column is correct unchanged. @@ -199,7 +199,7 @@ fn rollup_combinator(intent: &AggIntent, finer_measure_col: FieldId) -> Option Option bool { - let finer_set: HashSet<&FieldId> = finer.iter().collect(); - let coarser_set: HashSet<&FieldId> = coarser.iter().collect(); +fn is_strict_column_superset(finer: &[ColumnId], coarser: &[ColumnId]) -> bool { + let finer_set: HashSet<&ColumnId> = finer.iter().collect(); + let coarser_set: HashSet<&ColumnId> = coarser.iter().collect(); if finer_set.len() <= coarser_set.len() { return false; } @@ -341,14 +341,14 @@ impl ReplacementStrategy for RollupStrategy { /// measure column, with `child = finer` instead of the original shared /// source. /// -/// `coarser_by`'s `FieldId`s are positions into the *shared child's* -/// schema (the same schema `finer_by`'s `FieldId`s index into — see the -/// module docs' "`FieldId` comparability" section). `finer`'s own output +/// `coarser_by`'s `ColumnId`s are positions into the *shared child's* +/// schema (the same schema `finer_by`'s `ColumnId`s index into — see the +/// module docs' "`ColumnId` comparability" section). `finer`'s own output /// schema is a *different* schema (`finer_by`'s columns, in order, followed /// by its one measure column — `aggregate_output_schema`'s `by ++ measures` /// shape), so each of `coarser_by`'s columns must be translated from its /// position in the shared child to its position in `finer`'s output: the -/// index its `FieldId` occupies within `finer_by`'s own ordered list. +/// index its `ColumnId` occupies within `finer_by`'s own ordered list. fn build_rollup( finer: &Rc, coarser_by: &GroupKeys, @@ -358,10 +358,10 @@ fn build_rollup( let (finer_by, _, _) = bindable_grouped_aggregate(finer)?; // `finer`'s own single measure sits right after its `by` columns in its // output schema (`aggregate_output_schema`'s `by ++ measures` layout). - let finer_measure_col: FieldId = finer_by.len(); + let finer_measure_col: ColumnId = finer_by.len(); let combinator = rollup_combinator(intent, finer_measure_col)?; - let remapped_by: Vec = coarser_by + let remapped_by: Vec = coarser_by .keys() .iter() .map(|id| finer_by.keys().iter().position(|f| f == id)) @@ -416,7 +416,7 @@ mod tests { } } - fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { + fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { Rc::new(QueryExpr::Aggregate { reduction: Reduction::by(by), measures: vec![intent], @@ -428,7 +428,7 @@ mod tests { } fn without_agg( - excluded: Vec, + excluded: Vec, intent: AggIntent, child: &Rc, ) -> Rc { diff --git a/crates/frontend-promql/src/promql.rs b/crates/frontend-promql/src/promql.rs index f0c615a7b..83251053e 100644 --- a/crates/frontend-promql/src/promql.rs +++ b/crates/frontend-promql/src/promql.rs @@ -8,14 +8,14 @@ //! label matchers) and emits `UnresolvedQueryExpr` nodes with unresolved //! `ColumnRef`s — the same DAG shape //! [`resolve_root`](asap_types::pre_asap::resolve_root) later binds to -//! canonical, positional `QueryExpr`. The structural decisions a +//! canonical, positional `QueryExpr`. The structural decisions a //! separate converter stage would otherwise have to make (heavy-hitter //! `topk` recognition, the `PerEntity`/`Reduce` reduction choice, //! `without(...)` grouping) are made right here, since a front end //! building this shape already knows the answer at parse time — see //! `reduction_for` and `mark_without`. `resolve_root` is left with exactly //! the schema-*dependent* work: binding every `ColumnRef` to its -//! positional `FieldId`. +//! positional `ColumnId`. //! //! # PromQL → canonical unresolved-DAG mapping (summary) //! diff --git a/crates/frontend-promql/tests/promql_lowering.rs b/crates/frontend-promql/tests/promql_lowering.rs index 25e21ebfc..243bf6d18 100644 --- a/crates/frontend-promql/tests/promql_lowering.rs +++ b/crates/frontend-promql/tests/promql_lowering.rs @@ -605,7 +605,7 @@ fn topk_over_count_is_heavy_hitter_topk() { else { panic!("expected Aggregate with TopK, got {qe:?}"); }; - // `service` is the only group key → resolved to a positional FieldId. + // `service` is the only group key → resolved to a positional ColumnId. assert_eq!(reduction.expect_reduce().len(), 1); assert!(matches!( measures.as_slice(), diff --git a/crates/frontend-sql/src/sql/mod.rs b/crates/frontend-sql/src/sql/mod.rs index 2fd17f3b0..d1401988c 100644 --- a/crates/frontend-sql/src/sql/mod.rs +++ b/crates/frontend-sql/src/sql/mod.rs @@ -6,7 +6,7 @@ //! walks the unoptimized `LogicalPlan` and emits `UnresolvedQueryExpr` nodes with //! unresolved `ColumnRef`s directly (issue #179) — the same DAG shape //! [`resolve_root`](asap_types::pre_asap::resolve_root) binds to canonical, -//! positional `QueryExpr`. Unlike PromQL's front end, SQL's +//! positional `QueryExpr`. Unlike PromQL's front end, SQL's //! Ordinary SQL `Aggregate` nodes are `Reduction::Reduce`. The explicit //! `asap_rate`/`asap_increase` bridge is the narrow exception: it //! spells a time-series range reducer with an explicit value, time-index, and @@ -1704,7 +1704,7 @@ fn conditional_count_arm(expr: &Expr) -> Option<(&Expr, &Expr)> { /// modifier rule, the "reducer argument must be a bare column" rule /// (`reducer_col`), φ extraction from a literal argument, and the ambient /// `AccuracyTarget`. `resolve_root` resolves `col` to a positional -/// `FieldId`; the output name (DataFusion's own, e.g. +/// `ColumnId`; the output name (DataFusion's own, e.g. /// `"sum(metrics.bytes)"`) is threaded separately as `Aggregate.output_names`, /// not carried here. fn lower_agg_intent(expr: &Expr) -> Result, LoweringError> { @@ -1734,7 +1734,7 @@ fn lower_agg_intent(expr: &Expr) -> Result, LoweringError> // Value reducers (`reducer_col`) require a real column — `SUM(a*b)` // is rejected, not silently reduced over a probe column. Quantile // and CountDistinct reduce a column too, so they take the same path: - // `col` is `Option` once resolved, where `None` means "the + // `col` is `Option` once resolved, where `None` means "the // PromQL sample value", which a SQL query never has. Taking an // expression here would set `col: None` and silently drop it (#115). let col = |args: &[Expr]| -> Result, LoweringError> { @@ -1890,7 +1890,7 @@ fn scalar_positive_u64(value: &DfScalarValue) -> Option { /// core variant (issue #232). Core treats `Extension` opaquely: both columns /// are kept only as validated bare-column names in `payload` (`reducer_col`'s /// same "no expression arguments" rule, issue #115) — they are **not** run -/// through `resolve_agg_intent`'s positional `ColumnRef` -> `FieldId` +/// through `resolve_agg_intent`'s positional `ColumnRef` -> `ColumnId` /// binding the way a real reducer's `col` is, since `Extension` carries no /// typed column field for core to resolve. Shared `arg_selector_columns` validates /// and resolves those names during aggregate schema derivation, preserving the @@ -2104,7 +2104,7 @@ fn expand_grouping_set(gs: &logical_expr::GroupingSet) -> Vec> { /// Derived columns materialized in a `Project` beneath an `Aggregate` (#110). /// -/// `Aggregate.by` holds positional `FieldId`s and each reducer holds one input +/// `Aggregate.by` holds positional `ColumnId`s and each reducer holds one input /// column, so neither can hold an expression. `GROUP BY date_trunc('minute', t)` /// and `SUM(bytes * 8)` are therefore rewritten to group/reduce over a projected /// column that carries the expression's value. diff --git a/crates/frontend-sql/tests/sql_lowering.rs b/crates/frontend-sql/tests/sql_lowering.rs index 2f0e6727e..883966db9 100644 --- a/crates/frontend-sql/tests/sql_lowering.rs +++ b/crates/frontend-sql/tests/sql_lowering.rs @@ -424,7 +424,7 @@ async fn count_distinct_is_cardinality() { async fn select_distinct_lowers_to_distinct_with_positional_cols() { // SELECT DISTINCT → a `Dedup` node whose `cols` are positional ColumnIds // (not name-based ColumnRefs). DataFusion's `Distinct::All` dedups on every - // column, so `cols` is empty here — but the field type is now `Vec`. + // column, so `cols` is empty here — but the field type is now `Vec`. let qe = lower("SELECT DISTINCT service FROM metrics").await; let QueryExpr::Dedup { cols, .. } = &qe else { panic!("expected a Dedup at the root, got {qe:?}"); @@ -453,7 +453,7 @@ async fn inner_join_lowers_to_join_over_two_scans() { assert!(matches!(right.as_ref(), QueryExpr::Scan { .. })); } -/// The two `FieldId`s an equijoin predicate `Column(l) = Column(r)` binds to, +/// The two `ColumnId`s an equijoin predicate `Column(l) = Column(r)` binds to, /// returned sorted so the assertion is independent of left/right ordering. fn join_eq_columns(join: &QueryExpr) -> [usize; 2] { let QueryExpr::Join { pred, .. } = join else { @@ -1955,7 +1955,7 @@ async fn arg_min_lowers_to_its_own_extension_kind() { async fn arg_max_payload_preserves_both_column_names() { // Core never resolves an `Extension`'s payload, so both columns are kept // as validated bare-column `ColumnRef`s in `payload`, not run through - // positional `FieldId` binding -- see `lower_arg_selector`'s doc. + // positional `ColumnId` binding -- see `lower_arg_selector`'s doc. let qe = lower_clickhouse("SELECT argMax(service, latency) AS m FROM metrics").await; let (_, measures) = find_aggregate(&qe).expect("expected an Aggregate"); let AggIntent::Extension { payload, .. } = &measures[0] else { diff --git a/crates/integration-tests/tests/aggregate.rs b/crates/integration-tests/tests/aggregate.rs index 2d93d5774..051eeb6b7 100644 --- a/crates/integration-tests/tests/aggregate.rs +++ b/crates/integration-tests/tests/aggregate.rs @@ -4,7 +4,7 @@ //! //! Cross-series aggregates lower to a single `Aggregate` node with no //! `TimeRange` child (range functions use `TimeRange` — see `time_range.rs`). -//! Group keys land on `Aggregate.by` as positional `FieldId`s. +//! Group keys land on `Aggregate.by` as positional `ColumnId`s. //! Single-stat PromQL aggregates always get `output_names: [""]` (no alias) //! and `having: None`. diff --git a/crates/types/src/post_asap/expr.rs b/crates/types/src/post_asap/expr.rs index bac6a5079..9312d95c6 100644 --- a/crates/types/src/post_asap/expr.rs +++ b/crates/types/src/post_asap/expr.rs @@ -182,7 +182,7 @@ pub enum SummaryExpr { /// How this aggregation's output rows relate to `child`'s — the /// same [`Reduction`] the pre-ASAP `Aggregate` node it was bound /// from carried (issue #165), reused verbatim rather than - /// flattened to a bare `Vec`. `Reduction::Reduce(by)` + /// flattened to a bare `Vec`. `Reduction::Reduce(by)` /// with an empty `by` is a genuine full reduction (merge every /// candidate into one group); `Reduction::PerEntity` has no /// grouping concept at all (never merge across entities) — the diff --git a/crates/types/src/pre_asap/agg_intent.rs b/crates/types/src/pre_asap/agg_intent.rs index e1f5b4e35..c60dd55e3 100644 --- a/crates/types/src/pre_asap/agg_intent.rs +++ b/crates/types/src/pre_asap/agg_intent.rs @@ -16,19 +16,19 @@ use serde::{Deserialize, Serialize}; use crate::pre_asap::query_expr::DataModel; -use crate::pre_asap::schema::{DataType, Field, FieldDataType, FieldId}; +use crate::pre_asap::schema::{ColumnId, DataType, Field, FieldDataType}; use crate::types::AccuracyTarget; /// "What to compute" — the vocabulary the planner pivots on. /// /// Grouping for `TopK` rides on the enclosing `QueryExpr::Aggregate.by` -/// (positional `FieldId`s), like every other aggregate; the intent itself +/// (positional `ColumnId`s), like every other aggregate; the intent itself /// carries only `k` + the accuracy target. /// /// The single-column reducers (`Sum` / `Min` / `Max` / `Avg` / `StdDev` / /// `Variance` / `Quantile`) carry `col: Option` — the input /// column they reduce, generic over the column-reference state the same way -/// [`QueryExpr`](super::query_expr::QueryExpr) is: positional `FieldId` once +/// [`QueryExpr`](super::query_expr::QueryExpr) is: positional `ColumnId` once /// bound (the default, and every existing use of the bare `AggIntent` name), /// or an unresolved name-based `ColumnRef` for a front end constructing this /// intent directly, before the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has run. @@ -43,7 +43,7 @@ use crate::types::AccuracyTarget; #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(tag = "kind", rename_all = "snake_case", deny_unknown_fields)] #[serde(bound(serialize = "C: Serialize", deserialize = "C: Deserialize<'de>"))] -pub enum AggIntent { +pub enum AggIntent { // ── Data-model-agnostic ────────────────────────────────────────────── Count { accuracy: AccuracyTarget, @@ -386,7 +386,7 @@ pub enum MathFunc { // the `PerEntity`/`Reduce` reduction shape right at construction time (see // `asap_frontend_promql::promql::reduction_for`) — so they stay generic // alongside `input_col`, in one `impl` block. -impl AggIntent { +impl AggIntent { /// Resolve the existing SQL arg-selector extension using its child schema. /// The tuple is (selected value column, ordering column). /// Unknown extensions remain owned by their deployment model. Recognized @@ -394,7 +394,7 @@ impl AggIntent { pub fn arg_selector_columns( &self, schema: &super::schema::Schema, - ) -> Result, String> { + ) -> Result, String> { let Self::Extension { ext_kind, payload } = self else { return Ok(None); }; @@ -407,7 +407,7 @@ impl AggIntent { if fields.len() != 2 { return Err("arg selector requires arg_col and val_col only".into()); } - let resolve = |field: &str| -> Result { + let resolve = |field: &str| -> Result { let reference: super::expr_ir::ColumnRef = serde_json::from_value( fields .get(field) @@ -782,7 +782,7 @@ mod tests { fn output_column_names_are_intent_keyed() { let v = c("value", DataType::Float64); assert_eq!( - AggIntent::::Count { + AggIntent::::Count { accuracy: AccuracyTarget::Exact } .output_column(&v) @@ -790,13 +790,13 @@ mod tests { "count" ); assert_eq!( - AggIntent::::Sum { col: None } + AggIntent::::Sum { col: None } .output_column(&v) .name, "sum" ); assert_eq!( - AggIntent::::Quantile { + AggIntent::::Quantile { col: None, q: 0.99, accuracy: AccuracyTarget::Epsilon(0.01) @@ -810,7 +810,7 @@ mod tests { #[test] fn sum_preserves_input_dtype() { assert!(matches!( - AggIntent::::Sum { col: None } + AggIntent::::Sum { col: None } .output_column(&c("c", DataType::Int64)) .dtype, FieldDataType::Plain(DataType::Int64) @@ -871,12 +871,12 @@ mod tests { fn input_cols_tracks_only_reducers() { assert_eq!(AggIntent::Sum { col: Some(3) }.input_cols(), vec![3]); assert!( - AggIntent::::Avg { col: None } + AggIntent::::Avg { col: None } .input_cols() .is_empty(), "empty = PromQL sample value" ); - assert!(AggIntent::::Count { + assert!(AggIntent::::Count { accuracy: AccuracyTarget::Exact } .input_cols() diff --git a/crates/types/src/pre_asap/canonicalize.rs b/crates/types/src/pre_asap/canonicalize.rs index ddab95df7..b9e6a653a 100644 --- a/crates/types/src/pre_asap/canonicalize.rs +++ b/crates/types/src/pre_asap/canonicalize.rs @@ -38,13 +38,13 @@ pub fn canonicalize(mut expr: QueryExpr) -> QueryExpr { fn canon(expr: &mut QueryExpr) { // A `Concat` asserting a caller-proven `discriminator_unique_key` (issue - // #228) had that key's `FieldId`s resolved, in `resolve.rs`, against + // #228) had that key's `ColumnId`s resolved, in `resolve.rs`, against // exactly the first branch's output schema *as it stood before this // pass ran*. `try_promote_additive_top_ranking`/`try_rewrite_rownumber_topk` // below can restructure that branch (anywhere within it — not only at // its own top level, since the same recursive walk can rewrite a node // nested under a pass-through wrapper too) into a shape with a - // different output schema, which would leave those `FieldId`s + // different output schema, which would leave those `ColumnId`s // pointing at the wrong column, or out of bounds, of the // post-canonicalize schema. Snapshot the schema the discriminator key // was actually resolved against, right here, before recursing into the @@ -475,10 +475,10 @@ mod tests { // ── Concat's discriminator_unique_key vs. canonicalize (issue #228 review) ── // - // `resolve.rs` resolves `discriminator_unique_key`'s `FieldId`s against + // `resolve.rs` resolves `discriminator_unique_key`'s `ColumnId`s against // the first branch's *pre-canonicalize* output schema. If canonicalize // then restructures that branch (heavy-hitter promotion, the - // `ROW_NUMBER()` top-k rewrite), those `FieldId`s can end up pointing at + // `ROW_NUMBER()` top-k rewrite), those `ColumnId`s can end up pointing at // the wrong column — or out of bounds — of the new schema. The two tests // below pin the fix: the key is dropped whenever the branch's schema // actually changed, and survives untouched otherwise. Never guessed at. diff --git a/crates/types/src/pre_asap/column_resolution.rs b/crates/types/src/pre_asap/column_resolution.rs index 0cab6362d..f137906dd 100644 --- a/crates/types/src/pre_asap/column_resolution.rs +++ b/crates/types/src/pre_asap/column_resolution.rs @@ -1,7 +1,7 @@ //! Schema-driven column resolution. //! //! Front ends (issue #179) emit `ColumnRef` (name-based, optionally -//! table-qualified); the canonical tree uses positional [`FieldId`] resolved +//! table-qualified); the canonical tree uses positional [`ColumnId`] resolved //! against a per-node [`Schema`]. These helpers bridge the two — the //! [`SchemaResolver`](super::schema_resolver) builds the schema, and [`resolve_column_refs`] //! turns name-based refs (group keys, dedup columns) into positional ids, @@ -17,7 +17,7 @@ use super::query_expr::{ aggregate_output_schema, GroupKeys, QueryExpr, QueryExprError, Reduction, ResolvedQueryExpr, UnresolvedQueryExpr, }; -use super::schema::{DataType, FieldDataType, FieldId, Schema}; +use super::schema::{ColumnId, DataType, FieldDataType, Schema}; /// Errors returned by the resolution helpers. #[derive(Debug, Error, PartialEq, Eq)] @@ -29,12 +29,12 @@ pub enum ResolveError { }, #[error("ColumnRef::SampleValue has no `value` column in schema (have: {available:?})")] NoSampleValue { available: Vec }, - #[error("ColumnRef::Wildcard cannot be resolved to a single FieldId")] + #[error("ColumnRef::Wildcard cannot be resolved to a single ColumnId")] WildcardNotPositional, } -/// Resolve a single [`ColumnRef`] to a positional [`FieldId`]. -pub fn resolve_column_ref(col: &ColumnRef, schema: &Schema) -> Result { +/// Resolve a single [`ColumnRef`] to a positional [`ColumnId`]. +pub fn resolve_column_ref(col: &ColumnRef, schema: &Schema) -> Result { match col { ColumnRef::Named(name) => schema .column_id(name) @@ -62,7 +62,7 @@ pub fn resolve_column_ref(col: &ColumnRef, schema: &Schema) -> Result = (0..schema.fields.len()) + let numeric: Vec = (0..schema.fields.len()) .filter(|&i| Some(i) != schema.time_index) .filter(|&i| { matches!( @@ -84,7 +84,7 @@ pub fn resolve_column_ref(col: &ColumnRef, schema: &Schema) -> Result Result, ResolveError> { +) -> Result, ResolveError> { cols.iter().map(|c| resolve_column_ref(c, schema)).collect() } @@ -106,7 +106,7 @@ pub fn resolve_column_refs( pub fn resolve_group_keys_promql( cols: &[ColumnRef], schema: &Schema, -) -> Result, ResolveError> { +) -> Result, ResolveError> { cols.iter() .filter_map(|c| match resolve_column_ref(c, schema) { Err(ResolveError::NotFound { .. }) if schema.closed => None, diff --git a/crates/types/src/pre_asap/expr_ir.rs b/crates/types/src/pre_asap/expr_ir.rs index 27809c8fe..56330530c 100644 --- a/crates/types/src/pre_asap/expr_ir.rs +++ b/crates/types/src/pre_asap/expr_ir.rs @@ -8,7 +8,7 @@ //! DAG, not two type families joined by wrappers — generic over the same //! column-reference state `C` the rest of `QueryExpr` already carries //! (issue #179): [`ColumnRef`] (name-based, front-end-emitted) or -//! [`FieldId`](super::schema::FieldId) (positional, once bound). +//! [`ColumnId`](super::schema::ColumnId) (positional, once bound). //! //! What's left here is the vocabulary those scalar variants are built from — //! [`ScalarValue`], [`CompareOpKind`], [`ArithmeticOpKind`] — the **union** of what the two @@ -21,8 +21,9 @@ use serde::{Deserialize, Serialize}; /// A name-based column reference — the front-end-emitted, unresolved state of /// [`QueryExpr::Column`](super::query_expr::QueryExpr::Column) (`C = /// ColumnRef`); the [`SchemaResolver`](super::schema_resolver::SchemaResolver) resolves it to a -/// positional [`FieldId`](super::schema::FieldId). Includes the two -/// PromQL-conventional synthetic columns. +/// positional [`ColumnId`](super::schema::ColumnId). This is a logical reference, +/// not schema metadata or a runtime data array. `SampleValue` names the implicit +/// PromQL sample column; `Wildcard` represents an all-columns/rows request. #[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] pub enum ColumnRef { Named(String), diff --git a/crates/types/src/pre_asap/mod.rs b/crates/types/src/pre_asap/mod.rs index 8e06fbaba..d1c448756 100644 --- a/crates/types/src/pre_asap/mod.rs +++ b/crates/types/src/pre_asap/mod.rs @@ -3,7 +3,7 @@ //! - [`query_expr`] — the canonical, language- and deployment-independent //! intent algebra: one recursive [`QueryExpr`] DAG (relational operators //! *and* scalar expression shapes both, since issue #205) + [`AggIntent`], -//! generic over the column-reference state (positional [`FieldId`] once +//! generic over the column-reference state (positional [`ColumnId`] once //! bound, name-based [`ColumnRef`] before). //! - [`agg_intent`] — the aggregation-intent vocabulary. //! - [`expr_ir`] — the [`ColumnRef`] column-reference type and the scalar @@ -11,7 +11,7 @@ //! [`QueryExpr`]'s scalar variants are built from. //! - [`schema`] — the per-edge [`Schema`] every node carries. //! - [`schema_resolver`] / [`column_resolution`] — name resolution: turn a `ColumnRef` -//! into a positional `FieldId` against an in-scope [`Schema`]. +//! into a positional `ColumnId` against an in-scope [`Schema`]. //! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedQueryExpr`] tree to //! canonical [`ResolvedQueryExpr`] (issue #179): both front ends //! (`asap-frontend-promql`, `asap-frontend-sql`) construct `UnresolvedQueryExpr` @@ -60,5 +60,5 @@ pub use query_expr::{ WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, }; pub use resolve::{resolve_root, ResolveDAGError}; -pub use schema::{DataType, Field, FieldDataType, FieldId, Schema}; +pub use schema::{ColumnId, DataType, Field, FieldDataType, Schema}; pub use schema_resolver::{SchemaCatalog, SchemaResolver, UsageDerivedCatalog}; diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs index ba88103f0..79fdd854e 100644 --- a/crates/types/src/pre_asap/query_expr.rs +++ b/crates/types/src/pre_asap/query_expr.rs @@ -21,10 +21,10 @@ use thiserror::Error; use super::agg_intent::AggIntent; use super::expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; -use super::schema::{DataType, Field, FieldDataType, FieldId, Schema}; +use super::schema::{ColumnId, DataType, Field, FieldDataType, Schema}; /// The column-reference resolution state a [`QueryExpr`] tree carries — -/// [`FieldId`] (the default, and what the bare `QueryExpr` name has always +/// [`ColumnId`] (the default, and what the bare `QueryExpr` name has always /// meant) once the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has resolved every /// reference positionally, or the front-end-emitted, name-based [`ColumnRef`] /// before binding. The only place the two states differ in *shape* rather @@ -41,7 +41,7 @@ pub trait ColState: type ScanSchema: Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de>; } -impl ColState for FieldId { +impl ColState for ColumnId { type ScanSchema = Schema; } @@ -55,7 +55,7 @@ pub enum QueryExprError { #[error("invalid scalar function signature: {0}")] InvalidScalarSignature(String), #[error("by-column id {0} out of range (input has {1} columns)")] - InvalidGroupByColumn(FieldId, usize), + InvalidGroupByColumn(ColumnId, usize), #[error("Concat requires at least one child")] EmptyConcat, /// [`QueryExpr::output_schema`] called on (or reached, while recursing, a @@ -92,10 +92,10 @@ pub enum QueryExprError { /// `PromqlSeriesSample` groupings are always `by`. /// /// Serialises as a bare array for the (overwhelmingly common) `by` case — -/// wire-compatible with the `Vec` this field held before — and as +/// wire-compatible with the `Vec` this field held before — and as /// `{"without": [...]}` for the exclusion case. #[derive(Debug, Clone, PartialEq, Eq, Hash)] -pub struct GroupKeys { +pub struct GroupKeys { keys: Vec, without: bool, } @@ -430,7 +430,7 @@ pub enum SampleKind { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct SortKey { +pub struct SortKey { pub expr: QueryExpr, pub ascending: bool, pub nulls_first: bool, @@ -505,7 +505,7 @@ pub enum GroupSide { /// part of it — the box is what makes the recursive type's size finite there. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct Predicate(pub Rc>); +pub struct Predicate(pub Rc>); /// Whether any entry of an `Aggregate.filters` vector is set — the shape /// no binding rule accepts yet (issue #466): a filtered measure stays @@ -517,7 +517,7 @@ pub fn any_measure_filtered(filters: &[Option>]) -> bo /// One item in a SELECT projection list. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct ProjectItem { +pub struct ProjectItem { pub alias: Option, pub expr: QueryExpr, } @@ -531,7 +531,7 @@ pub struct ProjectItem { /// whether a grouping-key list happens to be empty or from a neighboring /// node's shape. See design proposal #165. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub enum Reduction { +pub enum Reduction { /// Collapses input rows via `by` — `by`/`without` semantics are exactly /// [`GroupKeys`]'s. May still collapse every row into one (an empty, /// non-`without` `by`) — that's a genuine reduction with zero grouping @@ -623,7 +623,7 @@ impl Reduction { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(deny_unknown_fields)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct ConcatDiscriminatorKey { +pub struct ConcatDiscriminatorKey { discriminator: C, inner_key: Vec, } @@ -650,9 +650,9 @@ impl ConcatDiscriminatorKey { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub enum QueryExpr { +pub enum QueryExpr { /// Outermost leaf. `schema` is the **binding schema** — the resolved column - /// set every positional `FieldId` in the tree indexes into, *not* a full + /// set every positional `ColumnId` in the tree indexes into, *not* a full /// description of the runtime row — once bound (`schema: Schema`, always /// present: the [`SchemaResolver`](super::schema_resolver) is total). Before binding, a /// front-end-emitted `Scan` (`C = ColumnRef`) knows it only when the front @@ -959,7 +959,7 @@ pub enum QueryExpr { // has to get right structurally anyway (a `Filter` is never built with an // operator sub-DAG as its `pred`). /// A column reference — unresolved [`ColumnRef`] (front-end-emitted, `C = - /// ColumnRef`) or positional [`FieldId`] (once bound, `C = FieldId`). + /// ColumnRef`) or positional [`ColumnId`] (once bound, `C = ColumnId`). Column(C), /// A constant literal value. Literal(ScalarValue), @@ -1147,11 +1147,11 @@ impl QueryExpr { } /// The canonical, positional, resolved tree — what the bare `QueryExpr` name -/// has always meant (the default `C = FieldId`). Every existing consumer +/// has always meant (the default `C = ColumnId`). Every existing consumer /// keeps using `QueryExpr` unparameterized; this alias exists only to name /// the resolved state explicitly at a use site that also wants to name /// [`UnresolvedQueryExpr`] nearby. -pub type ResolvedQueryExpr = QueryExpr; +pub type ResolvedQueryExpr = QueryExpr; /// The front-end-emitted, name-based, unresolved DAG — /// `QueryExpr`: front ends construct this directly during their @@ -1164,7 +1164,7 @@ pub type UnresolvedQueryExpr = QueryExpr; // only on the resolved instantiation, not `impl QueryExpr`. // Same reasoning as `AggIntent`'s `output_column`/`requires`/`is_per_series` // (#205): a schema-shaped property that is only meaningful post-binding. -impl QueryExpr { +impl QueryExpr { /// Infer a scalar expression against its input relation using the same /// canonical rules as projection schema derivation. pub fn scalar_type(&self, input: &Schema) -> Result<(DataType, bool), QueryExprError> { @@ -1668,7 +1668,7 @@ pub fn aggregate_output_schema( /// so the schema can't freeze to closed and claims no unique key (issue #39). fn without_output_schema( in_schema: &Schema, - excluded: &[FieldId], + excluded: &[ColumnId], measures: &[AggIntent], output_names: &[String], ) -> Result { @@ -1734,7 +1734,7 @@ fn without_output_schema( /// must be one of the scalar variants (issue #205) — an operator variant here /// is a construction bug, not a shape this needs to handle silently. fn infer_expr_type( - expr: &QueryExpr, + expr: &QueryExpr, schema: &Schema, ) -> Result<(DataType, bool), QueryExprError> { Ok(match expr { @@ -1852,7 +1852,7 @@ fn infer_expr_type( /// Default output-column name for a projection item with no explicit alias: /// a bare column keeps its (schema) name; anything else gets `col_{i}`. -fn default_proj_name(expr: &QueryExpr, idx: usize, schema: &Schema) -> String { +fn default_proj_name(expr: &QueryExpr, idx: usize, schema: &Schema) -> String { match expr { QueryExpr::Column(id) => schema .fields @@ -1922,7 +1922,11 @@ mod tests { ); } - fn scan(columns: Vec, time_index: Option, uk: Vec>) -> QueryExpr { + fn scan( + columns: Vec, + time_index: Option, + uk: Vec>, + ) -> QueryExpr { QueryExpr::Scan { source: Source::Table { table_ref: "t".into(), @@ -2634,7 +2638,7 @@ mod tests { /// different DAG position. `as_promql_scalar` is the round-trip inverse. #[test] fn promql_scalar_bridges_a_literal_float_at_an_operator_position() { - let bridge = QueryExpr::::promql_scalar(2.5); + let bridge = QueryExpr::::promql_scalar(2.5); assert_eq!( bridge, QueryExpr::PromqlScalarBridge(Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.5)))) @@ -2645,7 +2649,7 @@ mod tests { // its native (unwrapped, no row schema) scalar-sub-language position — // no longer a different variant, just not bridged to this DAG // position. - let sql_literal = QueryExpr::::Literal(ScalarValue::Float64(2.5)); + let sql_literal = QueryExpr::::Literal(ScalarValue::Float64(2.5)); assert_eq!(bridge.as_promql_scalar(), Some(2.5)); assert_ne!( bridge, sql_literal, @@ -2666,7 +2670,7 @@ mod tests { /// duplicate variants was used, only by whether the wrapper is present. #[test] fn row_schema_rides_on_the_bridge_wrapper_not_the_literal_variant() { - let bridged = QueryExpr::::promql_scalar(42.0); + let bridged = QueryExpr::::promql_scalar(42.0); let schema = bridged.output_schema().expect("bridge has a row schema"); assert_eq!(schema.fields.len(), 1); assert_eq!(schema.fields[0].name, "value"); @@ -2677,7 +2681,7 @@ mod tests { // `Compare`/`Arithmetic` operand would occupy) has no row schema of // its own — it's a construction bug to call `output_schema` on it // directly, caught as `ScalarHasNoRowSchema` rather than panicking. - let bare = QueryExpr::::Literal(ScalarValue::Float64(42.0)); + let bare = QueryExpr::::Literal(ScalarValue::Float64(42.0)); assert!(matches!( bare.output_schema(), Err(QueryExprError::ScalarHasNoRowSchema) diff --git a/crates/types/src/pre_asap/resolve.rs b/crates/types/src/pre_asap/resolve.rs index 3297767a1..44411f38f 100644 --- a/crates/types/src/pre_asap/resolve.rs +++ b/crates/types/src/pre_asap/resolve.rs @@ -1,5 +1,5 @@ //! Resolve a front-end-emitted, unresolved [`UnresolvedQueryExpr`] (`QueryExpr`) -//! into the canonical, positional [`ResolvedQueryExpr`] (`QueryExpr`). +//! into the canonical, positional [`ResolvedQueryExpr`] (`QueryExpr`). //! //! Both front ends (`asap-frontend-promql`, `asap-frontend-sql`) construct //! canonical `QueryExpr` shapes directly during their own `interpret` step @@ -10,9 +10,9 @@ //! "mechanical, schema-dependent substitution" #179 describes: a single //! generic, shape-preserving walk — every [`UnresolvedQueryExpr`] variant maps to the //! identical [`ResolvedQueryExpr`] variant — that resolves every [`ColumnRef`] to -//! the [`SchemaResolver`](super::schema_resolver::SchemaResolver)-computed positional [`FieldId`]. +//! the [`SchemaResolver`](super::schema_resolver::SchemaResolver)-computed positional [`ColumnId`]. //! -//! ## Why positional `FieldId`, not just carrying names all the way through (issue #216) +//! ## Why positional `ColumnId`, not just carrying names all the way through (issue #216) //! //! A mature query engine can legitimately choose either design — DataFusion's //! own logical plan (what `asap-frontend-sql` walks to build its `QueryExpr`) @@ -28,7 +28,7 @@ //! (`crates/frontend-sql/tests/sql_lowering.rs`) exists specifically because //! `metrics.service` and `hosts.service` are both just `"service"` once their //! schemas are concatenated. A bare name is ambiguous the moment two sources -//! share one; `FieldId` is what makes "the second `service`, position 4, not +//! share one; `ColumnId` is what makes "the second `service`, position 4, not //! the first" a fact recorded once, instead of a lookup redone at every use site. //! 2. **A name's meaning changes going up the DAG.** `Project` renames/aliases, //! `Aggregate` collapses columns and introduces synthetic ones, `Join` @@ -38,7 +38,7 @@ //! pins each reference to "this exact column of this exact node's //! already-derived output schema," so nothing downstream re-derives that scope. //! 3. **It concentrates scoping logic in one place instead of ~6.** Every -//! downstream pass just compares/indexes `FieldId`s — O(1), unambiguous. If +//! downstream pass just compares/indexes `ColumnId`s — O(1), unambiguous. If //! they worked on names instead, each would need its own qualifier-aware, //! join-collision-aware name resolver, or risk silently binding to the wrong //! `"service"`. @@ -61,7 +61,7 @@ use super::query_expr::{ aggregate_output_schema, any_measure_filtered, ConcatDiscriminatorKey, GroupKeys, Predicate, ProjectItem, QueryExprError, Reduction, ResolvedQueryExpr, SortKey, UnresolvedQueryExpr, }; -use super::schema::{FieldId, Schema}; +use super::schema::{ColumnId, Schema}; use super::schema_resolver::SchemaResolver; /// Errors from resolving a canonical, unresolved [`UnresolvedQueryExpr`] DAG. @@ -77,7 +77,7 @@ pub enum ResolveDAGError { } /// Resolve a whole [`UnresolvedQueryExpr`] tree rooted at `tree` into canonical -/// [`ResolvedQueryExpr`]: binds every `ColumnRef` to a `FieldId` via the +/// [`ResolvedQueryExpr`]: binds every `ColumnRef` to a `ColumnId` via the /// [`SchemaResolver`], then [`canonicalize`](super::canonicalize::canonicalize)s the /// result. pub fn resolve_root(dag: &UnresolvedQueryExpr) -> Result { @@ -474,11 +474,11 @@ fn inherited_names(schema: &Schema) -> Vec { } /// Resolve a name-based [`GroupKeys`] into positional -/// [`GroupKeys`], preserving its `by`/`without` mode. +/// [`GroupKeys`], preserving its `by`/`without` mode. fn resolve_group_keys( keys: &GroupKeys, schema: &Schema, -) -> Result, ResolveError> { +) -> Result, ResolveError> { let ids = resolve_column_refs(keys.keys(), schema)?; Ok(if keys.is_without() { GroupKeys::without(ids) @@ -488,7 +488,7 @@ fn resolve_group_keys( } /// Resolve a name-based [`Reduction`] into positional -/// [`Reduction`]. +/// [`Reduction`]. /// /// Uses [`resolve_group_keys_promql`] rather than the strict /// [`resolve_group_keys`], unlike every other group-key site in `resolve` @@ -506,7 +506,7 @@ fn resolve_group_keys( fn resolve_reduction( reduction: &Reduction, schema: &Schema, -) -> Result, ResolveError> { +) -> Result, ResolveError> { Ok(match reduction { Reduction::Reduce(by) => { let ids = resolve_group_keys_promql(by.keys(), schema)?; @@ -521,14 +521,14 @@ fn resolve_reduction( } /// Resolve a name-based [`AggIntent`] into positional -/// [`AggIntent`] — every `col: Option` resolves to -/// `Option` (`None` stays `None`, the sample-value convention); +/// [`AggIntent`] — every `col: Option` resolves to +/// `Option` (`None` stays `None`, the sample-value convention); /// every other field carries straight through unchanged. fn resolve_agg_intent( intent: &AggIntent, schema: &Schema, -) -> Result, ResolveError> { - let col = |c: &Option| -> Result, ResolveError> { +) -> Result, ResolveError> { + let col = |c: &Option| -> Result, ResolveError> { c.as_ref() .map(|r| resolve_column_ref(r, schema)) .transpose() @@ -821,7 +821,7 @@ mod tests { /// SchemaResolver's fallback schema wouldn't contain `phi` at all, and this /// `resolve_column_ref` call would fail `NotFound` for a column the /// caller correctly named. It must resolve cleanly, and the resolved - /// `ConcatDiscriminatorKey` must carry the *positional* `FieldId`s of + /// `ConcatDiscriminatorKey` must carry the *positional* `ColumnId`s of /// the branch's own (usage-derived) schema. #[test] fn resolve_root_seeds_and_resolves_an_otherwise_unreferenced_discriminator_column() { diff --git a/crates/types/src/pre_asap/schema.rs b/crates/types/src/pre_asap/schema.rs index 55220dcb4..77c8ebe92 100644 --- a/crates/types/src/pre_asap/schema.rs +++ b/crates/types/src/pre_asap/schema.rs @@ -20,12 +20,14 @@ use crate::post_asap::sketch::{ StatModelKind, StatModelParams, WaveletKind, WaveletParams, }; -/// Index into [`Schema::fields`] used everywhere a column position is -/// referenced (group-by keys, unique-key sets, the time axis index). +/// Zero-based position of a column in a particular operator's input or output. /// -/// Kept as a named type so downstream code can pattern on the intent ("this -/// is a column position, not just any number"). -pub type FieldId = usize; +/// During planning, the position indexes [`Schema::fields`] to obtain metadata; +/// during execution, it identifies the corresponding value in each input row. +/// Expressions, grouping keys, and schema key/time metadata use the same position. +/// It is local to that schema, not a stable field identity across projections or +/// joins, and does not own data or prescribe a row/column-oriented storage layout. +pub type ColumnId = usize; /// One field of a [`Schema`]: `name + dtype + nullable`, plus an optional /// table qualifier. The struct describes a column and holds none of its data. @@ -212,7 +214,7 @@ pub enum DataType { /// Per-edge schema. Flowing between any two operators, on every node's /// input and output. /// -/// `unique_keys` is metadata for reuse-aware planning: each inner `Vec` +/// `unique_keys` is metadata for reuse-aware planning: each inner `Vec` /// is a set of column indices that together uniquely identify rows. The /// outer `Vec` allows multiple unique-key sets (primary key + another /// unique constraint). Populated by per-node input/output spec — @@ -226,12 +228,12 @@ pub struct Schema { /// Index into `fields` for the time axis, if any. PromQL leaves /// always carry one; SQL leaves may or may not. #[serde(default)] - pub time_index: Option, + pub time_index: Option, /// Unique-key sets — each inner vec is a tuple of column indices /// that together uniquely identifies a row. Empty `Vec` means /// "no provable unique constraint" (the conservative default). #[serde(default)] - pub unique_keys: Vec>, + pub unique_keys: Vec>, /// Whether this schema **completely enumerates** the columns at this point. /// /// - `true` (**closed**): there are no columns beyond these — a catalog-backed @@ -265,9 +267,9 @@ struct SchemaWire { fields: Option>, columns: Option>, #[serde(default)] - time_index: Option, + time_index: Option, #[serde(default)] - unique_keys: Vec>, + unique_keys: Vec>, closed: Option, } @@ -426,8 +428,8 @@ impl Schema { /// inferred unique keys (e.g. PromQL leaves: `[time_index, label_set]`). pub fn with_time_index( fields: Vec, - time_index: FieldId, - unique_keys: Vec>, + time_index: ColumnId, + unique_keys: Vec>, ) -> Self { Self { fields, @@ -440,7 +442,7 @@ impl Schema { /// The schema of a summary-planning node: `fields` and a time axis, no /// unique-key claim, closed. The shape every post-ASAP operator output /// carried before pre- and post-ASAP schemas were one type. - pub fn lifted(fields: Vec, time_index: Option) -> Self { + pub fn lifted(fields: Vec, time_index: Option) -> Self { Self { fields, time_index, @@ -455,14 +457,14 @@ impl Schema { } /// Look up a field by name (first match). `None` if not present. - pub fn column_id(&self, name: &str) -> Option { + pub fn column_id(&self, name: &str) -> Option { self.fields.iter().position(|c| c.name == name) } /// Look up a field by `(table, name)` qualifier — disambiguates columns /// that share a `name` across a join (`a.k` vs `b.k`). `None` if no field /// has both that qualifier and name. - pub fn column_id_qualified(&self, table: &str, name: &str) -> Option { + pub fn column_id_qualified(&self, table: &str, name: &str) -> Option { self.fields .iter() .position(|c| c.name == name && c.table.as_deref() == Some(table)) @@ -478,7 +480,7 @@ impl Schema { /// Append `cols` as an additional unique-key set if not already present. /// Used by `Dedup { cols }`: "the input schema with `unique_keys` /// tightened to include `cols`". - pub fn add_unique_key(&mut self, cols: Vec) { + pub fn add_unique_key(&mut self, cols: Vec) { if !self.unique_keys.contains(&cols) { self.unique_keys.push(cols); } diff --git a/crates/types/src/pre_asap/schema_resolver.rs b/crates/types/src/pre_asap/schema_resolver.rs index 6e84f954d..e051bc2c6 100644 --- a/crates/types/src/pre_asap/schema_resolver.rs +++ b/crates/types/src/pre_asap/schema_resolver.rs @@ -1,7 +1,7 @@ //! The **SchemaResolver** — name resolution as an explicit pass. //! //! [`SchemaResolver::resolve_schema`] produces the complete, self-contained [`Schema`] every -//! `FieldId` in the canonical tree indexes into. [`resolve`](super::resolve) +//! `ColumnId` in the canonical tree indexes into. [`resolve`](super::resolve) //! then becomes purely structural: it threads the SchemaResolver's schema and //! positional resolution downstream is **total**. //! @@ -70,7 +70,7 @@ impl SchemaResolver { /// /// Contains the time axis, the synthetic `value` column, and one column /// per distinct name referenced anywhere in the tree — so positional - /// `FieldId` resolution downstream is total. + /// `ColumnId` resolution downstream is total. pub fn resolve_schema(&self, tree: &UnresolvedQueryExpr) -> Schema { self.resolve_schema_with_inherited(tree, &[]) } diff --git a/docs/design_docs/decisions/concat-unique-keys.md b/docs/design_docs/decisions/concat-unique-keys.md index 93164e8d1..9c0148abf 100644 --- a/docs/design_docs/decisions/concat-unique-keys.md +++ b/docs/design_docs/decisions/concat-unique-keys.md @@ -123,7 +123,7 @@ for why it's fine to ship unused. adds `(discriminator, inner_key)` as the sole unique key, trusting the caller's claim without checking it. - `resolve.rs`'s `Concat` arm resolves a pre-bind (`ColumnRef`) discriminator - key into its post-bind (`FieldId`) equivalent against the first resolved + key into its post-bind (`ColumnId`) equivalent against the first resolved branch's own output schema — the same schema `output_schema()` derives the merged shape from — so the feature works correctly end-to-end for a future caller upstream of `resolve_root`, even though no such caller exists yet. @@ -243,9 +243,9 @@ accuracy issue. All three are fixed on the same PR: the `Concat` arm now pushes `key.discriminator()` and every `key.inner_key()` column into the walk, mirroring `Dedup.cols` exactly. -2. **Resolved `FieldId`s in `discriminator_unique_key` could go stale after +2. **Resolved `ColumnId`s in `discriminator_unique_key` could go stale after `canonicalize()` runs.** `resolve_root_with_inherited` calls `resolve()` - first — which resolves the key's `ColumnRef`s into `FieldId`s against + first — which resolves the key's `ColumnRef`s into `ColumnId`s against `children.first()`'s output schema *as it stood at that point* — then `canonicalize()` runs afterward and can restructure that same first branch: `try_promote_heavy_hitter` and `try_rewrite_rownumber_topk` both @@ -253,7 +253,7 @@ accuracy issue. All three are fixed on the same PR: differently-shaped `Aggregate`, anywhere within the branch (not only at its own top level — the walk is recursive), potentially changing its column count/order. `output_schema()` read the previously-resolved - `FieldId`s with no consistency check, so a future branch matching one of + `ColumnId`s with no consistency check, so a future branch matching one of these rewrite triggers could silently produce a wrong `unique_keys` claim — a wrong query answer, not a missed optimization (per `cse.rs`'s own module doc). Fixed in `canon()` (`canonicalize.rs`): before recursing @@ -264,7 +264,7 @@ accuracy issue. All three are fixed on the same PR: (`Schema` is `PartialEq`/`Eq`); any difference at all — not just a column-count/type change, since a same-shaped-but-different schema is just as unsafe to trust positionally — drops the key (`None`) rather - than risk keeping a `FieldId` that now points at the wrong column or is + than risk keeping a `ColumnId` that now points at the wrong column or is out of bounds. The key is never *re-derived* by guessing at name or position: the two rewrites don't preserve column identity in a way that's safe to infer, so dropping is the only sound outcome once the diff --git a/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md b/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md index 613acf8e9..520501731 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md +++ b/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md @@ -665,7 +665,7 @@ per-series intents such as exact quantile, cardinality, and Top-K aggregate intents remain unavailable until they have an explicit physical algorithm. Hash-join lowering also uses the bound left and right output schemas to prove that every equality -compares one column from each side; same-side or out-of-range `FieldId`s fail +compares one column from each side; same-side or out-of-range `ColumnId`s fail closed. An `Rc` address is not physical identity. Every logical occurrence diff --git a/docs/design_docs/proposals/decoupling_op_and_expr.md b/docs/design_docs/proposals/decoupling_op_and_expr.md index 66c3a69e1..fe458bb80 100644 --- a/docs/design_docs/proposals/decoupling_op_and_expr.md +++ b/docs/design_docs/proposals/decoupling_op_and_expr.md @@ -58,7 +58,7 @@ operator inputs and scalar query-result references use `Rc`. `NonASAPOp` is the payload of an ordinary operator, not a second DAG-node type. `BinaryOp` likewise uses the single `BinaryOperator` payload specified there. -Names are resolved to `FieldId` before constructing these nodes. Parsing and +Names are resolved to `ColumnId` before constructing these nodes. Parsing and unresolved `ColumnRef` handling remain frontend concerns; no alternative generic operator definition is proposed here. These wrappers belong to operator fields and use the `ScalarExpr` defined in §2.2: @@ -100,12 +100,12 @@ time of that whole input. Existing signed offsets and `AtModifier` anchors remai Scalar recursion uses owned `Box` and `Vec` children. The only plan references are explicit operations that consume a query result to compute a value. Those edges remain visible to plan traversal and costing; they cannot hide a separate plan. -The definitions below use resolved `FieldId`s and the common `OperatorNode`; +The definitions below use resolved `ColumnId`s and the common `OperatorNode`; there is no separate pre-ASAP scalar representation. ```rust enum ScalarExpr { - Column(FieldId), + Column(ColumnId), Literal(ScalarValue), Negative { expr: Box, semantics: ExprSemantics }, Compare { diff --git a/docs/design_docs/proposals/operator-sharing.md b/docs/design_docs/proposals/operator-sharing.md index 6e1df70bb..884226238 100644 --- a/docs/design_docs/proposals/operator-sharing.md +++ b/docs/design_docs/proposals/operator-sharing.md @@ -89,7 +89,7 @@ introduce state construction and readout. `CurrentTimestamp`, `EvalTimestamp` and `PromqlScalarFromVector` belong to `ScalarExpr`, defined in the [companion proposal](decoupling_op_and_expr.md#22-scalar-expressions). -A constant needs no bridge operator. The sketches use resolved `FieldId`s and +A constant needs no bridge operator. The sketches use resolved `ColumnId`s and `Schema`; name resolution precedes construction of these nodes. `NonASAPOp` retains the query semantics needed before and after optimization: @@ -113,7 +113,7 @@ enum NonASAPOp { Concat { children: Vec>, discriminator_unique_key: Option, }, - Dedup { child: Rc, cols: Vec }, + Dedup { child: Rc, cols: Vec }, Sort { child: Rc, keys: Vec, partition_by: GroupKeys }, Limit { child: Rc, n: Option, offset: usize, partition_by: GroupKeys }, BinaryOp { @@ -168,9 +168,9 @@ enum ASAPOp { // Reserved operations; semantics and support require further design. SummaryMerge { children: Vec> }, SummarySubtract { left: Rc, right: Rc }, - SummaryDelete { summary_input: Rc, key: FieldId }, + SummaryDelete { summary_input: Rc, key: ColumnId }, SummaryJoin { - outer: Rc, inner: Rc, key: FieldId, family: FieldDataType, + outer: Rc, inner: Rc, key: ColumnId, family: FieldDataType, }, Extension { child: Rc, name: String }, } @@ -255,7 +255,7 @@ Scan node: OperatorNode ``` This is abbreviated structural notation: `Column` and `Literal` above are -`ScalarExpr` variants; column names stand for resolved `FieldId`s. The arithmetic +`ScalarExpr` variants; column names stand for resolved `ColumnId`s. The arithmetic and comparison use `ExprSemantics::Sql`. The aggregate has no grouping keys and names its output `sum_bytes`; the projection names its output `total_bytes`. @@ -351,7 +351,7 @@ enum ExecutionTiming { `ResultGuarantee` retains its existing definition. `Operator`, `OperatorNode`, `OperatorResultKind` and the common node layout are proposed; `Schema` is unified as specified below. This is a resolved-plan interface: name resolution must finish -before producing these concrete `FieldId`/`Schema` nodes. +before producing these concrete `ColumnId`/`Schema` nodes. | Plan stage | Required property state | |---|---| @@ -381,8 +381,8 @@ struct Field { struct Schema { fields: Vec, - time_index: Option, - unique_keys: Vec>, + time_index: Option, + unique_keys: Vec>, closed: bool, } diff --git a/docs/develop_docs/pre-asap-ir.md b/docs/develop_docs/pre-asap-ir.md index 2d1e08b7a..db5681329 100644 --- a/docs/develop_docs/pre-asap-ir.md +++ b/docs/develop_docs/pre-asap-ir.md @@ -15,6 +15,27 @@ Only operations that are semantically relevant to answering the query and select The pre-ASAP IR is defined using the `QueryExpr` enum. We discuss some of important enum types below. +## Fields and column references + +`Schema` owns `Field` metadata: name, type, nullability, and an optional table +qualifier. A `Field` contains no runtime values. The former schema `Column` +struct served this same metadata role; it was renamed to `Field`, not retained +as a second data container. + +`ColumnRef` is an unresolved logical reference (`Named`, `Qualified`, +`SampleValue`, or `Wildcard`). Resolution binds a reference to `ColumnId`, a +`usize` position within a particular schema. `QueryExpr::Column(ColumnId)` +reads that column; the same position indexes `Schema::fields` for type checking +and a runtime row for its value. Group keys, unique keys, and `time_index` also +use these column positions. They are not stable identities across projections +or joins, so the positional reference remains `ColumnId`, not `FieldId`. + +The native runtime currently stores `Batch { schema, rows: Vec> }`. +It has no physical `Column`/array container. A column reference expresses what +to read independently of whether an executor stores its data as rows or arrays. +For example, resolving `t.bytes` to `ColumnId = 1` obtains its type from +`schema.fields[1]`; native execution reads `row[1]`. + ## Node index Grouped to match the sections below — common relational nodes first, then the nodes specific @@ -65,7 +86,7 @@ list of aggregate intents (`measures`). - **`Reduce(GroupKeys)`** — a cross-row reduction: group by some columns, or group by every column *except* some listed ones. `GroupKeys` holds positional column references - (`FieldId`s — indexes into the input schema, not column names) and carries a `by`/`without` + (`ColumnId`s — indexes into the input schema, not column names) and carries a `by`/`without` flag, not just a plain list: - `by(keys)` — group by exactly these columns (SQL `GROUP BY`, PromQL `by(...)`). - `without(keys)` — group by every column *except* these (PromQL `without(...)`); the From 93ed4976380ffef05f15a99508d839322501a140 Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Fri, 2 Oct 2026 19:33:45 +0000 Subject: [PATCH 5/7] docs(schema): clarify fields and column references after DAG rebase --- crates/types/src/pre_asap/expr_ir.rs | 8 +++--- crates/types/src/pre_asap/schema_resolver.rs | 2 +- .../proposals/decoupling_op_and_expr.md | 6 +++++ .../design_docs/proposals/operator-sharing.md | 25 +++++++++++++++++++ 4 files changed, 36 insertions(+), 5 deletions(-) diff --git a/crates/types/src/pre_asap/expr_ir.rs b/crates/types/src/pre_asap/expr_ir.rs index 56330530c..21ffe21bf 100644 --- a/crates/types/src/pre_asap/expr_ir.rs +++ b/crates/types/src/pre_asap/expr_ir.rs @@ -1,8 +1,8 @@ -//! Field-reference and scalar-operator vocabulary shared by the whole -//! canonical [`QueryExpr`](super::query_expr::QueryExpr) tree. +//! Column-reference and scalar-operator vocabulary shared by the whole +//! canonical [`QueryExpr`](super::query_expr::QueryExpr) DAG. //! -//! Issue #205: the scalar expression shapes (`Field`/`Literal`/`Compare`/…) -//! used to live in a separate, self-recursive `Expr` tree here, reachable +//! Issue #205: the scalar expression shapes (`Column`/`Literal`/`Compare`/…) +//! used to live in a separate, self-recursive `Expr` DAG here, reachable //! from `QueryExpr` only through wrapper fields (`Predicate`, `ProjectItem`, //! `SortKey`). They're variants of `QueryExpr` itself now — one recursive //! DAG, not two type families joined by wrappers — generic over the same diff --git a/crates/types/src/pre_asap/schema_resolver.rs b/crates/types/src/pre_asap/schema_resolver.rs index e051bc2c6..d35bdd6b1 100644 --- a/crates/types/src/pre_asap/schema_resolver.rs +++ b/crates/types/src/pre_asap/schema_resolver.rs @@ -86,7 +86,7 @@ impl SchemaResolver { dag: &UnresolvedQueryExpr, inherited: &[String], ) -> Schema { - let mut columns: Vec = leftmost_scan_name(tree) + let mut columns: Vec = leftmost_scan_name(dag) .and_then(|name| self.catalog.columns_for(name)) .unwrap_or_else(default_leaf_columns); diff --git a/docs/design_docs/proposals/decoupling_op_and_expr.md b/docs/design_docs/proposals/decoupling_op_and_expr.md index fe458bb80..f87456a9f 100644 --- a/docs/design_docs/proposals/decoupling_op_and_expr.md +++ b/docs/design_docs/proposals/decoupling_op_and_expr.md @@ -58,6 +58,12 @@ operator inputs and scalar query-result references use `Rc`. `NonASAPOp` is the payload of an ordinary operator, not a second DAG-node type. `BinaryOp` likewise uses the single `BinaryOperator` payload specified there. +`Field` describes schema metadata; `ColumnRef` and `ColumnId` identify columns +read by expressions. Keep `ScalarExpr::Column(ColumnId)`: the ID selects a field +for type checking and the corresponding input value for evaluation, independently +of the executor's row/column storage layout. See the +[fields versus column references contract](operator-sharing.md#21-one-schema-model-for-values-and-state). + Names are resolved to `ColumnId` before constructing these nodes. Parsing and unresolved `ColumnRef` handling remain frontend concerns; no alternative generic operator definition is proposed here. These wrappers belong to operator fields diff --git a/docs/design_docs/proposals/operator-sharing.md b/docs/design_docs/proposals/operator-sharing.md index 884226238..3420c7d58 100644 --- a/docs/design_docs/proposals/operator-sharing.md +++ b/docs/design_docs/proposals/operator-sharing.md @@ -420,6 +420,31 @@ impl ScalarExpr { } ``` +**Fields versus column references.** These names describe different roles, not +competing representations of the same object: + +| Name | Role | Holds runtime values? | +|---|---|---| +| `Schema` | Ordered `Field` metadata, plus key/time/closedness information | No | +| `Field` | Name, type, nullability and optional qualifier for one output column | No | +| `ColumnRef` | Unresolved logical reference: `Named`, `Qualified`, `SampleValue`, or `Wildcard` | No | +| `ColumnId = usize` | Resolved column position in a particular input/output schema | No | +| Runtime batch | Values conforming to a schema; storage layout is executor-specific | Yes | + +Keep `ColumnRef`, `ColumnId`, and `ScalarExpr::Column(ColumnId)`. Renaming the +metadata struct `Column` to `Field` does not rename column references to field +references. The same position identifies metadata during planning and values +during execution; it is not a stable field identity across projections or joins. +Schema `unique_keys` and `time_index` also use these column positions. + +For example, resolving `t.bytes` to position `1` produces `ColumnId = 1`. +`schema.fields[1]` supplies its type and nullability; evaluating +`ScalarExpr::Column(1)` reads the corresponding value. The native executor +currently reads `row[1]` from `Batch { schema, rows: Vec> }`. A columnar +executor would select array `1` instead. No physical `Column` container is +introduced by the metadata rename, and the old metadata `Column` struct is not +retained as a second type. + **Relationship to current types.** `Field` is today's pre-ASAP `Column` with `dtype` widened from `DataType` to `FieldDataType`. `FieldDataType` is today's `SummaryFamilyType` under a name that also fits its `Plain` case. The proposed common `Schema` replaces From 6154dbccde15d75594d4d313244810e570f92932 Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Fri, 2 Oct 2026 19:46:02 +0000 Subject: [PATCH 6/7] fix(schema): preserve DAG wording and clarify schema import aliases --- crates/asap-aware-mapping/src/cost_model.rs | 4 ++-- crates/asap-aware-mapping/src/replacement.rs | 2 +- .../src/operators/aggregate/temporal.rs | 4 ++-- .../asap-physical-operators/src/operators/common.rs | 4 ++-- .../src/operators/vector_window.rs | 4 ++-- .../src/physical_planner/mod.rs | 5 ++--- .../src/physical_planner/precompute.rs | 12 +++++++----- .../src/runtime/batch_execution.rs | 10 +++++----- crates/asap-physical-operators/src/sources/mod.rs | 4 ++-- crates/asap-physical-operators/src/values.rs | 6 ++++-- .../tests/blocking_resources.rs | 6 +++--- .../tests/current_series_heap.rs | 8 ++++---- .../tests/deployment_computation.rs | 4 ++-- crates/asap-physical-operators/tests/physical_dag.rs | 6 +++--- .../tests/physical_plan_recovery.rs | 4 ++-- .../tests/physical_semantics.rs | 4 ++-- .../asap-physical-operators/tests/plan_properties.rs | 6 +++--- .../tests/precompute_population.rs | 8 ++++---- .../asap-physical-operators/tests/promql_binary.rs | 6 +++--- crates/asap-physical-operators/tests/raw_scan.rs | 4 ++-- .../tests/summary_projection.rs | 6 +++--- crates/integration-tests/tests/kll_pane_execution.rs | 4 ++-- crates/types/src/pre_asap/column_resolution.rs | 2 +- crates/types/src/pre_asap/mod.rs | 2 +- crates/types/src/pre_asap/query_expr.rs | 10 +++++----- crates/types/src/pre_asap/resolve.rs | 2 +- crates/types/src/pre_asap/schema_resolver.rs | 8 ++++---- 27 files changed, 74 insertions(+), 71 deletions(-) diff --git a/crates/asap-aware-mapping/src/cost_model.rs b/crates/asap-aware-mapping/src/cost_model.rs index 8a51aa790..d434efbe9 100644 --- a/crates/asap-aware-mapping/src/cost_model.rs +++ b/crates/asap-aware-mapping/src/cost_model.rs @@ -275,9 +275,9 @@ fn finite_rate(units_per_second: f64) -> Option { /// onto one `Rc` for two or more workload roots. See /// `docs/design_docs/cse-cost-model-decision.md`. pub struct CseCandidate<'a> { - /// The shared pre-ASAP sub_dag itself. + /// The shared pre-ASAP sub-DAG itself. pub sub_dag: &'a QueryExpr, - /// The `SummaryNode` this sub_dag bound to — gives the cost model the + /// The `SummaryNode` this sub-DAG bound to — gives the cost model the /// concrete `FieldDataType`/`(kind, params)` actually at stake, not /// just the pre-ASAP shape. pub bound_summary: &'a SummaryNode, diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 4b766579a..32c6d1123 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -2334,7 +2334,7 @@ fn override_accuracy(intent: &AggIntent, target: &AccuracyTarget) -> AggIntent { out } -/// Wrap an unrewritten pre-ASAP sub_dag, lifting its schema with every column +/// Wrap an unrewritten pre-ASAP sub-DAG, lifting its schema with every column /// `FieldDataType::Plain`. `pub` so a caller can fall back to this /// explicitly — e.g. when `SketchAlgorithmStrategy::replacements()` returns no /// candidate for a target, or a deployment wants to force a node its own diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs index 09c0ac1a8..8c987effa 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -308,7 +308,7 @@ mod tests { values::Batch, }; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as LogicalSchema}, + post_asap::{Field, FieldDataType, Schema as PlannerSchema}, pre_asap::DataType, types::AccuracyTarget, }; @@ -317,7 +317,7 @@ mod tests { // The same window operator must give the same answer in either engine phase. #[test] fn temporal_windows_execute_in_both_phases_and_count_is_integer() { - let schema = Arc::new(LogicalSchema { + let schema = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![ diff --git a/crates/asap-physical-operators/src/operators/common.rs b/crates/asap-physical-operators/src/operators/common.rs index fd237ccc0..930a1000c 100644 --- a/crates/asap-physical-operators/src/operators/common.rs +++ b/crates/asap-physical-operators/src/operators/common.rs @@ -3,7 +3,7 @@ pub(super) fn invalid(message: &str) -> Error { Error::Invalid(message.into()) } pub(super) fn schema(fields: Vec) -> Schema { - Arc::new(LogicalSchema { + Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields, @@ -76,4 +76,4 @@ pub(super) fn key_bytes(key: &[Vec]) -> usize { .map(|part| std::mem::size_of::>() + part.len()) .sum::() } -use planner_types::pre_asap::Schema as LogicalSchema; +use planner_types::pre_asap::Schema as PlannerSchema; diff --git a/crates/asap-physical-operators/src/operators/vector_window.rs b/crates/asap-physical-operators/src/operators/vector_window.rs index 1464fc330..93bb6e9c6 100644 --- a/crates/asap-physical-operators/src/operators/vector_window.rs +++ b/crates/asap-physical-operators/src/operators/vector_window.rs @@ -1,14 +1,14 @@ //! Window bounds are typed input data; aggregation and histogram semantics stay native. use super::*; use planner_types::pre_asap::AggIntent; -use planner_types::pre_asap::Schema as LogicalSchema; +use planner_types::pre_asap::Schema as PlannerSchema; pub(crate) fn matrix_schema() -> Schema { let mut fields = vector_binary::value_schema(false).fields.clone(); fields.insert(1, result_field("timestamp", DataType::Timestamp, false)); fields.push(result_field("window_start", DataType::Timestamp, false)); fields.push(result_field("window_end", DataType::Timestamp, false)); - Arc::new(LogicalSchema { + Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields, diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index c6d2cf0f7..76511b03c 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -978,9 +978,8 @@ fn summary_column(input: &Schema) -> Result { } fn named_column(input: &Schema, column: &ColumnRef) -> Result { let name = match column { - // Executable Schema as LogicalSchema retains column names, not table qualifiers. - // Frontend binding has resolved the qualifier; still reject ambiguous - // names here rather than guessing a join side. + // This summary-update lookup matches column names without qualifiers. + // Reject ambiguous names rather than guessing a join side. ColumnRef::Named(name) | ColumnRef::Qualified { name, .. } => name.as_str(), ColumnRef::SampleValue => "value", _ => { diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index 03a8047f1..9f2a8ccf2 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -2,14 +2,14 @@ use super::promql_rows::SERIES_IDENTITY_COLUMN as SERIES_IDENTITY; use super::*; use planner_types::{ - post_asap::{ExecutionTiming, GroupingStrategy, Schema as LogicalSchema}, + post_asap::{ExecutionTiming, GroupingStrategy, Schema as PlannerSchema}, pre_asap::DataType, }; /// Physical rows carry the population and pane coordinate alongside the logical value. /// These fields preserve identities which are implicit in a stored summary instance. pub fn population_schema(family: FieldDataType) -> Schema { - Arc::new(LogicalSchema { + Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![ @@ -128,7 +128,9 @@ pub fn boundary_schema(node: &PostAsapDAGNode) -> Result { } /// Validate the adapter layout during installed-plan recovery without lowering operators. -pub fn source_schema(logical: &LogicalSchema) -> Result { +/// `PlannerSchema` is an import alias for the shared planner `Schema`; the return +/// type is the runtime `Arc` handle for the population adapter layout. +pub fn source_schema(logical: &PlannerSchema) -> Result { let states = logical .fields .iter() @@ -518,7 +520,7 @@ fn fragment( } let item_columns = (3..fields.len()).collect::>(); let project = Operator::project(input.clone(), columns)?.with_output_schema( - Arc::new(LogicalSchema { + Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields, @@ -572,7 +574,7 @@ fn fragment( /// of the label set less excluded labels. fn raw_items( expr: &SummaryInputExpr, - scan: &LogicalSchema, + scan: &PlannerSchema, items: &mut Vec<(Expression, DataType)>, ) -> Result<(), Error> { // Open PromQL scans need not list every label, so any name that is not diff --git a/crates/asap-physical-operators/src/runtime/batch_execution.rs b/crates/asap-physical-operators/src/runtime/batch_execution.rs index d85b56a9c..3749d11b6 100644 --- a/crates/asap-physical-operators/src/runtime/batch_execution.rs +++ b/crates/asap-physical-operators/src/runtime/batch_execution.rs @@ -88,7 +88,7 @@ mod tests { values::Value, }; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as LogicalSchema}, + post_asap::{Field, FieldDataType, Schema as PlannerSchema}, pre_asap::DataType, }; use std::sync::Arc; @@ -96,7 +96,7 @@ mod tests { // Engine adapters can run the identical native chain from an outer executor. #[test] fn same_native_chain_inside_query_and_ingestion_execution() { - let schema = Arc::new(LogicalSchema { + let schema = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![Field { @@ -142,7 +142,7 @@ mod tests { // Native sources may cross the runtime's cooperative batch quantum. #[test] fn in_memory_source_drives_cooperative_yields() { - let schema = Arc::new(LogicalSchema { + let schema = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![], @@ -164,7 +164,7 @@ mod tests { // An adapter-held output must retain its parent's reservation after execution. #[test] fn returned_batches_keep_their_resource_reservation() { - let schema = Arc::new(LogicalSchema { + let schema = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![], @@ -195,7 +195,7 @@ mod tests { // A cancelled surrounding execution also prevents its native computation. #[test] fn cancellation_is_not_bypassed_by_in_memory_execution() { - let schema = Arc::new(LogicalSchema { + let schema = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![], diff --git a/crates/asap-physical-operators/src/sources/mod.rs b/crates/asap-physical-operators/src/sources/mod.rs index dfe6b1f0d..7f8018f52 100644 --- a/crates/asap-physical-operators/src/sources/mod.rs +++ b/crates/asap-physical-operators/src/sources/mod.rs @@ -8,7 +8,7 @@ use crate::{ }; use futures::{stream, StreamExt}; use planner_types::{ - post_asap::{FieldDataType, Schema as LogicalSchema}, + post_asap::{FieldDataType, Schema as PlannerSchema}, pre_asap::{DataType, QueryExpr, Source}, }; use std::sync::Arc; @@ -51,7 +51,7 @@ impl DataSources { "raw Scan requires a Planner Scan leaf".into(), )); }; - let output = Arc::new(LogicalSchema::lifted( + let output = Arc::new(PlannerSchema::lifted( schema.fields.clone(), schema.time_index, )); diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs index 9476fcd10..591e44d8f 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -2,11 +2,13 @@ use crate::AggregateCore; use crate::Error; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as LogicalSchema}, + post_asap::{Field, FieldDataType, Schema as PlannerSchema}, pre_asap::DataType, }; use std::{cmp::Ordering, sync::Arc}; -pub type Schema = Arc; +/// Shared runtime handle to the same schema metadata used by the planner. +/// This alias changes ownership, not the schema model or its field types. +pub type Schema = Arc; #[derive(Clone, serde::Serialize, serde::Deserialize)] pub enum Value { Null, diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/asap-physical-operators/tests/blocking_resources.rs index 7ba650211..6b5a531eb 100644 --- a/crates/asap-physical-operators/tests/blocking_resources.rs +++ b/crates/asap-physical-operators/tests/blocking_resources.rs @@ -8,13 +8,13 @@ use asap_physical_operators::{ }; use futures::{executor::block_on, FutureExt, StreamExt}; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as LogicalSchema}, + post_asap::{Field, FieldDataType, Schema as PlannerSchema}, pre_asap::{DataType, JoinKind, Predicate, QueryExpr, ScalarValue}, }; use std::sync::Arc; fn schema(width: usize) -> Schema { - Arc::new(LogicalSchema { + Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: (0..width) @@ -189,7 +189,7 @@ fn cooperative_sort_preserves_ties_across_chunks() { #[test] fn weighted_summary_build_yields_within_a_batch() { use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; - let input = Arc::new(LogicalSchema { + let input = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![ diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs index b0b496df9..1701e6586 100644 --- a/crates/asap-physical-operators/tests/current_series_heap.rs +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -9,12 +9,12 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema as LogicalSchema; +use planner_types::pre_asap::Schema as PlannerSchema; use planner_types::{post_asap::*, pre_asap::DataType}; use std::{collections::BTreeMap, sync::Arc}; -fn schema() -> Arc { - Arc::new(LogicalSchema { +fn schema() -> Arc { + Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: [ @@ -169,7 +169,7 @@ fn spatial_heap_ranks_latest_values_in_independent_runs() { }; let family = FieldDataType::Sketch(SketchKind::new(algorithm, params), Default::default()); let build = Operator::keyed_summary_build(schema(), family, 1, vec![3], vec![2]).unwrap(); - let output = Arc::new(LogicalSchema { + let output = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![ diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index 4fd00c1a3..5d6266b31 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -7,7 +7,7 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema as LogicalSchema; +use planner_types::pre_asap::Schema as PlannerSchema; use planner_types::{post_asap::*, pre_asap::QueryExpr, types::AccuracyTarget, workload::*}; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; @@ -77,7 +77,7 @@ fn population_dag(query: &str) -> PostAsapDAG { } /// Raw scan nodes are the frontier; everything above them is compiled. -fn raw_inputs(dag: &PostAsapDAG) -> Vec<(u64, Arc, String)> { +fn raw_inputs(dag: &PostAsapDAG) -> Vec<(u64, Arc, String)> { dag.nodes .iter() .filter_map(|node| match &node.payload { diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 8f138ecf4..29827dd54 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -9,12 +9,12 @@ use asap_physical_operators::{ }; use futures::{executor::block_on, StreamExt}; use planner_types::{ - post_asap::{ExactKind, ExactParams, Field, FieldDataType, Schema as LogicalSchema}, + post_asap::{ExactKind, ExactParams, Field, FieldDataType, Schema as PlannerSchema}, pre_asap::DataType, }; use std::sync::Arc; fn schema(fields: &[(&str, DataType, bool)]) -> Schema { - Arc::new(LogicalSchema { + Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: fields @@ -423,7 +423,7 @@ fn exact_state_and_family_validation() { let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); acc.update(None, 7., 0); - let schema = Arc::new(LogicalSchema { + let schema = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![Field { diff --git a/crates/asap-physical-operators/tests/physical_plan_recovery.rs b/crates/asap-physical-operators/tests/physical_plan_recovery.rs index 3f2956af4..0526e31cb 100644 --- a/crates/asap-physical-operators/tests/physical_plan_recovery.rs +++ b/crates/asap-physical-operators/tests/physical_plan_recovery.rs @@ -5,13 +5,13 @@ use asap_physical_operators::{ physical_planner::{CompiledPhysicalDAG, InputContract}, }; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as LogicalSchema}, + post_asap::{Field, FieldDataType, Schema as PlannerSchema}, pre_asap::DataType, }; use std::{collections::BTreeMap, sync::Arc}; fn sorted() -> CompiledPhysicalDAG { - let schema = Arc::new(LogicalSchema { + let schema = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![Field { diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs index 2813506a6..d082b6578 100644 --- a/crates/asap-physical-operators/tests/physical_semantics.rs +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -10,13 +10,13 @@ use asap_physical_operators::{ }; use futures::{executor::block_on, StreamExt}; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as LogicalSchema}, + post_asap::{Field, FieldDataType, Schema as PlannerSchema}, pre_asap::{CompareOpKind, DataType, JoinKind, Predicate, QueryExpr}, }; use std::{rc::Rc, sync::Arc}; fn schema(fields: &[(&str, DataType, bool)]) -> Schema { - Arc::new(LogicalSchema { + Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: fields diff --git a/crates/asap-physical-operators/tests/plan_properties.rs b/crates/asap-physical-operators/tests/plan_properties.rs index 41cba5424..4b84f4a96 100644 --- a/crates/asap-physical-operators/tests/plan_properties.rs +++ b/crates/asap-physical-operators/tests/plan_properties.rs @@ -8,7 +8,7 @@ use asap_physical_operators::{ Error, }; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as LogicalSchema}, + post_asap::{Field, FieldDataType, Schema as PlannerSchema}, pre_asap::{DataType, QueryExpr, Source}, }; use std::sync::{ @@ -35,7 +35,7 @@ impl RawSource for DeclaredSource { // A blocking parent must reject unknown and unbounded Scan inputs without opening a reader. #[test] fn blocking_inputs_require_an_explicit_finite_source() { - let schema = Arc::new(LogicalSchema { + let schema = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![Field { @@ -69,7 +69,7 @@ fn blocking_inputs_require_an_explicit_finite_source() { let scan = registry .bind(&QueryExpr::Scan { source: identity, - schema: LogicalSchema::new(vec![Field::plain("v", DataType::Int64, false)]), + schema: PlannerSchema::new(vec![Field::plain("v", DataType::Int64, false)]), predicates: vec![], }) .unwrap(); diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs index 59ece54e2..909364abd 100644 --- a/crates/asap-physical-operators/tests/precompute_population.rs +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -8,7 +8,7 @@ use asap_physical_operators::{ Statistic, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema as LogicalSchema; +use planner_types::pre_asap::Schema as PlannerSchema; use planner_types::{ post_asap::*, pre_asap::{ArithmeticOpKind, BinaryOpKind, ColumnRef, DataType, GroupKeys, Reduction}, @@ -19,7 +19,7 @@ use std::{collections::BTreeMap, sync::Arc}; #[test] fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let schema = |dtype| LogicalSchema { + let schema = |dtype| PlannerSchema { closed: true, unique_keys: vec![], fields: vec![Field { @@ -220,8 +220,8 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { } } -fn logical_schema(family: FieldDataType) -> LogicalSchema { - LogicalSchema { +fn logical_schema(family: FieldDataType) -> PlannerSchema { + PlannerSchema { closed: true, unique_keys: vec![], fields: vec![Field { diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs index c620b0342..fc5d97ead 100644 --- a/crates/asap-physical-operators/tests/promql_binary.rs +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -9,14 +9,14 @@ use futures::{executor::block_on, StreamExt}; use planner_types::{ post_asap::{ BinaryOperator, ExecutionDataState, Field, FieldDataType, PostAsapDAGNode, PostAsapNodeId, - PostAsapOperatorPayload, Schema as LogicalSchema, + PostAsapOperatorPayload, Schema as PlannerSchema, }, pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, }; use std::{collections::BTreeMap, sync::Arc}; fn schema() -> Schema { - Arc::new(LogicalSchema { + Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![ @@ -324,7 +324,7 @@ fn stored_series_readouts_support_filters_and_sets() { (ExactKind::Count, ExactParams::Count), ] { let family = FieldDataType::ExactAggregate(exact_kind.clone(), params); - let state_schema = Arc::new(LogicalSchema { + let state_schema = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![ diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs index b95987796..57e6edd4e 100644 --- a/crates/asap-physical-operators/tests/raw_scan.rs +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -6,7 +6,7 @@ use asap_physical_operators::dag::{ Error, Limits, OutputStream, RunContext, Scope, }; use futures::{executor::block_on, stream, StreamExt}; -use planner_types::pre_asap::Schema as LogicalSchema; +use planner_types::pre_asap::Schema as PlannerSchema; use planner_types::{ post_asap::*, pre_asap::{DataType, Field, GroupKeys, Predicate, QueryExpr, Source}, @@ -23,7 +23,7 @@ use std::{ fn fixture() -> (QueryExpr, Schema, Vec) { let schema = planner_types::pre_asap::Schema::new(vec![Field::plain("value", DataType::Int64, true)]); - let output = Arc::new(LogicalSchema { + let output = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![Field { diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs index ee54b3d45..d018fe511 100644 --- a/crates/asap-physical-operators/tests/summary_projection.rs +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -8,7 +8,7 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema as LogicalSchema; +use planner_types::pre_asap::Schema as PlannerSchema; use planner_types::{ post_asap::*, pre_asap::{ColumnRef, DataType, ProjectItem, QueryExpr}, @@ -20,7 +20,7 @@ use std::{collections::BTreeMap, sync::Arc}; #[test] fn post_asap_summary_projection_survives_recovery() { let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let schema = Arc::new(LogicalSchema { + let schema = Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![ @@ -39,7 +39,7 @@ fn post_asap_summary_projection_survives_recovery() { ], time_index: None, }); - let output = LogicalSchema { + let output = PlannerSchema { closed: true, unique_keys: vec![], fields: vec![ diff --git a/crates/integration-tests/tests/kll_pane_execution.rs b/crates/integration-tests/tests/kll_pane_execution.rs index b7f51711c..31fe62dc0 100644 --- a/crates/integration-tests/tests/kll_pane_execution.rs +++ b/crates/integration-tests/tests/kll_pane_execution.rs @@ -11,7 +11,7 @@ use asap_physical_operators::{ }; use asap_types::{ post_asap::{ - Field, FieldDataType, Schema as LogicalSchema, SketchAlgorithm, SketchKind, SketchParams, + Field, FieldDataType, Schema as PlannerSchema, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, }, pre_asap::DataType, @@ -32,7 +32,7 @@ fn family(k: u32) -> FieldDataType { ) } fn raw_schema() -> Schema { - Arc::new(LogicalSchema { + Arc::new(PlannerSchema { closed: true, unique_keys: vec![], fields: vec![Field { diff --git a/crates/types/src/pre_asap/column_resolution.rs b/crates/types/src/pre_asap/column_resolution.rs index f137906dd..cfafbe9e0 100644 --- a/crates/types/src/pre_asap/column_resolution.rs +++ b/crates/types/src/pre_asap/column_resolution.rs @@ -1,7 +1,7 @@ //! Schema-driven column resolution. //! //! Front ends (issue #179) emit `ColumnRef` (name-based, optionally -//! table-qualified); the canonical tree uses positional [`ColumnId`] resolved +//! table-qualified); the canonical DAG uses positional [`ColumnId`] resolved //! against a per-node [`Schema`]. These helpers bridge the two — the //! [`SchemaResolver`](super::schema_resolver) builds the schema, and [`resolve_column_refs`] //! turns name-based refs (group keys, dedup columns) into positional ids, diff --git a/crates/types/src/pre_asap/mod.rs b/crates/types/src/pre_asap/mod.rs index d1c448756..4cfcfbd96 100644 --- a/crates/types/src/pre_asap/mod.rs +++ b/crates/types/src/pre_asap/mod.rs @@ -12,7 +12,7 @@ //! - [`schema`] — the per-edge [`Schema`] every node carries. //! - [`schema_resolver`] / [`column_resolution`] — name resolution: turn a `ColumnRef` //! into a positional `ColumnId` against an in-scope [`Schema`]. -//! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedQueryExpr`] tree to +//! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedQueryExpr`] DAG to //! canonical [`ResolvedQueryExpr`] (issue #179): both front ends //! (`asap-frontend-promql`, `asap-frontend-sql`) construct `UnresolvedQueryExpr` //! directly during their own `interpret` step and call diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs index 79fdd854e..1051157d5 100644 --- a/crates/types/src/pre_asap/query_expr.rs +++ b/crates/types/src/pre_asap/query_expr.rs @@ -6,8 +6,8 @@ //! one parent, within one query or across a `QueryWorkload` batch, instead of //! being duplicated. Nothing in this module produces that sharing on its //! own — construction still allocates a fresh `Rc` per node, the same shape -//! as the old `Box` tree — a separate CSE pass is what turns two -//! independently constructed, structurally-equal subtrees into two +//! as the old `Box` DAG — a separate CSE pass is what turns two +//! independently constructed, structurally-equal sub-DAGs into two //! references to one `Rc` (issue #212, #222). Field identity is //! **positional** (`Aggregate.reduction: Reduction`, wrapping `GroupKeys` //! for the grouped case), resolved by the [`SchemaResolver`](super::schema_resolver) against @@ -23,7 +23,7 @@ use super::agg_intent::AggIntent; use super::expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; use super::schema::{ColumnId, DataType, Field, FieldDataType, Schema}; -/// The column-reference resolution state a [`QueryExpr`] tree carries — +/// The column-reference resolution state a [`QueryExpr`] DAG carries — /// [`ColumnId`] (the default, and what the bare `QueryExpr` name has always /// meant) once the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has resolved every /// reference positionally, or the front-end-emitted, name-based [`ColumnRef`] @@ -652,7 +652,7 @@ impl ConcatDiscriminatorKey { #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] pub enum QueryExpr { /// Outermost leaf. `schema` is the **binding schema** — the resolved column - /// set every positional `ColumnId` in the tree indexes into, *not* a full + /// set every positional `ColumnId` in the DAG indexes into, *not* a full /// description of the runtime row — once bound (`schema: Schema`, always /// present: the [`SchemaResolver`](super::schema_resolver) is total). Before binding, a /// front-end-emitted `Scan` (`C = ColumnRef`) knows it only when the front @@ -1146,7 +1146,7 @@ impl QueryExpr { } } -/// The canonical, positional, resolved tree — what the bare `QueryExpr` name +/// The canonical, positional, resolved DAG — what the bare `QueryExpr` name /// has always meant (the default `C = ColumnId`). Every existing consumer /// keeps using `QueryExpr` unparameterized; this alias exists only to name /// the resolved state explicitly at a use site that also wants to name diff --git a/crates/types/src/pre_asap/resolve.rs b/crates/types/src/pre_asap/resolve.rs index 44411f38f..b4a5c87a6 100644 --- a/crates/types/src/pre_asap/resolve.rs +++ b/crates/types/src/pre_asap/resolve.rs @@ -76,7 +76,7 @@ pub enum ResolveDAGError { Schema(#[from] QueryExprError), } -/// Resolve a whole [`UnresolvedQueryExpr`] tree rooted at `tree` into canonical +/// Resolve a whole [`UnresolvedQueryExpr`] DAG rooted at `dag` into canonical /// [`ResolvedQueryExpr`]: binds every `ColumnRef` to a `ColumnId` via the /// [`SchemaResolver`], then [`canonicalize`](super::canonicalize::canonicalize)s the /// result. diff --git a/crates/types/src/pre_asap/schema_resolver.rs b/crates/types/src/pre_asap/schema_resolver.rs index d35bdd6b1..a9afff2cb 100644 --- a/crates/types/src/pre_asap/schema_resolver.rs +++ b/crates/types/src/pre_asap/schema_resolver.rs @@ -1,7 +1,7 @@ //! The **SchemaResolver** — name resolution as an explicit pass. //! //! [`SchemaResolver::resolve_schema`] produces the complete, self-contained [`Schema`] every -//! `ColumnId` in the canonical tree indexes into. [`resolve`](super::resolve) +//! `ColumnId` in the canonical DAG indexes into. [`resolve`](super::resolve) //! then becomes purely structural: it threads the SchemaResolver's schema and //! positional resolution downstream is **total**. //! @@ -69,10 +69,10 @@ impl SchemaResolver { /// Resolve the complete [`Schema`] in scope for a query rooted at `dag`. /// /// Contains the time axis, the synthetic `value` column, and one column - /// per distinct name referenced anywhere in the tree — so positional + /// per distinct name referenced anywhere in the DAG — so positional /// `ColumnId` resolution downstream is total. - pub fn resolve_schema(&self, tree: &UnresolvedQueryExpr) -> Schema { - self.resolve_schema_with_inherited(tree, &[]) + pub fn resolve_schema(&self, dag: &UnresolvedQueryExpr) -> Schema { + self.resolve_schema_with_inherited(dag, &[]) } /// Like [`resolve_schema`](Self::resolve_schema), but also seeds `inherited` label names that are From 118f491c3498df889fb476155320f226882b8018 Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Fri, 2 Oct 2026 19:50:57 +0000 Subject: [PATCH 7/7] refactor(runtime): name shared schema ownership SchemaRef --- .../src/expressions/mod.rs | 4 +-- .../src/expressions/planner.rs | 6 ++-- .../src/operators/aggregate/mod.rs | 8 ++--- .../src/operators/aggregate/temporal.rs | 4 +-- .../src/operators/aligned_binary.rs | 4 +-- .../src/operators/common.rs | 8 ++--- .../src/operators/current_series.rs | 2 +- .../src/operators/filter.rs | 2 +- .../src/operators/joins/mod.rs | 10 +++--- .../src/operators/limit.rs | 2 +- .../src/operators/mod.rs | 16 ++++----- .../src/operators/projection.rs | 2 +- .../src/operators/scope_timestamp.rs | 2 +- .../src/operators/series_labels.rs | 20 +++++------ .../src/operators/series_window.rs | 2 +- .../src/operators/sort.rs | 2 +- .../src/operators/source.rs | 6 ++-- .../src/operators/summary/mod.rs | 16 +++++---- .../src/operators/unchecked.rs | 4 +-- .../src/operators/vector_binary.rs | 8 ++--- .../src/operators/vector_window.rs | 6 ++-- .../src/physical_planner/compiled.rs | 14 ++++---- .../src/physical_planner/mod.rs | 34 +++++++++---------- .../src/physical_planner/precompute.rs | 26 +++++++------- .../src/physical_planner/promql_fallback.rs | 6 ++-- .../src/physical_planner/promql_rows.rs | 2 +- .../src/physical_planner/promql_values.rs | 12 +++---- .../src/physical_planner/row_values.rs | 2 +- .../src/runtime/batch_execution.rs | 12 +++---- .../src/sources/memory.rs | 6 ++-- .../src/sources/mod.rs | 19 +++++------ crates/asap-physical-operators/src/values.rs | 20 +++++------ .../tests/blocking_resources.rs | 12 +++---- .../tests/current_series_heap.rs | 8 ++--- .../tests/deployment_computation.rs | 4 +-- .../tests/physical_dag.rs | 24 ++++++------- .../tests/physical_plan_recovery.rs | 4 +-- .../tests/physical_semantics.rs | 12 +++---- .../tests/plan_properties.rs | 12 +++---- .../tests/precompute_population.rs | 8 ++--- .../tests/promql_binary.rs | 10 +++--- .../tests/promql_values.rs | 8 ++--- .../asap-physical-operators/tests/raw_scan.rs | 20 +++++------ .../tests/summary_projection.rs | 6 ++-- .../tests/kll_pane_execution.rs | 21 ++++++------ .../summary_maintenance_lifecycle_e2e.rs | 2 +- docs/develop_docs/pre-asap-ir.md | 4 ++- 47 files changed, 222 insertions(+), 220 deletions(-) diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs index 627fd9314..8947743dc 100644 --- a/crates/asap-physical-operators/src/expressions/mod.rs +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -1,6 +1,6 @@ //! Scalar semantics and typed expression binding. Planner expressions enter through CompiledExpression. use crate::{ - values::{plain, Schema, Value}, + values::{plain, SchemaRef, Value}, Error, }; use planner_types::pre_asap::{ArithmeticOpKind, DataType}; @@ -56,7 +56,7 @@ impl Expression { pub fn planner(expression: crate::expressions::CompiledExpression) -> Self { Self::Planner(Box::new(expression)) } - pub(crate) fn dtype(&self, input: &Schema) -> Result<(DataType, bool), Error> { + pub(crate) fn dtype(&self, input: &SchemaRef) -> Result<(DataType, bool), Error> { use Expression::*; match self { Binary { diff --git a/crates/asap-physical-operators/src/expressions/planner.rs b/crates/asap-physical-operators/src/expressions/planner.rs index 0c33e5743..92042fa8e 100644 --- a/crates/asap-physical-operators/src/expressions/planner.rs +++ b/crates/asap-physical-operators/src/expressions/planner.rs @@ -1,6 +1,6 @@ //! Planner scalar expressions evaluated over native typed rows. use crate::{ - values::{Schema, Value}, + values::{SchemaRef, Value}, Error, }; use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind, DataType, QueryExpr, ScalarValue}; @@ -340,7 +340,7 @@ impl CompiledExpression { &self.expression } - pub fn compile(expression: &QueryExpr, input: &Schema) -> Result { + pub fn compile(expression: &QueryExpr, input: &SchemaRef) -> Result { let schema = input .fields .iter() @@ -371,7 +371,7 @@ impl CompiledExpression { pub(crate) fn dtype(&self) -> (DataType, bool) { self.output.clone() } - pub(crate) fn validate_input(&self, input: &Schema) -> Result<(), Error> { + pub(crate) fn validate_input(&self, input: &SchemaRef) -> Result<(), Error> { let checked = Self::compile(&self.expression, input)?; if checked.output != self.output { return Err(Error::Invalid( diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs index 89bd9ec5e..76eccd5ee 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/mod.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -1,7 +1,7 @@ use super::*; impl Operator { pub fn aggregate( - input: Schema, + input: SchemaRef, groups: Vec, measures: Vec<(String, Reduction)>, ) -> Result { @@ -53,7 +53,7 @@ impl Operator { }) } pub fn window( - input: Schema, + input: SchemaRef, intent: planner_types::pre_asap::AggIntent, coordinate: usize, value: usize, @@ -177,7 +177,7 @@ async fn reduce( rows: Vec>, groups: &[usize], measures: &[Reduction], - input: &Schema, + input: &SchemaRef, context: &RunContext, ) -> Result>, Error> { let mut work = Cooperative::new(context); @@ -239,7 +239,7 @@ pub(super) fn quantile(q: f64, mut values: Vec) -> f64 { async fn reduce_one( rows: &[Vec], measure: &Reduction, - input: &Schema, + input: &SchemaRef, work: &mut Cooperative, ) -> Result { let column = match measure { diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs index 8c987effa..e40a1fa56 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -308,7 +308,7 @@ mod tests { values::Batch, }; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as PlannerSchema}, + post_asap::{Field, FieldDataType, Schema}, pre_asap::DataType, types::AccuracyTarget, }; @@ -317,7 +317,7 @@ mod tests { // The same window operator must give the same answer in either engine phase. #[test] fn temporal_windows_execute_in_both_phases_and_count_is_integer() { - let schema = Arc::new(PlannerSchema { + let schema = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![ diff --git a/crates/asap-physical-operators/src/operators/aligned_binary.rs b/crates/asap-physical-operators/src/operators/aligned_binary.rs index 4f58886c4..188d28d6f 100644 --- a/crates/asap-physical-operators/src/operators/aligned_binary.rs +++ b/crates/asap-physical-operators/src/operators/aligned_binary.rs @@ -7,8 +7,8 @@ impl Operator { /// Match every row by the declared identity columns. Unlike an inner join, /// incomplete or duplicate keys are errors: dropping an update changes state. pub fn aligned_binary( - left: Schema, - right: Schema, + left: SchemaRef, + right: SchemaRef, keys: Vec<(usize, usize)>, values: (usize, usize), operator: BinaryOperator, diff --git a/crates/asap-physical-operators/src/operators/common.rs b/crates/asap-physical-operators/src/operators/common.rs index 930a1000c..5ef5023cd 100644 --- a/crates/asap-physical-operators/src/operators/common.rs +++ b/crates/asap-physical-operators/src/operators/common.rs @@ -2,8 +2,8 @@ use super::*; pub(super) fn invalid(message: &str) -> Error { Error::Invalid(message.into()) } -pub(super) fn schema(fields: Vec) -> Schema { - Arc::new(PlannerSchema { +pub(super) fn schema(fields: Vec) -> SchemaRef { + Arc::new(Schema { closed: true, unique_keys: vec![], fields, @@ -19,7 +19,7 @@ pub(super) fn result_field(name: &str, dtype: DataType, nullable: bool) -> Field } } -pub(super) fn validate_groups(input: &Schema, groups: &[usize]) -> Result<(), Error> { +pub(super) fn validate_groups(input: &SchemaRef, groups: &[usize]) -> Result<(), Error> { for &i in groups { plain(input, i)?; } @@ -76,4 +76,4 @@ pub(super) fn key_bytes(key: &[Vec]) -> usize { .map(|part| std::mem::size_of::>() + part.len()) .sum::() } -use planner_types::pre_asap::Schema as PlannerSchema; +use planner_types::pre_asap::Schema; diff --git a/crates/asap-physical-operators/src/operators/current_series.rs b/crates/asap-physical-operators/src/operators/current_series.rs index 937036c52..1dd9a2c03 100644 --- a/crates/asap-physical-operators/src/operators/current_series.rs +++ b/crates/asap-physical-operators/src/operators/current_series.rs @@ -3,7 +3,7 @@ use super::*; impl Operator { pub fn current_series( - input: Schema, + input: SchemaRef, identity: usize, coordinate: usize, value: usize, diff --git a/crates/asap-physical-operators/src/operators/filter.rs b/crates/asap-physical-operators/src/operators/filter.rs index 8862aaccd..84442cf71 100644 --- a/crates/asap-physical-operators/src/operators/filter.rs +++ b/crates/asap-physical-operators/src/operators/filter.rs @@ -1,6 +1,6 @@ use super::*; impl Operator { - pub fn filter(input: Schema, predicate: Expression) -> Result { + pub fn filter(input: SchemaRef, predicate: Expression) -> Result { if predicate.dtype(&input)?.0 != DataType::Bool { return Err(invalid("filter predicate must be boolean")); } diff --git a/crates/asap-physical-operators/src/operators/joins/mod.rs b/crates/asap-physical-operators/src/operators/joins/mod.rs index a43970337..76e2f748c 100644 --- a/crates/asap-physical-operators/src/operators/joins/mod.rs +++ b/crates/asap-physical-operators/src/operators/joins/mod.rs @@ -1,8 +1,8 @@ use super::*; impl Operator { pub fn semi_join( - left: Schema, - right: Schema, + left: SchemaRef, + right: SchemaRef, keys: Vec<(usize, usize)>, ) -> Result { if keys.is_empty() { @@ -42,11 +42,11 @@ impl Operator { } } pub fn relational_join( - left: Schema, - right: Schema, + left: SchemaRef, + right: SchemaRef, kind: planner_types::pre_asap::JoinKind, predicate: &planner_types::pre_asap::Predicate, - output: Schema, + output: SchemaRef, ) -> Result { use planner_types::pre_asap::JoinKind; let mut joined = left.fields.clone(); diff --git a/crates/asap-physical-operators/src/operators/limit.rs b/crates/asap-physical-operators/src/operators/limit.rs index 5f5e601fb..2e788db25 100644 --- a/crates/asap-physical-operators/src/operators/limit.rs +++ b/crates/asap-physical-operators/src/operators/limit.rs @@ -1,6 +1,6 @@ use super::*; impl Operator { - pub fn limit(input: Schema, n: u64, offset: u64, groups: Vec) -> Result { + pub fn limit(input: SchemaRef, n: u64, offset: u64, groups: Vec) -> Result { validate_groups(&input, &groups)?; Ok(Self { kind: Kind::Limit { n, offset, groups }, diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index fd5dac32d..4c7e5f5c6 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -3,7 +3,7 @@ mod aligned_binary; use crate::plan::{Boundedness, Emission, PhysicalOperator, PlanProperties}; use crate::{ runtime::{Cooperative, Input, OutputStream, Reservation, RunContext}, - values::{field, group_key, plain, Batch, Schema, Value}, + values::{field, group_key, plain, Batch, SchemaRef, Value}, Error, }; use futures::StreamExt; @@ -162,8 +162,8 @@ enum Kind { #[serde(try_from = "unchecked::UncheckedOperator")] pub struct Operator { kind: Kind, - inputs: Vec, - output: Schema, + inputs: Vec, + output: SchemaRef, } impl Operator { pub(crate) fn row_preserving_input(&self) -> Option { @@ -237,7 +237,7 @@ impl Operator { Ok(Some((start, end))) } - pub(crate) fn with_output_schema(mut self, output: Schema) -> Result { + pub(crate) fn with_output_schema(mut self, output: SchemaRef) -> Result { if self.output.fields.len() != output.fields.len() || self .output @@ -259,11 +259,11 @@ impl Operator { self.output = output; Ok(self) } - pub fn schema(&self) -> Schema { + pub fn schema(&self) -> SchemaRef { self.output.clone() } } -impl PhysicalOperator for Operator { +impl PhysicalOperator for Operator { fn requires_bounded_input(&self) -> bool { matches!( self.kind, @@ -345,10 +345,10 @@ impl PhysicalOperator for Operator { series_window::validate_context(self, context)?; self.readout_range(context).map(|_| ()) } - fn input_schemas(&self) -> Vec { + fn input_schemas(&self) -> Vec { self.inputs.clone() } - fn output_schema(&self) -> Schema { + fn output_schema(&self) -> SchemaRef { self.output.clone() } fn output_bytes(&self, value: &Batch) -> usize { diff --git a/crates/asap-physical-operators/src/operators/projection.rs b/crates/asap-physical-operators/src/operators/projection.rs index 4a174ed92..d019b8146 100644 --- a/crates/asap-physical-operators/src/operators/projection.rs +++ b/crates/asap-physical-operators/src/operators/projection.rs @@ -1,6 +1,6 @@ use super::*; impl Operator { - pub fn project(input: Schema, columns: Vec<(String, Expression)>) -> Result { + pub fn project(input: SchemaRef, columns: Vec<(String, Expression)>) -> Result { let fields = columns .iter() .map(|(name, e)| { diff --git a/crates/asap-physical-operators/src/operators/scope_timestamp.rs b/crates/asap-physical-operators/src/operators/scope_timestamp.rs index bdbd7b0d8..bb7a4e855 100644 --- a/crates/asap-physical-operators/src/operators/scope_timestamp.rs +++ b/crates/asap-physical-operators/src/operators/scope_timestamp.rs @@ -3,7 +3,7 @@ use super::*; use crate::runtime::Scope; impl Operator { - pub(crate) fn scope_timestamp(input: Schema, output: Schema) -> Result { + pub(crate) fn scope_timestamp(input: SchemaRef, output: SchemaRef) -> Result { crate::values::validate_schema(&output)?; let coordinate = output .time_index diff --git a/crates/asap-physical-operators/src/operators/series_labels.rs b/crates/asap-physical-operators/src/operators/series_labels.rs index 7f468964d..2a8c5d1ee 100644 --- a/crates/asap-physical-operators/src/operators/series_labels.rs +++ b/crates/asap-physical-operators/src/operators/series_labels.rs @@ -18,7 +18,7 @@ struct Layout { value: usize, } -fn layout(input: &Schema) -> Result { +fn layout(input: &SchemaRef) -> Result { let mut identity = None; let mut labels = Vec::new(); let mut value = None; @@ -39,7 +39,7 @@ fn layout(input: &Schema) -> Result { } impl Layout { - fn read(&self, input: &Schema, row: &[Value]) -> Result { + fn read(&self, input: &SchemaRef, row: &[Value]) -> Result { if let Some(i) = self.identity { let Value::Utf8(encoded) = &row[i] else { return Err(invalid("series identity must be Utf8")); @@ -64,7 +64,7 @@ impl Layout { } /// Replace the row's labels; a label column absent from `labels` is empty. - fn write(&self, input: &Schema, row: &mut [Value], labels: &Labels) -> Result<(), Error> { + fn write(&self, input: &SchemaRef, row: &mut [Value], labels: &Labels) -> Result<(), Error> { if let Some(i) = self.identity { let encoded = serde_json::to_string(labels).map_err(|e| invalid(&e.to_string()))?; row[i] = Value::Utf8(encoded.into()); @@ -79,8 +79,8 @@ impl Layout { impl Operator { pub(crate) fn series_relabel( - input: Schema, - output: Schema, + input: SchemaRef, + output: SchemaRef, destination: String, replacement: String, source_regex: Option<(String, String)>, @@ -127,7 +127,7 @@ impl Operator { /// Rewrite each row's label set to PromQL's matching labels: `On` keeps /// only `labels`; `Ignoring` drops `labels` and the metric name. pub fn series_labels( - input: Schema, + input: SchemaRef, kind: VectorMatchKind, labels: Vec, ) -> Result { @@ -146,7 +146,7 @@ impl Operator { /// Drop the metric name from each row's label set, as PromQL arithmetic /// with a literal does. Unlike matching labels, the result is itself a /// vector, so two rows that become equal are an error, as in Prometheus. - pub fn series_without_name(input: Schema) -> Result { + pub fn series_without_name(input: SchemaRef) -> Result { layout(&input)?; Ok(Self { kind: Kind::SeriesLabels { @@ -162,7 +162,7 @@ impl Operator { /// PromQL `histogram_quantile` over classic buckets: one histogram per /// label set other than the `le` column's label. The result drops `le`, /// `__name__` and the time column; its value is the quantile. - pub fn series_histogram_quantile(input: Schema, q: f64, le: usize) -> Result { + pub fn series_histogram_quantile(input: SchemaRef, q: f64, le: usize) -> Result { let layout = layout(&input)?; if !layout.labels.contains(&le) { return Err(invalid("histogram bucket bound must be a label column")); @@ -190,8 +190,8 @@ impl Operator { /// between vectors; label columns without a series identity must hold /// every label the result can take from the right side. pub fn series_binary( - left: Schema, - right: Schema, + left: SchemaRef, + right: SchemaRef, operator: BinaryOperator, scalars: [bool; 2], ) -> Result { diff --git a/crates/asap-physical-operators/src/operators/series_window.rs b/crates/asap-physical-operators/src/operators/series_window.rs index 91bbb11ce..b22c8e0ac 100644 --- a/crates/asap-physical-operators/src/operators/series_window.rs +++ b/crates/asap-physical-operators/src/operators/series_window.rs @@ -54,7 +54,7 @@ impl Operator { /// markers. A series is every column except the time and `value` columns. /// Output rows keep the input schema, with time `t` and the result value. pub fn series_window( - input: Schema, + input: SchemaRef, function: Option>, range_ms: i64, offset_ms: i64, diff --git a/crates/asap-physical-operators/src/operators/sort.rs b/crates/asap-physical-operators/src/operators/sort.rs index 71f81de01..116e55336 100644 --- a/crates/asap-physical-operators/src/operators/sort.rs +++ b/crates/asap-physical-operators/src/operators/sort.rs @@ -1,6 +1,6 @@ use super::*; impl Operator { - pub fn sort(input: Schema, keys: Vec, groups: Vec) -> Result { + pub fn sort(input: SchemaRef, keys: Vec, groups: Vec) -> Result { validate_groups(&input, &groups)?; for key in &keys { if !ordered(plain(&input, key.column)?.0) { diff --git a/crates/asap-physical-operators/src/operators/source.rs b/crates/asap-physical-operators/src/operators/source.rs index 95b1d2b03..a92c2e0db 100644 --- a/crates/asap-physical-operators/src/operators/source.rs +++ b/crates/asap-physical-operators/src/operators/source.rs @@ -8,7 +8,7 @@ impl Operator { } } - pub fn source(output: Schema, batches: Vec) -> Result { + pub fn source(output: SchemaRef, batches: Vec) -> Result { crate::values::validate_schema(&output)?; if batches.iter().any(|b| b.schema() != &output) { return Err(invalid("source schema mismatch")); @@ -32,7 +32,7 @@ impl Operator { output, }) } - pub fn vector_to_scalar(input: Schema, column: usize) -> Result { + pub fn vector_to_scalar(input: SchemaRef, column: usize) -> Result { if plain(&input, column)? != (&DataType::Float64, false) { return Err(invalid("scalar conversion requires non-null Float64")); } @@ -42,7 +42,7 @@ impl Operator { output: schema(vec![result_field("value", DataType::Float64, false)]), }) } - pub fn union(input: Schema, arity: usize) -> Result { + pub fn union(input: SchemaRef, arity: usize) -> Result { if arity == 0 { return Err(invalid("union needs at least one input")); } diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs index b6fff4f39..00a4d0b19 100644 --- a/crates/asap-physical-operators/src/operators/summary/mod.rs +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -8,7 +8,7 @@ pub enum ReadoutQuery { impl Operator { pub fn keyed_summary_build( - input: Schema, + input: SchemaRef, family: FieldDataType, value: usize, items: Vec, @@ -61,10 +61,10 @@ impl Operator { }) } pub fn keyed_readout( - input: Schema, + input: SchemaRef, state: usize, k: usize, - output: Schema, + output: SchemaRef, ) -> Result { use crate::summary_kernels::weighted_frequency::WeightedFrequency; crate::values::validate_family(&field(&input, state)?.dtype)?; @@ -91,7 +91,7 @@ impl Operator { }) } pub fn summary_build( - input: Schema, + input: SchemaRef, family: FieldDataType, value: usize, time: Option, @@ -146,7 +146,11 @@ impl Operator { output: schema(fields), }) } - pub fn summary_merge(input: Schema, state: usize, groups: Vec) -> Result { + pub fn summary_merge( + input: SchemaRef, + state: usize, + groups: Vec, + ) -> Result { validate_groups(&input, &groups)?; crate::values::validate_family(&field(&input, state)?.dtype)?; if matches!(field(&input, state)?.dtype, FieldDataType::Plain(_)) { @@ -163,7 +167,7 @@ impl Operator { output: schema(fields), }) } - pub fn readout(input: Schema, state: usize, query: ReadoutQuery) -> Result { + pub fn readout(input: SchemaRef, state: usize, query: ReadoutQuery) -> Result { let family = &field(&input, state)?.dtype; crate::values::validate_family(family)?; match &query { diff --git a/crates/asap-physical-operators/src/operators/unchecked.rs b/crates/asap-physical-operators/src/operators/unchecked.rs index 7a8d9d95f..d3504f9a9 100644 --- a/crates/asap-physical-operators/src/operators/unchecked.rs +++ b/crates/asap-physical-operators/src/operators/unchecked.rs @@ -5,8 +5,8 @@ use super::*; #[serde(deny_unknown_fields)] pub(super) struct UncheckedOperator { kind: Kind, - inputs: Vec, - output: Schema, + inputs: Vec, + output: SchemaRef, } impl TryFrom for Operator { type Error = Error; diff --git a/crates/asap-physical-operators/src/operators/vector_binary.rs b/crates/asap-physical-operators/src/operators/vector_binary.rs index 88f019c25..7027fb9be 100644 --- a/crates/asap-physical-operators/src/operators/vector_binary.rs +++ b/crates/asap-physical-operators/src/operators/vector_binary.rs @@ -2,7 +2,7 @@ use super::*; use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; -pub(crate) fn value_schema(scalar: bool) -> Schema { +pub(crate) fn value_schema(scalar: bool) -> SchemaRef { let mut fields = Vec::new(); if !scalar { fields.push(result_field( @@ -23,7 +23,7 @@ pub(crate) fn value_schema(scalar: bool) -> Schema { schema(fields) } -fn is_scalar(input: &Schema) -> Result { +fn is_scalar(input: &SchemaRef) -> Result { for scalar in [true, false] { let expected = value_schema(scalar); if input.fields.len() == expected.fields.len() @@ -43,8 +43,8 @@ fn is_scalar(input: &Schema) -> Result { impl Operator { pub fn vector_binary( - left: Schema, - right: Schema, + left: SchemaRef, + right: SchemaRef, operator: BinaryOperator, mut return_bool: bool, ) -> Result { diff --git a/crates/asap-physical-operators/src/operators/vector_window.rs b/crates/asap-physical-operators/src/operators/vector_window.rs index 93bb6e9c6..6b4d58347 100644 --- a/crates/asap-physical-operators/src/operators/vector_window.rs +++ b/crates/asap-physical-operators/src/operators/vector_window.rs @@ -1,14 +1,14 @@ //! Window bounds are typed input data; aggregation and histogram semantics stay native. use super::*; use planner_types::pre_asap::AggIntent; -use planner_types::pre_asap::Schema as PlannerSchema; +use planner_types::pre_asap::Schema; -pub(crate) fn matrix_schema() -> Schema { +pub(crate) fn matrix_schema() -> SchemaRef { let mut fields = vector_binary::value_schema(false).fields.clone(); fields.insert(1, result_field("timestamp", DataType::Timestamp, false)); fields.push(result_field("window_start", DataType::Timestamp, false)); fields.push(result_field("window_end", DataType::Timestamp, false)); - Arc::new(PlannerSchema { + Arc::new(Schema { closed: true, unique_keys: vec![], fields, diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs index 66f476c2d..f5f092354 100644 --- a/crates/asap-physical-operators/src/physical_planner/compiled.rs +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -4,11 +4,11 @@ use super::*; /// A typed execution boundary, without storage identity or a live reader. #[derive(Clone, Debug, serde::Serialize, serde::Deserialize)] pub struct InputContract { - pub schema: Schema, + pub schema: SchemaRef, pub properties: PlanProperties, } impl InputContract { - pub fn bounded(schema: Schema) -> Self { + pub fn bounded(schema: SchemaRef) -> Self { Self { schema, properties: PlanProperties { @@ -17,7 +17,7 @@ impl InputContract { }, } } - pub fn from_source(source: &dyn PhysicalOperator) -> Self { + pub fn from_source(source: &dyn PhysicalOperator) -> Self { Self { schema: source.output_schema(), properties: source.properties(&[]), @@ -285,7 +285,7 @@ impl CompiledPhysicalDAG { pub fn instantiate<'a>( &self, mut sources: BTreeMap>, - ) -> Result, Error> { + ) -> Result, Error> { let mut dag = PhysicalDAG::default(); for (&id, node) in &self.nodes { match node { @@ -326,14 +326,14 @@ impl CompiledPhysicalDAG { Ok(dag) } } -impl PhysicalOperator for InputContract { +impl PhysicalOperator for InputContract { fn name(&self) -> &str { "UnresolvedInput" } - fn input_schemas(&self) -> Vec { + fn input_schemas(&self) -> Vec { vec![] } - fn output_schema(&self) -> Schema { + fn output_schema(&self) -> SchemaRef { self.schema.clone() } fn properties(&self, _: &[PlanProperties]) -> PlanProperties { diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 76511b03c..71a3eb351 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -5,7 +5,7 @@ use crate::summary_kernels::exact::ExactReadout; use crate::{ operators::{Expression, Operator, Reduction, SortKey}, plan::{Boundedness, Emission, NodeId, PhysicalDAG, PhysicalOperator, PlanProperties}, - values::{Batch, Schema}, + values::{Batch, SchemaRef}, Error, }; use planner_types::{ @@ -29,7 +29,7 @@ fn invalid(message: impl Into) -> Error { /// Source nodes cut the DAG at an installed storage/ingestion frontier. The /// binding must have exactly the declared schema and no upstream dependencies. /// A deployment must authorize these frontiers before calling this function. -pub type Source<'a> = Box + 'a>; +pub type Source<'a> = Box + 'a>; pub mod precompute; pub mod promql_fallback; @@ -63,7 +63,7 @@ pub fn bind<'a>( dag: &PostAsapDAG, sources: BTreeMap>, roots: &[NodeId], -) -> Result, Error> { +) -> Result, Error> { let inputs = sources .iter() .map(|(&id, source)| (id, InputContract::from_source(source.as_ref()))) @@ -77,7 +77,7 @@ pub fn bind_with_data_sources<'a>( mut sources: BTreeMap>, roots: &[NodeId], data_sources: &crate::sources::DataSources, -) -> Result, Error> { +) -> Result, Error> { // Only resolve scans reachable below the selected input boundaries. let mut pending = roots.to_vec(); let mut seen = BTreeSet::new(); @@ -495,7 +495,7 @@ fn compile_internal( auxiliary -= 1; continue; } - let label_map = |schema: &Schema| { + let label_map = |schema: &SchemaRef| { schema .fields .iter() @@ -634,20 +634,20 @@ fn temporal_readout_drops_name(node: &PostAsapDAGNode) -> bool { /// Bind a Planner node against the schemas supplied by its deployment edges. /// This is the same checked path used by complete DAG binding. -pub fn compile_node(node: &PostAsapDAGNode, inputs: &[Schema]) -> Result { +pub fn compile_node(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result { for schema in inputs { crate::values::validate_schema(schema)?; } bind_operation(node, inputs)?.with_output_schema(Arc::new(node.output_schema.clone())) } -fn bind_operation(node: &PostAsapDAGNode, inputs: &[Schema]) -> Result { +fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result { if let Payload::Binary { operator } = &node.payload { let [left, right] = inputs else { return Err(invalid("binary requires two inputs")); }; if node.output_state.timing == planner_types::post_asap::ExecutionTiming::IngestionTime { - let value = |schema: &Schema| -> Result { + let value = |schema: &SchemaRef| -> Result { let columns = schema .fields .iter() @@ -887,7 +887,7 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[Schema]) -> Result, ) -> Result<(), Error> { match expr { @@ -963,7 +963,7 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[Schema]) -> Result Result { +fn summary_column(input: &SchemaRef) -> Result { let columns = input .fields .iter() @@ -976,7 +976,7 @@ fn summary_column(input: &Schema) -> Result { _ => Err(invalid("one summary state column required")), } } -fn named_column(input: &Schema, column: &ColumnRef) -> Result { +fn named_column(input: &SchemaRef, column: &ColumnRef) -> Result { let name = match column { // This summary-update lookup matches column names without qualifiers. // Reject ambiguous names rather than guessing a join side. @@ -1000,7 +1000,7 @@ fn named_column(input: &Schema, column: &ColumnRef) -> Result { _ => Err(invalid("summary update column missing or ambiguous")), } } -fn groups(input: &Schema, groups: &GroupKeys) -> Result, Error> { +fn groups(input: &SchemaRef, groups: &GroupKeys) -> Result, Error> { if groups.is_without() { return Err(invalid("grouping without requires resolved label columns")); } @@ -1009,7 +1009,7 @@ fn groups(input: &Schema, groups: &GroupKeys) -> Result, Error> { } Ok(groups.keys().to_vec()) } -fn expression(expr: &QueryExpr, input: &Schema) -> Result { +fn expression(expr: &QueryExpr, input: &SchemaRef) -> Result { Ok(Expression::planner( crate::expressions::CompiledExpression::compile(expr, input)?, )) @@ -1017,9 +1017,9 @@ fn expression(expr: &QueryExpr, input: &Schema) -> Result { struct CheckedSource<'a> { source: Source<'a>, - output: Schema, + output: SchemaRef, } -impl PhysicalOperator for CheckedSource<'_> { +impl PhysicalOperator for CheckedSource<'_> { fn properties(&self, inputs: &[crate::plan::PlanProperties]) -> crate::plan::PlanProperties { self.source.properties(inputs) } @@ -1027,10 +1027,10 @@ impl PhysicalOperator for CheckedSource<'_> { fn name(&self) -> &str { self.source.name() } - fn input_schemas(&self) -> Vec { + fn input_schemas(&self) -> Vec { vec![] } - fn output_schema(&self) -> Schema { + fn output_schema(&self) -> SchemaRef { self.output.clone() } fn output_bytes(&self, batch: &Batch) -> usize { diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index 9f2a8ccf2..10dffdc0c 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -2,14 +2,14 @@ use super::promql_rows::SERIES_IDENTITY_COLUMN as SERIES_IDENTITY; use super::*; use planner_types::{ - post_asap::{ExecutionTiming, GroupingStrategy, Schema as PlannerSchema}, + post_asap::{ExecutionTiming, GroupingStrategy, Schema}, pre_asap::DataType, }; /// Physical rows carry the population and pane coordinate alongside the logical value. /// These fields preserve identities which are implicit in a stored summary instance. -pub fn population_schema(family: FieldDataType) -> Schema { - Arc::new(PlannerSchema { +pub fn population_schema(family: FieldDataType) -> SchemaRef { + Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![ @@ -47,7 +47,7 @@ pub fn population_schema(family: FieldDataType) -> Schema { /// selected; the deployment decides which rows and panes they are. Label sets /// must be canonical (sorted, unique, no empty values), since they are the /// population identity: build rows with [`raw_sample_row`]. -pub fn raw_sample_schema() -> Schema { +pub fn raw_sample_schema() -> SchemaRef { let mut schema = (*population_schema(FieldDataType::Plain(DataType::Float64))).clone(); schema.fields[1].name = "$timestamp".into(); Arc::new(schema) @@ -82,7 +82,7 @@ pub fn raw_sample_row( /// Input contract of a precompute boundary: raw sample rows for a raw time /// series scan, otherwise the stored population of its summary state. -pub fn boundary_schema(node: &PostAsapDAGNode) -> Result { +pub fn boundary_schema(node: &PostAsapDAGNode) -> Result { let Payload::Fallback { expression } = &node.payload else { return source_schema(&node.output_schema); }; @@ -128,9 +128,9 @@ pub fn boundary_schema(node: &PostAsapDAGNode) -> Result { } /// Validate the adapter layout during installed-plan recovery without lowering operators. -/// `PlannerSchema` is an import alias for the shared planner `Schema`; the return -/// type is the runtime `Arc` handle for the population adapter layout. -pub fn source_schema(logical: &PlannerSchema) -> Result { +/// Borrows shared schema metadata and returns an Arc-owned schema for the +/// population adapter layout. +pub fn source_schema(logical: &Schema) -> Result { let states = logical .fields .iter() @@ -151,7 +151,7 @@ pub fn source_schema(logical: &PlannerSchema) -> Result { Ok(population_schema(state.dtype.clone())) } -pub fn is_population_schema(schema: &Schema) -> bool { +pub fn is_population_schema(schema: &SchemaRef) -> bool { schema .fields .get(2) @@ -223,7 +223,7 @@ pub fn compile( } let mut sources = BTreeMap::new(); let mut fragments = BTreeMap::new(); - let mut outputs = BTreeMap::::new(); + let mut outputs = BTreeMap::::new(); for id in ordered { let node = nodes[&id]; if frontier.contains(&id) { @@ -303,7 +303,7 @@ fn validate_value_output(node: &PostAsapDAGNode) -> Result<(), Error> { fn fragment( node: &PostAsapDAGNode, - schemas: &[Schema], + schemas: &[SchemaRef], parents: &[&PostAsapDAGNode], ) -> Result { let sources = schemas @@ -520,7 +520,7 @@ fn fragment( } let item_columns = (3..fields.len()).collect::>(); let project = Operator::project(input.clone(), columns)?.with_output_schema( - Arc::new(PlannerSchema { + Arc::new(Schema { closed: true, unique_keys: vec![], fields, @@ -574,7 +574,7 @@ fn fragment( /// of the label set less excluded labels. fn raw_items( expr: &SummaryInputExpr, - scan: &PlannerSchema, + scan: &Schema, items: &mut Vec<(Expression, DataType)>, ) -> Result<(), Error> { // Open PromQL scans need not list every label, so any name that is not diff --git a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs index bbc4d2986..237504c98 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs @@ -19,7 +19,7 @@ pub(super) fn raw_series_owner(slot: NodeId) -> Option { } /// A selector expression and its raw-series row schema. -pub type Selector = (QueryExpr, Schema); +pub type Selector = (QueryExpr, SchemaRef); /// The selectors a Fallback expression reads, left to right, and the row /// schema of the raw series the deployment supplies for each at @@ -49,7 +49,7 @@ pub(super) fn lower(expression: &QueryExpr) -> Result { Ok(lowering) } -fn declared(expression: &QueryExpr) -> Result { +fn declared(expression: &QueryExpr) -> Result { let schema = expression .output_schema() .map_err(|error| invalid(error.to_string()))?; @@ -107,7 +107,7 @@ pub(super) fn scalar(expression: &QueryExpr) -> bool { } impl Lowering { - fn schema(&self, input: &Input) -> Schema { + fn schema(&self, input: &Input) -> SchemaRef { match input { Input::Raw(i) => self.selectors[*i].1.clone(), Input::Step(i) => self.steps[*i].0.schema(), diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index 70cacec3f..106f4a522 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -31,7 +31,7 @@ pub fn with_series_identity(root: &QueryExpr) -> Result { /// Construct source rows only from full identities. The named label columns /// are projections of that same identity and cannot independently redefine it. pub fn series_row( - schema: &Schema, + schema: &SchemaRef, labels: &BTreeMap, timestamp: i64, value: f64, diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/asap-physical-operators/src/physical_planner/promql_values.rs index 73f86779c..880916ed2 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_values.rs @@ -1,14 +1,14 @@ //! Physical scalar/vector contracts preserve complete label sets across native computation. use super::*; -pub fn scalar_schema() -> Schema { +pub fn scalar_schema() -> SchemaRef { crate::operators::vector_binary::value_schema(true) } -pub fn vector_schema() -> Schema { +pub fn vector_schema() -> SchemaRef { crate::operators::vector_binary::value_schema(false) } -pub fn matrix_schema() -> Schema { +pub fn matrix_schema() -> SchemaRef { crate::operators::vector_window::matrix_schema() } @@ -81,7 +81,7 @@ pub fn compile_binary( ) } -fn unary(operators: Vec, input: Schema) -> Result { +fn unary(operators: Vec, input: SchemaRef) -> Result { let root = operators.len() as u64; CompiledPhysicalDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(input))]), @@ -120,7 +120,7 @@ fn grouped(grouping: &GroupKeys) -> Result { ) } -fn vector_output(input: Schema, labels: usize, value: usize) -> Result { +fn vector_output(input: SchemaRef, labels: usize, value: usize) -> Result { let value = Expression::ExactFloat64(value); Operator::project( input, @@ -209,7 +209,7 @@ pub fn compile_vector_to_scalar() -> Result { /// A stored exact-state input retains the complete population identity. The /// deployment supplies eligible panes; merging and finalization are computation. -pub fn exact_state_schema(family: FieldDataType) -> Result { +pub fn exact_state_schema(family: FieldDataType) -> Result { if !matches!(family, FieldDataType::ExactAggregate(..)) { return Err(invalid("exact-state input requires an exact family")); } diff --git a/crates/asap-physical-operators/src/physical_planner/row_values.rs b/crates/asap-physical-operators/src/physical_planner/row_values.rs index 8763437bb..78d0227a1 100644 --- a/crates/asap-physical-operators/src/physical_planner/row_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/row_values.rs @@ -14,7 +14,7 @@ pub(super) fn scalar_literal(expression: &QueryExpr) -> Option { /// Aggregate readouts of a maintained current-series population, as a chain. pub(super) fn population_aggregate( - input: &Schema, + input: &SchemaRef, grouping: &[String], readout: &PopulationReadout, ) -> Result, Error> { diff --git a/crates/asap-physical-operators/src/runtime/batch_execution.rs b/crates/asap-physical-operators/src/runtime/batch_execution.rs index 3749d11b6..ca59877cb 100644 --- a/crates/asap-physical-operators/src/runtime/batch_execution.rs +++ b/crates/asap-physical-operators/src/runtime/batch_execution.rs @@ -61,7 +61,7 @@ pub fn evaluate_source( } fn evaluate_dag( - dag: PhysicalDAG<'_, Batch, crate::values::Schema>, + dag: PhysicalDAG<'_, Batch, crate::values::SchemaRef>, root: crate::plan::NodeId, context: RunContext, ) -> Result>, Error> { @@ -88,7 +88,7 @@ mod tests { values::Value, }; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as PlannerSchema}, + post_asap::{Field, FieldDataType, Schema}, pre_asap::DataType, }; use std::sync::Arc; @@ -96,7 +96,7 @@ mod tests { // Engine adapters can run the identical native chain from an outer executor. #[test] fn same_native_chain_inside_query_and_ingestion_execution() { - let schema = Arc::new(PlannerSchema { + let schema = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![Field { @@ -142,7 +142,7 @@ mod tests { // Native sources may cross the runtime's cooperative batch quantum. #[test] fn in_memory_source_drives_cooperative_yields() { - let schema = Arc::new(PlannerSchema { + let schema = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![], @@ -164,7 +164,7 @@ mod tests { // An adapter-held output must retain its parent's reservation after execution. #[test] fn returned_batches_keep_their_resource_reservation() { - let schema = Arc::new(PlannerSchema { + let schema = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![], @@ -195,7 +195,7 @@ mod tests { // A cancelled surrounding execution also prevents its native computation. #[test] fn cancellation_is_not_bypassed_by_in_memory_execution() { - let schema = Arc::new(PlannerSchema { + let schema = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![], diff --git a/crates/asap-physical-operators/src/sources/memory.rs b/crates/asap-physical-operators/src/sources/memory.rs index 52ffbc303..1055888de 100644 --- a/crates/asap-physical-operators/src/sources/memory.rs +++ b/crates/asap-physical-operators/src/sources/memory.rs @@ -2,11 +2,11 @@ use super::*; /// Immutable in-memory raw data. The connector owns the resident input; each /// cursor clones only the next requested batch, not the entire data set. pub struct MemorySource { - schema: Schema, + schema: SchemaRef, batches: Vec, } impl MemorySource { - pub fn new(schema: Schema, batches: Vec) -> Result { + pub fn new(schema: SchemaRef, batches: Vec) -> Result { crate::values::validate_schema(&schema)?; if schema .fields @@ -27,7 +27,7 @@ impl RawSource for MemorySource { fn boundedness(&self) -> crate::plan::Boundedness { crate::plan::Boundedness::Bounded } - fn schema(&self) -> Schema { + fn schema(&self) -> SchemaRef { self.schema.clone() } fn scan(&self, context: RunContext) -> Result, Error> { diff --git a/crates/asap-physical-operators/src/sources/mod.rs b/crates/asap-physical-operators/src/sources/mod.rs index 7f8018f52..4b401d666 100644 --- a/crates/asap-physical-operators/src/sources/mod.rs +++ b/crates/asap-physical-operators/src/sources/mod.rs @@ -3,12 +3,12 @@ use crate::{ expressions::CompiledExpression, plan::PhysicalOperator, runtime::{Input, OutputStream, RunContext}, - values::{Batch, Schema, Value}, + values::{Batch, SchemaRef, Value}, Error, }; use futures::{stream, StreamExt}; use planner_types::{ - post_asap::{FieldDataType, Schema as PlannerSchema}, + post_asap::{FieldDataType, Schema}, pre_asap::{DataType, QueryExpr, Source}, }; use std::sync::Arc; @@ -18,7 +18,7 @@ use std::sync::Arc; /// and must honor cancellation and bound their own I/O buffers. Dropping a cursor /// must release its resources. A connector error is never an empty successful scan. pub trait RawSource { - fn schema(&self) -> Schema; + fn schema(&self) -> SchemaRef; /// Declare a finite snapshot/window explicitly; execution scope alone does not bound a cursor. fn boundedness(&self) -> crate::plan::Boundedness { crate::plan::Boundedness::Unknown @@ -51,10 +51,7 @@ impl DataSources { "raw Scan requires a Planner Scan leaf".into(), )); }; - let output = Arc::new(PlannerSchema::lifted( - schema.fields.clone(), - schema.time_index, - )); + let output = Arc::new(Schema::lifted(schema.fields.clone(), schema.time_index)); crate::values::validate_schema(&output)?; let reader = self .sources @@ -87,10 +84,10 @@ impl DataSources { pub struct Scan { reader: Arc, - output: Schema, + output: SchemaRef, predicates: Vec, } -impl PhysicalOperator for Scan { +impl PhysicalOperator for Scan { fn properties(&self, _: &[crate::plan::PlanProperties]) -> crate::plan::PlanProperties { crate::plan::PlanProperties { boundedness: self.reader.boundedness(), @@ -101,10 +98,10 @@ impl PhysicalOperator for Scan { fn name(&self) -> &str { "Scan" } - fn input_schemas(&self) -> Vec { + fn input_schemas(&self) -> Vec { vec![] } - fn output_schema(&self) -> Schema { + fn output_schema(&self) -> SchemaRef { self.output.clone() } fn output_bytes(&self, batch: &Batch) -> usize { diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs index 591e44d8f..b1186c171 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -2,13 +2,13 @@ use crate::AggregateCore; use crate::Error; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as PlannerSchema}, + post_asap::{Field, FieldDataType, Schema}, pre_asap::DataType, }; use std::{cmp::Ordering, sync::Arc}; -/// Shared runtime handle to the same schema metadata used by the planner. -/// This alias changes ownership, not the schema model or its field types. -pub type Schema = Arc; +/// Shared ownership of schema metadata; the field model is identical at planning +/// and execution time. +pub type SchemaRef = Arc; #[derive(Clone, serde::Serialize, serde::Deserialize)] pub enum Value { Null, @@ -200,11 +200,11 @@ impl Value { } #[derive(Clone, Debug)] pub struct Batch { - schema: Schema, + schema: SchemaRef, rows: Vec>, } impl Batch { - pub fn try_new(schema: Schema, rows: Vec>) -> Result { + pub fn try_new(schema: SchemaRef, rows: Vec>) -> Result { validate_schema(&schema)?; for row in &rows { if row.len() != schema.fields.len() { @@ -230,7 +230,7 @@ impl Batch { } Ok(Self { schema, rows }) } - pub fn schema(&self) -> &Schema { + pub fn schema(&self) -> &SchemaRef { &self.schema } pub fn rows(&self) -> &[Vec] { @@ -320,7 +320,7 @@ fn validate_state(family: &FieldDataType, state: &dyn AggregateCore) -> Result<( } } -pub(crate) fn validate_schema(schema: &Schema) -> Result<(), Error> { +pub(crate) fn validate_schema(schema: &SchemaRef) -> Result<(), Error> { if schema.time_index.is_some_and(|index| { schema .fields @@ -344,13 +344,13 @@ pub(crate) fn validate_schema(schema: &Schema) -> Result<(), Error> { Ok(()) } -pub(crate) fn field(schema: &Schema, column: usize) -> Result<&Field, Error> { +pub(crate) fn field(schema: &SchemaRef, column: usize) -> Result<&Field, Error> { schema .fields .get(column) .ok_or_else(|| Error::Invalid("column out of range".into())) } -pub(crate) fn plain(schema: &Schema, column: usize) -> Result<(&DataType, bool), Error> { +pub(crate) fn plain(schema: &SchemaRef, column: usize) -> Result<(&DataType, bool), Error> { let f = field(schema, column)?; let FieldDataType::Plain(dtype) = &f.dtype else { return Err(Error::Invalid("plain value required".into())); diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/asap-physical-operators/tests/blocking_resources.rs index 6b5a531eb..37f780812 100644 --- a/crates/asap-physical-operators/tests/blocking_resources.rs +++ b/crates/asap-physical-operators/tests/blocking_resources.rs @@ -3,18 +3,18 @@ use asap_physical_operators::{ operators::Operator, plan::{PhysicalDAG, PhysicalOperator}, runtime::{Limits, RunContext, Scope}, - values::{Batch, Schema, Value}, + values::{Batch, SchemaRef, Value}, Error, }; use futures::{executor::block_on, FutureExt, StreamExt}; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as PlannerSchema}, + post_asap::{Field, FieldDataType, Schema}, pre_asap::{DataType, JoinKind, Predicate, QueryExpr, ScalarValue}, }; use std::sync::Arc; -fn schema(width: usize) -> Schema { - Arc::new(PlannerSchema { +fn schema(width: usize) -> SchemaRef { + Arc::new(Schema { closed: true, unique_keys: vec![], fields: (0..width) @@ -41,7 +41,7 @@ fn context(max_bytes: usize) -> RunContext { ) .unwrap() } -fn source(n: usize) -> PhysicalDAG<'static, Batch, Schema> { +fn source(n: usize) -> PhysicalDAG<'static, Batch, SchemaRef> { let mut dag = PhysicalDAG::default(); dag.add( 0, @@ -189,7 +189,7 @@ fn cooperative_sort_preserves_ties_across_chunks() { #[test] fn weighted_summary_build_yields_within_a_batch() { use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; - let input = Arc::new(PlannerSchema { + let input = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![ diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs index 1701e6586..079c2bd69 100644 --- a/crates/asap-physical-operators/tests/current_series_heap.rs +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -9,12 +9,12 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema as PlannerSchema; +use planner_types::pre_asap::Schema; use planner_types::{post_asap::*, pre_asap::DataType}; use std::{collections::BTreeMap, sync::Arc}; -fn schema() -> Arc { - Arc::new(PlannerSchema { +fn schema() -> Arc { + Arc::new(Schema { closed: true, unique_keys: vec![], fields: [ @@ -169,7 +169,7 @@ fn spatial_heap_ranks_latest_values_in_independent_runs() { }; let family = FieldDataType::Sketch(SketchKind::new(algorithm, params), Default::default()); let build = Operator::keyed_summary_build(schema(), family, 1, vec![3], vec![2]).unwrap(); - let output = Arc::new(PlannerSchema { + let output = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![ diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index 5d6266b31..996552e7d 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -7,7 +7,7 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema as PlannerSchema; +use planner_types::pre_asap::Schema; use planner_types::{post_asap::*, pre_asap::QueryExpr, types::AccuracyTarget, workload::*}; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; @@ -77,7 +77,7 @@ fn population_dag(query: &str) -> PostAsapDAG { } /// Raw scan nodes are the frontier; everything above them is compiled. -fn raw_inputs(dag: &PostAsapDAG) -> Vec<(u64, Arc, String)> { +fn raw_inputs(dag: &PostAsapDAG) -> Vec<(u64, Arc, String)> { dag.nodes .iter() .filter_map(|node| match &node.payload { diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 29827dd54..df8650816 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -2,19 +2,19 @@ use asap_physical_operators::{ dag::{ operators::{Expression, Operator, Reduction, SortKey}, - values::{Batch, Schema, Value}, + values::{Batch, SchemaRef, Value}, Limits, PhysicalDAG, RunContext, Scope, }, Statistic, }; use futures::{executor::block_on, StreamExt}; use planner_types::{ - post_asap::{ExactKind, ExactParams, Field, FieldDataType, Schema as PlannerSchema}, + post_asap::{ExactKind, ExactParams, Field, FieldDataType, Schema}, pre_asap::DataType, }; use std::sync::Arc; -fn schema(fields: &[(&str, DataType, bool)]) -> Schema { - Arc::new(PlannerSchema { +fn schema(fields: &[(&str, DataType, bool)]) -> SchemaRef { + Arc::new(Schema { closed: true, unique_keys: vec![], fields: fields @@ -29,7 +29,7 @@ fn schema(fields: &[(&str, DataType, bool)]) -> Schema { time_index: None, }) } -fn run(dag: &PhysicalDAG<'_, Batch, Schema>, root: u64, scope: Scope) -> Vec> { +fn run(dag: &PhysicalDAG<'_, Batch, SchemaRef>, root: u64, scope: Scope) -> Vec> { let context = RunContext::new( scope, Limits { @@ -423,7 +423,7 @@ fn exact_state_and_family_validation() { let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); acc.update(None, 7., 0); - let schema = Arc::new(PlannerSchema { + let schema = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![Field { @@ -607,17 +607,17 @@ fn source_batches_must_match_the_bound_schema() { }; use std::{cell::Cell, collections::BTreeMap, rc::Rc}; struct WrongSource { - schema: Schema, + schema: SchemaRef, starts: Rc>, } - impl PhysicalOperator for WrongSource { + impl PhysicalOperator for WrongSource { fn name(&self) -> &str { "ExternalSource" } - fn input_schemas(&self) -> Vec { + fn input_schemas(&self) -> Vec { vec![] } - fn output_schema(&self) -> Schema { + fn output_schema(&self) -> SchemaRef { self.schema.clone() } fn output_bytes(&self, value: &Batch) -> usize { @@ -718,14 +718,14 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { ("score", DataType::Float64, false), ]); let keys_schema = schema(&[("key", DataType::Utf8, false)]); - let node = |id, payload, schema: &asap_physical_operators::values::Schema| PostAsapDAGNode { + let node = |id, payload, schema: &asap_physical_operators::values::SchemaRef| PostAsapDAGNode { id: PostAsapNodeId(id), payload, output_schema: (**schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, guarantee: None, }; - let edge = |producer, consumer, role, schema: &asap_physical_operators::values::Schema| { + let edge = |producer, consumer, role, schema: &asap_physical_operators::values::SchemaRef| { PostAsapDAGEdge { producer: PostAsapNodeId(producer), consumer: PostAsapNodeId(consumer), diff --git a/crates/asap-physical-operators/tests/physical_plan_recovery.rs b/crates/asap-physical-operators/tests/physical_plan_recovery.rs index 0526e31cb..fda1b502a 100644 --- a/crates/asap-physical-operators/tests/physical_plan_recovery.rs +++ b/crates/asap-physical-operators/tests/physical_plan_recovery.rs @@ -5,13 +5,13 @@ use asap_physical_operators::{ physical_planner::{CompiledPhysicalDAG, InputContract}, }; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as PlannerSchema}, + post_asap::{Field, FieldDataType, Schema}, pre_asap::DataType, }; use std::{collections::BTreeMap, sync::Arc}; fn sorted() -> CompiledPhysicalDAG { - let schema = Arc::new(PlannerSchema { + let schema = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![Field { diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs index d082b6578..61621c56f 100644 --- a/crates/asap-physical-operators/tests/physical_semantics.rs +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -6,17 +6,17 @@ use asap_physical_operators::{ operators::{Expression, Operator, Reduction, SortKey}, plan::PhysicalDAG, runtime::{Limits, RunContext, Scope}, - values::{Batch, Schema, Value}, + values::{Batch, SchemaRef, Value}, }; use futures::{executor::block_on, StreamExt}; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as PlannerSchema}, + post_asap::{Field, FieldDataType, Schema}, pre_asap::{CompareOpKind, DataType, JoinKind, Predicate, QueryExpr}, }; use std::{rc::Rc, sync::Arc}; -fn schema(fields: &[(&str, DataType, bool)]) -> Schema { - Arc::new(PlannerSchema { +fn schema(fields: &[(&str, DataType, bool)]) -> SchemaRef { + Arc::new(Schema { closed: true, unique_keys: vec![], fields: fields @@ -44,7 +44,7 @@ fn context() -> RunContext { ) .unwrap() } -fn collect(dag: &PhysicalDAG<'_, Batch, Schema>, root: u64) -> Vec> { +fn collect(dag: &PhysicalDAG<'_, Batch, SchemaRef>, root: u64) -> Vec> { let run = context(); let rows = block_on(async { let mut stream = dag.execute(&[root], run.clone()).unwrap().remove(0); @@ -57,7 +57,7 @@ fn collect(dag: &PhysicalDAG<'_, Batch, Schema>, root: u64) -> Vec> { assert_eq!(run.retained_bytes(), 0); rows } -fn unary(input: Schema, batches: Vec>>, op: Operator) -> Vec> { +fn unary(input: SchemaRef, batches: Vec>>, op: Operator) -> Vec> { let mut dag = PhysicalDAG::default(); let batches = batches .into_iter() diff --git a/crates/asap-physical-operators/tests/plan_properties.rs b/crates/asap-physical-operators/tests/plan_properties.rs index 4b84f4a96..fe2ad378b 100644 --- a/crates/asap-physical-operators/tests/plan_properties.rs +++ b/crates/asap-physical-operators/tests/plan_properties.rs @@ -4,11 +4,11 @@ use asap_physical_operators::{ plan::{Boundedness, Emission, PhysicalDAG}, runtime::{Limits, OutputStream, RunContext, Scope}, sources::{DataSources, RawSource}, - values::{Batch, Schema}, + values::{Batch, SchemaRef}, Error, }; use planner_types::{ - post_asap::{Field, FieldDataType, Schema as PlannerSchema}, + post_asap::{Field, FieldDataType, Schema}, pre_asap::{DataType, QueryExpr, Source}, }; use std::sync::{ @@ -16,12 +16,12 @@ use std::sync::{ Arc, }; struct DeclaredSource { - schema: Schema, + schema: SchemaRef, boundedness: Boundedness, opens: Arc, } impl RawSource for DeclaredSource { - fn schema(&self) -> Schema { + fn schema(&self) -> SchemaRef { self.schema.clone() } fn boundedness(&self) -> Boundedness { @@ -35,7 +35,7 @@ impl RawSource for DeclaredSource { // A blocking parent must reject unknown and unbounded Scan inputs without opening a reader. #[test] fn blocking_inputs_require_an_explicit_finite_source() { - let schema = Arc::new(PlannerSchema { + let schema = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![Field { @@ -69,7 +69,7 @@ fn blocking_inputs_require_an_explicit_finite_source() { let scan = registry .bind(&QueryExpr::Scan { source: identity, - schema: PlannerSchema::new(vec![Field::plain("v", DataType::Int64, false)]), + schema: Schema::new(vec![Field::plain("v", DataType::Int64, false)]), predicates: vec![], }) .unwrap(); diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs index 909364abd..5b91c6e72 100644 --- a/crates/asap-physical-operators/tests/precompute_population.rs +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -8,7 +8,7 @@ use asap_physical_operators::{ Statistic, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema as PlannerSchema; +use planner_types::pre_asap::Schema; use planner_types::{ post_asap::*, pre_asap::{ArithmeticOpKind, BinaryOpKind, ColumnRef, DataType, GroupKeys, Reduction}, @@ -19,7 +19,7 @@ use std::{collections::BTreeMap, sync::Arc}; #[test] fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let schema = |dtype| PlannerSchema { + let schema = |dtype| Schema { closed: true, unique_keys: vec![], fields: vec![Field { @@ -220,8 +220,8 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { } } -fn logical_schema(family: FieldDataType) -> PlannerSchema { - PlannerSchema { +fn logical_schema(family: FieldDataType) -> Schema { + Schema { closed: true, unique_keys: vec![], fields: vec![Field { diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs index fc5d97ead..4eec88aa1 100644 --- a/crates/asap-physical-operators/tests/promql_binary.rs +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -3,20 +3,20 @@ use asap_physical_operators::{ operators::Operator, physical_planner::{compile_node, CompiledPhysicalDAG, InputContract, Source}, runtime::{Limits, RunContext, Scope}, - values::{Batch, Schema, Value}, + values::{Batch, SchemaRef, Value}, }; use futures::{executor::block_on, StreamExt}; use planner_types::{ post_asap::{ BinaryOperator, ExecutionDataState, Field, FieldDataType, PostAsapDAGNode, PostAsapNodeId, - PostAsapOperatorPayload, Schema as PlannerSchema, + PostAsapOperatorPayload, Schema, }, pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, }; use std::{collections::BTreeMap, sync::Arc}; -fn schema() -> Schema { - Arc::new(PlannerSchema { +fn schema() -> SchemaRef { + Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![ @@ -324,7 +324,7 @@ fn stored_series_readouts_support_filters_and_sets() { (ExactKind::Count, ExactParams::Count), ] { let family = FieldDataType::ExactAggregate(exact_kind.clone(), params); - let state_schema = Arc::new(PlannerSchema { + let state_schema = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![ diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs index ffb726b1b..e7deb78b1 100644 --- a/crates/asap-physical-operators/tests/promql_values.rs +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -211,7 +211,7 @@ fn composed_ensemble_shares_a_producer_across_roots() { physical_planner::InputContract, plan::{PhysicalOperator, PlanProperties}, runtime::{Input, OutputStream}, - values::Schema, + values::SchemaRef, }; use planner_types::{ post_asap::BinaryOperator, @@ -221,14 +221,14 @@ fn composed_ensemble_shares_a_producer_across_roots() { source: Operator, starts: std::rc::Rc>, } - impl PhysicalOperator for Counted { + impl PhysicalOperator for Counted { fn name(&self) -> &str { "CountedInput" } - fn input_schemas(&self) -> Vec { + fn input_schemas(&self) -> Vec { vec![] } - fn output_schema(&self) -> Schema { + fn output_schema(&self) -> SchemaRef { self.source.schema() } fn output_bytes(&self, batch: &Batch) -> usize { diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs index 57e6edd4e..cf56a8682 100644 --- a/crates/asap-physical-operators/tests/raw_scan.rs +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -2,11 +2,11 @@ use asap_physical_operators::dag::{ planner::bind_with_data_sources, scan::{DataSources, MemorySource, RawSource}, - values::{Batch, Schema, Value}, + values::{Batch, SchemaRef, Value}, Error, Limits, OutputStream, RunContext, Scope, }; use futures::{executor::block_on, stream, StreamExt}; -use planner_types::pre_asap::Schema as PlannerSchema; +use planner_types::pre_asap::Schema; use planner_types::{ post_asap::*, pre_asap::{DataType, Field, GroupKeys, Predicate, QueryExpr, Source}, @@ -20,10 +20,10 @@ use std::{ }, }; -fn fixture() -> (QueryExpr, Schema, Vec) { +fn fixture() -> (QueryExpr, SchemaRef, Vec) { let schema = planner_types::pre_asap::Schema::new(vec![Field::plain("value", DataType::Int64, true)]); - let output = Arc::new(PlannerSchema { + let output = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![Field { @@ -57,7 +57,7 @@ fn fixture() -> (QueryExpr, Schema, Vec) { ]; (scan, output, batches) } -fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsapDAG { +fn plan(scan: QueryExpr, schema: &SchemaRef, state: ExecutionDataState) -> PostAsapDAG { let node = |id, payload| PostAsapDAGNode { id: PostAsapNodeId(id), payload, @@ -168,7 +168,7 @@ fn raw_scan_to_sort_limit_at_both_phases() { } } struct CountingSource { - schema: Schema, + schema: SchemaRef, opened: Arc, fail: bool, } @@ -176,7 +176,7 @@ impl RawSource for CountingSource { fn boundedness(&self) -> asap_physical_operators::plan::Boundedness { asap_physical_operators::plan::Boundedness::Bounded } - fn schema(&self) -> Schema { + fn schema(&self) -> SchemaRef { self.schema.clone() } fn scan(&self, _: RunContext) -> Result, Error> { @@ -244,15 +244,15 @@ fn binding_errors_and_reader_errors_are_not_empty_results() { }); } -// Schema drift cannot enter the DAG, and connector batches obey execution limits. +// SchemaRef drift cannot enter the DAG, and connector batches obey execution limits. #[test] fn schema_drift_and_memory_limits_fail_the_scan() { struct Drift { - expected: Schema, + expected: SchemaRef, batch: Batch, } impl RawSource for Drift { - fn schema(&self) -> Schema { + fn schema(&self) -> SchemaRef { self.expected.clone() } fn scan(&self, _: RunContext) -> Result, Error> { diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs index d018fe511..8cac1cbda 100644 --- a/crates/asap-physical-operators/tests/summary_projection.rs +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -8,7 +8,7 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema as PlannerSchema; +use planner_types::pre_asap::Schema; use planner_types::{ post_asap::*, pre_asap::{ColumnRef, DataType, ProjectItem, QueryExpr}, @@ -20,7 +20,7 @@ use std::{collections::BTreeMap, sync::Arc}; #[test] fn post_asap_summary_projection_survives_recovery() { let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let schema = Arc::new(PlannerSchema { + let schema = Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![ @@ -39,7 +39,7 @@ fn post_asap_summary_projection_survives_recovery() { ], time_index: None, }); - let output = PlannerSchema { + let output = Schema { closed: true, unique_keys: vec![], fields: vec![ diff --git a/crates/integration-tests/tests/kll_pane_execution.rs b/crates/integration-tests/tests/kll_pane_execution.rs index 31fe62dc0..65f4fc6c4 100644 --- a/crates/integration-tests/tests/kll_pane_execution.rs +++ b/crates/integration-tests/tests/kll_pane_execution.rs @@ -6,13 +6,12 @@ use asap_physical_operators::{ plan::{PhysicalDAG, PhysicalOperator, PlanProperties}, runtime::{Input, Limits, OutputStream, RunContext, Scope}, summary_kernels::datasketches_kll::DatasketchesKLLAccumulator, - values::{Batch, Schema, Value}, + values::{Batch, SchemaRef, Value}, AggregateCore, Error, }; use asap_types::{ post_asap::{ - Field, FieldDataType, Schema as PlannerSchema, SketchAlgorithm, SketchKind, SketchParams, - SketchQuery, + Field, FieldDataType, Schema, SketchAlgorithm, SketchKind, SketchParams, SketchQuery, }, pre_asap::DataType, }; @@ -31,8 +30,8 @@ fn family(k: u32) -> FieldDataType { Default::default(), ) } -fn raw_schema() -> Schema { - Arc::new(PlannerSchema { +fn raw_schema() -> SchemaRef { + Arc::new(Schema { closed: true, unique_keys: vec![], fields: vec![Field { @@ -50,7 +49,7 @@ fn query_scope() -> Scope { revision: 1, } } -fn fixture() -> (CompiledPhysicalDAG, Operator, Schema) { +fn fixture() -> (CompiledPhysicalDAG, Operator, SchemaRef) { let raw = raw_schema(); let build = Operator::summary_build(raw.clone(), family(200), 0, None, vec![]).unwrap(); let state = build.schema(); @@ -84,7 +83,7 @@ fn pane_state(maintenance: &CompiledPhysicalDAG, pane: i64) -> Arc]) -> Batch { +fn restore(schema: SchemaRef, states: &[Arc]) -> Batch { Batch::try_new( schema, states @@ -99,14 +98,14 @@ fn restore(schema: Schema, states: &[Arc]) -> Batch { ) .unwrap() } -fn readout(schema: Schema, q: f64) -> Operator { +fn readout(schema: SchemaRef, q: f64) -> Operator { Operator::readout(schema, 0, ReadoutQuery::Sketch(SketchQuery::Quantile { q })).unwrap() } struct CountStarts { operator: Operator, starts: Arc, } -impl PhysicalOperator for CountStarts { +impl PhysicalOperator for CountStarts { fn name(&self) -> &str { self.operator.name() } @@ -116,10 +115,10 @@ impl PhysicalOperator for CountStarts { fn requires_bounded_input(&self) -> bool { self.operator.requires_bounded_input() } - fn input_schemas(&self) -> Vec { + fn input_schemas(&self) -> Vec { self.operator.input_schemas() } - fn output_schema(&self) -> Schema { + fn output_schema(&self) -> SchemaRef { self.operator.output_schema() } fn output_bytes(&self, batch: &Batch) -> usize { diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index 5a28a13fc..eeade3ae5 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -1033,7 +1033,7 @@ fn execute_timed( } } } - let batch = |schema: &asap_physical_operators::values::Schema, name: &str| { + let batch = |schema: &asap_physical_operators::values::SchemaRef, name: &str| { let rows = samples .iter() .filter(|sample| sample.0 == name) diff --git a/docs/develop_docs/pre-asap-ir.md b/docs/develop_docs/pre-asap-ir.md index db5681329..abf5dc50b 100644 --- a/docs/develop_docs/pre-asap-ir.md +++ b/docs/develop_docs/pre-asap-ir.md @@ -30,7 +30,9 @@ and a runtime row for its value. Group keys, unique keys, and `time_index` also use these column positions. They are not stable identities across projections or joins, so the positional reference remains `ColumnId`, not `FieldId`. -The native runtime currently stores `Batch { schema, rows: Vec> }`. +The native runtime names shared ownership `SchemaRef = Arc` and stores +`Batch { schema: SchemaRef, rows: Vec> }`. `Schema` is the same metadata +model during planning and execution; the `Ref` suffix only distinguishes ownership. It has no physical `Column`/array container. A column reference expresses what to read independently of whether an executor stores its data as rows or arrays. For example, resolving `t.bytes` to `ColumnId = 1` obtains its type from