diff --git a/Cargo.lock b/Cargo.lock index 319fda30..7f0a3755 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -309,7 +309,7 @@ version = "0.1.0" dependencies = [ "asap-frontend-promql", "asap-types", - "asap_sketchlib", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", "serde", "serde_json", "thiserror 2.0.18", @@ -368,12 +368,29 @@ dependencies = [ "asap-aware-mapping", "asap-frontend-promql", "asap-frontend-sql", + "asap-physical-operators", "asap-types", - "asap_sketchlib", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", + "futures", "serde_json", "tokio", ] +[[package]] +name = "asap-physical-operators" +version = "0.1.0" +dependencies = [ + "asap-aware-mapping", + "asap-frontend-promql", + "asap-types", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369)", + "futures", + "serde", + "serde_json", + "thiserror 2.0.18", + "tracing", +] + [[package]] name = "asap-planner" version = "0.1.0" @@ -400,6 +417,24 @@ dependencies = [ "thiserror 2.0.18", ] +[[package]] +name = "asap_sketchlib" +version = "0.3.0" +source = "git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369#5f03ccbd798ed5fec62bdd839bcb331123cab369" +dependencies = [ + "bincode", + "bytes", + "prost", + "rand 0.9.5", + "rmp-serde", + "serde", + "serde-big-array", + "serde_bytes", + "smallvec", + "twox-hash 2.1.2", + "xxhash-rust", +] + [[package]] name = "asap_sketchlib" version = "0.3.0" diff --git a/Cargo.toml b/Cargo.toml index a1daf49d..a2b019af 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,5 +1,6 @@ [workspace] members = [ + "crates/asap-physical-operators", "crates/types", "crates/sql-function-catalog", "crates/asap-aware-mapping", diff --git a/crates/asap-aware-mapping/src/lib.rs b/crates/asap-aware-mapping/src/lib.rs index 76c3a88c..921f504e 100644 --- a/crates/asap-aware-mapping/src/lib.rs +++ b/crates/asap-aware-mapping/src/lib.rs @@ -221,13 +221,14 @@ pub use summary_maintenance_dag_export::{ }; pub use summary_maintenance_lifecycle::{ assemble_selected_dag_with_summary_maintenance_lifecycles, - global_selection_with_summary_maintenance_lifecycles, plan_summary_maintenance_lifecycles, - SummaryMaintenanceCapabilities, SummaryMaintenanceDeployment, - SummaryMaintenanceLifecycleAlternative, SummaryMaintenanceLifecycleAssemblyError, - SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecycleCostInputs, - SummaryMaintenanceLifecyclePlan, SummaryMaintenanceLifecyclePlanError, - SummaryMaintenanceLifecycleRejection, SummaryMaintenanceLifecycleSelectionError, - WorkloadDemand, + enumerate_summary_maintenance_lifecycles, global_selection_with_summary_maintenance_lifecycles, + plan_summary_maintenance_lifecycles, SummaryMaintenanceCapabilities, + SummaryMaintenanceDeployment, SummaryMaintenanceLifecycleAlternative, + SummaryMaintenanceLifecycleAssemblyError, SummaryMaintenanceLifecycleCandidates, + SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecycleChoiceError, + SummaryMaintenanceLifecycleCostInputs, SummaryMaintenanceLifecyclePlan, + SummaryMaintenanceLifecyclePlanError, SummaryMaintenanceLifecycleRejection, + SummaryMaintenanceLifecycleSelectionError, SummaryMaintenanceTimingError, WorkloadDemand, }; pub use topk_reuse::TopKLimitReuseStrategy; diff --git a/crates/asap-aware-mapping/src/maintained_population.rs b/crates/asap-aware-mapping/src/maintained_population.rs index 504c8100..29cd3c54 100644 --- a/crates/asap-aware-mapping/src/maintained_population.rs +++ b/crates/asap-aware-mapping/src/maintained_population.rs @@ -258,11 +258,15 @@ impl MaintainedPopulationStrategy { schema: input_schema.clone(), guarantee: Some(ResultGuarantee::exact("source samples")), }); + // Query time is only the initial layout: whether the population is + // retained at ingestion or rebuilt per query is its lifecycle choice + // (`SummaryMaintenanceLifecyclePlan::execution_timed_dag`). The readout + // and projection above it are query-time by construction. let maintained = Rc::new(SummaryNode { expr: SummaryExpr::ValueOperation { child: scan, operation: ValueOperation::MaintainPopulation { population }, - timing: ExecutionTiming::IngestionTime, + timing: ExecutionTiming::QueryTime, }, schema: input_schema, guarantee: Some(ResultGuarantee::exact( @@ -435,6 +439,31 @@ mod tests { assert_eq!(p.grouping, ["instance"]); assert_eq!(p.matchers[0].operation, CurrentSeriesMatch::Regex); } + // Population timing is a lifecycle choice: a retained or rebuilt + // population both validate, while its readout must stay at query time. + #[test] + fn population_timing_is_not_structural() { + let root = lower("topk(5,a)"); + let candidate = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) + .candidate(&root) + .unwrap(); + let with_timings = |population: ExecutionTiming, readout: ExecutionTiming| { + let mut node = (*candidate).clone(); + let SummaryExpr::ValueOperation { child, timing, .. } = &mut node.expr else { + unreachable!() + }; + *timing = readout; + let SummaryExpr::ValueOperation { timing, .. } = &mut Rc::make_mut(child).expr else { + unreachable!() + }; + *timing = population; + compile_post_asap_dag(&Rc::new(node)) + }; + use ExecutionTiming::{IngestionTime, QueryTime}; + assert!(with_timings(IngestionTime, QueryTime).is_ok()); + assert!(with_timings(QueryTime, QueryTime).is_ok()); + assert!(with_timings(IngestionTime, IngestionTime).is_err()); + } // A readout cannot reinterpret arbitrary rows as maintained state or exceed its producer's contract. #[test] fn malformed_population_dags_fail_closed() { diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index f8f55a1d..8c826b37 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -560,6 +560,12 @@ pub enum ReplacementProvenance { /// [`Replacement::ExactComposition`] with /// [`OperationPlacement::Maintenance`] (issue #171). ValueOperationAtIngestionTime, + /// A finalized whole-query result over rows carrying the PromQL series + /// identity, which the logical root does not expose (see + /// [`ReplacementStrategy::propose_for_root`]). Default selection never + /// commits it, because its readout must be validated and priced by + /// deployment; otherwise it would silently replace the logical plan. + RootPhysicalRealization, } /// A candidate a strategy considered for a target but refused to propose on @@ -634,6 +640,15 @@ pub trait ReplacementStrategy { domain_error: None, } } + + /// Whole-query logical alternatives for a workload root under its + /// end-to-end `target`. These may need input rows the root does not expose + /// (for example, the PromQL series identity), so + /// [`search_workload_with_targets`] asks only workload roots, once each. + /// They decide what to compute, never placement. Default: none. + fn propose_for_root(&self, _root: &Rc, _target: &AccuracyTarget) -> Proposals { + Proposals::default() + } } // ── Realization: how one AggIntent may be realised ─────────────────────── @@ -1341,114 +1356,6 @@ impl<'a> SketchAlgorithmStrategy<'a> { self.propose_with(&ranked, None) } - /// Fixed-window maintenance can finalize each series' counter state and - /// build a fresh heap or grouped Sum for that evaluation window. Deployment must provide - /// a complete, synchronized population and bind the matching window; this - /// candidate never incrementally adds one window's rates to another. - pub fn fixed_window_rate_candidates(&self, root: &Rc) -> Proposals { - fn place(node: &Rc) -> Option> { - let mut next = node.as_ref().clone(); - match &mut next.expr { - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing, - } if matches!(&child.expr, SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), - reduction: Reduction::PerEntity, child: source, .. - } if matches!(&source.expr, SummaryExpr::KeepPreAsap(source) if matches!(source.as_ref(), QueryExpr::TimeRange { .. }))) => - { - *timing = ExecutionTiming::IngestionTime; - } - SummaryExpr::ValueOperation { child, .. } - | SummaryExpr::SummaryAgg { child, .. } => *child = place(child)?, - SummaryExpr::SummaryEstimate { summary_input, .. } => { - *summary_input = place(summary_input)? - } - _ => return None, - } - Some(Rc::new(next)) - } - let mut proposals = self.propose_with(root, None); - proposals.candidates.retain_mut(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { - return false; - }; - let Ok(dag) = asap_types::post_asap::compile_post_asap_dag(node) else { - return false; - }; - if !dag.nodes.iter().any(|node| match &node.payload { - asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::Sketch(kind, _), - .. - } => matches!( - kind.algorithm(), - SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap - ), - asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), - .. - } => true, - _ => false, - }) { - return false; - } - let Some(placed) = place(node) else { - return false; - }; - if asap_types::post_asap::compile_post_asap_dag(&placed).is_err() { - return false; - } - let Ok(placed) = finalize_query_candidate(placed, root) else { - return false; - }; - candidate.replacement = Replacement::Summary(placed); - candidate - .rationale - .push_str("; fixed-window precompute over complete per-series counter states"); - true - }); - proposals - } - - /// Retain grouped Sum after a per-series Rate readout as a query-time - /// candidate alongside its complete-window maintenance placement. - pub fn query_time_rate_aggregation_candidates(&self, root: &Rc) -> Proposals { - fn query_time(node: &Rc) -> Rc { - let mut next = node.as_ref().clone(); - match &mut next.expr { - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing, - } if matches!( - &child.expr, - SummaryExpr::SummaryAgg { - family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), - .. - } - ) => - { - *timing = ExecutionTiming::QueryTime; - } - SummaryExpr::ValueOperation { child, .. } - | SummaryExpr::SummaryAgg { child, .. } => *child = query_time(child), - _ => {} - } - Rc::new(next) - } - let mut proposals = self.fixed_window_rate_candidates(root); - proposals.candidates.retain_mut(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { return false }; - if !matches!(&node.expr, SummaryExpr::ValueOperation { child, operation: ValueOperation::FinalizeExactAccumulator, .. } - if matches!(&child.expr, SummaryExpr::SummaryAgg { family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), .. })) { return false; } - candidate.replacement = Replacement::Summary(query_time(node)); - candidate.rationale = "query-time grouped Sum over complete per-series Rate readouts".into(); - true - }); - proposals - } - pub(crate) fn from_planning_inputs(planning_inputs: CandidatePlanningInputs<'a>) -> Self { Self { planning_inputs } } @@ -1705,6 +1612,37 @@ impl ReplacementStrategy for SketchAlgorithmStrategy<'_> { fn propose(&self, target: &TargetSubDAG<'_>) -> Proposals { self.propose_with(target.root, None) } + + /// Heap realizations of an instant-vector ranking (current-series TopK). + /// They rank rows that carry the complete PromQL series identity, which + /// the logical root does not expose, so each is a finalized query result + /// for the identity-carrying root. Placement variants (for example, + /// fixed-window or query-time Rate aggregation) are not listed here: the + /// lifecycle assigns timing and the physical compiler reads it. + fn propose_for_root(&self, root: &Rc, target: &AccuracyTarget) -> Proposals { + let Ok(typed) = asap_types::pre_asap::schema::with_promql_series_identity(root) else { + return Proposals::default(); + }; + let typed = Rc::new(typed); + let mut proposals = self.current_series_topk_candidates(&typed, target); + for mut candidate in std::mem::take(&mut proposals.candidates) { + let Replacement::Summary(node) = candidate.replacement else { + continue; + }; + let Ok(node) = finalize_query_candidate(node, &typed) else { + continue; + }; + let duplicate = proposals.candidates.iter().any(|existing| { + matches!(&existing.replacement, Replacement::Summary(other) if *other == node) + }); + if !duplicate { + candidate.replacement = Replacement::Summary(node); + candidate.provenance = ReplacementProvenance::RootPhysicalRealization; + proposals.candidates.push(candidate); + } + } + proposals + } } /// A human-readable rationale for one candidate `Realization`, for @@ -2718,7 +2656,9 @@ fn realize_physical_summary_input( /// Emit `SummaryAgg` (recursively binding the child), plus the /// `SummaryEstimate` readout when `estimate` is set. // Retain the exact expression and schema while placing its value production -// on the update path. Read-time consumers keep their original shared nodes. +// on the update path. This is the initial layout for values feeding a summary; +// lifecycle timing is authoritative. Read-time consumers keep their original +// shared nodes. fn maintenance_exact_values(node: Rc) -> Option> { let expr = match &node.expr { // These guards can fall back at read time, but cannot recover a parent @@ -2947,8 +2887,9 @@ fn construct_summary_agg( }; Rc::clone(child) } else if snapshot_weighted { - // A fresh query-time summary consumes this evaluation's finalized rates. - // Moving rate snapshots must never accumulate across evaluations. + // Each evaluation's finalized rates feed a fresh summary; rate snapshots + // must never accumulate across evaluations. Query time is only the + // initial layout; a retained summary's lifecycle moves it to ingestion. finalize_query_candidate(bound_child, &input.child)? } else { let child = finalize_exact_accumulator_at( @@ -5203,6 +5144,9 @@ impl<'a> GlobalSelection<'a> { if let Some(node) = self.assembled_nodes.borrow().get(&ptr) { return Ok(Rc::clone(node)); } + // A selected summary that realizes its inner aggregate, instead of + // hiding it in `KeepPreAsap`, is kept; lifecycle assignment decides + // whether it runs in precompute or at query time. let selected_composed_summary = self .groups .get(&ptr) @@ -5210,7 +5154,7 @@ impl<'a> GlobalSelection<'a> { .is_some_and(|candidate| matches!(&candidate.replacement, Replacement::Summary(node) if matches!(&node.expr, SummaryExpr::SummaryAgg { child, .. } - if matches!(&child.expr, SummaryExpr::KeepPreAsap(raw) if !contains_aggregate(raw))))); + if !matches!(&child.expr, SummaryExpr::KeepPreAsap(raw) if contains_aggregate(raw))))); let node = if query_time_nested_sum(target) && !selected_composed_summary { self.assemble_residual(target)? } else { @@ -5966,7 +5910,8 @@ fn is_cse_candidate(candidate: &ReplacementSubDAG) -> bool { } fn is_automatically_selectable(candidate: &ReplacementSubDAG, cost_model: &dyn CostModel) -> bool { - !candidate.has_missing_accuracy_evidence() + candidate.provenance != ReplacementProvenance::RootPhysicalRealization + && !candidate.has_missing_accuracy_evidence() && candidate.runtime_support_evidence(cost_model) != Some(false) } @@ -6468,6 +6413,28 @@ pub fn search_workload_with_targets<'s, Id>( .zip(targets) .filter_map(|((_, root), target)| target.map(|t| (Rc::as_ptr(root), t))) .collect(); + // Whole-root proposals join the root group before its target check. + for (index, (ptr, target)) in root_ptrs.iter().enumerate() { + if root_ptrs[..index].contains(&(*ptr, target.clone())) { + continue; + } + let group = space.groups.get_mut(ptr).expect("every root has a group"); + let root = Rc::clone(&group.target); + for strategy in strategies { + let name = strategy.name(); + let proposals = strategy.propose_for_root(&root, target); + for mut candidate in proposals.candidates { + candidate.strategy = name; + group.add_candidate(candidate); + } + group + .rejected + .extend(proposals.rejected.into_iter().map(|mut rejection| { + rejection.strategy = name; + rejection + })); + } + } let mut composition_targets: HashMap<_, Vec<_>> = HashMap::new(); for (ptr, target) in root_ptrs { composition_targets @@ -6886,6 +6853,77 @@ mod tests { use asap_types::types::AccuracyTarget; use std::collections::HashMap; + // Candidate shape without execution timing: what is computed, not where. + fn timing_free_shape(node: &Rc) -> serde_json::Value { + fn strip(value: &mut serde_json::Value) { + match value { + serde_json::Value::Object(fields) => { + fields.remove("timing"); + fields.values_mut().for_each(strip); + } + serde_json::Value::Array(values) => values.iter_mut().for_each(strip), + _ => {} + } + } + let mut shape = + serde_json::to_value(asap_types::post_asap::compile_post_asap_dag(node).unwrap()) + .unwrap(); + strip(&mut shape); + shape + } + + // Rate inventories never offer two candidates that differ only in timing. + #[test] + fn rate_candidate_inventories_have_no_timing_only_duplicates() { + for (query, accuracy) in [ + ("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact), + ("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)), + ] { + let root = Rc::new(lower_promql(query, accuracy)); + let inventory = search_workload(vec![(0usize, root)]) + .enumerate_candidate_dags(4096) + .unwrap(); + let shapes = inventory + .candidates + .iter() + .map(|forest| timing_free_shape(&forest[0].1)) + .collect::>(); + for (i, shape) in shapes.iter().enumerate() { + assert!(!shapes[..i].contains(shape), "{query}: duplicate {i}"); + } + } + } + + // Grouped Sum over Rate readouts stays a summary state in the inventory, + // so lifecycle assignment can place it in precompute or at query time. + #[test] + fn grouped_rate_sum_inventory_keeps_sum_state_for_lifecycle_placement() { + let root = Rc::new(lower_promql( + "sum by(job)(rate(m[1m]))", + AccuracyTarget::Exact, + )); + let inventory = search_workload(vec![(0usize, root)]) + .enumerate_candidate_dags(4096) + .unwrap(); + let is_exact = |node: &SummaryNode, kind: ExactKind| { + matches!(&node.expr, SummaryExpr::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(k, _), .. + } if *k == kind) + }; + assert!(inventory.candidates.iter().any(|forest| { + let SummaryExpr::ValueOperation { child: sum, .. } = &forest[0].1.expr else { + return false; + }; + let SummaryExpr::SummaryAgg { child: rate, .. } = &sum.expr else { + return false; + }; + is_exact(sum, ExactKind::Sum) + && matches!(&rate.expr, SummaryExpr::ValueOperation { + child, operation: ValueOperation::FinalizeExactAccumulator, .. + } if is_exact(child, ExactKind::Rate)) + })); + } + // Every exposed query result has a readout; internal accumulator frontiers stay states. #[test] fn query_candidate_roots_do_not_leak_exact_accumulator_state() { diff --git a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs index efc8cf13..129d06ed 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs @@ -117,7 +117,7 @@ pub fn export_summary_maintenance_plan( } /// Walk in the same post-order as `dag_export::export_summary` and attach a -/// deployment directly to every flattened occurrence of its `SummaryAgg`. +/// deployment directly to every flattened occurrence of its state node. /// This makes the decision visible to graph consumers without asking them to /// reconstruct pointer identity from graph position. fn annotate_lifecycle_deployments( diff --git a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs index b0d9d79f..cb1dd9db 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs @@ -11,7 +11,8 @@ //! //! This module enumerates and costs `Ephemeral`, `Prepared`, `Shared`, and //! `ContinuouslyMaintained` alternatives for every unique `SummaryAgg` in a -//! materialized plan. [`SummaryMaintenanceMode`] is an orthogonal detail of +//! materialized plan, and for every maintained population (`MaintainPopulation`) +//! that is not an input of a `SummaryAgg`. [`SummaryMaintenanceMode`] is an orthogonal detail of //! the selected deployment: state is either built directly or updated //! incrementally. Unknown evidence stays unknown and therefore cannot make a //! long-lived alternative win. @@ -21,9 +22,10 @@ use std::rc::Rc; use asap_types::post_asap::{ compile_post_asap_dag_with_node_ids, EvaluationSchedule, ExecutionDataStateError, - OutputRepresentation, PostAsapNodeId, ResultGuarantee, SummaryExpr, - SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, - SummaryNode, SummaryWindowFramework, + ExecutionTiming, OutputRepresentation, PostAsapDag, PostAsapDagValidationError, PostAsapNodeId, + ResultGuarantee, SummaryExpr, SummaryMaintenanceLifecycle, + SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, SummaryNode, + SummaryWindowFramework, ValueOperation, }; use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; @@ -158,14 +160,16 @@ impl SummaryMaintenanceLifecycleAlternative { } } -/// One unique summary-state deployment. Shared `Rc` nodes are emitted once. +/// One unique retained-state deployment. Shared `Rc` nodes are emitted once. #[derive(Debug, Clone)] pub struct SummaryMaintenanceDeployment { /// Identity of this summary in the exported post-ASAP semantic DAG. /// It is scoped to one plan version and is not a summary definition or /// summary instance identity. pub post_asap_node_id: PostAsapNodeId, - /// The unique materialized `SummaryAgg` represented by this deployment. + /// The unique materialized `SummaryAgg`, or maintained population + /// (`MaintainPopulation`) not consumed by a `SummaryAgg`, represented by + /// this deployment. Cost-model lifecycle hooks receive this node. pub summary: Rc, /// Lifecycle, evaluation, and representation commitment selected for this /// state, or `None` when no alternative is selectable. @@ -183,8 +187,9 @@ pub struct SummaryMaintenanceDeployment { pub struct SummaryMaintenanceLifecyclePlan { /// Root of the materialized post-ASAP DAG being deployed. pub root: Rc, - /// One entry per unique reachable `SummaryAgg`; shared `Rc` nodes appear - /// only once. + /// One entry per unique reachable `SummaryAgg`, then per unique + /// maintained population outside any `SummaryAgg`'s inputs; shared `Rc` + /// nodes appear only once. pub deployments: Vec, /// Caller-supplied optimization horizon used to turn rates into total /// costs. `None` keeps horizon-dependent alternatives unselectable. @@ -211,6 +216,89 @@ pub struct SummaryMaintenanceLifecyclePlan { pub raw_recompute_total_cost: Option, } +/// Why a lifecycle plan cannot assign execution timing to its DAG. +#[derive(Debug, thiserror::Error, PartialEq)] +pub enum SummaryMaintenanceTimingError { + #[error(transparent)] + InvalidPostAsapDag(#[from] ExecutionDataStateError), + #[error("summary {0:?} has no selected lifecycle")] + UnselectedLifecycle(PostAsapNodeId), + /// A maintained population outside any `SummaryAgg`'s inputs has no + /// deployment, so its timing would be guessed. Enumeration always emits + /// one; this arises only for a plan whose root or deployments were edited. + #[error("node {0:?} maintains state that has no summary-maintenance lifecycle")] + UnplannedMaintainedState(PostAsapNodeId), + #[error(transparent)] + InvalidPhases(#[from] PostAsapDagValidationError), +} + +impl SummaryMaintenanceLifecyclePlan { + /// The post-ASAP DAG of [`Self::root`] with every node's timing derived + /// from the selected lifecycles, so physical compilation places it. + /// + /// A retained (non-`Ephemeral`) state outlives one query, so it and every + /// input it consumes run at ingestion time. Every other node runs at query + /// time: readouts and consumers of retained state, and each `Ephemeral` + /// state not consumed by retained state together with its inputs, whose + /// raw data the deployment must supply as a query source. This applies to + /// maintained populations as to `SummaryAgg` states; a population feeding + /// a `SummaryAgg` is one of its inputs. Timings already on the root are + /// ignored. + pub fn execution_timed_dag(&self) -> Result { + let compiled = compile_post_asap_dag_with_node_ids(&self.root)?; + let dag = compiled.dag; + for population in &standalone_populations(&self.root) { + let id = compiled + .node_ids + .node_id(population) + .expect("collected population belongs to the compiled DAG"); + if !self + .deployments + .iter() + .any(|deployment| deployment.post_asap_node_id == id) + { + return Err(SummaryMaintenanceTimingError::UnplannedMaintainedState(id)); + } + } + let mut pending = Vec::new(); + for deployment in &self.deployments { + let guarantee = deployment + .summary_maintenance_lifecycle_guarantee + .as_ref() + .ok_or(SummaryMaintenanceTimingError::UnselectedLifecycle( + deployment.post_asap_node_id, + ))?; + if guarantee.summary_maintenance_lifecycle != SummaryMaintenanceLifecycle::Ephemeral { + pending.push(deployment.post_asap_node_id); + } + } + let mut ingestion = HashSet::new(); + while let Some(id) = pending.pop() { + if ingestion.insert(id) { + pending.extend( + dag.edges + .iter() + .filter(|edge| edge.consumer == id) + .map(|edge| edge.producer), + ); + } + } + let phases = dag + .nodes + .iter() + .map(|node| { + let timing = if ingestion.contains(&node.id) { + ExecutionTiming::IngestionTime + } else { + ExecutionTiming::QueryTime + }; + (node.id, timing) + }) + .collect(); + Ok(dag.with_execution_phases(&phases)?) + } +} + /// Explicit association between a materialized target and the normalized /// workload entries whose demand consumes it. /// @@ -287,6 +375,171 @@ pub enum SummaryMaintenanceLifecycleSelectionError { SummaryMaintenance(#[from] SummaryMaintenanceLifecyclePlanError), } +/// Every lifecycle alternative for each unique retained state of one fixed +/// root, before any lifecycle is chosen. +/// +/// Planner selection ([`plan_summary_maintenance_lifecycles`]) and a +/// deployment's explicit choice ([`Self::select`]) both finish from this value, +/// so they produce the same [`SummaryMaintenanceLifecyclePlan`] shape. +pub struct SummaryMaintenanceLifecycleCandidates<'a> { + /// Unselected plan: deployments carry alternatives but no guarantee or + /// window framework. + plan: SummaryMaintenanceLifecyclePlan, + components: Vec, + arrival: DataArrival, + required_accuracy: Vec, + cost_model: &'a dyn CostModel, + comparison_target: Option<&'a QueryExpr>, +} + +/// Why an explicit per-state lifecycle choice cannot be bound. +#[derive(Debug, thiserror::Error, PartialEq)] +pub enum SummaryMaintenanceLifecycleChoiceError { + #[error("summary {0:?} is not a deployment of this root")] + UnknownSummary(PostAsapNodeId), + #[error("summary {0:?} is chosen more than once")] + DuplicateChoice(PostAsapNodeId), + #[error("summary {0:?} has no chosen lifecycle")] + MissingChoice(PostAsapNodeId), + #[error("chosen lifecycle is not an enumerated alternative of summary {0:?}")] + NotAnAlternative(PostAsapNodeId), + #[error("chosen lifecycle of summary {post_asap_node_id:?} is rejected: {rejection:?}")] + Rejected { + post_asap_node_id: PostAsapNodeId, + rejection: Option, + }, + #[error("summary states on one maintenance path have different evaluation schedules")] + IncompatibleEvaluationSchedules, + #[error("the cost model supplied no complete estimate for the chosen combination")] + NoCompleteEstimate, +} + +impl SummaryMaintenanceLifecycleCandidates<'_> { + /// One entry per unique retained state (see + /// [`SummaryMaintenanceLifecyclePlan::deployments`]), with every + /// alternative and its rejection; no lifecycle or window framework is + /// selected. + pub fn deployments(&self) -> &[SummaryMaintenanceDeployment] { + &self.plan.deployments + } + + /// Guarantee that binding `lifecycle` would attach under this workload's + /// data arrival, so a caller can price an alternative before choosing it. + pub fn guarantee( + &self, + lifecycle: &SummaryMaintenanceLifecycle, + ) -> SummaryMaintenanceLifecycleGuarantee { + lifecycle_guarantee(lifecycle, self.arrival) + } + + fn context(&self) -> CompleteCostContext<'_> { + CompleteCostContext { + root: &self.plan.root, + components: &self.components, + cost_model: self.cost_model, + comparison_target: self.comparison_target, + horizon: self.plan.horizon, + expected_reads: self.plan.expected_reads, + required_accuracy: &self.required_accuracy, + } + } + + fn finish( + mut self, + estimate: Option, + ) -> SummaryMaintenanceLifecyclePlan { + if let Some(estimate) = estimate { + self.plan.summary_total_cost = Some(estimate.cost); + self.plan.selected_window_implementation_id = estimate.physical_plan_id; + self.plan.window_accuracy_guarantee = estimate.window_accuracy_guarantee; + } + self.plan + } + + /// Planner's choice: the cheapest complete combination of eligible + /// alternatives. + fn select_cheapest(mut self) -> SummaryMaintenanceLifecyclePlan { + let estimate = select_complete_lifecycle_combination( + &self.plan.root, + &mut self.plan.deployments, + &self.components, + self.arrival, + self.cost_model, + self.comparison_target, + self.plan.horizon, + self.plan.expected_reads, + &self.required_accuracy, + ); + self.finish(estimate) + } + + /// Bind one caller-chosen lifecycle per summary state. Each choice must be + /// an alternative Planner itself could select; the complete estimate is + /// then obtained exactly as for Planner selection, so window framework and + /// cost are the model's and unknown cost is never replaced by zero. + pub fn select( + mut self, + choices: &[(PostAsapNodeId, SummaryMaintenanceLifecycle)], + ) -> Result { + use SummaryMaintenanceLifecycleChoiceError as E; + let deployments = &self.plan.deployments; + let mut chosen: Vec> = + vec![None; deployments.len()]; + let context = self.context(); + for (id, lifecycle) in choices { + let index = deployments + .iter() + .position(|deployment| deployment.post_asap_node_id == *id) + .ok_or(E::UnknownSummary(*id))?; + if chosen[index].is_some() { + return Err(E::DuplicateChoice(*id)); + } + let alternative = deployments[index] + .alternatives + .iter() + .find(|alternative| alternative.summary_maintenance_lifecycle == *lifecycle) + .ok_or(E::NotAnAlternative(*id))?; + if !context.eligible(alternative) { + return Err(E::Rejected { + post_asap_node_id: *id, + rejection: alternative.rejection.clone(), + }); + } + chosen[index] = Some(alternative); + } + let selected = chosen + .into_iter() + .enumerate() + .map(|(index, alternative)| { + let alternative = + alternative.ok_or(E::MissingChoice(deployments[index].post_asap_node_id))?; + Ok(( + index, + lifecycle_guarantee(&alternative.summary_maintenance_lifecycle, self.arrival), + // Reached only for costed alternatives or when the + // complete hook is authoritative, matching Planner search. + alternative.total_cost.unwrap_or(Cost::ZERO), + )) + }) + .collect::, E>>()?; + if selected.is_empty() { + return Ok(self.finish(None)); + } + if !context.schedules_compatible(&selected) { + return Err(E::IncompatibleEvaluationSchedules); + } + let estimate = context + .estimate(deployments, &selected) + .ok_or(E::NoCompleteEstimate)?; + let guarantees = selected + .into_iter() + .map(|(index, guarantee, _)| (index, guarantee)) + .collect(); + apply_selection(&mut self.plan.deployments, guarantees, &estimate); + Ok(self.finish(Some(estimate))) + } +} + /// Workload-wide evidence derived specifically for summary-maintenance /// lifecycle enumeration and costing. /// @@ -333,7 +586,30 @@ pub fn plan_summary_maintenance_lifecycles( capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &dyn CostModel, ) -> Result { - plan_summary_maintenance_lifecycles_with_profile( + Ok(enumerate_summary_maintenance_lifecycles( + root, + demand, + now_ms, + horizon, + capabilities, + cost_model, + )? + .select_cheapest()) +} + +/// Validate a materialized plan and enumerate lifecycle alternatives for each +/// unique summary state without choosing one. A deployment that prices the +/// alternatives itself binds its choice with +/// [`SummaryMaintenanceLifecycleCandidates::select`]. +pub fn enumerate_summary_maintenance_lifecycles<'a>( + root: Rc, + demand: WorkloadDemand<'_>, + now_ms: u64, + horizon: Option, + capabilities: SummaryMaintenanceLifecycleCapabilities, + cost_model: &'a dyn CostModel, +) -> Result, SummaryMaintenanceLifecyclePlanError> { + enumerate_with_profile( root, demand, now_ms, @@ -349,16 +625,16 @@ pub fn plan_summary_maintenance_lifecycles( /// eligibility and data-arrival facts; `profile` supplies effective uses after /// DAG path multiplicity has been propagated by `PlanSpace`. #[expect(clippy::too_many_arguments, reason = "internal bound planning context")] -fn plan_summary_maintenance_lifecycles_with_profile( +fn enumerate_with_profile<'a>( root: Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, capabilities: SummaryMaintenanceLifecycleCapabilities, - cost_model: &dyn CostModel, + cost_model: &'a dyn CostModel, profile: Option, - comparison_target: Option<&QueryExpr>, -) -> Result { + comparison_target: Option<&'a QueryExpr>, +) -> Result, SummaryMaintenanceLifecyclePlanError> { demand.workload.validate()?; if let Some(data) = demand.data_workload { data.validate()?; @@ -389,10 +665,16 @@ fn plan_summary_maintenance_lifecycles_with_profile( }; } let mut summaries = Vec::new(); - collect_summary_aggs(&root, &mut HashSet::new(), &mut summaries); + collect_states( + &root, + &mut HashSet::new(), + &mut summaries, + StateKind::SummaryAgg, + ); + summaries.extend(standalone_populations(&root)); let node_ids = compile_post_asap_dag_with_node_ids(&root)?.node_ids; let components = summary_state_components(&summaries); - let mut deployments: Vec = summaries + let deployments: Vec = summaries .into_iter() .map(|summary| { let alternatives = alternatives_for( @@ -413,37 +695,26 @@ fn plan_summary_maintenance_lifecycles_with_profile( } }) .collect(); - let complete_estimate = select_complete_lifecycle_combination( - &root, - &mut deployments, - &components, - facts.arrival, + let selected_raw_recompute = matches!(root.expr, SummaryExpr::KeepPreAsap(_)); + Ok(SummaryMaintenanceLifecycleCandidates { + plan: SummaryMaintenanceLifecyclePlan { + root, + deployments, + horizon, + evaluation_rate: facts.evaluation_rate, + update_rate: facts.update_rate, + expected_reads: facts.reads, + selected_raw_recompute, + selected_window_implementation_id: None, + summary_total_cost: None, + window_accuracy_guarantee: None, + raw_recompute_total_cost: None, + }, + components, + arrival: facts.arrival, + required_accuracy: facts.required_accuracy, cost_model, comparison_target, - horizon, - facts.reads, - &facts.required_accuracy, - ); - let summary_total_cost = complete_estimate.as_ref().map(|estimate| estimate.cost); - let selected_window_implementation_id = complete_estimate - .as_ref() - .and_then(|estimate| estimate.physical_plan_id.clone()); - let window_accuracy_guarantee = complete_estimate - .as_ref() - .and_then(|estimate| estimate.window_accuracy_guarantee.clone()); - let selected_raw_recompute = matches!(root.expr, SummaryExpr::KeepPreAsap(_)); - Ok(SummaryMaintenanceLifecyclePlan { - root, - deployments, - horizon, - evaluation_rate: facts.evaluation_rate, - update_rate: facts.update_rate, - expected_reads: facts.reads, - selected_raw_recompute, - selected_window_implementation_id, - summary_total_cost, - window_accuracy_guarantee, - raw_recompute_total_cost: None, }) } @@ -483,7 +754,7 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( continue; }; costs.finalize_target(&group.target); - let plan = plan_summary_maintenance_lifecycles_with_profile( + let plan = enumerate_with_profile( Rc::clone(summary), WorkloadDemand { workload, @@ -496,7 +767,8 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( cost_model, Some(profiles.for_target(&group.target)), Some(&group.target), - )?; + )? + .select_cheapest(); let raw = plan .expected_reads .and_then(|reads| cost_model.raw_query_recompute_total_cost(&group.target, reads)); @@ -529,7 +801,7 @@ pub fn assemble_selected_dag_with_summary_maintenance_lifecycles( selection .assemble_selected_dag(target)? .map(|root| { - let mut plan = plan_summary_maintenance_lifecycles_with_profile( + let mut plan = enumerate_with_profile( root, demand, now_ms, @@ -538,7 +810,8 @@ pub fn assemble_selected_dag_with_summary_maintenance_lifecycles( cost_model, None, Some(target), - )?; + )? + .select_cheapest(); plan.raw_recompute_total_cost = plan .expected_reads .and_then(|reads| cost_model.raw_query_recompute_total_cost(target, reads)); @@ -993,20 +1266,39 @@ fn rejected( } } -fn collect_summary_aggs( +#[derive(Clone, Copy, PartialEq)] +enum StateKind { + SummaryAgg, + Population, +} + +/// Collect every unique node of `kind` reachable from `node`. +fn collect_states( node: &Rc, seen: &mut HashSet<*const SummaryNode>, output: &mut Vec>, + kind: StateKind, ) { if !seen.insert(Rc::as_ptr(node)) { return; } match &node.expr { SummaryExpr::SummaryAgg { child, .. } => { - output.push(Rc::clone(node)); - collect_summary_aggs(child, seen, output); + if kind == StateKind::SummaryAgg { + output.push(Rc::clone(node)); + } + collect_states(child, seen, output, kind); + } + SummaryExpr::ValueOperation { + child, operation, .. + } => { + if kind == StateKind::Population + && matches!(operation, ValueOperation::MaintainPopulation { .. }) + { + output.push(Rc::clone(node)); + } + collect_states(child, seen, output, kind) } - SummaryExpr::ValueOperation { child, .. } => collect_summary_aggs(child, seen, output), SummaryExpr::SummaryJoin { outer, inner, .. } | SummaryExpr::RelationalJoin { left: outer, @@ -1022,22 +1314,50 @@ fn collect_summary_aggs( left: outer, right: inner, } => { - collect_summary_aggs(outer, seen, output); - collect_summary_aggs(inner, seen, output); + collect_states(outer, seen, output, kind); + collect_states(inner, seen, output, kind); } SummaryExpr::SummaryDelete { summary_input, .. } | SummaryExpr::SummaryEstimate { summary_input, .. } => { - collect_summary_aggs(summary_input, seen, output) + collect_states(summary_input, seen, output, kind) } SummaryExpr::SummaryMerge { children, .. } => { for child in children { - collect_summary_aggs(child, seen, output); + collect_states(child, seen, output, kind); } } SummaryExpr::KeepPreAsap(_) => {} } } +/// Maintained populations that are not an input of any `SummaryAgg`. A +/// population feeding summary state is on that state's maintenance path, so +/// that state's lifecycle times it, even when a readout also reads it directly. +fn standalone_populations(root: &Rc) -> Vec> { + let mut summaries = Vec::new(); + collect_states( + root, + &mut HashSet::new(), + &mut summaries, + StateKind::SummaryAgg, + ); + let mut nested = Vec::new(); + let mut seen = HashSet::new(); + for summary in &summaries { + collect_states(summary, &mut seen, &mut nested, StateKind::Population); + } + let nested: HashSet<_> = nested.iter().map(Rc::as_ptr).collect(); + let mut populations = Vec::new(); + collect_states( + root, + &mut HashSet::new(), + &mut populations, + StateKind::Population, + ); + populations.retain(|population| !nested.contains(&Rc::as_ptr(population))); + populations +} + pub(crate) fn evaluation_schedule( lifecycle: &SummaryMaintenanceLifecycle, arrival: DataArrival, @@ -1060,7 +1380,7 @@ pub(crate) fn evaluation_schedule( } /// Summary states composed on one maintenance path must be produced on the -/// same schedule. Return a component id for each collected `SummaryAgg`. +/// same schedule. Return a component id for each collected state. fn summary_state_components(summaries: &[Rc]) -> Vec { let indices: HashMap<_, _> = summaries .iter() @@ -1091,7 +1411,12 @@ fn summary_state_components(summaries: &[Rc]) -> Vec { continue; } let mut descendants = Vec::new(); - collect_summary_aggs(child, &mut HashSet::new(), &mut descendants); + collect_states( + child, + &mut HashSet::new(), + &mut descendants, + StateKind::SummaryAgg, + ); for descendant in descendants { let child_index = indices[&Rc::as_ptr(&descendant)]; let parent_root = find(&mut parents, parent_index); @@ -1104,6 +1429,99 @@ fn summary_state_components(summaries: &[Rc]) -> Vec { .collect() } +/// Inputs shared by every complete lifecycle-combination evaluation of one +/// root, whether Planner searches combinations or a caller supplies one. +struct CompleteCostContext<'a> { + root: &'a SummaryNode, + components: &'a [usize], + cost_model: &'a dyn CostModel, + comparison_target: Option<&'a QueryExpr>, + horizon: Option, + expected_reads: Option, + required_accuracy: &'a [AccuracyTarget], +} + +impl CompleteCostContext<'_> { + /// Planner's own admission rule for one alternative. Uncosted alternatives + /// are admitted only when the complete-candidate hook is authoritative. + fn eligible(&self, alternative: &SummaryMaintenanceLifecycleAlternative) -> bool { + alternative.selectable() + || (self + .cost_model + .complete_summary_candidate_estimate_covers_lifecycle_costs() + && alternative.rejection + == Some(SummaryMaintenanceLifecycleRejection::MissingCostEvidence)) + } + + /// `selected` holds one entry per deployment, in deployment order. + fn schedules_compatible( + &self, + selected: &[(usize, SummaryMaintenanceLifecycleGuarantee, Cost)], + ) -> bool { + !selected.iter().enumerate().any(|(left, (_, a, _))| { + selected.iter().enumerate().any(|(right, (_, b, _))| { + self.components[left] == self.components[right] + && a.evaluation_schedule != b.evaluation_schedule + }) + }) + } + + fn estimate( + &self, + deployments: &[SummaryMaintenanceDeployment], + selected: &[(usize, SummaryMaintenanceLifecycleGuarantee, Cost)], + ) -> Option { + if !self.schedules_compatible(selected) { + return None; + } + let costed: Vec<_> = selected + .iter() + .map(|(index, guarantee, cost)| CostedSummaryDeployment { + summary: &deployments[*index].summary, + guarantee, + selected_cost: *cost, + }) + .collect(); + let estimate = self.cost_model.complete_summary_candidate_estimate( + self.root, + self.comparison_target, + &costed, + self.horizon, + self.expected_reads, + self.required_accuracy, + )?; + (estimate.window_frameworks.len() == deployments.len()).then_some(estimate) + } +} + +fn lifecycle_guarantee( + lifecycle: &SummaryMaintenanceLifecycle, + arrival: DataArrival, +) -> SummaryMaintenanceLifecycleGuarantee { + SummaryMaintenanceLifecycleGuarantee { + summary_maintenance_mode: maintenance_mode(lifecycle, arrival), + evaluation_schedule: evaluation_schedule(lifecycle, arrival), + summary_maintenance_lifecycle: lifecycle.clone(), + output_representation: OutputRepresentation::SummaryState, + } +} + +fn apply_selection( + deployments: &mut [SummaryMaintenanceDeployment], + guarantees: Vec<(usize, SummaryMaintenanceLifecycleGuarantee)>, + estimate: &CompleteSummaryCandidateEstimate, +) { + for (index, guarantee) in guarantees { + deployments[index].summary_maintenance_lifecycle_guarantee = Some(guarantee); + } + for (deployment, framework) in deployments + .iter_mut() + .zip(estimate.window_frameworks.iter().cloned()) + { + deployment.selected_window_framework = framework; + } +} + #[expect(clippy::too_many_arguments, reason = "complete combination context")] fn select_complete_lifecycle_combination( root: &SummaryNode, @@ -1120,82 +1538,48 @@ fn select_complete_lifecycle_combination( if deployments.is_empty() { return None; } + let context = CompleteCostContext { + root, + components, + cost_model, + comparison_target, + horizon, + expected_reads, + required_accuracy, + }; // The whole-candidate hook is intentionally arbitrary, so partial costs // cannot soundly prune the search. Bound exhaustive enumeration and fail // closed instead of allowing an adversarial DAG to consume exponential // planner time. - let complete_costing = cost_model.complete_summary_candidate_estimate_covers_lifecycle_costs(); - let eligible = |alternative: &SummaryMaintenanceLifecycleAlternative| { - alternative.selectable() - || (complete_costing - && alternative.rejection - == Some(SummaryMaintenanceLifecycleRejection::MissingCostEvidence)) - }; let combinations = deployments .iter() .try_fold(1_usize, |product, deployment| { let selectable = deployment .alternatives .iter() - .filter(|alternative| eligible(alternative)) + .filter(|alternative| context.eligible(alternative)) .count(); product.checked_mul(selectable) })?; if combinations == 0 || combinations > MAX_COMPLETE_LIFECYCLE_COMBINATIONS { return None; } - #[expect( - clippy::too_many_arguments, - reason = "recursive lifecycle-combination search state" - )] + type Best = Option<( + CompleteSummaryCandidateEstimate, + Vec<(usize, SummaryMaintenanceLifecycleGuarantee)>, + )>; fn visit( index: usize, - root: &SummaryNode, + context: &CompleteCostContext<'_>, deployments: &[SummaryMaintenanceDeployment], - components: &[usize], arrival: DataArrival, - cost_model: &dyn CostModel, - comparison_target: Option<&QueryExpr>, - horizon: Option, - expected_reads: Option, - required_accuracy: &[AccuracyTarget], - complete_costing: bool, selected: &mut Vec<(usize, SummaryMaintenanceLifecycleGuarantee, Cost)>, - best: &mut Option<( - CompleteSummaryCandidateEstimate, - Vec<(usize, SummaryMaintenanceLifecycleGuarantee)>, - )>, + best: &mut Best, ) { if index == deployments.len() { - if selected.iter().enumerate().any(|(left, (_, a, _))| { - selected.iter().enumerate().any(|(right, (_, b, _))| { - components[left] == components[right] - && a.evaluation_schedule != b.evaluation_schedule - }) - }) { - return; - } - let costed: Vec<_> = selected - .iter() - .map(|(index, guarantee, cost)| CostedSummaryDeployment { - summary: &deployments[*index].summary, - guarantee, - selected_cost: *cost, - }) - .collect(); - let Some(estimate) = cost_model.complete_summary_candidate_estimate( - root, - comparison_target, - &costed, - horizon, - expected_reads, - required_accuracy, - ) else { + let Some(estimate) = context.estimate(deployments, selected) else { return; }; - if estimate.window_frameworks.len() != deployments.len() { - return; - } if best .as_ref() .is_none_or(|(best_estimate, _)| estimate.cost.0 < best_estimate.cost.0) @@ -1213,40 +1597,14 @@ fn select_complete_lifecycle_combination( for alternative in deployments[index] .alternatives .iter() - .filter(|alternative| { - alternative.selectable() - || (complete_costing - && alternative.rejection - == Some(SummaryMaintenanceLifecycleRejection::MissingCostEvidence)) - }) + .filter(|alternative| context.eligible(alternative)) { - let lifecycle = alternative.summary_maintenance_lifecycle.clone(); - let guarantee = SummaryMaintenanceLifecycleGuarantee { - summary_maintenance_mode: maintenance_mode(&lifecycle, arrival), - evaluation_schedule: evaluation_schedule(&lifecycle, arrival), - summary_maintenance_lifecycle: lifecycle, - output_representation: OutputRepresentation::SummaryState, - }; selected.push(( index, - guarantee, + lifecycle_guarantee(&alternative.summary_maintenance_lifecycle, arrival), alternative.total_cost.unwrap_or(Cost::ZERO), )); - visit( - index + 1, - root, - deployments, - components, - arrival, - cost_model, - comparison_target, - horizon, - expected_reads, - required_accuracy, - complete_costing, - selected, - best, - ); + visit(index + 1, context, deployments, arrival, selected, best); selected.pop(); } } @@ -1254,29 +1612,14 @@ fn select_complete_lifecycle_combination( let mut best = None; visit( 0, - root, + &context, deployments, - components, arrival, - cost_model, - comparison_target, - horizon, - expected_reads, - required_accuracy, - complete_costing, &mut Vec::new(), &mut best, ); let (estimate, guarantees) = best?; - for (index, guarantee) in guarantees { - deployments[index].summary_maintenance_lifecycle_guarantee = Some(guarantee); - } - for (deployment, framework) in deployments - .iter_mut() - .zip(estimate.window_frameworks.iter().cloned()) - { - deployment.selected_window_framework = framework; - } + apply_selection(deployments, guarantees, &estimate); Some(estimate) } @@ -1323,8 +1666,8 @@ mod tests { } use super::*; use asap_types::post_asap::{ - ExactKind, ExactParams, GroupingStrategy, ResultGuarantee, SketchAlgorithm, - SummaryFamilyType, SummaryField, SummarySchema, + ExactKind, ExactParams, GroupingStrategy, PostAsapOperatorPayload, ResultGuarantee, + SketchAlgorithm, SummaryFamilyType, SummaryField, SummarySchema, }; use asap_types::pre_asap::AggIntent; use asap_types::pre_asap::{Column, ColumnRef, DataType, QueryExpr, Reduction, Schema, Source}; @@ -2402,4 +2745,842 @@ mod tests { assert_eq!(batch.evaluation_rate, None); assert_eq!(batch.one_shot_consumers, 1); } + + fn continuous_candidates<'a>( + workload: &QueryWorkload, + data: &DataWorkload, + model: &'a dyn CostModel, + ) -> SummaryMaintenanceLifecycleCandidates<'a> { + enumerate_summary_maintenance_lifecycles( + summary(), + WorkloadDemand::new_with_data(workload, data, &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities { + supports_shared: false, + ..SummaryMaintenanceLifecycleCapabilities::ALL + }, + model, + ) + .unwrap() + } + + fn choose( + candidates: &SummaryMaintenanceLifecycleCandidates<'_>, + lifecycle: SummaryMaintenanceLifecycle, + ) -> Vec<(PostAsapNodeId, SummaryMaintenanceLifecycle)> { + candidates + .deployments() + .iter() + .map(|deployment| (deployment.post_asap_node_id, lifecycle.clone())) + .collect() + } + + // Enumeration reports all four lifecycle kinds with their rejections and + // selects nothing. + #[test] + fn enumeration_exposes_every_lifecycle_without_selecting() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let candidates = continuous_candidates(&workload, &data, &UnitCosts); + let [deployment] = candidates.deployments() else { + panic!("one summary state"); + }; + assert_eq!(deployment.summary_maintenance_lifecycle_guarantee, None); + assert_eq!(deployment.selected_window_framework, None); + let outcome: Vec<_> = deployment + .alternatives + .iter() + .map(|alternative| { + ( + &alternative.summary_maintenance_lifecycle, + alternative.rejection.clone(), + alternative.total_cost.is_some(), + ) + }) + .collect(); + assert!(matches!( + outcome.as_slice(), + [ + (SummaryMaintenanceLifecycle::Ephemeral, None, true), + ( + SummaryMaintenanceLifecycle::Prepared { .. }, + Some(SummaryMaintenanceLifecycleRejection::RequiresPredictableOneTimeQuery), + false + ), + ( + SummaryMaintenanceLifecycle::Shared { .. }, + Some(SummaryMaintenanceLifecycleRejection::UnsupportedByRuntime), + false + ), + ( + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + None, + true + ), + ] + )); + let guarantee = candidates.guarantee(&SummaryMaintenanceLifecycle::ContinuouslyMaintained); + assert_eq!( + guarantee.summary_maintenance_mode, + SummaryMaintenanceMode::Incremental + ); + assert_eq!(guarantee.evaluation_schedule, EvaluationSchedule::PerUpdate); + } + + // Explicitly choosing Planner's own selection reproduces Planner's plan. + #[test] + fn explicit_choice_of_planner_selection_reproduces_planner_plan() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let planned = plan_summary_maintenance_lifecycles( + summary(), + WorkloadDemand::new_with_data(&workload, &data, &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities { + supports_shared: false, + ..SummaryMaintenanceLifecycleCapabilities::ALL + }, + &UnitCosts, + ) + .unwrap(); + let candidates = continuous_candidates(&workload, &data, &UnitCosts); + let choice = choose( + &candidates, + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + ); + let chosen = candidates.select(&choice).unwrap(); + assert_eq!(format!("{chosen:?}"), format!("{planned:?}")); + } + + // A deployment may bind a legal alternative Planner's estimate does not + // prefer; the plan carries that alternative's guarantee and cost. + #[test] + fn explicit_choice_may_bind_a_costlier_legal_alternative() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let candidates = continuous_candidates(&workload, &data, &UnitCosts); + let ephemeral_cost = candidates.deployments()[0].alternatives[0].total_cost; + let choice = choose(&candidates, SummaryMaintenanceLifecycle::Ephemeral); + let plan = candidates.select(&choice).unwrap(); + assert_eq!( + selected_summary_maintenance_lifecycle(&plan.deployments[0]), + Some(&SummaryMaintenanceLifecycle::Ephemeral) + ); + assert_eq!(plan.summary_total_cost, ephemeral_cost); + } + + // Choices that Planner could not select, or that do not cover exactly the + // enumerated states, are refused rather than bound. + #[test] + fn explicit_choice_rejects_illegal_or_incomplete_choices() { + use SummaryMaintenanceLifecycleChoiceError as E; + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let select = |model: &dyn CostModel, choice: &dyn Fn(PostAsapNodeId) -> Vec<_>| { + let candidates = continuous_candidates(&workload, &data, model); + let id = candidates.deployments()[0].post_asap_node_id; + (id, candidates.select(&choice(id)).unwrap_err()) + }; + let shared = SummaryMaintenanceLifecycle::Shared { + retention: DurationMs(10_000), + }; + let (id, error) = select(&UnitCosts, &|id| vec![(id, shared.clone())]); + assert_eq!( + error, + E::Rejected { + post_asap_node_id: id, + rejection: Some(SummaryMaintenanceLifecycleRejection::UnsupportedByRuntime), + } + ); + let continuous = SummaryMaintenanceLifecycle::ContinuouslyMaintained; + let (id, error) = select(&crate::cost_model::DefaultCostModel, &|id| { + vec![(id, SummaryMaintenanceLifecycle::Ephemeral)] + }); + assert_eq!( + error, + E::Rejected { + post_asap_node_id: id, + rejection: Some(SummaryMaintenanceLifecycleRejection::MissingCostEvidence), + } + ); + let (id, error) = select(&UnitCosts, &|id| { + vec![( + id, + SummaryMaintenanceLifecycle::Shared { + retention: DurationMs(1), + }, + )] + }); + assert_eq!(error, E::NotAnAlternative(id)); + let (id, error) = select(&UnitCosts, &|_| vec![]); + assert_eq!(error, E::MissingChoice(id)); + let (id, error) = select(&UnitCosts, &|id| { + vec![(id, continuous.clone()), (id, continuous.clone())] + }); + assert_eq!(error, E::DuplicateChoice(id)); + let (_, error) = select(&UnitCosts, &|_| { + vec![(PostAsapNodeId(u32::MAX), continuous.clone())] + }); + assert_eq!(error, E::UnknownSummary(PostAsapNodeId(u32::MAX))); + } + + // Nested states on one maintenance path must share an evaluation schedule. + #[test] + fn explicit_choice_rejects_incompatible_nested_schedules() { + let workload = workload(vec![], vec![repeating()], continuous(1_000, 20_000)); + let candidates = enumerate_summary_maintenance_lifecycles( + nested_summary(), + WorkloadDemand::new_with_data(&workload, &continuous(1_000, 20_000), &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &IncompatibleNestedCosts, + ) + .unwrap(); + let [outer, inner] = candidates.deployments() else { + panic!("two summary states"); + }; + let choice = vec![ + ( + outer.post_asap_node_id, + SummaryMaintenanceLifecycle::Ephemeral, + ), + ( + inner.post_asap_node_id, + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + ), + ]; + assert_eq!( + candidates.select(&choice).unwrap_err(), + SummaryMaintenanceLifecycleChoiceError::IncompatibleEvaluationSchedules + ); + } + + // A multi-summary root yields one candidate entry per unique state, with + // a shared `Rc` state listed once. + #[test] + fn enumeration_lists_each_unique_summary_state_once() { + let shared = summary(); + let root = Rc::new(SummaryNode { + expr: SummaryExpr::SummaryMerge { + timing: asap_types::post_asap::ExecutionTiming::IngestionTime, + children: vec![Rc::clone(&shared), Rc::clone(&shared), summary()], + }, + schema: shared.schema.clone(), + guarantee: None, + }); + let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); + let candidates = enumerate_summary_maintenance_lifecycles( + root, + WorkloadDemand::new_with_data(&workload, &at_rest(), &[0]), + 1_000, + None, + SummaryMaintenanceLifecycleCapabilities::ALL, + &UnitCosts, + ) + .unwrap(); + let ids: HashSet<_> = candidates + .deployments() + .iter() + .map(|deployment| deployment.post_asap_node_id) + .collect(); + assert_eq!(candidates.deployments().len(), 2); + assert_eq!(ids.len(), 2); + assert!(candidates + .deployments() + .iter() + .any(|deployment| Rc::ptr_eq(&deployment.summary, &shared))); + } + + fn readout(state: &Rc) -> Rc { + Rc::new(SummaryNode { + expr: SummaryExpr::ValueOperation { + child: Rc::clone(state), + operation: ValueOperation::FinalizeExactAccumulator, + timing: ExecutionTiming::QueryTime, + }, + schema: SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }, + guarantee: Some(ResultGuarantee::exact("sum")), + }) + } + + fn lifecycle_matching( + alternatives: &[SummaryMaintenanceLifecycleAlternative], + kind: fn(&SummaryMaintenanceLifecycle) -> bool, + ) -> SummaryMaintenanceLifecycle { + alternatives + .iter() + .map(|alternative| &alternative.summary_maintenance_lifecycle) + .find(|lifecycle| kind(lifecycle)) + .expect("lifecycle kind is an alternative") + .clone() + } + + /// Bind the lifecycle `choose` picks for every state of `root`, then + /// derive the timed DAG. + fn timed_dag( + root: Rc, + workload: &QueryWorkload, + data: &DataWorkload, + horizon: Option, + choose: impl Fn(&SummaryMaintenanceDeployment) -> SummaryMaintenanceLifecycle, + ) -> PostAsapDag { + let candidates = enumerate_summary_maintenance_lifecycles( + root, + WorkloadDemand::new_with_data(workload, data, &[0]), + 1_000, + horizon, + SummaryMaintenanceLifecycleCapabilities::ALL, + &UnitCosts, + ) + .unwrap(); + let choice: Vec<_> = candidates + .deployments() + .iter() + .map(|deployment| (deployment.post_asap_node_id, choose(deployment))) + .collect(); + let dag = candidates + .select(&choice) + .unwrap() + .execution_timed_dag() + .unwrap(); + dag.validate().unwrap(); + dag + } + + /// Operator kinds in node-id order, each paired with its timing. + fn timings(dag: &PostAsapDag) -> Vec<(&'static str, ExecutionTiming)> { + dag.nodes + .iter() + .map(|node| { + let kind = match node.payload { + PostAsapOperatorPayload::Fallback { .. } => "raw", + PostAsapOperatorPayload::SummaryAgg { .. } => "state", + PostAsapOperatorPayload::Value { .. } => "readout", + PostAsapOperatorPayload::Binary { .. } => "binary", + _ => "other", + }; + (kind, node.output_state.timing) + }) + .collect() + } + + const INGEST: ExecutionTiming = ExecutionTiming::IngestionTime; + const QUERY: ExecutionTiming = ExecutionTiming::QueryTime; + + // Every retained lifecycle kind runs its state and inputs at ingestion + // time and its readout at query time. + #[test] + fn retained_lifecycles_time_state_and_inputs_at_ingestion() { + let mut scheduled = batch(Predictability::Predictable { + known_at: Some(TimestampMs(1_000)), + }); + scheduled.execute_at = Some(TimestampMs(11_000)); + type Case = ( + QueryWorkload, + DataWorkload, + Option, + fn(&SummaryMaintenanceLifecycle) -> bool, + ); + let cases: [Case; 3] = [ + ( + workload(vec![], vec![repeating()], continuous(1_000, 60_000)), + continuous(1_000, 60_000), + Some(Horizon(10.0)), + |lifecycle| { + matches!( + lifecycle, + SummaryMaintenanceLifecycle::ContinuouslyMaintained + ) + }, + ), + ( + workload(vec![], vec![repeating()], at_rest()), + at_rest(), + Some(Horizon(10.0)), + |lifecycle| matches!(lifecycle, SummaryMaintenanceLifecycle::Shared { .. }), + ), + ( + workload(vec![scheduled], vec![], at_rest()), + at_rest(), + None, + |lifecycle| matches!(lifecycle, SummaryMaintenanceLifecycle::Prepared { .. }), + ), + ]; + for (workload, data, horizon, kind) in cases { + let dag = timed_dag( + readout(&summary()), + &workload, + &data, + horizon, + |deployment| lifecycle_matching(&deployment.alternatives, kind), + ); + assert_eq!( + timings(&dag), + [("raw", INGEST), ("state", INGEST), ("readout", QUERY)] + ); + } + } + + // An Ephemeral state, its raw input, and its readout all run at query time. + #[test] + fn ephemeral_lifecycle_times_state_and_downstream_at_query() { + let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); + let dag = timed_dag(readout(&summary()), &workload, &at_rest(), None, |_| { + SummaryMaintenanceLifecycle::Ephemeral + }); + assert_eq!( + timings(&dag), + [("raw", QUERY), ("state", QUERY), ("readout", QUERY)] + ); + } + + // One state read by two consumers is one deployment; its timing follows + // that single choice while both consumers run at query time. + #[test] + fn shared_state_is_timed_once_for_all_consumers() { + let state = summary(); + let lhs = readout(&state); + let rhs = Rc::new(lhs.as_ref().clone()); + let root = Rc::new(SummaryNode { + expr: SummaryExpr::BinaryOp { + lhs, + rhs, + operator: asap_types::post_asap::BinaryOperator { + kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( + asap_types::pre_asap::ArithmeticOpKind::Add, + ), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + timing: QUERY, + }, + schema: readout(&state).schema.clone(), + guarantee: None, + }); + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let dag = timed_dag(root, &workload, &data, Some(Horizon(10.0)), |deployment| { + assert!(Rc::ptr_eq(&deployment.summary, &state)); + SummaryMaintenanceLifecycle::ContinuouslyMaintained + }); + assert_eq!( + timings(&dag), + [ + ("raw", INGEST), + ("state", INGEST), + ("readout", QUERY), + ("readout", QUERY), + ("binary", QUERY), + ] + ); + } + + // An Ephemeral state consumed by retained state is built on the retained + // state's ingestion path; it is not retained, but cannot run at query time. + #[test] + fn ephemeral_state_feeding_retained_state_runs_at_ingestion() { + let mut scheduled = batch(Predictability::Predictable { + known_at: Some(TimestampMs(1_000)), + }); + scheduled.execute_at = Some(TimestampMs(11_000)); + let root = nested_summary(); + let workload = workload(vec![scheduled], vec![], at_rest()); + let dag = timed_dag( + Rc::clone(&root), + &workload, + &at_rest(), + None, + |deployment| { + if Rc::ptr_eq(&deployment.summary, &root) { + lifecycle_matching(&deployment.alternatives, |lifecycle| { + matches!(lifecycle, SummaryMaintenanceLifecycle::Prepared { .. }) + }) + } else { + SummaryMaintenanceLifecycle::Ephemeral + } + }, + ); + assert_eq!( + timings(&dag), + [("raw", INGEST), ("state", INGEST), ("state", INGEST)] + ); + } + + // Timing is not derived for a state without a selected lifecycle, and a + // raw-recompute plan runs entirely at query time. + #[test] + fn timing_requires_a_selected_lifecycle_for_every_state() { + let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); + let data = at_rest(); + let demand = WorkloadDemand::new_with_data(&workload, &data, &[0]); + let plan = plan_summary_maintenance_lifecycles( + readout(&summary()), + demand, + 1_000, + None, + SummaryMaintenanceLifecycleCapabilities::ALL, + &crate::cost_model::DefaultCostModel, + ) + .unwrap(); + assert_eq!( + plan.execution_timed_dag().unwrap_err(), + SummaryMaintenanceTimingError::UnselectedLifecycle( + plan.deployments[0].post_asap_node_id + ) + ); + let raw = plan_summary_maintenance_lifecycles( + crate::replacement::keep_pre_asap(&sum_query()).unwrap(), + demand, + 1_000, + None, + SummaryMaintenanceLifecycleCapabilities::ALL, + &UnitCosts, + ) + .unwrap(); + assert_eq!( + timings(&raw.execution_timed_dag().unwrap()), + [("raw", QUERY)] + ); + } + + /// A strategy-built `sum(a)` over one maintained current-series population. + fn population_readout() -> Rc { + let target = Rc::new(crate::test_support::lower_promql( + "sum(a)", + AccuracyTarget::Exact, + )); + crate::maintained_population::MaintainedPopulationStrategy::new(std::slice::from_ref( + &target, + )) + .candidate(&target) + .unwrap() + } + + fn is_population(node: &SummaryNode) -> bool { + matches!( + node.expr, + SummaryExpr::ValueOperation { + operation: ValueOperation::MaintainPopulation { .. }, + .. + } + ) + } + + fn population_timings(dag: &PostAsapDag) -> Vec<(&'static str, ExecutionTiming)> { + dag.nodes + .iter() + .zip(timings(dag)) + .map(|(node, (kind, timing))| match node.payload { + PostAsapOperatorPayload::Value { + operation: ValueOperation::MaintainPopulation { .. }, + } => ("population", timing), + _ => (kind, timing), + }) + .collect() + } + + // A maintained population is enumerated as retained state, with costs + // from the caller's model for both the maintained and the rebuilt choice. + #[test] + fn enumeration_includes_maintained_population() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let candidates = enumerate_summary_maintenance_lifecycles( + population_readout(), + WorkloadDemand::new_with_data(&workload, &data, &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &UnitCosts, + ) + .unwrap(); + let [deployment] = candidates.deployments() else { + panic!("one population state"); + }; + assert!(is_population(&deployment.summary)); + let cost = |lifecycle: SummaryMaintenanceLifecycle| { + deployment + .alternatives + .iter() + .find(|alternative| alternative.summary_maintenance_lifecycle == lifecycle) + .and_then(|alternative| alternative.total_cost) + }; + // Ephemeral: (build 10 + read 1 + retire 1) x 10 reads. Maintained over + // 10 s at 1 update/s: build 10 + updates 10 + reads 10 + retention 1 + retire 1. + assert_eq!( + cost(SummaryMaintenanceLifecycle::Ephemeral), + Some(Cost(120.0)) + ); + assert_eq!( + cost(SummaryMaintenanceLifecycle::ContinuouslyMaintained), + Some(Cost(32.0)) + ); + } + + // Without cost evidence a population's alternatives stay unknown: Planner + // selects none and timing is refused rather than guessed. + #[test] + fn population_without_cost_evidence_stays_unselected() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let plan = plan_summary_maintenance_lifecycles( + population_readout(), + WorkloadDemand::new_with_data(&workload, &data, &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &crate::cost_model::DefaultCostModel, + ) + .unwrap(); + let [deployment] = plan.deployments.as_slice() else { + panic!("one population state"); + }; + assert!(deployment + .alternatives + .iter() + .all(|alternative| alternative.total_cost.is_none())); + assert!(deployment.summary_maintenance_lifecycle_guarantee.is_none()); + assert_eq!( + plan.execution_timed_dag().unwrap_err(), + SummaryMaintenanceTimingError::UnselectedLifecycle(deployment.post_asap_node_id) + ); + } + + // A retained population and its raw input run at ingestion time; an + // Ephemeral population is rebuilt from raw input at query time. + #[test] + fn population_lifecycle_choice_decides_its_timing() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let timed = |lifecycle: SummaryMaintenanceLifecycle| { + population_timings(&timed_dag( + population_readout(), + &workload, + &data, + Some(Horizon(10.0)), + |_| lifecycle.clone(), + )) + }; + assert_eq!( + timed(SummaryMaintenanceLifecycle::ContinuouslyMaintained), + [("raw", INGEST), ("population", INGEST), ("readout", QUERY)] + ); + assert_eq!( + timed(SummaryMaintenanceLifecycle::Ephemeral), + [("raw", QUERY), ("population", QUERY), ("readout", QUERY)] + ); + } + + // When retaining is cheaper, Planner's own selection keeps the population + // maintained at ingestion time, as realization strategies placed it before + // population timing became a lifecycle decision. + #[test] + fn planner_selection_retains_population_at_ingestion() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let plan = plan_summary_maintenance_lifecycles( + population_readout(), + WorkloadDemand::new_with_data(&workload, &data, &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &UnitCosts, + ) + .unwrap(); + // Shared and ContinuouslyMaintained tie at 32; the first wins. + assert!(matches!( + selected_summary_maintenance_lifecycle(&plan.deployments[0]), + Some(SummaryMaintenanceLifecycle::Shared { .. }) + )); + assert_eq!( + population_timings(&plan.execution_timed_dag().unwrap()), + [("raw", INGEST), ("population", INGEST), ("readout", QUERY)] + ); + } + + // A population feeding summary state is that state's input, not a separate + // deployment: the state's lifecycle times it. + #[test] + fn population_feeding_summary_state_follows_that_state() { + let SummaryExpr::ValueOperation { + child: population, .. + } = &population_readout().expr + else { + unreachable!() + }; + let state = summary(); + let SummaryExpr::SummaryAgg { + family, + input, + reduction, + grouping, + .. + } = &state.expr + else { + unreachable!() + }; + let state = Rc::new(SummaryNode { + expr: SummaryExpr::SummaryAgg { + child: Rc::clone(population), + family: family.clone(), + input: input.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + }, + ..state.as_ref().clone() + }); + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let timed = |lifecycle: SummaryMaintenanceLifecycle| { + population_timings(&timed_dag( + readout(&state), + &workload, + &data, + Some(Horizon(10.0)), + |deployment| { + assert!(Rc::ptr_eq(&deployment.summary, &state)); + lifecycle.clone() + }, + )) + }; + assert_eq!( + timed(SummaryMaintenanceLifecycle::ContinuouslyMaintained), + [ + ("raw", INGEST), + ("population", INGEST), + ("state", INGEST), + ("readout", QUERY) + ] + ); + assert_eq!( + timed(SummaryMaintenanceLifecycle::Ephemeral), + [ + ("raw", QUERY), + ("population", QUERY), + ("state", QUERY), + ("readout", QUERY) + ] + ); + } + + // A population both read directly and consumed by summary state is that + // state's input in either traversal order: not a separate deployment, and + // timed by the state's lifecycle. + #[test] + fn shared_population_follows_its_summary_consumer() { + let direct = population_readout(); + let SummaryExpr::ValueOperation { + child: population, .. + } = &direct.expr + else { + unreachable!() + }; + let state = summary(); + let SummaryExpr::SummaryAgg { + family, + input, + reduction, + grouping, + .. + } = &state.expr + else { + unreachable!() + }; + let state = Rc::new(SummaryNode { + expr: SummaryExpr::SummaryAgg { + child: Rc::clone(population), + family: family.clone(), + input: input.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + }, + ..state.as_ref().clone() + }); + let binary = |lhs: Rc, rhs: Rc| { + Rc::new(SummaryNode { + schema: lhs.schema.clone(), + expr: SummaryExpr::BinaryOp { + lhs, + rhs, + operator: asap_types::post_asap::BinaryOperator { + kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( + asap_types::pre_asap::ArithmeticOpKind::Add, + ), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + timing: QUERY, + }, + guarantee: None, + }) + }; + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + for root in [ + binary(Rc::clone(&direct), readout(&state)), + binary(readout(&state), Rc::clone(&direct)), + ] { + for lifecycle in [ + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + SummaryMaintenanceLifecycle::Ephemeral, + ] { + let dag = timed_dag( + Rc::clone(&root), + &workload, + &data, + Some(Horizon(10.0)), + |deployment| { + assert!(Rc::ptr_eq(&deployment.summary, &state)); + lifecycle.clone() + }, + ); + let expected = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { + QUERY + } else { + INGEST + }; + for (kind, timing) in population_timings(&dag) { + if matches!(kind, "raw" | "population" | "state") { + assert_eq!(timing, expected, "{kind}"); + } else { + assert_eq!(timing, QUERY, "{kind}"); + } + } + } + } + } + + // A plan whose population deployment was removed after enumeration is + // refused rather than timed by a guess. + #[test] + fn timing_refuses_population_without_deployment() { + let data = continuous(1_000, 60_000); + let workload = workload(vec![], vec![repeating()], data.clone()); + let mut plan = plan_summary_maintenance_lifecycles( + population_readout(), + WorkloadDemand::new_with_data(&workload, &data, &[0]), + 1_000, + Some(Horizon(10.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &UnitCosts, + ) + .unwrap(); + let id = plan.deployments.remove(0).post_asap_node_id; + assert_eq!( + plan.execution_timed_dag(), + Err(SummaryMaintenanceTimingError::UnplannedMaintainedState(id)) + ); + } } diff --git a/crates/asap-physical-operators/Cargo.toml b/crates/asap-physical-operators/Cargo.toml new file mode 100644 index 00000000..74c172e3 --- /dev/null +++ b/crates/asap-physical-operators/Cargo.toml @@ -0,0 +1,18 @@ +[package] +name = "asap-physical-operators" +version = "0.1.0" +edition = "2021" + +[dependencies] +futures = "0.3" +planner-types = { package = "asap-types", path = "../types" } +asap_sketchlib = { git = "https://github.com/ProjectASAP/asap_sketchlib", rev = "5f03ccbd798ed5fec62bdd839bcb331123cab369" } +serde = { version = "1", features = ["derive", "rc"] } +serde_json = "1" +tracing = "0.1" +thiserror = "2" + + +[dev-dependencies] +asap-aware-mapping = { path = "../asap-aware-mapping" } +asap-frontend-promql = { path = "../frontend-promql" } diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md new file mode 100644 index 00000000..5f638465 --- /dev/null +++ b/crates/asap-physical-operators/README.md @@ -0,0 +1,121 @@ +# ASAP physical operators + +An independent Rust physical operator DAG runtime shared by ingestion time and +query time execution. The library requires neither backend engine, a server, +a storage implementation, Arrow nor DataFusion. DataFusion informed the design; +it is not the execution framework. + +`plan::PhysicalDag` binds typed operator inputs to node IDs. Each execution starts +one producer per reachable node, shares output batches among its consumers, and +bounds buffering. Dropping one consumer does not cancel other consumers. A +`RunContext` carries query or ingestion scope, cancellation and byte accounting. +Executions use the caller's worker and worker-local streams, with no internal +thread pool. Poll multiple root streams concurrently when they share inputs. + +`operators::Operator` implements native batch sources, scalar values, +projection, filtering, grouped exact aggregation, semi-join, grouped Sort and +Limit, vector-to-scalar conversion, Union, and summary construction/merge/readout. +Sort followed by Limit implements grouped ranking; no dedicated TopK physical +operator is needed. Summary construction updates state batch by batch. End of +input means the supplied query range or ingestion window is complete. + +```rust +use asap_physical_operators::{ + expressions::Expression, + operators::Operator, + values::Value, + plan::PhysicalDag, + runtime::{Limits, RunContext, Scope}, +}; +use asap_physical_operators::planner::pre_asap::DataType; +use futures::{executor::block_on, StreamExt}; + +let source = Operator::scalar(Value::Int64(7), DataType::Int64)?; +let negate = Operator::project(source.schema(), vec![ + ("value".into(), Expression::Negate(Box::new(Expression::Column(0)))), +])?; +let mut plan = PhysicalDag::default(); +plan.add(0, vec![], source)?; +plan.add(1, vec![0], negate)?; +let run = RunContext::new( + Scope::Query { evaluation_time_ms: 1000, revision: 1 }, + Limits::default(), +)?; +let mut output = plan.execute(&[1], run)?.remove(0); +let batch = block_on(output.next()).unwrap()?; +assert!(matches!(batch.rows()[0][0], Value::Int64(-7))); +# Ok::<(), asap_physical_operators::dag::Error>(()) +``` + +`physical_planner::compile` accepts a logical Post-ASAP DAG (`PostAsapDag`) and typed input contracts. +The resulting candidate is instantiated with deployment readers after selection. It rejects unsupported operations and +schema mismatches before starting a source. Implement `PhysicalOperator` for a +deployment source, including asynchronous I/O; computation operators remain in +the library. The public `planner` export identifies the exact Planner types used +by the crate. The physical compiler currently supports a subset of those types and +operations; it does not interpret an unknown node as external fallback. + +Plain values preserve Planner scalar/collection types and nullability. Numeric +arithmetic uses matching Int64 or Float64 inputs; integer overflow is an error. +Boolean predicates use three-valued logic. Native summary states currently cover +exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS and CountSketch with candidate heaps. Binding checks family, +parameters and readout compatibility; source batches also validate state payloads. +Existing accumulator algorithms are reused as kernels behind these operators. + +This crate is owned by ASAPPlanner. Its `planner-types` dependency is the local +IR crate, so a contract change and its execution tests belong in the same PR. +Deployments supply storage/ingestion sources and adapt output protocols. The +library has no ASAPQuery-backend dependency. Backend raw Scan remains a separate +deployment capability. + +See [the design](../../docs/design_docs/physical-planning-and-deployment.md). + +## Module boundaries + +- `plan`: immutable graph, operator interface, schemas and execution properties. +- `runtime`: per-run streams, shared producers, memory reservations and cancellation. +- `expressions`: scalar evaluation; typed builders and the Planner expression adapter. +- `operators`: projection, filter, joins, aggregate/window, sort, limit and summary implementations. +- `sources`: raw-source interface, Scan and the memory connector. +- `physical_planner`: native operator lowering, typed input contracts and checked instantiation. +- `summary_kernels`: in-memory summary state over `asap_sketchlib` and exact Planner state: merge, typed readout and update adapters. +- `readout`: readouts over merged exact summary states. +- `capability`: explicit kernel, native-batch and typed readout validation. + +The `dag`, `factory`, `traits` and `arithmetic` paths are re-exports. They contain no alternative +execution implementations. + +Sketch algorithms and their state encodings belong to `asap_sketchlib`. Wire decoding, delta +frames, edge sampling and storage statistics belong to deployments. Kernels hold one population's +state; operators own grouping. + +A source must declare `Boundedness::Bounded` to feed a blocking operator. +The default for a custom raw source is `Unknown`; query or ingestion scope alone +does not promise that its cursor ends. `PhysicalDag::properties` validates these +requirements before any source starts and returns boundedness and emission mode +for every reachable node. The memory connector declares finite input. Custom +physical sources expose the same facts through `PhysicalOperator::properties`. + +Blocking operators reserve estimated workspace and yield cooperatively during +row processing and sort merges. Cancellation releases reservations when the +stream is polled or dropped. Individual scalar evaluations, bounded sort chunks +and sketch kernel calls are synchronous; this is not preemptive execution. +There is no spill or partitioned parallel execution in this implementation. + +## Physical compilation and deployment inputs + +`physical_planner::compile` accepts a Planner `PostAsapDag`, typed +`InputContract`s and output roots. It returns a reusable `CompiledPhysicalDag` +containing selected native operators and no live readers. Compilation validates +schemas, input ordering, sharing and boundedness before deployment source access. + +A deployment calls `CompiledPhysicalDag::instantiate` with exactly the declared +inputs. This checks source schemas and execution properties and constructs the +runnable graph without repeating logical lowering. The graph executes through +the shared runtime with independent per-run state. Window coverage, revision and +maintenance-policy admission remain deployment/planning contracts; this compiler +does not discover storage or silently change a selected maintenance strategy. + +Pane construction and geometry belong to deployment. A deployment runs the +precompute DAG once per pane it constructs and binds the selected pane states to +query input slots; the query DAG merges and reads them out as computation. diff --git a/crates/asap-physical-operators/src/capability.rs b/crates/asap-physical-operators/src/capability.rs new file mode 100644 index 00000000..2a1cc095 --- /dev/null +++ b/crates/asap-physical-operators/src/capability.rs @@ -0,0 +1,241 @@ +//! Capability boundaries, checked without constructing accumulator state. +//! +//! `validate_summary_kernel` checks update kernels, including families without a +//! native batch representation. `validate_native_family` and +//! `validate_sketch_readout` / `validate_exact_readout` check native state and readout support. +//! Keyed weighted-frequency readouts are checked by `Operator::keyed_readout`. +//! A successful kernel check alone does not mean a physical DAG will bind. +//! +//! Stored-state encodings belong to deployments. Full plan acceptance is +//! owned by `binding`, which also validates schemas, expressions and inputs. +use crate::Error; +use planner_types::post_asap::{ + ExactKind, ExactParams, GroupingStrategy, SketchAlgorithm, SketchParams, SketchQuery, + SummaryFamilyType, SummaryUpdate, +}; + +/// Check the same contract used by `create_planner_accumulator` before a plan +/// is accepted. Execution timing is deliberately not a kernel property. +pub fn validate_summary_kernel( + family: &SummaryFamilyType, + input: &SummaryUpdate, + grouping: &GroupingStrategy, +) -> Result<(), String> { + if grouping != &GroupingStrategy::PerSubpopulationInstance { + return Err("shared summary grouping has no registered kernel".into()); + } + let keyed = match family { + SummaryFamilyType::ExactAggregate(kind, params) => { + use ExactKind as K; + use ExactParams as P; + if !matches!( + (kind, params), + (K::Sum, P::Sum) + | (K::Count, P::Count) + | (K::Min, P::Min) + | (K::Max, P::Max) + | (K::Rate, P::Rate) + | (K::Increase, P::Increase) + ) { + return Err(format!("unsupported exact kernel {family:?}")); + } + input.item.is_some() + } + SummaryFamilyType::Sketch(kind, layout) => { + if layout != grouping { + return Err("Planner family and operator grouping disagree".into()); + } + use SketchAlgorithm as A; + use SketchParams as P; + match (kind.algorithm(), kind.params()) { + (A::Kll, P::Kll { k }) if (8..=u16::MAX as u32).contains(k) => false, + (A::DDSketch, P::DDSketch { alpha }) + if alpha.is_finite() && *alpha > 0.0 && *alpha < 1.0 => + { + false + } + (A::Hll, P::Hll { precision }) if (4..=18).contains(precision) => false, + (A::Cms, P::Cms { width, depth }) + | (A::CountSketch, P::CountSketch { width, depth }) + if valid_matrix(*width, *depth) => + { + true + } + ( + A::CmsWithHeap, + P::CmsWithHeap { + width, + depth, + heap_size, + }, + ) + | ( + A::CountSketchWithHeap, + P::CountSketchWithHeap { + width, + depth, + heap_size, + }, + ) if valid_matrix(*width, *depth) && *heap_size > 0 => true, + ( + A::UnivMon, + P::UnivMon { + heap_size, + sketch_rows, + sketch_cols, + layers, + }, + ) if *heap_size > 0 + && *sketch_cols > 0 + && (1..=20).contains(sketch_rows) + && (1..=64).contains(layers) + && (*sketch_rows as usize) + .checked_mul(*sketch_cols as usize) + .and_then(|n| n.checked_mul(*layers as usize)) + .is_some() => + { + false + } + _ => { + return Err(format!( + "unsupported kernel or invalid parameters: {kind:?}" + )) + } + } + } + _ => return Err(format!("unsupported summary kernel {family:?}")), + }; + if keyed != input.item.is_some() && !is_unit_sample_frequency(input) { + return Err("Planner item expression does not match kernel layout".into()); + } + Ok(()) +} + +fn valid_matrix(width: u32, depth: u32) -> bool { + // Construction uses the kernel's native row hashing, so no encoded-size + // limit applies here. + width > 0 + && depth > 0 + && (width as usize) + .checked_mul(depth as usize) + .and_then(|n| n.checked_mul(std::mem::size_of::())) + .is_some() +} + +pub(crate) fn is_unit_sample_frequency(update: &planner_types::post_asap::SummaryUpdate) -> bool { + use planner_types::post_asap::{NonNegativeWeightProof, SummaryInputExpr, WeightDomain}; + matches!( + update.item, + Some(SummaryInputExpr::Column( + planner_types::pre_asap::ColumnRef::SampleValue + )) + ) && matches!(update.weight, SummaryInputExpr::Constant(1.0)) + && matches!( + update.weight_domain, + WeightDomain::NonNegative { + proof: NonNegativeWeightProof::UnitCount + } + ) +} + +pub fn validate_native_family(family: &SummaryFamilyType) -> Result<(), Error> { + use planner_types::post_asap::SketchAlgorithm as A; + if let SummaryFamilyType::Sketch(kind, grouping) = family { + if matches!(kind.algorithm(), A::CmsWithHeap | A::CountSketchWithHeap) { + let (_, width, depth, _) = + crate::summary_kernels::weighted_frequency::WeightedFrequency::configuration(kind)?; + return if valid_matrix(width as u32, depth as u32) && grouping == &Default::default() { + Ok(()) + } else { + Err(Error::Invalid( + "invalid weighted frequency dimensions or grouping strategy".into(), + )) + }; + } + } + match family { + SummaryFamilyType::ExactAggregate(..) => {} + SummaryFamilyType::Sketch(kind, _) + if matches!(kind.algorithm(), A::Kll | A::DDSketch | A::Hll) => {} + _ => { + return Err(Error::Invalid( + "summary family has no native DAG state implementation".into(), + )) + } + } + crate::capability::validate_summary_kernel( + family, + &planner_types::post_asap::SummaryUpdate::column( + planner_types::pre_asap::ColumnRef::SampleValue, + ), + &Default::default(), + ) + .map_err(Error::Invalid) +} + +/// A sketch readout is native only for the families Planner can read directly. +pub fn validate_sketch_readout( + family: &SummaryFamilyType, + query: &SketchQuery, +) -> Result<(), Error> { + validate_native_family(family)?; + use planner_types::post_asap::SketchAlgorithm as A; + // A point count without an item value reads the total count. + let bare_count = matches!(query, SketchQuery::PointCount { value: None, .. }); + let supported = match family { + SummaryFamilyType::Sketch(kind, _) => match (kind.algorithm(), query) { + (A::Kll, SketchQuery::Quantile { q }) | (A::DDSketch, SketchQuery::Quantile { q }) => { + if !(0.0..=1.0).contains(q) { + return Err(Error::Invalid( + "quantile readout requires quantile in [0,1]".into(), + )); + } + true + } + (A::DDSketch, _) => bare_count, + (A::Hll, SketchQuery::Cardinality) => true, + (A::Hll, _) => bare_count, + _ => false, + }, + _ => false, + }; + if !supported { + return Err(Error::Invalid( + "readout is not implemented for this summary family".into(), + )); + } + Ok(()) +} + +/// An exact readout must match the exact family it reads. +pub fn validate_exact_readout( + family: &SummaryFamilyType, + readout: &crate::summary_kernels::exact::ExactReadout, +) -> Result<(), Error> { + validate_native_family(family)?; + use crate::Statistic as S; + use planner_types::post_asap::ExactKind as E; + let supported = matches!( + (family, readout.statistic), + (SummaryFamilyType::ExactAggregate(E::Sum, _), S::Sum) + | (SummaryFamilyType::ExactAggregate(E::Count, _), S::Count) + | (SummaryFamilyType::ExactAggregate(E::Min, _), S::Min) + | (SummaryFamilyType::ExactAggregate(E::Max, _), S::Max) + | (SummaryFamilyType::ExactAggregate(E::Rate, _), S::Rate) + | ( + SummaryFamilyType::ExactAggregate(E::Increase, _), + S::Increase + ) + ); + if !supported { + return Err(Error::Invalid( + "readout is not implemented for this summary family".into(), + )); + } + if readout.lookback_ms.is_some_and(|lookback| { + lookback <= 0 || !matches!(readout.statistic, S::Rate | S::Increase) + }) { + return Err(Error::Invalid("invalid exact counter lookback".into())); + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/dag/mod.rs b/crates/asap-physical-operators/src/dag/mod.rs new file mode 100644 index 00000000..c7837672 --- /dev/null +++ b/crates/asap-physical-operators/src/dag/mod.rs @@ -0,0 +1,8 @@ +//! Compatibility imports. New code should use plan, runtime, operators, physical_planner and sources directly. +pub use crate::plan::{NodeId, PhysicalDag, PhysicalOperator}; +pub use crate::runtime::batch_execution; +pub use crate::runtime::{ + Input, Limits, OutputStream, Reservation, RunContext, Scope, SharedValue, +}; +pub use crate::Error; +pub use crate::{expressions, operators, physical_planner as planner, sources as scan, values}; diff --git a/crates/asap-physical-operators/src/error.rs b/crates/asap-physical-operators/src/error.rs new file mode 100644 index 00000000..afee33d8 --- /dev/null +++ b/crates/asap-physical-operators/src/error.rs @@ -0,0 +1,18 @@ +use crate::plan::NodeId; +#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] +pub enum Error { + #[error("invalid DAG: {0}")] + Invalid(String), + #[error("operator failed: {0}")] + Operator(String), + #[error("node {node} ({operation}) failed: {source}")] + AtNode { + node: NodeId, + operation: String, + source: Box, + }, + #[error("execution memory limit exceeded")] + MemoryLimit, + #[error("execution cancelled")] + Cancelled, +} diff --git a/crates/asap-physical-operators/src/expressions/arithmetic.rs b/crates/asap-physical-operators/src/expressions/arithmetic.rs new file mode 100644 index 00000000..30e277d4 --- /dev/null +++ b/crates/asap-physical-operators/src/expressions/arithmetic.rs @@ -0,0 +1,63 @@ +//! Float64 arithmetic shared by ASAP execution engines. +//! Preserve IEEE non-finite results; callers own their output policies. + +pub fn evaluate_float64_arithmetic( + operator: &planner_types::pre_asap::ArithmeticOpKind, + left: f64, + right: f64, +) -> f64 { + use planner_types::pre_asap::ArithmeticOpKind::*; + match operator { + Add => left + right, + Sub => left - right, + Mul => left * right, + Div => left / right, + Mod => left % right, + Pow => left.powf(right), + Atan2 => left.atan2(right), + } +} + +/// Execute the Planner binary contract after a deployment has resolved matching rows. +pub fn evaluate_binary( + operator: &planner_types::post_asap::BinaryOperator, + left: f64, + right: f64, +) -> Result { + use crate::{values::Value, Error}; + use planner_types::pre_asap::{ArithmeticOpKind, BinaryOpKind, CompareOpKind}; + let invalid = + || Error::Invalid("unsupported binary operation or invalid checked-division domain".into()); + if operator.vector_match.is_some() { + return Err(invalid()); + } + if operator.checked_relative_division || operator.checked_finite_division { + if operator.kind != BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + || !left.is_finite() + || !right.is_finite() + || right == 0. + { + return Err(invalid()); + } + let value = left / right; + if !value.is_finite() || (operator.checked_relative_division && !value.is_normal()) { + return Err(invalid()); + } + return Ok(Value::Float64(value)); + } + Ok(match operator.kind { + BinaryOpKind::Arithmetic(ref op) => { + Value::Float64(evaluate_float64_arithmetic(op, left, right)) + } + BinaryOpKind::Compare(ref op) => Value::Bool(match op { + CompareOpKind::Eq => left == right, + CompareOpKind::Ne => left != right, + CompareOpKind::Lt => left < right, + CompareOpKind::Le => left <= right, + CompareOpKind::Gt => left > right, + CompareOpKind::Ge => left >= right, + _ => return Err(invalid()), + }), + _ => return Err(invalid()), + }) +} diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs new file mode 100644 index 00000000..a16a3bde --- /dev/null +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -0,0 +1,410 @@ +//! Scalar semantics and typed expression binding. Planner expressions enter through CompiledExpression. +use crate::{ + values::{plain, Schema, Value}, + Error, +}; +use planner_types::pre_asap::{ArithmeticOpKind, DataType}; +pub mod arithmetic; +mod planner; +pub use planner::CompiledExpression; +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] +pub enum Expression { + Binary { + operator: planner_types::post_asap::BinaryOperator, + left: Box, + right: Box, + }, + Planner(Box), + Column(usize), + ExactFloat64(usize), + FiniteFloat64(Box), + LabelSet { + column: usize, + labels: Vec, + without: bool, + }, + /// One label of a label map; an absent label reads as empty, as in PromQL. + Label { + column: usize, + name: String, + }, + /// Canonical encoding of a label map less `excluding`, identical to + /// `promql_rows::encode_series_identity`. + LabelIdentity { + column: usize, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + excluding: Vec, + }, + Literal { + value: Value, + dtype: DataType, + }, + Negate(Box), + Arithmetic { + op: ArithmeticOpKind, + left: Box, + right: Box, + }, + Equal(Box, Box), + Less(Box, Box), + And(Box, Box), + Or(Box, Box), + Not(Box), + IsNull(Box), +} +impl Expression { + pub fn planner(expression: crate::expressions::CompiledExpression) -> Self { + Self::Planner(Box::new(expression)) + } + pub(crate) fn dtype(&self, input: &Schema) -> Result<(DataType, bool), Error> { + use Expression::*; + match self { + Binary { + operator, + left, + right, + } => { + use planner_types::pre_asap::{BinaryOpKind, CompareOpKind}; + let (a, n) = left.dtype(input)?; + let (b, m) = right.dtype(input)?; + if a != DataType::Float64 || b != a || operator.vector_match.is_some() { + return Err(invalid( + "binary expression requires resolved Float64 operands", + )); + } + if (operator.checked_relative_division || operator.checked_finite_division) + && operator.kind != BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + { + return Err(invalid("checked division contract on non-division")); + } + let dtype = match operator.kind { + BinaryOpKind::Arithmetic(_) => DataType::Float64, + BinaryOpKind::Compare( + CompareOpKind::Eq + | CompareOpKind::Ne + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge, + ) => DataType::Bool, + _ => return Err(invalid("unsupported binary operation")), + }; + Ok((dtype, n || m)) + } + Planner(expression) => { + expression.validate_input(input)?; + Ok(expression.dtype()) + } + FiniteFloat64(expression) => { + if expression.dtype(input)? != (DataType::Float64, false) { + return Err(invalid("finite update requires non-null Float64")); + } + Ok((DataType::Float64, false)) + } + ExactFloat64(column) => { + let (dtype, nullable) = plain(input, *column)?; + if nullable || !matches!(dtype, DataType::Int64 | DataType::Float64) { + return Err(invalid( + "exact Float64 conversion requires non-null numeric input", + )); + } + Ok((DataType::Float64, false)) + } + LabelSet { column, labels, .. } => { + let (dtype, nullable) = plain(input, *column)?; + let expected = DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }; + if dtype != &expected + || nullable + || labels + .iter() + .collect::>() + .len() + != labels.len() + { + return Err(invalid( + "label projection requires a non-null Utf8 map and unique label names", + )); + } + Ok((expected, false)) + } + Label { column, .. } | LabelIdentity { column, .. } => { + if plain(input, *column)? + != ( + &DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }, + false, + ) + { + return Err(invalid("label read requires a non-null Utf8 map")); + } + Ok((DataType::Utf8, false)) + } + Column(i) => { + let (t, n) = plain(input, *i)?; + Ok((t.clone(), n)) + } + Literal { value, dtype } => { + if value.matches(dtype, true) { + Ok((dtype.clone(), matches!(value, Value::Null))) + } else { + Err(invalid("literal type mismatch")) + } + } + Negate(v) => { + let (t, n) = v.dtype(input)?; + if matches!(t, DataType::Int64 | DataType::Float64) { + Ok((t, n)) + } else { + Err(invalid("numeric negation required")) + } + } + Arithmetic { op, left, right } => { + let (a, n) = left.dtype(input)?; + let (b, m) = right.dtype(input)?; + if a == b + && matches!(a, DataType::Int64 | DataType::Float64) + && !(a == DataType::Int64 && *op == ArithmeticOpKind::Atan2) + { + Ok((a, n || m)) + } else { + Err(invalid("arithmetic requires matching numeric types")) + } + } + Equal(a, b) | Less(a, b) => { + let (a, n) = a.dtype(input)?; + let (b, m) = b.dtype(input)?; + if a == b && ordered(&a) { + Ok((DataType::Bool, n || m)) + } else { + Err(invalid("comparison requires matching ordered types")) + } + } + And(a, b) | Or(a, b) => { + let (a, n) = a.dtype(input)?; + let (b, m) = b.dtype(input)?; + if a == DataType::Bool && b == DataType::Bool { + Ok((DataType::Bool, n || m)) + } else { + Err(invalid("boolean operands required")) + } + } + Not(v) => { + let (t, n) = v.dtype(input)?; + if t == DataType::Bool { + Ok((t, n)) + } else { + Err(invalid("boolean operand required")) + } + } + IsNull(v) => { + v.dtype(input)?; + Ok((DataType::Bool, false)) + } + } + } + pub(crate) fn evaluate(&self, row: &[Value]) -> Result { + use Expression::*; + Ok(match self { + FiniteFloat64(expression) => match expression.evaluate(row)? { + Value::Float64(value) if value.is_finite() => Value::Float64(value), + _ => return Err(invalid("summary update must be finite")), + }, + ExactFloat64(column) => match row[*column] { + Value::Float64(value) => Value::Float64(value), + Value::Int64(value) if value.unsigned_abs() <= (1u64 << 53) => { + Value::Float64(value as f64) + } + _ => { + return Err(invalid( + "numeric result cannot be represented exactly as Float64", + )) + } + }, + LabelSet { + column, + labels, + without, + } => { + let Value::Map(entries) = &row[*column] else { + return Err(invalid("label projection requires a map")); + }; + let mut selected = std::collections::BTreeMap::new(); + let mut seen = std::collections::BTreeSet::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("label projection requires Utf8 entries")); + }; + if !seen.insert(key.clone()) { + return Err(invalid("duplicate label name")); + } + let keep = if *without { + key.as_ref() != "__name__" + && !labels.iter().any(|label| label.as_str() == key.as_ref()) + } else { + labels.iter().any(|label| label.as_str() == key.as_ref()) + }; + if keep && !value.is_empty() { + selected.insert(key.clone(), value.clone()); + } + } + Value::Map( + selected + .into_iter() + .map(|(k, v)| (Value::Utf8(k), Value::Utf8(v))) + .collect::>() + .into(), + ) + } + Binary { + operator, + left, + right, + } => { + let (a, b) = (left.evaluate(row)?, right.evaluate(row)?); + if matches!(a, Value::Null) || matches!(b, Value::Null) { + Value::Null + } else { + let (Value::Float64(a), Value::Float64(b)) = (a, b) else { + return Err(invalid("binary value schema mismatch")); + }; + arithmetic::evaluate_binary(operator, a, b)? + } + } + Planner(expression) => expression.evaluate(row)?, + Label { column, name } => { + let Value::Map(entries) = &row[*column] else { + return Err(invalid("label read requires a map")); + }; + let mut found = None; + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("label read requires Utf8 entries")); + }; + if key.as_ref() == name.as_str() && found.replace(value.clone()).is_some() { + return Err(invalid("duplicate label name")); + } + } + Value::Utf8(found.unwrap_or_else(|| "".into())) + } + LabelIdentity { column, excluding } => { + let Value::Map(entries) = &row[*column] else { + return Err(invalid("label identity requires a map")); + }; + let mut labels = std::collections::BTreeMap::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("label identity requires Utf8 entries")); + }; + if excluding.iter().any(|label| label.as_str() == key.as_ref()) { + continue; + } + if labels.insert(key.to_string(), value.to_string()).is_some() { + return Err(invalid("duplicate label name")); + } + } + Value::Utf8( + crate::physical_planner::promql_rows::encode_series_identity(&labels)?.into(), + ) + } + Column(i) => row[*i].clone(), + Literal { value, .. } => value.clone(), + Negate(v) => match v.evaluate(row)? { + Value::Int64(v) => Value::Int64( + v.checked_neg() + .ok_or_else(|| invalid("integer negation overflow"))?, + ), + Value::Float64(v) => Value::Float64(-v), + Value::Null => Value::Null, + _ => return Err(invalid("numeric negation required")), + }, + Arithmetic { op, left, right } => { + numeric(op, left.evaluate(row)?, right.evaluate(row)?)? + } + Equal(a, b) | Less(a, b) => { + let (a, b) = (a.evaluate(row)?, b.evaluate(row)?); + if matches!(a, Value::Null) || matches!(b, Value::Null) { + Value::Null + } else if matches!((&a,&b),(Value::Float64(a),Value::Float64(b)) if a.is_nan() || b.is_nan()) + { + Value::Bool(false) + } else { + let c = a.compare(&b)?; + Value::Bool(if matches!(self, Equal(..)) { + c.is_eq() + } else { + c.is_lt() + }) + } + } + And(a, b) | Or(a, b) => { + let (a, b) = (a.evaluate(row)?, b.evaluate(row)?); + match (a, b, matches!(self, And(..))) { + (Value::Bool(false), _, true) | (_, Value::Bool(false), true) => { + Value::Bool(false) + } + (Value::Bool(true), _, false) | (_, Value::Bool(true), false) => { + Value::Bool(true) + } + (Value::Null, _, _) | (_, Value::Null, _) => Value::Null, + (Value::Bool(a), Value::Bool(b), true) => Value::Bool(a && b), + (Value::Bool(a), Value::Bool(b), false) => Value::Bool(a || b), + _ => return Err(invalid("boolean operands required")), + } + } + Not(v) => match v.evaluate(row)? { + Value::Bool(v) => Value::Bool(!v), + Value::Null => Value::Null, + _ => return Err(invalid("boolean operand required")), + }, + IsNull(v) => Value::Bool(matches!(v.evaluate(row)?, Value::Null)), + }) + } +} +pub(crate) fn ordered(dtype: &DataType) -> bool { + if let DataType::Map { key, value, .. } = dtype { + return ordered(key) && ordered(value); + } + matches!( + dtype, + DataType::Null + | DataType::Int64 + | DataType::Float64 + | DataType::Utf8 + | DataType::Bool + | DataType::Timestamp + | DataType::Date + ) +} +pub(crate) fn numeric(op: &ArithmeticOpKind, a: Value, b: Value) -> Result { + use ArithmeticOpKind::*; + Ok(match (a, b) { + (Value::Null, _) | (_, Value::Null) => Value::Null, + (Value::Float64(a), Value::Float64(b)) => { + Value::Float64(arithmetic::evaluate_float64_arithmetic(op, a, b)) + } + (Value::Int64(a), Value::Int64(b)) => Value::Int64( + match op { + Add => a.checked_add(b), + Sub => a.checked_sub(b), + Mul => a.checked_mul(b), + Div => a.checked_div(b), + Mod => a.checked_rem(b), + Pow => u32::try_from(b).ok().and_then(|b| a.checked_pow(b)), + Atan2 => None, + } + .ok_or_else(|| invalid("invalid integer arithmetic or overflow"))?, + ), + _ => return Err(invalid("arithmetic type mismatch")), + }) +} + +fn invalid(message: &str) -> Error { + Error::Invalid(message.into()) +} diff --git a/crates/asap-physical-operators/src/expressions/planner.rs b/crates/asap-physical-operators/src/expressions/planner.rs new file mode 100644 index 00000000..2130a2f7 --- /dev/null +++ b/crates/asap-physical-operators/src/expressions/planner.rs @@ -0,0 +1,549 @@ +//! Planner scalar expressions evaluated over native typed rows. +use crate::{ + values::{Schema, Value}, + Error, +}; +use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind, DataType, QueryExpr, ScalarValue}; +use std::{cmp::Ordering, sync::Arc}; + +pub(super) fn evaluate( + expr: &QueryExpr, + row: &[Value], + schema: &planner_types::pre_asap::Schema, +) -> Result { + match expr { + QueryExpr::Column(index) => row.get(*index).cloned().ok_or(Error::Invalid(format!( + "column {index} outside row width {}", + row.len() + ))), + QueryExpr::Literal(value) => Ok(match value { + ScalarValue::Interval { + months, + days, + nanos, + } => Value::Interval { + months: *months, + days: *days, + nanos: *nanos, + }, + ScalarValue::Int64(value) => Value::Int64(*value), + ScalarValue::Float64(value) => Value::Float64(*value), + ScalarValue::Utf8(value) => Value::Utf8(value.clone().into()), + ScalarValue::Boolean(value) => Value::Bool(*value), + ScalarValue::Null => Value::Null, + }), + QueryExpr::Compare { left, op, right } => { + let left = evaluate(left, row, schema)?; + let right = evaluate(right, row, schema)?; + compare(op, left, right) + } + QueryExpr::Arithmetic { op, left, right } => arithmetic( + op, + evaluate(left, row, schema)?, + evaluate(right, row, schema)?, + ), + QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + let and = matches!(expr, QueryExpr::BoolAnd(_)); + let mut null = false; + for part in parts { + match evaluate(part, row, schema)? { + Value::Bool(value) if value != and => return Ok(Value::Bool(value)), + Value::Bool(_) => {} + Value::Null => null = true, + _ => return Err(Error::Invalid("boolean predicate required".into())), + } + } + Ok(if null { Value::Null } else { Value::Bool(and) }) + } + QueryExpr::Not(value) => match evaluate(value, row, schema)? { + Value::Bool(value) => Ok(Value::Bool(!value)), + Value::Null => Ok(Value::Null), + _ => Err(Error::Invalid("boolean predicate required".into())), + }, + QueryExpr::IsNull(value) => Ok(Value::Bool(matches!( + evaluate(value, row, schema)?, + Value::Null + ))), + QueryExpr::IsNotNull(value) => Ok(Value::Bool(!matches!( + evaluate(value, row, schema)?, + Value::Null + ))), + QueryExpr::FunctionCall { name, args } => { + use planner_types::pre_asap::scalar_signature::MapScalarFunction; + if name.eq_ignore_ascii_case("asap_struct_field") { + expr.scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))?; + let DataType::Struct { fields } = args[0] + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))? + .0 + else { + unreachable!() + }; + let offset = match &args[1] { + QueryExpr::Literal(ScalarValue::Int64(index)) => { + usize::try_from(index - 1).ok() + } + QueryExpr::Literal(ScalarValue::Utf8(name)) => { + fields.iter().position(|field| &field.name == name) + } + _ => None, + } + .ok_or_else(|| Error::Invalid("struct field selector".into()))?; + let Value::Struct(values) = evaluate(&args[0], row, schema)? else { + return Err(Error::Invalid("struct field input".into())); + }; + return values + .get(offset) + .cloned() + .ok_or_else(|| Error::Invalid("struct field value".into())); + } + if name.eq_ignore_ascii_case("asap_element_access") { + let (output_type, _) = expr + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))?; + if let DataType::List { element } = args[0] + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))? + .0 + { + let Value::List(values) = evaluate(&args[0], row, schema)? else { + return Err(Error::Invalid("array access input".into())); + }; + let index = match evaluate(&args[1], row, schema)? { + Value::Null => return Ok(Value::Null), + Value::Int64(index) => index, + _ => return Err(Error::Invalid("array access index".into())), + }; + let offset = if index > 0 { + usize::try_from(index - 1).ok() + } else if index < 0 { + usize::try_from(index.unsigned_abs()) + .ok() + .and_then(|distance| values.len().checked_sub(distance)) + } else { + None + }; + return match offset.and_then(|offset| values.get(offset)) { + Some(value) => Ok(value.clone()), + None => default_collection_element(&output_type, element.nullable), + }; + } + } + let function = (if name.eq_ignore_ascii_case("asap_element_access") { + Some(MapScalarFunction::Access) + } else { + MapScalarFunction::from_name(name) + }) + .ok_or_else(|| Error::Invalid(format!("scalar function {name}")))?; + expr.scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))?; + let values = args + .iter() + .map(|arg| evaluate(arg, row, schema)) + .collect::, _>>()?; + match function { + MapScalarFunction::Construct => { + let mut values = values.into_iter(); + let mut entries = Vec::new(); + while let Some(key) = values.next() { + if !matches!(key, Value::Int64(_) | Value::Utf8(_) | Value::Bool(_)) { + return Err(Error::Invalid("map key value type".into())); + } + entries.push(( + key, + values + .next() + .ok_or_else(|| Error::Invalid("odd map argument count".into()))?, + )); + } + Ok(Value::Map(entries.into())) + } + MapScalarFunction::Concat => { + let mut entries = Vec::new(); + for value in values { + let Value::Map(next) = value else { + return Err(Error::Invalid("map concat argument".into())); + }; + entries.extend(next.iter().cloned()); + } + Ok(Value::Map(entries.into())) + } + MapScalarFunction::Access => { + let [Value::Map(entries), key] = values.as_slice() else { + return Err(Error::Invalid("map access arguments".into())); + }; + if matches!(key, Value::Null) { + return Ok(Value::Null); + } + if !matches!(key, Value::Int64(_) | Value::Utf8(_) | Value::Bool(_)) { + return Err(Error::Invalid("map lookup key type".into())); + } + if let Some((_, value)) = entries + .iter() + .find(|(candidate, _)| cell_cmp(candidate, key) == Some(Ordering::Equal)) + { + return Ok(value.clone()); + } + let ( + DataType::Map { + value, + value_nullable, + .. + }, + _, + ) = args[0] + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))? + else { + unreachable!() + }; + default_collection_element(&value, value_nullable) + } + } + } + other => Err(Error::Invalid(format!("scalar expression {other:?}"))), + } +} + +fn default_collection_element(dtype: &DataType, nullable: bool) -> Result { + if nullable { + return Ok(Value::Null); + } + Ok(match dtype { + DataType::Interval | DataType::Date => { + return Err(Error::Invalid("temporal value transport".into())) + } + DataType::Null => Value::Null, + DataType::Int64 => Value::Int64(0), + DataType::Float64 => Value::Float64(0.0), + DataType::Utf8 => Value::Utf8("".into()), + DataType::Bool => Value::Bool(false), + DataType::Map { .. } => Value::Map(Arc::from([])), + DataType::List { .. } => Value::List(Arc::from([])), + DataType::Struct { fields } => Value::Struct( + fields + .iter() + .map(|field| default_collection_element(&field.dtype, field.nullable)) + .collect::, _>>()? + .into(), + ), + _ => { + return Err(Error::Invalid( + "collection missing-element default type".into(), + )) + } + }) +} + +fn compare(op: &CompareOpKind, left: Value, right: Value) -> Result { + if matches!(left, Value::Null) || matches!(right, Value::Null) { + return Ok(Value::Null); + } + // NaN is unordered, not a type mismatch. Match the native scalar path. + if matches!(&left, Value::Float64(v) if v.is_nan()) + || matches!(&right, Value::Float64(v) if v.is_nan()) + { + return match op { + CompareOpKind::Ne => Ok(Value::Bool(true)), + CompareOpKind::Eq + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge => Ok(Value::Bool(false)), + _ => Err(Error::Invalid(format!("comparison {op:?}"))), + }; + } + let ordering = cell_cmp(&left, &right) + .ok_or_else(|| Error::Invalid("comparison of incompatible values".into()))?; + let value = match op { + CompareOpKind::Eq => ordering == Ordering::Equal, + CompareOpKind::Ne => ordering != Ordering::Equal, + CompareOpKind::Lt => ordering == Ordering::Less, + CompareOpKind::Le => ordering != Ordering::Greater, + CompareOpKind::Gt => ordering == Ordering::Greater, + CompareOpKind::Ge => ordering != Ordering::Less, + _ => return Err(Error::Invalid(format!("comparison {op:?}"))), + }; + Ok(Value::Bool(value)) +} + +fn arithmetic(op: &ArithmeticOpKind, left: Value, right: Value) -> Result { + let (left, right) = match (left, right) { + (Value::Int64(a), Value::Float64(b)) => (Value::Float64(a as f64), Value::Float64(b)), + (Value::Float64(a), Value::Int64(b)) => (Value::Float64(a), Value::Float64(b as f64)), + pair => pair, + }; + super::numeric(op, left, right) +} + +fn integer_float_cmp(integer: i64, float: f64) -> Option { + if float.is_nan() { + return None; + } + // These bounds are powers of two, exactly representable as Float64. + if float >= 9_223_372_036_854_775_808.0 { + return Some(Ordering::Less); + } + if float < -9_223_372_036_854_775_808.0 { + return Some(Ordering::Greater); + } + let integral = float as i64; + match integer.cmp(&integral) { + Ordering::Equal => 0.0_f64.partial_cmp(&float.fract()), + other => Some(other), + } +} + +fn cell_cmp(left: &Value, right: &Value) -> Option { + match (left, right) { + (Value::Int64(left), Value::Int64(right)) => Some(left.cmp(right)), + (Value::Float64(left), Value::Float64(right)) => left.partial_cmp(right), + (Value::Int64(left), Value::Float64(right)) => integer_float_cmp(*left, *right), + (Value::Float64(left), Value::Int64(right)) => { + integer_float_cmp(*right, *left).map(Ordering::reverse) + } + (Value::Utf8(left), Value::Utf8(right)) => Some(left.cmp(right)), + (Value::Bool(left), Value::Bool(right)) => Some(left.cmp(right)), + (Value::Timestamp(left), Value::Timestamp(right)) => Some(left.cmp(right)), + (Value::Map(left), Value::Map(right)) => { + for ((left_key, left_value), (right_key, right_value)) in left.iter().zip(right.iter()) + { + let order = cell_cmp(left_key, right_key)?; + if order != Ordering::Equal { + return Some(order); + } + let order = match (left_value, right_value) { + (Value::Null, Value::Null) => Ordering::Equal, + (Value::Null, _) => Ordering::Greater, + (_, Value::Null) => Ordering::Less, + _ => cell_cmp(left_value, right_value)?, + }; + if order != Ordering::Equal { + return Some(order); + } + } + Some(left.len().cmp(&right.len())) + } + _ => None, + } +} + +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] +pub struct CompiledExpression { + expression: QueryExpr, + schema: planner_types::pre_asap::Schema, + output: (DataType, bool), +} +impl CompiledExpression { + pub(crate) fn expression(&self) -> &QueryExpr { + &self.expression + } + + pub fn compile(expression: &QueryExpr, input: &Schema) -> Result { + let schema = input + .fields + .iter() + .map(|field| { + let planner_types::post_asap::SummaryFamilyType::Plain(dtype) = &field.dtype else { + return Err(Error::Invalid( + "scalar expression cannot consume opaque summary state".into(), + )); + }; + Ok(planner_types::pre_asap::Column::new( + field.name.clone(), + dtype.clone(), + field.nullable, + )) + }) + .collect::, Error>>()?; + let schema = planner_types::pre_asap::Schema::new(schema); + validate(expression, &schema)?; + let output = expression + .scalar_type(&schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + Ok(Self { + expression: expression.clone(), + schema, + output, + }) + } + pub(crate) fn dtype(&self) -> (DataType, bool) { + self.output.clone() + } + pub(crate) fn validate_input(&self, input: &Schema) -> Result<(), Error> { + let checked = Self::compile(&self.expression, input)?; + if checked.output != self.output { + return Err(Error::Invalid( + "persisted expression type differs from its semantics".into(), + )); + } + if input.fields.len() != self.schema.columns.len() + || input + .fields + .iter() + .zip(&self.schema.columns) + .any(|(field, column)| { + field.dtype + != planner_types::post_asap::SummaryFamilyType::Plain(column.dtype.clone()) + || field.nullable != column.nullable + }) + { + return Err(Error::Invalid( + "expression input differs from its bound schema".into(), + )); + } + Ok(()) + } + /// Evaluate a row under the same typed schema used when binding the expression. + pub fn evaluate(&self, row: &[Value]) -> Result { + if row.len() != self.schema.columns.len() + || row + .iter() + .zip(&self.schema.columns) + .any(|(value, column)| !value.matches(&column.dtype, column.nullable)) + { + return Err(Error::Invalid( + "expression input differs from its bound schema".into(), + )); + } + evaluate(&self.expression, row, &self.schema) + } +} +fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Result<(), Error> { + let invalid = || Error::Invalid(format!("unsupported scalar expression: {expr:?}")); + expr.scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + match expr { + QueryExpr::Column(_) | QueryExpr::Literal(_) => Ok(()), + QueryExpr::Arithmetic { left, right, .. } => { + for value in [left, right] { + validate(value, schema)?; + if !matches!( + value + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0, + DataType::Int64 | DataType::Float64 | DataType::Null + ) { + return Err(invalid()); + } + } + Ok(()) + } + QueryExpr::Compare { left, right, op } => { + if !matches!( + op, + CompareOpKind::Eq + | CompareOpKind::Ne + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge + ) { + return Err(invalid()); + } + validate(left, schema)?; + validate(right, schema)?; + let (a, _) = left + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + let (b, _) = right + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + fn comparable(dtype: &DataType) -> bool { + match dtype { + DataType::Null + | DataType::Int64 + | DataType::Float64 + | DataType::Utf8 + | DataType::Bool + | DataType::Timestamp => true, + DataType::Map { key, value, .. } => comparable(key) && comparable(value), + _ => false, + } + } + let numeric = |dtype: &DataType| matches!(dtype, DataType::Int64 | DataType::Float64); + if !comparable(&a) + || !comparable(&b) + || (a != b + && !matches!(a, DataType::Null) + && !matches!(b, DataType::Null) + && !(numeric(&a) && numeric(&b))) + { + return Err(invalid()); + } + Ok(()) + } + QueryExpr::FunctionCall { name, args } => { + if name != "asap_struct_field" + && name != "asap_element_access" + && planner_types::pre_asap::scalar_signature::MapScalarFunction::from_name(name) + .is_none() + { + return Err(invalid()); + } + for arg in args { + validate(arg, schema)?; + } + Ok(()) + } + QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + for part in parts { + validate(part, schema)?; + if !matches!( + part.scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0, + DataType::Bool | DataType::Null + ) { + return Err(invalid()); + } + } + Ok(()) + } + QueryExpr::Not(value) => { + validate(value, schema)?; + if !matches!( + value + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0, + DataType::Bool | DataType::Null + ) { + return Err(invalid()); + } + Ok(()) + } + QueryExpr::IsNull(value) | QueryExpr::IsNotNull(value) => validate(value, schema), + _ => Err(invalid()), + } +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn mixed_comparison_preserves_integer_precision_and_boundaries() { + assert_eq!( + integer_float_cmp(9_007_199_254_740_993, 9_007_199_254_740_992.0), + Some(Ordering::Greater) + ); + assert_eq!( + integer_float_cmp(i64::MAX, 9_223_372_036_854_775_808.0), + Some(Ordering::Less) + ); + assert_eq!( + integer_float_cmp(i64::MIN, -9_223_372_036_854_775_808.0), + Some(Ordering::Equal) + ); + assert_eq!(integer_float_cmp(-1, -1.5), Some(Ordering::Greater)); + assert_eq!(integer_float_cmp(1, 1.5), Some(Ordering::Less)); + assert_eq!(integer_float_cmp(0, f64::INFINITY), Some(Ordering::Less)); + assert_eq!( + integer_float_cmp(0, f64::NEG_INFINITY), + Some(Ordering::Greater) + ); + assert_eq!(integer_float_cmp(0, f64::NAN), None); + } +} diff --git a/crates/asap-physical-operators/src/key_by_label_values.rs b/crates/asap-physical-operators/src/key_by_label_values.rs new file mode 100644 index 00000000..e574da23 --- /dev/null +++ b/crates/asap-physical-operators/src/key_by_label_values.rs @@ -0,0 +1,126 @@ +use serde::{Deserialize, Serialize}; +// use std::collections::HashMap; +use std::hash::{Hash, Hasher}; + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct KeyByLabelValues { + // pub labels: HashMap, + pub labels: Vec, +} + +impl KeyByLabelValues { + pub fn new() -> Self { + Self { labels: Vec::new() } + } + + pub fn new_with_labels(labels: Vec) -> Self { + Self { labels } + } + + pub fn insert(&mut self, value: String) { + self.labels.push(value); + } + + pub fn get(&self, index: usize) -> Option<&String> { + self.labels.get(index) + } + + /// Encode labels as a semicolon-joined string — the canonical key format used + /// for sketch item hashing (CountMinSketch, CountSketch, HydraKLL). + pub fn to_semicolon_str(&self) -> String { + self.labels.join(";") + } + + #[cfg(test)] + /// Decode a semicolon-joined string back into a KeyByLabelValues. + pub fn from_semicolon_str(s: &str) -> Self { + Self { + labels: s.split(';').map(|s| s.to_string()).collect(), + } + } + + pub fn is_empty(&self) -> bool { + self.labels.is_empty() + } + + pub fn len(&self) -> usize { + self.labels.len() + } +} + +impl Hash for KeyByLabelValues { + fn hash(&self, state: &mut H) { + // Create a sorted vector of key-value pairs for consistent hashing + let mut sorted_pairs: Vec<_> = self.labels.iter().collect(); + sorted_pairs.sort(); + + for value in sorted_pairs { + value.hash(state); + } + } +} + +impl Default for KeyByLabelValues { + fn default() -> Self { + Self::new() + } +} + +impl std::fmt::Display for KeyByLabelValues { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{{")?; + let mut first = true; + for value in &self.labels { + if !first { + write!(f, ", ")?; + } + write!(f, "{value}")?; + first = false; + } + write!(f, "}}") + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_key_by_label_values() { + let mut key = KeyByLabelValues::new(); + key.insert("localhost:8080".to_string()); + key.insert("prometheus".to_string()); + + assert_eq!(key.len(), 2); + assert_eq!(key.get(0), Some(&"localhost:8080".to_string())); + assert_eq!(key.get(1), Some(&"prometheus".to_string())); + } + + #[test] + fn test_semicolon_roundtrip() { + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string(), "prod".to_string()]); + assert_eq!(key.to_semicolon_str(), "web;prod"); + let roundtripped = KeyByLabelValues::from_semicolon_str("web;prod"); + assert_eq!(roundtripped, key); + } + + #[test] + fn test_hash_consistency() { + let mut key1 = KeyByLabelValues::new(); + key1.insert("a".to_string()); + key1.insert("b".to_string()); + + let mut key2 = KeyByLabelValues::new(); + key2.insert("b".to_string()); + key2.insert("a".to_string()); + + // Should hash to the same value regardless of insertion order + let mut hasher1 = std::collections::hash_map::DefaultHasher::new(); + let mut hasher2 = std::collections::hash_map::DefaultHasher::new(); + + key1.hash(&mut hasher1); + key2.hash(&mut hasher2); + + assert_eq!(hasher1.finish(), hasher2.finish()); + } +} diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs new file mode 100644 index 00000000..345ee762 --- /dev/null +++ b/crates/asap-physical-operators/src/lib.rs @@ -0,0 +1,33 @@ +#![doc = include_str!("../README.md")] + +pub mod key_by_label_values; +pub mod measurement; +pub mod summary_kernels; +pub use summary_kernels::traits; + +mod statistic; +pub use key_by_label_values::KeyByLabelValues; +pub use measurement::Measurement; +pub use statistic::Statistic; +pub use traits::*; + +pub use expressions::arithmetic; +pub mod capability; +pub use summary_kernels::factory; + +/// The exact Planner contract used by these kernels. +pub use planner_types as planner; + +pub mod dag; + +pub mod readout; + +mod error; +pub use error::Error; +pub mod expressions; +pub mod operators; +pub mod physical_planner; +pub mod plan; +pub mod runtime; +pub mod sources; +pub mod values; diff --git a/crates/asap-physical-operators/src/measurement.rs b/crates/asap-physical-operators/src/measurement.rs new file mode 100644 index 00000000..57234f01 --- /dev/null +++ b/crates/asap-physical-operators/src/measurement.rs @@ -0,0 +1,48 @@ +use serde::{Deserialize, Serialize}; +use std::ops::Add; + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct Measurement { + pub value: f64, +} + +impl Measurement { + pub fn new(value: f64) -> Self { + Self { value } + } +} + +impl Add for Measurement { + type Output = Measurement; + + fn add(self, other: Measurement) -> Measurement { + Measurement::new(self.value + other.value) + } +} + +impl Add for &Measurement { + type Output = Measurement; + + fn add(self, other: &Measurement) -> Measurement { + Measurement::new(self.value + other.value) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_measurement_creation() { + let measurement = Measurement::new(42.5); + assert_eq!(measurement.value, 42.5); + } + + #[test] + fn test_measurement_addition() { + let m1 = Measurement::new(10.0); + let m2 = Measurement::new(20.0); + let result = m1 + m2; + assert_eq!(result.value, 30.0); + } +} diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs new file mode 100644 index 00000000..d929fc07 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -0,0 +1,355 @@ +use super::*; +impl Operator { + pub fn aggregate( + input: Schema, + groups: Vec, + measures: Vec<(String, Reduction)>, + ) -> Result { + validate_groups(&input, &groups)?; + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + for (name, reduction) in &measures { + let (t, n) = match reduction { + Reduction::Count => (DataType::Int64, false), + Reduction::Sum(i) | Reduction::Avg(i) => { + let (t, _) = plain(&input, *i)?; + if !matches!(t, DataType::Int64 | DataType::Float64) { + return Err(invalid("numeric aggregate input required")); + } + ( + if matches!(reduction, Reduction::Avg(_)) { + DataType::Float64 + } else { + t.clone() + }, + false, + ) + } + Reduction::Quantile { column, q } => { + if plain(&input, *column)?.0 != &DataType::Float64 || q.is_nan() { + return Err(invalid("quantile requires Float64 input and a numeric q")); + } + (DataType::Float64, false) + } + Reduction::Min(i) | Reduction::Max(i) => { + let (t, nullable) = plain(&input, *i)?; + if !ordered(t) { + return Err(invalid("ordered aggregate input required")); + } + (t.clone(), nullable || groups.is_empty()) + } + }; + fields.push(result_field(name, t, n)); + } + Ok(Self { + kind: Kind::Aggregate { + groups, + measures: measures.into_iter().map(|(_, r)| r).collect(), + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn window( + input: Schema, + intent: planner_types::pre_asap::AggIntent, + coordinate: usize, + value: usize, + groups: Vec, + window: Option<(i64, i64)>, + ) -> Result { + use planner_types::pre_asap::AggIntent; + validate_groups(&input, &groups)?; + let histogram = matches!(intent, AggIntent::HistogramQuantile { .. }); + if !matches!( + intent, + AggIntent::Rate + | AggIntent::Increase + | AggIntent::Count { .. } + | AggIntent::Sum { col: None } + | AggIntent::Avg { col: None } + | AggIntent::Min { col: None } + | AggIntent::Max { col: None } + | AggIntent::HistogramQuantile { .. } + ) { + return Err(invalid( + "unsupported temporal intent or unresolved value column", + )); + } + let coordinate_type = if histogram { + DataType::Float64 + } else { + DataType::Timestamp + }; + if plain(&input, coordinate)? != (&coordinate_type, false) + || plain(&input, value)? != (&DataType::Float64, false) + { + return Err(invalid("window coordinate/value schema mismatch")); + } + if (!histogram && !matches!(window, Some((start, end)) if start < end)) + || (histogram && window.is_some()) + { + return Err(invalid("invalid temporal window")); + } + let mut fields = groups + .iter() + .map(|i| input.fields[*i].clone()) + .collect::>(); + fields.push(result_field( + "value", + if matches!(intent, AggIntent::Count { .. }) { + DataType::Int64 + } else { + DataType::Float64 + }, + false, + )); + Ok(Self { + kind: Kind::Window { + intent: Box::new(intent), + coordinate, + value, + groups, + window, + }, + inputs: vec![input], + output: schema(fields), + }) + } +} +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] +pub enum Reduction { + Count, + Sum(usize), + Avg(usize), + Min(usize), + Max(usize), + /// PromQL `quantile`: linear interpolation between closest ranks. + Quantile { + column: usize, + q: f64, + }, +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &operator.kind { + Kind::Window { + intent, + coordinate, + value, + groups, + window, + } => { + crate::operators::aggregate::temporal::reduce( + rows, + intent, + groups, + *coordinate, + *value, + *window, + &context, + ) + .await? + } + Kind::Aggregate { groups, measures } => { + reduce(rows, groups, measures, &operator.inputs[0], &context).await? + } + _ => unreachable!(), + }; + Batch::try_new(output, result) + }) + .boxed_local()) +} + +pub(super) mod temporal; +async fn reduce( + rows: Vec>, + groups: &[usize], + measures: &[Reduction], + input: &Schema, + context: &RunContext, +) -> Result>, Error> { + let mut work = Cooperative::new(context); + let mut workspace = Workspace::new(context)?; + let mut grouped = BTreeMap::>, Vec>>::new(); + if rows.is_empty() && groups.is_empty() { + grouped.insert(vec![], vec![]); + } + for row in rows { + work.checkpoint().await?; + let key = group_key(&row, groups)?; + workspace.grow(std::mem::size_of::>())?; + if !grouped.contains_key(&key) { + workspace.grow(key_bytes(&key))?; + } + grouped.entry(key).or_default().push(row); + } + let mut output = Vec::new(); + for rows in grouped.into_values() { + work.checkpoint().await?; + let mut result = groups + .iter() + .map(|&i| rows[0][i].clone()) + .collect::>(); + for measure in measures { + result.push(reduce_one(&rows, measure, input, &mut work).await?); + } + workspace.grow(row_bytes(&result))?; + output.push(result); + } + Ok(output) +} + +// Matches Prometheus `quantile`: NaN for no values, ±Inf outside [0, 1], +// and NaN samples ordered first. +pub(super) fn quantile(q: f64, mut values: Vec) -> f64 { + if values.is_empty() { + return f64::NAN; + } + if q < 0. { + return f64::NEG_INFINITY; + } + if q > 1. { + return f64::INFINITY; + } + values.sort_by(|a, b| match (a.is_nan(), b.is_nan()) { + (true, true) => std::cmp::Ordering::Equal, + (true, false) => std::cmp::Ordering::Less, + (false, true) => std::cmp::Ordering::Greater, + _ => a.total_cmp(b), + }); + let rank = q * (values.len() - 1) as f64; + let low = rank.floor() as usize; + let high = (low + 1).min(values.len() - 1); + let weight = rank - low as f64; + values[low] * (1. - weight) + values[high] * weight +} + +async fn reduce_one( + rows: &[Vec], + measure: &Reduction, + input: &Schema, + work: &mut Cooperative, +) -> Result { + let column = match measure { + Reduction::Count => { + return Ok(Value::Int64( + i64::try_from(rows.len()).map_err(|_| invalid("count overflow"))?, + )) + } + Reduction::Sum(i) | Reduction::Avg(i) | Reduction::Min(i) | Reduction::Max(i) => *i, + Reduction::Quantile { column, q } => { + let mut values = Vec::with_capacity(rows.len()); + for row in rows { + work.checkpoint().await?; + match &row[*column] { + Value::Float64(value) => values.push(*value), + Value::Null => {} + _ => return Err(invalid("floating quantile value required")), + } + } + return Ok(Value::Float64(quantile(*q, values))); + } + }; + let values = rows + .iter() + .map(|r| &r[column]) + .filter(|v| !matches!(v, Value::Null)); + if matches!(measure, Reduction::Min(_) | Reduction::Max(_)) { + if plain(input, column)?.0 == &DataType::Float64 { + // Match exact-state kernels: ignore NaN when a numeric value exists. + let mut best: Option = None; + for value in values { + work.checkpoint().await?; + let Value::Float64(value) = value else { + return Err(invalid("floating aggregate value required")); + }; + best = Some(best.map_or(*value, |old| { + if matches!(measure, Reduction::Min(_)) { + old.min(*value) + } else { + old.max(*value) + } + })); + } + return Ok(best.map(Value::Float64).unwrap_or(Value::Null)); + } + let mut best: Option<&Value> = None; + for value in values { + work.checkpoint().await?; + if best + .map(|b| value.compare(b)) + .transpose()? + .is_none_or(|order| { + if matches!(measure, Reduction::Min(_)) { + order.is_lt() + } else { + order.is_gt() + } + }) + { + best = Some(value); + } + } + return Ok(best.cloned().unwrap_or(Value::Null)); + } + let mut count = 0usize; + let dtype = plain(input, column)?.0; + if dtype == &DataType::Int64 { + let mut sum = 0i128; + for v in values { + work.checkpoint().await?; + let Value::Int64(v) = v else { + return Err(invalid("integer aggregate value required")); + }; + sum = sum + .checked_add(i128::from(*v)) + .ok_or_else(|| invalid("integer aggregate overflow"))?; + count += 1; + } + return if matches!(measure, Reduction::Avg(_)) { + Ok(Value::Float64(sum as f64 / count as f64)) + } else { + Ok(Value::Int64( + i64::try_from(sum).map_err(|_| invalid("integer sum overflow"))?, + )) + }; + } + let mut sum = -0.0; + for v in values { + work.checkpoint().await?; + let Value::Float64(v) = v else { + return Err(invalid("floating aggregate value required")); + }; + sum += v; + count += 1; + } + Ok(Value::Float64(if matches!(measure, Reduction::Avg(_)) { + sum / count as f64 + } else { + sum + })) +} + +#[cfg(test)] +mod tests { + // Quantile follows Prometheus: interpolate ranks, NaN when empty, ±Inf outside [0, 1]. + #[test] + fn quantile_matches_prometheus_edge_cases() { + assert!(super::quantile(0.5, vec![]).is_nan()); + assert_eq!(super::quantile(0.5, vec![3.]), 3.); + assert_eq!(super::quantile(0.75, vec![4., 1., 2., 3.]), 3.25); + assert_eq!(super::quantile(-0.1, vec![1.]), f64::NEG_INFINITY); + assert_eq!(super::quantile(1.1, vec![1.]), f64::INFINITY); + assert_eq!(super::quantile(1., vec![2., f64::NAN, 1.]), 2.); + } +} diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs new file mode 100644 index 00000000..3ac0a3d1 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -0,0 +1,338 @@ +//! Windowed computations use Planner intents; deployments supply the input window. +use crate::{ + operators::{ + common::{key_bytes, row_bytes, Workspace}, + sort::cooperative_sort, + }, + runtime::{Cooperative, RunContext}, +}; +use crate::{ + values::{group_key, Value}, + Error, +}; +use planner_types::pre_asap::{AggIntent, ColumnRef}; +use std::collections::BTreeMap; + +pub(in crate::operators) async fn reduce( + rows: Vec>, + intent: &AggIntent, + groups: &[usize], + coordinate: usize, + value: usize, + window: Option<(i64, i64)>, + context: &RunContext, +) -> Result>, Error> { + let mut work = Cooperative::new(context); + let mut workspace = Workspace::new(context)?; + let mut grouped = + BTreeMap::>, (Vec, Vec<(f64, f64)>, Vec<(i64, f64)>)>::new(); + for row in rows { + work.checkpoint().await?; + let key = group_key(&row, groups)?; + workspace.grow(32)?; + if !grouped.contains_key(&key) { + workspace.grow(key_bytes(&key) + row_bytes(&row))?; + } + let entry = grouped.entry(key).or_insert_with(|| { + ( + groups.iter().map(|i| row[*i].clone()).collect(), + vec![], + vec![], + ) + }); + let Value::Float64(v) = row[value] else { + return Err(Error::Invalid("window value must be Float64".into())); + }; + match row[coordinate] { + Value::Timestamp(t) => entry.2.push((t, v)), + Value::Float64(bound) => entry.1.push((bound, v)), + _ => return Err(Error::Invalid("invalid window coordinate".into())), + } + } + let mut output = Vec::new(); + for (_, (mut keys, buckets, points)) in grouped { + work.checkpoint().await?; + let result = if let AggIntent::HistogramQuantile { q } = intent { + Some(Value::Float64(bucket_quantile(*q, buckets, context).await?)) + } else { + let points = cooperative_sort(points, |a, b| a.0.cmp(&b.0), context).await?; + let (start, end) = + window.ok_or_else(|| Error::Invalid("missing temporal window".into()))?; + if points.iter().any(|p| p.0 < start || p.0 > end) + || points.windows(2).any(|p| p[0].0 == p[1].0) + { + return Err(Error::Invalid( + "duplicate or out-of-window timestamp".into(), + )); + } + window_value(intent, &points, start, end)? + }; + if let Some(result) = result { + keys.push(result); + output.push(keys); + } + } + Ok(output) +} + +/// One series' value over its sorted samples in the window `(start, end]`. +/// `None` means PromQL emits no sample for this series. +pub(in crate::operators) fn window_value( + intent: &AggIntent, + points: &[(i64, f64)], + start: i64, + end: i64, +) -> Result, Error> { + Ok(match intent { + AggIntent::Rate => rate(points, start, end, true).map(Value::Float64), + AggIntent::Delta => rate(points, start, end, false) + .map(|v| Value::Float64(v * (end as f64 - start as f64) / 1000.)), + AggIntent::Increase => rate(points, start, end, true) + .map(|v| Value::Float64(v * (end as f64 - start as f64) / 1000.)), + AggIntent::Count { .. } => Some(Value::Int64( + i64::try_from(points.len()).map_err(|_| Error::Invalid("count overflow".into()))?, + )), + AggIntent::Sum { .. } => Some(Value::Float64(points.iter().map(|p| p.1).sum())), + AggIntent::Avg { .. } => Some(Value::Float64( + points.iter().map(|p| p.1).sum::() / points.len() as f64, + )), + AggIntent::Min { .. } => Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { + if a.is_nan() || p.1 < a { + p.1 + } else { + a + } + }))), + AggIntent::Max { .. } => Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { + if a.is_nan() || p.1 > a { + p.1 + } else { + a + } + }))), + AggIntent::IRate | AggIntent::IDelta => { + let [.., (t0, v0), (t1, v1)] = points else { + return Ok(None); + }; + let rate = matches!(intent, AggIntent::IRate); + // A counter reset makes the last value the increase. + let delta = if rate && v1 < v0 { *v1 } else { v1 - v0 }; + match (rate, t1 - t0) { + (_, 0) => None, + (true, interval) => Some(Value::Float64(delta / (interval as f64 / 1000.))), + (false, _) => Some(Value::Float64(delta)), + } + } + AggIntent::Changes | AggIntent::Resets => { + let changed = |(a, b): (f64, f64)| match intent { + AggIntent::Changes => a != b && !(a.is_nan() && b.is_nan()), + _ => b < a, + }; + let count = points + .windows(2) + .filter(|pair| changed((pair[0].1, pair[1].1))) + .count(); + Some(Value::Float64(count as f64)) + } + AggIntent::LastOverTime => points.last().map(|p| Value::Float64(p.1)), + AggIntent::Quantile { col: None, q, .. } => Some(Value::Float64(super::quantile( + *q, + points.iter().map(|p| p.1).collect(), + ))), + _ => return Err(Error::Invalid("unsupported temporal intent".into())), + }) +} + +/// Prometheus `extrapolatedRate`; `counter` enables reset correction and the zero bound. +fn rate(points: &[(i64, f64)], start: i64, end: i64, counter: bool) -> Option { + if points.len() < 2 { + return None; + } + let (first_t, first) = points[0]; + let (last_t, last) = *points.last()?; + let span = (last_t as f64 - first_t as f64) / 1000.; + if span <= 0. { + return None; + } + let mut delta = last - first; + for pair in points.windows(2) { + if counter && pair[1].1 < pair[0].1 { + delta += pair[0].1; + } + } + let average = span / (points.len() - 1) as f64; + let mut to_start = (first_t as f64 - start as f64) / 1000.; + let mut to_end = (end as f64 - last_t as f64) / 1000.; + if to_start >= average * 1.1 { + to_start = average / 2.; + } + // Apply the zero bound after the sparse-window half-interval cap. + if counter && delta > 0. && first >= 0. { + to_start = to_start.min(span * first / delta); + } + if to_end >= average * 1.1 { + to_end = average / 2.; + } + Some(delta * (span + to_start + to_end) / span / ((end as f64 - start as f64) / 1000.)) +} + +async fn bucket_quantile( + q: f64, + mut b: Vec<(f64, f64)>, + context: &RunContext, +) -> Result { + let mut work = Cooperative::new(context); + let _scratch = context.reserve(b.len().checked_mul(16).ok_or(Error::MemoryLimit)?)?; + if q.is_nan() { + return Ok(f64::NAN); + } + if q < 0. { + return Ok(f64::NEG_INFINITY); + } + if q > 1. { + return Ok(f64::INFINITY); + } + b.retain(|p| !p.0.is_nan()); + b = cooperative_sort(b, |a, b| a.0.total_cmp(&b.0), context).await?; + let mut buckets: Vec<(f64, f64)> = Vec::new(); + for p in b { + work.checkpoint().await?; + if let Some(last) = buckets.last_mut() { + if last.0 == p.0 { + last.1 += p.1; + continue; + } + } + buckets.push(p); + } + if buckets.len() < 2 || buckets.last().unwrap().0 != f64::INFINITY { + return Ok(f64::NAN); + } + let mut prev = buckets[0].1; + for p in buckets.iter_mut().skip(1) { + work.checkpoint().await?; + if p.1 < prev || (p.1 - prev).abs() <= 1e-12 * (p.1.abs() + prev.abs()) { + p.1 = prev; + } + prev = p.1; + } + let count = buckets.last().unwrap().1; + if count == 0. { + return Ok(f64::NAN); + } + let rank = q * count; + let idx = buckets[..buckets.len() - 1].partition_point(|p| p.1 < rank); + if idx == buckets.len() - 1 { + return Ok(buckets[idx - 1].0); + } + if idx == 0 && buckets[0].0 <= 0. { + return Ok(buckets[0].0); + } + let (start, base) = if idx == 0 { (0., 0.) } else { buckets[idx - 1] }; + let (end, upper) = buckets[idx]; + Ok(start + (end - start) * (rank - base) / (upper - base)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::{ + operators::Operator, + runtime::{batch_execution::evaluate_batch, Limits, RunContext, Scope}, + values::Batch, + }; + use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, + types::AccuracyTarget, + }; + use std::sync::Arc; + + // The same window operator must give the same answer in either engine phase. + #[test] + fn temporal_windows_execute_in_both_phases_and_count_is_integer() { + let schema = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "time".into(), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), + nullable: false, + }, + SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }, + ], + time_index: Some(0), + }); + for scope in [ + Scope::Query { + evaluation_time_ms: 2000, + revision: 1, + }, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }, + ] { + for (intent, expected) in [ + (AggIntent::Rate, Value::Float64(2.)), + (AggIntent::Increase, Value::Float64(4.)), + ( + AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }, + Value::Int64(3), + ), + ] { + let batch = Batch::try_new( + schema.clone(), + vec![ + vec![Value::Timestamp(0), Value::Float64(2.)], + vec![Value::Timestamp(1000), Value::Float64(4.)], + vec![Value::Timestamp(2000), Value::Float64(2.)], + ], + ) + .unwrap(); + let operator = + Operator::window(schema.clone(), intent, 0, 1, vec![], Some((0, 2000))) + .unwrap(); + let result = evaluate_batch( + batch, + vec![operator], + RunContext::new(scope.clone(), Limits::default()).unwrap(), + ) + .unwrap(); + assert_eq!( + format!("{:?}", result[0].rows()[0][0]), + format!("{expected:?}") + ); + } + } + assert!(Operator::window(schema, AggIntent::Rate, 0, 1, vec![], Some((1, 1))).is_err()); + } + + // Histogram interpolation requires an infinite terminal bucket and coalesces duplicates. + #[test] + fn histogram_boundaries_and_duplicate_buckets() { + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + let bucket_quantile = |q, buckets| { + futures::executor::block_on(super::bucket_quantile(q, buckets, &context)).unwrap() + }; + assert_eq!( + bucket_quantile(0.5, vec![(1., 1.), (1., 1.), (2., 4.), (f64::INFINITY, 4.)]), + 1. + ); + assert!(bucket_quantile(0.5, vec![(1., 2.), (2., 4.)]).is_nan()); + assert_eq!(bucket_quantile(-0.1, vec![]), f64::NEG_INFINITY); + } +} diff --git a/crates/asap-physical-operators/src/operators/aligned_binary.rs b/crates/asap-physical-operators/src/operators/aligned_binary.rs new file mode 100644 index 00000000..941a9f61 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/aligned_binary.rs @@ -0,0 +1,144 @@ +//! Arithmetic on complete, aligned population/window rows used by precomputation. +use super::*; +use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; +use std::collections::BTreeSet; + +impl Operator { + /// Match every row by the declared identity columns. Unlike an inner join, + /// incomplete or duplicate keys are errors: dropping an update changes state. + pub fn aligned_binary( + left: Schema, + right: Schema, + keys: Vec<(usize, usize)>, + values: (usize, usize), + operator: BinaryOperator, + ) -> Result { + if keys.is_empty() + || !matches!(operator.kind, BinaryOpKind::Arithmetic(_)) + || operator.vector_match.is_some() + { + return Err(invalid( + "aligned arithmetic requires explicit keys and arithmetic semantics", + )); + } + for (input, value) in [(&left, values.0), (&right, values.1)] { + if input.fields.get(value).is_none_or(|f| { + f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Float64) + }) { + return Err(invalid( + "aligned arithmetic requires non-null Float64 values", + )); + } + } + let mut left_keys = BTreeSet::new(); + let mut right_keys = BTreeSet::new(); + for &(l, r) in &keys { + if l == values.0 + || r == values.1 + || !left_keys.insert(l) + || !right_keys.insert(r) + || left + .fields + .get(l) + .zip(right.fields.get(r)) + .is_none_or(|(l, r)| l.nullable || r.nullable || l.dtype != r.dtype) + { + return Err(invalid("invalid aligned arithmetic keys")); + } + } + if left_keys.len() + 1 != left.fields.len() || right_keys.len() + 1 != right.fields.len() { + return Err(invalid( + "aligned arithmetic must account for every input column", + )); + } + Ok(Self { + output: left.clone(), + inputs: vec![left, right], + kind: Kind::AlignedBinary { + keys, + values, + operator, + }, + }) + } +} + +pub(super) fn execute<'a>( + op: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::AlignedBinary { + keys, + values, + operator, + } = &op.kind + else { + unreachable!() + }; + let right = inputs + .pop() + .ok_or_else(|| invalid("missing aligned right input"))?; + let left = inputs + .pop() + .ok_or_else(|| invalid("missing aligned left input"))?; + Ok(futures::stream::once(async move { + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + if left.is_empty() || left.len() != right.len() { + return Err(invalid( + "aligned arithmetic requires matching nonempty key sets", + )); + } + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + let columns = |side: bool| { + keys.iter() + .map(|&(l, r)| if side { r } else { l }) + .collect::>() + }; + let left_columns = columns(false); + let right_columns = columns(true); + let mut indexed = BTreeMap::new(); + for row in right { + work.checkpoint().await?; + let key = group_key(&row, &right_columns)?; + let Value::Float64(value) = row[values.1] else { + return Err(invalid("invalid aligned value")); + }; + if !value.is_finite() { + return Err(invalid("aligned arithmetic input is non-finite")); + } + workspace.grow(key_bytes(&key) + 64)?; + if indexed.insert(key, value).is_some() { + return Err(invalid("aligned arithmetic input has duplicate keys")); + } + } + let mut rows = Vec::new(); + for mut row in left { + work.checkpoint().await?; + let key = group_key(&row, &left_columns)?; + let right = indexed + .remove(&key) + .ok_or_else(|| invalid("aligned arithmetic input has missing or duplicate keys"))?; + let Value::Float64(left) = row[values.0] else { + return Err(invalid("invalid aligned value")); + }; + if !left.is_finite() { + return Err(invalid("aligned arithmetic input is non-finite")); + } + let result = crate::expressions::arithmetic::evaluate_binary(operator, left, right)?; + if !matches!(result, Value::Float64(value) if value.is_finite()) { + return Err(invalid("aligned arithmetic produced a non-finite update")); + } + row[values.0] = result; + workspace.grow(std::mem::size_of::>())?; + rows.push(row); + } + if !indexed.is_empty() { + return Err(invalid("aligned arithmetic has unmatched input keys")); + } + Batch::try_new(op.output.clone(), rows) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/common.rs b/crates/asap-physical-operators/src/operators/common.rs new file mode 100644 index 00000000..9f1709da --- /dev/null +++ b/crates/asap-physical-operators/src/operators/common.rs @@ -0,0 +1,75 @@ +use super::*; +pub(super) fn invalid(message: &str) -> Error { + Error::Invalid(message.into()) +} +pub(super) fn schema(fields: Vec) -> Schema { + Arc::new(SummarySchema { + fields, + time_index: None, + }) +} +pub(super) fn result_field(name: &str, dtype: DataType, nullable: bool) -> SummaryField { + SummaryField { + name: name.into(), + dtype: SummaryFamilyType::Plain(dtype), + nullable, + } +} + +pub(super) fn validate_groups(input: &Schema, groups: &[usize]) -> Result<(), Error> { + for &i in groups { + plain(input, i)?; + } + if groups + .iter() + .collect::>() + .len() + != groups.len() + { + return Err(invalid("duplicate group columns")); + } + Ok(()) +} +pub(super) async fn collect_rows( + mut input: Input<'_, Batch>, + context: &RunContext, +) -> Result<(Vec>, Vec), Error> { + let mut rows = Vec::new(); + let mut work = Cooperative::new(context); + let mut reservations = Vec::new(); + while let Some(batch) = input.next().await { + let batch = batch?; + reservations.push(context.reserve(batch.bytes())?); + for row in batch.rows() { + work.checkpoint().await?; + rows.push(row.clone()); + } + } + Ok((rows, reservations)) +} +/// Estimates retained workspace before growing collections. It is not an RSS limit. +pub(super) struct Workspace { + reservation: Reservation, + bytes: usize, +} +impl Workspace { + pub(super) fn new(context: &RunContext) -> Result { + Ok(Self { + reservation: context.reserve(0)?, + bytes: 0, + }) + } + pub(super) fn grow(&mut self, bytes: usize) -> Result<(), Error> { + self.bytes = self.bytes.checked_add(bytes).ok_or(Error::MemoryLimit)?; + self.reservation.resize(self.bytes) + } +} +pub(super) fn row_bytes(row: &[Value]) -> usize { + std::mem::size_of::>() + row.iter().map(Value::bytes).sum::() +} +pub(super) fn key_bytes(key: &[Vec]) -> usize { + 64 + key + .iter() + .map(|part| std::mem::size_of::>() + part.len()) + .sum::() +} diff --git a/crates/asap-physical-operators/src/operators/current_series.rs b/crates/asap-physical-operators/src/operators/current_series.rs new file mode 100644 index 00000000..937036c5 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/current_series.rs @@ -0,0 +1,135 @@ +//! A bounded instant-vector snapshot: select latest before removing stale markers. +use super::*; + +impl Operator { + pub fn current_series( + input: Schema, + identity: usize, + coordinate: usize, + value: usize, + lookback_ms: i64, + ) -> Result { + if lookback_ms <= 0 + || plain(&input, identity)? != (&DataType::Utf8, false) + || plain(&input, coordinate)? != (&DataType::Timestamp, false) + || plain(&input, value)? != (&DataType::Float64, false) + { + return Err(invalid("invalid current-series input contract")); + } + Ok(Self { + kind: Kind::CurrentSeries { + identity, + coordinate, + value, + lookback_ms, + }, + inputs: vec![input.clone()], + output: input, + }) + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::CurrentSeries { + identity, + coordinate, + value, + lookback_ms, + } = operator.kind + else { + unreachable!() + }; + let input = inputs + .pop() + .ok_or_else(|| invalid("current-series input missing"))?; + let output = operator.output.clone(); + let (start, end) = window(lookback_ms, &context)?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let mut latest = BTreeMap::, usize>::new(); + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + for (index, row) in rows.iter().enumerate() { + work.checkpoint().await?; + let Value::Timestamp(timestamp) = row[coordinate] else { + unreachable!() + }; + if timestamp <= start || timestamp > end { + continue; + } + let key = row[identity].key()?; + if let Some(&previous) = latest.get(&key) { + let Value::Timestamp(previous_time) = rows[previous][coordinate] else { + unreachable!() + }; + if timestamp < previous_time { + continue; + } + if timestamp == previous_time { + let (Value::Float64(a), Value::Float64(b)) = + (&row[value], &rows[previous][value]) + else { + unreachable!() + }; + if a.to_bits() != b.to_bits() { + return Err(invalid("conflicting samples for one series timestamp")); + } + continue; + } + } else { + workspace.grow(64 + key.len())?; + } + latest.insert(key, index); + } + let mut result = Vec::new(); + for index in latest.into_values() { + work.checkpoint().await?; + let Value::Float64(sample) = rows[index][value] else { + unreachable!() + }; + if sample.to_bits() == 0x7ff0_0000_0000_0002 { + continue; + } + workspace.grow(row_bytes(&rows[index]))?; + let mut row = rows[index].clone(); + row[coordinate] = Value::Timestamp(end); + result.push(row); + } + Batch::try_new(output, result) + }) + .boxed_local()) +} + +fn window(lookback_ms: i64, context: &RunContext) -> Result<(i64, i64), Error> { + let end = match context.scope { + crate::runtime::Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + crate::runtime::Scope::Ingestion { window_end_ms, .. } => window_end_ms, + }; + let start = end + .checked_sub(lookback_ms) + .ok_or_else(|| invalid("current-series window overflows"))?; + if let crate::runtime::Scope::Ingestion { + window_start_ms, .. + } = context.scope + { + if window_start_ms != start { + return Err(invalid( + "current-series maintenance window differs from lookback", + )); + } + } + Ok((start, end)) +} + +pub(super) fn validate_context(operator: &Operator, context: &RunContext) -> Result<(), Error> { + if let Kind::CurrentSeries { lookback_ms, .. } = operator.kind { + window(lookback_ms, context)?; + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/operators/filter.rs b/crates/asap-physical-operators/src/operators/filter.rs new file mode 100644 index 00000000..8862aacc --- /dev/null +++ b/crates/asap-physical-operators/src/operators/filter.rs @@ -0,0 +1,39 @@ +use super::*; +impl Operator { + pub fn filter(input: Schema, predicate: Expression) -> Result { + if predicate.dtype(&input)?.0 != DataType::Bool { + return Err(invalid("filter predicate must be boolean")); + } + Ok(Self { + kind: Kind::Filter(predicate), + inputs: vec![input.clone()], + output: input, + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::Filter(predicate) => Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + if matches!(predicate.evaluate(row)?, Value::Bool(true)) { + rows.push(row.clone()); + } + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/joins/mod.rs b/crates/asap-physical-operators/src/operators/joins/mod.rs new file mode 100644 index 00000000..a4397033 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/joins/mod.rs @@ -0,0 +1,228 @@ +use super::*; +impl Operator { + pub fn semi_join( + left: Schema, + right: Schema, + keys: Vec<(usize, usize)>, + ) -> Result { + if keys.is_empty() { + return Err(invalid("semi-join needs matching keys")); + } + for &(l, r) in &keys { + if plain(&left, l)?.0 != plain(&right, r)?.0 { + return Err(invalid("join key types differ")); + } + } + Ok(Self { + kind: Kind::SemiJoin { + keys, + require_complete_right: false, + }, + inputs: vec![left.clone(), right], + output: left, + }) + } + pub(crate) fn require_complete_right(mut self) -> Self { + if let Kind::SemiJoin { + require_complete_right, + .. + } = &mut self.kind + { + *require_complete_right = true; + } + self + } + pub(crate) fn certified_pruning_keys(&self) -> Option<&[(usize, usize)]> { + match &self.kind { + Kind::SemiJoin { + keys, + require_complete_right: true, + } => Some(keys), + _ => None, + } + } + pub fn relational_join( + left: Schema, + right: Schema, + kind: planner_types::pre_asap::JoinKind, + predicate: &planner_types::pre_asap::Predicate, + output: Schema, + ) -> Result { + use planner_types::pre_asap::JoinKind; + let mut joined = left.fields.clone(); + joined.extend(right.fields.clone()); + let predicate = + crate::expressions::CompiledExpression::compile(&predicate.0, &schema(joined.clone()))?; + if predicate.dtype().0 != DataType::Bool { + return Err(invalid("join predicate must be boolean")); + } + let fields = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + left.fields.clone() + } else { + for field in &mut joined[..left.fields.len()] { + if matches!(kind, JoinKind::Right | JoinKind::Full) { + field.nullable = true; + } + } + for field in &mut joined[left.fields.len()..] { + if matches!(kind, JoinKind::Left | JoinKind::Full) { + field.nullable = true; + } + } + joined + }; + Self { + kind: Kind::Join { + kind, + predicate: Box::new(predicate), + }, + inputs: vec![left, right], + output: schema(fields), + } + .with_output_schema(output) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + if let Kind::Join { kind, predicate } = &operator.kind { + let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; + let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; + return Ok(futures::stream::once(async move { + use planner_types::pre_asap::JoinKind; + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + let mut workspace = Workspace::new(&context)?; + let mut work = Cooperative::new(&context); + workspace.grow(right.len())?; + let mut result = Vec::new(); + let mut right_matched = vec![false; right.len()]; + for left_row in &left { + work.checkpoint().await?; + let mut matched = false; + for (i, right_row) in right.iter().enumerate() { + work.checkpoint().await?; + let mut joined = left_row.clone(); + joined.extend(right_row.iter().cloned()); + if *kind == JoinKind::Cross + || matches!(predicate.evaluate(&joined)?, Value::Bool(true)) + { + matched = true; + right_matched[i] = true; + match kind { + JoinKind::Semi => { + workspace.grow(row_bytes(left_row))?; + result.push(left_row.clone()); + break; + } + JoinKind::Anti => break, + _ => { + workspace.grow(row_bytes(&joined))?; + result.push(joined); + } + } + } + } + if !matched { + match kind { + JoinKind::Left | JoinKind::Full => { + let mut joined = left_row.clone(); + joined.resize( + joined.len() + operator.inputs[1].fields.len(), + Value::Null, + ); + workspace.grow(row_bytes(&joined))?; + result.push(joined); + } + JoinKind::Anti => { + workspace.grow(row_bytes(left_row))?; + result.push(left_row.clone()); + } + _ => {} + } + } + } + if matches!(kind, JoinKind::Right | JoinKind::Full) { + for (matched, row) in right_matched.into_iter().zip(right) { + work.checkpoint().await?; + if !matched { + let mut joined = vec![Value::Null; operator.inputs[0].fields.len()]; + joined.extend(row); + workspace.grow(row_bytes(&joined))?; + result.push(joined); + } + } + } + Batch::try_new(output, result) + }) + .boxed_local()); + } + if let Kind::SemiJoin { + keys, + require_complete_right, + } = &operator.kind + { + let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; + let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; + return Ok(futures::stream::once(async move { + // Poll both branches together: either may depend on a common producer. + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + let right_cols = keys.iter().map(|(_, r)| *r).collect::>(); + let left_cols = keys.iter().map(|(l, _)| *l).collect::>(); + let mut members = std::collections::BTreeSet::new(); + let mut workspace = Workspace::new(&context)?; + let mut work = Cooperative::new(&context); + for row in &right { + work.checkpoint().await?; + if right_cols.iter().all(|&i| matchable_key(&row[i])) { + let key = group_key(row, &right_cols)?; + if !members.contains(&key) { + workspace.grow(key_bytes(&key))?; + members.insert(key); + } + } else if *require_complete_right { + return Err(invalid( + "certified pruning candidate has an unmatchable key", + )); + } + } + let mut rows = Vec::new(); + let mut covered = std::collections::BTreeSet::new(); + for row in left { + work.checkpoint().await?; + if left_cols.iter().all(|&i| matchable_key(&row[i])) + && members.contains(&group_key(&row, &left_cols)?) + { + if *require_complete_right { + let key = group_key(&row, &left_cols)?; + if !covered.contains(&key) { + workspace.grow(key_bytes(&key))?; + covered.insert(key); + } + } + workspace.grow(std::mem::size_of::>())?; + rows.push(row); + } + } + if *require_complete_right && members != covered { + return Err(invalid("certified pruning key has no authoritative value")); + } + Batch::try_new(output, rows) + }) + .boxed_local()); + } + unreachable!() +} + +// Group keys canonicalize NaNs, but equality joins must not match them. +fn matchable_key(value: &Value) -> bool { + match value { + Value::Null => false, + Value::Float64(v) => !v.is_nan(), + _ => true, + } +} diff --git a/crates/asap-physical-operators/src/operators/limit.rs b/crates/asap-physical-operators/src/operators/limit.rs new file mode 100644 index 00000000..5f5e601f --- /dev/null +++ b/crates/asap-physical-operators/src/operators/limit.rs @@ -0,0 +1,69 @@ +use super::*; +impl Operator { + pub fn limit(input: Schema, n: u64, offset: u64, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + Ok(Self { + kind: Kind::Limit { n, offset, groups }, + inputs: vec![input.clone()], + output: input, + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::Limit { n, offset, groups } => { + let counts = BTreeMap::>, u64>::new(); + Ok(futures::stream::try_unfold( + (input, counts, Vec::::new(), false), + move |(mut input, mut counts, mut memory, done)| { + let output = output.clone(); + let context = context.clone(); + async move { + if done || *n == 0 { + return Ok(None); + } + let Some(batch) = input.next().await else { + return Ok(None); + }; + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + let key = group_key(row, groups)?; + if !counts.contains_key(&key) { + memory.push( + context.reserve( + key.iter() + .map(|part| part.len() + std::mem::size_of::>()) + .sum::() + + 64, + )?, + ); + } + let count = counts.entry(key).or_default(); + if *count >= *offset && count.saturating_sub(*offset) < *n { + rows.push(row.clone()); + } + *count = count.saturating_add(1); + } + let done = groups.is_empty() + && counts + .get(&vec![]) + .is_some_and(|count| count.saturating_sub(*offset) >= *n); + Ok(Some(( + Batch::try_new(output, rows)?, + (input, counts, memory, done), + ))) + } + }, + ) + .boxed_local()) + } + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs new file mode 100644 index 00000000..196e9e8a --- /dev/null +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -0,0 +1,370 @@ +//! Native physical operators. Each module owns its constructors and execution. +mod aligned_binary; +use crate::plan::{Boundedness, Emission, PhysicalOperator, PlanProperties}; +use crate::{ + runtime::{Cooperative, Input, OutputStream, Reservation, RunContext}, + values::{field, group_key, plain, Batch, Schema, Value}, + Error, +}; +use futures::StreamExt; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema, SummaryUpdate}, + pre_asap::{ColumnRef, DataType}, +}; +use std::{collections::BTreeMap, sync::Arc}; +pub(crate) mod common; +use crate::expressions::ordered; +pub use crate::expressions::Expression; +use common::*; +mod aggregate; +mod current_series; +mod filter; +mod joins; +mod limit; +mod projection; +mod scope_timestamp; +mod series_labels; +mod series_window; +mod sort; +mod source; +mod summary; +mod unchecked; +pub(crate) mod vector_binary; +pub(crate) mod vector_window; +pub use aggregate::Reduction; +pub use series_window::SubquerySteps; +pub use sort::SortKey; +pub use summary::ReadoutQuery; +#[derive(Clone, serde::Serialize, serde::Deserialize)] +enum Kind { + #[serde(skip)] + Source(Vec), + Constant { + value: Value, + dtype: DataType, + }, + ScopeTimestamp { + columns: Vec>, + }, + CurrentSeries { + identity: usize, + coordinate: usize, + value: usize, + lookback_ms: i64, + }, + Union, + VectorToScalar { + column: usize, + }, + VectorBinary { + operator: planner_types::post_asap::BinaryOperator, + return_bool: bool, + }, + AlignedBinary { + keys: Vec<(usize, usize)>, + values: (usize, usize), + operator: planner_types::post_asap::BinaryOperator, + }, + RangeWindow { + intent: Box>, + }, + HistogramQuantile, + SeriesWindow { + function: Option>>, + coordinate: usize, + value: usize, + range_ms: i64, + offset_ms: i64, + at_ms: Option, + steps: Option, + }, + SeriesLabels { + kind: planner_types::pre_asap::VectorMatchKind, + labels: Vec, + }, + SeriesBinary { + operator: planner_types::post_asap::BinaryOperator, + }, + Project(Vec), + Filter(Expression), + Limit { + n: u64, + offset: u64, + groups: Vec, + }, + Sort { + keys: Vec, + groups: Vec, + }, + Window { + intent: Box>, + coordinate: usize, + value: usize, + groups: Vec, + window: Option<(i64, i64)>, + }, + Aggregate { + groups: Vec, + measures: Vec, + }, + SemiJoin { + keys: Vec<(usize, usize)>, + require_complete_right: bool, + }, + Join { + kind: planner_types::pre_asap::JoinKind, + predicate: Box, + }, + SummaryBuild { + family: SummaryFamilyType, + value: usize, + time: Option, + groups: Vec, + }, + KeyedSummaryBuild { + family: SummaryFamilyType, + value: usize, + items: Vec, + groups: Vec, + }, + KeyedReadout { + state: usize, + k: usize, + }, + SummaryMerge { + state: usize, + groups: Vec, + }, + Readout { + state: usize, + query: ReadoutQuery, + }, +} +/// A bound operation has a fully checked input/output contract before execution. +#[derive(Clone, serde::Serialize, serde::Deserialize)] +#[serde(try_from = "unchecked::UncheckedOperator")] +pub struct Operator { + kind: Kind, + inputs: Vec, + output: Schema, +} +impl Operator { + pub(crate) fn row_preserving_input(&self) -> Option { + match self.kind { + Kind::Filter(_) | Kind::Sort { .. } | Kind::Limit { .. } | Kind::SemiJoin { .. } => { + Some(0) + } + _ => None, + } + } + + pub(crate) fn is_counter_readout(&self) -> bool { + matches!( + self.kind, + Kind::Readout { + query: ReadoutQuery::Exact(crate::summary_kernels::exact::ExactReadout { + statistic: crate::Statistic::Rate | crate::Statistic::Increase, + .. + }), + .. + } + ) + } + + pub(crate) fn with_counter_lookback(mut self, lookback: i64) -> Result { + if lookback <= 0 { + return Err(invalid("counter lookback must be positive")); + } + if let Kind::Readout { + query: ReadoutQuery::Exact(readout), + .. + } = &mut self.kind + { + readout.lookback_ms = Some(lookback); + } + Ok(self) + } + + /// Resolve a counter readout's logical lookback to this run's evaluation range. + pub(super) fn readout_range(&self, context: &RunContext) -> Result, Error> { + let Kind::Readout { + query: + ReadoutQuery::Exact(crate::summary_kernels::exact::ExactReadout { + lookback_ms: Some(lookback), + .. + }), + .. + } = &self.kind + else { + return Ok(None); + }; + let end = match context.scope { + crate::runtime::Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + crate::runtime::Scope::Ingestion { window_end_ms, .. } => window_end_ms, + }; + let start = end + .checked_sub(*lookback) + .ok_or_else(|| invalid("counter window overflows Int64"))?; + if let crate::runtime::Scope::Ingestion { + window_start_ms, .. + } = context.scope + { + if window_start_ms != start { + return Err(invalid( + "maintenance window differs from logical counter window", + )); + } + } + Ok(Some((start, end))) + } + + pub(crate) fn with_output_schema(mut self, output: Schema) -> Result { + if self.output.fields.len() != output.fields.len() + || self + .output + .fields + .iter() + .zip(&output.fields) + .any(|(actual, declared)| { + actual.dtype != declared.dtype || (actual.nullable && !declared.nullable) + }) + { + return Err(invalid("native output type differs from Planner output")); + } + if output.time_index.is_some_and(|i| { + i >= output.fields.len() + || output.fields[i].dtype != SummaryFamilyType::Plain(DataType::Timestamp) + }) { + return Err(invalid("invalid output time column")); + } + self.output = output; + Ok(self) + } + pub fn schema(&self) -> Schema { + self.output.clone() + } +} +impl PhysicalOperator for Operator { + fn requires_bounded_input(&self) -> bool { + matches!( + self.kind, + Kind::Sort { .. } + | Kind::AlignedBinary { .. } + | Kind::VectorBinary { .. } + | Kind::RangeWindow { .. } + | Kind::HistogramQuantile + | Kind::CurrentSeries { .. } + | Kind::SeriesWindow { .. } + | Kind::SeriesLabels { .. } + | Kind::SeriesBinary { .. } + | Kind::Aggregate { .. } + | Kind::Window { .. } + | Kind::Join { .. } + | Kind::SemiJoin { .. } + | Kind::SummaryBuild { .. } + | Kind::KeyedSummaryBuild { .. } + | Kind::SummaryMerge { .. } + | Kind::VectorToScalar { .. } + ) + } + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + let boundedness = match &self.kind { + Kind::Source(_) | Kind::Constant { .. } => Boundedness::Bounded, + Kind::Limit { groups, .. } if groups.is_empty() => Boundedness::Bounded, + _ => Boundedness::from_inputs(inputs), + }; + PlanProperties { + boundedness, + emission: if matches!(self.kind, Kind::ScopeTimestamp { .. }) { + inputs + .first() + .map_or(Emission::Unknown, |input| input.emission) + } else if self.requires_bounded_input() { + Emission::AfterInput + } else { + Emission::Incremental + }, + } + } + + fn name(&self) -> &str { + match self.kind { + Kind::Source(_) => "Source", + Kind::Constant { .. } => "Constant", + Kind::ScopeTimestamp { .. } => "ScopeTimestamp", + Kind::Union => "Union", + Kind::CurrentSeries { .. } => "CurrentSeries", + Kind::VectorToScalar { .. } => "VectorToScalar", + Kind::VectorBinary { .. } => "VectorBinary", + Kind::AlignedBinary { .. } => "AlignedBinary", + Kind::RangeWindow { .. } => "RangeWindow", + Kind::HistogramQuantile => "HistogramQuantile", + Kind::SeriesWindow { .. } => "SeriesWindow", + Kind::SeriesLabels { .. } => "SeriesLabels", + Kind::SeriesBinary { .. } => "SeriesBinary", + Kind::Project(_) => "Project", + Kind::Filter(_) => "Filter", + Kind::Limit { .. } => "Limit", + Kind::Sort { .. } => "Sort", + Kind::Aggregate { .. } => "Aggregate", + Kind::Window { .. } => "WindowAggregate", + Kind::SemiJoin { .. } => "SemiJoin", + Kind::Join { .. } => "RelationalJoin", + Kind::SummaryBuild { .. } | Kind::KeyedSummaryBuild { .. } => "SummaryAgg", + Kind::KeyedReadout { .. } => "SummaryEstimate", + Kind::SummaryMerge { .. } => "SummaryMerge", + Kind::Readout { .. } => "SummaryReadout", + } + } + fn validate_context(&self, context: &RunContext) -> Result<(), Error> { + current_series::validate_context(self, context)?; + series_window::validate_context(self, context)?; + self.readout_range(context).map(|_| ()) + } + fn input_schemas(&self) -> Vec { + self.inputs.clone() + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, value: &Batch) -> usize { + value.bytes() + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + match self.kind { + Kind::Source(_) | Kind::Constant { .. } | Kind::Union | Kind::VectorToScalar { .. } => { + source::execute(self, inputs, context) + } + Kind::VectorBinary { .. } => vector_binary::execute(self, inputs, context), + Kind::AlignedBinary { .. } => aligned_binary::execute(self, inputs, context), + Kind::RangeWindow { .. } | Kind::HistogramQuantile => { + vector_window::execute(self, inputs, context) + } + Kind::Project(_) => projection::execute(self, inputs, context), + Kind::CurrentSeries { .. } => current_series::execute(self, inputs, context), + Kind::ScopeTimestamp { .. } => scope_timestamp::execute(self, inputs, context), + Kind::SeriesWindow { .. } => series_window::execute(self, inputs, context), + Kind::SeriesLabels { .. } | Kind::SeriesBinary { .. } => { + series_labels::execute(self, inputs, context) + } + Kind::Filter(_) => filter::execute(self, inputs, context), + Kind::Limit { .. } => limit::execute(self, inputs, context), + Kind::Sort { .. } => sort::execute(self, inputs, context), + Kind::Window { .. } | Kind::Aggregate { .. } => { + aggregate::execute(self, inputs, context) + } + Kind::Join { .. } | Kind::SemiJoin { .. } => joins::execute(self, inputs, context), + Kind::SummaryMerge { .. } => summary::execute_merge(self, inputs, context), + Kind::SummaryBuild { .. } + | Kind::Readout { .. } + | Kind::KeyedSummaryBuild { .. } + | Kind::KeyedReadout { .. } => summary::execute(self, inputs, context), + } + } +} diff --git a/crates/asap-physical-operators/src/operators/projection.rs b/crates/asap-physical-operators/src/operators/projection.rs new file mode 100644 index 00000000..4a174ed9 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/projection.rs @@ -0,0 +1,56 @@ +use super::*; +impl Operator { + pub fn project(input: Schema, columns: Vec<(String, Expression)>) -> Result { + let fields = columns + .iter() + .map(|(name, e)| { + if let Expression::Column(index) = e { + let mut field = input + .fields + .get(*index) + .ok_or_else(|| invalid("projection column out of range"))? + .clone(); + field.name = name.clone(); + return Ok(field); + } + let (t, n) = e.dtype(&input)?; + Ok(result_field(name, t, n)) + }) + .collect::>()?; + Ok(Self { + kind: Kind::Project(columns.into_iter().map(|(_, e)| e).collect()), + inputs: vec![input], + output: schema(fields), + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::Project(expressions) => Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + let rows = batch + .rows() + .iter() + .map(|r| { + expressions + .iter() + .map(|e| e.evaluate(r)) + .collect::, _>>() + }) + .collect::, _>>()?; + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/scope_timestamp.rs b/crates/asap-physical-operators/src/operators/scope_timestamp.rs new file mode 100644 index 00000000..3561d6a6 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/scope_timestamp.rs @@ -0,0 +1,91 @@ +//! Run-scoped timestamp restoration after reduction. +use super::*; +use crate::runtime::Scope; + +impl Operator { + pub(crate) fn scope_timestamp(input: Schema, output: Schema) -> Result { + crate::values::validate_schema(&output)?; + let coordinate = output + .time_index + .ok_or_else(|| invalid("temporal output requires a time index"))?; + if plain(&output, coordinate)? != (&DataType::Timestamp, false) { + return Err(invalid("temporal output requires a non-null timestamp")); + } + let mut columns = Vec::new(); + let mut used = std::collections::BTreeSet::new(); + for (index, field) in output.fields.iter().enumerate() { + if index == coordinate { + columns.push(None); + continue; + } + let matches: Vec<_> = input + .fields + .iter() + .enumerate() + .filter(|(_, candidate)| { + candidate.dtype == field.dtype + && candidate.nullable == field.nullable + && (candidate.name == field.name + || !matches!(field.dtype, SummaryFamilyType::Plain(_))) + }) + .map(|(index, _)| index) + .collect(); + let [column] = matches.as_slice() else { + return Err(invalid("temporal output column missing or ambiguous")); + }; + if !used.insert(*column) { + return Err(invalid("temporal output repeats an input column")); + } + columns.push(Some(*column)); + } + if used.len() != input.fields.len() { + return Err(invalid("temporal output drops an input column")); + } + Ok(Self { + kind: Kind::ScopeTimestamp { columns }, + inputs: vec![input], + output, + }) + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::ScopeTimestamp { columns } = &operator.kind else { + return Err(invalid("scope timestamp operator required")); + }; + let input = inputs + .pop() + .ok_or_else(|| invalid("scope timestamp input missing"))?; + let output = operator.output.clone(); + let timestamp = match context.scope { + Scope::Ingestion { window_end_ms, .. } => window_end_ms, + Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + }; + Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + let rows = batch + .rows() + .iter() + .map(|row| { + columns + .iter() + .map(|column| { + column.map_or(Value::Timestamp(timestamp), |column| row[column].clone()) + }) + .collect() + }) + .collect(); + Batch::try_new(output.clone(), rows) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/series_labels.rs b/crates/asap-physical-operators/src/operators/series_labels.rs new file mode 100644 index 00000000..be248319 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/series_labels.rs @@ -0,0 +1,204 @@ +//! PromQL label-set rewriting and one-to-one vector matching over rows that +//! carry a series identity or plain label columns. +use super::*; +use planner_types::{ + post_asap::BinaryOperator, + pre_asap::{schema::PROMQL_SERIES_IDENTITY, BinaryOpKind, VectorMatchKind}, +}; + +type Labels = BTreeMap; + +/// Where a row's label set lives: the encoded series identity when present, +/// otherwise the non-empty Utf8 label columns. The one Float64 column is the +/// sample value, whatever an aggregate named it. +#[derive(Clone, Debug)] +struct Layout { + identity: Option, + labels: Vec, + value: usize, +} + +fn layout(input: &Schema) -> Result { + let mut identity = None; + let mut labels = Vec::new(); + let mut value = None; + for (i, field) in input.fields.iter().enumerate() { + match plain(input, i)? { + (DataType::Utf8, false) if field.name == PROMQL_SERIES_IDENTITY => identity = Some(i), + (DataType::Utf8, _) if field.name != PROMQL_SERIES_IDENTITY => labels.push(i), + (DataType::Float64, false) if value.is_none() => value = Some(i), + (DataType::Timestamp, false) if input.time_index == Some(i) => {} + _ => return Err(invalid("PromQL vector requires labels, time and one value")), + } + } + Ok(Layout { + identity, + labels, + value: value.ok_or_else(|| invalid("PromQL vector requires one value"))?, + }) +} + +impl Layout { + fn read(&self, input: &Schema, row: &[Value]) -> Result { + if let Some(i) = self.identity { + let Value::Utf8(encoded) = &row[i] else { + return Err(invalid("series identity must be Utf8")); + }; + let mut labels: Labels = + serde_json::from_str(encoded).map_err(|e| invalid(&e.to_string()))?; + // PromQL treats an empty label value as an absent label. + labels.retain(|_, v| !v.is_empty()); + return Ok(labels); + } + let mut labels = Labels::new(); + for &i in &self.labels { + match &row[i] { + Value::Utf8(v) if !v.is_empty() => { + labels.insert(input.fields[i].name.clone(), v.to_string()); + } + Value::Utf8(_) | Value::Null => {} + _ => return Err(invalid("label must be Utf8")), + } + } + Ok(labels) + } + + /// Replace the row's labels; a label column absent from `labels` is empty. + fn write(&self, input: &Schema, row: &mut [Value], labels: &Labels) -> Result<(), Error> { + if let Some(i) = self.identity { + let encoded = serde_json::to_string(labels).map_err(|e| invalid(&e.to_string()))?; + row[i] = Value::Utf8(encoded.into()); + } + for &i in &self.labels { + let value = labels.get(&input.fields[i].name).map_or("", String::as_str); + row[i] = Value::Utf8(value.into()); + } + Ok(()) + } +} + +impl Operator { + /// Rewrite each row's label set to PromQL's matching labels: `On` keeps + /// only `labels`; `Ignoring` drops `labels` and the metric name. + pub fn series_labels( + input: Schema, + kind: VectorMatchKind, + labels: Vec, + ) -> Result { + layout(&input)?; + Ok(Self { + kind: Kind::SeriesLabels { kind, labels }, + inputs: vec![input.clone()], + output: input, + }) + } + + /// PromQL one-to-one arithmetic between rows with equal label sets. The + /// result keeps the left row, without the metric name. + pub fn series_binary( + left: Schema, + right: Schema, + operator: BinaryOperator, + ) -> Result { + layout(&left)?; + layout(&right)?; + if !matches!(operator.kind, BinaryOpKind::Arithmetic(_)) || operator.vector_match.is_some() + { + return Err(invalid("series binary requires unmatched arithmetic")); + } + Ok(Self { + kind: Kind::SeriesBinary { operator }, + inputs: vec![left.clone(), right], + output: left, + }) + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let left_layout = layout(&operator.inputs[0])?; + let right = match operator.kind { + Kind::SeriesBinary { .. } => { + Some(inputs.pop().ok_or_else(|| invalid("missing right input"))?) + } + _ => None, + }; + let left = inputs.pop().ok_or_else(|| invalid("missing left input"))?; + Ok(futures::stream::once(async move { + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + let (rows, _memory) = collect_rows(left, &context).await?; + let mut result = Vec::new(); + match (&operator.kind, right) { + (Kind::SeriesLabels { kind, labels }, None) => { + for mut row in rows { + work.checkpoint().await?; + let mut set = left_layout.read(&output, &row)?; + match kind { + VectorMatchKind::On => set.retain(|k, _| labels.contains(k)), + VectorMatchKind::Ignoring => { + set.retain(|k, _| k != "__name__" && !labels.contains(k)) + } + } + left_layout.write(&output, &mut row, &set)?; + workspace.grow(row_bytes(&row))?; + result.push(row); + } + } + (Kind::SeriesBinary { operator: binary }, Some(right)) => { + let right_schema = &operator.inputs[1]; + let right_layout = layout(right_schema)?; + let (right, _right_memory) = collect_rows(right, &context).await?; + // Prometheus returns before matching when either side is empty. + if rows.is_empty() || right.is_empty() { + return Batch::try_new(output.clone(), vec![]); + } + let mut matches = BTreeMap::new(); + for row in &right { + work.checkpoint().await?; + let set = right_layout.read(right_schema, row)?; + workspace.grow(set.iter().map(|(k, v)| 64 + k.len() + v.len()).sum())?; + let Value::Float64(value) = row[right_layout.value] else { + return Err(invalid("vector value must be Float64")); + }; + // Prometheus rejects a duplicate on the one side. + if matches.insert(set, (value, false)).is_some() { + return Err(invalid( + "duplicate series for a match group on the right-hand side", + )); + } + } + for mut row in rows { + work.checkpoint().await?; + let mut set = left_layout.read(&output, &row)?; + let Some((value, matched)) = matches.get_mut(&set) else { + continue; + }; + // Only a left duplicate that finds a match is ambiguous. + if std::mem::replace(matched, true) { + return Err(invalid( + "many-to-one matching must be explicit (group_left/group_right)", + )); + } + let Value::Float64(left_value) = row[left_layout.value] else { + return Err(invalid("vector value must be Float64")); + }; + row[left_layout.value] = crate::expressions::arithmetic::evaluate_binary( + binary, left_value, *value, + )?; + set.remove("__name__"); + left_layout.write(&output, &mut row, &set)?; + workspace.grow(row_bytes(&row))?; + result.push(row); + } + } + _ => return Err(invalid("series label operator inputs mismatch")), + } + Batch::try_new(output.clone(), result) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/series_window.rs b/crates/asap-physical-operators/src/operators/series_window.rs new file mode 100644 index 00000000..20ab1aed --- /dev/null +++ b/crates/asap-physical-operators/src/operators/series_window.rs @@ -0,0 +1,243 @@ +//! PromQL per-series evaluation over the samples before an evaluation instant. +use super::*; +use planner_types::pre_asap::AggIntent; + +/// A PromQL subquery grid: every multiple of `step_ms` in +/// `(T - offset_ms - range_ms, T - offset_ms]`. `T` is `at_ms` (the subquery's +/// `@`) when present, else the query time. +#[derive(Clone, Copy, Debug, PartialEq, Eq, serde::Serialize, serde::Deserialize)] +pub struct SubquerySteps { + pub range_ms: i64, + pub step_ms: i64, + pub offset_ms: i64, + pub at_ms: Option, +} + +const STALE_MARKER: u64 = 0x7ff0_0000_0000_0002; +const MAX_SUBQUERY_STEPS: i64 = 100_000; + +impl Operator { + /// Evaluate each series at instant `t` over its samples in + /// `(e - offset_ms - range_ms, e - offset_ms]`. `t` is the query time, or + /// each step of `steps`; `e` is `at_ms` (the selector's `@`) when present, + /// else `t`. `function: None` is instant selection: the latest + /// sample, absent if it is a stale marker. Range functions ignore stale + /// markers. A series is every column except the time and `value` columns. + /// Output rows keep the input schema, with time `t` and the result value. + pub fn series_window( + input: Schema, + function: Option>, + range_ms: i64, + offset_ms: i64, + at_ms: Option, + steps: Option, + ) -> Result { + let coordinate = input + .time_index + .ok_or_else(|| invalid("series window requires a time column"))?; + let values = input + .fields + .iter() + .enumerate() + .filter(|(_, f)| f.name == "value") + .map(|(i, _)| i) + .collect::>(); + let [value] = values.as_slice() else { + return Err(invalid("series window requires one value column")); + }; + if plain(&input, coordinate)? != (&DataType::Timestamp, false) + || plain(&input, *value)? != (&DataType::Float64, false) + { + return Err(invalid( + "series window requires non-null time and Float64 value", + )); + } + if range_ms <= 0 || steps.is_some_and(|s| s.range_ms <= 0 || s.step_ms <= 0) { + return Err(invalid("series window ranges and steps must be positive")); + } + // Bounds per-run work independently of the data, as the backend's grid does. + if steps.is_some_and(|s| s.range_ms / s.step_ms > MAX_SUBQUERY_STEPS) { + return Err(invalid("subquery exceeds 100000 steps")); + } + if !matches!( + function, + None | Some( + AggIntent::Rate + | AggIntent::Increase + | AggIntent::Delta + | AggIntent::Count { .. } + | AggIntent::Sum { col: None } + | AggIntent::Avg { col: None } + | AggIntent::Min { col: None } + | AggIntent::Max { col: None } + | AggIntent::IRate + | AggIntent::IDelta + | AggIntent::Changes + | AggIntent::Resets + | AggIntent::LastOverTime + | AggIntent::Quantile { col: None, .. } + ) + ) { + return Err(invalid("unsupported PromQL range function")); + } + Ok(Self { + kind: Kind::SeriesWindow { + function: function.map(Box::new), + coordinate, + value: *value, + range_ms, + offset_ms, + at_ms, + steps, + }, + inputs: vec![input.clone()], + output: input, + }) + } +} + +/// The first and last evaluation instants and the step between them. The grid +/// is iterated, not allocated: its size depends only on the query. +fn evaluation_times( + context: &RunContext, + steps: Option, +) -> Result<(i64, i64, i64), Error> { + let crate::runtime::Scope::Query { + evaluation_time_ms, .. + } = context.scope + else { + return Err(invalid("series window requires a query evaluation time")); + }; + let Some(steps) = steps else { + return Ok((evaluation_time_ms, evaluation_time_ms, 1)); + }; + let overflow = || invalid("subquery grid overflows"); + let end = steps + .at_ms + .unwrap_or(evaluation_time_ms) + .checked_sub(steps.offset_ms) + .ok_or_else(overflow)?; + let start = end.checked_sub(steps.range_ms).ok_or_else(overflow)?; + let first = (start.div_euclid(steps.step_ms) + 1) + .checked_mul(steps.step_ms) + .ok_or_else(overflow)?; + Ok((first, end, steps.step_ms)) +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::SeriesWindow { + function, + coordinate, + value, + range_ms, + offset_ms, + at_ms, + steps, + } = &operator.kind + else { + unreachable!() + }; + let (coordinate, value) = (*coordinate, *value); + let input = inputs + .pop() + .ok_or_else(|| invalid("series window input missing"))?; + let times = evaluation_times(&context, *steps)?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + let identity = (0..operator.output.fields.len()) + .filter(|&i| i != coordinate && i != value) + .collect::>(); + let mut series = BTreeMap::>, (usize, Vec<(i64, f64)>)>::new(); + for (index, row) in rows.iter().enumerate() { + work.checkpoint().await?; + let key = group_key(row, &identity)?; + let (Value::Timestamp(time), Value::Float64(sample)) = (&row[coordinate], &row[value]) + else { + return Err(invalid("series window requires time and value samples")); + }; + workspace.grow(16)?; + if !series.contains_key(&key) { + workspace.grow(key_bytes(&key) + 64)?; + } + series + .entry(key) + .or_insert_with(|| (index, Vec::new())) + .1 + .push((*time, *sample)); + } + for (_, points) in series.values_mut() { + work.checkpoint().await?; + points.sort_by_key(|p| p.0); + if points.windows(2).any(|p| p[0].0 == p[1].0) { + return Err(invalid("duplicate sample timestamp for one series")); + } + } + let mut output = Vec::new(); + let (mut time, last, step) = times; + while time <= last && !series.is_empty() { + work.checkpoint().await?; + let overflow = || invalid("series window overflows"); + let end = at_ms + .unwrap_or(time) + .checked_sub(*offset_ms) + .ok_or_else(overflow)?; + let start = end.checked_sub(*range_ms).ok_or_else(overflow)?; + for (template, points) in series.values() { + work.checkpoint().await?; + // PromQL ranges are left-open: a sample at `start` is outside. + let first = points.partition_point(|p| p.0 <= start); + let last = points.partition_point(|p| p.0 <= end); + let points = &points[first..last]; + let result = match function { + None => points + .last() + .filter(|p| p.1.to_bits() != STALE_MARKER) + .map(|p| p.1), + Some(intent) => { + let fresh = points + .iter() + .copied() + .filter(|p| p.1.to_bits() != STALE_MARKER) + .collect::>(); + if fresh.is_empty() { + None + } else { + match aggregate::temporal::window_value(intent, &fresh, start, end)? { + Some(Value::Float64(v)) => Some(v), + Some(Value::Int64(v)) => Some(v as f64), + Some(_) => return Err(invalid("invalid range function result")), + None => None, + } + } + } + }; + if let Some(result) = result { + let mut row = rows[*template].clone(); + row[coordinate] = Value::Timestamp(time); + row[value] = Value::Float64(result); + workspace.grow(row_bytes(&row))?; + output.push(row); + } + } + let Some(next) = time.checked_add(step) else { + break; + }; + time = next; + } + Batch::try_new(operator.output.clone(), output) + }) + .boxed_local()) +} + +pub(super) fn validate_context(operator: &Operator, context: &RunContext) -> Result<(), Error> { + if let Kind::SeriesWindow { steps, .. } = operator.kind { + evaluation_times(context, steps)?; + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/operators/sort.rs b/crates/asap-physical-operators/src/operators/sort.rs new file mode 100644 index 00000000..71f81de0 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/sort.rs @@ -0,0 +1,172 @@ +use super::*; +impl Operator { + pub fn sort(input: Schema, keys: Vec, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + for key in &keys { + if !ordered(plain(&input, key.column)?.0) { + return Err(invalid("unsupported sort type")); + } + } + Ok(Self { + kind: Kind::Sort { keys, groups }, + inputs: vec![input.clone()], + output: input, + }) + } +} +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] +pub struct SortKey { + pub column: usize, + pub descending: bool, + pub nulls_first: bool, +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &operator.kind { + Kind::Sort { keys, groups } => { + let mut grouped = BTreeMap::>, Vec>>::new(); + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + for row in rows { + work.checkpoint().await?; + for key in keys { + if matches!(row[key.column], Value::Map(_)) && nested_nan(&row[key.column]) + { + return Err(invalid("NaN in collection sort key")); + } + } + let key = group_key(&row, groups)?; + workspace.grow(std::mem::size_of::>())?; + if !grouped.contains_key(&key) { + workspace.grow(key_bytes(&key))?; + } + grouped.entry(key).or_default().push(row); + } + let mut result = Vec::new(); + for rows in grouped.into_values() { + let rows = + cooperative_sort(rows, |a, b| compare_rows(a, b, keys), &context).await?; + result.extend(rows); + } + result + } + _ => unreachable!(), + }; + Batch::try_new(output, result) + }) + .boxed_local()) +} + +fn compare_rows(a: &[Value], b: &[Value], keys: &[SortKey]) -> std::cmp::Ordering { + use std::cmp::Ordering::*; + for key in keys { + let (a, b) = (&a[key.column], &b[key.column]); + let order = match (a, b) { + (Value::Null, Value::Null) => Equal, + (Value::Null, _) => { + if key.nulls_first { + Less + } else { + Greater + } + } + (_, Value::Null) => { + if key.nulls_first { + Greater + } else { + Less + } + } + (Value::Float64(a), Value::Float64(b)) if a.is_nan() || b.is_nan() => { + match (a.is_nan(), b.is_nan()) { + (true, true) => Equal, + (true, false) => Greater, + _ => Less, + } + } + _ => { + let order = a.compare(b).expect("bound ordered types"); + if key.descending { + order.reverse() + } else { + order + } + } + }; + if order != Equal { + return order; + } + } + Equal +} +fn nested_nan(value: &Value) -> bool { + match value { + Value::Float64(value) => value.is_nan(), + Value::Map(values) => values + .iter() + .any(|(key, value)| nested_nan(key) || nested_nan(value)), + Value::List(values) | Value::Struct(values) => values.iter().any(nested_nan), + _ => false, + } +} +/// Stable in-memory merge sort with bounded synchronous chunks. Scratch storage +/// is reserved before allocation; comparisons yield between merge steps. +pub(super) async fn cooperative_sort( + rows: Vec, + compare: impl Fn(&T, &T) -> std::cmp::Ordering, + context: &RunContext, +) -> Result, Error> { + use std::collections::VecDeque; + let bytes = rows + .len() + .checked_mul(std::mem::size_of::() + std::mem::size_of::>()) + .and_then(|n| n.checked_mul(3)) + .ok_or(Error::MemoryLimit)?; + let _scratch = context.reserve(bytes)?; + let mut work = Cooperative::new(context); + let mut rows = rows.into_iter(); + let mut runs = VecDeque::new(); + loop { + work.checkpoint().await?; + let mut chunk = rows.by_ref().take(256).collect::>(); + if chunk.is_empty() { + break; + } + chunk.sort_by(&compare); + runs.push_back(VecDeque::from(chunk)); + } + // Merge adjacent runs in rounds to preserve ties in original input order. + while runs.len() > 1 { + let mut next = VecDeque::new(); + while let Some(mut left) = runs.pop_front() { + let Some(mut right) = runs.pop_front() else { + next.push_back(left); + break; + }; + let mut merged = VecDeque::with_capacity(left.len() + right.len()); + while !left.is_empty() || !right.is_empty() { + work.checkpoint().await?; + let take_left = match (left.front(), right.front()) { + (Some(a), Some(b)) => !compare(a, b).is_gt(), + (Some(_), None) => true, + _ => false, + }; + merged.push_back(if take_left { + left.pop_front().unwrap() + } else { + right.pop_front().unwrap() + }); + } + next.push_back(merged); + } + runs = next; + } + Ok(runs.pop_front().unwrap_or_default().into()) +} diff --git a/crates/asap-physical-operators/src/operators/source.rs b/crates/asap-physical-operators/src/operators/source.rs new file mode 100644 index 00000000..42701e17 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/source.rs @@ -0,0 +1,96 @@ +use super::*; +impl Operator { + pub fn source(output: Schema, batches: Vec) -> Result { + crate::values::validate_schema(&output)?; + if batches.iter().any(|b| b.schema() != &output) { + return Err(invalid("source schema mismatch")); + } + Ok(Self { + kind: Kind::Source(batches), + inputs: vec![], + output, + }) + } + pub fn scalar(value: Value, dtype: DataType) -> Result { + let output = schema(vec![result_field( + "value", + dtype.clone(), + matches!(value, Value::Null), + )]); + Batch::try_new(output.clone(), vec![vec![value.clone()]])?; + Ok(Self { + kind: Kind::Constant { value, dtype }, + inputs: vec![], + output, + }) + } + pub fn vector_to_scalar(input: Schema, column: usize) -> Result { + if plain(&input, column)? != (&DataType::Float64, false) { + return Err(invalid("scalar conversion requires non-null Float64")); + } + Ok(Self { + kind: Kind::VectorToScalar { column }, + inputs: vec![input], + output: schema(vec![result_field("value", DataType::Float64, false)]), + }) + } + pub fn union(input: Schema, arity: usize) -> Result { + if arity == 0 { + return Err(invalid("union needs at least one input")); + } + Ok(Self { + kind: Kind::Union, + inputs: vec![input.clone(); arity], + output: input, + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + if let Kind::Source(batches) = &operator.kind { + return Ok(futures::stream::iter(batches.iter().cloned().map(Ok)).boxed_local()); + } + if let Kind::Constant { value, .. } = &operator.kind { + return Ok(futures::stream::once(async move { + Batch::try_new(output, vec![vec![value.clone()]]) + }) + .boxed_local()); + } + if matches!(operator.kind, Kind::Union) { + return Ok(futures::stream::select_all(inputs) + .map(|batch| batch.map(|batch| batch.value().clone())) + .boxed_local()); + } + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::VectorToScalar { column } => Ok(futures::stream::once(async move { + let mut input = input; + let mut work = Cooperative::new(&context); + let mut value = f64::NAN; + let mut count = 0usize; + while let Some(batch) = input.next().await { + for row in batch?.rows() { + work.checkpoint().await?; + count = count.saturating_add(1); + if let Value::Float64(v) = row[*column] { + value = v; + } + } + } + Batch::try_new( + output, + vec![vec![Value::Float64(if count == 1 { + value + } else { + f64::NAN + })]], + ) + }) + .boxed_local()), + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs new file mode 100644 index 00000000..cc87a75a --- /dev/null +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -0,0 +1,567 @@ +use super::*; +/// A summary readout: a sketch query, or an exact readout with typed parameters. +#[derive(Clone, Debug, PartialEq, serde::Serialize, serde::Deserialize)] +pub enum ReadoutQuery { + Sketch(planner_types::post_asap::SketchQuery), + Exact(crate::summary_kernels::exact::ExactReadout), +} + +impl Operator { + pub fn keyed_summary_build( + input: Schema, + family: SummaryFamilyType, + value: usize, + items: Vec, + groups: Vec, + ) -> Result { + use crate::summary_kernels::weighted_frequency::WeightedFrequency; + crate::values::validate_family(&family)?; + let SummaryFamilyType::Sketch(kind, _) = &family else { + return Err(invalid("keyed sketch required")); + }; + WeightedFrequency::configuration(kind)?; + validate_groups(&input, &groups)?; + if items.is_empty() || plain(&input, value)? != (&DataType::Float64, false) { + return Err(invalid( + "keyed summary requires identities and non-null Float64 weights", + )); + } + for &item in &items { + if !matches!( + plain(&input, item)?.0, + DataType::Utf8 + | DataType::Timestamp + | DataType::Int64 + | DataType::Float64 + | DataType::Bool + | DataType::Null + ) { + return Err(invalid("unsupported keyed summary identity type")); + } + } + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }); + Ok(Self { + kind: Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn keyed_readout( + input: Schema, + state: usize, + k: usize, + output: Schema, + ) -> Result { + use crate::summary_kernels::weighted_frequency::WeightedFrequency; + crate::values::validate_family(&field(&input, state)?.dtype)?; + let SummaryFamilyType::Sketch(kind, _) = &field(&input, state)?.dtype else { + return Err(invalid("keyed readout requires summary state")); + }; + let (_, _, _, capacity) = WeightedFrequency::configuration(kind)?; + if k > capacity || output.fields.len() <= input.fields.len() { + return Err(invalid("invalid keyed readout shape or capacity")); + } + if state + 1 != input.fields.len() + || output.fields[..state] != input.fields[..state] + || output.fields.last().unwrap().dtype != SummaryFamilyType::Plain(DataType::Float64) + { + return Err(invalid( + "keyed readout must preserve partitions and return a Float64 score", + )); + } + crate::values::validate_schema(&output)?; + Ok(Self { + kind: Kind::KeyedReadout { state, k }, + inputs: vec![input], + output, + }) + } + pub fn summary_build( + input: Schema, + family: SummaryFamilyType, + value: usize, + time: Option, + groups: Vec, + ) -> Result { + crate::values::validate_family(&family)?; + validate_groups(&input, &groups)?; + if plain(&input, value)?.0 != &DataType::Float64 { + return Err(invalid("summary numeric update requires Float64")); + } + if let Some(time) = time { + if plain(&input, time)? != (&DataType::Timestamp, false) { + return Err(invalid("summary time column must be a timestamp")); + } + } + if time.is_none() + && matches!( + family, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Rate + | planner_types::post_asap::ExactKind::Increase, + _ + ) + ) + { + return Err(invalid("counter summary requires a timestamp column")); + } + crate::capability::validate_summary_kernel( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .map_err(Error::Invalid)?; + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }); + Ok(Self { + kind: Kind::SummaryBuild { + family, + value, + time, + groups, + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn summary_merge(input: Schema, state: usize, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + crate::values::validate_family(&field(&input, state)?.dtype)?; + if matches!(field(&input, state)?.dtype, SummaryFamilyType::Plain(_)) { + return Err(invalid("summary state required")); + } + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(input.fields[state].clone()); + Ok(Self { + kind: Kind::SummaryMerge { state, groups }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn readout(input: Schema, state: usize, query: ReadoutQuery) -> Result { + let family = &field(&input, state)?.dtype; + crate::values::validate_family(family)?; + match &query { + ReadoutQuery::Sketch(query) => { + crate::capability::validate_sketch_readout(family, query)? + } + ReadoutQuery::Exact(readout) => { + crate::capability::validate_exact_readout(family, readout)? + } + } + let mut fields = input.fields.clone(); + let result_type = if matches!( + fields[state].dtype, + SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) + ) { + DataType::Int64 + } else { + DataType::Float64 + }; + // A state-only row represents the global population. Its extrema may + // be empty, just like an ordinary ungrouped MIN/MAX aggregate. + let nullable = fields.len() == 1 + && matches!( + fields[state].dtype, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Min + | planner_types::post_asap::ExactKind::Max, + _ + ) + ); + fields[state] = result_field("value", result_type, nullable); + Ok(Self { + kind: Kind::Readout { state, query }, + inputs: vec![input], + output: schema(fields), + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let range_ms = operator.readout_range(&context)?; + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::SummaryBuild { + family, + value, + time, + groups, + } => Ok(futures::stream::once(async move { + Batch::try_new( + output, + build_summary(input, family, *value, *time, groups, &context).await?, + ) + }) + .boxed_local()), + Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + } => Ok(futures::stream::once(async move { + Batch::try_new( + output, + build_keyed_summary(input, family, *value, items, groups, &context).await?, + ) + }) + .boxed_local()), + Kind::KeyedReadout { state, k } => Ok(input + .map(move |batch| { + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + let Value::Summary { state: summary, .. } = &row[*state] else { + return Err(invalid("summary value required")); + }; + let summary = summary + .as_any() + .downcast_ref::() + .ok_or_else(|| invalid("weighted frequency typed state required"))?; + for items in summary.rows(*k) { + let mut values = row[..*state].to_vec(); + values.extend(items); + // The typed output schema restores epoch-millisecond + // timestamp keys from the kernel's Int64 representation. + for (value, field) in values.iter_mut().zip(&output.fields) { + if field.dtype == SummaryFamilyType::Plain(DataType::Timestamp) { + if let Value::Int64(time) = value { + *value = Value::Timestamp(*time); + } + } + } + rows.push(values); + } + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + Kind::Readout { state, query } => Ok(input + .map(move |batch| { + let batch = batch?; + let mut rows = batch.rows().to_vec(); + if let ReadoutQuery::Exact(readout) = query { + rows.retain(|row| !matches!(&row[*state], Value::Summary { state: summary, .. } + if crate::readout::insufficient_counter_samples(summary.as_ref(), readout.statistic))); + } + for row in &mut rows { + let Value::Summary { state: summary, .. } = &row[*state] else { + return Err(invalid("summary value required")); + }; + row[*state] = match query { + ReadoutQuery::Sketch(query) => Value::Float64( + summary + .estimate(query) + .map_err(|e| Error::Operator(e.to_string()))?, + ), + ReadoutQuery::Exact(readout) => { + let exact = summary + .as_any() + .downcast_ref::() + .ok_or_else(|| invalid("exact readout requires exact state"))?; + if output.fields[*state].dtype == SummaryFamilyType::Plain(DataType::Int64) { + let count = exact.count().ok_or_else(|| { + Error::Operator("exact count state lacks an integer count".into()) + })?; + Value::Int64(i64::try_from(count).map_err(|_| { + Error::Operator("exact count exceeds Int64".into()) + })?) + } else { + match exact + .readout(readout.statistic, range_ms, None) + .map_err(|e| Error::Operator(e.to_string()))? + { + Some(value) => Value::Float64(value), + None if output.fields[*state].nullable => Value::Null, + None => { + return Err(Error::Operator( + "empty exact population".into(), + )) + } + } + } + } + }; + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => unreachable!(), + } +} +pub(super) fn execute_merge<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &operator.kind { + Kind::SummaryMerge { state, groups } => { + merge_summary(rows, *state, groups, &context).await? + } + _ => unreachable!(), + }; + Batch::try_new(output, result) + }) + .boxed_local()) +} + +async fn build_summary( + mut input: Input<'_, Batch>, + family: &SummaryFamilyType, + value: usize, + time: Option, + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + type State = ( + Vec, + Box, + Reservation, + usize, + Option, + ); + let create = |labels: Vec, key_bytes: usize| -> Result { + let updater = crate::factory::create_planner_accumulator( + family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .map_err(Error::Operator)?; + let overhead = labels.iter().map(Value::bytes).sum::() + key_bytes + 64; + let memory = context.reserve(updater.memory_usage_bytes() + overhead)?; + Ok((labels, updater, memory, overhead, None)) + }; + let mut work = Cooperative::new(context); + let mut states = BTreeMap::>, State>::new(); + if groups.is_empty() { + states.insert(vec![], create(vec![], 0)?); + } + let ordered_time = matches!( + family, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Rate + | planner_types::post_asap::ExactKind::Increase, + _ + ) + ); + while let Some(batch) = input.next().await { + let batch = batch?; + for row in batch.rows() { + work.checkpoint().await?; + let key = group_key(row, groups)?; + if !states.contains_key(&key) { + let labels = groups.iter().map(|&i| row[i].clone()).collect(); + let state = create( + labels, + key.iter() + .map(|v| v.len() + std::mem::size_of::>()) + .sum(), + )?; + states.insert(key.clone(), state); + } + let (_, updater, memory, overhead, previous) = + states.get_mut(&key).expect("inserted group"); + // SQL aggregates ignore NULL samples while retaining the group. + // A missing counter sample also contributes no observation. + let value = match row[value] { + Value::Float64(value) => value, + Value::Null => continue, + _ => return Err(invalid("summary update type")), + }; + let timestamp = if let Some(time) = time { + let Value::Timestamp(time) = row[time] else { + return Err(invalid("summary time type")); + }; + time + } else { + 0 + }; + if ordered_time && previous.is_some_and(|prior| timestamp <= prior) { + return Err(Error::Operator( + "counter samples must have strictly increasing timestamps within each group" + .into(), + )); + } + updater + .validate_single_input(value) + .map_err(Error::Operator)?; + updater.update_single(value, timestamp); + *previous = Some(timestamp); + memory.resize(updater.memory_usage_bytes() + *overhead)?; + } + } + Ok(states + .into_values() + .map(|(mut labels, updater, _memory, _, _)| { + labels.push(Value::Summary { + family: family.clone(), + state: Arc::from(updater.into_accumulator()), + }); + labels + }) + .collect()) +} +async fn merge_summary( + rows: Vec>, + state_column: usize, + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + type GroupState = (Vec, SummaryFamilyType, Arc); + let mut states: BTreeMap>, GroupState> = BTreeMap::new(); + let mut work = Cooperative::new(context); + let mut memory = context.reserve(0)?; + let mut retained = 0usize; + for row in rows { + work.checkpoint().await?; + let Value::Summary { family, state } = &row[state_column] else { + return Err(invalid("summary state required")); + }; + let key = group_key(&row, groups)?; + if let Some((_, expected, existing)) = states.get_mut(&key) { + if expected != family { + return Err(invalid("incompatible summary family")); + } + let old_bytes = existing.approx_memory_bytes(); + // Reserve an estimate for the replacement while both input states remain live. + memory.resize( + retained + .checked_add(old_bytes) + .and_then(|n| n.checked_add(state.approx_memory_bytes())) + .ok_or(Error::MemoryLimit)?, + )?; + *existing = Arc::from( + existing + .merge_with(state.as_ref()) + .map_err(|e| Error::Operator(e.to_string()))?, + ); + retained = retained + .checked_sub(old_bytes) + .and_then(|n| n.checked_add(existing.approx_memory_bytes())) + .ok_or(Error::MemoryLimit)?; + memory.resize(retained)?; + } else { + retained = retained + .checked_add(key_bytes(&key) + row_bytes(&row)) + .ok_or(Error::MemoryLimit)?; + memory.resize(retained)?; + states.insert( + key, + ( + groups.iter().map(|&i| row[i].clone()).collect(), + family.clone(), + state.clone(), + ), + ); + } + } + Ok(states + .into_values() + .map(|(mut keys, family, state)| { + keys.push(Value::Summary { family, state }); + keys + }) + .collect()) +} + +async fn build_keyed_summary( + mut input: Input<'_, Batch>, + family: &SummaryFamilyType, + value: usize, + items: &[usize], + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + use crate::{summary_kernels::weighted_frequency::WeightedFrequency, AggregateCore}; + let SummaryFamilyType::Sketch(kind, _) = family else { + unreachable!() + }; + let (algorithm, width, depth, capacity) = WeightedFrequency::configuration(kind)?; + let mut work = Cooperative::new(context); + let mut states = + BTreeMap::>, (Vec, WeightedFrequency, Reservation, usize)>::new(); + while let Some(batch) = input.next().await { + let batch = batch?; + for row in batch.rows() { + work.checkpoint().await?; + let key = group_key(row, groups)?; + if !states.contains_key(&key) { + let labels = groups.iter().map(|&i| row[i].clone()).collect::>(); + let overhead = labels.iter().map(Value::bytes).sum::() + + key.iter().map(|v| v.len() + 24).sum::() + + 128; + let bytes = width + .checked_mul(depth) + .and_then(|n| n.checked_mul(8)) + .and_then(|n| n.checked_add(overhead)) + .ok_or_else(|| invalid("weighted frequency memory size overflow"))?; + let reservation = context.reserve(bytes)?; + states.insert( + key.clone(), + ( + labels, + WeightedFrequency::new(algorithm, width, depth, capacity)?, + reservation, + overhead, + ), + ); + } + let (_, summary, reservation, overhead) = states.get_mut(&key).unwrap(); + let Value::Float64(weight) = row[value] else { + return Err(invalid("weighted frequency weight type")); + }; + summary.update( + &items + .iter() + .map(|&i| match &row[i] { + Value::Timestamp(time) => Value::Int64(*time), + value => value.clone(), + }) + .collect::>(), + weight, + )?; + reservation.resize(summary.approx_memory_bytes() + *overhead)?; + } + } + Ok(states + .into_values() + .map(|(mut labels, summary, _, _)| { + labels.push(Value::Summary { + family: family.clone(), + state: Arc::new(summary), + }); + labels + }) + .collect()) +} diff --git a/crates/asap-physical-operators/src/operators/unchecked.rs b/crates/asap-physical-operators/src/operators/unchecked.rs new file mode 100644 index 00000000..46bc6a0c --- /dev/null +++ b/crates/asap-physical-operators/src/operators/unchecked.rs @@ -0,0 +1,156 @@ +//! Deserialized operators are validated before use, whatever the encoding. +use super::*; + +#[derive(serde::Deserialize)] +#[serde(deny_unknown_fields)] +pub(super) struct UncheckedOperator { + kind: Kind, + inputs: Vec, + output: Schema, +} +impl TryFrom for Operator { + type Error = Error; + fn try_from(unchecked: UncheckedOperator) -> Result { + let expected_kind = + serde_json::to_value(&unchecked.kind).map_err(|error| invalid(&error.to_string()))?; + let UncheckedOperator { + kind, + inputs, + output, + } = unchecked; + for schema in inputs.iter().chain(std::iter::once(&output)) { + crate::values::validate_schema(schema)?; + } + let input = |index| { + inputs + .get(index) + .cloned() + .ok_or_else(|| invalid("missing operator input")) + }; + let op = match kind { + Kind::Source(_) => return Err(invalid("physical plans cannot serialize live sources")), + Kind::Constant { value, dtype } => Operator::scalar(value, dtype)?, + Kind::ScopeTimestamp { .. } => Operator::scope_timestamp(input(0)?, output.clone())?, + Kind::Union => Operator::union(input(0)?, inputs.len())?, + Kind::CurrentSeries { + identity, + coordinate, + value, + lookback_ms, + } => Operator::current_series(input(0)?, identity, coordinate, value, lookback_ms)?, + Kind::VectorToScalar { column } => Operator::vector_to_scalar(input(0)?, column)?, + Kind::VectorBinary { + operator, + return_bool, + } => Operator::vector_binary(input(0)?, input(1)?, operator, return_bool)?, + Kind::AlignedBinary { + keys, + values, + operator, + } => Operator::aligned_binary(input(0)?, input(1)?, keys, values, operator)?, + Kind::RangeWindow { intent } => Operator::range_window(*intent)?, + Kind::HistogramQuantile => Operator::histogram_quantile(), + Kind::SeriesWindow { + function, + range_ms, + offset_ms, + at_ms, + steps, + .. + } => Operator::series_window( + input(0)?, + function.map(|f| *f), + range_ms, + offset_ms, + at_ms, + steps, + )?, + Kind::SeriesLabels { kind, labels } => { + Operator::series_labels(input(0)?, kind, labels)? + } + Kind::SeriesBinary { operator } => { + Operator::series_binary(input(0)?, input(1)?, operator)? + } + Kind::Project(expressions) => { + if expressions.len() != output.fields.len() { + return Err(invalid("projection width mismatch")); + } + Operator::project( + input(0)?, + output + .fields + .iter() + .zip(expressions) + .map(|(f, e)| (f.name.clone(), e)) + .collect(), + )? + } + Kind::Filter(expression) => Operator::filter(input(0)?, expression)?, + Kind::Limit { n, offset, groups } => Operator::limit(input(0)?, n, offset, groups)?, + Kind::Sort { keys, groups } => Operator::sort(input(0)?, keys, groups)?, + Kind::Window { + intent, + coordinate, + value, + groups, + window, + } => Operator::window(input(0)?, *intent, coordinate, value, groups, window)?, + Kind::Aggregate { groups, measures } => { + if groups.len() + measures.len() != output.fields.len() { + return Err(invalid("aggregate width mismatch")); + } + let names = output.fields[groups.len()..].iter().map(|f| f.name.clone()); + Operator::aggregate(input(0)?, groups, names.zip(measures).collect())? + } + Kind::SemiJoin { + keys, + require_complete_right, + } => { + let operator = Operator::semi_join(input(0)?, input(1)?, keys)?; + if require_complete_right { + operator.require_complete_right() + } else { + operator + } + } + Kind::Join { kind, predicate } => Operator::relational_join( + input(0)?, + input(1)?, + kind, + &planner_types::pre_asap::Predicate(std::rc::Rc::new( + predicate.expression().clone(), + )), + output.clone(), + )?, + Kind::SummaryBuild { + family, + value, + time, + groups, + } => Operator::summary_build(input(0)?, family, value, time, groups)?, + Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + } => Operator::keyed_summary_build(input(0)?, family, value, items, groups)?, + Kind::KeyedReadout { state, k } => { + Operator::keyed_readout(input(0)?, state, k, output.clone())? + } + Kind::SummaryMerge { state, groups } => { + Operator::summary_merge(input(0)?, state, groups)? + } + Kind::Readout { state, query } => Operator::readout(input(0)?, state, query)?, + } + .with_output_schema(output)?; + if serde_json::to_value(&op.kind).map_err(|error| invalid(&error.to_string()))? + != expected_kind + { + return Err(invalid("operator contains inconsistent compiled fields")); + } + if op.inputs != inputs { + return Err(invalid("operator input contracts differ")); + } + Ok(op) + } +} diff --git a/crates/asap-physical-operators/src/operators/vector_binary.rs b/crates/asap-physical-operators/src/operators/vector_binary.rs new file mode 100644 index 00000000..cc60ff4c --- /dev/null +++ b/crates/asap-physical-operators/src/operators/vector_binary.rs @@ -0,0 +1,237 @@ +//! Label matching and scalar broadcasting are physical computation, not source binding. +use super::*; +use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; + +pub(crate) fn value_schema(scalar: bool) -> Schema { + let mut fields = Vec::new(); + if !scalar { + fields.push(result_field( + "labels", + DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }, + false, + )); + } + fields.push(result_field( + if scalar { "$promql_scalar" } else { "value" }, + DataType::Float64, + false, + )); + schema(fields) +} + +fn is_scalar(input: &Schema) -> Result { + for scalar in [true, false] { + let expected = value_schema(scalar); + if input.fields.len() == expected.fields.len() + && input + .fields + .iter() + .zip(&expected.fields) + .all(|(a, b)| a.dtype == b.dtype && !a.nullable) + { + return Ok(scalar); + } + } + Err(invalid( + "vector binary requires Float64 scalars or complete label-map vectors", + )) +} + +impl Operator { + pub fn vector_binary( + left: Schema, + right: Schema, + operator: BinaryOperator, + return_bool: bool, + ) -> Result { + let scalar = is_scalar(&left)? && is_scalar(&right)?; + is_scalar(&right)?; + let expression = Expression::Binary { + operator: operator.clone(), + left: Box::new(Expression::Column(0)), + right: Box::new(Expression::Column(1)), + }; + expression.dtype(&schema(vec![ + result_field("left", DataType::Float64, false), + result_field("right", DataType::Float64, false), + ]))?; + let comparison = matches!(operator.kind, BinaryOpKind::Compare(_)); + if (return_bool && !comparison) || (scalar && comparison && !return_bool) { + return Err(invalid("invalid scalar/vector comparison bool mode")); + } + Ok(Self { + inputs: vec![left, right], + output: value_schema(scalar), + kind: Kind::VectorBinary { + operator, + return_bool, + }, + }) + } +} + +type Labels = BTreeMap, Arc>; +fn labels(row: &[Value]) -> Result { + let Some(Value::Map(entries)) = row.first() else { + return Err(invalid("vector requires label map")); + }; + let mut result = BTreeMap::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("labels must be Utf8")); + }; + if result.insert(key.clone(), value.clone()).is_some() { + return Err(invalid("duplicate label name")); + } + } + Ok(result) +} +fn identity(mut labels: Labels) -> Labels { + labels.remove("__name__"); + labels.retain(|_, value| !value.is_empty()); + labels +} +fn value(row: &[Value]) -> Result { + match row.last() { + Some(Value::Float64(value)) => Ok(*value), + _ => Err(invalid("binary value must be Float64")), + } +} +fn label_bytes(labels: &Labels) -> usize { + labels.iter().map(|(k, v)| 64 + k.len() + v.len()).sum() +} + +pub(super) fn execute<'a>( + op: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::VectorBinary { + operator, + return_bool, + } = &op.kind + else { + unreachable!() + }; + let left_scalar = is_scalar(&op.inputs[0])?; + let right_scalar = is_scalar(&op.inputs[1])?; + let right = inputs.pop().ok_or_else(|| invalid("missing right input"))?; + let left = inputs.pop().ok_or_else(|| invalid("missing left input"))?; + Ok(futures::stream::once(async move { + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + if (left_scalar && left.len() != 1) || (right_scalar && right.len() != 1) { + return Err(invalid("scalar input must contain exactly one value")); + } + let mut workspace = Workspace::new(&context)?; + let mut work = Cooperative::new(&context); + let mut rows = Vec::new(); + let mut result_identities = std::collections::BTreeSet::new(); + let mut emit = |labels: Labels, a: f64, b: f64| -> Result<(), Error> { + let arithmetic = matches!(operator.kind, BinaryOpKind::Arithmetic(_)); + let result = match crate::expressions::arithmetic::evaluate_binary(operator, a, b)? { + Value::Float64(value) => value, + Value::Bool(value) if *return_bool => { + if value { + 1. + } else { + 0. + } + } + Value::Bool(true) => { + if left_scalar { + b + } else { + a + } + } + Value::Bool(false) => return Ok(()), + _ => return Err(invalid("invalid binary result")), + }; + let mut row = Vec::new(); + if !left_scalar || !right_scalar { + let labels = if arithmetic || *return_bool { + identity(labels) + } else { + labels + }; + workspace.grow(label_bytes(&labels) + 64)?; + if !result_identities.insert(labels.clone()) { + return Err(invalid("duplicate vector result labels")); + } + workspace.grow( + label_bytes(&labels) + + std::mem::size_of::>() + + 2 * std::mem::size_of::(), + )?; + row.push(Value::Map( + labels + .into_iter() + .map(|(k, v)| (Value::Utf8(k), Value::Utf8(v))) + .collect::>() + .into(), + )); + } else { + workspace.grow(std::mem::size_of::>() + std::mem::size_of::())?; + } + row.push(Value::Float64(result)); + rows.push(row); + Ok(()) + }; + if left_scalar || right_scalar { + let vectors = if left_scalar { &right } else { &left }; + for row in vectors { + work.checkpoint().await?; + let labels = if left_scalar && right_scalar { + Labels::new() + } else { + labels(row)? + }; + emit( + labels, + if left_scalar { + value(&left[0])? + } else { + value(row)? + }, + if right_scalar { + value(&right[0])? + } else { + value(row)? + }, + )?; + } + } else { + let mut rhs = BTreeMap::new(); + // Keep matching workspace separate from the output reservation captured by emit. + let mut matching = Workspace::new(&context)?; + for row in &right { + work.checkpoint().await?; + let key = identity(labels(row)?); + matching.grow(label_bytes(&key) + 64)?; + if rhs.insert(key, value(row)?).is_some() { + return Err(invalid("duplicate vector matching labels")); + } + } + let mut seen = std::collections::BTreeSet::new(); + for row in &left { + work.checkpoint().await?; + let labels = labels(row)?; + let key = identity(labels.clone()); + matching.grow(label_bytes(&key) + 64)?; + if !seen.insert(key.clone()) { + return Err(invalid("duplicate vector matching labels")); + } + if let Some(b) = rhs.get(&key) { + emit(labels, value(row)?, *b)?; + } + } + } + Batch::try_new(op.output.clone(), rows) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/vector_window.rs b/crates/asap-physical-operators/src/operators/vector_window.rs new file mode 100644 index 00000000..407f04ea --- /dev/null +++ b/crates/asap-physical-operators/src/operators/vector_window.rs @@ -0,0 +1,147 @@ +//! Window bounds are typed input data; aggregation and histogram semantics stay native. +use super::*; +use planner_types::pre_asap::AggIntent; + +pub(crate) fn matrix_schema() -> Schema { + let mut fields = vector_binary::value_schema(false).fields.clone(); + fields.insert(1, result_field("timestamp", DataType::Timestamp, false)); + fields.push(result_field("window_start", DataType::Timestamp, false)); + fields.push(result_field("window_end", DataType::Timestamp, false)); + Arc::new(SummarySchema { + fields, + time_index: Some(1), + }) +} + +impl Operator { + pub fn range_window(intent: AggIntent) -> Result { + // Reuse the window constructor's semantic admission, without fixing request time. + Self::window(matrix_schema(), intent.clone(), 1, 2, vec![0], Some((0, 1)))?; + Ok(Self { + kind: Kind::RangeWindow { + intent: Box::new(intent), + }, + inputs: vec![matrix_schema()], + output: vector_binary::value_schema(false), + }) + } + pub fn histogram_quantile() -> Self { + Self { + kind: Kind::HistogramQuantile, + inputs: vec![ + vector_binary::value_schema(true), + vector_binary::value_schema(false), + ], + output: vector_binary::value_schema(false), + } + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + match &operator.kind { + Kind::RangeWindow { intent } => { + let input = inputs + .pop() + .ok_or_else(|| invalid("missing matrix input"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let mut window = None; + let mut work = Cooperative::new(&context); + for row in &rows { + work.checkpoint().await?; + let (Value::Timestamp(start), Value::Timestamp(end)) = (&row[3], &row[4]) + else { + return Err(invalid("missing matrix window bounds")); + }; + if start >= end || window.is_some_and(|bounds| bounds != (*start, *end)) { + return Err(invalid("matrix rows must share one nonempty window")); + } + window = Some((*start, *end)); + } + let mut rows = + aggregate::temporal::reduce(rows, intent, &[0], 1, 2, window, &context).await?; + for row in &mut rows { + work.checkpoint().await?; + row[1] = Expression::ExactFloat64(1).evaluate(row)?; + } + Batch::try_new(operator.output.clone(), rows) + }) + .boxed_local()) + } + Kind::HistogramQuantile => { + let buckets = inputs + .pop() + .ok_or_else(|| invalid("missing histogram buckets"))?; + let quantile = inputs.pop().ok_or_else(|| invalid("missing quantile"))?; + Ok(futures::stream::once(async move { + let ((quantile, _q_memory), (buckets, _bucket_memory)) = futures::try_join!( + collect_rows(quantile, &context), + collect_rows(buckets, &context) + )?; + let [row] = quantile.as_slice() else { + return Err(invalid("histogram quantile requires one scalar")); + }; + let [Value::Float64(q)] = row.as_slice() else { + return Err(invalid("invalid quantile scalar")); + }; + let mut rows = Vec::new(); + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + for row in buckets { + work.checkpoint().await?; + let Value::Map(entries) = &row[0] else { + return Err(invalid("histogram buckets require labels")); + }; + let mut bound = None; + let mut labels = BTreeMap::new(); + let mut seen = std::collections::BTreeSet::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("histogram labels must be Utf8")); + }; + if !seen.insert(key) { + return Err(invalid("duplicate histogram label")); + } + if key.as_ref() == "le" { + bound = value.parse::().ok(); + } else if key.as_ref() != "__name__" && !value.is_empty() { + labels.insert(key.clone(), value.clone()); + } + } + if let Some(bound) = bound { + let projected = vec![ + Value::Map( + labels + .into_iter() + .map(|(k, v)| (Value::Utf8(k), Value::Utf8(v))) + .collect::>() + .into(), + ), + Value::Float64(bound), + row[1].clone(), + ]; + workspace.grow(row_bytes(&projected))?; + rows.push(projected); + } + } + let result = aggregate::temporal::reduce( + rows, + &AggIntent::HistogramQuantile { q: *q }, + &[0], + 1, + 2, + None, + &context, + ) + .await?; + Batch::try_new(operator.output.clone(), result) + }) + .boxed_local()) + } + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs new file mode 100644 index 00000000..f5658a9a --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -0,0 +1,483 @@ +//! Compile maintenance-selected frontiers without deployment-specific graph rewrites. +use super::*; + +/// One computation realization; lifecycle/window/revision requirements accompany +/// it during optimization and deployment. Stored outputs have no storage identity. +/// Deserialization validates the producer/reader boundary. +#[derive(Clone, serde::Serialize, serde::Deserialize)] +#[serde(try_from = "UncheckedCandidate")] +pub struct PhysicalCandidate { + pub precompute: Option, + pub query: CompiledPhysicalDag, + pub materialized_outputs: BTreeMap, +} + +/// Compile an explicit materialization frontier selected by Planner maintenance +/// search. Operators upstream of that frontier run in precompute, including +/// readouts/reductions; query execution receives their typed output values. +/// Empty frontiers retain the full computation in the query DAG. +/// +/// Repeated windows must be instantiated with the same evaluation/population +/// contract used to build each output. This API never treats a result from a +/// different window or revision as interchangeable merely because types match. +pub fn compile_candidate( + dag: &PostAsapDag, + inputs: BTreeMap, + roots: &[NodeId], + frontier: &[NodeId], +) -> Result { + cut_candidate(&compile(dag, inputs, roots)?, frontier) +} + +/// Derive one frontier's candidate from a complete [`compile`] result by +/// partitioning its operators; nothing is lowered again. A deployment compiles +/// each query DAG once and derives every placement choice from that result. +/// The candidate is identical to [`compile_candidate`] for the same frontier. +pub fn cut_candidate( + compiled: &CompiledPhysicalDag, + frontier: &[NodeId], +) -> Result { + if frontier.is_empty() { + return Ok(PhysicalCandidate { + precompute: None, + query: compiled.clone(), + materialized_outputs: BTreeMap::new(), + }); + } + let frontier_set: BTreeSet<_> = frontier.iter().copied().collect(); + // `compile` retains only reachable nodes and numbers its helper operators + // above the u32 Planner ID range; only Planner outputs are boundaries. + if frontier_set.len() != frontier.len() + || frontier + .iter() + .any(|&id| !compiled.is_operator(id) || u32::try_from(id).is_err()) + { + return Err(invalid("frontier must contain distinct computed outputs")); + } + let inputs: BTreeMap<_, _> = compiled + .input_contracts() + .map(|(id, contract)| (id, contract.clone())) + .collect(); + let precompute = compiled.cut(&inputs, frontier)?; + let mut materialized_outputs = BTreeMap::new(); + for &id in frontier { + let mut output = precompute.output_contract(id)?; + if output.properties.boundedness != Boundedness::Bounded { + return Err(invalid("materialized output requires bounded execution")); + } + // A stored reader may stream batches even when the producer blocked. + // Its timing is independent; the retained result still must be finite. + output.properties.emission = Emission::Unknown; + materialized_outputs.insert(id, output); + } + let mut query_inputs = inputs; + query_inputs.extend(materialized_outputs.clone()); + let query = compiled.cut(&query_inputs, compiled.roots())?; + let used: BTreeSet<_> = query.input_contracts().map(|(id, _)| id).collect(); + if !frontier.iter().all(|id| used.contains(id)) { + return Err(invalid( + "frontier contains an output shadowed by another boundary", + )); + } + Ok(PhysicalCandidate { + precompute: Some(precompute), + query, + materialized_outputs, + }) +} + +/// Materialization frontier implied by lifecycle-assigned timing: ingestion-time +/// nodes read by a query-time node, plus the root when it is ingestion-timed. +/// `cut_candidate` of one [`compile`] result with this frontier realizes the +/// assignment, so different assignments are different cuts of one lowering. +/// That holds while timing-dependent lowering (an ingestion-time `Binary` +/// aligns by value column) has the same timing at compile time as here. +/// A query-time node feeding an ingestion-time node has no valid placement. +pub fn frontier_from_timing(dag: &PostAsapDag) -> Result, Error> { + use planner_types::post_asap::ExecutionTiming::IngestionTime; + let timing = dag + .nodes + .iter() + .map(|node| (node.id, node.output_state.timing)) + .collect::>(); + let mut frontier = BTreeSet::new(); + if timing.get(&dag.root) == Some(&IngestionTime) { + frontier.insert(u64::from(dag.root.0)); + } + for edge in &dag.edges { + let (Some(&producer), Some(&consumer)) = + (timing.get(&edge.producer), timing.get(&edge.consumer)) + else { + return Err(invalid("timed DAG edge names an unknown node")); + }; + match (producer == IngestionTime, consumer == IngestionTime) { + (true, false) => { + frontier.insert(u64::from(edge.producer.0)); + } + (false, true) => return Err(invalid("query-time node feeds an ingestion-time node")), + _ => {} + } + } + Ok(frontier.into_iter().collect()) +} + +/// Enumerate bounded, reachable materialization frontiers above explicit inputs. +/// Each frontier is an antichain: storing an output and its ancestor together +/// would leave the ancestor unused by query execution. Lifecycle eligibility +/// and deployment feasibility are evaluated separately before cost selection. +/// Exceeding the search budget returns an error, never a partial inventory. +pub fn enumerate_frontiers( + dag: &PostAsapDag, + inputs: &BTreeMap, + roots: &[NodeId], + max_candidates: usize, +) -> Result>, Error> { + enumerate_compiled_frontiers(&compile(dag, inputs.clone(), roots)?, max_candidates) +} + +fn enumerate_compiled_frontiers( + compiled: &CompiledPhysicalDag, + max_candidates: usize, +) -> Result>, Error> { + if max_candidates == 0 { + return Err(invalid( + "frontier search requires a positive candidate budget", + )); + } + let mut ancestors = BTreeMap::>::new(); + let mut eligible = Vec::new(); + for (id, properties) in compiled.output_properties()? { + if !compiled.is_operator(id) + || u32::try_from(id).is_err() + || properties.boundedness != Boundedness::Bounded + { + continue; + } + let mut seen = BTreeSet::new(); + let mut pending = vec![id]; + while let Some(current) = pending.pop() { + if seen.insert(current) { + pending.extend(compiled.dependencies(current)); + } + } + ancestors.insert(id, seen); + eligible.push(id); + } + let mut frontiers = vec![vec![]]; + for id in eligible { + let additions = frontiers + .iter() + .filter(|frontier| { + frontier.iter().all(|previous| { + !ancestors[&id].contains(previous) && !ancestors[previous].contains(&id) + }) + }) + .map(|frontier| { + let mut next = frontier.clone(); + next.push(id); + next + }) + .collect::>(); + if additions.len() > max_candidates.saturating_sub(frontiers.len()) { + return Err(invalid( + "materialization frontier search exceeds candidate budget", + )); + } + frontiers.extend(additions); + } + Ok(frontiers) +} + +/// Lower every maintenance candidate before feasibility/cost evaluation. Keep +/// individual failures visible; do not substitute another computation on error. +/// The DAG is lowered once; each frontier is a [`cut_candidate`] of it. +pub fn compile_candidates( + dag: &PostAsapDag, + inputs: BTreeMap, + roots: &[NodeId], + frontiers: &[Vec], +) -> Vec> { + match compile(dag, inputs, roots) { + Ok(compiled) => frontiers + .iter() + .map(|frontier| cut_candidate(&compiled, frontier)) + .collect(), + Err(error) => frontiers.iter().map(|_| Err(error.clone())).collect(), + } +} + +/// Complete workload cost supplied by scoped optimizer/deployment evidence. +/// The evaluator includes build/update work, retained state, shared producers +/// and recurrent reads over the same horizon; these are not per-query timings. +#[derive(Clone, Debug)] +pub struct CandidateCost { + pub workload_scope: String, + pub horizon_seconds: f64, + pub total_cost: f64, +} + +pub struct CandidateSelection { + pub candidate: T, + pub candidate_index: usize, + pub cost: CandidateCost, +} + +/// Select only compiled and deployment-feasible physical candidates. `None` +/// rejects an unbindable candidate before pricing. Comparable scoped costs are +/// required; deployment never rewrites the selected frontier after this step. +/// The payload is generic so deployments can retain binding/diagnostic metadata +/// alongside each compiled computation without duplicating winner selection. +pub fn select_candidate( + candidates: Vec>, + mut evaluate: impl FnMut(&T) -> Result, Error>, +) -> Result, Error> { + let mut scope: Option<(String, f64)> = None; + let mut selected: Option> = None; + for (candidate_index, candidate) in candidates.into_iter().enumerate() { + let Ok(candidate) = candidate else { continue }; + let Some(cost) = evaluate(&candidate)? else { + continue; + }; + if cost.workload_scope.is_empty() + || !cost.horizon_seconds.is_finite() + || cost.horizon_seconds <= 0. + || !cost.total_cost.is_finite() + || cost.total_cost < 0. + { + return Err(invalid( + "candidate cost lacks a valid workload scope/horizon", + )); + } + let current_scope = (cost.workload_scope.clone(), cost.horizon_seconds); + if scope.as_ref().is_some_and(|scope| scope != ¤t_scope) { + return Err(invalid( + "candidate costs describe different workloads or horizons", + )); + } + scope = Some(current_scope); + if selected + .as_ref() + .is_none_or(|selected| cost.total_cost < selected.cost.total_cost) + { + selected = Some(CandidateSelection { + candidate, + candidate_index, + cost, + }); + } + } + selected.ok_or_else(|| invalid("no feasible priced physical candidate")) +} + +#[derive(serde::Deserialize)] +#[serde(deny_unknown_fields)] +struct UncheckedCandidate { + precompute: Option, + query: CompiledPhysicalDag, + materialized_outputs: BTreeMap, +} +impl TryFrom for PhysicalCandidate { + type Error = Error; + fn try_from(candidate: UncheckedCandidate) -> Result { + let result = Self { + precompute: candidate.precompute, + query: candidate.query, + materialized_outputs: candidate.materialized_outputs, + }; + result.validate()?; + Ok(result) + } +} + +impl PhysicalCandidate { + /// Validate the physical handoff, including the producer/reader boundary. + pub fn validate(&self) -> Result<(), Error> { + self.query.validate()?; + let Some(precompute) = &self.precompute else { + return if self.materialized_outputs.is_empty() { + Ok(()) + } else { + Err(invalid("materialized outputs have no producer DAG")) + }; + }; + precompute.validate()?; + let outputs: BTreeSet<_> = self.materialized_outputs.keys().copied().collect(); + if outputs.is_empty() || outputs != precompute.roots().iter().copied().collect() { + return Err(invalid("physical frontier differs from precompute outputs")); + } + let readers: BTreeMap<_, _> = self.query.input_contracts().collect(); + for (&id, contract) in &self.materialized_outputs { + let produced = precompute.output_contract(id)?; + // Direct frontiers retain their node IDs. Temporal candidates can + // read several window instances through distinct input slots; + // their deployment bindings must validate those slots separately. + let reader = readers.get(&id); + if contract.schema != produced.schema + || reader.is_some_and(|reader| contract.schema != reader.schema) + || produced.properties.boundedness != Boundedness::Bounded + || contract.properties.boundedness != Boundedness::Bounded + || reader + .is_some_and(|reader| reader.properties.boundedness != Boundedness::Bounded) + { + return Err(invalid("physical frontier schema or boundedness mismatch")); + } + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use planner_types::workload::*; + + fn grouped_rate() -> (PostAsapDag, BTreeMap, NodeId) { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("sum by(job)(rate(m[1m]))".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit( + planner_types::types::AccuracyTarget::Exact, + ), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let root = asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0); + let root = std::rc::Rc::new(promql_rows::with_series_identity(&root).unwrap()); + let space = asap_aware_mapping::search_workload(vec![("q", root)]); + let selected = space + .global_selection(&asap_aware_mapping::cost_model::DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + let dag = planner_types::post_asap::compile_post_asap_dag(&selected).unwrap(); + let state = dag + .nodes + .iter() + .find(|node| matches!(node.payload, Payload::SummaryAgg { .. })) + .unwrap(); + let inputs = BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + (dag.clone(), inputs, u64::from(dag.root.0)) + } + + /// Enumerating and cutting every frontier lowers each Planner node once. + #[test] + fn candidates_for_all_frontiers_share_one_lowering() { + let (dag, inputs, root) = grouped_rate(); + let lowered = || crate::physical_planner::LOWERED_NODES.with(|count| count.get()); + let before = lowered(); + let compiled = compile(&dag, inputs, &[root]).unwrap(); + let once = lowered() - before; + let frontiers = enumerate_compiled_frontiers(&compiled, 4096).unwrap(); + assert!(frontiers.len() >= 3, "{frontiers:?}"); + for frontier in &frontiers { + cut_candidate(&compiled, frontier).unwrap(); + } + assert!(once > 0); + assert_eq!(lowered() - before, once); + } + + fn with_timing( + dag: &PostAsapDag, + timing: impl Fn(&PostAsapDagNode) -> planner_types::post_asap::ExecutionTiming, + ) -> PostAsapDag { + let mut timed = dag.clone(); + for node in &mut timed.nodes { + node.output_state.timing = timing(node); + } + for edge in &mut timed.edges { + let producer = timed.nodes.iter().find(|node| node.id == edge.producer); + edge.data_state = producer.unwrap().output_state; + } + timed + } + + fn raw_input(dag: &PostAsapDag) -> BTreeMap { + let raw = dag + .nodes + .iter() + .find(|node| matches!(node.payload, Payload::Fallback { .. })) + .unwrap(); + BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(Arc::new(raw.output_schema.clone())), + )]) + } + + /// Cutting one compilation by a retained-state timing and by the all + /// query-time timing (what ContinuouslyMaintained and Ephemeral assign) + /// lowers each Planner node once and matches `compile_candidate`. + #[test] + fn timing_cuts_share_one_lowering() { + use planner_types::post_asap::ExecutionTiming::QueryTime; + let (retained, _, root) = grouped_rate(); + let ephemeral = with_timing(&retained, |_| QueryTime); + let inputs = raw_input(&retained); + let lowered = || crate::physical_planner::LOWERED_NODES.with(|count| count.get()); + let before = lowered(); + let compiled = compile(&ephemeral, inputs.clone(), &[root]).unwrap(); + let once = lowered() - before; + let cuts = [&retained, &ephemeral].map(|timed| { + let frontier = frontier_from_timing(timed).unwrap(); + let cut = cut_candidate(&compiled, &frontier).unwrap(); + (timed, frontier, cut) + }); + assert!(once > 0); + assert_eq!(lowered() - before, once); + assert_eq!(cuts[0].1.len(), 1); + assert!(cuts[1].1.is_empty()); + for (timed, frontier, cut) in cuts { + let expected = compile_candidate(timed, inputs.clone(), &[root], &frontier).unwrap(); + assert_eq!( + serde_json::to_vec(&cut).unwrap(), + serde_json::to_vec(&expected).unwrap() + ); + } + } + + /// The frontier is the ingestion-time nodes read at query time; an + /// ingestion-time root is itself the frontier. + #[test] + fn frontier_from_timing_includes_ingestion_root() { + use planner_types::post_asap::ExecutionTiming::IngestionTime; + let (dag, _, root) = grouped_rate(); + let timed = with_timing(&dag, |_| IngestionTime); + assert_eq!(frontier_from_timing(&timed).unwrap(), [root]); + } + + /// A query-time node feeding an ingestion-time node is rejected. + #[test] + fn frontier_from_timing_rejects_query_time_input_to_ingestion() { + use planner_types::post_asap::ExecutionTiming::{IngestionTime, QueryTime}; + let (dag, _, _) = grouped_rate(); + let timed = with_timing(&dag, |node| { + if node.id == dag.root { + IngestionTime + } else { + QueryTime + } + }); + assert!(frontier_from_timing(&timed).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs new file mode 100644 index 00000000..af0bbe39 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -0,0 +1,352 @@ +//! Reader-independent physical computation and checked deployment instantiation. +use super::*; + +/// A typed execution boundary, without storage identity or a live reader. +#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)] +pub struct InputContract { + pub schema: Schema, + pub properties: PlanProperties, +} +impl InputContract { + pub fn bounded(schema: Schema) -> Self { + Self { + schema, + properties: PlanProperties { + boundedness: Boundedness::Bounded, + emission: Emission::Unknown, + }, + } + } + pub fn from_source(source: &dyn PhysicalOperator) -> Self { + Self { + schema: source.output_schema(), + properties: source.properties(&[]), + } + } +} +#[derive(Clone, serde::Serialize, serde::Deserialize)] +enum Node { + Input(InputContract), + Operator { + inputs: Vec, + operator: Operator, + }, +} + +/// Selected native operators and input slots. Rebinding never repeats lowering. +/// Serde is format-agnostic; deployments choose the encoding and its versioning. +/// Deserialization validates the graph before it is usable. +#[derive(Clone, serde::Serialize, serde::Deserialize)] +#[serde(try_from = "UncheckedDag")] +pub struct CompiledPhysicalDag { + nodes: BTreeMap, + roots: Vec, +} +#[derive(serde::Deserialize)] +#[serde(deny_unknown_fields)] +struct UncheckedDag { + nodes: BTreeMap, + roots: Vec, +} +impl TryFrom for CompiledPhysicalDag { + type Error = Error; + fn try_from(dag: UncheckedDag) -> Result { + let result = Self { + nodes: dag.nodes, + roots: dag.roots, + }; + result.validate()?; + Ok(result) + } +} + +impl CompiledPhysicalDag { + /// Link already-selected physical fragments without lowering operators again. + /// Fragment keys and source keys share a namespace; repeated dependency IDs + /// therefore remain one producer in the composed graph. + pub fn compose( + sources: BTreeMap, + fragments: BTreeMap, Self)>, + roots: Vec, + ) -> Result { + if sources.keys().any(|id| fragments.contains_key(id)) { + return Err(invalid("physical source and fragment IDs overlap")); + } + let mut contracts = sources.clone(); + for (&id, (_, fragment)) in &fragments { + fragment.validate()?; + let [root] = fragment.roots() else { + return Err(invalid("composed fragment requires one root")); + }; + if fragment.input_contracts().any(|(id, _)| id == *root) { + return Err(invalid("fragment root must be a computed output")); + } + contracts.insert(id, fragment.output_contract(*root)?); + } + let mut next = contracts + .keys() + .next_back() + .copied() + .unwrap_or(0) + .checked_add(1) + .ok_or_else(|| invalid("physical node ID overflow"))?; + let mut result = Self::new(roots); + for (id, contract) in sources { + result.add_input(id, contract)?; + } + for (id, (inputs, fragment)) in fragments { + if inputs.len() != fragment.input_contracts().count() { + return Err(invalid("physical fragment input arity mismatch")); + } + let mut mapping = BTreeMap::new(); + for ((local, expected), global) in fragment.input_contracts().zip(inputs) { + let actual = contracts + .get(&global) + .ok_or_else(|| invalid("missing physical fragment dependency"))?; + if expected.schema != actual.schema + || (expected.properties.boundedness == Boundedness::Bounded + && actual.properties.boundedness != Boundedness::Bounded) + { + return Err(invalid("physical fragment dependency contract mismatch")); + } + mapping.insert(local, global); + } + mapping.insert(fragment.roots[0], id); + for local in fragment.nodes.keys() { + if !mapping.contains_key(local) { + mapping.insert(*local, next); + next = next + .checked_add(1) + .ok_or_else(|| invalid("physical node ID overflow"))?; + } + } + for (local, node) in fragment.nodes { + if let Node::Operator { inputs, operator } = node { + result.add( + mapping[&local], + inputs.into_iter().map(|input| mapping[&input]).collect(), + operator, + )?; + } + } + } + result.validate()?; + Ok(result) + } + + /// Assemble already-lowered operators and typed external inputs. This is + /// useful for engines that compose multiple compiled computation fragments. + pub fn from_operators( + inputs: BTreeMap, + operators: BTreeMap, Operator)>, + roots: Vec, + ) -> Result { + let mut result = Self::new(roots); + for (id, contract) in inputs { + result.add_input(id, contract)?; + } + for (id, (inputs, operator)) in operators { + result.add(id, inputs, operator)?; + } + result.validate()?; + Ok(result) + } + pub(super) fn new(roots: Vec) -> Self { + Self { + nodes: BTreeMap::new(), + roots, + } + } + pub(super) fn add_input(&mut self, id: NodeId, contract: InputContract) -> Result<(), Error> { + self.insert(id, Node::Input(contract)) + } + pub(super) fn add( + &mut self, + id: NodeId, + inputs: Vec, + operator: Operator, + ) -> Result<(), Error> { + self.insert(id, Node::Operator { inputs, operator }) + } + fn insert(&mut self, id: NodeId, node: Node) -> Result<(), Error> { + if self.nodes.insert(id, node).is_some() { + return Err(invalid(format!("duplicate physical node {id}"))); + } + Ok(()) + } + /// Identify the external input whose rows survive unchanged at this output. + /// Protocol adapters can retain labels that are outside a closed physical schema. + pub fn row_source(&self, id: NodeId) -> Option { + match self.nodes.get(&id)? { + Node::Input(_) => Some(id), + Node::Operator { inputs, operator } => { + let index = operator.row_preserving_input()?; + self.row_source(*inputs.get(index)?) + } + } + } + + /// Selected operator name, for plan inspection without decoding its wire format. + /// Certified candidate pruning checks authoritative-key coverage inside this operator. + pub fn certified_pruning_keys(&self, id: NodeId) -> Option<&[(usize, usize)]> { + match self.nodes.get(&id)? { + Node::Operator { operator, .. } => operator.certified_pruning_keys(), + Node::Input(_) => None, + } + } + pub fn operator_name(&self, id: NodeId) -> Option<&str> { + match self.nodes.get(&id)? { + Node::Input(_) => Some("Input"), + Node::Operator { operator, .. } => Some(operator.name()), + } + } + + pub fn roots(&self) -> &[NodeId] { + &self.roots + } + pub fn input_contracts(&self) -> impl Iterator { + self.nodes.iter().filter_map(|(&id, node)| match node { + Node::Input(contract) => Some((id, contract)), + Node::Operator { .. } => None, + }) + } + /// Derive a reachable output contract without opening deployment readers. + pub fn output_contract(&self, id: NodeId) -> Result { + let properties = *self + .output_properties()? + .get(&id) + .ok_or_else(|| invalid("output is not reachable"))?; + let schema = match self + .nodes + .get(&id) + .ok_or_else(|| invalid("missing output"))? + { + Node::Input(contract) => contract.schema.clone(), + Node::Operator { operator, .. } => operator.output_schema(), + }; + Ok(InputContract { schema, properties }) + } + /// Properties of every reachable node, derived in one contract-only pass. + pub(super) fn output_properties(&self) -> Result, Error> { + let sources = self + .input_contracts() + .map(|(id, contract)| (id, Box::new(contract.clone()) as Source<'_>)) + .collect(); + self.instantiate(sources)?.properties(&self.roots) + } + /// Direct physical dependencies; empty for inputs and unknown IDs. + pub(super) fn dependencies(&self, id: NodeId) -> &[NodeId] { + match self.nodes.get(&id) { + Some(Node::Operator { inputs, .. }) => inputs, + _ => &[], + } + } + pub(super) fn is_operator(&self, id: NodeId) -> bool { + matches!(self.nodes.get(&id), Some(Node::Operator { .. })) + } + /// Keep the already-lowered operators reachable from `roots`, replacing + /// each node in `boundaries` by a typed input. Nothing is lowered again. + pub(super) fn cut( + &self, + boundaries: &BTreeMap, + roots: &[NodeId], + ) -> Result { + let mut result = Self::new(roots.to_vec()); + let mut pending = roots.to_vec(); + while let Some(id) = pending.pop() { + if result.nodes.contains_key(&id) { + continue; + } + let node = match boundaries.get(&id) { + Some(contract) => Node::Input(contract.clone()), + None => self + .nodes + .get(&id) + .cloned() + .ok_or_else(|| invalid(format!("missing physical node {id}")))?, + }; + if let Node::Operator { inputs, .. } = &node { + pending.extend(inputs); + } + result.nodes.insert(id, node); + } + result.validate()?; + Ok(result) + } + /// Validate using contract-only sources. No deployment reader is available. + pub fn validate(&self) -> Result<(), Error> { + let sources = self + .input_contracts() + .map(|(id, c)| (id, Box::new(c.clone()) as Source<'_>)) + .collect(); + self.instantiate(sources).map(|_| ()) + } + /// Resolve exactly the declared inputs and validate before any source starts. + pub fn instantiate<'a>( + &self, + mut sources: BTreeMap>, + ) -> Result, Error> { + let mut graph = PhysicalDag::default(); + for (&id, node) in &self.nodes { + match node { + Node::Input(contract) => { + let source = sources + .remove(&id) + .ok_or_else(|| invalid(format!("missing physical input {id}")))?; + let actual = source.properties(&[]); + if !source.input_schemas().is_empty() + || source.output_schema() != contract.schema + || (contract.properties.boundedness != Boundedness::Unknown + && actual.boundedness != contract.properties.boundedness) + || (contract.properties.emission != Emission::Unknown + && actual.emission != contract.properties.emission) + { + return Err(invalid(format!( + "physical input {id} violates its compiled contract" + ))); + } + graph.add_boxed( + id, + vec![], + Box::new(CheckedSource { + source, + output: contract.schema.clone(), + }), + )?; + } + Node::Operator { inputs, operator } => { + graph.add(id, inputs.clone(), operator.clone())?; + } + } + } + if !sources.is_empty() { + return Err(invalid("unexpected physical input binding")); + } + graph.validate(&self.roots)?; + Ok(graph) + } +} +impl PhysicalOperator for InputContract { + fn name(&self) -> &str { + "UnresolvedInput" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.schema.clone() + } + fn properties(&self, _: &[PlanProperties]) -> PlanProperties { + self.properties + } + fn output_bytes(&self, batch: &Batch) -> usize { + batch.bytes() + } + fn start<'a>( + &'a self, + _: Vec>, + _: crate::runtime::RunContext, + ) -> Result, Error> { + Err(invalid("physical input must be resolved before execution")) + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs new file mode 100644 index 00000000..655c024a --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -0,0 +1,1079 @@ +//! Compile logical computation to native operators with typed external inputs. +//! Compilation needs no readers; deployment resolves inputs after selection. +use crate::operators::ReadoutQuery; +use crate::summary_kernels::exact::ExactReadout; +use crate::{ + operators::{Expression, Operator, Reduction, SortKey}, + plan::{Boundedness, Emission, NodeId, PhysicalDag, PhysicalOperator, PlanProperties}, + values::{Batch, Schema}, + Error, +}; +use planner_types::{ + post_asap::{ + ExactOperation, PostAsapDag, PostAsapDagNode, PostAsapOperatorPayload as Payload, + SketchQuery, SummaryFamilyType, SummaryInputExpr, ValueOperation, + }, + pre_asap::{ + AggIntent, ColumnRef, CompareOpKind, DataType, GroupKeys, QueryExpr, + Reduction as PlannerReduction, + }, +}; +use std::{ + collections::{BTreeMap, BTreeSet}, + sync::Arc, +}; +fn invalid(message: impl Into) -> Error { + Error::Invalid(message.into()) +} + +/// Source nodes cut the DAG at an installed storage/ingestion frontier. The +/// binding must have exactly the declared schema and no upstream dependencies. +/// A deployment must authorize these frontiers before calling this function. +pub type Source<'a> = Box + 'a>; + +pub mod precompute; +pub mod promql_fallback; +pub mod promql_rows; +pub mod promql_values; + +mod candidates; +pub use candidates::{ + compile_candidate, compile_candidates, cut_candidate, enumerate_frontiers, + frontier_from_timing, select_candidate, CandidateCost, CandidateSelection, PhysicalCandidate, +}; + +mod compiled; +pub use compiled::{CompiledPhysicalDag, InputContract}; + +mod row_values; + +/// Compile computation without opening or retaining deployment readers. +/// Input contracts identify explicit boundaries selected by maintenance planning. +pub fn compile( + dag: &PostAsapDag, + inputs: BTreeMap, + roots: &[NodeId], +) -> Result { + compile_internal(dag, inputs, roots) +} + +/// Convenience for callers that already resolved inputs. Lowering still uses +/// only their contracts, and instantiation checks those contracts again. +pub fn bind<'a>( + dag: &PostAsapDag, + sources: BTreeMap>, + roots: &[NodeId], +) -> Result, Error> { + let inputs = sources + .iter() + .map(|(&id, source)| (id, InputContract::from_source(source.as_ref()))) + .collect(); + compile(dag, inputs, roots)?.instantiate(sources) +} + +/// Resolve raw scan connectors before invoking the reader-independent compiler. +pub fn bind_with_data_sources<'a>( + dag: &PostAsapDag, + mut sources: BTreeMap>, + roots: &[NodeId], + data_sources: &crate::sources::DataSources, +) -> Result, Error> { + // Only resolve scans reachable below the selected input boundaries. + let mut pending = roots.to_vec(); + let mut seen = BTreeSet::new(); + while let Some(id) = pending.pop() { + if !seen.insert(id) || sources.contains_key(&id) { + continue; + } + let node = dag + .nodes + .iter() + .find(|n| u64::from(n.id.0) == id) + .ok_or_else(|| invalid(format!("missing node {id}")))?; + if let Payload::Fallback { + expression: expression @ QueryExpr::Scan { .. }, + } = &node.payload + { + sources.insert(id, Box::new(data_sources.bind(expression)?)); + } else { + pending.extend( + dag.edges + .iter() + .filter(|e| u64::from(e.consumer.0) == id) + .map(|e| u64::from(e.producer.0)), + ); + } + } + bind(dag, sources, roots) +} + +#[cfg(test)] +thread_local! { + /// Planner nodes lowered by this thread, for compile-once tests. + static LOWERED_NODES: std::cell::Cell = const { std::cell::Cell::new(0) }; +} + +/// Helper operators are numbered from their Planner node alone, above the u32 +/// Planner ID range, so every boundary choice yields a subgraph of the same +/// lowering and candidate cuts need not renumber operators. A node lowering to +/// several helpers takes consecutive indices below its base. +fn helper_id(node: NodeId, index: u64) -> NodeId { + debug_assert!(node <= u64::from(u32::MAX) && index < 1 << 16); + u64::MAX - (node << 16) - index +} + +fn compile_internal( + dag: &PostAsapDag, + mut sources: BTreeMap, + roots: &[NodeId], +) -> Result { + preflight_depth(dag)?; + dag.validate().map_err(|e| invalid(e.to_string()))?; + let nodes = dag + .nodes + .iter() + .map(|node| (u64::from(node.id.0), node)) + .collect::>(); + let mut dependencies = BTreeMap::>::new(); + // Binary input order is semantic; serialized edge order is not. + let mut edges = dag.edges.iter().collect::>(); + edges.sort_by_key(|edge| { + ( + edge.consumer.0, + match edge.role { + planner_types::post_asap::EdgeRole::Left => 0, + planner_types::post_asap::EdgeRole::Input => 1, + planner_types::post_asap::EdgeRole::Right => 2, + }, + ) + }); + // Scalar literal operands of query-time arithmetic are folded into the consumer. + let mut literals = BTreeMap::::new(); + for edge in edges { + let consumer = u64::from(edge.consumer.0); + if let ( + Payload::Fallback { expression }, + Some(PostAsapDagNode { + payload: Payload::Binary { .. }, + .. + }), + ) = ( + &nodes[&u64::from(edge.producer.0)].payload, + nodes.get(&consumer), + ) { + if let Some(value) = row_values::scalar_literal(expression) { + let left = edge.role == planner_types::post_asap::EdgeRole::Left; + if literals.insert(consumer, (value, left)).is_some() { + return Err(invalid("binary with two scalar literals is not folded")); + } + continue; + } + } + dependencies + .entry(u64::from(edge.consumer.0)) + .or_default() + .push(u64::from(edge.producer.0)); + } + let known = |id: &NodeId| { + nodes.contains_key(id) + || promql_fallback::raw_series_owner(*id).is_some_and(|owner| { + matches!( + nodes.get(&owner), + Some(PostAsapDagNode { + payload: Payload::Fallback { .. }, + .. + }) + ) + }) + }; + if !sources.keys().all(known) { + return Err(invalid("source binding names an unknown node")); + } + let mut ordered = Vec::new(); + let mut seen = BTreeSet::new(); + let mut pending = roots.iter().map(|&id| (id, false)).collect::>(); + while let Some((id, expanded)) = pending.pop() { + if expanded { + ordered.push(id); + continue; + } + if !seen.insert(id) { + continue; + } + if !nodes.contains_key(&id) { + return Err(invalid(format!("missing root {id}"))); + } + pending.push((id, true)); + if !sources.contains_key(&id) { + for &input in dependencies.get(&id).into_iter().flatten() { + pending.push((input, false)); + } + } + } + let mut graph = CompiledPhysicalDag::new(roots.to_vec()); + for id in ordered { + let node = nodes[&id]; + let mut auxiliary = helper_id(id, 0); + let output = Arc::new(node.output_schema.clone()); + crate::values::validate_schema(&output)?; + if let Some(source) = sources.remove(&id) { + if source.schema != output { + return Err(invalid("frontier does not have the declared schema")); + } + graph.add_input(id, source)?; + } else { + #[cfg(test)] + LOWERED_NODES.with(|count| count.set(count.get() + 1)); + let mut inputs = dependencies.get(&id).cloned().unwrap_or_default(); + let mut schemas = inputs + .iter() + .map(|id| Arc::new(nodes[id].output_schema.clone())) + .collect::>(); + if matches!(node.payload, Payload::SummaryMerge) && inputs.len() > 1 { + if schemas.iter().any(|s| s != &schemas[0]) { + return Err(invalid("summary merge inputs have different schemas")); + } + graph.add( + auxiliary, + inputs, + Operator::union(schemas[0].clone(), schemas.len())?, + )?; + inputs = vec![auxiliary]; + schemas.truncate(1); + } + // A consumed bare selector supplies raw range rows (e.g. to a + // per-entity summary), not an instant vector, so only its consumer computes. + let raw_rows = matches!( + &node.payload, + Payload::Fallback { + expression: QueryExpr::TimeRange { .. } + } + ) && dag.edges.iter().any(|e| u64::from(e.producer.0) == id); + if let (Payload::Fallback { expression }, false) = (&node.payload, raw_rows) { + let promql_fallback::Lowering { + selectors, + mut steps, + } = promql_fallback::lower(expression) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + let mut slots = Vec::new(); + for (i, (_, schema)) in selectors.iter().enumerate() { + let slot = promql_fallback::raw_series_input(id, i); + match sources.remove(&slot) { + Some(contract) if &contract.schema == schema => { + graph.add_input(slot, contract)? + } + Some(_) => { + return Err(invalid(format!( + "node {id}: raw series input {slot} differs from the selector schema" + ))) + } + None => { + return Err(invalid(format!( + "node {id}: PromQL fallback requires raw series input {slot}" + ))) + } + } + slots.push(slot); + } + let (last, last_inputs) = steps + .pop() + .ok_or_else(|| invalid("empty PromQL lowering"))?; + let mut ids = Vec::new(); + let resolve = |inputs: Vec, ids: &[NodeId]| { + inputs + .into_iter() + .map(|input| match input { + promql_fallback::Input::Raw(i) => slots[i], + promql_fallback::Input::Step(i) => ids[i], + }) + .collect::>() + }; + for (operator, inputs) in steps { + graph.add(auxiliary, resolve(inputs, &ids), operator)?; + ids.push(auxiliary); + auxiliary -= 1; + } + graph.add( + id, + resolve(last_inputs, &ids), + last.with_output_schema(output)?, + )?; + continue; + } + if let Payload::Value { + operation: ValueOperation::MaintainPopulation { population }, + } = &node.payload + { + use planner_types::post_asap::maintained_population::PopulationInput; + let PopulationInput::CurrentSeries(spec) = &population.input else { + return Err(invalid( + "native maintained population requires a current-series input", + )); + }; + let [input] = schemas.as_slice() else { + return Err(invalid("current-series population requires one input")); + }; + if spec.without { + return Err(invalid( + "dynamic without grouping requires label-set projection", + )); + } + let identity = named_column( + input, + &ColumnRef::Named(promql_rows::SERIES_IDENTITY_COLUMN.into()), + )?; + let coordinate = input + .time_index + .ok_or_else(|| invalid("current-series input lacks timestamp"))?; + let value = named_column(input, &ColumnRef::SampleValue)?; + let lookback = i64::try_from(spec.lookback_ms) + .map_err(|_| invalid("current-series lookback overflows"))?; + graph.add( + id, + inputs, + Operator::current_series(input.clone(), identity, coordinate, value, lookback)? + .with_output_schema(output)?, + )?; + continue; + } + if let Payload::Value { + operation: ValueOperation::ReadPopulation { readout }, + } = &node.payload + { + use planner_types::post_asap::maintained_population::{ + PopulationInput, PopulationReadout, + }; + let [producer] = inputs.as_slice() else { + return Err(invalid("population readout requires one input")); + }; + let Payload::Value { + operation: ValueOperation::MaintainPopulation { population }, + } = &nodes[producer].payload + else { + return Err(invalid( + "population readout requires its declared population", + )); + }; + let PopulationInput::CurrentSeries(spec) = &population.input else { + return Err(invalid("current-series population required")); + }; + if spec.without { + return Err(invalid( + "dynamic without ranking requires label-set projection", + )); + } + let input = schemas[0].clone(); + let PopulationReadout::TopK { k } = readout else { + let mut chain = + row_values::population_aggregate(&input, &spec.grouping, readout)?; + let last = chain.pop().expect("nonempty chain"); + let mut inputs = inputs; + for operator in chain { + graph.add(auxiliary, inputs, operator)?; + inputs = vec![auxiliary]; + auxiliary -= 1; + } + graph.add(id, inputs, last.with_output_schema(output)?)?; + continue; + }; + let groups = spec + .grouping + .iter() + .map(|name| named_column(&input, &ColumnRef::Named(name.clone()))) + .collect::, _>>()?; + let value = named_column(&input, &ColumnRef::SampleValue)?; + graph.add( + auxiliary, + inputs, + Operator::sort( + input.clone(), + vec![SortKey { + column: value, + descending: true, + nulls_first: false, + }], + groups.clone(), + )?, + )?; + graph.add( + id, + vec![auxiliary], + Operator::limit(input, *k as u64, 0, groups)?.with_output_schema(output)?, + )?; + continue; + } + // A closed row must include either all source labels or the explicit + // complete-label identity. Projected labels alone are insufficient. + if let Payload::SummaryAgg { + family, + input: update, + reduction: PlannerReduction::PerEntity, + grouping, + } = &node.payload + { + let [input_id] = inputs.as_slice() else { + return Err(invalid("per-entity summary requires one input")); + }; + let Payload::Fallback { + expression: QueryExpr::TimeRange { child, .. }, + } = &nodes[input_id].payload + else { + return Err(invalid( + "per-entity summary requires a resolved raw time range", + )); + }; + let QueryExpr::Scan { schema, .. } = child.as_ref() else { + return Err(invalid("per-entity summary requires a resolved source")); + }; + if !schema.closed || update.item.is_some() { + return Err(invalid( + "per-entity summary requires complete source identity", + )); + } + crate::capability::validate_summary_kernel(family, update, grouping) + .map_err(Error::Invalid)?; + let SummaryInputExpr::Column(value) = &update.weight else { + return Err(invalid( + "per-entity update requires a projected value column", + )); + }; + let input = schemas[0].clone(); + let value = named_column(&input, value)?; + let coordinate = input + .time_index + .ok_or_else(|| invalid("temporal input lacks time"))?; + let groups = (0..input.fields.len()) + .filter(|&column| column != value && column != coordinate) + .collect(); + let build = Operator::summary_build( + input, + family.clone(), + value, + Some(coordinate), + groups, + )?; + let compact = build.schema(); + graph.add(auxiliary, inputs, build)?; + graph.add( + id, + vec![auxiliary], + Operator::scope_timestamp(compact, output)?, + )?; + continue; + } + if let Payload::Binary { operator } = &node.payload { + let query_time = node.output_state.timing + == planner_types::post_asap::ExecutionTiming::QueryTime; + if let Some(&(value, left)) = literals.get(&id) { + let [input] = schemas.as_slice() else { + return Err(invalid("scalar binary requires one row input")); + }; + if !query_time { + return Err(invalid("scalar literal binary must run at query time")); + } + let project = row_values::scalar_binary(input, operator, value, left) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + graph.add(id, inputs, project.with_output_schema(output)?)?; + continue; + } + let label_map = |schema: &Schema| { + schema + .fields + .iter() + .any(|f| matches!(f.dtype, SummaryFamilyType::Plain(DataType::Map { .. }))) + }; + if let (true, [left, right]) = (query_time, schemas.as_slice()) { + if !label_map(left) && !label_map(right) { + let (join, project) = row_values::grouped_binary(left, right, operator) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + graph.add(auxiliary, inputs, join)?; + graph.add(id, vec![auxiliary], project.with_output_schema(output)?)?; + auxiliary -= 1; + continue; + } + } + } + if let Payload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + } = &node.payload + { + // Exact counts read out as Int64; PromQL declares a Float64 sample. + let readout = bind_operation(node, &schemas) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + let actual = readout.schema(); + let converted = actual.fields.iter().zip(&output.fields).position(|(a, d)| { + a.dtype == SummaryFamilyType::Plain(DataType::Int64) + && d.dtype == SummaryFamilyType::Plain(DataType::Float64) + }); + if let Some(column) = converted { + let columns = actual + .fields + .iter() + .enumerate() + .map(|(i, field)| { + ( + field.name.clone(), + if i == column { + Expression::ExactFloat64(i) + } else { + Expression::Column(i) + }, + ) + }) + .collect(); + let project = Operator::project(actual, columns)?.with_output_schema(output)?; + graph.add(auxiliary, inputs, readout)?; + graph.add(id, vec![auxiliary], project)?; + auxiliary -= 1; + continue; + } + } + let mut operator = compile_node(node, &schemas) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + if operator.is_counter_readout() { + let mut pending = vec![id]; + let mut visited = BTreeSet::new(); + let mut ranges = BTreeSet::new(); + while let Some(ancestor) = pending.pop() { + if !visited.insert(ancestor) { + continue; + } + if let Payload::Fallback { + expression: QueryExpr::TimeRange { range, .. }, + } = &nodes[&ancestor].payload + { + ranges.insert( + i64::try_from(range.as_millis()) + .map_err(|_| invalid("counter lookback exceeds Int64"))?, + ); + continue; + } + pending.extend(dependencies.get(&ancestor).into_iter().flatten().copied()); + } + if ranges.len() > 1 { + return Err(invalid("counter readout has ambiguous logical windows")); + } + if let Some(lookback) = ranges.into_iter().next() { + operator = operator.with_counter_lookback(lookback)?; + } + } + graph.add(id, inputs, operator)?; + } + } + graph.validate()?; + Ok(graph) +} + +/// Bind a Planner node against the schemas supplied by its deployment edges. +/// This is the same checked path used by complete DAG binding. +pub fn compile_node(node: &PostAsapDagNode, inputs: &[Schema]) -> Result { + for schema in inputs { + crate::values::validate_schema(schema)?; + } + bind_operation(node, inputs)?.with_output_schema(Arc::new(node.output_schema.clone())) +} + +fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result { + if let Payload::Binary { operator } = &node.payload { + let [left, right] = inputs else { + return Err(invalid("binary requires two inputs")); + }; + if node.output_state.timing == planner_types::post_asap::ExecutionTiming::IngestionTime { + let value = |schema: &Schema| -> Result { + let columns = schema + .fields + .iter() + .enumerate() + .filter(|(_, field)| { + field.dtype + == SummaryFamilyType::Plain(planner_types::pre_asap::DataType::Float64) + }) + .map(|(i, _)| i) + .collect::>(); + match columns.as_slice() { + [value] => Ok(*value), + _ => Err(invalid("aligned binary requires one value column")), + } + }; + let (l, r) = (value(left)?, value(right)?); + let keys = left + .fields + .iter() + .enumerate() + .filter(|(i, _)| *i != l) + .map(|(i, field)| { + right + .fields + .iter() + .position(|other| other.name == field.name && other.dtype == field.dtype) + .map(|j| (i, j)) + .ok_or_else(|| invalid("aligned input identities differ")) + }) + .collect::, _>>()?; + return Operator::aligned_binary( + left.clone(), + right.clone(), + keys, + (l, r), + operator.clone(), + ); + } + return Operator::vector_binary(left.clone(), right.clone(), operator.clone(), false); + } + if let Payload::RelationalJoin { + join_kind, + pred, + pruning, + } = &node.payload + { + use planner_types::{post_asap::CandidateCompleteness, pre_asap::JoinKind}; + if pruning.is_some() && *join_kind != JoinKind::Semi { + return Err(invalid("pruning certificate requires a semi-join")); + } + if matches!(pruning,Some(CandidateCompleteness::Certified { guarantee }) if guarantee.has_unknown() || guarantee.metric != planner_types::post_asap::ErrorMetric::TopKMembership) + { + return Err(invalid("invalid pruning certificate")); + } + let [left, right] = inputs else { + return Err(invalid("join requires two inputs")); + }; + if *join_kind == JoinKind::Semi { + if let Ok(keys) = equijoin_keys(pred, left, right) { + let operator = Operator::semi_join(left.clone(), right.clone(), keys)?; + return Ok( + if matches!(pruning, Some(CandidateCompleteness::Certified { .. })) { + operator.require_complete_right() + } else { + operator + }, + ); + } + } + if matches!(pruning, Some(CandidateCompleteness::Certified { .. })) { + return Err(invalid("certified pruning requires explicit equijoin keys")); + } + return Operator::relational_join( + left.clone(), + right.clone(), + join_kind.clone(), + pred, + Arc::new(node.output_schema.clone()), + ); + } + let [input] = inputs else { + return Err(invalid( + "native Planner binding currently requires a unary operation or an explicit source", + )); + }; + match &node.payload { + Payload::Value { operation, .. } => match operation { + ValueOperation::Project { cols, .. } => Operator::project( + input.clone(), + cols.iter() + .enumerate() + .map(|(i, col)| { + Ok(( + node.output_schema + .fields + .get(i) + .ok_or_else(|| invalid("projection width mismatch"))? + .name + .clone(), + match &col.expr { + QueryExpr::Column(index) => Expression::Column(*index), + expr => expression(expr, input)?, + }, + )) + }) + .collect::>()?, + ), + ValueOperation::Filter { pred } => { + Operator::filter(input.clone(), expression(&pred.0, input)?) + } + ValueOperation::Sort { keys, partition_by } => Operator::sort( + input.clone(), + keys.iter() + .map(|key| { + let QueryExpr::Column(column) = key.expr else { + return Err(invalid( + "sort expression must be projected before sorting", + )); + }; + Ok(SortKey { + column, + descending: !key.ascending, + nulls_first: key.nulls_first, + }) + }) + .collect::>()?, + groups(input, partition_by)?, + ), + ValueOperation::Limit { + n, + offset, + partition_by, + } => Operator::limit( + input.clone(), + *n as u64, + *offset as u64, + groups(input, partition_by)?, + ), + ValueOperation::Exact(ExactOperation::Aggregate { + reduction, + measures, + output_names, + having: None, + }) => { + if measures.len() != output_names.len() { + return Err(invalid("aggregate output names differ from measures")); + } + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid( + "per-entity aggregate requires an explicit entity binding", + )); + }; + let measures = measures + .iter() + .zip(output_names) + .map(|(m, name)| { + let column = |col: Option| { + col.map(Ok) + .unwrap_or_else(|| named_column(input, &ColumnRef::SampleValue)) + }; + let m = match m { + AggIntent::Count { .. } => Reduction::Count, + AggIntent::Sum { col } => Reduction::Sum(column(*col)?), + AggIntent::Avg { col } => Reduction::Avg(column(*col)?), + AggIntent::Min { col } => Reduction::Min(column(*col)?), + AggIntent::Max { col } => Reduction::Max(column(*col)?), + _ => { + return Err(invalid( + "aggregate intent has no native implementation", + )) + } + }; + Ok((name.clone(), m)) + }) + .collect::>()?; + Operator::aggregate(input.clone(), groups(input, keys)?, measures) + } + ValueOperation::FinalizeExactAccumulator => { + let state = summary_column(input)?; + use crate::Statistic as S; + use planner_types::post_asap::ExactKind as E; + let statistic = match &input.fields[state].dtype { + SummaryFamilyType::ExactAggregate(kind, _) => match kind { + E::Sum => S::Sum, + E::Count => S::Count, + E::Min => S::Min, + E::Max => S::Max, + E::Rate => S::Rate, + E::Increase => S::Increase, + _ => return Err(invalid("exact family readout is unsupported")), + }, + _ => return Err(invalid("exact finalization requires exact state")), + }; + Operator::readout( + input.clone(), + state, + ReadoutQuery::Exact(ExactReadout { + statistic, + lookback_ms: None, + }), + ) + } + _ => Err(invalid("value operation has no native implementation")), + }, + Payload::SummaryAgg { + family, + input: update, + reduction, + grouping, + } => { + if let Some(item) = &update.item { + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid("keyed summary requires explicit partitions")); + }; + let SummaryInputExpr::Column(weight) = &update.weight else { + return Err(invalid( + "keyed summary weight must be a finalized value column", + )); + }; + if matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::CmsWithHeap) + && !matches!( + update.weight_domain, + planner_types::post_asap::WeightDomain::NonNegative { .. } + ) + { + return Err(invalid("CMS requires a nonnegative weight contract")); + } + fn columns( + expr: &SummaryInputExpr, + input: &Schema, + result: &mut Vec, + ) -> Result<(), Error> { + match expr { + SummaryInputExpr::Column(column) => { + result.push(named_column(input, column)?) + } + SummaryInputExpr::Tuple(items) => { + for item in items { + columns(item, input, result)?; + } + } + _ => return Err(invalid("keyed summary needs explicit item columns")), + } + Ok(()) + } + let mut items = Vec::new(); + columns(item, input, &mut items)?; + return Operator::keyed_summary_build( + input.clone(), + family.clone(), + named_column(input, weight)?, + items, + groups(input, keys)?, + ); + } + crate::capability::validate_summary_kernel(family, update, grouping) + .map_err(Error::Invalid)?; + let SummaryInputExpr::Column(column) = &update.weight else { + return Err(invalid( + "summary update expression must be projected to a column", + )); + }; + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid( + "summary construction requires explicit grouping columns", + )); + }; + Operator::summary_build( + input.clone(), + family.clone(), + named_column(input, column)?, + input.time_index, + groups(input, keys)?, + ) + } + Payload::SummaryMerge => { + let state = summary_column(input)?; + Operator::summary_merge( + input.clone(), + state, + (0..input.fields.len()) + .filter(|&i| i != state && Some(i) != input.time_index) + .collect(), + ) + } + Payload::SummaryEstimate { query } => { + if let SketchQuery::TopK { k } = query { + return Operator::keyed_readout( + input.clone(), + summary_column(input)?, + *k, + Arc::new(node.output_schema.clone()), + ); + } + Operator::readout( + input.clone(), + summary_column(input)?, + ReadoutQuery::Sketch(query.clone()), + ) + } + _ => Err(invalid( + "physical operation has no native binding; no fallback is installed", + )), + } +} +fn summary_column(input: &Schema) -> Result { + let columns = input + .fields + .iter() + .enumerate() + .filter(|(_, f)| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + .map(|(i, _)| i) + .collect::>(); + match columns.as_slice() { + [column] => Ok(*column), + _ => Err(invalid("one summary state column required")), + } +} +fn named_column(input: &Schema, column: &ColumnRef) -> Result { + let name = match column { + // Executable SummarySchema retains column names, not table qualifiers. + // Frontend binding has resolved the qualifier; still reject ambiguous + // names here rather than guessing a join side. + ColumnRef::Named(name) | ColumnRef::Qualified { name, .. } => name.as_str(), + ColumnRef::SampleValue => "value", + _ => { + return Err(invalid( + "summary update requires an unambiguous bound column", + )) + } + }; + let matches = input + .fields + .iter() + .enumerate() + .filter(|(_, field)| field.name == name) + .map(|(i, _)| i) + .collect::>(); + match matches.as_slice() { + [column] => Ok(*column), + _ => Err(invalid("summary update column missing or ambiguous")), + } +} +fn groups(input: &Schema, groups: &GroupKeys) -> Result, Error> { + if groups.is_without() { + return Err(invalid("grouping without requires resolved label columns")); + } + if groups.keys().iter().any(|&i| i >= input.fields.len()) { + return Err(invalid("grouping column out of range")); + } + Ok(groups.keys().to_vec()) +} +fn expression(expr: &QueryExpr, input: &Schema) -> Result { + Ok(Expression::planner( + crate::expressions::CompiledExpression::compile(expr, input)?, + )) +} + +struct CheckedSource<'a> { + source: Source<'a>, + output: Schema, +} +impl PhysicalOperator for CheckedSource<'_> { + fn properties(&self, inputs: &[crate::plan::PlanProperties]) -> crate::plan::PlanProperties { + self.source.properties(inputs) + } + + fn name(&self) -> &str { + self.source.name() + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, batch: &Batch) -> usize { + self.source.output_bytes(batch) + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: crate::runtime::RunContext, + ) -> Result, Error> { + use futures::StreamExt; + Ok(self + .source + .start(inputs, context)? + .map(|batch| { + let batch = batch?; + if batch.schema() != &self.output { + return Err(invalid("source batch differs from its bound schema")); + } + Ok(batch) + }) + .boxed_local()) + } +} + +// Bound recursion before invoking the upstream recursive provenance validator. +fn preflight_depth(dag: &PostAsapDag) -> Result<(), Error> { + let mut remaining = dag + .nodes + .iter() + .map(|node| (node.id, 0usize)) + .collect::>(); + if remaining.len() != dag.nodes.len() { + return Err(invalid("duplicate Planner node")); + } + let mut consumers = BTreeMap::<_, Vec<_>>::new(); + for edge in &dag.edges { + if !remaining.contains_key(&edge.producer) { + return Err(invalid("missing Planner edge producer")); + } + *remaining + .get_mut(&edge.consumer) + .ok_or_else(|| invalid("missing Planner edge consumer"))? += 1; + consumers + .entry(edge.producer) + .or_default() + .push(edge.consumer); + } + let mut ready = remaining + .iter() + .filter(|(_, n)| **n == 0) + .map(|(id, _)| *id) + .collect::>(); + let mut depths = BTreeMap::new(); + let mut visited = 0; + while let Some(id) = ready.pop_front() { + visited += 1; + let depth = *depths.get(&id).unwrap_or(&1usize); + if depth > 128 { + return Err(invalid("DAG exceeds the supported execution depth of 128")); + } + for &consumer in consumers.get(&id).into_iter().flatten() { + let next = depths.entry(consumer).or_insert(1); + *next = (*next).max(depth + 1); + let count = remaining.get_mut(&consumer).expect("validated endpoint"); + *count -= 1; + if *count == 0 { + ready.push_back(consumer); + } + } + } + if visited != dag.nodes.len() { + return Err(invalid("Planner DAG contains a cycle")); + } + Ok(()) +} + +/// Join predicates address the concatenated left/right schema. +fn semi_join_keys( + expr: &QueryExpr, + left: usize, + right: usize, + keys: &mut Vec<(usize, usize)>, +) -> Result<(), Error> { + match expr { + QueryExpr::BoolAnd(parts) => { + for part in parts { + semi_join_keys(part, left, right, keys)?; + } + } + QueryExpr::Compare { + left: a, + op: CompareOpKind::Eq, + right: b, + } => { + let (QueryExpr::Column(a), QueryExpr::Column(b)) = (a.as_ref(), b.as_ref()) else { + return Err(invalid("semi-join requires column equality keys")); + }; + let (a, b) = if a < b { (*a, *b) } else { (*b, *a) }; + if a >= left || b < left || b >= left + right { + return Err(invalid("semi-join key must match left to right")); + } + keys.push((a, b - left)); + } + _ => return Err(invalid("unsupported semi-join predicate")), + } + Ok(()) +} + +/// Resolve equality keys against the Planner join's concatenated input schema. +/// Deployments may use these positions to bind their source columns. +pub fn equijoin_keys( + pred: &planner_types::pre_asap::Predicate, + left: &planner_types::post_asap::SummarySchema, + right: &planner_types::post_asap::SummarySchema, +) -> Result, Error> { + let mut keys = Vec::new(); + semi_join_keys(&pred.0, left.fields.len(), right.fields.len(), &mut keys)?; + if keys.is_empty() { + return Err(invalid("semi-join requires explicit matching keys")); + } + Ok(keys) +} diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs new file mode 100644 index 00000000..4fa40c68 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -0,0 +1,609 @@ +//! Compile immutable summary-input computation with explicit population and pane identity. +use super::promql_rows::SERIES_IDENTITY_COLUMN as SERIES_IDENTITY; +use super::*; +use planner_types::{ + post_asap::{ExecutionTiming, GroupingStrategy, SummarySchema}, + pre_asap::DataType, +}; + +/// Physical rows carry the population and pane coordinate alongside the logical value. +/// These fields preserve identities which are implicit in a stored summary instance. +pub fn population_schema(family: SummaryFamilyType) -> Schema { + Arc::new(SummarySchema { + fields: vec![ + planner_types::post_asap::SummaryField { + name: "$population".into(), + dtype: SummaryFamilyType::Plain(DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }), + nullable: false, + }, + planner_types::post_asap::SummaryField { + name: "$window_end".into(), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), + nullable: false, + }, + planner_types::post_asap::SummaryField { + name: "value".into(), + dtype: family, + nullable: false, + }, + ], + time_index: Some(1), + }) +} + +/// Raw sample rows at a precompute boundary. `$population` holds the series' +/// complete label set, so it is the complete source identity of per-series +/// summaries; `$timestamp` is the sample time and `value` a finite sample +/// (stale markers are not samples). Rows are what the boundary's source scan +/// selected; the deployment decides which rows and panes they are. Label sets +/// must be canonical (sorted, unique, no empty values), since they are the +/// population identity: build rows with [`raw_sample_row`]. +pub fn raw_sample_schema() -> Schema { + let mut schema = (*population_schema(SummaryFamilyType::Plain(DataType::Float64))).clone(); + schema.fields[1].name = "$timestamp".into(); + Arc::new(schema) +} + +/// A raw sample row whose label set is sorted, unique and omits empty values, +/// so one series always has one population identity. +pub fn raw_sample_row( + labels: &BTreeMap, + timestamp_ms: i64, + value: f64, +) -> Vec { + use crate::values::Value; + vec![ + Value::Map( + labels + .iter() + .filter(|(_, v)| !v.is_empty()) + .map(|(k, v)| { + ( + Value::Utf8(k.as_str().into()), + Value::Utf8(v.as_str().into()), + ) + }) + .collect::>() + .into(), + ), + Value::Timestamp(timestamp_ms), + Value::Float64(value), + ] +} + +/// Input contract of a precompute boundary: raw sample rows for a raw time +/// series scan, otherwise the stored population of its summary state. +pub fn boundary_schema(node: &PostAsapDagNode) -> Result { + let Payload::Fallback { expression } = &node.payload else { + return source_schema(&node.output_schema); + }; + let scan = match expression { + planner_types::pre_asap::QueryExpr::TimeRange { child, .. } => child.as_ref(), + expression => expression, + }; + if !matches!( + scan, + planner_types::pre_asap::QueryExpr::Scan { + source: planner_types::pre_asap::Source::TimeSeries { .. }, + .. + } + ) { + return source_schema(&node.output_schema); + } + let logical = &node.output_schema; + // Labels may be absent from a series; its label map then omits them. + let valid = logical + .fields + .iter() + .enumerate() + .all(|(i, field)| match &field.dtype { + SummaryFamilyType::Plain(DataType::Timestamp) => { + Some(i) == logical.time_index && !field.nullable + } + SummaryFamilyType::Plain(DataType::Float64) => field.name == "value" && !field.nullable, + SummaryFamilyType::Plain(DataType::Utf8) => true, + _ => false, + }) + && !logical + .fields + .iter() + .any(|f| f.name.starts_with('$') && f.name != SERIES_IDENTITY) + && logical.time_index.is_some() + && logical.fields.iter().filter(|f| f.name == "value").count() == 1; + if !valid { + return Err(invalid( + "raw sample boundary requires labels, a timestamp and one Float64 value", + )); + } + Ok(raw_sample_schema()) +} + +/// Validate the adapter layout during installed-plan recovery without lowering operators. +pub fn source_schema(logical: &SummarySchema) -> Result { + let states = logical + .fields + .iter() + .filter(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + .collect::>(); + let [state] = states.as_slice() else { + return Err(invalid( + "stored population requires one typed summary state", + )); + }; + if logical.fields.iter().enumerate().any(|(i, field)| matches!(&field.dtype, SummaryFamilyType::Plain(dtype) + if field.nullable || !matches!(dtype, DataType::Utf8) && !(Some(i) == logical.time_index && *dtype == DataType::Timestamp))) { + return Err(invalid("stored population metadata cannot reconstruct extra value columns")); + } + if state.nullable { + return Err(invalid("stored population state cannot be null")); + } + Ok(population_schema(state.dtype.clone())) +} + +pub fn is_population_schema(schema: &Schema) -> bool { + schema + .fields + .get(2) + .is_some_and(|field| *schema == population_schema(field.dtype.clone())) +} + +/// Compile a complete selected precompute sub-DAG. Inputs are already-computed +/// state boundaries; the deployment supplies groups, panes and states, never operations. +pub fn compile( + dag: &PostAsapDag, + frontiers: &[NodeId], + roots: &[NodeId], +) -> Result { + preflight_depth(dag)?; + dag.validate().map_err(|e| invalid(e.to_string()))?; + let nodes = dag + .nodes + .iter() + .map(|n| (u64::from(n.id.0), n)) + .collect::>(); + let frontier = frontiers.iter().copied().collect::>(); + if frontier.len() != frontiers.len() || roots.iter().any(|r| frontier.contains(r)) { + return Err(invalid( + "precompute boundaries must be distinct from outputs", + )); + } + let mut dependencies = BTreeMap::>::new(); + let mut edges = dag.edges.iter().collect::>(); + edges.sort_by_key(|edge| { + ( + edge.consumer.0, + match edge.role { + planner_types::post_asap::EdgeRole::Left => 0, + planner_types::post_asap::EdgeRole::Input => 1, + planner_types::post_asap::EdgeRole::Right => 2, + }, + ) + }); + for edge in edges { + dependencies + .entry(u64::from(edge.consumer.0)) + .or_default() + .push(u64::from(edge.producer.0)); + } + let mut ordered = Vec::new(); + let mut seen = BTreeSet::new(); + let mut pending = roots.iter().map(|&id| (id, false)).collect::>(); + while let Some((id, expanded)) = pending.pop() { + if expanded { + ordered.push(id); + continue; + } + if !seen.insert(id) { + continue; + } + if !nodes.contains_key(&id) { + return Err(invalid("missing precompute node")); + } + pending.push((id, true)); + if !frontier.contains(&id) { + pending.extend( + dependencies + .get(&id) + .into_iter() + .flatten() + .map(|id| (*id, false)), + ); + } + } + let mut sources = BTreeMap::new(); + let mut fragments = BTreeMap::new(); + let mut outputs = BTreeMap::::new(); + for id in ordered { + let node = nodes[&id]; + if frontier.contains(&id) { + let schema = boundary_schema(node)?; + sources.insert(id, InputContract::bounded(schema.clone())); + outputs.insert(id, schema); + continue; + } + if node.output_state.timing != ExecutionTiming::IngestionTime { + return Err(invalid("precompute graph contains a query-time operation")); + } + let inputs = dependencies.get(&id).cloned().unwrap_or_default(); + let schemas = inputs + .iter() + .map(|id| { + outputs + .get(id) + .cloned() + .ok_or_else(|| invalid("missing precompute input")) + }) + .collect::, _>>()?; + let graph = fragment( + node, + &schemas, + &inputs.iter().map(|id| nodes[id]).collect::>(), + )?; + outputs.insert(id, graph.output_contract(graph.roots()[0])?.schema); + fragments.insert(id, (inputs, graph)); + } + CompiledPhysicalDag::compose(sources, fragments, roots.to_vec()) +} + +fn validate_value_output(node: &PostAsapDagNode) -> Result<(), Error> { + let schema = &node.output_schema; + let values = schema + .fields + .iter() + .enumerate() + .filter(|(i, _)| Some(*i) != schema.time_index) + .collect::>(); + if !matches!(values.as_slice(), [(_, field)] if !field.nullable && field.dtype == SummaryFamilyType::Plain(DataType::Float64)) + || schema.time_index.is_some_and(|i| { + schema.fields.get(i).is_none_or(|f| { + f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Timestamp) + }) + }) + { + return Err(invalid( + "precompute value schema requires Float64 and an optional declared timestamp", + )); + } + Ok(()) +} + +fn fragment( + node: &PostAsapDagNode, + schemas: &[Schema], + parents: &[&PostAsapDagNode], +) -> Result { + let sources = schemas + .iter() + .enumerate() + .map(|(id, schema)| (id as u64, InputContract::bounded(schema.clone()))) + .collect(); + let mut operators = BTreeMap::new(); + let mut next = schemas.len() as u64; + let mut add = |inputs: Vec, op: Operator| -> Result { + let id = next; + next += 1; + operators.insert(id, (inputs, op)); + Ok(id) + }; + let root = match &node.payload { + Payload::Binary { operator } => { + validate_value_output(node)?; + if node.output_schema.time_index.is_none() + || parents.iter().any(|p| p.output_schema.time_index.is_none()) + { + return Err(invalid( + "precompute binary requires declared window timestamps", + )); + } + let [left, right] = schemas else { + return Err(invalid("precompute binary requires two inputs")); + }; + add( + vec![0, 1], + Operator::aligned_binary( + left.clone(), + right.clone(), + vec![(0, 0), (1, 1)], + (2, 2), + operator.clone(), + )?, + )? + } + Payload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + } => { + let [input] = schemas else { + return Err(invalid("finalize requires one state input")); + }; + validate_value_output(node)?; + let statistic = match &input.fields[2].dtype { + SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Sum, _) => { + crate::Statistic::Sum + } + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Count, + _, + ) => crate::Statistic::Count, + _ => { + return Err(invalid( + "precompute finalization requires explicit Sum or Count semantics", + )) + } + }; + let read = Operator::readout( + input.clone(), + 2, + ReadoutQuery::Exact(ExactReadout { + statistic, + lookback_ms: None, + }), + )?; + let output = read.schema(); + let read = add(vec![0], read)?; + let project = Operator::project( + output, + vec![ + ("$population".into(), Expression::Column(0)), + ("$window_end".into(), Expression::Column(1)), + ( + "value".into(), + Expression::FiniteFloat64(Box::new(Expression::ExactFloat64(2))), + ), + ], + )? + .with_output_schema(population_schema(SummaryFamilyType::Plain( + DataType::Float64, + )))?; + add(vec![read], project)? + } + Payload::SummaryAgg { + family, + input: update, + reduction, + grouping, + } => { + let [input] = schemas else { + return Err(invalid("summary update requires one input")); + }; + // Item identities resolve against the complete label set of raw + // samples; finalized readouts carry no such identity. + let raw = *input == raw_sample_schema(); + // A unit-frequency summary (HLL) observes each raw sample value. + let unit_frequency = raw + && crate::capability::is_unit_sample_frequency(update) + && matches!(family, SummaryFamilyType::Sketch(kind, _) if !matches!( + kind.algorithm(), + planner_types::post_asap::SketchAlgorithm::Cms + | planner_types::post_asap::SketchAlgorithm::CountSketch + | planner_types::post_asap::SketchAlgorithm::CmsWithHeap + | planner_types::post_asap::SketchAlgorithm::CountSketchWithHeap + )); + let keyed = update.item.is_some() && !unit_frequency; + if (keyed && !raw) || !matches!(grouping, GroupingStrategy::PerSubpopulationInstance) { + return Err(invalid( + "precompute keyed/shared update needs its dedicated physical candidate", + )); + } + crate::capability::validate_summary_kernel(family, update, grouping) + .map_err(Error::Invalid)?; + if raw + && matches!( + update.weight_domain, + planner_types::post_asap::WeightDomain::NonNegative { + proof: planner_types::post_asap::NonNegativeWeightProof::ResetAwareCounterDerivative + } + ) + { + return Err(invalid( + "a counter-derivative weight cannot be read from raw cumulative samples", + )); + } + if keyed + && matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::CmsWithHeap) + && !matches!( + update.weight_domain, + planner_types::post_asap::WeightDomain::NonNegative { .. } + ) + { + return Err(invalid("CMS requires a nonnegative weight contract")); + } + let labels = match reduction { + PlannerReduction::PerEntity => Expression::Column(0), + PlannerReduction::Reduce(keys) => Expression::LabelSet { + column: 0, + labels: keys + .keys() + .iter() + .map(|key| { + parents[0] + .output_schema + .fields + .get(*key) + // A raw label map omits absent labels; the + // series identity is not one of its labels. + .filter(|field| { + (raw || !field.nullable) + && field.name != SERIES_IDENTITY + && field.dtype == SummaryFamilyType::Plain(DataType::Utf8) + }) + .map(|f| f.name.clone()) + .ok_or_else(|| { + invalid("summary grouping must identify population labels") + }) + }) + .collect::, _>>()?, + without: keys.is_without(), + }, + }; + let weight = match &update.weight { + _ if unit_frequency => Expression::Column(2), + SummaryInputExpr::Constant(value) => Expression::Literal { + value: crate::values::Value::Float64(*value), + dtype: DataType::Float64, + }, + SummaryInputExpr::Column(ColumnRef::SampleValue) => Expression::Column(2), + SummaryInputExpr::Column(ColumnRef::Named(name)) + if parents[0].output_schema.fields.iter().any(|f| { + f.name == *name && f.dtype == SummaryFamilyType::Plain(DataType::Float64) + }) => + { + Expression::Column(2) + } + _ => { + return Err(invalid( + "summary weight does not resolve to the input value", + )) + } + }; + let mut columns = vec![ + ("$population".into(), labels), + ("$window_end".into(), Expression::Column(1)), + ("value".into(), Expression::FiniteFloat64(Box::new(weight))), + ]; + let mut fields = population_schema(SummaryFamilyType::Plain(DataType::Float64)) + .fields + .clone(); + if keyed { + let mut items = Vec::new(); + raw_items( + update.item.as_ref().expect("keyed item"), + &parents[0].output_schema, + &mut items, + )?; + for (index, (expression, dtype)) in items.into_iter().enumerate() { + let name = format!("$item{index}"); + fields.push(planner_types::post_asap::SummaryField { + name: name.clone(), + dtype: SummaryFamilyType::Plain(dtype), + nullable: false, + }); + columns.push((name, expression)); + } + } + let item_columns = (3..fields.len()).collect::>(); + let project = Operator::project(input.clone(), columns)?.with_output_schema( + Arc::new(SummarySchema { + fields, + time_index: Some(1), + }), + )?; + let projected = project.schema(); + let project = add(vec![0], project)?; + let build = if keyed { + Operator::keyed_summary_build(projected, family.clone(), 2, item_columns, vec![0])? + } else { + Operator::summary_build(projected, family.clone(), 2, Some(1), vec![0])? + }; + let built = build.schema(); + let build = add(vec![project], build)?; + add( + vec![build], + Operator::scope_timestamp(built, population_schema(family.clone()))?, + )? + } + Payload::SummaryMerge => { + let Some(input) = schemas.first() else { + return Err(invalid("summary merge requires inputs")); + }; + if schemas.iter().any(|s| s != input) { + return Err(invalid("summary merge inputs differ")); + } + let union = add( + (0..schemas.len() as u64).collect(), + Operator::union(input.clone(), schemas.len())?, + )?; + let merge = Operator::summary_merge(input.clone(), 2, vec![0])?; + let merged = merge.schema(); + let merge = add(vec![union], merge)?; + add( + vec![merge], + Operator::scope_timestamp(merged, input.clone())?, + )? + } + _ => { + return Err(invalid( + "precompute operation has no native population implementation", + )) + } + }; + CompiledPhysicalDag::from_operators(sources, operators, vec![root]) +} + +/// Resolve keyed item identities over raw sample rows: labels (absent labels +/// read as empty, as in PromQL), the sample value, or the canonical encoding +/// of the label set less excluded labels. +fn raw_items( + expr: &SummaryInputExpr, + scan: &SummarySchema, + items: &mut Vec<(Expression, DataType)>, +) -> Result<(), Error> { + // Open PromQL scans need not list every label, so any name that is not + // another scan column (value, time, series identity) reads as a label. + let label = |column: &ColumnRef| match column { + ColumnRef::Named(name) | ColumnRef::Qualified { name, .. } + if !name.starts_with('$') + && scan.fields.iter().all(|f| { + &f.name != name || f.dtype == SummaryFamilyType::Plain(DataType::Utf8) + }) => + { + Some(name.clone()) + } + _ => None, + }; + let identity = |excluding: Vec| { + ( + Expression::LabelIdentity { + column: 0, + excluding, + }, + DataType::Utf8, + ) + }; + match expr { + SummaryInputExpr::Column(ColumnRef::SampleValue) => { + items.push((Expression::Column(2), DataType::Float64)) + } + SummaryInputExpr::Column(ColumnRef::Named(name) | ColumnRef::Qualified { name, .. }) + if name == "value" => + { + items.push((Expression::Column(2), DataType::Float64)) + } + SummaryInputExpr::Column(ColumnRef::Named(name) | ColumnRef::Qualified { name, .. }) + if name == SERIES_IDENTITY => + { + items.push(identity(vec![])) + } + SummaryInputExpr::Column(column) if label(column).is_some() => items.push(( + Expression::Label { + column: 0, + name: label(column).expect("resolved label"), + }, + DataType::Utf8, + )), + SummaryInputExpr::EntityIdentity( + planner_types::post_asap::EntityIdentity::PromqlLabelSet { excluding }, + ) => items.push(identity( + excluding + .iter() + .map(|column| { + label(column).ok_or_else(|| invalid("excluded identity label is not a label")) + }) + .collect::>()?, + )), + SummaryInputExpr::Tuple(parts) if !parts.is_empty() => { + for part in parts { + raw_items(part, scan, items)?; + } + } + _ => { + return Err(invalid( + "keyed summary item does not resolve over raw samples", + )) + } + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs new file mode 100644 index 00000000..12c60ad1 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs @@ -0,0 +1,515 @@ +//! Compile a retained PromQL subtree (`Fallback`) from its typed expression. +//! The deployment supplies the raw series of each selector; the Planner +//! computes selection, range functions, subqueries, matching and aggregation. +use super::*; +use crate::operators::SubquerySteps; +use planner_types::post_asap::execution_data_state::lift_plain; +use planner_types::pre_asap::{AtModifier, BinaryOpKind, VectorMatch, VectorMatchKind}; + +/// Input slot for the raw series read by the `selector`th selector (in +/// [`raw_series`] order) of Fallback node `node`. The node's own ID names its +/// computed output, so the raw rows need another. +pub fn raw_series_input(node: NodeId, selector: usize) -> NodeId { + node | ((selector as u64 + 1) << 32) +} + +/// The Fallback node that owns a raw-series input slot. +pub(super) fn raw_series_owner(slot: NodeId) -> Option { + (slot >> 32 != 0).then_some(slot & u64::from(u32::MAX)) +} + +/// A selector expression and its raw-series row schema. +pub type Selector = (QueryExpr, Schema); + +/// The selectors a Fallback expression reads, left to right, and the row +/// schema of the raw series the deployment supplies for each at +/// [`raw_series_input`]. The rows must cover the selector's window at every +/// evaluation instant `T`, or at its `@` time: `(T - offset - range, T - offset]`; +/// under a subquery `[R:S] offset O` that is `(T - O - R - offset - range, T - O - offset]`. +pub fn raw_series(expression: &QueryExpr) -> Result, Error> { + Ok(lower(expression)?.selectors) +} + +/// An operator input: a selector's raw rows or an earlier step. +pub(super) enum Input { + Raw(usize), + Step(usize), +} + +/// Operators computing an expression; the last step is its result. +#[derive(Default)] +pub(super) struct Lowering { + pub selectors: Vec, + pub steps: Vec<(Operator, Vec)>, +} + +pub(super) fn lower(expression: &QueryExpr) -> Result { + let mut lowering = Lowering::default(); + lowering.value(expression)?; + Ok(lowering) +} + +fn declared(expression: &QueryExpr) -> Result { + let schema = expression + .output_schema() + .map_err(|error| invalid(error.to_string()))?; + Ok(Arc::new(lift_plain(&schema))) +} + +fn millis(duration: &std::time::Duration) -> Result { + i64::try_from(duration.as_millis()).map_err(|_| invalid("PromQL duration exceeds Int64")) +} + +/// A fixed `@` time. `start()`/`end()` depend on the deployment's range query. +fn at(shift: &planner_types::pre_asap::TimeShift) -> Result, Error> { + match shift.at { + None => Ok(None), + Some(AtModifier::Timestamp(at)) => Ok(Some(at)), + Some(_) => Err(invalid("@ start() and @ end() depend on the range query")), + } +} + +/// `TimeRange { range, [TimeShift { offset, @ }], Scan }`: range, offset, `@`. +fn selector(expression: &QueryExpr) -> Result<(i64, i64, Option), Error> { + let QueryExpr::TimeRange { range, child } = expression else { + return Err(invalid("PromQL operand must be a series selector")); + }; + let (offset, at, scan) = match child.as_ref() { + QueryExpr::TimeShift { shift, child } => (shift.offset_ms, at(shift)?, child.as_ref()), + scan => (0, None, scan), + }; + if !matches!(scan, QueryExpr::Scan { .. }) { + return Err(invalid("PromQL selector must read one scan")); + } + Ok((millis(range)?, offset, at)) +} + +/// PromQL scalar-valued expressions have no labels to match. +fn scalar(expression: &QueryExpr) -> bool { + matches!( + expression, + QueryExpr::PromqlScalarBridge(_) + | QueryExpr::PromqlScalarFromVector(_) + | QueryExpr::EvalTimestamp + ) +} + +impl Lowering { + fn schema(&self, input: &Input) -> Schema { + match input { + Input::Raw(i) => self.selectors[*i].1.clone(), + Input::Step(i) => self.steps[*i].0.schema(), + } + } + + fn add(&mut self, operator: Operator, inputs: Vec) -> Input { + self.steps.push((operator, inputs)); + Input::Step(self.steps.len() - 1) + } + + /// Conform `operator` to the logical schema of the expression it computes. + fn push( + &mut self, + operator: Operator, + inputs: Vec, + logical: &QueryExpr, + ) -> Result { + Ok(self.add(operator.with_output_schema(declared(logical)?)?, inputs)) + } + + fn read(&mut self, selector: &QueryExpr) -> Result { + let schema = declared(selector)?; + if !schema + .fields + .iter() + .any(|f| f.name == promql_rows::SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "PromQL fallback requires the complete series identity", + )); + } + self.selectors.push((selector.clone(), schema)); + Ok(Input::Raw(self.selectors.len() - 1)) + } + + /// An instant vector, or a scalar for scalar-valued expressions. + fn value(&mut self, expression: &QueryExpr) -> Result { + match expression { + QueryExpr::TimeRange { .. } => { + let (range, offset, at) = selector(expression)?; + let input = self.read(expression)?; + let schema = self.schema(&input); + self.push( + Operator::series_window(schema, None, range, offset, at, None)?, + vec![input], + expression, + ) + } + QueryExpr::Aggregate { + reduction: planner_types::pre_asap::Reduction::PerEntity, + measures, + having: None, + child, + .. + } => { + let [function] = measures.as_slice() else { + return Err(invalid("range function requires one measure")); + }; + self.range_function(function, child, expression) + } + QueryExpr::Aggregate { + reduction: planner_types::pre_asap::Reduction::Reduce(keys), + measures, + having: None, + child, + .. + } => { + let [measure] = measures.as_slice() else { + return Err(invalid("vector aggregation requires one measure")); + }; + let input = self.value(child)?; + self.aggregate(input, measure, keys, expression) + } + QueryExpr::Sort { + keys, + partition_by, + child, + } => { + let step = self.value(child)?; + let input = self.schema(&step); + let keys = keys + .iter() + .map(|key| match key.expr { + QueryExpr::Column(column) => Ok(SortKey { + column, + descending: !key.ascending, + nulls_first: key.nulls_first, + }), + _ => Err(invalid("sort key must be a column")), + }) + .collect::>()?; + let groups = groups(&input, partition_by)?; + self.push(Operator::sort(input, keys, groups)?, vec![step], expression) + } + QueryExpr::Limit { n, offset, child } => { + let step = self.value(child)?; + let input = self.schema(&step); + // `topk by (...)` partitions through the Sort it limits. + let groups = match child.as_ref() { + QueryExpr::Sort { partition_by, .. } => groups(&input, partition_by)?, + _ => vec![], + }; + self.push( + Operator::limit(input, *n as u64, *offset as u64, groups)?, + vec![step], + expression, + ) + } + QueryExpr::BinaryOp { + op: BinaryOpKind::Arithmetic(op), + lhs, + rhs, + vector_match, + } => { + let (vector, literal, literal_left) = match ( + row_values::scalar_literal(lhs), + row_values::scalar_literal(rhs), + ) { + (None, Some(value)) => (lhs, value, false), + (Some(value), None) => (rhs, value, true), + (None, None) => return self.match_vectors(expression, vector_match), + _ => return Err(invalid("PromQL arithmetic between two literals")), + }; + let step = self.value(vector)?; + let input = self.schema(&step); + let value = named_column(&input, &ColumnRef::SampleValue)?; + let literal = Expression::Literal { + value: crate::values::Value::Float64(literal), + dtype: DataType::Float64, + }; + let operator = planner_types::post_asap::BinaryOperator { + kind: planner_types::pre_asap::BinaryOpKind::Arithmetic(op.clone()), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }; + let (left, right) = if literal_left { + (literal, Expression::Column(value)) + } else { + (Expression::Column(value), literal) + }; + let columns = input + .fields + .iter() + .enumerate() + .map(|(i, field)| { + let expression = if i == value { + Expression::Binary { + operator: operator.clone(), + left: Box::new(left.clone()), + right: Box::new(right.clone()), + } + } else { + Expression::Column(i) + }; + (field.name.clone(), expression) + }) + .collect(); + self.push(Operator::project(input, columns)?, vec![step], expression) + } + QueryExpr::PromqlScalarFromVector(child) => { + let step = self.value(child)?; + let input = self.schema(&step); + let value = named_column(&input, &ColumnRef::SampleValue)?; + self.push( + Operator::vector_to_scalar(input, value)?, + vec![step], + expression, + ) + } + QueryExpr::PromqlVectorFromScalar(child) => { + let step = self.value(child)?; + let input = self.schema(&step); + Ok(self.add( + Operator::scope_timestamp(input, declared(expression)?)?, + vec![step], + )) + } + QueryExpr::PromqlScalarBridge(_) => { + let value = row_values::scalar_literal(expression) + .ok_or_else(|| invalid("PromQL scalar must be a literal"))?; + self.push( + Operator::scalar(crate::values::Value::Float64(value), DataType::Float64)?, + vec![], + expression, + ) + } + _ => Err(invalid("PromQL expression has no native fallback lowering")), + } + } + + /// One-to-one vector arithmetic: both sides reduce to their matching + /// labels, which are also the result's labels. + fn match_vectors( + &mut self, + logical: &QueryExpr, + vector_match: &Option, + ) -> Result { + let QueryExpr::BinaryOp { + op: kind, lhs, rhs, .. + } = logical + else { + unreachable!() + }; + if scalar(lhs) || scalar(rhs) { + return Err(invalid("PromQL arithmetic with a non-literal scalar")); + } + let (matching, labels) = match vector_match { + None => (VectorMatchKind::Ignoring, vec![]), + Some(VectorMatch { + kind, + labels, + grouping: None, + }) => (kind.clone(), labels.clone()), + Some(_) => return Err(invalid("group_left/group_right matching is unsupported")), + }; + let mut sides = Vec::new(); + for side in [lhs, rhs] { + let step = self.value(side)?; + let input = self.schema(&step); + sides.push(self.add( + Operator::series_labels(input, matching.clone(), labels.clone())?, + vec![step], + )); + } + let (left, right) = (self.schema(&sides[0]), self.schema(&sides[1])); + let operator = planner_types::post_asap::BinaryOperator { + kind: kind.clone(), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }; + self.push( + Operator::series_binary(left, right, operator)?, + sides, + logical, + ) + } + + /// `function(matrix)`, where the matrix is a range selector or a subquery. + fn range_function( + &mut self, + function: &AggIntent, + matrix: &QueryExpr, + logical: &QueryExpr, + ) -> Result { + let function = unbound(function)?; + let (subquery, offset, at_ms) = match matrix { + QueryExpr::TimeShift { shift, child } => (child.as_ref(), shift.offset_ms, at(shift)?), + other => (other, 0, None), + }; + let QueryExpr::PromqlSubquery { + range: outer, + resolution, + child, + } = subquery + else { + let (range, offset, at) = selector(matrix)?; + let input = self.read(matrix)?; + let schema = self.schema(&input); + return self.push( + Operator::series_window(schema, Some(function), range, offset, at, None)?, + vec![input], + logical, + ); + }; + let step = resolution.as_ref().ok_or_else(|| { + invalid("subquery resolution defaults to the deployment evaluation interval") + })?; + let steps = SubquerySteps { + range_ms: millis(outer)?, + step_ms: millis(step)?, + offset_ms: offset, + at_ms, + }; + // Each step evaluates a per-series selection or range function. + let (inner, selected) = match child.as_ref() { + QueryExpr::Aggregate { + reduction: planner_types::pre_asap::Reduction::PerEntity, + measures, + having: None, + child: selected, + .. + } => match measures.as_slice() { + [inner] => (Some(unbound(inner)?), selected.as_ref()), + _ => return Err(invalid("range function requires one measure")), + }, + selected => (None, selected), + }; + let (range, inner_offset, inner_at) = selector(selected)?; + let raw = self.read(selected)?; + let schema = self.schema(&raw); + let step = self.push( + Operator::series_window(schema, inner, range, inner_offset, inner_at, Some(steps))?, + vec![raw], + child, + )?; + let input = self.schema(&step); + self.push( + Operator::series_window(input, Some(function), steps.range_ms, offset, at_ms, None)?, + vec![step], + logical, + ) + } + + /// Cross-series aggregation. A global aggregate groups by one constant so + /// that no input series yields an empty vector, not one row. + fn aggregate( + &mut self, + mut step: Input, + measure: &AggIntent, + keys: &GroupKeys, + logical: &QueryExpr, + ) -> Result { + let mut input = self.schema(&step); + let value = named_column(&input, &ColumnRef::SampleValue)?; + let reduction = match measure { + AggIntent::Sum { col: None } => Reduction::Sum(value), + AggIntent::Avg { col: None } => Reduction::Avg(value), + AggIntent::Min { col: None } => Reduction::Min(value), + AggIntent::Max { col: None } => Reduction::Max(value), + AggIntent::Count { .. } => Reduction::Count, + _ => return Err(invalid("vector aggregate has no native lowering")), + }; + let mut groups = if keys.is_without() { + // Group by every remaining label, including the rewritten identity. + let excluded = keys.keys(); + if excluded.iter().any(|&i| i >= input.fields.len()) { + return Err(invalid("grouping column out of range")); + } + let names = excluded.iter().map(|&i| input.fields[i].name.clone()); + let relabel = + Operator::series_labels(input.clone(), VectorMatchKind::Ignoring, names.collect())?; + step = self.add(relabel, vec![step]); + (0..input.fields.len()) + .filter(|&i| Some(i) != input.time_index && i != value && !excluded.contains(&i)) + .collect() + } else { + groups(&input, keys)? + }; + let global = groups.is_empty(); + if global { + let mut columns = (0..input.fields.len()) + .map(|i| (input.fields[i].name.clone(), Expression::Column(i))) + .collect::>(); + columns.push(( + "$promql_global_group".into(), + Expression::Literal { + value: crate::values::Value::Utf8("".into()), + dtype: DataType::Utf8, + }, + )); + let project = Operator::project(input, columns)?; + input = project.schema(); + groups = vec![input.fields.len() - 1]; + step = self.add(project, vec![step]); + } + let output = declared(logical)?; + let name = output + .fields + .last() + .ok_or_else(|| invalid("aggregate output lacks a value"))? + .name + .clone(); + let aggregate = Operator::aggregate(input, groups, vec![(name, reduction)])?; + let actual = aggregate.schema(); + let step = self.add(aggregate, vec![step]); + // Drop the constant group; convert counts where PromQL declares Float64. + let skip = usize::from(global); + let columns = actual.fields[skip..] + .iter() + .zip(&output.fields) + .enumerate() + .map(|(i, (field, declared))| { + let column = i + skip; + let expression = if field.dtype != declared.dtype { + Expression::ExactFloat64(column) + } else { + Expression::Column(column) + }; + (field.name.clone(), expression) + }) + .collect(); + self.push(Operator::project(actual, columns)?, vec![step], logical) + } +} + +fn unbound(intent: &AggIntent) -> Result, Error> { + Ok(match intent { + AggIntent::Rate => AggIntent::Rate, + AggIntent::Increase => AggIntent::Increase, + AggIntent::Delta => AggIntent::Delta, + AggIntent::Count { accuracy } => AggIntent::Count { + accuracy: accuracy.clone(), + }, + AggIntent::Sum { col: None } => AggIntent::Sum { col: None }, + AggIntent::Avg { col: None } => AggIntent::Avg { col: None }, + AggIntent::Min { col: None } => AggIntent::Min { col: None }, + AggIntent::Max { col: None } => AggIntent::Max { col: None }, + AggIntent::IRate => AggIntent::IRate, + AggIntent::IDelta => AggIntent::IDelta, + AggIntent::Changes => AggIntent::Changes, + AggIntent::Resets => AggIntent::Resets, + AggIntent::LastOverTime => AggIntent::LastOverTime, + AggIntent::Quantile { + col: None, + q, + accuracy, + } => AggIntent::Quantile { + col: None, + q: *q, + accuracy: accuracy.clone(), + }, + _ => return Err(invalid("unsupported PromQL range function")), + }) +} diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs new file mode 100644 index 00000000..d2c13328 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -0,0 +1,318 @@ +//! A bounded PromQL source row carries the entire label set, not just labels +//! mentioned by the query. The source adapter owns this lossless encoding. +use super::*; +use planner_types::pre_asap::DataType; +use std::rc::Rc; + +/// Not a legal PromQL label name, so it cannot shadow a user label. +pub use planner_types::pre_asap::schema::PROMQL_SERIES_IDENTITY as SERIES_IDENTITY_COLUMN; + +/// Canonical, reversible identity. JSON object encoding preserves label names, +/// empty values and escaping; sorting makes ingestion order irrelevant. +pub fn encode_series_identity(labels: &BTreeMap) -> Result { + serde_json::to_string(labels).map_err(|error| invalid(error.to_string())) +} + +pub fn decode_series_identity(encoded: &str) -> Result, Error> { + let labels: BTreeMap = + serde_json::from_str(encoded).map_err(|error| invalid(error.to_string()))?; + if encode_series_identity(&labels)? != encoded { + return Err(invalid("series identity is not canonically encoded")); + } + Ok(labels) +} + +/// Resolve the row representation before candidate search; see +/// [`planner_types::pre_asap::schema::with_promql_series_identity`]. +pub fn with_series_identity(root: &QueryExpr) -> Result { + planner_types::pre_asap::schema::with_promql_series_identity(root).map_err(invalid) +} + +/// Construct source rows only from full identities. The named label columns +/// are projections of that same identity and cannot independently redefine it. +pub fn series_row( + schema: &Schema, + labels: &BTreeMap, + timestamp: i64, + value: f64, +) -> Result, Error> { + use crate::values::Value; + let identity = encode_series_identity(labels)?; + let mut found = false; + let row = schema + .fields + .iter() + .enumerate() + .map(|(index, field)| { + if field.name == SERIES_IDENTITY_COLUMN { + if field.dtype != SummaryFamilyType::Plain(DataType::Utf8) + || field.nullable + || found + { + return Err(invalid("invalid series identity column")); + } + found = true; + Ok(Value::Utf8(identity.clone().into())) + } else if Some(index) == schema.time_index { + Ok(Value::Timestamp(timestamp)) + } else if field.name == "value" + && field.dtype == SummaryFamilyType::Plain(DataType::Float64) + { + Ok(Value::Float64(value)) + } else if field.dtype == SummaryFamilyType::Plain(DataType::Utf8) { + Ok(labels.get(&field.name).map_or_else( + || Value::Utf8("".into()), + |value| Value::Utf8(value.clone().into()), + )) + } else { + Err(invalid("unsupported PromQL source column")) + } + }) + .collect::, _>>()?; + if !found { + return Err(invalid("source lacks its full series identity")); + } + Ok(row) +} + +/// Compile the selected TopK computation above an existing maintained-population +/// source. The boundary supplies the complete eligible vector, not a truncated +/// TopK result; ranking remains a native physical operator. +pub fn compile_current_series_readout( + selected: &Rc, +) -> Result { + use planner_types::post_asap::{ + compile_post_asap_dag, maintained_population::PopulationReadout, SummaryField, + }; + let mut dag = compile_post_asap_dag(selected).map_err(|error| invalid(error.to_string()))?; + // Typed snapshot candidates already carry full identity throughout the DAG. + // Cut at the population output, preserving all selected heap/readout nodes. + let populations = dag.nodes.iter().filter(|node| matches!(&node.payload, + Payload::Value { operation: ValueOperation::MaintainPopulation { population } } + if matches!(population.input, planner_types::post_asap::maintained_population::PopulationInput::CurrentSeries(_)) + )).collect::>(); + if let [population] = populations.as_slice() { + if population + .output_schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + { + return compile( + &dag, + BTreeMap::from([( + u64::from(population.id.0), + InputContract::bounded(Arc::new(population.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + ); + } + } + if dag.nodes.len() != 3 + || !dag.nodes.iter().any(|node| { + node.id == dag.root + && matches!( + node.payload, + Payload::Value { + operation: ValueOperation::ReadPopulation { + readout: PopulationReadout::TopK { .. } + } + } + ) + }) + { + return Err(invalid( + "expected one selected current-series TopK computation", + )); + } + let mut frontier = None; + for node in &mut dag.nodes { + match &mut node.payload { + Payload::Fallback { expression } => { + *expression = with_series_identity(expression)?; + } + Payload::Value { + operation: ValueOperation::MaintainPopulation { .. }, + } => { + frontier = Some(u64::from(node.id.0)); + } + Payload::Value { + operation: + ValueOperation::ReadPopulation { + readout: PopulationReadout::TopK { .. }, + }, + } => {} + _ => return Err(invalid("unsupported current-series readout dependency")), + } + if node + .output_schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "current-series input already has a physical identity column", + )); + } + node.output_schema.fields.push(SummaryField { + name: SERIES_IDENTITY_COLUMN.into(), + dtype: SummaryFamilyType::Plain(DataType::Utf8), + nullable: false, + }); + } + for edge in &mut dag.edges { + edge.intermediate_schema = dag + .nodes + .iter() + .find(|node| node.id == edge.producer) + .unwrap() + .output_schema + .clone(); + } + let frontier = frontier.ok_or_else(|| invalid("missing current-series population"))?; + let schema = Arc::new( + dag.nodes + .iter() + .find(|node| u64::from(node.id.0) == frontier) + .unwrap() + .output_schema + .clone(), + ); + compile( + &dag, + BTreeMap::from([(frontier, InputContract::bounded(schema))]), + &[u64::from(dag.root.0)], + ) +} + +/// Compile selected ranking or aggregation above an exact per-series Rate +/// readout. Deployments bind complete window readouts at this boundary; +/// the heap is rebuilt independently for each evaluation. This does not move +/// that frontier to ingestion time or authorize combining finalized rates. +pub fn compile_rate_ranking( + selected: &Rc, +) -> Result< + ( + Rc, + CompiledPhysicalDag, + ), + Error, +> { + use planner_types::post_asap::{ + compile_post_asap_dag_with_node_ids, ExactKind, SummaryExpr, SummaryNode, + }; + fn frontier(node: &Rc) -> Option> { + match &node.expr { + SummaryExpr::ValueOperation { + child, + operation: ValueOperation::FinalizeExactAccumulator, + timing: planner_types::post_asap::ExecutionTiming::QueryTime, + } if matches!(&child.expr, SummaryExpr::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + reduction: planner_types::pre_asap::Reduction::PerEntity, + child: raw, .. + } if matches!(&raw.expr, SummaryExpr::KeepPreAsap(expr) if matches!(expr.as_ref(), QueryExpr::TimeRange { .. }))) => + { + Some(Rc::clone(node)) + } + SummaryExpr::ValueOperation { child, .. } | SummaryExpr::SummaryAgg { child, .. } => { + frontier(child) + } + SummaryExpr::SummaryEstimate { summary_input, .. } => frontier(summary_input), + _ => None, + } + } + let source = frontier(selected) + .ok_or_else(|| invalid("ranking requires one exact per-series Rate frontier"))?; + if !source + .schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid("Rate ranking requires complete series identity")); + } + let compiled = compile_post_asap_dag_with_node_ids(selected) + .map_err(|error| invalid(error.to_string()))?; + let id = u64::from( + compiled + .node_ids + .node_id(&source) + .ok_or_else(|| invalid("missing Rate frontier"))? + .0, + ); + let program = compile( + &compiled.dag, + BTreeMap::from([(id, InputContract::bounded(Arc::new(source.schema.clone())))]), + &[u64::from(compiled.dag.root.0)], + )?; + Ok((source, program)) +} + +/// Compile a lifecycle-timed DAG whose heap or grouped Sum over per-series +/// Rate readouts runs at ingestion time: fresh aggregate state per closed +/// window. The input is the complete collection of per-series counter states. +pub fn compile_fixed_window_rate_aggregation( + dag: &planner_types::post_asap::PostAsapDag, +) -> Result { + use planner_types::post_asap::{ExactKind, ExecutionTiming, SketchAlgorithm}; + let sources = dag + .nodes + .iter() + .filter(|n| { + matches!( + &n.payload, + Payload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + reduction: planner_types::pre_asap::Reduction::PerEntity, + .. + } + ) + }) + .collect::>(); + let heaps = dag + .nodes + .iter() + .filter(|n| { + n.output_state.timing == ExecutionTiming::IngestionTime + && match &n.payload { + Payload::SummaryAgg { + family: SummaryFamilyType::Sketch(kind, _), + .. + } => matches!( + kind.algorithm(), + SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap + ), + Payload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + .. + } => true, + _ => false, + } + }) + .collect::>(); + let ([source], [heap]) = (sources.as_slice(), heaps.as_slice()) else { + return Err(invalid( + "expected one selected fixed-window Rate aggregation", + )); + }; + if !source + .output_schema + .fields + .iter() + .any(|f| f.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "fixed-window Rate aggregation requires complete series identity", + )); + } + compile_candidate( + dag, + BTreeMap::from([( + u64::from(source.id.0), + InputContract::bounded(Arc::new(source.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &[u64::from(heap.id.0)], + ) +} diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/asap-physical-operators/src/physical_planner/promql_values.rs new file mode 100644 index 00000000..98032505 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/promql_values.rs @@ -0,0 +1,278 @@ +//! Physical scalar/vector contracts preserve complete label sets across native computation. +use super::*; + +pub fn scalar_schema() -> Schema { + crate::operators::vector_binary::value_schema(true) +} +pub fn vector_schema() -> Schema { + crate::operators::vector_binary::value_schema(false) +} + +pub fn matrix_schema() -> Schema { + crate::operators::vector_window::matrix_schema() +} + +pub fn compile_scalar(value: f64) -> Result { + let operator = Operator::scalar( + crate::values::Value::Float64(value), + planner_types::pre_asap::DataType::Float64, + )? + .with_output_schema(scalar_schema())?; + CompiledPhysicalDag::from_operators( + BTreeMap::new(), + BTreeMap::from([(0, (vec![], operator))]), + vec![0], + ) +} + +pub fn compile_temporal( + intent: &AggIntent, + preserve_metric_name: bool, +) -> Result { + let operator = Operator::range_window(intent.clone())?; + let mut operators = vec![operator]; + if !preserve_metric_name { + operators.push(Operator::project( + vector_schema(), + vec![ + ( + "labels".into(), + Expression::LabelSet { + column: 0, + labels: vec![], + without: true, + }, + ), + ("value".into(), Expression::Column(1)), + ], + )?); + } + unary(operators, matrix_schema()) +} + +pub fn compile_histogram_quantile() -> Result { + CompiledPhysicalDag::from_operators( + BTreeMap::from([ + (0, InputContract::bounded(scalar_schema())), + (1, InputContract::bounded(vector_schema())), + ]), + BTreeMap::from([(2, (vec![0, 1], Operator::histogram_quantile()))]), + vec![2], + ) +} + +/// Compile before deployment chooses readers. Input slots 0 and 1 retain operand order. +pub fn compile_binary( + operator: &planner_types::post_asap::BinaryOperator, + return_bool: bool, + left_scalar: bool, + right_scalar: bool, +) -> Result { + let left = crate::operators::vector_binary::value_schema(left_scalar); + let right = crate::operators::vector_binary::value_schema(right_scalar); + let op = Operator::vector_binary(left.clone(), right.clone(), operator.clone(), return_bool)?; + CompiledPhysicalDag::from_operators( + BTreeMap::from([ + (0, InputContract::bounded(left)), + (1, InputContract::bounded(right)), + ]), + BTreeMap::from([(2, (vec![0, 1], op))]), + vec![2], + ) +} + +fn unary(operators: Vec, input: Schema) -> Result { + let root = operators.len() as u64; + CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(input))]), + operators + .into_iter() + .enumerate() + .map(|(i, op)| ((i + 1) as u64, (vec![i as u64], op))) + .collect(), + vec![root], + ) +} + +fn grouped(grouping: &GroupKeys) -> Result { + let labels = grouping + .keys() + .iter() + .map(|key| match key { + ColumnRef::Named(label) => Ok(label.clone()), + _ => Err(invalid("vector grouping requires label names")), + }) + .collect::, _>>()?; + Operator::project( + vector_schema(), + vec![ + ("labels".into(), Expression::Column(0)), + ("value".into(), Expression::Column(1)), + ( + "group".into(), + Expression::LabelSet { + column: 0, + labels, + without: grouping.is_without(), + }, + ), + ], + ) +} + +fn vector_output(input: Schema, labels: usize, value: usize) -> Result { + let value = Expression::ExactFloat64(value); + Operator::project( + input, + vec![ + ("labels".into(), Expression::Column(labels)), + ("value".into(), value), + ], + ) +} + +pub fn compile_aggregate( + intent: &AggIntent, + grouping: &GroupKeys, +) -> Result { + let project = grouped(grouping)?; + let reduction = match intent { + AggIntent::Sum { .. } => Reduction::Sum(1), + AggIntent::Avg { .. } => Reduction::Avg(1), + AggIntent::Count { .. } => Reduction::Count, + AggIntent::Min { .. } => Reduction::Min(1), + AggIntent::Max { .. } => Reduction::Max(1), + _ => return Err(invalid("unsupported vector aggregate")), + }; + let aggregate = + Operator::aggregate(project.schema(), vec![2], vec![("value".into(), reduction)])?; + let output = vector_output(aggregate.schema(), 0, 1)?; + unary(vec![project, aggregate, output], vector_schema()) +} + +pub fn compile_sort( + descending: bool, + grouping: &GroupKeys, +) -> Result { + let project = grouped(grouping)?; + let sort = Operator::sort( + project.schema(), + vec![SortKey { + column: 1, + descending, + nulls_first: false, + }], + vec![2], + )?; + let output = vector_output(sort.schema(), 0, 1)?; + unary(vec![project, sort, output], vector_schema()) +} + +pub fn compile_limit( + n: u64, + offset: u64, + grouping: &GroupKeys, +) -> Result { + let project = grouped(grouping)?; + let limit = Operator::limit(project.schema(), n, offset, vec![2])?; + let output = vector_output(limit.schema(), 0, 1)?; + unary(vec![project, limit, output], vector_schema()) +} + +pub fn compile_negate(scalar: bool) -> Result { + let input = if scalar { + scalar_schema() + } else { + vector_schema() + }; + let mut columns = Vec::new(); + if !scalar { + columns.push(("labels".into(), Expression::Column(0))); + } + columns.push(( + if scalar { + "$promql_scalar".into() + } else { + "value".into() + }, + Expression::Negate(Box::new(Expression::Column(if scalar { 0 } else { 1 }))), + )); + unary(vec![Operator::project(input.clone(), columns)?], input) +} + +pub fn compile_vector_to_scalar() -> Result { + unary( + vec![Operator::vector_to_scalar(vector_schema(), 1)?.with_output_schema(scalar_schema())?], + vector_schema(), + ) +} + +/// A stored exact-state input retains the complete population identity. The +/// deployment supplies eligible panes; merging and finalization are computation. +pub fn exact_state_schema(family: SummaryFamilyType) -> Result { + if !matches!(family, SummaryFamilyType::ExactAggregate(..)) { + return Err(invalid("exact-state input requires an exact family")); + } + crate::values::validate_family(&family)?; + let mut schema = (*vector_schema()).clone(); + schema.fields[1].dtype = family; + Ok(Arc::new(schema)) +} + +/// Retain exact readout semantics before any deployment state is opened. +pub fn compile_exact_readout( + family: SummaryFamilyType, + lookback_ms: u64, + preserve_metric_name: bool, +) -> Result { + use planner_types::post_asap::ExactKind; + let statistic = match &family { + SummaryFamilyType::ExactAggregate(kind, _) => match kind { + ExactKind::Sum => crate::Statistic::Sum, + ExactKind::Count => crate::Statistic::Count, + ExactKind::Min => crate::Statistic::Min, + ExactKind::Max => crate::Statistic::Max, + ExactKind::Rate => crate::Statistic::Rate, + ExactKind::Increase => crate::Statistic::Increase, + ExactKind::IRate => return Err(invalid("instant-rate state readout is not supported")), + }, + _ => return Err(invalid("exact readout requires an exact family")), + }; + let input = exact_state_schema(family)?; + let merge = Operator::summary_merge(input.clone(), 1, vec![0])?; + let mut readout = Operator::readout( + merge.schema(), + 1, + ReadoutQuery::Exact(ExactReadout { + statistic, + lookback_ms: None, + }), + )?; + if matches!( + statistic, + crate::Statistic::Rate | crate::Statistic::Increase + ) { + readout = readout.with_counter_lookback( + i64::try_from(lookback_ms).map_err(|_| invalid("counter lookback exceeds Int64"))?, + )?; + } + let project = Operator::project( + readout.schema(), + vec![ + ( + "labels".into(), + if preserve_metric_name { + Expression::Column(0) + } else { + Expression::LabelSet { + column: 0, + labels: vec![], + without: true, + } + }, + ), + ("value".into(), Expression::ExactFloat64(1)), + ], + )?; + unary(vec![merge, readout, project], input) +} diff --git a/crates/asap-physical-operators/src/physical_planner/row_values.rs b/crates/asap-physical-operators/src/physical_planner/row_values.rs new file mode 100644 index 00000000..737c8065 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/row_values.rs @@ -0,0 +1,227 @@ +//! Query-time PromQL value computation over logical row schemas. +use super::*; +use planner_types::post_asap::{maintained_population::PopulationReadout, BinaryOperator}; +use planner_types::pre_asap::{BinaryOpKind, DataType, Predicate, ScalarValue}; +use std::rc::Rc; + +/// A PromQL number literal has no row schema; its consumer folds it in. +pub(super) fn scalar_literal(expression: &QueryExpr) -> Option { + match expression { + QueryExpr::PromqlScalarBridge(child) => scalar_literal(child), + QueryExpr::Literal(ScalarValue::Float64(value)) => Some(*value), + _ => None, + } +} + +/// Rows without a time column, label map, or series identity carry only +/// their group labels, so those labels are the complete PromQL identity. +fn grouped_value(input: &Schema) -> Result<(usize, Vec), Error> { + if input.time_index.is_some() + || input + .fields + .iter() + .any(|field| field.name == promql_rows::SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "row binary requires grouped rows; per-series matching needs a name-free identity", + )); + } + let mut value = None; + let mut labels = Vec::new(); + for (i, field) in input.fields.iter().enumerate() { + match &field.dtype { + SummaryFamilyType::Plain(DataType::Float64) if value.is_none() => value = Some(i), + SummaryFamilyType::Plain(DataType::Utf8) => labels.push(i), + _ => { + return Err(invalid( + "row binary requires Utf8 labels and one Float64 value", + )) + } + } + } + Ok(( + value.ok_or_else(|| invalid("row binary requires a Float64 value"))?, + labels, + )) +} + +fn arithmetic(operator: &BinaryOperator) -> Result<(), Error> { + if !matches!(operator.kind, BinaryOpKind::Arithmetic(_)) { + return Err(invalid( + "row comparison requires filter or bool semantics, which Binary does not carry", + )); + } + Ok(()) +} + +/// Apply `vector op scalar` (or `scalar op vector`) to each row's value. +pub(super) fn scalar_binary( + input: &Schema, + operator: &BinaryOperator, + literal: f64, + literal_left: bool, +) -> Result { + arithmetic(operator)?; + let (value, _) = grouped_value(input)?; + let literal = Expression::Literal { + value: crate::values::Value::Float64(literal), + dtype: DataType::Float64, + }; + let columns = input + .fields + .iter() + .enumerate() + .map(|(i, field)| { + let expression = if i != value { + Expression::Column(i) + } else if literal_left { + binary(operator, literal.clone(), Expression::Column(i)) + } else { + binary(operator, Expression::Column(i), literal.clone()) + }; + (field.name.clone(), expression) + }) + .collect(); + Operator::project(input.clone(), columns) +} + +/// One-to-one PromQL matching of grouped rows on equal label sets. Returns +/// the inner equi-join and the projection that applies the operator. +pub(super) fn grouped_binary( + left: &Schema, + right: &Schema, + operator: &BinaryOperator, +) -> Result<(Operator, Operator), Error> { + arithmetic(operator)?; + let (left_value, left_labels) = grouped_value(left)?; + let (right_value, right_labels) = grouped_value(right)?; + if left_labels.len() != right_labels.len() { + return Err(invalid("row binary inputs have different label sets")); + } + let width = left.fields.len(); + let keys = left_labels + .iter() + .map(|&l| { + let name = &left.fields[l].name; + let r = right_labels + .iter() + .copied() + .find(|&r| &right.fields[r].name == name) + .ok_or_else(|| invalid("row binary inputs have different label sets"))?; + let (a, b) = ( + Rc::new(QueryExpr::Column(l)), + Rc::new(QueryExpr::Column(width + r)), + ); + let equal = QueryExpr::Compare { + left: a.clone(), + op: CompareOpKind::Eq, + right: b.clone(), + }; + // A nullable label compares like PromQL's empty label: absent on both sides matches. + Ok(if left.fields[l].nullable || right.fields[r].nullable { + QueryExpr::BoolOr(vec![ + equal, + QueryExpr::BoolAnd(vec![QueryExpr::IsNull(a), QueryExpr::IsNull(b)]), + ]) + } else { + equal + }) + }) + .collect::, Error>>()?; + let predicate = Predicate(Rc::new(QueryExpr::BoolAnd(keys))); + let mut joined = left.fields.clone(); + joined.extend(right.fields.iter().cloned()); + let join = Operator::relational_join( + left.clone(), + right.clone(), + planner_types::pre_asap::JoinKind::Inner, + &predicate, + Arc::new(planner_types::post_asap::SummarySchema { + fields: joined, + time_index: None, + }), + )?; + let columns = left + .fields + .iter() + .enumerate() + .map(|(i, field)| { + let expression = if i == left_value { + binary( + operator, + Expression::Column(i), + Expression::Column(width + right_value), + ) + } else { + Expression::Column(i) + }; + (field.name.clone(), expression) + }) + .collect(); + let project = Operator::project(join.schema(), columns)?; + Ok((join, project)) +} + +fn binary(operator: &BinaryOperator, left: Expression, right: Expression) -> Expression { + Expression::Binary { + operator: operator.clone(), + left: Box::new(left), + right: Box::new(right), + } +} + +/// Aggregate readouts of a maintained current-series population, as a chain. +pub(super) fn population_aggregate( + input: &Schema, + grouping: &[String], + readout: &PopulationReadout, +) -> Result, Error> { + let groups = grouping + .iter() + .map(|name| named_column(input, &ColumnRef::Named(name.clone()))) + .collect::, _>>()?; + let value = named_column(input, &ColumnRef::SampleValue)?; + let reduction = match readout { + PopulationReadout::Sum => Reduction::Sum(value), + PopulationReadout::Count => Reduction::Count, + PopulationReadout::Average => Reduction::Avg(value), + PopulationReadout::Quantile { q } => Reduction::Quantile { + column: value, + q: *q, + }, + PopulationReadout::TopK { .. } => { + return Err(invalid( + "TopK population readout ranks; it does not aggregate", + )) + } + }; + if !groups.is_empty() { + return Ok(vec![Operator::aggregate( + input.clone(), + groups, + vec![("value".into(), reduction)], + )?]); + } + // A global aggregate over no members is an empty PromQL vector, not one row. + let aggregate = Operator::aggregate( + input.clone(), + vec![], + vec![ + ("value".into(), reduction), + ("members".into(), Reduction::Count), + ], + )?; + let zero = Expression::Literal { + value: crate::values::Value::Int64(0), + dtype: DataType::Int64, + }; + let filter = Operator::filter( + aggregate.schema(), + Expression::Less(Box::new(zero), Box::new(Expression::Column(1))), + )?; + let project = Operator::project( + filter.schema(), + vec![("value".into(), Expression::Column(0))], + )?; + Ok(vec![aggregate, filter, project]) +} diff --git a/crates/asap-physical-operators/src/plan/mod.rs b/crates/asap-physical-operators/src/plan/mod.rs new file mode 100644 index 00000000..0dce4dd8 --- /dev/null +++ b/crates/asap-physical-operators/src/plan/mod.rs @@ -0,0 +1,160 @@ +//! Immutable physical graph, operator contracts and pre-execution validation. +use crate::{ + runtime::{Input, OutputStream, RunContext}, + Error, +}; +use std::{ + collections::{BTreeMap, BTreeSet}, + fmt::Debug, +}; +pub type NodeId = u64; +mod properties; +pub use properties::{Boundedness, Emission, PlanProperties}; +/// Operators own computation. The runtime provides already-connected inputs; +/// an operator must not recursively execute another plan node itself. +pub trait PhysicalOperator { + fn name(&self) -> &str; + /// Source implementations must explicitly declare finite input before feeding blocking operators. + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + PlanProperties { + boundedness: Boundedness::from_inputs(inputs), + emission: Emission::Unknown, + } + } + fn requires_bounded_input(&self) -> bool { + false + } + + /// Validate run-specific contracts before any source is opened. + fn validate_context(&self, _context: &RunContext) -> Result<(), Error> { + Ok(()) + } + fn input_schemas(&self) -> Vec; + fn output_schema(&self) -> S; + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error>; + fn output_bytes(&self, value: &V) -> usize; +} +pub(crate) struct Node<'a, V, S> { + pub(crate) inputs: Vec, + pub(crate) operator: Box + 'a>, +} +pub struct PhysicalDag<'a, V, S> { + pub(crate) nodes: BTreeMap>, +} +impl Default for PhysicalDag<'_, V, S> { + fn default() -> Self { + Self { + nodes: BTreeMap::new(), + } + } +} +impl<'a, V: 'a, S: Clone + PartialEq + Debug + 'a> PhysicalDag<'a, V, S> { + pub fn add( + &mut self, + id: NodeId, + inputs: Vec, + operator: impl PhysicalOperator + 'a, + ) -> Result<(), Error> { + self.add_boxed(id, inputs, Box::new(operator)) + } + pub fn add_boxed( + &mut self, + id: NodeId, + inputs: Vec, + operator: Box + 'a>, + ) -> Result<(), Error> { + if self.nodes.contains_key(&id) { + return Err(Error::Invalid(format!("duplicate node {id}"))); + } + self.nodes.insert(id, Node { inputs, operator }); + Ok(()) + } + pub fn validate(&self, roots: &[NodeId]) -> Result<(), Error> { + self.properties(roots).map(|_| ()) + } + /// Derive properties while checking topology and schemas, before starting sources. + pub fn properties(&self, roots: &[NodeId]) -> Result, Error> { + fn visit( + dag: &PhysicalDag<'_, V, S>, + id: NodeId, + active: &mut BTreeSet, + done: &mut BTreeMap, + ) -> Result { + if let Some((depth, _)) = done.get(&id) { + return Ok(*depth); + } + if active.len() >= 128 { + return Err(Error::Invalid( + "DAG exceeds the supported execution depth of 128".into(), + )); + } + if !active.insert(id) { + return Err(Error::Invalid(format!("cycle at node {id}"))); + } + let node = dag + .nodes + .get(&id) + .ok_or_else(|| Error::Invalid(format!("missing node {id}")))?; + let expected = node.operator.input_schemas(); + if expected.len() != node.inputs.len() { + return Err(Error::Invalid(format!("node {id} input arity mismatch"))); + } + let mut depth = 1; + let mut input_properties = Vec::new(); + for (input, schema) in node.inputs.iter().zip(expected) { + depth = depth.max(1 + visit(dag, *input, active, done)?); + input_properties.push(done[input].1); + let actual = dag.nodes[input].operator.output_schema(); + if actual != schema { + return Err(Error::Invalid(format!( + "node {id} input {input} schema mismatch: {actual:?} vs {schema:?}" + ))); + } + } + if depth > 128 { + return Err(Error::Invalid( + "DAG exceeds the supported execution depth of 128".into(), + )); + } + if node.operator.requires_bounded_input() + && input_properties + .iter() + .any(|p| p.boundedness != Boundedness::Bounded) + { + return Err(Error::Invalid(format!( + "node {id} ({}) requires bounded inputs", + node.operator.name() + ))); + } + let properties = node.operator.properties(&input_properties); + active.remove(&id); + done.insert(id, (depth, properties)); + Ok(depth) + } + if roots.is_empty() { + return Err(Error::Invalid("execution needs a root".into())); + } + let mut done = BTreeMap::new(); + for &root in roots { + visit(self, root, &mut BTreeSet::new(), &mut done)?; + } + Ok(done + .into_iter() + .map(|(id, (_, properties))| (id, properties)) + .collect()) + } + pub fn execute<'r>( + &'r self, + roots: &[NodeId], + context: RunContext, + ) -> Result>, Error> + where + 'a: 'r, + { + crate::runtime::execute(self, roots, context) + } +} diff --git a/crates/asap-physical-operators/src/plan/properties.rs b/crates/asap-physical-operators/src/plan/properties.rs new file mode 100644 index 00000000..7ec770ea --- /dev/null +++ b/crates/asap-physical-operators/src/plan/properties.rs @@ -0,0 +1,32 @@ +//! Execution facts used to reject operators that cannot finish on their inputs. +#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug, PartialEq, Eq)] +pub enum Boundedness { + /// The source or operator promises a finite result for this run. + Bounded, + Unbounded, + /// No finite-input guarantee has been supplied. + Unknown, +} +impl Boundedness { + pub fn from_inputs(inputs: &[PlanProperties]) -> Self { + if inputs.iter().any(|p| p.boundedness == Self::Unbounded) { + Self::Unbounded + } else if inputs.is_empty() || inputs.iter().any(|p| p.boundedness == Self::Unknown) { + Self::Unknown + } else { + Self::Bounded + } + } +} +#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug, PartialEq, Eq)] +pub enum Emission { + Incremental, + /// Produces its result only after all inputs end, even if accumulation is incremental. + AfterInput, + Unknown, +} +#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug, PartialEq, Eq)] +pub struct PlanProperties { + pub boundedness: Boundedness, + pub emission: Emission, +} diff --git a/crates/asap-physical-operators/src/readout.rs b/crates/asap-physical-operators/src/readout.rs new file mode 100644 index 00000000..e884ecc3 --- /dev/null +++ b/crates/asap-physical-operators/src/readout.rs @@ -0,0 +1,111 @@ +//! Readouts over merged exact summary states. +use crate::summary_kernels::exact::ExactAccumulator; +use crate::{AggregateCore, KeyByLabelValues, Statistic}; +use std::sync::Arc; + +fn merge_exact_states( + states: impl IntoIterator>, +) -> Result { + let mut states = states.into_iter(); + let exact = |state: &Arc| { + state + .as_any() + .downcast_ref::() + .cloned() + .ok_or_else(|| "readout requires Planner exact state".to_string()) + }; + let mut merged = exact(&states.next().ok_or("empty exact state input")?)?; + for state in states { + merged + .merge_from(&exact(&state)?) + .map_err(|error| error.to_string())?; + } + Ok(merged) +} + +/// PromQL counter readouts omit a series with fewer than two samples. Other +/// state/type/range failures remain errors rather than empty results. +pub fn insufficient_counter_samples(state: &dyn AggregateCore, statistic: Statistic) -> bool { + matches!(statistic, Statistic::Rate | Statistic::Increase) + && state + .as_any() + .downcast_ref::() + .is_some_and(|state| state.insufficient_counter_samples(statistic, &None)) +} + +/// Merge already selected exact panes and read one population. `None` means +/// the population is absent from the result: a counter with too few samples, +/// or an empty MIN/MAX. +pub fn exact_readout( + states: impl IntoIterator>, + statistic: Statistic, + range_ms: Option<(i64, i64)>, + key: Option<&KeyByLabelValues>, +) -> Result, String> { + let merged = merge_exact_states(states)?; + if merged.insufficient_counter_samples(statistic, &key.cloned()) { + return Ok(None); + } + merged + .readout(statistic, range_ms, key) + .map_err(|error| error.to_string()) +} + +#[cfg(test)] +mod counter_tests { + use super::*; + use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; + + fn counter(kind: ExactKind, params: ExactParams, keyed: bool) -> ExactAccumulator { + ExactAccumulator::new(SummaryFamilyType::ExactAggregate(kind, params), keyed).unwrap() + } + + // A counter population with a single sample is absent, keyed or not. + #[test] + fn planner_counter_population_omits_insufficient_samples() { + for (kind, params, statistic) in [ + (ExactKind::Rate, ExactParams::Rate, Statistic::Rate), + ( + ExactKind::Increase, + ExactParams::Increase, + Statistic::Increase, + ), + ] { + for keyed in [false, true] { + let mut state = counter(kind.clone(), params.clone(), keyed); + let key = keyed.then(|| KeyByLabelValues::new_with_labels(vec!["checkout".into()])); + state.update(key.as_ref(), 10., 10_000); + assert_eq!( + exact_readout( + [Arc::new(state) as Arc], + statistic, + None, + key.as_ref() + ) + .unwrap(), + None + ); + } + } + } + + // Two ordered samples read a rate; an inverted range and empty input fail. + #[test] + fn sparse_counter_is_absent_but_invalid_ranges_still_fail() { + let mut state = counter(ExactKind::Rate, ExactParams::Rate, false); + state.update(None, 10., 10_000); + let rate = Statistic::Rate; + let one = [Arc::new(state.clone()) as Arc]; + assert_eq!( + exact_readout(one, rate, Some((0, 60_000)), None).unwrap(), + None + ); + state.update(None, 20., 20_000); + let two = || [Arc::new(state.clone()) as Arc]; + assert!(exact_readout(two(), rate, Some((0, 60_000)), None) + .unwrap() + .is_some()); + assert!(exact_readout(two(), rate, Some((60_000, 0)), None).is_err()); + assert!(exact_readout([], rate, Some((0, 60_000)), None).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/runtime/batch_execution.rs b/crates/asap-physical-operators/src/runtime/batch_execution.rs new file mode 100644 index 00000000..f6300440 --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/batch_execution.rs @@ -0,0 +1,210 @@ +//! Execute a bounded in-memory batch through native operators. This is also the +//! bridge for deployments whose boundary values are not yet streaming batches. +use crate::{ + operators::Operator, + plan::PhysicalDag, + runtime::{RunContext, SharedValue}, + values::Batch, + Error, +}; +use futures::{FutureExt, StreamExt}; + +/// Every input is already in memory; the chain contains native operators only. +/// This deliberately does not enter a nested executor when called from a DAG +/// adapter. I/O belongs to source operators in the surrounding execution. +pub fn evaluate_batch( + input: Batch, + operators: Vec, + context: RunContext, +) -> Result>, Error> { + let mut graph = PhysicalDag::default(); + graph.add( + 0, + vec![], + Operator::source(input.schema().clone(), vec![input])?, + )?; + let mut root = 0; + for operator in operators { + graph.add(root + 1, vec![root], operator)?; + root += 1; + } + evaluate_graph(graph, root, context) +} + +/// Bind the ordered in-memory inputs of a native multi-input operator. +pub fn evaluate_inputs( + inputs: Vec, + operator: Operator, + context: RunContext, +) -> Result>, Error> { + let mut graph = PhysicalDag::default(); + let root = inputs.len() as u64; + for (id, input) in inputs.into_iter().enumerate() { + graph.add( + id as u64, + vec![], + Operator::source(input.schema().clone(), vec![input])?, + )?; + } + graph.add(root, (0..root).collect(), operator)?; + evaluate_graph(graph, root, context) +} + +/// Evaluate a native in-memory source, including scalar sources, in the caller's scope. +pub fn evaluate_source( + source: Operator, + context: RunContext, +) -> Result>, Error> { + let mut graph = PhysicalDag::default(); + graph.add(0, vec![], source)?; + evaluate_graph(graph, 0, context) +} + +fn evaluate_graph( + graph: PhysicalDag<'_, Batch, crate::values::Schema>, + root: crate::plan::NodeId, + context: RunContext, +) -> Result>, Error> { + let mut output = graph.execute(&[root], context)?.remove(0); + let mut batches = Vec::new(); + loop { + match output.next().now_or_never() { + Some(Some(Ok(batch))) => batches.push(batch), + Some(Some(Err(error))) => return Err(error), + Some(None) => return Ok(batches), + // Native operators have no I/O sources here. Pending is the + // shared runtime's cooperative yield after a batch quantum. + None => continue, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::{ + operators::Expression, + runtime::{Limits, Scope}, + values::Value, + }; + use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, + }; + use std::sync::Arc; + + // Engine adapters can run the identical native chain from an outer executor. + #[test] + fn same_native_chain_inside_query_and_ingestion_execution() { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }); + for scope in [ + Scope::Query { + evaluation_time_ms: 20, + revision: 1, + }, + Scope::Ingestion { + window_start_ms: 10, + window_end_ms: 20, + revision: 1, + }, + ] { + let batch = Batch::try_new(schema.clone(), vec![vec![Value::Float64(7.)]]).unwrap(); + let negate = Operator::project( + schema.clone(), + vec![( + "value".into(), + Expression::Negate(Box::new(Expression::Column(0))), + )], + ) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let result = futures::executor::block_on(async { + evaluate_batch(batch, vec![negate], context.clone()) + }) + .unwrap(); + assert!(matches!(result[0].rows()[0][0], Value::Float64(-7.))); + let source = Operator::scalar(Value::Float64(9.), DataType::Float64).unwrap(); + let scalar = evaluate_source(source, context).unwrap(); + assert!(matches!(scalar[0].rows()[0][0], Value::Float64(9.))); + } + } + + // Native sources may cross the runtime's cooperative batch quantum. + #[test] + fn in_memory_source_drives_cooperative_yields() { + let schema = Arc::new(SummarySchema { + fields: vec![], + time_index: None, + }); + let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); + let source = Operator::source(schema, vec![batch; 65]).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + assert_eq!(evaluate_source(source, context).unwrap().len(), 65); + } + + // An adapter-held output must retain its parent's reservation after execution. + #[test] + fn returned_batches_keep_their_resource_reservation() { + let schema = Arc::new(SummarySchema { + fields: vec![], + time_index: None, + }); + let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); + let bytes = batch.bytes(); + let source = Operator::source(schema, vec![batch]).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_bytes: bytes, + max_buffered_batches: 1, + }, + ) + .unwrap(); + let held = evaluate_source(source.clone(), context.clone()).unwrap(); + assert_eq!(context.retained_bytes(), bytes); + assert!(evaluate_source(source.clone(), context.clone()).is_err()); + drop(held); + assert_eq!(context.retained_bytes(), 0); + assert!(evaluate_source(source, context).is_ok()); + } + + // A cancelled surrounding execution also prevents its native computation. + #[test] + fn cancellation_is_not_bypassed_by_in_memory_execution() { + let schema = Arc::new(SummarySchema { + fields: vec![], + time_index: None, + }); + let batch = Batch::try_new(schema, vec![vec![]]).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + context.cancel(); + assert!(matches!( + evaluate_batch(batch, vec![], context), + Err(Error::Cancelled) + )); + } +} diff --git a/crates/asap-physical-operators/src/runtime/context.rs b/crates/asap-physical-operators/src/runtime/context.rs new file mode 100644 index 00000000..c145a457 --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/context.rs @@ -0,0 +1,133 @@ +use crate::Error; +use std::{ + cell::{Cell, RefCell}, + rc::Rc, + task::Waker, +}; +/// Scope is part of an execution instance, never mutable state in a reusable plan. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Scope { + Ingestion { + window_start_ms: i64, + window_end_ms: i64, + revision: u64, + }, + Query { + evaluation_time_ms: i64, + revision: u64, + }, +} +#[derive(Clone, Debug)] +pub struct Limits { + pub max_buffered_batches: usize, + pub max_bytes: usize, +} +impl Default for Limits { + fn default() -> Self { + Self { + max_buffered_batches: 8, + max_bytes: 64 * 1024 * 1024, + } + } +} +pub(super) struct Control { + cancelled: Cell, + bytes: Cell, + peak: Cell, + pub(super) limits: Limits, + waiters: RefCell>, +} +#[derive(Clone)] +pub struct RunContext { + pub scope: Scope, + pub(super) control: Rc, +} +impl RunContext { + pub fn new(scope: Scope, limits: Limits) -> Result { + if limits.max_buffered_batches == 0 || limits.max_bytes == 0 { + return Err(Error::Invalid("execution limits must be positive".into())); + } + if matches!(&scope, Scope::Ingestion { window_start_ms, window_end_ms, .. } if window_start_ms > window_end_ms) + { + return Err(Error::Invalid("inverted ingestion window".into())); + } + Ok(Self { + scope, + control: Rc::new(Control { + cancelled: Cell::new(false), + bytes: Cell::new(0), + peak: Cell::new(0), + limits, + waiters: RefCell::new(Vec::new()), + }), + }) + } + pub fn cancel(&self) { + self.control.cancelled.set(true); + for waiter in self.control.waiters.borrow_mut().drain(..) { + waiter.wake(); + } + } + pub fn is_cancelled(&self) -> bool { + self.control.cancelled.get() + } + pub fn retained_bytes(&self) -> usize { + self.control.bytes.get() + } + pub fn peak_bytes(&self) -> usize { + self.control.peak.get() + } + pub fn reserve(&self, bytes: usize) -> Result { + let total = self + .control + .bytes + .get() + .checked_add(bytes) + .ok_or(Error::MemoryLimit)?; + if total > self.control.limits.max_bytes { + return Err(Error::MemoryLimit); + } + self.control.bytes.set(total); + self.control.peak.set(self.control.peak.get().max(total)); + Ok(Reservation { + bytes, + control: Rc::clone(&self.control), + }) + } + pub(super) fn register(&self, waker: &Waker) { + let mut waiters = self.control.waiters.borrow_mut(); + if !waiters.iter().any(|old| old.will_wake(waker)) { + waiters.push(waker.clone()); + } + } +} +pub struct Reservation { + bytes: usize, + pub(super) control: Rc, +} +impl Reservation { + /// Adjust an operator-owned allocation without accumulating bookkeeping entries. + pub fn resize(&mut self, bytes: usize) -> Result<(), Error> { + let total = self + .control + .bytes + .get() + .checked_sub(self.bytes) + .and_then(|total| total.checked_add(bytes)) + .ok_or(Error::MemoryLimit)?; + if total > self.control.limits.max_bytes { + return Err(Error::MemoryLimit); + } + self.control.bytes.set(total); + self.control.peak.set(self.control.peak.get().max(total)); + self.bytes = bytes; + Ok(()) + } +} +impl Drop for Reservation { + fn drop(&mut self) { + self.control + .bytes + .set(self.control.bytes.get().saturating_sub(self.bytes)); + } +} diff --git a/crates/asap-physical-operators/src/runtime/cooperative.rs b/crates/asap-physical-operators/src/runtime/cooperative.rs new file mode 100644 index 00000000..fae869d6 --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/cooperative.rs @@ -0,0 +1,40 @@ +//! Worker-local CPU loops yield so other consumers and cancellation can progress. +use super::RunContext; +use crate::Error; +use std::task::Poll; + +pub(crate) struct Cooperative { + context: RunContext, + remaining: usize, +} +impl Cooperative { + pub(crate) fn new(context: &RunContext) -> Self { + Self { + context: context.clone(), + remaining: 1024, + } + } + pub(crate) async fn checkpoint(&mut self) -> Result<(), Error> { + if self.context.is_cancelled() { + return Err(Error::Cancelled); + } + self.remaining -= 1; + if self.remaining == 0 { + self.remaining = 1024; + let mut yielded = false; + futures::future::poll_fn(|cx| { + if self.context.is_cancelled() { + return Poll::Ready(Err(Error::Cancelled)); + } + if yielded { + return Poll::Ready(Ok(())); + } + yielded = true; + cx.waker().wake_by_ref(); + Poll::Pending + }) + .await?; + } + Ok(()) + } +} diff --git a/crates/asap-physical-operators/src/runtime/mod.rs b/crates/asap-physical-operators/src/runtime/mod.rs new file mode 100644 index 00000000..f72a57ef --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/mod.rs @@ -0,0 +1,272 @@ +//! Per-run producer sharing, streams, backpressure and resource ownership. +use crate::{ + plan::{NodeId, PhysicalDag}, + Error, +}; +use futures::{stream::LocalBoxStream, Stream}; +use std::{ + cell::RefCell, + collections::{BTreeMap, VecDeque}, + fmt::Debug, + pin::Pin, + rc::Rc, + sync::Arc, + task::{Context, Poll, Waker}, +}; +mod context; +pub use context::{Limits, Reservation, RunContext, Scope}; +pub type OutputStream<'a, V> = LocalBoxStream<'a, Result>; +/// An output owns its memory reservation even after it leaves the DAG's queue. +pub struct SharedValue { + value: Arc, + _reservation: Rc, +} +impl Clone for SharedValue { + fn clone(&self) -> Self { + Self { + value: Arc::clone(&self.value), + _reservation: Rc::clone(&self._reservation), + } + } +} +impl std::ops::Deref for SharedValue { + type Target = V; + fn deref(&self) -> &V { + &self.value + } +} +impl SharedValue { + pub fn value(&self) -> &V { + &self.value + } +} + +pub(crate) fn execute<'r, V: 'r, S: Clone + PartialEq + Debug + 'r>( + dag: &'r PhysicalDag<'_, V, S>, + roots: &[NodeId], + context: RunContext, +) -> Result>, Error> { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + dag.validate(roots)?; + let mut pending = roots.to_vec(); + let mut visited = std::collections::BTreeSet::new(); + while let Some(id) = pending.pop() { + if visited.insert(id) { + let node = &dag.nodes[&id]; + node.operator.validate_context(&context)?; + pending.extend(node.inputs.iter().copied()); + } + } + fn build<'r, V: 'r, S: 'r>( + dag: &'r PhysicalDag<'_, V, S>, + id: NodeId, + context: &RunContext, + states: &mut BTreeMap>>>, + ) -> Result>>, Error> { + if let Some(state) = states.get(&id) { + return Ok(Rc::clone(state)); + } + let node = &dag.nodes[&id]; + let mut inputs = Vec::new(); + for &child in &node.inputs { + inputs.push(Input::subscribe(build(dag, child, context, states)?)); + } + let stream = node + .operator + .start(inputs, context.clone()) + .map_err(|source| Error::AtNode { + node: id, + operation: node.operator.name().into(), + source: Box::new(source), + })?; + let op = node.operator.as_ref(); + let state = Rc::new(RefCell::new(Producer { + stream: Some(stream), + node: id, + operation: node.operator.name().into(), + size: Box::new(move |value| op.output_bytes(value)), + context: context.clone(), + queue: VecDeque::new(), + base: 0, + next_reader: 0, + batches_polled: 0, + readers: BTreeMap::new(), + waiters: BTreeMap::new(), + finished: false, + failure: None, + })); + states.insert(id, Rc::clone(&state)); + Ok(state) + } + let mut states = BTreeMap::new(); + roots + .iter() + .map(|&id| build(dag, id, &context, &mut states).map(Input::subscribe)) + .collect() +} +struct Producer<'a, V> { + node: NodeId, + operation: String, + stream: Option>, + size: Box usize + 'a>, + context: RunContext, + queue: VecDeque>, + base: u64, + next_reader: u64, + batches_polled: usize, + readers: BTreeMap, + waiters: BTreeMap, + finished: bool, + failure: Option, +} +impl Producer<'_, V> { + fn trim(&mut self) { + let minimum = self + .readers + .values() + .copied() + .min() + .unwrap_or(self.base + self.queue.len() as u64); + while self.base < minimum { + self.queue.pop_front(); + self.base += 1; + } + for (_, waker) in std::mem::take(&mut self.waiters) { + waker.wake(); + } + if self.readers.is_empty() { + self.stream = None; + self.queue.clear(); + } + } +} +pub struct Input<'a, V> { + producer: Rc>>, + reader: u64, + done: bool, +} +impl<'a, V> Input<'a, V> { + fn subscribe(producer: Rc>>) -> Self { + let reader = { + let mut state = producer.borrow_mut(); + let id = state.next_reader; + state.next_reader += 1; + let base = state.base; + state.readers.insert(id, base); + id + }; + Self { + producer, + reader, + done: false, + } + } +} +impl Drop for Input<'_, V> { + fn drop(&mut self) { + let mut state = self.producer.borrow_mut(); + state.readers.remove(&self.reader); + state.waiters.remove(&self.reader); + state.trim(); + } +} +impl Stream for Input<'_, V> { + type Item = Result, Error>; + fn poll_next(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + let this = self.get_mut(); + if this.done { + return Poll::Ready(None); + } + let mut state = this.producer.borrow_mut(); + state.context.register(cx.waker()); + if state.context.is_cancelled() { + state.failure = Some(Error::Cancelled); + state.finished = true; + state.stream = None; + state.queue.clear(); + } + let position = state.readers[&this.reader]; + let index = (position - state.base) as usize; + if let Some(value) = state.queue.get(index).cloned() { + state.readers.insert(this.reader, position + 1); + state.trim(); + return Poll::Ready(Some(Ok(value))); + } + if state.finished { + this.done = true; + state.readers.remove(&this.reader); + let failure = state.failure.clone(); + state.trim(); + return Poll::Ready(failure.map(Err)); + } + state.waiters.insert(this.reader, cx.waker().clone()); + if state.queue.len() >= state.context.control.limits.max_buffered_batches { + return Poll::Pending; + } + // Always-ready sources must still give cancellation and other roots a turn. + if state.batches_polled >= 32 { + state.batches_polled = 0; + cx.waker().wake_by_ref(); + return Poll::Pending; + } + let polled = state + .stream + .as_mut() + .expect("unfinished producer") + .as_mut() + .poll_next(cx); + if matches!(&polled, Poll::Ready(Some(Ok(_)))) { + state.batches_polled += 1; + } + match polled { + Poll::Pending => Poll::Pending, + Poll::Ready(Some(Ok(value))) => match state.context.reserve((state.size)(&value)) { + Ok(reservation) => { + let value = SharedValue { + value: Arc::new(value), + _reservation: Rc::new(reservation), + }; + state.queue.push_back(value.clone()); + state.readers.insert(this.reader, position + 1); + state.trim(); + Poll::Ready(Some(Ok(value))) + } + Err(error) => { + state.failure = Some(error.clone()); + state.finished = true; + state.stream = None; + this.done = true; + state.readers.remove(&this.reader); + state.trim(); + Poll::Ready(Some(Err(error))) + } + }, + Poll::Ready(result) => { + let error = result.and_then(Result::err).map(|source| match source { + Error::AtNode { .. } | Error::Cancelled | Error::MemoryLimit => source, + source => Error::AtNode { + node: state.node, + operation: state.operation.clone(), + source: Box::new(source), + }, + }); + state.failure = error.clone(); + state.finished = true; + state.stream = None; + this.done = true; + state.readers.remove(&this.reader); + state.trim(); + Poll::Ready(error.map(Err)) + } + } + } +} + +pub mod batch_execution; +#[cfg(test)] +mod tests; + +mod cooperative; +pub(crate) use cooperative::Cooperative; diff --git a/crates/asap-physical-operators/src/runtime/tests.rs b/crates/asap-physical-operators/src/runtime/tests.rs new file mode 100644 index 00000000..f683041c --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/tests.rs @@ -0,0 +1,262 @@ +use super::*; +use crate::plan::PhysicalOperator; +use futures::{executor::block_on, stream, StreamExt}; +use std::cell::Cell; + +struct Source { + starts: Rc>, + polls: Rc>, + fail: bool, + end: u64, +} +impl PhysicalOperator for Source { + fn name(&self) -> &str { + "CountingSource" + } + fn input_schemas(&self) -> Vec<()> { + vec![] + } + fn output_schema(&self) {} + fn output_bytes(&self, _: &u64) -> usize { + 8 + } + fn start<'a>( + &'a self, + _: Vec>, + _: RunContext, + ) -> Result, Error> { + self.starts.set(self.starts.get() + 1); + Ok(stream::iter(0..self.end) + .map(move |n| { + self.polls.set(self.polls.get() + 1); + if self.fail && n == 1 { + Err(Error::Operator("source failure".into())) + } else { + Ok(n) + } + }) + .boxed_local()) + } +} +struct Identity; +impl PhysicalOperator for Identity { + fn name(&self) -> &str { + "Identity" + } + fn input_schemas(&self) -> Vec<()> { + vec![()] + } + fn output_schema(&self) {} + fn output_bytes(&self, _: &u64) -> usize { + 8 + } + fn start<'a>( + &'a self, + mut inputs: Vec>, + _: RunContext, + ) -> Result, Error> { + Ok(inputs + .remove(0) + .map(|value| value.map(|v| *v)) + .boxed_local()) + } +} +fn context() -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 100, + revision: 1, + }, + Limits { + max_buffered_batches: 1, + max_bytes: 1024, + }, + ) + .unwrap() +} +fn source(fail: bool) -> (Source, Rc>, Rc>) { + let starts = Rc::new(Cell::new(0)); + let polls = Rc::new(Cell::new(0)); + ( + Source { + starts: starts.clone(), + polls: polls.clone(), + fail, + end: 4, + }, + starts, + polls, + ) +} + +// A shared producer runs once, and the slow reader bounds producer progress. +#[test] +fn shared_source_backpressure_and_reader_drop() { + let (source, starts, polls) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let context = context(); + let mut readers = dag.execute(&[0, 0], context.clone()).unwrap(); + let mut slow = readers.pop().unwrap(); + let mut fast = readers.pop().unwrap(); + assert_eq!(starts.get(), 1); + let first = block_on(fast.next()).unwrap().unwrap(); + assert_eq!(*first, 0); + let mut cx = Context::from_waker(futures::task::noop_waker_ref()); + assert!(Pin::new(&mut fast).poll_next(&mut cx).is_pending()); + assert_eq!(polls.get(), 1); + let same = block_on(slow.next()).unwrap().unwrap(); + assert!(Arc::ptr_eq(&first.value, &same.value)); + drop(same); + drop(first); + assert_eq!(context.retained_bytes(), 0); + assert_eq!(*block_on(fast.next()).unwrap().unwrap(), 1); + drop(slow); + assert_eq!(*block_on(fast.next()).unwrap().unwrap(), 2); + assert_eq!(*block_on(fast.next()).unwrap().unwrap(), 3); + assert!(block_on(fast.next()).is_none()); + assert_eq!(polls.get(), 4); + drop(fast); + assert_eq!(context.retained_bytes(), 0); +} + +// Independent branches consume a common node concurrently without duplicate work. +#[test] +fn diamond_and_run_isolation() { + let (source, starts, polls) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + dag.add(1, vec![0], Identity).unwrap(); + dag.add(2, vec![0], Identity).unwrap(); + for _ in 0..2 { + let mut outputs = dag.execute(&[1, 2], context()).unwrap(); + let a = outputs.pop().unwrap(); + let b = outputs.pop().unwrap(); + let (a, b) = + block_on(async { futures::join!(a.collect::>(), b.collect::>()) }); + assert_eq!( + a.iter().map(|v| **v.as_ref().unwrap()).collect::>(), + vec![0, 1, 2, 3] + ); + assert_eq!( + b.iter().map(|v| **v.as_ref().unwrap()).collect::>(), + vec![0, 1, 2, 3] + ); + } + assert_eq!(starts.get(), 2); + assert_eq!(polls.get(), 8); +} + +// Failure reaches every subscriber; cancellation stops further producer work. +#[test] +fn broadcast_error_and_cancel() { + let (source, _, polls) = source(true); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let mut outputs = dag.execute(&[0, 0], context()).unwrap(); + let a = outputs.pop().unwrap(); + let b = outputs.pop().unwrap(); + let (a, b) = block_on(async { futures::join!(a.collect::>(), b.collect::>()) }); + for values in [a, b] { + assert_eq!(values.len(), 2); + assert!(matches!(values[1], Err(Error::AtNode { node: 0, .. }))); + } + assert_eq!(polls.get(), 2); + let run = context(); + let mut output = dag.execute(&[0], run.clone()).unwrap().remove(0); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + assert!(block_on(output.next()).is_none()); + assert_eq!(polls.get(), 2); +} + +// Retaining a consumer output retains its budget lease after queue eviction. +#[test] +fn retained_outputs_count_against_budget() { + let (source, _, _) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let run = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_buffered_batches: 1, + max_bytes: 8, + }, + ) + .unwrap(); + let mut input = dag.execute(&[0], run.clone()).unwrap().remove(0); + let held = block_on(input.next()).unwrap().unwrap(); + assert_eq!(run.retained_bytes(), 8); + assert!(matches!( + block_on(input.next()), + Some(Err(Error::MemoryLimit)) + )); + drop(input); + assert_eq!(run.retained_bytes(), 8); + drop(held); + assert_eq!(run.retained_bytes(), 0); +} + +// Invalid graphs fail before even starting a source. +#[test] +fn invalid_graphs_do_not_start_sources() { + let (source, starts, _) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + dag.add(1, vec![2], Identity).unwrap(); + dag.add(2, vec![1], Identity).unwrap(); + assert!(dag.execute(&[0, 1], context()).is_err()); + assert_eq!(starts.get(), 0); + let mut missing = PhysicalDag::default(); + missing.add(1, vec![9], Identity).unwrap(); + assert!(missing.validate(&[1]).is_err()); + let mut arity = PhysicalDag::default(); + arity.add(1, vec![], Identity).unwrap(); + assert!(arity.validate(&[1]).is_err()); +} + +// An always-ready source must yield so cancellation can be polled on this worker. +#[test] +fn ready_sources_cooperate_with_cancellation() { + let (mut source, _, polls) = source(false); + source.end = 10_000; + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let context = context(); + let mut input = dag.execute(&[0], context.clone()).unwrap().remove(0); + block_on(async { + let drain = async { + while let Some(result) = input.next().await { + if let Err(error) = result { + assert_eq!(error, Error::Cancelled); + return; + } + } + panic!("source completed without yielding"); + }; + let cancel = async { + context.cancel(); + }; + futures::join!(drain, cancel); + }); + assert_eq!(polls.get(), 32); + assert_eq!(context.retained_bytes(), 0); +} + +// Cached shorter paths must not hide an over-deep path through shared nodes. +#[test] +fn depth_limit_covers_shared_paths() { + let (source, _, _) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + for id in 1..129 { + dag.add(id, vec![id - 1], Identity).unwrap(); + } + assert!(dag.validate(&(0..129).collect::>()).is_err()); +} diff --git a/crates/asap-physical-operators/src/sources/memory.rs b/crates/asap-physical-operators/src/sources/memory.rs new file mode 100644 index 00000000..7f77c54f --- /dev/null +++ b/crates/asap-physical-operators/src/sources/memory.rs @@ -0,0 +1,44 @@ +use super::*; +/// Immutable in-memory raw data. The connector owns the resident input; each +/// cursor clones only the next requested batch, not the entire data set. +pub struct MemorySource { + schema: Schema, + batches: Vec, +} +impl MemorySource { + pub fn new(schema: Schema, batches: Vec) -> Result { + crate::values::validate_schema(&schema)?; + if schema + .fields + .iter() + .any(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + { + return Err(Error::Invalid( + "raw source cannot contain summary states".into(), + )); + } + if batches.iter().any(|batch| batch.schema() != &schema) { + return Err(Error::Invalid("memory source batch schema mismatch".into())); + } + Ok(Self { schema, batches }) + } +} +impl RawSource for MemorySource { + fn boundedness(&self) -> crate::plan::Boundedness { + crate::plan::Boundedness::Bounded + } + fn schema(&self) -> Schema { + self.schema.clone() + } + fn scan(&self, context: RunContext) -> Result, Error> { + Ok(stream::iter(self.batches.iter()) + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let _allocation = context.reserve(batch.bytes())?; + Ok(batch.clone()) + }) + .boxed_local()) + } +} diff --git a/crates/asap-physical-operators/src/sources/mod.rs b/crates/asap-physical-operators/src/sources/mod.rs new file mode 100644 index 00000000..4da778b0 --- /dev/null +++ b/crates/asap-physical-operators/src/sources/mod.rs @@ -0,0 +1,187 @@ +//! Raw data access. Connectors provide rows; Scan owns Planner predicate semantics. +use crate::{ + expressions::CompiledExpression, + plan::PhysicalOperator, + runtime::{Input, OutputStream, RunContext}, + values::{Batch, Schema, Value}, + Error, +}; +use futures::{stream, StreamExt}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{DataType, QueryExpr, Source}, +}; +use std::sync::Arc; + +/// A bound data source. Metadata must be stable for the lifetime of the binding. +/// Each scan opens an independent cursor. Connectors return raw, unfiltered rows +/// and must honor cancellation and bound their own I/O buffers. Dropping a cursor +/// must release its resources. A connector error is never an empty successful scan. +pub trait RawSource { + fn schema(&self) -> Schema; + /// Declare a finite snapshot/window explicitly; execution scope alone does not bound a cursor. + fn boundedness(&self) -> crate::plan::Boundedness { + crate::plan::Boundedness::Unknown + } + fn scan(&self, context: RunContext) -> Result, Error>; +} + +/// Explicit source identities; no implicit network discovery or fallback. +#[derive(Default)] +pub struct DataSources { + sources: Vec<(Source, Arc)>, +} +impl DataSources { + pub fn register(&mut self, identity: Source, source: Arc) -> Result<(), Error> { + if self.sources.iter().any(|(key, _)| key == &identity) { + return Err(Error::Invalid("duplicate data source".into())); + } + crate::values::validate_schema(&source.schema())?; + self.sources.push((identity, source)); + Ok(()) + } + pub fn bind(&self, expression: &QueryExpr) -> Result { + let QueryExpr::Scan { + source, + predicates, + schema, + } = expression + else { + return Err(Error::Invalid( + "raw Scan requires a Planner Scan leaf".into(), + )); + }; + let output = Arc::new(SummarySchema { + fields: schema + .columns + .iter() + .map(|column| SummaryField { + name: column.name.clone(), + dtype: SummaryFamilyType::Plain(column.dtype.clone()), + nullable: column.nullable, + }) + .collect(), + time_index: schema.time_index, + }); + crate::values::validate_schema(&output)?; + let reader = self + .sources + .iter() + .find(|(key, _)| key == source) + .map(|(_, reader)| reader.clone()) + .ok_or_else(|| Error::Invalid(format!("unbound raw source: {source:?}")))?; + if reader.schema() != output { + return Err(Error::Invalid( + "raw source differs from Planner Scan schema".into(), + )); + } + let predicates = predicates + .iter() + .map(|predicate| { + let predicate = CompiledExpression::compile(&predicate.0, &output)?; + if predicate.dtype().0 != DataType::Bool { + return Err(Error::Invalid("Scan predicate must be boolean".into())); + } + Ok(predicate) + }) + .collect::, Error>>()?; + Ok(Scan { + reader, + output, + predicates, + }) + } +} + +pub struct Scan { + reader: Arc, + output: Schema, + predicates: Vec, +} +impl PhysicalOperator for Scan { + fn properties(&self, _: &[crate::plan::PlanProperties]) -> crate::plan::PlanProperties { + crate::plan::PlanProperties { + boundedness: self.reader.boundedness(), + emission: crate::plan::Emission::Incremental, + } + } + + fn name(&self) -> &str { + "Scan" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, batch: &Batch) -> usize { + batch.bytes() + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + if !inputs.is_empty() { + return Err(Error::Invalid("Scan cannot have inputs".into())); + } + if context.is_cancelled() { + return Err(Error::Cancelled); + } + // Opening is lazy: validation and construction of a run perform no I/O. + let opening = context.clone(); + let stream = stream::once(async move { + if opening.is_cancelled() { + return Err(Error::Cancelled); + } + self.reader.scan(opening) + }); + use futures::TryStreamExt; + Ok(stream + .try_flatten() + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + if batch.schema() != &self.output { + return Err(Error::Invalid( + "connector returned a different Scan schema".into(), + )); + } + if self.predicates.is_empty() { + return Ok(batch); + } + let _workspace = + context.reserve(batch.bytes().checked_mul(2).ok_or(Error::MemoryLimit)?)?; + let mut rows = Vec::new(); + for row in batch.rows() { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let mut keep = true; + for predicate in &self.predicates { + match predicate.evaluate(row)? { + Value::Bool(true) => {} + Value::Bool(false) | Value::Null => { + keep = false; + break; + } + _ => { + return Err(Error::Invalid("Scan predicate is not boolean".into())) + } + } + } + if keep { + rows.push(row.clone()); + } + } + Batch::try_new(self.output.clone(), rows) + }) + .boxed_local()) + } +} + +mod memory; +pub use memory::MemorySource; diff --git a/crates/asap-physical-operators/src/statistic.rs b/crates/asap-physical-operators/src/statistic.rs new file mode 100644 index 00000000..7053308d --- /dev/null +++ b/crates/asap-physical-operators/src/statistic.rs @@ -0,0 +1,67 @@ +use std::{fmt, str::FromStr}; +use tracing::debug; +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] +pub enum Statistic { + Count, + Sum, + Cardinality, + FrequencyL2, + FrequencyEntropy, + Increase, + Rate, + Min, + Max, + Quantile, + Topk, +} + +impl fmt::Display for Statistic { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + debug!("Formatting Statistic: {:?}", self); + match self { + Statistic::Count => write!(f, "count"), + Statistic::Sum => write!(f, "sum"), + Statistic::Cardinality => write!(f, "cardinality"), + Statistic::FrequencyL2 => write!(f, "frequency_l2"), + Statistic::FrequencyEntropy => write!(f, "frequency_entropy"), + Statistic::Increase => write!(f, "increase"), + Statistic::Rate => write!(f, "rate"), + Statistic::Min => write!(f, "min"), + Statistic::Max => write!(f, "max"), + Statistic::Quantile => write!(f, "quantile"), + Statistic::Topk => write!(f, "topk"), + } + } +} + +#[allow(clippy::should_implement_trait)] +impl Statistic { + pub fn from_str(s: &str) -> Option { + debug!("Parsing Statistic from string: {}", s); + match s.to_lowercase().as_str() { + "count" => Some(Statistic::Count), + "sum" => Some(Statistic::Sum), + "cardinality" => Some(Statistic::Cardinality), + "frequency_l2" => Some(Statistic::FrequencyL2), + "frequency_entropy" => Some(Statistic::FrequencyEntropy), + "increase" => Some(Statistic::Increase), + "rate" => Some(Statistic::Rate), + "min" => Some(Statistic::Min), + "max" => Some(Statistic::Max), + "quantile" => Some(Statistic::Quantile), + "topk" => Some(Statistic::Topk), + _ => None, + } + } +} + +impl FromStr for Statistic { + type Err = (); + + /// Parse a statistic from a string (case-insensitive). + /// Use `s.parse::()` or `Statistic::from_str(s)`. + fn from_str(s: &str) -> Result { + debug!("FromStr trait parsing Statistic: {}", s); + Statistic::from_str(s).ok_or(()) + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs new file mode 100644 index 00000000..5a78fc48 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs @@ -0,0 +1,76 @@ +//! Count-Min Sketch frequency summary over `asap_sketchlib::CountMinSketch`. +use crate::{AggregateCore, KernelError, KeyByLabelValues}; +use asap_sketchlib::CountMinSketch; + +#[derive(Debug, Clone)] +pub struct CountMinSketchAccumulator { + pub inner: CountMinSketch, +} + +impl CountMinSketchAccumulator { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + inner: CountMinSketch::new(row_num, col_num), + } + } + + /// Estimated frequency of one item. + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + self.inner.estimate(&key.to_semicolon_str()) + } +} + +impl AggregateCore for CountMinSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("Count-Min Sketch merges only with Count-Min Sketch")?; + Ok(Box::new(Self { + inner: CountMinSketch::merge_refs(&[&self.inner, &other.inner])?, + })) + } + + fn approx_memory_bytes(&self) -> usize { + 16 * 1024 + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Merged point counts add item frequencies and never underestimate. + #[test] + fn merged_point_counts_add() { + let (mut a, mut b) = ( + CountMinSketchAccumulator::new(3, 128), + CountMinSketchAccumulator::new(3, 128), + ); + let key = KeyByLabelValues::new_with_labels(vec!["checkout".into()]); + a.inner.update(&key.to_semicolon_str(), 2.0); + b.inner.update(&key.to_semicolon_str(), 3.0); + let merged = a.merge_with(&b).unwrap(); + let merged = merged + .as_any() + .downcast_ref::() + .unwrap(); + assert!(merged.query_key(&key) >= 5.0); + } + + // Merge rejects a different summary family. + #[test] + fn rejects_foreign_merge() { + let cms = CountMinSketchAccumulator::new(3, 128); + let kll = crate::summary_kernels::DatasketchesKLLAccumulator::new(200); + assert!(cms.merge_with(&kll).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs new file mode 100644 index 00000000..da07aeee --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs @@ -0,0 +1,234 @@ +use crate::{AggregateCore, KeyByLabelValues}; +use asap_sketchlib::CountMinSketchWithHeap; + +/// Count-Min Sketch with a top-k heap over `asap_sketchlib::CountMinSketchWithHeap`. +#[derive(Debug, Clone)] +pub struct CountMinSketchWithHeapAccumulator { + pub inner: CountMinSketchWithHeap, +} + +impl CountMinSketchWithHeapAccumulator { + pub fn new(row_num: usize, col_num: usize, heap_size: usize) -> Self { + Self { + inner: CountMinSketchWithHeap::new(row_num, col_num, heap_size), + } + } + + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + let key_string = key.labels.join(";"); + self.inner.estimate(&key_string) + } + + /// VALUE-WEIGHTED heavy-hitter update (FIX: CountSketch/CMS topk + /// recall-0). The default ingest path inserts `+1` per occurrence keyed + /// by the raw `item`, so the heap ranks groups by OCCURRENCE COUNT — the + /// wrong answer for `topk(k, sum by (label) (metric))`, which asks for + /// the top groups by SUM OF VALUE. This update adds the sample `value` + /// (not `+1`) into both the CMS matrix and the top-k heap, keyed by the + /// GROUP LABEL (e.g. the `host` / `zone` value), so the heap's ranking is + /// by summed value. Repeated calls for the same `group_label` accumulate, + /// so after folding a window the heap holds Σvalue per group. + /// + /// Delegates to the library's value-weighted `CountMinSketchWithHeap:: + /// update(key, value)` (`sketchlib_cms_heap_update` → `insert_many(key, + /// round(value))`), which is the "separate update path" the evaluation + /// plan (Fig 3c) called for. + pub fn insert_value(&mut self, group_label: &str, value: f64) { + self.inner.update(group_label, value); + } + + /// Read the top-`k` GROUPS ranked by summed VALUE (descending), keyed by + /// the group label. Pairs with [`Self::insert_value`]: the heap built by + /// value-weighted updates ranks by Σvalue, so this returns the + /// value-weighted top-k (not the occurrence-count top-k the raw `item` + /// heap would give). Sorted descending by value; ties broken by key for + /// determinism; truncated to `k`. + pub fn topk_by_value(&self, k: usize) -> Vec<(String, f64)> { + let mut items: Vec<(String, f64)> = self + .inner + .topk_heap_items() + .into_iter() + .map(|it| (it.key, it.value)) + .collect(); + items.sort_by(|a, b| { + b.1.partial_cmp(&a.1) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.0.cmp(&b.0)) + }); + items.truncate(k); + items + } + + /// Get all keys from the top-k heap. + pub fn get_topk_keys(&self) -> Vec { + self.inner + .topk_heap_items() + .iter() + .map(|item| { + let labels: Vec = item.key.split(';').map(|s| s.to_string()).collect(); + KeyByLabelValues { labels } + }) + .collect() + } +} + +impl AggregateCore for CountMinSketchWithHeapAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other_cms = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountMinSketchWithHeapAccumulator")?; + + let mut merged = self.clone(); + merged.inner.merge(&other_cms.inner)?; + Ok(Box::new(merged)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_count_min_sketch_with_heap_creation() { + let cms = CountMinSketchWithHeapAccumulator::new(4, 1000, 20); + assert_eq!(cms.inner.rows(), 4); + assert_eq!(cms.inner.cols(), 1000); + assert_eq!(cms.inner.heap_size, 20); + assert_eq!(cms.inner.topk_heap_items().len(), 0); + } + + #[test] + fn test_get_topk_keys() { + let mut cms = CountMinSketchWithHeapAccumulator::new(2, 3, 5); + cms.inner.update("label1;label2", 100.0); + cms.inner.update("label3;label4", 50.0); + + let keys = cms.get_topk_keys(); + assert_eq!(keys.len(), 2); + // Heap order is not part of the contract; compare as a set. + let label_sets: std::collections::HashSet<_> = + keys.iter().map(|k| k.labels.clone()).collect(); + assert!(label_sets.contains(&vec!["label1".to_string(), "label2".to_string()])); + assert!(label_sets.contains(&vec!["label3".to_string(), "label4".to_string()])); + } + + // ---------------------------------------------------------------- + // FIX 1 — VALUE-WEIGHTED top-k (recall 0 → correct). + // + // `topk(k, sum by (host) (cpu_load))` asks for the top-k hosts by + // SUM OF VALUE. The heavy-hitter heap built by the default `+1`-per- + // occurrence update ranks by COUNT keyed by `item`, so its recall + // against the value-weighted ground truth is 0 when the busiest host + // (most samples) is NOT the heaviest host (largest Σvalue). + // `insert_value(group_label, value)` adds the sample VALUE keyed by the + // GROUP LABEL, so `topk_by_value` ranks by Σvalue — correct recall. + // ---------------------------------------------------------------- + + /// Crafted adversarial dataset: the host with the MOST samples + /// (`h_chatty`, 100 tiny samples) is NOT the host with the largest + /// value-sum (`h_heavy`, a handful of huge samples). A COUNT-ranked + /// heap would surface `h_chatty`; the value-weighted top-k must surface + /// the true heavy hitters by Σvalue, giving recall 1.0 against the + /// ground-truth top-k-by-value-sum. + #[test] + fn value_weighted_topk_has_full_recall_vs_count_topk() { + // (host, per-sample value, sample count) → true Σvalue: + // h_heavy : 1000 × 3 = 3000 (few samples, huge value) + // h_mid : 200 × 5 = 1000 + // h_small : 50 × 6 = 300 + // h_chatty: 1 × 100 = 100 (MOST samples, tiny value) + let data: &[(&str, f64, usize)] = &[ + ("h_heavy", 1000.0, 3), + ("h_mid", 200.0, 5), + ("h_small", 50.0, 6), + ("h_chatty", 1.0, 100), + ]; + + // Wide CMS + heap large enough to hold every group exactly (4 groups) + // so the estimate equals the true Σvalue with no hash collisions. + let mut acc = CountMinSketchWithHeapAccumulator::new(5, 4096, 16); + let mut truth: std::collections::HashMap<&str, f64> = std::collections::HashMap::new(); + for (host, value, count) in data { + for _ in 0..*count { + acc.insert_value(host, *value); + } + *truth.entry(*host).or_insert(0.0) += value * (*count as f64); + } + + // Ground-truth top-2 by value-sum: h_heavy (3000), h_mid (1000). + let mut truth_ranked: Vec<(&str, f64)> = truth.into_iter().collect(); + truth_ranked.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap()); + let truth_top2: std::collections::HashSet<&str> = + truth_ranked.iter().take(2).map(|(k, _)| *k).collect(); + assert!( + truth_top2.contains("h_heavy") && truth_top2.contains("h_mid"), + "ground-truth top-2 by value-sum should be h_heavy + h_mid" + ); + + // Value-weighted top-2 from the heap. + let got = acc.topk_by_value(2); + assert_eq!(got.len(), 2, "k=2 → two groups: {got:?}"); + let got_keys: std::collections::HashSet<&str> = + got.iter().map(|(k, _)| k.as_str()).collect(); + + // RECALL = |got ∩ truth| / |truth| must be 1.0. + let hits = got_keys.intersection(&truth_top2).count(); + let recall = hits as f64 / truth_top2.len() as f64; + assert_eq!( + recall, 1.0, + "value-weighted top-k recall must be 1.0 (count-ranked heap would \ + surface h_chatty and miss h_heavy → recall < 1): got={got:?}" + ); + + // The busiest-by-count host (h_chatty) must NOT be in the top-2, + // proving we rank by value-sum, not occurrence count. + assert!( + !got_keys.contains("h_chatty"), + "h_chatty (most samples, smallest value-sum) must be excluded: {got:?}" + ); + + // Estimates are exact here (no collisions, heap holds all groups): + // top-1 must be h_heavy with Σvalue 3000. + assert_eq!(got[0].0, "h_heavy"); + assert!( + (got[0].1 - 3000.0).abs() < 1e-6, + "h_heavy value-sum estimate ≈ 3000, got {}", + got[0].1 + ); + assert_eq!(got[1].0, "h_mid"); + assert!( + (got[1].1 - 1000.0).abs() < 1e-6, + "h_mid value-sum estimate ≈ 1000, got {}", + got[1].1 + ); + } + + /// A single value-weighted insert must put the full value (not +1) into + /// the heap, and repeated inserts for the same group must accumulate. + #[test] + fn insert_value_accumulates_summed_value_in_heap() { + let mut acc = CountMinSketchWithHeapAccumulator::new(4, 1024, 8); + acc.insert_value("g", 10.0); + acc.insert_value("g", 25.0); + let top = acc.topk_by_value(1); + assert_eq!(top.len(), 1); + assert_eq!(top[0].0, "g"); + assert!( + (top[0].1 - 35.0).abs() < 1e-6, + "summed value should be 35 (10+25), got {}", + top[0].1 + ); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs new file mode 100644 index 00000000..2728b5f2 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs @@ -0,0 +1,96 @@ +//! CountSketch accumulator backed by `asap_sketchlib::CountSketch`. +//! +//! Per-key queries delegate to sketchlib's median-of-signed-rows estimator. +//! Top-k requires the separate heap-bearing accumulator. + +use crate::{AggregateCore, KeyByLabelValues}; +use asap_sketchlib::CountSketch; + +/// Count Sketch accumulator — inner matrix of signed counts. +#[derive(Debug, Clone)] +pub struct CountSketchAccumulator { + pub inner: CountSketch, +} + +impl CountSketchAccumulator { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + inner: CountSketch::new(row_num, col_num), + } + } + + /// Median-of-signed-rows point estimate for `key`, via + /// `asap_sketchlib::CountSketch::estimate`. + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + self.inner.estimate(&key.to_semicolon_str()) + } +} + +impl AggregateCore for CountSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other_cs = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountSketchAccumulator")?; + + let merged_inner = CountSketch::merge_refs(&[&self.inner, &other_cs.inner])?; + Ok(Box::new(Self { + inner: merged_inner, + })) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_query_key_uses_real_sketchlib_estimator() { + // `query_key` must match sketchlib's estimator and hash specification. + let mut cs = CountSketchAccumulator::new(4, 1000); + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + cs.inner.update(&key.to_semicolon_str(), 10.0); + assert_eq!( + cs.query_key(&key), + cs.inner.estimate(&key.to_semicolon_str()) + ); + } + + #[test] + fn test_aggregate_core_merge_matches_matrix_add() { + let a = CountSketchAccumulator { + inner: CountSketch::from_legacy_matrix(vec![vec![1.0, -2.0], vec![3.0, -4.0]], 2, 2), + }; + let b = CountSketchAccumulator { + inner: CountSketch::from_legacy_matrix(vec![vec![-1.0, 2.0], vec![-3.0, 4.0]], 2, 2), + }; + let merged_box = a.merge_with(&b).expect("merge ok"); + let merged = merged_box + .as_any() + .downcast_ref::() + .expect("downcast ok"); + let m = merged.inner.sketch(); + assert_eq!(m[0], vec![0.0, 0.0]); + assert_eq!(m[1], vec![0.0, 0.0]); + } + + #[test] + fn test_aggregate_core_merge_wrong_type_rejects() { + use crate::summary_kernels::count_min_sketch::CountMinSketchAccumulator; + let cs = CountSketchAccumulator::new(2, 3); + let cms = CountMinSketchAccumulator::new(2, 3); + let result = cs.merge_with(&cms); + assert!(result.is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs b/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs new file mode 100644 index 00000000..a735e4bd --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs @@ -0,0 +1,143 @@ +//! CountSketch with a top-k heap over `asap_sketchlib::CountSketchWithHeap` +//! (median-of-signed-rows), distinct from the Count-Min heap variant. + +use crate::{AggregateCore, KeyByLabelValues}; +use asap_sketchlib::CountSketchWithHeap; + +#[derive(Debug, Clone)] +pub struct CountSketchWithHeapAccumulator { + pub inner: CountSketchWithHeap, +} + +impl CountSketchWithHeapAccumulator { + pub fn new(row_num: usize, col_num: usize, heap_size: usize) -> Self { + Self { + inner: CountSketchWithHeap::new(row_num, col_num, heap_size), + } + } + + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + let key_string = key.labels.join(";"); + self.inner.estimate(&key_string) + } + + /// Value-weighted heavy-hitter update -- see + /// `CountMinSketchWithHeapAccumulator::insert_value`'s doc for why + /// this (not a `+1`-per-occurrence update) is the correct semantics + /// for `topk(k, sum by (label) (metric))`-shaped queries. + pub fn insert_value(&mut self, group_label: &str, value: f64) { + self.inner.update(group_label, value); + } + + /// Read the top-`k` groups ranked by summed value (descending, tie-broken + /// by key for determinism). Mirrors `CountMinSketchWithHeapAccumulator::topk_by_value`. + pub fn topk_by_value(&self, k: usize) -> Vec<(String, f64)> { + let mut items: Vec<(String, f64)> = self + .inner + .topk_heap_items() + .into_iter() + .map(|it| (it.key, it.value)) + .collect(); + items.sort_by(|a, b| { + b.1.partial_cmp(&a.1) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.0.cmp(&b.0)) + }); + items.truncate(k); + items + } + + /// Get all keys from the top-k heap. + pub fn get_topk_keys(&self) -> Vec { + self.inner + .topk_heap_items() + .iter() + .map(|item| { + let labels: Vec = item.key.split(';').map(|s| s.to_string()).collect(); + KeyByLabelValues { labels } + }) + .collect() + } +} + +impl AggregateCore for CountSketchWithHeapAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other_cs = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountSketchWithHeapAccumulator")?; + + let mut merged = self.clone(); + merged.inner.merge(&other_cs.inner)?; + Ok(Box::new(merged)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_count_sketch_with_heap_creation() { + let cs = CountSketchWithHeapAccumulator::new(4, 1000, 20); + assert_eq!(cs.inner.rows(), 4); + assert_eq!(cs.inner.cols(), 1000); + assert_eq!(cs.inner.heap_size, 20); + assert_eq!(cs.inner.topk_heap_items().len(), 0); + } + + #[test] + fn test_get_topk_keys() { + let mut cs = CountSketchWithHeapAccumulator::new(2, 3, 5); + cs.inner.update("label1;label2", 100.0); + cs.inner.update("label3;label4", 50.0); + + let keys = cs.get_topk_keys(); + assert_eq!(keys.len(), 2); + let label_sets: std::collections::HashSet<_> = + keys.iter().map(|k| k.labels.clone()).collect(); + assert!(label_sets.contains(&vec!["label1".to_string(), "label2".to_string()])); + assert!(label_sets.contains(&vec!["label3".to_string(), "label4".to_string()])); + } + + #[test] + fn insert_value_accumulates_summed_value_in_heap() { + let mut acc = CountSketchWithHeapAccumulator::new(4, 1024, 8); + acc.insert_value("g", 10.0); + acc.insert_value("g", 25.0); + let top = acc.topk_by_value(1); + assert_eq!(top.len(), 1); + assert_eq!(top[0].0, "g"); + assert!( + (top[0].1 - 35.0).abs() < 1e-6, + "summed value should be 35 (10+25), got {}", + top[0].1 + ); + } + + /// CountSketch and Count-Min heap states are distinct families and never merge. + #[test] + fn test_rejects_merge_with_cms_family_accumulator() { + use crate::summary_kernels::count_min_sketch_with_heap::CountMinSketchWithHeapAccumulator; + + let cs = CountSketchWithHeapAccumulator::new(4, 64, 10); + let cms = CountMinSketchWithHeapAccumulator::new(4, 64, 10); + let result = cs.merge_with(&cms); + assert!( + result.is_err(), + "CountSketchWithHeapAccumulator must not merge with CountMinSketchWithHeapAccumulator \ + -- different algorithms sharing only a storage shape" + ); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs b/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs new file mode 100644 index 00000000..0f740a9d --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs @@ -0,0 +1,106 @@ +//! KLL quantile summary over `asap_sketchlib::KllSketch`. +use crate::{AggregateCore, KernelError}; +use asap_sketchlib::KllSketch; +use planner_types::post_asap::SketchQuery; + +#[derive(Clone)] +pub struct DatasketchesKLLAccumulator { + pub inner: KllSketch, +} + +impl DatasketchesKLLAccumulator { + pub fn new(k: u16) -> Self { + Self { + inner: KllSketch::new(k), + } + } + + pub fn update(&mut self, value: f64) { + self.inner.update(value); + } + + pub fn get_quantile(&self, quantile: f64) -> f64 { + self.inner.quantile(quantile) + } +} + +impl std::fmt::Debug for DatasketchesKLLAccumulator { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("DatasketchesKLLAccumulator") + .field("k", &self.inner.k) + .field("sketch_n", &self.inner.count()) + .finish() + } +} + +// SAFETY: `KllSketch` owns its buffers and has no interior mutability; the +// accumulator is only mutated through `&mut self`. +unsafe impl Send for DatasketchesKLLAccumulator {} +unsafe impl Sync for DatasketchesKLLAccumulator {} + +impl AggregateCore for DatasketchesKLLAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("KLL merges only with KLL")?; + Ok(Box::new(Self { + inner: KllSketch::merge_refs(&[&self.inner, &other.inner])?, + })) + } + + fn estimate(&self, query: &SketchQuery) -> Result { + match query { + SketchQuery::Quantile { q } if (0.0..=1.0).contains(q) => Ok(self.get_quantile(*q)), + SketchQuery::Quantile { .. } => Err("quantile must be in [0, 1]".into()), + other => Err(format!("KLL does not answer {other:?}").into()), + } + } + + fn approx_memory_bytes(&self) -> usize { + // KLL with default k=200 holds ~2*k items (~3 KiB); round up for overhead. + 4 * 1024 + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Merging two KLL states reads like one state built over both inputs. + #[test] + fn merged_quantile_matches_single_build() { + let (mut a, mut b, mut all) = ( + DatasketchesKLLAccumulator::new(200), + DatasketchesKLLAccumulator::new(200), + DatasketchesKLLAccumulator::new(200), + ); + for v in 0..100 { + a.update(f64::from(v)); + all.update(f64::from(v)); + } + for v in 100..200 { + b.update(f64::from(v)); + all.update(f64::from(v)); + } + let merged = a.merge_with(&b).unwrap(); + let q = SketchQuery::Quantile { q: 0.5 }; + assert_eq!(merged.estimate(&q).unwrap(), all.estimate(&q).unwrap()); + } + + // KLL answers only quantiles in [0, 1]. + #[test] + fn rejects_unsupported_or_out_of_range_queries() { + let kll = DatasketchesKLLAccumulator::new(200); + assert!(kll.estimate(&SketchQuery::Quantile { q: 1.5 }).is_err()); + assert!(kll.estimate(&SketchQuery::Cardinality).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs new file mode 100644 index 00000000..1d6be98c --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs @@ -0,0 +1,88 @@ +//! DDSketch quantile summary over `asap_sketchlib::DdSketch`. +use crate::{AggregateCore, KernelError}; +use asap_sketchlib::DdSketch; +use planner_types::post_asap::SketchQuery; + +#[derive(Debug, Clone)] +pub struct DDSketchAccumulator { + pub inner: DdSketch, +} + +impl DDSketchAccumulator { + pub fn new(alpha: f64) -> Self { + Self { + inner: DdSketch::new(alpha), + } + } +} + +impl AggregateCore for DDSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("DDSketch merges only with DDSketch")?; + Ok(Box::new(Self { + inner: DdSketch::merge_refs(&[&self.inner, &other.inner])?, + })) + } + + /// Quantiles, and the total sample count as a bare `PointCount`. + fn estimate(&self, query: &SketchQuery) -> Result { + match query { + SketchQuery::Quantile { q } if (0.0..=1.0).contains(q) => self + .inner + .quantile(*q) + .ok_or_else(|| "DDSketch quantile of an empty population".into()), + SketchQuery::Quantile { .. } => Err("quantile must be in [0, 1]".into()), + SketchQuery::PointCount { value: None, .. } => Ok(self.inner.total_count() as f64), + other => Err(format!("DDSketch does not answer {other:?}").into()), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use planner_types::pre_asap::ColumnRef; + + fn bare_count() -> SketchQuery { + SketchQuery::PointCount { + key: ColumnRef::SampleValue, + value: None, + } + } + + // A bare point count reads the total sample count, and merge adds counts. + #[test] + fn count_and_quantile_survive_merge() { + let (mut a, mut b) = ( + DDSketchAccumulator::new(0.01), + DDSketchAccumulator::new(0.01), + ); + for v in 1..=50 { + a.inner.update(f64::from(v)); + b.inner.update(f64::from(v + 50)); + } + let merged = a.merge_with(&b).unwrap(); + assert_eq!(merged.estimate(&bare_count()).unwrap(), 100.0); + let median = merged.estimate(&SketchQuery::Quantile { q: 0.5 }).unwrap(); + assert!((median - 50.0).abs() <= 1.0, "{median}"); + } + + // An empty DDSketch has no quantile, and unsupported queries are errors. + #[test] + fn empty_quantile_and_unsupported_queries_fail() { + let dd = DDSketchAccumulator::new(0.01); + assert!(dd.estimate(&SketchQuery::Quantile { q: 0.5 }).is_err()); + assert!(dd.estimate(&SketchQuery::Cardinality).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/exact.rs b/crates/asap-physical-operators/src/summary_kernels/exact.rs new file mode 100644 index 00000000..cb9d6099 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/exact.rs @@ -0,0 +1,237 @@ +//! Exact summary state identified by Planner family, independent of keyed layout. +use super::increase::IncreaseAccumulator; +use crate::Statistic; +use crate::{AggregateCore, KeyByLabelValues, Measurement}; +use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; +use serde::{Deserialize, Serialize}; +use std::collections::HashMap; + +type Error = Box; + +#[derive(Debug, Clone, Serialize, Deserialize)] +enum ScalarState { + Sum(f64), + Count(u64), + Min(Option), + Max(Option), + Counter(Option), +} + +/// Both the family and population layout survive persistence. Sharing counter +/// arithmetic never authorizes a Rate state to answer an Increase readout. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ExactAccumulator { + family: SummaryFamilyType, + scalar: ScalarState, + keyed: Option>, +} + +/// Planned readout of an exact summary. `lookback_ms` is the logical PromQL +/// counter window; the evaluation range is resolved from it at run time. +#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] +pub struct ExactReadout { + pub statistic: Statistic, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub lookback_ms: Option, +} + +impl ExactAccumulator { + /// Read one population. An empty MIN/MAX population reads as `None`. + /// `range_ms` extrapolates a counter Rate/Increase to that evaluation range. + pub fn readout( + &self, + statistic: Statistic, + range_ms: Option<(i64, i64)>, + key: Option<&KeyByLabelValues>, + ) -> Result, Error> { + if statistic != self.statistic() { + return Err("readout differs from Planner exact family".into()); + } + let state = match (&self.keyed, key) { + (Some(states), Some(key)) => states.get(key).ok_or("unknown exact population")?, + (None, None) => &self.scalar, + _ => return Err("readout population differs from installed layout".into()), + }; + match state { + ScalarState::Sum(sum) => Ok(Some(*sum)), + ScalarState::Count(count) => Ok(Some(*count as f64)), + ScalarState::Min(value) | ScalarState::Max(value) => Ok(*value), + ScalarState::Counter(Some(counter)) => counter + .extrapolated_value(range_ms, statistic == Statistic::Rate) + .map(Some), + ScalarState::Counter(None) => Err("empty counter population".into()), + } + } + + /// Exact integer count of an unkeyed Count state. + pub fn count(&self) -> Option { + match (&self.keyed, &self.scalar) { + (None, ScalarState::Count(count)) => Some(*count), + _ => None, + } + } + + /// Accumulate into run-local scratch state. Persistent input states remain + /// immutable; a failed merge discards this scratch state. + pub(crate) fn merge_from(&mut self, other: &Self) -> Result<(), Error> { + if self.family != other.family || self.is_keyed() != other.is_keyed() { + return Err("cannot merge different Planner families or layouts".into()); + } + if let (Some(target), Some(source)) = (&mut self.keyed, &other.keyed) { + for (key, state) in source { + let combined = match target.get(key) { + Some(old) => merge_scalar(old, state)?, + None => state.clone(), + }; + target.insert(key.clone(), combined); + } + } else { + self.scalar = merge_scalar(&self.scalar, &other.scalar)?; + } + Ok(()) + } + + pub fn new(family: SummaryFamilyType, keyed: bool) -> Result { + use ExactKind as K; + use ExactParams as P; + let scalar = match &family { + SummaryFamilyType::ExactAggregate(K::Sum, P::Sum) => ScalarState::Sum(0.0), + SummaryFamilyType::ExactAggregate(K::Count, P::Count) => ScalarState::Count(0), + SummaryFamilyType::ExactAggregate(K::Min, P::Min) => ScalarState::Min(None), + SummaryFamilyType::ExactAggregate(K::Max, P::Max) => ScalarState::Max(None), + SummaryFamilyType::ExactAggregate(K::Rate, P::Rate) + | SummaryFamilyType::ExactAggregate(K::Increase, P::Increase) => { + ScalarState::Counter(None) + } + _ => return Err(format!("unsupported exact Planner family: {family:?}")), + }; + Ok(Self { + family, + scalar, + keyed: keyed.then(HashMap::new), + }) + } + + pub fn family(&self) -> &SummaryFamilyType { + &self.family + } + pub(crate) fn insufficient_counter_samples( + &self, + statistic: Statistic, + key: &Option, + ) -> bool { + if statistic != self.statistic() { + return false; + } + let state = match (&self.keyed, key) { + (Some(states), Some(key)) => states.get(key), + (None, None) => Some(&self.scalar), + _ => None, + }; + match state { + Some(ScalarState::Counter(None)) => true, + Some(ScalarState::Counter(Some(counter))) => { + counter.sample_count < 2 + || counter.last_seen_timestamp == counter.starting_timestamp + } + _ => false, + } + } + pub fn is_keyed(&self) -> bool { + self.keyed.is_some() + } + + pub fn update(&mut self, key: Option<&KeyByLabelValues>, value: f64, timestamp: i64) { + let state = match (&mut self.keyed, key) { + (Some(states), Some(key)) => states + .entry(key.clone()) + .or_insert_with(|| self.scalar.clone()), + (None, None) => &mut self.scalar, + _ => panic!("exact update population layout differs from installed DAG"), + }; + match state { + ScalarState::Sum(sum) => *sum += value, + ScalarState::Count(count) => { + *count = count.checked_add(1).expect("exact count overflow") + } + ScalarState::Min(current) => { + *current = Some(current.map_or(value, |old| old.min(value))) + } + ScalarState::Max(current) => { + *current = Some(current.map_or(value, |old| old.max(value))) + } + ScalarState::Counter(current) => match current { + Some(counter) => counter.update(Measurement::new(value), timestamp), + None => { + *current = Some(IncreaseAccumulator::new( + Measurement::new(value), + timestamp, + Measurement::new(value), + timestamp, + )) + } + }, + } + } + + fn statistic(&self) -> Statistic { + match self.family { + SummaryFamilyType::ExactAggregate(ExactKind::Sum, _) => Statistic::Sum, + SummaryFamilyType::ExactAggregate(ExactKind::Count, _) => Statistic::Count, + SummaryFamilyType::ExactAggregate(ExactKind::Min, _) => Statistic::Min, + SummaryFamilyType::ExactAggregate(ExactKind::Max, _) => Statistic::Max, + SummaryFamilyType::ExactAggregate(ExactKind::Rate, _) => Statistic::Rate, + SummaryFamilyType::ExactAggregate(ExactKind::Increase, _) => Statistic::Increase, + _ => unreachable!("validated exact family"), + } + } +} + +fn merge_scalar(left: &ScalarState, right: &ScalarState) -> Result { + Ok(match (left, right) { + (ScalarState::Sum(a), ScalarState::Sum(b)) => ScalarState::Sum(a + b), + (ScalarState::Count(a), ScalarState::Count(b)) => { + ScalarState::Count(a.checked_add(*b).ok_or("exact count overflow")?) + } + (ScalarState::Min(a), ScalarState::Min(b)) => { + ScalarState::Min(a.iter().chain(b).copied().reduce(f64::min)) + } + (ScalarState::Max(a), ScalarState::Max(b)) => { + ScalarState::Max(a.iter().chain(b).copied().reduce(f64::max)) + } + (ScalarState::Counter(a), ScalarState::Counter(b)) => ScalarState::Counter(match (a, b) { + (Some(a), Some(b)) => Some(IncreaseAccumulator::merge_pair(a, b)), + (a, b) => a.clone().or_else(|| b.clone()), + }), + _ => return Err("exact scalar state families differ".into()), + }) +} + +impl AggregateCore for ExactAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + fn as_any(&self) -> &dyn std::any::Any { + self + } + fn merge_with(&self, other: &dyn AggregateCore) -> Result, Error> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("merge requires Planner exact state")?; + let mut merged = self.clone(); + merged.merge_from(other)?; + Ok(Box::new(merged)) + } + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::() + + self.keyed.as_ref().map_or(0, |m| { + m.keys() + .map(|k| { + std::mem::size_of::() + + k.labels.iter().map(String::len).sum::() + }) + .sum::() + }) + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/factory.rs b/crates/asap-physical-operators/src/summary_kernels/factory.rs new file mode 100644 index 00000000..d44e64ab --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/factory.rs @@ -0,0 +1,799 @@ +use crate::summary_kernels::hll_sketch::HllSketchAccumulator; +use crate::summary_kernels::univmon::UnivMonAccumulator; +use crate::summary_kernels::{ + CountMinSketchAccumulator, CountMinSketchWithHeapAccumulator, CountSketchAccumulator, + CountSketchWithHeapAccumulator, DDSketchAccumulator, DatasketchesKLLAccumulator, + HydraKllSketchAccumulator, +}; +use crate::{AggregateCore, KeyByLabelValues}; +use planner_types::post_asap::{SketchAlgorithm, SketchParams, SummaryFamilyType}; + +/// Generate the clone-based `AccumulatorUpdater` methods for updaters whose +/// inner `acc` field implements `Clone + AggregateCore`. +macro_rules! impl_clone_accumulator_methods { + ($acc_field:ident) => { + fn take_accumulator(&mut self) -> Box { + let result = Box::new(self.$acc_field.clone()); + self.reset(); + result + } + + fn snapshot_accumulator(&self) -> Box { + Box::new(self.$acc_field.clone()) + } + + fn into_accumulator(self: Box) -> Box { + // Consume the updater and MOVE the accumulator out — no clone. + // Avoids a clone when a pane is evicted at window close. + let this = *self; + Box::new(this.$acc_field) + } + }; +} + +/// Shared update interface for query-time and precompute-time accumulation. +/// +/// This provides a uniform interface over all accumulator types so that the +/// operators don't need to know which concrete type they're dealing with. +pub trait AccumulatorUpdater: Send { + /// Validate an immutable precompute input before an updater can silently + /// discard a value outside its representable domain. + fn validate_single_input(&self, value: f64) -> Result<(), String> { + if value.is_finite() { + Ok(()) + } else { + Err("accumulator input must be finite".into()) + } + } + + /// Feed a single (value, timestamp_ms) pair — for SingleSubpopulation types. + fn update_single(&mut self, value: f64, timestamp_ms: i64); + + /// Feed a keyed (key, value, timestamp_ms) triple, e.g. a frequency item or an exact keyed state. + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, timestamp_ms: i64); + + /// Extract the final accumulator as a boxed `AggregateCore`. + fn take_accumulator(&mut self) -> Box; + + /// Non-destructive read of the current accumulator state (clone without reset). + /// Used by pane-based sliding windows to read shared panes. + fn snapshot_accumulator(&self) -> Box; + + /// Consume the updater and return its accumulator by move, avoiding the + /// clone that `take_accumulator`/`snapshot_accumulator` pay. Default falls + /// back to a clone for updaters that can't move their inner state out. + fn into_accumulator(self: Box) -> Box { + self.snapshot_accumulator() + } + + /// Reset internal state for reuse (avoids re-allocation). + fn reset(&mut self); + + /// Whether this updater consumes keyed updates. + fn is_keyed(&self) -> bool; + + /// Estimated memory usage in bytes. + fn memory_usage_bytes(&self) -> usize; +} + +// --------------------------------------------------------------------------- +// KllAccumulatorUpdater +// --------------------------------------------------------------------------- + +pub struct KllAccumulatorUpdater { + acc: DatasketchesKLLAccumulator, + k: u16, +} + +impl KllAccumulatorUpdater { + pub fn new(k: u16) -> Self { + Self { + acc: DatasketchesKLLAccumulator::new(k), + k, + } + } +} + +impl AccumulatorUpdater for KllAccumulatorUpdater { + fn update_single(&mut self, value: f64, _timestamp_ms: i64) { + self.acc.update(value); + } + + fn update_keyed(&mut self, _key: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = DatasketchesKLLAccumulator::new(self.k); + } + + fn is_keyed(&self) -> bool { + false + } + + fn memory_usage_bytes(&self) -> usize { + // KLL sketch size is hard to estimate precisely; use a rough estimate + std::mem::size_of::() + 4096 + } +} + +// --------------------------------------------------------------------------- +// DDSketchAccumulatorUpdater +// --------------------------------------------------------------------------- +pub struct DDSketchAccumulatorUpdater { + acc: DDSketchAccumulator, + alpha: f64, +} + +impl DDSketchAccumulatorUpdater { + pub fn new(alpha: f64) -> Self { + Self { + acc: DDSketchAccumulator::new(alpha), + alpha, + } + } +} + +impl AccumulatorUpdater for DDSketchAccumulatorUpdater { + fn validate_single_input(&self, value: f64) -> Result<(), String> { + let (minimum, maximum) = + asap_sketchlib::sketches::ddsketch::ddsketch_indexable_bounds(self.alpha); + if value.is_finite() && value > 0.0 && value >= minimum && value <= maximum { + Ok(()) + } else { + Err("DDS maintenance input is outside its positive representable domain".into()) + } + } + + fn update_single(&mut self, value: f64, _timestamp_ms: i64) { + self.acc.inner.update(value); + } + + fn update_keyed(&mut self, _key: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = DDSketchAccumulator::new(self.alpha); + } + + fn is_keyed(&self) -> bool { + false + } + + fn memory_usage_bytes(&self) -> usize { + // Bucket store is variable; rough estimate matches KLL. + std::mem::size_of::() + 4096 + } +} + +// --------------------------------------------------------------------------- +// CmsAccumulatorUpdater (CountMinSketch) +// --------------------------------------------------------------------------- + +/// Keyed weighted-frequency updater. +/// +/// A raw Prometheus sample represents the observed metric value, so a bare CMS +/// adds `value` for its key. Counting each received sample as one is a distinct +/// event-count operation and requires an explicit typed plan contract; it must +/// not be inferred from the sketch algorithm alone. +pub struct CmsAccumulatorUpdater { + acc: CountMinSketchAccumulator, + row_num: usize, + col_num: usize, +} + +impl CmsAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + acc: CountMinSketchAccumulator::new(row_num, col_num), + row_num, + col_num, + } + } +} + +impl AccumulatorUpdater for CmsAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.inner.update(&key.to_semicolon_str(), value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = CountMinSketchAccumulator::new(self.row_num, self.col_num); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + } +} + +// --------------------------------------------------------------------------- +// CmsHeapAccumulatorUpdater — value-weighted / count-weighted top-k +// --------------------------------------------------------------------------- + +/// What quantity the top-k heap ranks keys by. +/// +/// These are DIFFERENT query semantics and must be chosen explicitly: +/// +/// * [`TopkWeight::Value`] — accumulate **Σ of the datapoint value** per key. +/// This answers "top-k by total " (e.g. "top-k hosts by +/// total CPU"). The heap value is the summed metric value, so the read-side +/// reducer's "sort heap descending by value" yields the correct ranking. +/// +/// * [`TopkWeight::Count`] — accumulate **+1 per event** per key (occurrence +/// frequency), the textbook heavy-hitter / frequency-top-k semantics +/// ("which keys appear most often"). +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum TopkWeight { + /// Σ datapoint value per key (value-weighted top-k). + Value, + /// +1 per event per key (count-weighted / frequency top-k). + Count, +} + +/// Keyed top-k updater backed by a real `CountMinSketchWithHeap` (a CMS +/// matrix PLUS a size-`heap_size` top-k heap). Unlike the heap-LESS +/// `CmsAccumulatorUpdater`, this enumerates top-k keys at read time +/// (`get_topk_keys` / `topk_heap_items`), which is what `topk(...)` queries +/// need. +/// +/// The key is the frequency item supplied by the operator (e.g. a `host` +/// value), not a group-by population. The accumulated quantity is selected +/// by [`TopkWeight`]: +/// * `Value` → `inner.update(key, value)` adds the datapoint value (Σ value). +/// * `Count` → `inner.update(key, 1.0)` adds one per event (Σ count). +/// +/// Both `CountMinSketchWithHeap` and `CountSketchWithHeap` raw-input policies +/// route here; the heap is the shared distinguishing payload. +pub struct CmsHeapAccumulatorUpdater { + acc: CountMinSketchWithHeapAccumulator, + row_num: usize, + col_num: usize, + heap_size: usize, + weight: TopkWeight, +} + +impl CmsHeapAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize, heap_size: usize, weight: TopkWeight) -> Self { + Self { + acc: CountMinSketchWithHeapAccumulator::new(row_num, col_num, heap_size), + row_num, + col_num, + heap_size, + weight, + } + } +} + +impl AccumulatorUpdater for CmsHeapAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + // Heap key = the group-by label-value vector (e.g. `host`), joined the + // same way the read-side `get_topk_keys` splits it back apart (`;`). + let weighted = match self.weight { + // Σ value: feed the datapoint value. sketchlib's CMS-heap + // `update(key, w)` adds `w.round()` occurrences of `key`, so the + // heap value accumulates the (rounded) summed metric value. + TopkWeight::Value => value, + // Σ count: one occurrence per event, regardless of value. + TopkWeight::Count => 1.0, + }; + self.acc.inner.update(&key.to_semicolon_str(), weighted); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = + CountMinSketchWithHeapAccumulator::new(self.row_num, self.col_num, self.heap_size); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + + self.heap_size * (std::mem::size_of::() + 32) + } +} + +// --------------------------------------------------------------------------- +// CountSketchAccumulatorUpdater (real median-of-signed-rows CountSketch) +// --------------------------------------------------------------------------- + +/// Keyed point-frequency updater backed by a real `asap_sketchlib::CountSketch` +/// (signed rows, median-of-rows estimator) — distinct math from +/// `CmsAccumulatorUpdater`'s CMS (min-of-rows). +/// +/// As with bare CMS, each raw Prometheus sample contributes its `value`. +/// Unit event counting must be selected explicitly by a future typed plan +/// contract rather than being implied by `SketchAlgorithm::CountSketch`. +pub struct CountSketchAccumulatorUpdater { + acc: CountSketchAccumulator, + row_num: usize, + col_num: usize, +} + +impl CountSketchAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + acc: CountSketchAccumulator::new(row_num, col_num), + row_num, + col_num, + } + } +} + +impl AccumulatorUpdater for CountSketchAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.inner.update(&key.to_semicolon_str(), value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = CountSketchAccumulator::new(self.row_num, self.col_num); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + } +} + +// --------------------------------------------------------------------------- +// CountSketchWithHeapAccumulatorUpdater (real CountSketch + top-k heap) +// --------------------------------------------------------------------------- + +/// Keyed top-k updater backed by a real `CountSketchWithHeap` (signed-row +/// CountSketch matrix PLUS a size-`heap_size` top-k heap). Distinct math from +/// `CmsHeapAccumulatorUpdater`'s CMS-with-heap (min-of-rows); shares the same +/// [`TopkWeight`] semantics and heap payload shape. +pub struct CountSketchWithHeapAccumulatorUpdater { + acc: CountSketchWithHeapAccumulator, + row_num: usize, + col_num: usize, + heap_size: usize, + weight: TopkWeight, +} + +impl CountSketchWithHeapAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize, heap_size: usize, weight: TopkWeight) -> Self { + Self { + acc: CountSketchWithHeapAccumulator::new(row_num, col_num, heap_size), + row_num, + col_num, + heap_size, + weight, + } + } +} + +impl AccumulatorUpdater for CountSketchWithHeapAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + let weighted = match self.weight { + TopkWeight::Value => value, + TopkWeight::Count => 1.0, + }; + self.acc.inner.update(&key.to_semicolon_str(), weighted); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = CountSketchWithHeapAccumulator::new(self.row_num, self.col_num, self.heap_size); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + + self.heap_size * (std::mem::size_of::() + 32) + } +} + +// --------------------------------------------------------------------------- +// HydraKllAccumulatorUpdater +// --------------------------------------------------------------------------- + +pub struct HydraKllAccumulatorUpdater { + acc: HydraKllSketchAccumulator, + row_num: usize, + col_num: usize, + k: u16, +} + +impl HydraKllAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize, k: u16) -> Self { + Self { + acc: HydraKllSketchAccumulator::new(row_num, col_num, k), + row_num, + col_num, + k, + } + } +} + +impl AccumulatorUpdater for HydraKllAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.update(key, value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = HydraKllSketchAccumulator::new(self.row_num, self.col_num, self.k); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + // Rough estimate: each cell is a KLL sketch + std::mem::size_of::() + self.row_num * self.col_num * 4096 + } +} + +// --------------------------------------------------------------------------- +// Config helpers +// --------------------------------------------------------------------------- + +fn cms_dims(params: &SketchParams) -> (usize, usize) { + match params { + SketchParams::Cms { width, depth } | SketchParams::CountSketch { width, depth } => { + (*depth as usize, *width as usize) + } + other => unreachable!( + "accumulator_spec() paired SketchAlgorithm::Cms/CountSketch with unexpected params: {other:?}" + ), + } +} + +/// Read `(rows = depth, columns = width, heap_size)` out of `SketchParams::CmsWithHeap` +/// or `::CountSketchWithHeap`. +fn cms_heap_dims(params: &SketchParams) -> (usize, usize, usize) { + match params { + SketchParams::CmsWithHeap { + width, + depth, + heap_size, + } + | SketchParams::CountSketchWithHeap { + width, + depth, + heap_size, + } => (*depth as usize, *width as usize, *heap_size as usize), + other => unreachable!( + "accumulator_spec() paired a WithHeap SketchAlgorithm with unexpected params: {other:?}" + ), + } +} + +/// Construct the kernel declared by a Planner SummaryAgg. No deployment config +/// tags participate in this dispatch and unsupported payloads are errors. +pub fn create_planner_accumulator( + family: &SummaryFamilyType, + input: &planner_types::post_asap::SummaryUpdate, + grouping: &planner_types::post_asap::GroupingStrategy, +) -> Result, String> { + if input.item.is_some() + && matches!( + input.weight_domain, + planner_types::post_asap::WeightDomain::NonNegative { + proof: + planner_types::post_asap::NonNegativeWeightProof::ResetAwareCounterDerivative + } + ) + { + return Err("window-weighted summaries require typed DAG binding; integer heap updaters cannot consume rates".into()); + } + + crate::capability::validate_summary_kernel(family, input, grouping)?; + use planner_types::post_asap::GroupingStrategy; + if grouping != &GroupingStrategy::PerSubpopulationInstance { + return Err("shared summary grouping requires a supported Planner Hydra kernel".into()); + } + if matches!(family, SummaryFamilyType::ExactAggregate(..)) { + return Ok(Box::new(PlannerExactUpdater { + acc: crate::summary_kernels::exact::ExactAccumulator::new( + family.clone(), + input.item.is_some(), + )?, + })); + } + let SummaryFamilyType::Sketch(kind, family_grouping) = family else { + return Err(format!("unsupported Planner summary family {family:?}")); + }; + if family_grouping != grouping { + return Err("Planner family and operator grouping disagree".into()); + } + let updater: Box = match (kind.algorithm(), kind.params()) { + (SketchAlgorithm::Kll, SketchParams::Kll { k }) => Box::new(KllAccumulatorUpdater::new( + u16::try_from(*k).map_err(|_| "KLL k exceeds runtime bound")?, + )), + (SketchAlgorithm::DDSketch, SketchParams::DDSketch { alpha }) => { + Box::new(DDSketchAccumulatorUpdater::new(*alpha)) + } + (SketchAlgorithm::Cms, params @ SketchParams::Cms { .. }) => { + let (r, c) = cms_dims(params); + Box::new(CmsAccumulatorUpdater::new(r, c)) + } + (SketchAlgorithm::CountSketch, params @ SketchParams::CountSketch { .. }) => { + let (r, c) = cms_dims(params); + Box::new(CountSketchAccumulatorUpdater::new(r, c)) + } + (SketchAlgorithm::CmsWithHeap, params @ SketchParams::CmsWithHeap { .. }) => { + let (r, c, h) = cms_heap_dims(params); + Box::new(CmsHeapAccumulatorUpdater::new(r, c, h, TopkWeight::Value)) + } + ( + SketchAlgorithm::CountSketchWithHeap, + params @ SketchParams::CountSketchWithHeap { .. }, + ) => { + let (r, c, h) = cms_heap_dims(params); + Box::new(CountSketchWithHeapAccumulatorUpdater::new( + r, + c, + h, + TopkWeight::Value, + )) + } + (SketchAlgorithm::Hll, SketchParams::Hll { precision }) => Box::new(HllUpdater { + acc: HllSketchAccumulator::new( + asap_sketchlib::HllVariant::Regular, + u32::from(*precision), + ), + }), + ( + SketchAlgorithm::UnivMon, + SketchParams::UnivMon { + heap_size, + sketch_rows, + sketch_cols, + layers, + }, + ) => Box::new(UnivMonUpdater { + acc: UnivMonAccumulator::new( + *heap_size as usize, + *sketch_rows as usize, + *sketch_cols as usize, + *layers as usize, + ) + .map_err(|e| e.to_string())?, + }), + _ => { + return Err(format!( + "unsupported Planner algorithm/parameters: {kind:?}" + )) + } + }; + if updater.is_keyed() != input.item.is_some() + && !crate::capability::is_unit_sample_frequency(input) + { + return Err("Planner item expression does not match the selected kernel layout".into()); + } + Ok(updater) +} + +struct PlannerExactUpdater { + acc: crate::summary_kernels::exact::ExactAccumulator, +} +impl AccumulatorUpdater for PlannerExactUpdater { + fn update_single(&mut self, value: f64, timestamp: i64) { + self.acc.update(None, value, timestamp); + } + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, timestamp: i64) { + self.acc.update(Some(key), value, timestamp); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc = crate::summary_kernels::exact::ExactAccumulator::new( + self.acc.family().clone(), + self.acc.is_keyed(), + ) + .expect("installed exact family"); + } + fn is_keyed(&self) -> bool { + self.acc.is_keyed() + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } +} + +struct UnivMonUpdater { + acc: UnivMonAccumulator, +} + +struct HllUpdater { + acc: HllSketchAccumulator, +} + +impl AccumulatorUpdater for HllUpdater { + fn is_keyed(&self) -> bool { + false + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } + fn update_single(&mut self, value: f64, _: i64) { + if !value.is_nan() { + let bits = if value == 0.0 { 0 } else { value.to_bits() }; + self.acc.inner.update(&bits.to_le_bytes()); + } + } + fn update_keyed(&mut self, _: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc.inner = + asap_sketchlib::HllSketch::new(self.acc.inner.variant, self.acc.inner.precision); + } +} + +impl AccumulatorUpdater for UnivMonUpdater { + fn is_keyed(&self) -> bool { + false + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } + fn update_single(&mut self, value: f64, _: i64) { + self.acc + .insert_sample(value) + .expect("UnivMon sample counter overflow"); + } + fn update_keyed(&mut self, _: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc.clear(); + } +} + +#[cfg(test)] +mod planner_parameter_regression { + use super::*; + use planner_types::post_asap::{SketchKind, SummaryInputExpr, SummaryUpdate}; + + // Planner width is the bucket count; depth is the independent hash-row count. + #[test] + fn planner_sketch_dimensions_are_not_transposed() { + for (algorithm, params) in [ + ( + SketchAlgorithm::Cms, + SketchParams::Cms { + width: 128, + depth: 3, + }, + ), + ( + SketchAlgorithm::CountSketch, + SketchParams::CountSketch { + width: 128, + depth: 3, + }, + ), + ( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 128, + depth: 3, + heap_size: 8, + }, + ), + ( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 128, + depth: 3, + heap_size: 8, + }, + ), + ] { + let family = SummaryFamilyType::Sketch( + SketchKind::new(algorithm.clone(), params), + Default::default(), + ); + let update = SummaryUpdate { + item: Some(SummaryInputExpr::Column( + planner_types::pre_asap::ColumnRef::Named("host".into()), + )), + weight: SummaryInputExpr::Constant(1.0), + weight_domain: Default::default(), + }; + let state = create_planner_accumulator(&family, &update, &Default::default()) + .unwrap() + .snapshot_accumulator(); + let dims = match algorithm { + SketchAlgorithm::Cms => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows(), s.inner.cols()) + } + SketchAlgorithm::CountSketch => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows, s.inner.cols) + } + SketchAlgorithm::CmsWithHeap => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows(), s.inner.cols()) + } + SketchAlgorithm::CountSketchWithHeap => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows(), s.inner.cols()) + } + _ => unreachable!(), + }; + assert_eq!(dims, (3, 128), "{algorithm:?}"); + } + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs new file mode 100644 index 00000000..089261c5 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs @@ -0,0 +1,79 @@ +//! HyperLogLog distinct-count summary over `asap_sketchlib::HllSketch`. +use crate::{AggregateCore, KernelError}; +use asap_sketchlib::{HllSketch, HllVariant}; +use planner_types::post_asap::SketchQuery; + +#[derive(Debug, Clone)] +pub struct HllSketchAccumulator { + pub inner: HllSketch, +} + +impl HllSketchAccumulator { + pub fn new(variant: HllVariant, precision: u32) -> Self { + Self { + inner: HllSketch::new(variant, precision), + } + } +} + +impl AggregateCore for HllSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("HLL merges only with HLL")?; + Ok(Box::new(Self { + inner: HllSketch::merge_refs(&[&self.inner, &other.inner])?, + })) + } + + /// Distinct count. A bare `PointCount` over an HLL also reads the distinct count. + fn estimate(&self, query: &SketchQuery) -> Result { + match query { + SketchQuery::Cardinality | SketchQuery::PointCount { value: None, .. } => { + Ok(self.inner.estimate()) + } + other => Err(format!("HLL does not answer {other:?}").into()), + } + } + + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::().saturating_add(self.inner.registers.capacity()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Distinct count after merge counts overlapping items once. + #[test] + fn merged_cardinality_deduplicates_overlap() { + let (mut a, mut b) = ( + HllSketchAccumulator::new(HllVariant::Regular, 12), + HllSketchAccumulator::new(HllVariant::Regular, 12), + ); + for v in 0..1000u32 { + a.inner.update(&v.to_le_bytes()); + b.inner.update(&(v + 500).to_le_bytes()); + } + let merged = a.merge_with(&b).unwrap(); + let estimate = merged.estimate(&SketchQuery::Cardinality).unwrap(); + assert!((estimate - 1500.0).abs() / 1500.0 < 0.05, "{estimate}"); + } + + // HLL does not answer quantiles. + #[test] + fn rejects_quantile() { + let hll = HllSketchAccumulator::new(HllVariant::Regular, 12); + assert!(hll.estimate(&SketchQuery::Quantile { q: 0.5 }).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs b/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs new file mode 100644 index 00000000..c0347e69 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs @@ -0,0 +1,55 @@ +use crate::{AggregateCore, KeyByLabelValues}; +use asap_sketchlib::HydraKllSketch; + +/// HydraKLL (shared-grouping quantiles) over `asap_sketchlib::HydraKllSketch`. +#[derive(Debug, Clone)] +pub struct HydraKllSketchAccumulator { + pub inner: HydraKllSketch, +} + +impl HydraKllSketchAccumulator { + pub fn new(row_num: usize, col_num: usize, k: u16) -> Self { + Self { + inner: HydraKllSketch::new(row_num, col_num, k), + } + } + + pub fn update(&mut self, key: &KeyByLabelValues, value: f64) { + self.inner.update(&key.to_semicolon_str(), value); + } + + pub fn query_key(&self, key: &KeyByLabelValues, quantile: f64) -> f64 { + self.inner.quantile(&key.to_semicolon_str(), quantile) + } +} + +impl AggregateCore for HydraKllSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let hk = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to HydraKllSketchAccumulator")?; + + let mut merged = self.clone(); + merged.inner.merge(&hk.inner)?; + Ok(Box::new(merged)) + } + + fn approx_memory_bytes(&self) -> usize { + // HydraKLL is a row*col grid of KLL sketches; typical instances + // are on the order of tens of KiB. 32 KiB is a conservative + // per-instance default. + 32 * 1024 + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/increase.rs b/crates/asap-physical-operators/src/summary_kernels/increase.rs new file mode 100644 index 00000000..f92ee582 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/increase.rs @@ -0,0 +1,213 @@ +use crate::{AggregateCore, Measurement}; +use serde::{Deserialize, Serialize}; + +/// Accumulator for tracking increases in counter metrics +/// Stores the starting and last seen measurements with timestamps +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct IncreaseAccumulator { + pub starting_measurement: Measurement, + pub starting_timestamp: i64, + pub last_seen_measurement: Measurement, + pub last_seen_timestamp: i64, + /// Sum of monotonic deltas, adding the post-reset value whenever the + /// counter decreases. This is the reset correction Prometheus applies. + #[serde(default)] + pub total_increase: f64, + #[serde(default)] + pub sample_count: u64, +} + +impl IncreaseAccumulator { + /// Merge two counter intervals without a temporary collection. Ties retain + /// the left input, matching the stable ordering of multi-pane merges. + pub(crate) fn merge_pair(left: &Self, right: &Self) -> Self { + let (first, second) = if left.starting_timestamp <= right.starting_timestamp { + (left, right) + } else { + (right, left) + }; + let mut merged = first.clone(); + if second.starting_timestamp > merged.last_seen_timestamp { + merged.total_increase += + if second.starting_measurement.value >= merged.last_seen_measurement.value { + second.starting_measurement.value - merged.last_seen_measurement.value + } else { + second.starting_measurement.value + }; + } + merged.total_increase += second.total_increase; + merged.sample_count = merged.sample_count.saturating_add(second.sample_count); + if second.last_seen_timestamp > merged.last_seen_timestamp { + merged.last_seen_measurement = second.last_seen_measurement.clone(); + merged.last_seen_timestamp = second.last_seen_timestamp; + } + + merged + } + + pub fn new( + starting_measurement: Measurement, + starting_timestamp: i64, + last_seen_measurement: Measurement, + last_seen_timestamp: i64, + ) -> Self { + let total_increase = if last_seen_timestamp <= starting_timestamp { + 0.0 + } else if last_seen_measurement.value >= starting_measurement.value { + last_seen_measurement.value - starting_measurement.value + } else { + last_seen_measurement.value + }; + let sample_count = if last_seen_timestamp > starting_timestamp { + 2 + } else { + 1 + }; + Self { + starting_measurement, + starting_timestamp, + last_seen_measurement, + last_seen_timestamp, + total_increase, + sample_count, + } + } + + pub fn update(&mut self, measurement: Measurement, timestamp: i64) { + if timestamp < self.last_seen_timestamp { + return; + } + if timestamp == self.last_seen_timestamp { + return; + } + if measurement.value >= self.last_seen_measurement.value { + self.total_increase += measurement.value - self.last_seen_measurement.value; + } else { + self.total_increase += measurement.value; + } + self.last_seen_measurement = measurement; + self.last_seen_timestamp = timestamp; + self.sample_count = self.sample_count.saturating_add(1); + } +} + +impl AggregateCore for IncreaseAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + // Downcast to IncreaseAccumulator + let other_increase = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to IncreaseAccumulator")?; + + let merged = Self::merge_pair(self, other_increase); + Ok(Box::new(merged)) + } + + fn approx_memory_bytes(&self) -> usize { + // Two Measurements + two i64s. Measurements are a few f64 fields. + std::mem::size_of::() + } +} + +impl IncreaseAccumulator { + /// PromQL-style increase or rate, extrapolated to `range_ms` when given. + pub(crate) fn extrapolated_value( + &self, + range_ms: Option<(i64, i64)>, + is_rate: bool, + ) -> Result> { + if self.sample_count < 2 || self.last_seen_timestamp <= self.starting_timestamp { + return Err("at least two ordered counter samples are required".into()); + } + let sampled_interval = (self.last_seen_timestamp - self.starting_timestamp) as f64 / 1000.0; + let Some((range_start, range_end)) = range_ms else { + return Ok(if is_rate { + self.total_increase / sampled_interval + } else { + self.total_increase + }); + }; + if range_end <= range_start { + return Err("invalid counter evaluation range".into()); + } + + let mut duration_to_start = + (self.starting_timestamp.saturating_sub(range_start)) as f64 / 1000.0; + let duration_to_end = (range_end.saturating_sub(self.last_seen_timestamp)) as f64 / 1000.0; + let average_sample_interval = sampled_interval / (self.sample_count - 1) as f64; + let extrapolation_threshold = average_sample_interval * 1.1; + + if self.total_increase > 0.0 && self.starting_measurement.value >= 0.0 { + let duration_to_zero = + sampled_interval * (self.starting_measurement.value / self.total_increase); + duration_to_start = duration_to_start.min(duration_to_zero); + } + let mut extrapolate_to = sampled_interval; + extrapolate_to += if duration_to_start < extrapolation_threshold { + duration_to_start.max(0.0) + } else { + average_sample_interval / 2.0 + }; + extrapolate_to += if duration_to_end < extrapolation_threshold { + duration_to_end.max(0.0) + } else { + average_sample_interval / 2.0 + }; + let mut factor = extrapolate_to / sampled_interval; + if is_rate { + factor /= (range_end - range_start) as f64 / 1000.0; + } + Ok(self.total_increase * factor) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_increase_accumulator_creation() { + let starting_measurement = Measurement::new(10.0); + let last_seen_measurement = Measurement::new(25.0); + let acc = IncreaseAccumulator::new( + starting_measurement.clone(), + 1000, + last_seen_measurement.clone(), + 2000, + ); + + assert_eq!(acc.starting_measurement.value, 10.0); + assert_eq!(acc.starting_timestamp, 1000); + assert_eq!(acc.last_seen_measurement.value, 25.0); + assert_eq!(acc.last_seen_timestamp, 2000); + } + + #[test] + fn test_increase_accumulator_update() { + let starting_measurement = Measurement::new(10.0); + let mut acc = IncreaseAccumulator::new( + starting_measurement.clone(), + 1000, + starting_measurement.clone(), + 1000, + ); + + let new_measurement = Measurement::new(25.0); + acc.update(new_measurement.clone(), 2000); + + assert_eq!(acc.last_seen_measurement.value, 25.0); + assert_eq!(acc.last_seen_timestamp, 2000); + assert_eq!(acc.starting_measurement.value, 10.0); // Should remain unchanged + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/mod.rs b/crates/asap-physical-operators/src/summary_kernels/mod.rs new file mode 100644 index 00000000..478e75d0 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/mod.rs @@ -0,0 +1,26 @@ +//! In-memory summary state: thin adapters over `asap_sketchlib` and exact Planner state. +pub mod count_min_sketch; +pub mod count_min_sketch_with_heap; +pub mod count_sketch; +pub mod count_sketch_with_heap; +pub mod datasketches_kll; +pub mod dd_sketch; +pub mod exact; +pub mod hll_sketch; +pub mod hydra_kll; +pub mod increase; +pub mod univmon; + +pub use count_min_sketch::*; +pub use count_min_sketch_with_heap::*; +pub use count_sketch::*; +pub use count_sketch_with_heap::*; +pub use datasketches_kll::*; +pub use dd_sketch::*; +pub use hll_sketch::*; +pub use hydra_kll::*; +pub use increase::*; + +pub mod factory; +pub mod traits; +pub mod weighted_frequency; diff --git a/crates/asap-physical-operators/src/summary_kernels/traits.rs b/crates/asap-physical-operators/src/summary_kernels/traits.rs new file mode 100644 index 00000000..7d866077 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/traits.rs @@ -0,0 +1,35 @@ +use planner_types::post_asap::SketchQuery; + +pub type KernelError = Box; + +/// In-memory state of one population's summary. +/// +/// Kernels adapt `asap_sketchlib` structures (or exact Planner state) to the +/// operations physical operators need: merge, typed readout and memory +/// accounting. Grouping belongs to operators; byte encodings belong to +/// `asap_sketchlib` and deployments. +pub trait AggregateCore: Send + Sync { + fn clone_boxed_core(&self) -> Box; + + fn as_any(&self) -> &dyn std::any::Any; + + /// Merge with a state of the same family and shape, leaving both inputs unchanged. + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError>; + + /// Answer a sketch readout. Exact states are read through + /// [`ExactAccumulator::readout`](super::exact::ExactAccumulator::readout). + fn estimate(&self, query: &SketchQuery) -> Result { + Err(format!("{query:?} is not supported by this summary").into()) + } + + /// Approximate in-memory footprint, used for execution memory reservations. + fn approx_memory_bytes(&self) -> usize { + 4096 + } +} + +impl Clone for Box { + fn clone(&self) -> Self { + self.clone_boxed_core() + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/univmon.rs b/crates/asap-physical-operators/src/summary_kernels/univmon.rs new file mode 100644 index 00000000..bffd8afa --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/univmon.rs @@ -0,0 +1,109 @@ +//! One frequency state shared by count, distinct, L2 and entropy readouts. + +use crate::AggregateCore; +use asap_sketchlib::{DataInput, UnivMon}; + +type Error = Box; + +#[derive(Debug, Clone)] +pub struct UnivMonAccumulator { + inner: UnivMon, +} + +impl UnivMonAccumulator { + /// Empty the sketch in place, keeping its shape. + pub(crate) fn clear(&mut self) { + self.inner.free(); + } + + pub fn new(heap_size: usize, rows: usize, cols: usize, layers: usize) -> Result { + if heap_size == 0 || cols == 0 || !(1..=20).contains(&rows) || !(1..=64).contains(&layers) { + return Err("invalid UnivMon dimensions".into()); + } + rows.checked_mul(cols) + .and_then(|n| n.checked_mul(layers)) + .ok_or("UnivMon dimensions overflow")?; + Ok(Self { + inner: UnivMon::init_univmon(heap_size, rows, cols, layers), + }) + } + + /// Each non-NaN sample is one occurrence. Signed zero has one identity. + pub fn insert_sample(&mut self, value: f64) -> Result<(), Error> { + if value.is_nan() { + return Ok(()); + } + self.inner + .bucket_size + .checked_add(1) + .ok_or("UnivMon count overflow")?; + let bits = if value == 0.0 { 0 } else { value.to_bits() }; + self.inner.insert(&DataInput::U64(bits), 1); + Ok(()) + } + + fn compatible(&self, other: &Self) -> bool { + ( + self.inner.heap_size, + self.inner.sketch_row, + self.inner.sketch_col, + self.inner.layer_size, + ) == ( + other.inner.heap_size, + other.inner.sketch_row, + other.inner.sketch_col, + other.inner.layer_size, + ) + } + + pub fn dimensions(&self) -> (usize, usize, usize, usize) { + ( + self.inner.heap_size, + self.inner.sketch_row, + self.inner.sketch_col, + self.inner.layer_size, + ) + } + + pub fn merge_in_place(&mut self, other: &Self) -> Result<(), Error> { + if !self.compatible(other) { + return Err("incompatible UnivMon dimensions".into()); + } + self.inner + .bucket_size + .checked_add(other.inner.bucket_size) + .ok_or("UnivMon count overflow")?; + self.inner.merge(&other.inner); + Ok(()) + } +} + +impl AggregateCore for UnivMonAccumulator { + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::().saturating_add( + self.inner.layer_size.saturating_mul( + self.inner + .sketch_row + .saturating_mul(self.inner.sketch_col) + .saturating_mul(16) + .saturating_add(self.inner.heap_size.saturating_mul(256)), + ), + ) + } + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, Error> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("expected UnivMon state")?; + let mut merged = self.clone(); + merged.merge_in_place(other)?; + Ok(Box::new(merged)) + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs new file mode 100644 index 00000000..6f008532 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs @@ -0,0 +1,152 @@ +//! ASAP type and trait adapter for sketchlib's Float64 weighted frequency kernel. +use crate::AggregateCore; +use crate::{values::Value, Error}; +pub use asap_sketchlib::FrequencyAlgorithm; +use asap_sketchlib::{FrequencyIdentity, WeightedFrequency as Kernel, WeightedFrequencyError}; +use serde::{Deserialize, Serialize}; + +fn adapt_error(error: WeightedFrequencyError) -> Error { + match error { + WeightedFrequencyError::Invalid(message) => Error::Invalid(message), + WeightedFrequencyError::Update(message) => Error::Operator(message), + } +} +fn identity(value: &Value) -> Result { + Ok(match value { + Value::Null => FrequencyIdentity::Null, + Value::Bool(v) => FrequencyIdentity::Bool(*v), + Value::Int64(v) => FrequencyIdentity::Int64(*v), + Value::Float64(v) => FrequencyIdentity::Float64(*v), + Value::Utf8(v) => FrequencyIdentity::Utf8(v.to_string()), + _ => { + return Err(Error::Invalid( + "unsupported weighted frequency identity".into(), + )) + } + }) +} +fn value(identity: FrequencyIdentity) -> Value { + match identity { + FrequencyIdentity::Null => Value::Null, + FrequencyIdentity::Bool(v) => Value::Bool(v), + FrequencyIdentity::Int64(v) => Value::Int64(v), + FrequencyIdentity::Float64(v) => Value::Float64(v), + FrequencyIdentity::Utf8(v) => Value::Utf8(v.into()), + } +} +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(transparent)] +pub struct WeightedFrequency { + inner: Kernel, +} +impl WeightedFrequency { + pub(crate) fn configuration( + kind: &planner_types::post_asap::SketchKind, + ) -> Result<(FrequencyAlgorithm, usize, usize, usize), Error> { + use planner_types::post_asap::{SketchAlgorithm as A, SketchParams as P}; + let (algorithm, width, depth, capacity) = match (kind.algorithm(), kind.params()) { + ( + A::CmsWithHeap, + P::CmsWithHeap { + width, + depth, + heap_size, + }, + ) => (FrequencyAlgorithm::Cms, *width, *depth, *heap_size), + ( + A::CountSketchWithHeap, + P::CountSketchWithHeap { + width, + depth, + heap_size, + }, + ) if depth % 2 == 1 => (FrequencyAlgorithm::CountSketch, *width, *depth, *heap_size), + _ => { + return Err(Error::Invalid( + "unsupported weighted frequency family or depth".into(), + )) + } + }; + if width == 0 || depth == 0 || capacity == 0 { + return Err(Error::Invalid( + "invalid weighted frequency dimensions".into(), + )); + } + Ok((algorithm, width as usize, depth as usize, capacity as usize)) + } + + pub(crate) fn algorithm(&self) -> FrequencyAlgorithm { + self.inner.algorithm() + } + pub(crate) fn shape(&self) -> (usize, usize, usize) { + self.inner.shape() + } + pub fn new( + algorithm: FrequencyAlgorithm, + width: usize, + depth: usize, + capacity: usize, + ) -> Result { + Kernel::new(algorithm, width, depth, capacity) + .map(|inner| Self { inner }) + .map_err(adapt_error) + } + pub fn update(&mut self, values: &[Value], weight: f64) -> Result<(), Error> { + let values = values.iter().map(identity).collect::, _>>()?; + self.inner.update(&values, weight).map_err(adapt_error) + } + pub fn rows(&self, n: usize) -> Vec> { + self.inner + .topk(n) + .into_iter() + .map(|(items, score)| { + let mut row = items.into_iter().map(value).collect::>(); + row.push(Value::Float64(score)); + row + }) + .collect() + } +} +impl AggregateCore for WeightedFrequency { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + fn as_any(&self) -> &dyn std::any::Any { + self + } + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("weighted frequency state type mismatch")?; + Ok(Box::new(Self { + inner: self.inner.merge(&other.inner)?, + })) + } + fn approx_memory_bytes(&self) -> usize { + self.inner.approx_memory_bytes() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Merge uses the same Float64 state representation and rejects other shapes. + #[test] + fn compatible_merge_preserves_fractional_weights() { + let mut left = WeightedFrequency::new(FrequencyAlgorithm::Cms, 4096, 5, 8).unwrap(); + let mut right = left.clone(); + left.update(&[Value::Int64(7)], 0.125).unwrap(); + right.update(&[Value::Int64(7)], 0.25).unwrap(); + let merged = left.merge_with(&right).unwrap(); + let merged = merged.as_any().downcast_ref::().unwrap(); + assert!(matches!(merged.rows(1)[0][1], Value::Float64(0.375))); + assert!(left + .merge_with(&WeightedFrequency::new(FrequencyAlgorithm::Cms, 32, 5, 8).unwrap()) + .is_err()); + } +} diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs new file mode 100644 index 00000000..56d43633 --- /dev/null +++ b/crates/asap-physical-operators/src/values.rs @@ -0,0 +1,405 @@ +//! Runtime values preserve Planner schemas; summary states are typed values too. +use crate::AggregateCore; +use crate::Error; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, +}; +use std::{cmp::Ordering, sync::Arc}; +pub type Schema = Arc; +#[derive(Clone, serde::Serialize, serde::Deserialize)] +pub enum Value { + Null, + Bool(bool), + Int64(i64), + Float64(f64), + Utf8(Arc), + Timestamp(i64), + Date(i32), + Interval { + months: i32, + days: i32, + nanos: i64, + }, + List(Arc<[Value]>), + Struct(Arc<[Value]>), + Map(Arc<[(Value, Value)]>), + #[serde(skip)] + Summary { + family: SummaryFamilyType, + state: Arc, + }, +} +impl std::fmt::Debug for Value { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Summary { family, .. } => f.debug_tuple("Summary").field(family).finish(), + _ => write!(f, "{:?}", self.key()), + } + } +} +impl Value { + pub fn bytes(&self) -> usize { + std::mem::size_of::() + + match self { + Self::Utf8(s) => s.len(), + Self::List(v) | Self::Struct(v) => v.iter().map(Self::bytes).sum(), + Self::Map(v) => v.iter().map(|(k, v)| k.bytes() + v.bytes()).sum(), + Self::Summary { state, .. } => state.approx_memory_bytes(), + _ => 0, + } + } + pub fn matches(&self, dtype: &DataType, nullable: bool) -> bool { + if matches!(self, Self::Null) { + return nullable || matches!(dtype, DataType::Null); + } + match (self, dtype) { + (Self::Bool(_), DataType::Bool) + | (Self::Int64(_), DataType::Int64) + | (Self::Float64(_), DataType::Float64) + | (Self::Utf8(_), DataType::Utf8) + | (Self::Timestamp(_), DataType::Timestamp) + | (Self::Date(_), DataType::Date) + | (Self::Interval { .. }, DataType::Interval) => true, + (Self::List(v), DataType::List { element }) => v + .iter() + .all(|v| v.matches(&element.dtype, element.nullable)), + (Self::Struct(v), DataType::Struct { fields }) => { + v.len() == fields.len() + && v.iter() + .zip(fields) + .all(|(v, f)| v.matches(&f.dtype, f.nullable)) + } + ( + Self::Map(v), + DataType::Map { + key, + value, + value_nullable, + }, + ) => v + .iter() + .all(|(k, v)| k.matches(key, false) && v.matches(value, *value_nullable)), + _ => false, + } + } + /// Stable typed equality key. Zero signs and NaN payloads form one group. + pub fn key(&self) -> Result, Error> { + let mut out = Vec::new(); + macro_rules! number { + ($tag:expr,$v:expr) => {{ + out.push($tag); + out.extend_from_slice(&$v.to_le_bytes()); + }}; + } + match self { + Self::Null => out.push(0), + Self::Bool(v) => out.extend([1, *v as u8]), + Self::Int64(v) => number!(2, v), + Self::Float64(v) => { + let bits = if *v == 0. { + 0 + } else if v.is_nan() { + f64::NAN.to_bits() + } else { + v.to_bits() + }; + number!(3, bits); + } + Self::Utf8(v) => { + out.push(4); + out.extend(v.as_bytes()); + } + Self::Timestamp(v) => number!(5, v), + Self::Date(v) => number!(6, v), + Self::Interval { + months, + days, + nanos, + } => { + number!(7, months); + number!(8, days); + number!(9, nanos); + } + Self::List(v) | Self::Struct(v) => { + out.push(if matches!(self, Self::List(_)) { + 10 + } else { + 11 + }); + for v in v.iter() { + let key = v.key()?; + out.extend((key.len() as u64).to_le_bytes()); + out.extend(key); + } + } + Self::Map(v) => { + out.push(12); + for (k, v) in v.iter() { + for value in [k, v] { + let key = value.key()?; + out.extend((key.len() as u64).to_le_bytes()); + out.extend(key); + } + } + } + Self::Summary { .. } => { + return Err(Error::Invalid( + "summary states cannot be grouping keys".into(), + )) + } + } + Ok(out) + } + pub fn compare(&self, other: &Self) -> Result { + Ok(match (self, other) { + (Self::Null, Self::Null) => Ordering::Equal, + (Self::Int64(a), Self::Int64(b)) | (Self::Timestamp(a), Self::Timestamp(b)) => a.cmp(b), + (Self::Float64(a), Self::Float64(b)) => { + if a == b { + Ordering::Equal + } else { + a.total_cmp(b) + } + } + (Self::Utf8(a), Self::Utf8(b)) => a.cmp(b), + (Self::Bool(a), Self::Bool(b)) => a.cmp(b), + (Self::Date(a), Self::Date(b)) => a.cmp(b), + (Self::Map(left), Self::Map(right)) => { + let mut result = Ordering::Equal; + for ((lk, lv), (rk, rv)) in left.iter().zip(right.iter()) { + result = lk.compare(rk)?; + if result != Ordering::Equal { + break; + } + result = match (lv, rv) { + (Self::Null, Self::Null) => Ordering::Equal, + (Self::Null, _) => Ordering::Greater, + (_, Self::Null) => Ordering::Less, + _ => lv.compare(rv)?, + }; + if result != Ordering::Equal { + break; + } + } + if result == Ordering::Equal { + left.len().cmp(&right.len()) + } else { + result + } + } + _ => { + return Err(Error::Operator( + "values do not have a supported common ordering".into(), + )) + } + }) + } +} +#[derive(Clone, Debug)] +pub struct Batch { + schema: Schema, + rows: Vec>, +} +impl Batch { + pub fn try_new(schema: Schema, rows: Vec>) -> Result { + validate_schema(&schema)?; + for row in &rows { + if row.len() != schema.fields.len() { + return Err(Error::Invalid( + "row width differs from Planner schema".into(), + )); + } + for (value, field) in row.iter().zip(&schema.fields) { + let matches = match (&field.dtype, value) { + (SummaryFamilyType::Plain(dtype), value) => { + value.matches(dtype, field.nullable) + } + (expected, Value::Summary { family, state }) => { + expected == family && validate_state(family, state.as_ref()).is_ok() + } + _ => false, + }; + if !matches { + return Err(Error::Invalid(format!( + "value differs from type of {}", + field.name + ))); + } + } + } + Ok(Self { schema, rows }) + } + pub fn schema(&self) -> &Schema { + &self.schema + } + pub fn rows(&self) -> &[Vec] { + &self.rows + } + pub fn bytes(&self) -> usize { + std::mem::size_of::() + + self.rows.capacity() * std::mem::size_of::>() + + self + .rows + .iter() + .flat_map(|r| r.iter()) + .map(Value::bytes) + .sum::() + } +} +pub(crate) fn group_key(row: &[Value], columns: &[usize]) -> Result>, Error> { + columns + .iter() + .map(|&i| { + row.get(i) + .ok_or_else(|| Error::Invalid("group column out of range".into()))? + .key() + }) + .collect() +} + +pub(crate) use crate::capability::validate_native_family as validate_family; + +fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Result<(), Error> { + use crate::summary_kernels::{ + datasketches_kll::DatasketchesKLLAccumulator, dd_sketch::DDSketchAccumulator, + exact::ExactAccumulator, hll_sketch::HllSketchAccumulator, + }; + use planner_types::post_asap::SketchParams; + validate_family(family)?; + let valid = match family { + SummaryFamilyType::Sketch(kind, _) + if matches!( + kind.params(), + SketchParams::CmsWithHeap { .. } | SketchParams::CountSketchWithHeap { .. } + ) => + { + use crate::summary_kernels::weighted_frequency::WeightedFrequency; + let (algorithm, width, depth, capacity) = WeightedFrequency::configuration(kind)?; + state + .as_any() + .downcast_ref::() + .is_some_and(|state| { + state.algorithm() == algorithm && state.shape() == (width, depth, capacity) + }) + } + + SummaryFamilyType::ExactAggregate(..) => state + .as_any() + .downcast_ref::() + .is_some_and(|s| s.family() == family && !s.is_keyed()), + SummaryFamilyType::Sketch(kind, _) => match kind.params() { + SketchParams::Kll { k } => state + .as_any() + .downcast_ref::() + .is_some_and(|s| u32::from(s.inner.k()) == *k), + SketchParams::DDSketch { alpha } => state + .as_any() + .downcast_ref::() + .is_some_and(|s| s.inner.alpha == *alpha), + SketchParams::Hll { precision } => state + .as_any() + .downcast_ref::() + .is_some_and(|s| s.inner.precision == u32::from(*precision)), + _ => false, + }, + _ => false, + }; + if valid { + Ok(()) + } else { + Err(Error::Invalid( + "state payload differs from declared family, parameters or population layout".into(), + )) + } +} + +pub(crate) fn validate_schema(schema: &Schema) -> Result<(), Error> { + if schema.time_index.is_some_and(|index| { + schema + .fields + .get(index) + .is_none_or(|field| field.dtype != SummaryFamilyType::Plain(DataType::Timestamp)) + }) { + return Err(Error::Invalid( + "time index must name a Timestamp column".into(), + )); + } + for field in &schema.fields { + if !matches!(field.dtype, SummaryFamilyType::Plain(_)) { + validate_family(&field.dtype)?; + if field.nullable { + return Err(Error::Invalid( + "nullable summary states are not supported".into(), + )); + } + } + } + Ok(()) +} + +pub(crate) fn field(schema: &Schema, column: usize) -> Result<&SummaryField, Error> { + schema + .fields + .get(column) + .ok_or_else(|| Error::Invalid("column out of range".into())) +} +pub(crate) fn plain(schema: &Schema, column: usize) -> Result<(&DataType, bool), Error> { + let f = field(schema, column)?; + let SummaryFamilyType::Plain(dtype) = &f.dtype else { + return Err(Error::Invalid("plain value required".into())); + }; + Ok((dtype, f.nullable)) +} + +#[cfg(test)] +mod weighted_state_tests { + use super::*; + use crate::summary_kernels::weighted_frequency::{FrequencyAlgorithm, WeightedFrequency}; + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + + // A state cannot acquire a different family or shape merely by relabeling its batch. + #[test] + fn weighted_state_family_and_shape_must_match() { + let cms = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 32, + depth: 5, + heap_size: 8, + }, + ), + Default::default(), + ); + let cs = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 32, + depth: 5, + heap_size: 8, + }, + ), + Default::default(), + ); + let state = WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 32, 5, 8).unwrap(); + assert!(validate_state(&cs, &state).is_ok()); + assert!(validate_state(&cms, &state).is_err()); + let wrong_shape = + WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 64, 5, 8).unwrap(); + assert!(validate_state(&cs, &wrong_shape).is_err()); + let even_depth = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 32, + depth: 4, + heap_size: 8, + }, + ), + Default::default(), + ); + assert!(validate_family(&even_depth).is_err()); + } +} diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/asap-physical-operators/tests/blocking_resources.rs new file mode 100644 index 00000000..7a312892 --- /dev/null +++ b/crates/asap-physical-operators/tests/blocking_resources.rs @@ -0,0 +1,242 @@ +//! Blocking operators enforce resources before returning their first batch. +use asap_physical_operators::{ + operators::Operator, + plan::{PhysicalDag, PhysicalOperator}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Schema, Value}, + Error, +}; +use futures::{executor::block_on, FutureExt, StreamExt}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{DataType, JoinKind, Predicate, QueryExpr, ScalarValue}, +}; +use std::sync::Arc; + +fn schema(width: usize) -> Schema { + Arc::new(SummarySchema { + fields: (0..width) + .map(|i| SummaryField { + name: format!("v{i}"), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: false, + }) + .collect(), + time_index: None, + }) +} +fn context(max_bytes: usize) -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_bytes, + ..Limits::default() + }, + ) + .unwrap() +} +fn source(n: usize) -> PhysicalDag<'static, Batch, Schema> { + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + schema(1), + vec![Batch::try_new(schema(1), vec![vec![Value::Int64(1)]; n]).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag +} +fn cross_join() -> Operator { + Operator::relational_join( + schema(1), + schema(1), + JoinKind::Cross, + &Predicate(std::rc::Rc::new(QueryExpr::Literal(ScalarValue::Boolean( + true, + )))), + schema(2), + ) + .unwrap() +} + +// Even callers starting an operator directly cannot bypass its workspace budget. +#[test] +fn join_reserves_result_growth_before_returning_output() { + let sources = source(64); + let run = context(32 * 1024); + let inputs = sources.execute(&[0, 0], run.clone()).unwrap(); + let join = cross_join(); + let mut output = join.start(inputs, run.clone()).unwrap(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::MemoryLimit)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); +} + +// A single large input batch must not monopolize the worker during a join. +#[test] +fn join_yields_during_computation_and_observes_cancellation() { + let sources = source(64); + let run = context(16 * 1024 * 1024); + let inputs = sources.execute(&[0, 0], run.clone()).unwrap(); + let join = cross_join(); + let mut output = join.start(inputs, run.clone()).unwrap(); + assert!( + output.next().now_or_never().is_none(), + "join should yield before producing all 4096 rows" + ); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); +} + +// Sorting and grouping yield even for one large batch. +#[test] +fn blocking_reductions_yield_and_release_memory_on_cancellation() { + use asap_physical_operators::{ + operators::{Reduction, SortKey}, + plan::PhysicalOperator, + }; + let operators = vec![ + Operator::sort( + schema(1), + vec![SortKey { + column: 0, + descending: false, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + Operator::aggregate(schema(1), vec![], vec![("sum".into(), Reduction::Sum(0))]).unwrap(), + ]; + for operator in operators { + let sources = source(768); + let run = context(16 * 1024 * 1024); + let inputs = sources.execute(&[0], run.clone()).unwrap(); + let mut output = operator.start(inputs, run.clone()).unwrap(); + assert!(output.next().now_or_never().is_none()); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); + } +} + +// Merge-sort rounds preserve input order for tied keys across chunk boundaries. +#[test] +fn cooperative_sort_preserves_ties_across_chunks() { + use asap_physical_operators::operators::SortKey; + let batch = Batch::try_new( + schema(2), + (0..1025) + .rev() + .map(|i| vec![Value::Int64(i % 3), Value::Int64(i)]) + .collect(), + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(schema(2), vec![batch]).unwrap()) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::sort( + schema(2), + vec![SortKey { + column: 0, + descending: false, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + ) + .unwrap(); + let mut output = dag + .execute(&[1], context(16 * 1024 * 1024)) + .unwrap() + .remove(0); + let batch = block_on(output.next()).unwrap().unwrap(); + let expected = (0..3) + .flat_map(|key| (0..1025).rev().filter(move |i| i % 3 == key)) + .collect::>(); + for (row, expected) in batch.rows().iter().zip(expected) { + assert!(matches!(row[1], Value::Int64(i) if i == expected)); + } + assert_eq!(batch.rows().len(), 1025); +} + +// The integrated weighted-summary path obeys the same cooperative cancellation contract. +#[test] +fn weighted_summary_build_yields_within_a_batch() { + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let input = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "item".into(), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: false, + }, + SummaryField { + name: "weight".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }, + ], + time_index: None, + }); + let mut sources = PhysicalDag::default(); + let batch = Batch::try_new( + input.clone(), + (0..1500) + .map(|i| vec![Value::Int64(i % 8), Value::Float64(0.25)]) + .collect(), + ) + .unwrap(); + sources + .add( + 0, + vec![], + Operator::source(input.clone(), vec![batch]).unwrap(), + ) + .unwrap(); + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 64, + depth: 3, + heap_size: 8, + }, + ), + Default::default(), + ); + let operator = Operator::keyed_summary_build(input, family, 1, vec![0], vec![]).unwrap(); + let run = context(16 * 1024 * 1024); + let inputs = sources.execute(&[0], run.clone()).unwrap(); + let mut output = operator.start(inputs, run.clone()).unwrap(); + assert!(output.next().now_or_never().is_none()); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); +} diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs new file mode 100644 index 00000000..322bd6fe --- /dev/null +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -0,0 +1,401 @@ +//! Spatial heap weights come from a fresh instant vector, never sample history. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{ + promql_rows::{decode_series_identity, series_row, SERIES_IDENTITY_COLUMN}, + CompiledPhysicalDag, InputContract, Source, + }, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{post_asap::*, pre_asap::DataType}; +use std::{collections::BTreeMap, sync::Arc}; + +fn schema() -> Arc { + Arc::new(SummarySchema { + fields: [ + ("ts", DataType::Timestamp), + ("value", DataType::Float64), + ("job", DataType::Utf8), + (SERIES_IDENTITY_COLUMN, DataType::Utf8), + ] + .into_iter() + .map(|(name, dtype)| SummaryField { + name: name.into(), + dtype: SummaryFamilyType::Plain(dtype), + nullable: false, + }) + .collect(), + time_index: Some(0), + }) +} +fn run(program: &CompiledPhysicalDag, data: Batch, end: i64) -> Result, String> { + let recovered = serde_json::from_slice::( + &serde_json::to_vec(&program).map_err(|e| e.to_string())?, + ) + .map_err(|e| e.to_string())?; + let input_id = recovered.input_contracts().next().unwrap().0; + let graph = recovered + .instantiate(BTreeMap::from([( + input_id, + Box::new(Operator::source(data.schema().clone(), vec![data]).unwrap()) as Source<'_>, + )])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: end, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = graph.execute(recovered.roots(), context).unwrap().remove(0); + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push((*batch.map_err(|e| e.to_string())?).clone()); + } + Ok(batches) + }) +} +fn input(samples: &[(&str, i64, f64)]) -> Batch { + let schema = schema(); + let rows = samples + .iter() + .map(|(instance, time, value)| { + series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("hidden_instance".into(), (*instance).into()), + ]), + *time, + *value, + ) + .unwrap() + }) + .collect(); + Batch::try_new(schema, rows).unwrap() +} +fn snapshot_plan() -> CompiledPhysicalDag { + CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema()))]), + BTreeMap::from([( + 1, + ( + vec![0], + Operator::current_series(schema(), 3, 0, 1, 60_000).unwrap(), + ), + )]), + vec![1], + ) + .unwrap() +} + +// Replacement, expiry and stale markers act before sketch updates. Hidden labels +// survive even when every series has the same projected `job` value. +#[test] +fn latest_snapshot_replaces_decreases_expires_and_retains_full_identity() { + let plan = snapshot_plan(); + let batches = run( + &plan, + input(&[ + ("decrease", 10_000, 100.), + ("decrease", 50_000, 1.), + ("steady", 40_000, 20.), + ("expired", 0, 1_000.), + ("stale", 20_000, 500.), + ("stale", 55_000, f64::from_bits(0x7ff0_0000_0000_0002)), + ("future", 60_001, 2_000.), + ]), + 60_000, + ) + .unwrap(); + let values = batches + .iter() + .flat_map(|batch| batch.rows()) + .map(|row| { + let Value::Utf8(identity) = &row[3] else { + panic!() + }; + let Value::Float64(value) = row[1] else { + panic!() + }; + assert!(matches!(row[0], Value::Timestamp(60_000))); + ( + decode_series_identity(identity).unwrap()["hidden_instance"].clone(), + value, + ) + }) + .collect::>(); + assert_eq!( + values, + BTreeMap::from([("decrease".into(), 1.), ("steady".into(), 20.)]) + ); + assert!(run(&plan, input(&[("steady", 40_000, 20.)]), 100_000) + .unwrap() + .iter() + .all(|batch| batch.rows().is_empty())); + assert!(run( + &plan, + input(&[("conflict", 50_000, 1.), ("conflict", 50_000, 2.)]), + 60_000 + ) + .is_err()); +} + +#[test] +fn spatial_heap_ranks_latest_values_in_independent_runs() { + for algorithm in [ + SketchAlgorithm::CmsWithHeap, + SketchAlgorithm::CountSketchWithHeap, + ] { + let params = match algorithm { + SketchAlgorithm::CmsWithHeap => SketchParams::CmsWithHeap { + width: 2048, + depth: 5, + heap_size: 100, + }, + _ => SketchParams::CountSketchWithHeap { + width: 2048, + depth: 5, + heap_size: 100, + }, + }; + let family = + SummaryFamilyType::Sketch(SketchKind::new(algorithm, params), Default::default()); + let build = Operator::keyed_summary_build(schema(), family, 1, vec![3], vec![2]).unwrap(); + let output = Arc::new(SummarySchema { + fields: vec![ + schema().fields[2].clone(), + schema().fields[3].clone(), + schema().fields[1].clone(), + ], + time_index: None, + }); + let read = Operator::keyed_readout(build.schema(), 1, 1, output).unwrap(); + let plan = CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema()))]), + BTreeMap::from([ + ( + 1, + ( + vec![0], + Operator::current_series(schema(), 3, 0, 1, 60_000).unwrap(), + ), + ), + (2, (vec![1], build)), + (3, (vec![2], read)), + ]), + vec![3], + ) + .unwrap(); + for (samples, end, winner, score) in [ + ( + vec![("a", 10_000, 100.), ("a", 50_000, 1.), ("b", 50_000, 20.)], + 60_000, + "b", + 20., + ), + ( + vec![("a", 110_000, 3.), ("b", 50_000, 20.)], + 120_000, + "a", + 3., + ), + ] { + let batches = run(&plan, input(&samples), end).unwrap(); + let rows = batches + .iter() + .flat_map(|batch| batch.rows()) + .collect::>(); + assert_eq!(rows.len(), 1); + let Value::Utf8(encoded) = &rows[0][1] else { + panic!() + }; + assert_eq!( + decode_series_identity(encoded).unwrap()["hidden_instance"], + winner + ); + assert!(matches!(rows[0][2], Value::Float64(actual) if actual == score)); + } + } +} + +// Blocking membership selection shares the run's cancellation and byte budget. +#[test] +fn current_series_observes_resource_limits() { + use asap_physical_operators::Error; + let plan = snapshot_plan(); + for cancelled in [false, true] { + let data = input(&[("one", 50_000, 1.)]); + let graph = plan + .instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(data.schema().clone(), vec![data]).unwrap()) + as Source<'_>, + )])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 60_000, + revision: 0, + }, + Limits { + max_bytes: if cancelled { 1 << 20 } else { 1 }, + ..Limits::default() + }, + ) + .unwrap(); + if cancelled { + context.cancel(); + } + let result = match graph.execute(&[1], context.clone()) { + Err(error) => Err(error), + Ok(mut streams) => block_on(streams.remove(0).next()).unwrap().map(|_| ()), + }; + assert!(matches!( + (cancelled, result), + (true, Err(Error::Cancelled)) | (false, Err(Error::MemoryLimit)) + )); + assert_eq!(context.retained_bytes(), 0); + } +} + +#[test] +fn identity_encoding_is_lossless_and_rejects_noncanonical_inputs() { + use asap_physical_operators::physical_planner::promql_rows::encode_series_identity; + let labels = BTreeMap::from([ + ("a".into(), "quote\"slash\\".into()), + ("other".into(), "".into()), + ]); + assert_eq!( + decode_series_identity(&encode_series_identity(&labels).unwrap()).unwrap(), + labels + ); + for invalid in [ + "[]", + "{\"a\":1}", + "{\"a\":\"x\",\"a\":\"x\"}", + "{ \"a\":\"x\"}", + ] { + assert!(decode_series_identity(invalid).is_err(), "{invalid}"); + } +} + +// The actual Planner population candidate lowers to native operators; this +// test does not manually assemble the computation or its dependency edges. +#[test] +fn planner_current_series_candidate_compiles_with_dynamic_identity() { + use asap_physical_operators::physical_planner::{compile, promql_rows::with_series_identity}; + use planner_types::{types::AccuracyTarget, workload::*}; + use std::rc::Rc; + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("topk by(job)(1, m)".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(60_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let original = asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0); + let open_root = Rc::new(original.clone()); + let open_selected = + asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&open_root), + ) + .candidate(&open_root) + .unwrap(); + let snapshot_program = + asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + &open_selected, + ) + .unwrap(); + let encoded = String::from_utf8(serde_json::to_vec(&snapshot_program).unwrap()).unwrap(); + assert!( + !encoded.contains("CurrentSeries"), + "maintained input must not be rebuilt" + ); + assert!(encoded.contains("Sort") && encoded.contains("Limit")); + assert_eq!(snapshot_program.input_contracts().count(), 1); + let root = Rc::new(with_series_identity(&original).unwrap()); + let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&root), + ) + .candidate(&root) + .unwrap(); + let logical = compile_post_asap_dag(&selected).unwrap(); + let raw = logical + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .unwrap(); + let raw_schema = Arc::new(raw.output_schema.clone()); + let physical = compile( + &logical, + BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(raw_schema.clone()), + )]), + &[u64::from(logical.root.0)], + ) + .unwrap(); + let bytes = String::from_utf8(serde_json::to_vec(&physical).unwrap()).unwrap(); + assert!(bytes.contains("CurrentSeries")); + assert!(bytes.contains("Sort")); + assert!(bytes.contains("Limit")); + let rows = [("a", 10_000, 100.), ("a", 50_000, 1.), ("b", 50_000, 20.)] + .into_iter() + .map(|(member, at, value)| { + series_row( + &raw_schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("unreferenced".into(), member.into()), + ]), + at, + value, + ) + .unwrap() + }) + .collect(); + let batches = run(&physical, Batch::try_new(raw_schema, rows).unwrap(), 60_000).unwrap(); + let rows = batches + .iter() + .flat_map(|batch| batch.rows()) + .collect::>(); + assert_eq!(rows.len(), 1); + assert!(matches!(rows[0][1], Value::Float64(20.))); + let id = batches[0] + .schema() + .fields + .iter() + .position(|field| field.name == SERIES_IDENTITY_COLUMN) + .unwrap(); + let Value::Utf8(encoded) = &rows[0][id] else { + panic!() + }; + assert_eq!( + decode_series_identity(encoded).unwrap()["unreferenced"], + "b" + ); +} diff --git a/crates/asap-physical-operators/tests/deployment.rs b/crates/asap-physical-operators/tests/deployment.rs new file mode 100644 index 00000000..0c261a03 --- /dev/null +++ b/crates/asap-physical-operators/tests/deployment.rs @@ -0,0 +1,90 @@ +//! Exercise the public library without a backend server, store, or scheduler. +use asap_physical_operators::planner::{ + post_asap::{ + GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryFamilyType, + SummaryUpdate, + }, + pre_asap::ColumnRef, +}; +use asap_physical_operators::{factory::create_planner_accumulator, AggregateCore}; + +fn family(k: u32) -> SummaryFamilyType { + SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), + GroupingStrategy::PerSubpopulationInstance, + ) +} +fn build(values: &[f64]) -> Box { + let mut operator = create_planner_accumulator( + &family(512), + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .unwrap(); + for (at, value) in values.iter().enumerate() { + operator.validate_single_input(*value).unwrap(); + operator.update_single(*value, at as i64); + } + operator.into_accumulator() +} +fn read(state: &dyn AggregateCore) -> f64 { + state + .estimate(&asap_physical_operators::planner::post_asap::SketchQuery::Quantile { q: 0.5 }) + .unwrap() +} + +// The same kernels work when every build is query-time, when only a prefix +// was precomputed, and when all state was precomputed before the readout. +#[test] +fn raw_partial_and_fully_precomputed_use_the_same_kernels() { + let raw: Vec = (0..128).map(f64::from).collect(); + let raw_only = build(&raw); + let stored_prefix = build(&raw[..64]); + let query_time_suffix = build(&raw[64..]); + let partial = stored_prefix.merge_with(&*query_time_suffix).unwrap(); + let stored_complete = build(&raw); + assert_eq!(read(&*raw_only), read(&*partial)); + assert_eq!(read(&*partial), read(&*stored_complete)); + assert!((read(&*raw_only) - 64.0).abs() <= 1.0); +} + +// A compiler must reject invalid physical parameters before starting execution. +#[test] +fn invalid_kll_parameters_are_rejected_at_binding() { + let result = create_planner_accumulator( + &family(0), + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ); + assert!(result.is_err()); +} + +// Native CountSketch supports the confidence-sized depth used by the backend; +// a packed-wire column-bit budget must not be imposed on this constructor. +#[test] +fn native_count_sketch_dimensions_are_not_packed_wire_dimensions() { + use asap_physical_operators::planner::post_asap::SummaryInputExpr; + use asap_physical_operators::KeyByLabelValues; + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 1200, + depth: 55, + heap_size: 3, + }, + ), + Default::default(), + ); + let mut update = SummaryUpdate::column(ColumnRef::SampleValue); + update.item = Some(SummaryInputExpr::Column(ColumnRef::Named("host".into()))); + let mut operator = create_planner_accumulator(&family, &update, &Default::default()).unwrap(); + let key = KeyByLabelValues::new_with_labels(vec!["a".into()]); + operator.update_keyed(&key, 7.0, 1000); + let state = operator.into_accumulator(); + let state = state + .as_any() + .downcast_ref::() + .unwrap(); + assert_eq!(state.query_key(&key), 7.0); +} diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs new file mode 100644 index 00000000..b026f47f --- /dev/null +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -0,0 +1,340 @@ +//! Planner-selected PromQL computation compiles from the timed DAG alone; +//! the deployment supplies only raw rows at the ingestion frontier. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{compile, promql_rows, CompiledPhysicalDag, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{post_asap::*, pre_asap::QueryExpr, types::AccuracyTarget, workload::*}; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; + +fn lower(query: &str) -> QueryExpr { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(60_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0) +} + +/// The first exact summary candidate, as Planner selection would hand it over. +fn exact_dag(query: &str) -> PostAsapDag { + use asap_aware_mapping::{Replacement, ReplacementStrategy, TargetSubDAG}; + let expression = lower(query); + let root = Rc::new(promql_rows::with_series_identity(&expression).unwrap_or(expression)); + asap_aware_mapping::SketchAlgorithmStrategy::new(&asap_aware_mapping::DefaultCostModel) + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .find_map(|candidate| match candidate.replacement { + Replacement::Summary(node) => { + let dag = compile_post_asap_dag(&node).ok()?; + dag.nodes + .iter() + .all(|n| !matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { family, .. } if !matches!(family, SummaryFamilyType::ExactAggregate(..)))) + .then_some(dag) + } + _ => None, + }) + .unwrap() +} + +fn population_dag(query: &str) -> PostAsapDag { + let root = Rc::new(promql_rows::with_series_identity(&lower(query)).unwrap()); + let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&root), + ) + .candidate(&root) + .unwrap(); + compile_post_asap_dag(&selected).unwrap() +} + +/// Raw scan nodes are the frontier; everything above them is compiled. +fn raw_inputs(dag: &PostAsapDag) -> Vec<(u64, Arc, String)> { + dag.nodes + .iter() + .filter_map(|node| match &node.payload { + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::TimeRange { child, .. }, + } => match child.as_ref() { + QueryExpr::Scan { + source: planner_types::pre_asap::Source::TimeSeries { metric }, + .. + } => Some(( + u64::from(node.id.0), + Arc::new(node.output_schema.clone()), + metric.clone(), + )), + _ => None, + }, + _ => None, + }) + .collect() +} + +type Sample = (&'static str, &'static str, &'static str, i64, f64); + +/// Compile, round-trip, bind raw `(metric, job, instance, ts, value)` samples, +/// and return `(job, value)` rows of the root. +fn run(dag: &PostAsapDag, samples: &[Sample], end: i64) -> Result, String> { + let inputs = raw_inputs(dag); + let program = compile( + dag, + inputs + .iter() + .map(|(id, schema, _)| (*id, InputContract::bounded(schema.clone()))) + .collect(), + &[u64::from(dag.root.0)], + ) + .map_err(|e| e.to_string())?; + let program: CompiledPhysicalDag = + serde_json::from_slice(&serde_json::to_vec(&program).unwrap()).unwrap(); + let sources = inputs + .iter() + .map(|(id, schema, metric)| { + let rows = samples + .iter() + .filter(|sample| sample.0 == metric) + .map(|(name, job, instance, at, value)| { + let labels = BTreeMap::from([ + ("__name__".to_string(), name.to_string()), + ("job".into(), job.to_string()), + ("instance".into(), instance.to_string()), + ]); + if schema + .fields + .iter() + .any(|f| f.name == promql_rows::SERIES_IDENTITY_COLUMN) + { + promql_rows::series_row(schema, &labels, *at, *value).unwrap() + } else { + schema + .fields + .iter() + .enumerate() + .map(|(i, f)| match f.name.as_str() { + _ if Some(i) == schema.time_index => Value::Timestamp(*at), + "value" => Value::Float64(*value), + label => Value::Utf8(labels[label].clone().into()), + }) + .collect() + } + }) + .collect(); + let batch = Batch::try_new(schema.clone(), rows).unwrap(); + ( + *id, + Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect(); + let graph = program.instantiate(sources).map_err(|e| e.to_string())?; + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: end, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = graph + .execute(program.roots(), context) + .map_err(|e| e.to_string())? + .remove(0); + let mut rows = BTreeMap::new(); + while let Some(batch) = stream.next().await { + let batch = batch.map_err(|e| e.to_string())?; + let job = batch.schema().fields.iter().position(|f| f.name == "job"); + for row in batch.rows() { + let key = match job.map(|i| &row[i]) { + Some(Value::Utf8(job)) => job.to_string(), + _ => String::new(), + }; + let value = match row.last() { + Some(Value::Float64(v)) => *v, + Some(Value::Int64(v)) => *v as f64, + other => return Err(format!("unexpected value {other:?}")), + }; + assert!(rows.insert(key, value).is_none(), "duplicate output group"); + } + } + Ok(rows) + }) +} + +const SAMPLES: &[Sample] = &[ + ("m", "api", "a", 10_000, 4.), + ("m", "api", "a", 50_000, 1.), + ("m", "api", "b", 40_000, 7.), + ("m", "api", "c", 30_000, 2.), + ("m", "db", "d", 20_000, 5.), +]; + +fn reference(pairs: &[(&str, f64)]) -> BTreeMap { + pairs.iter().map(|(k, v)| (k.to_string(), *v)).collect() +} + +// Current-series aggregates read the latest member values, matching the +// backend CurrentSeriesStore formulas (PromQL quantile interpolation). +#[test] +fn population_aggregates_match_current_series_reference() { + // Latest values: api = {a: 1, b: 7, c: 2}; db = {d: 5}. + for (query, expected) in [ + ("sum by (job) (m)", reference(&[("api", 10.), ("db", 5.)])), + ("count by (job) (m)", reference(&[("api", 3.), ("db", 1.)])), + ( + "avg by (job) (m)", + reference(&[("api", 10. / 3.), ("db", 5.)]), + ), + // Sorted api = [1, 2, 7]; rank 0.25 * 2 = 0.5 → 1.5. + ( + "quantile by (job) (0.25, m)", + reference(&[("api", 1.5), ("db", 5.)]), + ), + ] { + let dag = population_dag(query); + assert_eq!(run(&dag, SAMPLES, 60_000).unwrap(), expected, "{query}"); + } +} + +// A global readout of an empty population is an empty vector, as in PromQL. +#[test] +fn global_population_aggregate_of_no_members_is_empty() { + // Latest values are [1, 2, 5, 7] at 60s; every member has expired by 1000s. + for (query, expected) in [("sum(m)", 15.), ("count(m)", 4.), ("quantile(0.5, m)", 3.5)] { + let dag = population_dag(query); + let live = run(&dag, SAMPLES, 60_000).unwrap(); + assert_eq!(live, reference(&[("", expected)]), "{query}"); + assert!(run(&dag, SAMPLES, 1_000_000).unwrap().is_empty(), "{query}"); + } +} + +// Scalar operands on either side apply to every grouped value, including negation. +#[test] +fn scalar_literal_arithmetic_applies_to_grouped_values() { + // sum_over_time over 5m per job: api = 4 + 1 + 7 + 2 = 14, db = 5. + for (query, expected) in [ + ( + "sum by (job) (sum_over_time(m[5m])) * 2", + reference(&[("api", 28.), ("db", 10.)]), + ), + ( + "100 - sum by (job) (sum_over_time(m[5m]))", + reference(&[("api", 86.), ("db", 95.)]), + ), + ( + "-sum by (job) (sum_over_time(m[5m]))", + reference(&[("api", -14.), ("db", -5.)]), + ), + ] { + assert_eq!( + run(&exact_dag(query), SAMPLES, 60_000).unwrap(), + expected, + "{query}" + ); + } +} + +// Grouped vectors match one-to-one on labels; unmatched groups are dropped and +// unchecked division by zero yields +Inf as in PromQL. +#[test] +fn grouped_vector_arithmetic_matches_labels() { + let samples: &[Sample] = &[ + ("a", "api", "x", 10_000, 6.), + ("a", "api", "y", 20_000, 3.), + ("a", "db", "x", 10_000, 1.), + ("a", "web", "x", 10_000, 1.), + ("b", "api", "x", 10_000, 3.), + ("b", "db", "x", 10_000, 0.), + ("b", "cache", "x", 10_000, 1.), + ]; + let dag = + exact_dag("sum by (job) (sum_over_time(a[5m])) / sum by (job) (sum_over_time(b[5m]))"); + assert_eq!( + run(&dag, samples, 60_000).unwrap(), + reference(&[("api", 3.), ("db", f64::INFINITY)]) + ); +} + +// Exact observation counts finalize to the Float64 value PromQL declares, +// then roll up per job: api has 2 + 1 + 1 samples in 5m, db has 1. +#[test] +fn exact_count_finalizes_to_declared_float_value() { + let mut dag = exact_dag("sum by (job) (count_over_time(m[5m]))"); + let finalize = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator + } + ) + }) + .unwrap() + .clone(); + let root = dag.nodes.iter().find(|n| n.id == dag.root).unwrap().clone(); + let mut edge = dag + .edges + .iter() + .find(|e| e.producer == finalize.id) + .unwrap() + .clone(); + // Read the rolled-up exact state the same way the query path does. + let mut read = finalize.clone(); + read.id = PostAsapNodeId(root.id.0 + 1); + read.output_schema = root.output_schema.clone(); + read.output_schema.fields.last_mut().unwrap().dtype = + SummaryFamilyType::Plain(planner_types::pre_asap::DataType::Float64); + edge.producer = root.id; + edge.consumer = read.id; + edge.intermediate_schema = root.output_schema.clone(); + edge.data_state = root.output_state; + dag.root = read.id; + dag.nodes.push(read); + dag.edges.push(edge); + assert_eq!( + run(&dag, SAMPLES, 60_000).unwrap(), + reference(&[("api", 4.), ("db", 1.)]) + ); +} + +// Comparisons need filter/bool semantics that `Binary` does not carry, so +// they fail at compile time instead of emitting 0/1 values. +#[test] +fn row_comparison_fails_closed() { + let mut dag = exact_dag("sum by (job) (sum_over_time(m[5m])) * 2"); + for node in &mut dag.nodes { + if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { + operator.kind = planner_types::pre_asap::BinaryOpKind::Compare( + planner_types::pre_asap::CompareOpKind::Gt, + ); + } + } + let error = run(&dag, SAMPLES, 60_000).unwrap_err(); + assert!(error.contains("comparison"), "{error}"); +} diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs new file mode 100644 index 00000000..652880e0 --- /dev/null +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -0,0 +1,1445 @@ +//! Acceptance tests use the library directly, without either backend engine. +use asap_physical_operators::{ + dag::{ + operators::{Expression, Operator, Reduction, SortKey}, + values::{Batch, Schema, Value}, + Limits, PhysicalDag, RunContext, Scope, + }, + Statistic, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{ExactKind, ExactParams, SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, +}; +use std::sync::Arc; +fn schema(fields: &[(&str, DataType, bool)]) -> Schema { + Arc::new(SummarySchema { + fields: fields + .iter() + .map(|(name, dtype, nullable)| SummaryField { + name: (*name).into(), + dtype: SummaryFamilyType::Plain(dtype.clone()), + nullable: *nullable, + }) + .collect(), + time_index: None, + }) +} +fn run(dag: &PhysicalDag<'_, Batch, Schema>, root: u64, scope: Scope) -> Vec> { + let context = RunContext::new( + scope, + Limits { + max_buffered_batches: 1, + ..Limits::default() + }, + ) + .unwrap(); + block_on(async { + let mut stream = dag.execute(&[root], context.clone()).unwrap().remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend(batch.unwrap().rows().iter().cloned()); + } + assert_eq!(context.retained_bytes(), 0); + rows + }) +} +fn query() -> Scope { + Scope::Query { + evaluation_time_ms: 1000, + revision: 2, + } +} +fn floats(rows: &[Vec], column: usize) -> Vec { + rows.iter() + .map(|r| { + if let Value::Float64(v) = r[column] { + v + } else { + panic!("not Float64") + } + }) + .collect() +} + +// Sort followed by partitioned Limit implements ranking independently per group. +#[test] +fn grouped_sort_limit_across_batches() { + let schema = schema(&[ + ("group", DataType::Int64, false), + ("score", DataType::Float64, false), + ]); + let batches = [ + vec![(1, 1.), (2, 4.), (1, 9.)], + vec![(2, 8.), (1, 5.), (2, 2.)], + ] + .into_iter() + .map(|rows| { + Batch::try_new( + schema.clone(), + rows.into_iter() + .map(|(g, v)| vec![Value::Int64(g), Value::Float64(v)]) + .collect(), + ) + .unwrap() + }) + .collect(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(schema.clone(), batches).unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::sort( + schema.clone(), + vec![SortKey { + column: 1, + descending: true, + nulls_first: false, + }], + vec![0], + ) + .unwrap(), + ) + .unwrap(); + dag.add(2, vec![1], Operator::limit(schema, 1, 1, vec![0]).unwrap()) + .unwrap(); + assert_eq!(floats(&run(&dag, 2, query()), 1), vec![5., 4.]); +} + +// The same computation runs in either engine scope with fresh per-run state. +#[test] +fn summary_construction_merge_and_readout_at_both_phases() { + let schema = schema(&[("v", DataType::Float64, false)]); + let batches = (1..=20) + .map(|v| Batch::try_new(schema.clone(), vec![vec![Value::Float64(v as f64)]]).unwrap()) + .collect(); + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let build = Operator::summary_build(schema.clone(), family, 0, None, vec![]).unwrap(); + let state = build.schema(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(schema, batches).unwrap()) + .unwrap(); + dag.add(1, vec![0], build).unwrap(); + dag.add(2, vec![1, 1], Operator::union(state.clone(), 2).unwrap()) + .unwrap(); + dag.add( + 3, + vec![2], + Operator::summary_merge(state.clone(), 0, vec![]).unwrap(), + ) + .unwrap(); + dag.add( + 4, + vec![3], + Operator::readout( + state, + 0, + asap_physical_operators::operators::ReadoutQuery::Exact( + asap_physical_operators::summary_kernels::exact::ExactReadout { + statistic: Statistic::Sum, + lookback_ms: None, + }, + ), + ) + .unwrap(), + ) + .unwrap(); + for scope in [ + query(), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1000, + revision: 2, + }, + ] { + assert_eq!(floats(&run(&dag, 4, scope), 0), vec![420.]); + } +} + +// A semi-join can consume two branches of one producer with a one-batch buffer. +#[test] +fn diamond_semijoin_preserves_left_values_and_multiplicity() { + let schema = schema(&[("key", DataType::Int64, false)]); + let batches = [1, 2, 2, 3] + .into_iter() + .map(|v| Batch::try_new(schema.clone(), vec![vec![Value::Int64(v)]]).unwrap()) + .collect(); + let filter = Operator::filter( + schema.clone(), + Expression::Equal( + Box::new(Expression::Column(0)), + Box::new(Expression::Literal { + value: Value::Int64(2), + dtype: DataType::Int64, + }), + ), + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(schema.clone(), batches).unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], filter).unwrap(); + dag.add( + 2, + vec![0, 1], + Operator::semi_join(schema.clone(), schema, vec![(0, 0)]).unwrap(), + ) + .unwrap(); + let rows = run(&dag, 2, query()); + assert_eq!(rows.len(), 2); + assert!(rows.iter().all(|r| matches!(r[0], Value::Int64(2)))); +} + +// Integer aggregation must not silently lose precision through Float64. +#[test] +fn exact_integer_and_empty_extrema() { + let schema = schema(&[("v", DataType::Int64, false)]); + let aggregate = Operator::aggregate( + schema.clone(), + vec![], + vec![("sum".into(), Reduction::Sum(0))], + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + let value = 9_007_199_254_740_993; + dag.add( + 0, + vec![], + Operator::source( + schema.clone(), + vec![Batch::try_new( + schema.clone(), + vec![vec![Value::Int64(value)], vec![Value::Int64(2)]], + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], aggregate).unwrap(); + assert!(matches!(run(&dag,1,query())[0][0],Value::Int64(v) if v==value+2)); + let mut empty = PhysicalDag::default(); + empty + .add(0, vec![], Operator::source(schema.clone(), vec![]).unwrap()) + .unwrap(); + empty + .add( + 1, + vec![0], + Operator::aggregate(schema, vec![], vec![("min".into(), Reduction::Min(0))]).unwrap(), + ) + .unwrap(); + assert!(matches!(run(&empty, 1, query())[0][0], Value::Null)); +} + +// Plain value operators are library implementations, including NaN comparison. +#[test] +fn scalar_negation_and_vector_conversion() { + let scalar = Operator::scalar(Value::Float64(7.), DataType::Float64).unwrap(); + let project = Operator::project( + scalar.schema(), + vec![( + "v".into(), + Expression::Negate(Box::new(Expression::Column(0))), + )], + ) + .unwrap(); + let convert = Operator::vector_to_scalar(project.schema(), 0).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], scalar).unwrap(); + dag.add(1, vec![0], project).unwrap(); + dag.add(2, vec![1], convert).unwrap(); + assert_eq!(floats(&run(&dag, 2, query()), 0), vec![-7.]); + let scalar = Operator::scalar(Value::Float64(f64::NAN), DataType::Float64).unwrap(); + let predicate = Expression::Equal( + Box::new(Expression::Column(0)), + Box::new(Expression::Column(0)), + ); + let filter = Operator::filter(scalar.schema(), predicate).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], scalar).unwrap(); + dag.add(1, vec![0], filter).unwrap(); + assert!(run(&dag, 1, query()).is_empty()); +} + +// Invalid operations fail at binding rather than becoming external fallbacks. +#[test] +fn binding_rejects_unsupported_operations() { + let schema = schema(&[("v", DataType::Float64, false)]); + assert!(Operator::summary_build( + schema.clone(), + SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), + 0, + None, + vec![] + ) + .is_err()); + let sum = Operator::summary_build( + schema.clone(), + SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + 0, + None, + vec![], + ) + .unwrap(); + assert!(Operator::readout( + sum.schema(), + 0, + asap_physical_operators::operators::ReadoutQuery::Sketch( + planner_types::post_asap::SketchQuery::Quantile { q: 0.5 } + ) + ) + .is_err()); + assert!(Operator::filter(schema, Expression::Column(0)).is_err()); +} + +// KLL is one family example: precomputation changes input sources, not operators. +#[test] +fn kll_raw_partial_and_precomputed_are_native_dags() { + use planner_types::post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("value", DataType::Float64, false)]); + let family = SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), + GroupingStrategy::PerSubpopulationInstance, + ); + let build = Operator::summary_build(input.clone(), family, 0, None, vec![]).unwrap(); + let state = build.schema(); + let build_range = |start: u32, end: u32| { + let mut dag = PhysicalDag::default(); + let batch = Batch::try_new( + input.clone(), + (start..end) + .map(|v| vec![Value::Float64(f64::from(v))]) + .collect(), + ) + .unwrap(); + dag.add( + 0, + vec![], + Operator::source(input.clone(), vec![batch]).unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], build.clone()).unwrap(); + run( + &dag, + 1, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1000, + revision: 1, + }, + ) + }; + let prefix = build_range(0, 64); + let complete = build_range(0, 128); + let query_plan = |stored: Option>>, raw_start: Option| { + let mut dag = PhysicalDag::default(); + let mut states = vec![]; + if let Some(rows) = stored { + dag.add( + 0, + vec![], + Operator::source( + state.clone(), + vec![Batch::try_new(state.clone(), rows).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + states.push(0); + } + if let Some(start) = raw_start { + dag.add( + 1, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new( + input.clone(), + (start..128) + .map(|v| vec![Value::Float64(f64::from(v))]) + .collect(), + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add(2, vec![1], build.clone()).unwrap(); + states.push(2); + } + dag.add( + 3, + states.clone(), + Operator::union(state.clone(), states.len()).unwrap(), + ) + .unwrap(); + dag.add( + 4, + vec![3], + Operator::summary_merge(state.clone(), 0, vec![]).unwrap(), + ) + .unwrap(); + dag.add( + 5, + vec![4], + Operator::readout( + state.clone(), + 0, + asap_physical_operators::operators::ReadoutQuery::Sketch( + planner_types::post_asap::SketchQuery::Quantile { q: 0.5 }, + ), + ) + .unwrap(), + ) + .unwrap(); + floats(&run(&dag, 5, query()), 0)[0] + }; + let raw = query_plan(None, Some(0)); + let partial = query_plan(Some(prefix), Some(64)); + let full = query_plan(Some(complete), None); + assert_eq!(raw, partial); + assert_eq!(partial, full); + assert!((raw - 64.).abs() <= 1.); +} + +// Exact state must match its declared family; a mislabeled state is rejected. +#[test] +fn exact_state_and_family_validation() { + use asap_physical_operators::summary_kernels::exact::ExactAccumulator; + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); + acc.update(None, 7., 0); + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }], + time_index: None, + }); + let value = Value::Summary { + family: family.clone(), + state: Arc::new(acc), + }; + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + schema.clone(), + vec![Batch::try_new(schema.clone(), vec![vec![value]]).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::readout( + schema.clone(), + 0, + asap_physical_operators::operators::ReadoutQuery::Exact( + asap_physical_operators::summary_kernels::exact::ExactReadout { + statistic: Statistic::Sum, + lookback_ms: None, + }, + ), + ) + .unwrap(), + ) + .unwrap(); + assert_eq!(floats(&run(&dag, 1, query()), 0), vec![7.]); + let wrong = ExactAccumulator::new( + SummaryFamilyType::ExactAggregate(ExactKind::Max, ExactParams::Max), + false, + ) + .unwrap(); + assert!(Batch::try_new( + schema, + vec![vec![Value::Summary { + family, + state: Arc::new(wrong) + }]] + ) + .is_err()); +} + +// Planner binding rejects unknown computation instead of accepting a fallback. +#[test] +fn bind_post_asap_before_execution() { + use asap_physical_operators::dag::planner::bind; + use planner_types::{ + post_asap::{ + EdgeRole, ExecutionDataState, GroupingEdgeCompatibility, PostAsapDag, PostAsapDagEdge, + PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, ValueOperation, + WindowEdgeCompatibility, + }, + pre_asap::{ArithmeticOpKind, ProjectItem, QueryExpr, ScalarValue}, + }; + use std::{collections::BTreeMap, rc::Rc}; + let schema = schema(&[("value", DataType::Float64, false)]); + let node = |id, payload| PostAsapDagNode { + id: PostAsapNodeId(id), + payload, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*schema).clone(), + guarantee: None, + }; + let mut dag = PostAsapDag { + nodes: vec![ + node( + 0, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(1.), + }, + ), + node( + 1, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Project { + cols: vec![ProjectItem { + alias: None, + expr: QueryExpr::Arithmetic { + op: ArithmeticOpKind::Add, + left: Rc::new(QueryExpr::Column(0)), + right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.))), + }, + }], + qualifier: None, + }, + }, + ), + ], + edges: vec![PostAsapDagEdge { + producer: PostAsapNodeId(0), + consumer: PostAsapNodeId(1), + role: EdgeRole::Input, + intermediate_schema: (*schema).clone(), + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }], + root: PostAsapNodeId(1), + }; + let sources = || -> BTreeMap> { + BTreeMap::from([( + 0, + Box::new( + Operator::source( + schema.clone(), + vec![Batch::try_new(schema.clone(), vec![vec![Value::Float64(1.)]]).unwrap()], + ) + .unwrap(), + ) as asap_physical_operators::dag::planner::Source<'static>, + )]) + }; + let native = bind(&dag, sources(), &[1]).unwrap(); + assert_eq!(floats(&run(&native, 1, query()), 0), vec![3.]); + // A literal Fallback needs no deployment input. + let literal = bind(&dag, BTreeMap::new(), &[1]).unwrap(); + assert_eq!(floats(&run(&literal, 1, query()), 0), vec![3.]); + dag.nodes[1].payload = PostAsapOperatorPayload::Value { + operation: ValueOperation::Extension { + name: "unknown".into(), + }, + }; + assert!(bind(&dag, sources(), &[1]).is_err()); +} + +// A completed empty population has an exact zero count, with integer output. +#[test] +fn empty_exact_count_is_an_integer_state_readout() { + let input = schema(&[("value", DataType::Float64, false)]); + let build = Operator::summary_build( + input.clone(), + SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count), + 0, + None, + vec![], + ) + .unwrap(); + let read = Operator::readout( + build.schema(), + 0, + asap_physical_operators::operators::ReadoutQuery::Exact( + asap_physical_operators::summary_kernels::exact::ExactReadout { + statistic: Statistic::Count, + lookback_ms: None, + }, + ), + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(input, vec![]).unwrap()) + .unwrap(); + dag.add(1, vec![0], build).unwrap(); + dag.add(2, vec![1], read).unwrap(); + assert!(matches!(run(&dag, 2, query())[0][0], Value::Int64(0))); +} + +// A deployment source cannot pass a different row shape to bound expressions. +#[test] +fn source_batches_must_match_the_bound_schema() { + use asap_physical_operators::dag::{self, PhysicalOperator}; + use planner_types::{ + post_asap::{ + ExecutionDataState, PostAsapDag, PostAsapDagNode, PostAsapNodeId, + PostAsapOperatorPayload, + }, + pre_asap::QueryExpr, + }; + use std::{cell::Cell, collections::BTreeMap, rc::Rc}; + struct WrongSource { + schema: Schema, + starts: Rc>, + } + impl PhysicalOperator for WrongSource { + fn name(&self) -> &str { + "ExternalSource" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.schema.clone() + } + fn output_bytes(&self, value: &Batch) -> usize { + value.bytes() + } + fn start<'a>( + &'a self, + _: Vec>, + _: RunContext, + ) -> Result, dag::Error> { + self.starts.set(self.starts.get() + 1); + Ok( + futures::stream::once(async { Batch::try_new(schema(&[]), vec![vec![]]) }) + .boxed_local(), + ) + } + } + let expected = schema(&[("value", DataType::Float64, false)]); + let starts = Rc::new(Cell::new(0)); + let plan = PostAsapDag { + nodes: vec![PostAsapDagNode { + id: PostAsapNodeId(0), + payload: PostAsapOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(1.), + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*expected).clone(), + guarantee: None, + }], + edges: vec![], + root: PostAsapNodeId(0), + }; + let source = Box::new(WrongSource { + schema: expected, + starts: starts.clone(), + }) as dag::planner::Source<'static>; + let native = dag::planner::bind(&plan, BTreeMap::from([(0, source)]), &[0]).unwrap(); + assert_eq!(starts.get(), 0); + let context = RunContext::new(query(), Limits::default()).unwrap(); + let mut output = native.execute(&[0], context).unwrap().remove(0); + assert!(matches!( + block_on(output.next()), + Some(Err(dag::Error::AtNode { node: 0, .. })) + )); + assert_eq!(starts.get(), 1); +} + +// Float extrema have the same NaN behavior as the exact summary kernels. +#[test] +fn extrema_preserve_numeric_values_in_the_presence_of_nan() { + let input = schema(&[("v", DataType::Float64, false)]); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new( + input.clone(), + vec![vec![Value::Float64(-f64::NAN)], vec![Value::Float64(5.)]], + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::aggregate( + input, + vec![], + vec![ + ("min".into(), Reduction::Min(0)), + ("max".into(), Reduction::Max(0)), + ], + ) + .unwrap(), + ) + .unwrap(); + let rows = run(&dag, 1, query()); + assert_eq!(floats(&rows, 0), vec![5.]); + assert_eq!(floats(&rows, 1), vec![5.]); +} + +// Planner wire nodes, including grouping and edge roles, are executable at either phase. +#[test] +fn planner_semijoin_sort_limit_contract_at_both_phases() { + use asap_physical_operators::dag::planner::{bind, Source}; + use planner_types::{ + post_asap::*, + pre_asap::{CompareOpKind, GroupKeys, JoinKind, Predicate, QueryExpr, SortKey}, + }; + use std::{collections::BTreeMap, rc::Rc}; + let rows_schema = schema(&[ + ("group", DataType::Utf8, false), + ("key", DataType::Utf8, false), + ("score", DataType::Float64, false), + ]); + let keys_schema = schema(&[("key", DataType::Utf8, false)]); + let node = |id, payload, schema: &Schema| PostAsapDagNode { + id: PostAsapNodeId(id), + payload, + output_schema: (**schema).clone(), + output_state: ExecutionDataState::QUERY_ROWS, + guarantee: None, + }; + let edge = |producer, consumer, role, schema: &Schema| PostAsapDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role, + intermediate_schema: (**schema).clone(), + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }; + let groups = GroupKeys::by(vec![0]); + let dag = PostAsapDag { + nodes: vec![ + node( + 0, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(0.), + }, + &rows_schema, + ), + node( + 1, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(0.), + }, + &keys_schema, + ), + node( + 2, + PostAsapOperatorPayload::RelationalJoin { + join_kind: JoinKind::Semi, + pruning: None, + pred: Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(1)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(3)), + })), + }, + &rows_schema, + ), + node( + 3, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Sort { + keys: vec![SortKey { + expr: QueryExpr::Column(2), + ascending: false, + nulls_first: false, + }], + partition_by: groups.clone(), + }, + }, + &rows_schema, + ), + node( + 4, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Limit { + n: 1, + offset: 0, + partition_by: groups, + }, + }, + &rows_schema, + ), + ], + // Deliberately put Right before Left: list order must not swap inputs. + edges: vec![ + edge(1, 2, EdgeRole::Right, &keys_schema), + edge(0, 2, EdgeRole::Left, &rows_schema), + edge(2, 3, EdgeRole::Input, &rows_schema), + edge(3, 4, EdgeRole::Input, &rows_schema), + ], + root: PostAsapNodeId(4), + }; + let text = |v: &str| Value::Utf8(v.into()); + for (phase, scope) in [ + (ExecutionTiming::QueryTime, query()), + ( + ExecutionTiming::IngestionTime, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1000, + revision: 2, + }, + ), + ] { + let dag = dag + .with_execution_phases(&dag.nodes.iter().map(|node| (node.id, phase)).collect()) + .unwrap(); + let sources: BTreeMap> = BTreeMap::from([ + ( + 0, + Box::new( + Operator::source( + rows_schema.clone(), + vec![Batch::try_new( + rows_schema.clone(), + vec![ + vec![text("a"), text("x"), Value::Float64(8.)], + vec![text("a"), text("y"), Value::Float64(9.)], + vec![text("b"), text("x"), Value::Float64(2.)], + vec![text("b"), text("z"), Value::Float64(99.)], + ], + ) + .unwrap()], + ) + .unwrap(), + ) as Source<'static>, + ), + ( + 1, + Box::new( + Operator::source( + keys_schema.clone(), + vec![Batch::try_new( + keys_schema.clone(), + vec![vec![text("x")], vec![text("y")]], + ) + .unwrap()], + ) + .unwrap(), + ) as Source<'static>, + ), + ]); + let native = bind(&dag, sources, &[4]).unwrap(); + let mut scores = floats(&run(&native, 4, scope), 2); + scores.sort_by(f64::total_cmp); + assert_eq!(scores, vec![2., 9.]); + } +} + +// Planner scalar signatures, collection access and null predicates share native execution. +#[test] +fn planner_expressions_preserve_collection_and_nullable_types() { + use asap_physical_operators::dag::expressions::CompiledExpression; + use planner_types::pre_asap::{CompareOpKind, QueryExpr, ScalarValue}; + use std::rc::Rc; + let input_schema = schema(&[( + "items", + DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Int64), + value_nullable: false, + }, + false, + )]); + let access = QueryExpr::FunctionCall { + name: "asap_element_access".into(), + args: vec![ + QueryExpr::Column(0), + QueryExpr::Literal(ScalarValue::Utf8("count".into())), + ], + }; + let project = Operator::project( + input_schema.clone(), + vec![( + "count".into(), + Expression::planner(CompiledExpression::compile(&access, &input_schema).unwrap()), + )], + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + input_schema.clone(), + vec![Batch::try_new( + input_schema.clone(), + vec![ + vec![Value::Map( + vec![(Value::Utf8("count".into()), Value::Int64(7))].into(), + )], + vec![Value::Map(Arc::from([]))], + ], + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + let projected = project.schema(); + dag.add(1, vec![0], project).unwrap(); + let predicate = QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Ge, + right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + }; + dag.add( + 2, + vec![1], + Operator::filter( + projected.clone(), + Expression::planner(CompiledExpression::compile(&predicate, &projected).unwrap()), + ) + .unwrap(), + ) + .unwrap(); + let rows = run(&dag, 2, query()); + assert!(matches!(rows.as_slice(),[row] if matches!(row.as_slice(),[Value::Int64(7)]))); + let unknown = QueryExpr::FunctionCall { + name: "unregistered_function".into(), + args: vec![QueryExpr::Column(0)], + }; + assert!(CompiledExpression::compile(&unknown, &input_schema).is_err()); +} + +// Outer, semi and anti joins share Planner predicates and preserve SQL null behavior. +#[test] +fn native_relational_join_kinds_preserve_unmatched_rows() { + use planner_types::pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}; + use std::rc::Rc; + let input = schema(&[("key", DataType::Int64, true)]); + let predicate = Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(1)), + })); + for (kind, count) in [ + (JoinKind::Inner, 1), + (JoinKind::Left, 3), + (JoinKind::Right, 3), + (JoinKind::Full, 5), + (JoinKind::Semi, 1), + (JoinKind::Anti, 2), + (JoinKind::Cross, 9), + ] { + let output = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + input.clone() + } else { + schema(&[ + ("left", DataType::Int64, true), + ("right", DataType::Int64, true), + ]) + }; + let mut dag = PhysicalDag::default(); + for (id, rows) in [ + ( + 0, + vec![ + vec![Value::Int64(1)], + vec![Value::Int64(2)], + vec![Value::Null], + ], + ), + ( + 1, + vec![ + vec![Value::Int64(2)], + vec![Value::Int64(3)], + vec![Value::Null], + ], + ), + ] { + dag.add( + id, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new(input.clone(), rows).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + } + dag.add( + 2, + vec![0, 1], + Operator::relational_join( + input.clone(), + input.clone(), + kind.clone(), + &predicate, + output, + ) + .unwrap(), + ) + .unwrap(); + assert_eq!(run(&dag, 2, query()).len(), count, "{kind:?}"); + } +} + +// Per-series fractional rates feed either weighted frequency family per job, in either scope. +#[test] +fn weighted_rate_topk_preserves_partitions_fractional_scores_and_evaluation_scope() { + for count_sketch in [false, true] { + assert_weighted_rate_topk(count_sketch); + } +} +fn assert_weighted_rate_topk(count_sketch: bool) { + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let raw = schema(&[ + ("service", DataType::Utf8, false), + ("job", DataType::Utf8, false), + ("instance", DataType::Int64, false), + ("t", DataType::Timestamp, false), + ("value", DataType::Float64, false), + ]); + let mut rows = Vec::new(); + // Multiple instances of auth accumulate. Batch has a very different scale. + for (service, job, instance, rate) in [ + ("auth", "api", 1, 0.125), + ("auth", "api", 2, 0.25), + ("checkout", "api", 1, 0.3125), + ("search", "api", 1, 0.0625), + ("ingest", "batch", 1, 100.0), + ("export", "batch", 1, 80.0), + ("cleanup", "batch", 1, 20.0), + ] { + for (t, value) in [(0, 0.0), (30_000, rate * 30.0), (60_000, rate * 60.0)] { + rows.push(vec![ + Value::Utf8(service.into()), + Value::Utf8(job.into()), + Value::Int64(instance), + Value::Timestamp(t), + Value::Float64(value), + ]); + } + } + let rates = Operator::window( + raw.clone(), + planner_types::pre_asap::AggIntent::Rate, + 3, + 4, + vec![0, 1, 2], + Some((0, 60_000)), + ) + .unwrap(); + let family = SummaryFamilyType::Sketch( + SketchKind::new( + if count_sketch { + SketchAlgorithm::CountSketchWithHeap + } else { + SketchAlgorithm::CmsWithHeap + }, + if count_sketch { + SketchParams::CountSketchWithHeap { + width: 4096, + depth: 5, + heap_size: 8, + } + } else { + SketchParams::CmsWithHeap { + width: 4096, + depth: 5, + heap_size: 8, + } + }, + ), + Default::default(), + ); + let build = Operator::keyed_summary_build(rates.schema(), family, 3, vec![0], vec![1]).unwrap(); + let output = schema(&[ + ("job", DataType::Utf8, false), + ("service", DataType::Utf8, false), + ("score", DataType::Float64, false), + ]); + let readout = Operator::keyed_readout(build.schema(), 1, 8, output.clone()).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(raw.clone(), vec![Batch::try_new(raw, rows).unwrap()]).unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], rates).unwrap(); + dag.add(2, vec![1], build).unwrap(); + dag.add(3, vec![2], readout).unwrap(); + dag.add( + 4, + vec![3], + Operator::sort( + output.clone(), + vec![SortKey { + column: 2, + descending: true, + nulls_first: false, + }], + vec![0], + ) + .unwrap(), + ) + .unwrap(); + dag.add(5, vec![4], Operator::limit(output, 2, 0, vec![0]).unwrap()) + .unwrap(); + for scope in [ + query(), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 60_000, + revision: 2, + }, + query(), + ] { + let result = run(&dag, 5, scope); + assert_eq!(result.len(), 4); + assert_eq!(floats(&result, 2), vec![0.375, 0.3125, 100.0, 80.0]); + let services = result + .iter() + .map(|row| match &row[1] { + Value::Utf8(v) => v.as_ref(), + _ => panic!("service"), + }) + .collect::>(); + assert_eq!(services, vec!["auth", "checkout", "ingest", "export"]); + } +} + +// The grouped temporal reducer's sample schema must survive physical Sort/Limit binding. +#[test] +fn grouped_temporal_schema_compiles_and_executes_topk() { + use asap_physical_operators::physical_planner::{ + compile_node, CompiledPhysicalDag, InputContract, Source, + }; + use planner_types::post_asap::{ + ExecutionDataState, PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, + ValueOperation, + }; + use planner_types::pre_asap::{ + aggregate_output_schema, AggIntent, Column, GroupKeys, QueryExpr, Reduction as IrReduction, + Schema as IrSchema, + }; + let grouped = IrSchema::new(vec![ + Column::new("job", DataType::Utf8, false), + Column::new("sum", DataType::Float64, false), + ]); + let output = aggregate_output_schema( + &grouped, + &IrReduction::PerEntity, + &[AggIntent::Avg { col: None }], + &[], + ) + .unwrap(); + let input = schema( + &output + .columns + .iter() + .map(|c| (c.name.as_str(), c.dtype.clone(), c.nullable)) + .collect::>(), + ); + let node = |id, operation| PostAsapDagNode { + id: PostAsapNodeId(id), + payload: PostAsapOperatorPayload::Value { operation }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*input).clone(), + guarantee: None, + }; + let sort = compile_node( + &node( + 1, + ValueOperation::Sort { + keys: vec![planner_types::pre_asap::SortKey { + expr: QueryExpr::Column(1), + ascending: false, + nulls_first: false, + }], + partition_by: GroupKeys::none(), + }, + ), + std::slice::from_ref(&input), + ) + .unwrap(); + let limit = compile_node( + &node( + 2, + ValueOperation::Limit { + n: 1, + offset: 0, + partition_by: GroupKeys::none(), + }, + ), + std::slice::from_ref(&input), + ) + .unwrap(); + let compiled = CompiledPhysicalDag::from_operators( + [(0, InputContract::bounded(input.clone()))].into(), + [(1, (vec![0], sort)), (2, (vec![1], limit))].into(), + vec![2], + ) + .unwrap(); + let recovered = + serde_json::from_slice::(&serde_json::to_vec(&compiled).unwrap()) + .unwrap(); + assert_eq!(recovered.row_source(2), Some(0)); + assert_eq!(recovered.operator_name(2), Some("Limit")); + let expected = vec![Value::Utf8("api".into()), Value::Float64(9.)]; + let batch = Batch::try_new( + input.clone(), + vec![ + vec![Value::Utf8("worker".into()), Value::Float64(2.)], + expected.clone(), + ], + ) + .unwrap(); + let source = Box::new(Operator::source(input, vec![batch]).unwrap()) as Source<'_>; + let physical = recovered.instantiate([(0, source)].into()).unwrap(); + let mut stream = physical + .execute(&[2], RunContext::new(query(), Limits::default()).unwrap()) + .unwrap() + .remove(0); + let rows = block_on(async { + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend_from_slice(batch.unwrap().rows()); + } + rows + }); + assert_eq!(rows.len(), 1); + assert!(matches!(&rows[0][0], Value::Utf8(label) if label.as_ref() == "api")); + assert!(matches!(rows[0][1], Value::Float64(9.))); +} + +// A certified candidate set must have authoritative values for every key, including after recovery. +#[test] +fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { + use asap_physical_operators::physical_planner::{ + compile_node, CompiledPhysicalDag, InputContract, Source, + }; + use planner_types::{ + post_asap::*, + pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}, + }; + use std::{collections::BTreeMap, rc::Rc}; + let schema = schema(&[("key", DataType::Utf8, false)]); + for certified in [false, true] { + let node = PostAsapDagNode { + id: PostAsapNodeId(2), + output_schema: (*schema).clone(), + output_state: ExecutionDataState::QUERY_ROWS, + guarantee: None, + payload: PostAsapOperatorPayload::RelationalJoin { + join_kind: JoinKind::Semi, + pred: Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(1)), + })), + pruning: certified.then_some(CandidateCompleteness::Certified { + guarantee: ResultGuarantee { + metric: ErrorMetric::TopKMembership, + bound: BoundExpr::Zero, + failure_probability: ProbabilityExpr::Constant { value: 0.01 }, + provenance: vec![], + }, + }), + }, + }; + let graph = CompiledPhysicalDag::from_operators( + [ + (0, InputContract::bounded(schema.clone())), + (1, InputContract::bounded(schema.clone())), + ] + .into(), + [( + 2, + ( + vec![0, 1], + compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(), + ), + )] + .into(), + vec![2], + ) + .unwrap(); + let graph = + serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + .unwrap(); + assert_eq!( + graph.certified_pruning_keys(2), + certified.then_some(&[(0, 0)][..]) + ); + for complete in [false, true] { + let sources = [ + vec!["a"], + if complete { + vec!["a"] + } else { + vec!["a", "missing"] + }, + ] + .into_iter() + .enumerate() + .map(|(i, keys)| { + let batch = Batch::try_new( + schema.clone(), + keys.into_iter() + .map(|k| vec![Value::Utf8(k.into())]) + .collect(), + ) + .unwrap(); + ( + i as u64, + Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect::>(); + let bound = graph.instantiate(sources).unwrap(); + let result = block_on(async { + let mut stream = bound + .execute( + graph.roots(), + RunContext::new(query(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok::<_, asap_physical_operators::Error>(rows) + }); + if certified && !complete { + assert!(result + .unwrap_err() + .to_string() + .contains("no authoritative value")); + } else { + let rows = result.unwrap(); + assert_eq!(rows.len(), 1); + assert!(matches!(&rows[0][0], Value::Utf8(key) if key.as_ref() == "a")); + } + } + } +} + +// Precompute arithmetic must match population/window identities, never zip arrival order. +#[test] +fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { + use asap_physical_operators::physical_planner::{ + compile_node, CompiledPhysicalDag, InputContract, Source, + }; + use planner_types::{ + post_asap::*, + pre_asap::{ArithmeticOpKind, BinaryOpKind}, + }; + use std::collections::BTreeMap; + let input = schema(&[ + ("population", DataType::Utf8, false), + ("time", DataType::Timestamp, false), + ("value", DataType::Float64, false), + ]); + let node = PostAsapDagNode { + id: PostAsapNodeId(2), + output_schema: (*input).clone(), + output_state: ExecutionDataState::INGESTION_ROWS, + guarantee: None, + payload: PostAsapOperatorPayload::Binary { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + }, + }; + let program = CompiledPhysicalDag::from_operators( + [ + (0, InputContract::bounded(input.clone())), + (1, InputContract::bounded(input.clone())), + ] + .into(), + [( + 2, + ( + vec![0, 1], + compile_node(&node, &[input.clone(), input.clone()]).unwrap(), + ), + )] + .into(), + vec![2], + ) + .unwrap(); + let program = + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + .unwrap(); + for (right, expected) in [ + (vec![("b", 2, 3.), ("a", 1, 2.)], Some(vec![8., 17.])), + (vec![("b", 2, 3.)], None), + (vec![("a", 1, 2.), ("a", 1, 2.)], None), + (vec![("a", 2, 2.), ("b", 1, 3.)], None), + (vec![("a", 1, f64::NAN), ("b", 2, 3.)], None), + ] { + let sources = [vec![("a", 1, 10.), ("b", 2, 20.)], right] + .into_iter() + .enumerate() + .map(|(i, rows)| { + let rows = rows + .into_iter() + .map(|(group, time, value)| { + vec![ + Value::Utf8(group.into()), + Value::Timestamp(time), + Value::Float64(value), + ] + }) + .collect(); + let batch = Batch::try_new(input.clone(), rows).unwrap(); + ( + i as u64, + Box::new(Operator::source(input.clone(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect::>(); + let graph = program.instantiate(sources).unwrap(); + let result = block_on(async { + let mut stream = graph + .execute( + program.roots(), + RunContext::new(query(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok::<_, asap_physical_operators::Error>(rows) + }); + match expected { + Some(values) => assert_eq!(floats(&result.unwrap(), 2), values), + None => assert!(result.is_err()), + } + } +} diff --git a/crates/asap-physical-operators/tests/physical_plan_recovery.rs b/crates/asap-physical-operators/tests/physical_plan_recovery.rs new file mode 100644 index 00000000..5a920a8d --- /dev/null +++ b/crates/asap-physical-operators/tests/physical_plan_recovery.rs @@ -0,0 +1,105 @@ +//! Deserialized physical plans recover selected operators without logical lowering. +//! Deployments choose the encoding; JSON is used here only as a test format. +use asap_physical_operators::{ + operators::{Operator, SortKey}, + physical_planner::{CompiledPhysicalDag, InputContract}, +}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, +}; +use std::{collections::BTreeMap, sync::Arc}; + +fn sorted() -> CompiledPhysicalDag { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }); + CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), + BTreeMap::from([( + 1, + ( + vec![0], + Operator::sort( + schema, + vec![SortKey { + column: 0, + descending: true, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + ), + )]), + vec![1], + ) + .unwrap() +} + +#[test] +fn recovery_retains_selected_operator_and_rejects_invalid_contracts() { + let bytes = serde_json::to_vec(&sorted()).unwrap(); + let recovered = serde_json::from_slice::(&bytes).unwrap(); + assert_eq!(serde_json::to_vec(&recovered).unwrap(), bytes); + for mutation in ["column", "edge", "output"] { + let mut wire: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + match mutation { + "column" => { + wire["nodes"]["1"]["Operator"]["operator"]["kind"]["Sort"]["keys"][0]["column"] = + 7.into() + } + "edge" => wire["nodes"]["1"]["Operator"]["inputs"][0] = 999.into(), + "output" => { + wire["nodes"]["1"]["Operator"]["operator"]["output"]["fields"][0]["dtype"] = + serde_json::json!({"Plain":"utf8"}) + } + _ => unreachable!(), + } + assert!( + serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()) + .is_err(), + "accepted {mutation}" + ); + } +} + +#[test] +fn candidate_recovery_preserves_materialization_boundary() { + use asap_physical_operators::physical_planner::PhysicalCandidate; + let precompute = sorted(); + let output = InputContract::bounded(precompute.output_contract(1).unwrap().schema); + let query = CompiledPhysicalDag::from_operators( + BTreeMap::from([(1, output.clone())]), + BTreeMap::from([( + 2, + ( + vec![1], + Operator::limit(output.schema.clone(), 3, 0, vec![]).unwrap(), + ), + )]), + vec![2], + ) + .unwrap(); + let candidate = PhysicalCandidate { + precompute: Some(precompute), + query, + materialized_outputs: BTreeMap::from([(1, output)]), + }; + let bytes = serde_json::to_vec(&candidate).unwrap(); + let restored = serde_json::from_slice::(&bytes).unwrap(); + assert_eq!(restored.precompute.as_ref().unwrap().roots(), &[1]); + assert_eq!(restored.query.roots(), &[2]); + assert_eq!(serde_json::to_vec(&restored).unwrap(), bytes); + let mut wire: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + wire["materialized_outputs"]["1"]["schema"]["fields"][0]["dtype"] = + serde_json::json!({"Plain":"utf8"}); + assert!( + serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()).is_err() + ); +} diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs new file mode 100644 index 00000000..5770d7d6 --- /dev/null +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -0,0 +1,695 @@ +//! Contract tests inspired by DataFusion's limit, sort and join test matrices. +//! Expectations follow ASAP's IR (notably row-count and IEEE NaN equality). +//! Reference: apache/datafusion e2ca7f3, physical-plan/src/{limit.rs,sorts/sort.rs}. +use asap_physical_operators::{ + expressions::CompiledExpression, + operators::{Expression, Operator, Reduction, SortKey}, + plan::PhysicalDag, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Schema, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{CompareOpKind, DataType, JoinKind, Predicate, QueryExpr}, +}; +use std::{rc::Rc, sync::Arc}; + +fn schema(fields: &[(&str, DataType, bool)]) -> Schema { + Arc::new(SummarySchema { + fields: fields + .iter() + .map(|(name, dtype, nullable)| SummaryField { + name: (*name).into(), + dtype: SummaryFamilyType::Plain(dtype.clone()), + nullable: *nullable, + }) + .collect(), + time_index: None, + }) +} +fn context() -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits { + max_buffered_batches: 1, + ..Limits::default() + }, + ) + .unwrap() +} +fn collect(dag: &PhysicalDag<'_, Batch, Schema>, root: u64) -> Vec> { + let run = context(); + let rows = block_on(async { + let mut stream = dag.execute(&[root], run.clone()).unwrap().remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend_from_slice(batch.unwrap().rows()); + } + rows + }); + assert_eq!(run.retained_bytes(), 0); + rows +} +fn unary(input: Schema, batches: Vec>>, op: Operator) -> Vec> { + let mut dag = PhysicalDag::default(); + let batches = batches + .into_iter() + .map(|rows| Batch::try_new(input.clone(), rows).unwrap()) + .collect(); + dag.add(0, vec![], Operator::source(input, batches).unwrap()) + .unwrap(); + dag.add(1, vec![0], op).unwrap(); + collect(&dag, 1) +} +fn keys(rows: &[Vec]) -> Vec>> { + rows.iter() + .map(|r| r.iter().map(|v| v.key().unwrap()).collect()) + .collect() +} +fn eq_predicate() -> Predicate { + Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(1)), + })) +} +fn join(left: Vec, right: Vec, kind: JoinKind, keyed: bool) -> Vec> { + let input = schema(&[("key", DataType::Float64, true)]); + let output = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + input.clone() + } else { + schema(&[ + ("left", DataType::Float64, true), + ("right", DataType::Float64, true), + ]) + }; + let op = if keyed { + Operator::semi_join(input.clone(), input.clone(), vec![(0, 0)]).unwrap() + } else { + Operator::relational_join(input.clone(), input.clone(), kind, &eq_predicate(), output) + .unwrap() + }; + let mut dag = PhysicalDag::default(); + for (id, values) in [(0, left), (1, right)] { + let batches = values + .into_iter() + .map(|v| Batch::try_new(input.clone(), vec![vec![v]]).unwrap()) + .collect(); + dag.add( + id, + vec![], + Operator::source(input.clone(), batches).unwrap(), + ) + .unwrap(); + } + dag.add(2, vec![0, 1], op).unwrap(); + collect(&dag, 2) +} + +// OFFSET/FETCH must be invariant to empty batches and input batch boundaries. +#[test] +fn limit_offset_fetch_matrix() { + let input = schema(&[("v", DataType::Int64, false)]); + for chunk in [1, 2, 5, 12] { + let values = (0..9).map(|n| vec![Value::Int64(n)]).collect::>(); + let mut batches = vec![vec![]]; + for rows in values.chunks(chunk) { + batches.push(rows.to_vec()); + batches.push(vec![]); + } + for offset in [0, 1, 8, 9, 10, u64::MAX] { + for n in [0, 1, 3, 12, u64::MAX] { + let rows = unary( + input.clone(), + batches.clone(), + Operator::limit(input.clone(), n, offset, vec![]).unwrap(), + ); + let expected = values + .iter() + .skip(offset.min(9) as usize) + .take(n.min(9) as usize) + .cloned() + .collect::>(); + assert_eq!( + keys(&rows), + keys(&expected), + "chunk={chunk}, offset={offset}, n={n}" + ); + } + } + } +} + +// Zero-column batches still have rows: LIMIT must not infer cardinality from columns. +#[test] +fn limit_preserves_zero_column_row_count() { + let input = schema(&[]); + let rows = unary( + input.clone(), + vec![vec![vec![]; 5], vec![vec![]; 5]], + Operator::limit(input, 4, 3, vec![]).unwrap(), + ); + assert_eq!(rows.len(), 4); +} + +// NULL placement is independent of sort direction; ties retain original row order. +#[test] +fn sort_direction_null_placement_and_ties() { + let input = schema(&[("v", DataType::Int64, true), ("id", DataType::Int64, false)]); + let values = [Some(2), None, Some(1), Some(2), None]; + let rows = values + .iter() + .enumerate() + .map(|(i, v)| { + vec![ + v.map(Value::Int64).unwrap_or(Value::Null), + Value::Int64(i as i64), + ] + }) + .collect::>(); + for (descending, nulls_first, expected) in [ + (false, false, vec![2, 0, 3, 1, 4]), + (false, true, vec![1, 4, 2, 0, 3]), + (true, false, vec![0, 3, 2, 1, 4]), + (true, true, vec![1, 4, 0, 3, 2]), + ] { + let op = Operator::sort( + input.clone(), + vec![SortKey { + column: 0, + descending, + nulls_first, + }], + vec![], + ) + .unwrap(); + let result = unary( + input.clone(), + vec![rows[..2].to_vec(), vec![], rows[2..].to_vec()], + op, + ); + let ids = result + .iter() + .map(|r| match r[1] { + Value::Int64(n) => n, + _ => unreachable!(), + }) + .collect::>(); + assert_eq!(ids, expected); + } +} + +// Outer joins preserve unmatched NULLs, while semi/anti joins preserve left multiplicity. +#[test] +fn joins_nulls_duplicates_and_empty_sides() { + for (kind, expected_len) in [ + (JoinKind::Inner, 4), + (JoinKind::Left, 6), + (JoinKind::Right, 6), + (JoinKind::Full, 8), + (JoinKind::Semi, 2), + (JoinKind::Anti, 2), + ] { + let left = vec![ + Value::Float64(1.), + Value::Float64(1.), + Value::Float64(2.), + Value::Null, + ]; + let right = vec![ + Value::Float64(1.), + Value::Float64(1.), + Value::Float64(3.), + Value::Null, + ]; + let result = join(left, right, kind.clone(), false); + assert_eq!(result.len(), expected_len, "{kind:?}"); + } + for (kind, expected_len) in [ + (JoinKind::Inner, 0), + (JoinKind::Left, 1), + (JoinKind::Right, 0), + (JoinKind::Full, 1), + (JoinKind::Semi, 0), + (JoinKind::Anti, 1), + ] { + assert_eq!( + join(vec![Value::Float64(7.)], vec![], kind.clone(), false).len(), + expected_len, + "{kind:?}" + ); + } + let result = join(vec![Value::Float64(7.)], vec![], JoinKind::Left, false); + assert!(matches!( + result[0].as_slice(), + [Value::Float64(7.), Value::Null] + )); +} + +// Changing the semi-join algorithm must not turn IEEE NaN != NaN into a match. +#[test] +fn keyed_semijoin_obeys_ieee_equality_for_nan_and_zero() { + let left = vec![ + Value::Float64(f64::NAN), + Value::Float64(-0.), + Value::Float64(0.), + Value::Null, + ]; + let right = vec![Value::Float64(f64::NAN), Value::Float64(0.), Value::Null]; + let keyed = join(left, right, JoinKind::Semi, true); + let expected = vec![vec![Value::Float64(-0.)], vec![Value::Float64(0.)]]; + assert_eq!(keys(&keyed), keys(&expected)); +} + +// Group equality intentionally differs from predicate equality: NULL and NaNs group together. +#[test] +fn grouping_canonicalizes_null_nan_and_signed_zero() { + let input = schema(&[("v", DataType::Float64, true)]); + let op = Operator::aggregate( + input.clone(), + vec![0], + vec![("count".into(), Reduction::Count)], + ) + .unwrap(); + let values = vec![ + Value::Null, + Value::Null, + Value::Float64(0.), + Value::Float64(-0.), + Value::Float64(f64::NAN), + Value::Float64(f64::from_bits(0x7ff8000000000001)), + ]; + let result = unary( + input, + values.into_iter().map(|v| vec![vec![v]]).collect(), + op, + ); + assert_eq!(result.len(), 3); + assert!(result.iter().all(|r| matches!(r[1], Value::Int64(2)))); +} + +// Global empty input yields one aggregate row; grouped empty input yields none. +#[test] +fn aggregate_empty_and_all_null_follow_asap_contract() { + let input = schema(&[("v", DataType::Int64, true)]); + for batches in [ + vec![], + vec![vec![]], + vec![vec![vec![Value::Null], vec![Value::Null]]], + ] { + let n = batches.iter().map(Vec::len).sum::(); + let op = Operator::aggregate( + input.clone(), + vec![], + vec![ + ("count".into(), Reduction::Count), + ("min".into(), Reduction::Min(0)), + ("max".into(), Reduction::Max(0)), + ], + ) + .unwrap(); + let result = unary(input.clone(), batches, op); + assert_eq!(result.len(), 1); + assert!(matches!(result[0][0], Value::Int64(v) if v == n as i64)); + assert!(matches!(result[0][1], Value::Null)); + assert!(matches!(result[0][2], Value::Null)); + } + let op = Operator::aggregate( + input.clone(), + vec![0], + vec![("count".into(), Reduction::Count)], + ) + .unwrap(); + assert!(unary(input, vec![], op).is_empty()); +} + +// A precompiled expression with a different input contract must fail during binding. +#[test] +fn projection_rejects_expression_bound_to_another_schema() { + let original = schema(&[("a", DataType::Int64, false), ("b", DataType::Int64, false)]); + let current = schema(&[("a", DataType::Int64, false)]); + let expr = CompiledExpression::compile(&QueryExpr::Column(1), &original).unwrap(); + assert!(Operator::project(current, vec![("b".into(), Expression::planner(expr))]).is_err()); +} + +// A valid Planner MIN/MAX schema must bind even for a non-null input column. +#[test] +fn global_extrema_bind_with_planner_derived_schema() { + use asap_physical_operators::physical_planner::compile_node; + use planner_types::{ + post_asap::*, + pre_asap::{AggIntent, Column, GroupKeys, Reduction as PlanReduction}, + }; + let input = schema(&[("v", DataType::Int64, false)]); + for measure in [ + AggIntent::Min { col: Some(0) }, + AggIntent::Max { col: Some(0) }, + ] { + let planner_input = + planner_types::pre_asap::Schema::new(vec![Column::new("v", DataType::Int64, false)]); + let derived = planner_types::pre_asap::query_expr::aggregate_output_schema( + &planner_input, + &PlanReduction::Reduce(GroupKeys::by(vec![])), + std::slice::from_ref(&measure), + &[], + ) + .unwrap(); + let result = derived.columns[0].clone(); + let output = schema(&[(&result.name, result.dtype, result.nullable)]); + let node = PostAsapDagNode { + id: PostAsapNodeId(1), + payload: PostAsapOperatorPayload::Value { + operation: ValueOperation::Exact(ExactOperation::Aggregate { + reduction: PlanReduction::Reduce(GroupKeys::by(vec![])), + measures: vec![measure], + output_names: vec![result.name], + having: None, + }), + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*output).clone(), + guarantee: None, + }; + let operator = compile_node(&node, std::slice::from_ref(&input)) + .expect("global extremum should bind to its Planner schema"); + assert!(operator.schema().fields[0].nullable); + let empty = unary(input.clone(), vec![], operator.clone()); + assert!(matches!(empty[0][0], Value::Null)); + let nonempty = unary(input.clone(), vec![vec![vec![Value::Int64(7)]]], operator); + assert!(matches!(nonempty[0][0], Value::Int64(7))); + } +} + +// NaN is a valid numeric input, not a schema error; all six comparisons obey IEEE rules. +#[test] +fn planner_comparisons_handle_nan_without_execution_errors() { + let input = schema(&[ + ("a", DataType::Float64, false), + ("b", DataType::Float64, false), + ]); + for op in [ + CompareOpKind::Eq, + CompareOpKind::Ne, + CompareOpKind::Lt, + CompareOpKind::Le, + CompareOpKind::Gt, + CompareOpKind::Ge, + ] { + let expression = QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: op.clone(), + right: Rc::new(QueryExpr::Column(1)), + }; + let compiled = CompiledExpression::compile(&expression, &input).unwrap(); + for row in [ + [Value::Float64(f64::NAN), Value::Float64(1.)], + [Value::Float64(1.), Value::Float64(f64::NAN)], + [Value::Float64(f64::NAN), Value::Float64(f64::NAN)], + ] { + let actual = compiled.evaluate(&row).unwrap(); + assert!(matches!(actual,Value::Bool(value) if value == (op == CompareOpKind::Ne))); + } + } +} + +// A bounded LIMIT branch must unsubscribe so another branch can drain the producer. +#[test] +fn limit_branch_finishes_without_blocking_shared_sibling() { + let input = schema(&[("v", DataType::Int64, false)]); + let mut dag = PhysicalDag::default(); + let batches = (0..100) + .map(|v| Batch::try_new(input.clone(), vec![vec![Value::Int64(v)]]).unwrap()) + .collect(); + dag.add(0, vec![], Operator::source(input.clone(), batches).unwrap()) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::limit(input.clone(), 1, 0, vec![]).unwrap(), + ) + .unwrap(); + dag.add(2, vec![0, 1], Operator::union(input, 2).unwrap()) + .unwrap(); + // Bound polls as well as rows so a backpressure regression cannot hang the suite. + use futures::{task::noop_waker_ref, Stream}; + use std::{ + pin::Pin, + task::{Context, Poll}, + }; + let run = context(); + let mut stream = dag.execute(&[2], run.clone()).unwrap().remove(0); + let mut cx = Context::from_waker(noop_waker_ref()); + let mut count = 0; + for _ in 0..2000 { + match Pin::new(&mut stream).poll_next(&mut cx) { + Poll::Ready(Some(batch)) => count += batch.unwrap().rows().len(), + Poll::Ready(None) => { + assert_eq!(count, 101); + drop(stream); + assert_eq!(run.retained_bytes(), 0); + return; + } + Poll::Pending => {} + } + } + panic!("shared LIMIT/Union failed to make progress"); +} + +// Mixed numeric comparisons must not round Int64 values through f64 before comparing. +#[test] +fn mixed_numeric_comparisons_preserve_large_integer_precision() { + let input = schema(&[ + ("a", DataType::Int64, false), + ("b", DataType::Float64, false), + ]); + let expr = QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Gt, + right: Rc::new(QueryExpr::Column(1)), + }; + let compiled = CompiledExpression::compile(&expr, &input).unwrap(); + for (a, b, expected) in [ + (9_007_199_254_740_993, 9_007_199_254_740_992.0, true), + (i64::MAX, 9_223_372_036_854_775_808.0, false), + (i64::MIN, f64::NEG_INFINITY, true), + ] { + assert!( + matches!(compiled.evaluate(&[Value::Int64(a),Value::Float64(b)]).unwrap(), Value::Bool(v) if v == expected) + ); + } +} + +// Both expression paths must implement all nine combinations of three-valued booleans. +#[test] +fn boolean_truth_tables_agree_between_expression_paths() { + let input = schema(&[("a", DataType::Bool, true), ("b", DataType::Bool, true)]); + for and in [true, false] { + for a in [None, Some(false), Some(true)] { + for b in [None, Some(false), Some(true)] { + let parts = vec![QueryExpr::Column(0), QueryExpr::Column(1)]; + let planner = if and { + QueryExpr::BoolAnd(parts) + } else { + QueryExpr::BoolOr(parts) + }; + let native = if and { + Expression::And( + Box::new(Expression::Column(0)), + Box::new(Expression::Column(1)), + ) + } else { + Expression::Or( + Box::new(Expression::Column(0)), + Box::new(Expression::Column(1)), + ) + }; + let expected = match (a, b, and) { + (Some(false), _, true) | (_, Some(false), true) => Some(false), + (Some(true), _, false) | (_, Some(true), false) => Some(true), + (None, _, _) | (_, None, _) => None, + (Some(a), Some(b), true) => Some(a && b), + (Some(a), Some(b), false) => Some(a || b), + } + .map(Value::Bool) + .unwrap_or(Value::Null); + let row = vec![ + a.map(Value::Bool).unwrap_or(Value::Null), + b.map(Value::Bool).unwrap_or(Value::Null), + ]; + let compiled = CompiledExpression::compile(&planner, &input).unwrap(); + assert_eq!( + compiled.evaluate(&row).unwrap().key().unwrap(), + expected.key().unwrap() + ); + let op = Operator::project(input.clone(), vec![("result".into(), native)]).unwrap(); + let result = unary(input.clone(), vec![vec![row]], op); + assert_eq!(result[0][0].key().unwrap(), expected.key().unwrap()); + } + } + } +} + +// Partial/final execution must agree with one build for an uncompacted KLL population. +#[test] +fn kll_partial_merge_and_multiple_readouts_preserve_population() { + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("v", DataType::Float64, false)]); + let family = SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), + Default::default(), + ); + let mut dag = PhysicalDag::default(); + for (id, range) in [(0, 0..64), (1, 64..128), (2, 0..128)] { + let rows = range.map(|n| vec![Value::Float64(n as f64)]).collect(); + dag.add( + id, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new(input.clone(), rows).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add( + id + 3, + vec![id], + Operator::summary_build(input.clone(), family.clone(), 0, None, vec![]).unwrap(), + ) + .unwrap(); + } + let state = Operator::summary_build(input, family, 0, None, vec![]) + .unwrap() + .schema(); + dag.add(6, vec![3, 4], Operator::union(state.clone(), 2).unwrap()) + .unwrap(); + dag.add( + 7, + vec![6], + Operator::summary_merge(state.clone(), 0, vec![]).unwrap(), + ) + .unwrap(); + let mut roots = vec![]; + for (i, q) in [0.0, 0.5, 1.0].into_iter().enumerate() { + for (j, build) in [5, 7].into_iter().enumerate() { + let id = 8 + (i * 2 + j) as u64; + dag.add( + id, + vec![build], + Operator::readout( + state.clone(), + 0, + asap_physical_operators::operators::ReadoutQuery::Sketch( + planner_types::post_asap::SketchQuery::Quantile { q }, + ), + ) + .unwrap(), + ) + .unwrap(); + roots.push(id); + } + } + for _ in 0..2 { + let run = context(); + let outputs = block_on(futures::future::join_all( + dag.execute(&roots, run.clone()) + .unwrap() + .into_iter() + .map(|s| s.collect::>()), + )); + for (pair, expected) in outputs.chunks(2).zip([0., 64., 127.]) { + let value = |batches: &[Result< + asap_physical_operators::runtime::SharedValue, + asap_physical_operators::Error, + >]| { + assert_eq!(batches.len(), 1); + match batches[0].as_ref().unwrap().rows()[0][0] { + Value::Float64(v) => v, + _ => panic!("quantile must be Float64"), + } + }; + assert_eq!(value(&pair[0]), value(&pair[1])); + assert!((value(&pair[0]) - expected).abs() <= 1.); + } + drop(outputs); + assert_eq!(run.retained_bytes(), 0); + } +} + +// Retained zero-column rows still own Vec headers and must consume the output budget. +#[test] +fn zero_column_output_obeys_memory_limit() { + use asap_physical_operators::Error; + let input = schema(&[]); + let batch = Batch::try_new(input.clone(), vec![vec![]; 200]).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(input, vec![batch]).unwrap()) + .unwrap(); + let run = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_bytes: 1024, + ..Limits::default() + }, + ) + .unwrap(); + let mut stream = dag.execute(&[0], run.clone()).unwrap().remove(0); + assert!(matches!( + block_on(stream.next()), + Some(Err(Error::MemoryLimit)) + )); + drop(stream); + assert_eq!(run.retained_bytes(), 0); +} + +// Empty exact-state finalization must preserve ordinary global MIN/MAX null semantics. +#[test] +fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { + use asap_physical_operators::Statistic; + use planner_types::post_asap::{ExactKind, ExactParams}; + let input = schema(&[("v", DataType::Float64, false)]); + for (kind, params, statistic) in [ + (ExactKind::Min, ExactParams::Min, Statistic::Min), + (ExactKind::Max, ExactParams::Max, Statistic::Max), + ] { + let build = Operator::summary_build( + input.clone(), + SummaryFamilyType::ExactAggregate(kind, params), + 0, + None, + vec![], + ) + .unwrap(); + let state = build.schema(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(input.clone(), vec![]).unwrap()) + .unwrap(); + dag.add(1, vec![0], build).unwrap(); + dag.add( + 2, + vec![1], + Operator::readout( + state, + 0, + asap_physical_operators::operators::ReadoutQuery::Exact( + asap_physical_operators::summary_kernels::exact::ExactReadout { + statistic, + lookback_ms: None, + }, + ), + ) + .unwrap(), + ) + .unwrap(); + let rows = collect(&dag, 2); + assert_eq!(rows.len(), 1); + assert!(matches!(rows[0][0], Value::Null)); + } +} diff --git a/crates/asap-physical-operators/tests/plan_properties.rs b/crates/asap-physical-operators/tests/plan_properties.rs new file mode 100644 index 00000000..f9a509c2 --- /dev/null +++ b/crates/asap-physical-operators/tests/plan_properties.rs @@ -0,0 +1,151 @@ +//! Finite-input contracts are validated before source execution. +use asap_physical_operators::{ + operators::{Operator, SortKey}, + plan::{Boundedness, Emission, PhysicalDag}, + runtime::{Limits, OutputStream, RunContext, Scope}, + sources::{DataSources, RawSource}, + values::{Batch, Schema}, + Error, +}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{Column, DataType, QueryExpr, Schema as LogicalSchema, Source}, +}; +use std::sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, +}; +struct DeclaredSource { + schema: Schema, + boundedness: Boundedness, + opens: Arc, +} +impl RawSource for DeclaredSource { + fn schema(&self) -> Schema { + self.schema.clone() + } + fn boundedness(&self) -> Boundedness { + self.boundedness + } + fn scan(&self, _: RunContext) -> Result, Error> { + self.opens.fetch_add(1, Ordering::SeqCst); + Ok(Box::pin(futures::stream::empty())) + } +} +// A blocking parent must reject unknown and unbounded Scan inputs without opening a reader. +#[test] +fn blocking_inputs_require_an_explicit_finite_source() { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "v".into(), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: false, + }], + time_index: None, + }); + for boundedness in [ + Boundedness::Unknown, + Boundedness::Unbounded, + Boundedness::Bounded, + ] { + let opens = Arc::new(AtomicUsize::new(0)); + let mut registry = DataSources::default(); + let identity = Source::Table { + table_ref: "t".into(), + }; + registry + .register( + identity.clone(), + Arc::new(DeclaredSource { + schema: schema.clone(), + boundedness, + opens: opens.clone(), + }), + ) + .unwrap(); + let scan = registry + .bind(&QueryExpr::Scan { + source: identity, + schema: LogicalSchema::new(vec![Column::new("v", DataType::Int64, false)]), + predicates: vec![], + }) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], scan).unwrap(); + dag.add( + 1, + vec![0], + Operator::sort( + schema.clone(), + vec![SortKey { + column: 0, + descending: false, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + ) + .unwrap(); + let run = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + if boundedness == Boundedness::Bounded { + let properties = dag.properties(&[1]).unwrap(); + assert_eq!(properties[&1].emission, Emission::AfterInput); + assert_eq!(properties[&1].boundedness, Boundedness::Bounded); + assert!(dag.execute(&[1], run).is_ok()); + } else { + assert!( + matches!(dag.execute(&[1], run), Err(Error::Invalid(message)) if message.contains("requires bounded inputs")) + ); + } + assert_eq!(opens.load(Ordering::SeqCst), 0); + } +} + +// Kernel support must not be mistaken for executable native state/readout support. +#[test] +fn summary_capability_levels_are_distinct() { + use asap_physical_operators::{ + capability::{validate_native_family, validate_sketch_readout, validate_summary_kernel}, + planner::post_asap::SketchQuery, + }; + use planner_types::{ + post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate}, + pre_asap::ColumnRef, + }; + let grouping = GroupingStrategy::default(); + let cms = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::Cms, + SketchParams::Cms { + width: 64, + depth: 4, + }, + ), + grouping.clone(), + ); + let update = SummaryUpdate { + item: Some(planner_types::post_asap::SummaryInputExpr::Column( + ColumnRef::Named("host".into()), + )), + weight: planner_types::post_asap::SummaryInputExpr::Constant(1.0), + weight_domain: Default::default(), + }; + assert!(validate_summary_kernel(&cms, &update, &grouping).is_ok()); + assert!(validate_native_family(&cms).is_err()); + let kll = SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 128 }), + grouping, + ); + assert!(validate_native_family(&kll).is_ok()); + assert!(validate_sketch_readout(&kll, &SketchQuery::Quantile { q: 1.5 }).is_err()); + assert!(validate_sketch_readout(&kll, &SketchQuery::Cardinality).is_err()); + assert!(validate_sketch_readout(&kll, &SketchQuery::Quantile { q: 0.5 }).is_ok()); +} diff --git a/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs b/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs new file mode 100644 index 00000000..2f8d5249 --- /dev/null +++ b/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs @@ -0,0 +1,250 @@ +//! Logical heap alternatives that need the PromQL series identity are part of +//! Planner's search space: `enumerate_candidate_dags_for_root` lists +//! current-series TopK heaps without a caller-side series-identity pass, cost +//! ranking, or workload Cartesian expansion. Placement variants are not listed. +use asap_aware_mapping::{ + accuracy::{AccuracyEvidenceProvider, DefaultAccuracyModel, PropagationStats}, + cost_model::DefaultCostModel, + replacement::{default_strategies_with_evidence, ReplacementProvenance}, + search_workload_with_targets, Proposals, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, +}; +use asap_physical_operators::physical_planner::promql_rows::{ + compile_current_series_readout, SERIES_IDENTITY_COLUMN, +}; +use planner_types::{ + post_asap::*, + pre_asap::QueryExpr, + types::AccuracyTarget, + workload::{ + AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence as WorkloadEvidence, + PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, + TimeSelection, + }, +}; +use std::rc::Rc; + +struct Evidence; +impl AccuracyEvidenceProvider for Evidence { + fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + Some(1000) + } + fn propagation_stats( + &self, + op: &CompositionOperator, + _: &SummaryFamilyType, + _: Option<&SketchQuery>, + ) -> PropagationStats { + if matches!(op, CompositionOperator::TopKSelection) { + PropagationStats { + topk_selected_lower_bound: Some(101.), + topk_excluded_upper_bound: Some(100.), + topk_interval_failure_probability: Some(0.001), + ..Default::default() + } + } else { + Default::default() + } + } +} + +/// Forwards everything except whole-root proposals: the pre-change search. +struct LogicalOnly(Box); +impl ReplacementStrategy for LogicalOnly { + fn name(&self) -> &'static str { + self.0.name() + } + fn matches(&self, target: &TargetSubDAG<'_>) -> bool { + self.0.matches(target) + } + fn replacements(&self, target: &TargetSubDAG<'_>) -> Vec { + self.0.replacements(target) + } + fn propose(&self, target: &TargetSubDAG<'_>) -> Proposals { + self.0.propose(target) + } +} + +fn lower(query: &str, accuracy: &AccuracyTarget) -> Rc { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy.clone()), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: WorkloadEvidence { + value: Some(DurationMs(1_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + Rc::new( + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0), + ) +} + +type Dag = Vec<(usize, Rc)>; + +/// Candidate DAGs for query 1 of a two-query workload, with and without +/// whole-root proposals. Query 0 is a bystander that must not multiply them. +fn inventories(query: &str, accuracy: AccuracyTarget) -> (Vec, Vec) { + let roots = vec![ + ( + 0, + lower("sum by(job)(m)", &AccuracyTarget::Exact), + Some(AccuracyTarget::Exact), + ), + (1, lower(query, &accuracy), Some(accuracy)), + ]; + let full = default_strategies_with_evidence(&DefaultCostModel, &Evidence); + let logical: Vec> = + default_strategies_with_evidence(&DefaultCostModel, &Evidence) + .into_iter() + .map(|strategy| Box::new(LogicalOnly(strategy)) as Box) + .collect(); + let enumerate = |strategies: &[Box]| { + search_workload_with_targets(roots.clone(), strategies, &DefaultAccuracyModel) + .enumerate_candidate_dags_for_root(&1, 65_536) + .unwrap() + .candidates + }; + (enumerate(&full), enumerate(&logical)) +} + +fn carries_identity(dag: &Dag) -> bool { + dag.iter().any(|(_, root)| { + compile_post_asap_dag(root) + .unwrap() + .nodes + .iter() + .any(|node| { + node.output_schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + }) + }) +} + +/// Shared acceptance checks; returns the added identity-carrying alternatives. +fn added_alternatives(query: &str, accuracy: AccuracyTarget) -> Vec> { + let (full, logical) = inventories(query, accuracy); + for (index, dag) in full.iter().enumerate() { + assert_eq!(dag.len(), 1, "one root per candidate, no workload product"); + assert!( + !full[..index].contains(dag), + "{query}: identical DAG listed twice" + ); + } + let (added, kept): (Vec<_>, Vec<_>) = full.into_iter().partition(carries_identity); + assert_eq!( + kept, logical, + "{query}: existing candidates must be unchanged" + ); + added.into_iter().map(|mut dag| dag.remove(0).1).collect() +} + +const CURRENT_SERIES_TOPK: &str = "topk by(job)(1, m)"; + +// Instant-vector TopK lists finalized current-series heap readouts. +#[test] +fn current_series_topk_lists_heap_readouts() { + let added = added_alternatives(CURRENT_SERIES_TOPK, AccuracyTarget::Epsilon(0.1)); + assert!(!added.is_empty()); + for root in added { + assert!(!matches!(root.expr, SummaryExpr::SummaryAgg { .. })); + assert!( + compile_current_series_readout(&root).is_ok(), + "unbindable alternative {root:?}" + ); + } +} + +// Rate queries gain no fixed-window or query-time placement variants. +#[test] +fn rate_placement_variants_are_not_listed() { + for (query, accuracy) in [ + ("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact), + ("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)), + ("rate(m[1m])", AccuracyTarget::Exact), + ] { + assert!(added_alternatives(query, accuracy).is_empty(), "{query}"); + } +} + +// Queries without a current-series heap realization are unchanged. +#[test] +fn unrelated_queries_keep_their_inventory() { + for (query, accuracy) in [ + ("sum by(job)(m)", AccuracyTarget::Exact), + ( + "quantile_over_time(0.9, m[1m])", + AccuracyTarget::Epsilon(0.05), + ), + ("max_over_time(m[1m])", AccuracyTarget::Exact), + ] { + assert!(added_alternatives(query, accuracy).is_empty(), "{query}"); + } +} + +// Default cost-based selection keeps the logical plan; deployment prices heaps. +#[test] +fn global_selection_never_commits_a_series_identity_heap() { + let accuracy = AccuracyTarget::Epsilon(0.1); + let root = lower(CURRENT_SERIES_TOPK, &accuracy); + let strategies = default_strategies_with_evidence(&DefaultCostModel, &Evidence); + let space = search_workload_with_targets( + vec![(0, root, Some(accuracy))], + &strategies, + &DefaultAccuracyModel, + ); + let selected = space + .global_selection(&DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + assert!(!carries_identity(&vec![(0, selected)])); +} + +// A query repeated in the workload is proposed once, not once per copy. +#[test] +fn repeated_roots_do_not_duplicate_alternatives() { + let accuracy = AccuracyTarget::Epsilon(0.1); + let strategies = default_strategies_with_evidence(&DefaultCostModel, &Evidence); + let count = |copies: usize| { + let roots = (0..copies) + .map(|id| { + ( + id, + lower(CURRENT_SERIES_TOPK, &accuracy), + Some(accuracy.clone()), + ) + }) + .collect(); + let space = search_workload_with_targets(roots, &strategies, &DefaultAccuracyModel); + space + .candidates_for_target(&space.roots[0].1) + .unwrap() + .candidates + .iter() + .filter(|candidate| { + candidate.provenance == ReplacementProvenance::RootPhysicalRealization + }) + .count() + }; + assert!(count(1) > 0); + assert_eq!(count(2), count(1)); +} diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs new file mode 100644 index 00000000..f8050118 --- /dev/null +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -0,0 +1,726 @@ +//! Materialized frontiers are compiled by Planner, never rewritten by deployment. +use asap_aware_mapping::{cost_model::DefaultCostModel, search_workload}; +use asap_physical_operators::{ + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{ + compile, compile_candidate, compile_candidates, cut_candidate, enumerate_frontiers, + select_candidate, CandidateCost, CompiledPhysicalDag, InputContract, PhysicalCandidate, + Source, + }, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{post_asap::*, pre_asap::DataType, types::AccuracyTarget, workload::*}; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; + +fn grouped_rate_space() -> asap_aware_mapping::PlanSpace<&'static str> { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("sum by(job)(rate(m[1m]))".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let root = Rc::new( + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0), + ); + let root = Rc::new( + asap_physical_operators::physical_planner::promql_rows::with_series_identity(&root) + .unwrap(), + ); + search_workload(vec![("grouped-rate", root)]) +} + +fn grouped_rate() -> PostAsapDag { + let space = grouped_rate_space(); + let selected = space + .global_selection(&DefaultCostModel) + .assemble_selected_query(&space.roots[0].1) + .unwrap() + .unwrap(); + compile_post_asap_dag(&selected).unwrap() +} +fn run(plan: &CompiledPhysicalDag, inputs: BTreeMap, scope: Scope) -> Vec { + let sources = inputs + .into_iter() + .map(|(id, batch)| { + let source = Operator::source(batch.schema().clone(), vec![batch]).unwrap(); + (id, Box::new(source) as Source<'_>) + }) + .collect(); + let dag = plan.instantiate(sources).unwrap(); + block_on(async { + let context = RunContext::new(scope, Limits::default()).unwrap(); + let mut output = dag.execute(plan.roots(), context).unwrap().remove(0); + let mut batches = vec![]; + while let Some(batch) = output.next().await { + batches.push((*batch.unwrap()).clone()); + } + batches + }) +} + +/// Rate readouts and grouped Sum can run together during bounded precompute; +/// storing per-series rates instead leaves the same Sum in the query DAG. +#[test] +fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { + let dag = grouped_rate(); + let state = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) + .unwrap(); + let readout = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator + } + ) && dag + .edges + .iter() + .any(|edge| edge.producer == state.id && edge.consumer == node.id) + }) + .unwrap(); + let input_schema = Arc::new(state.output_schema.clone()); + let (family, update, grouping) = match &state.payload { + PostAsapOperatorPayload::SummaryAgg { + family, + input, + grouping, + .. + } => (family, input, grouping), + _ => unreachable!(), + }; + let range_ms = Some((-58_000, 2_000)); + let mut expected_rate_sum = 0.; + let rows = [[100., 0., 100.], [100., 200., 0.]] + .into_iter() + .enumerate() + .map(|(index, values)| { + let mut accumulator = create_planner_accumulator(family, update, grouping).unwrap(); + for (i, value) in values.into_iter().enumerate() { + accumulator.update_single(value, i as i64 * 1000); + } + let state = accumulator.into_accumulator(); + expected_rate_sum += state + .as_any() + .downcast_ref::() + .unwrap() + .readout(asap_physical_operators::Statistic::Rate, range_ms, None) + .unwrap() + .unwrap(); + let summary = Value::Summary { + family: family.clone(), + state: Arc::from(state), + }; + input_schema + .fields + .iter() + .map(|field| match &field.dtype { + SummaryFamilyType::ExactAggregate(..) => summary.clone(), + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(2000), + SummaryFamilyType::Plain(DataType::Utf8) => { + Value::Utf8(if field.name == "job" { + "api".into() + } else { + format!("series-{index}").into() + }) + } + _ => panic!("unexpected input field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(input_schema.clone(), rows).unwrap(); + let root = u64::from(dag.root.0); + let state_id = u64::from(state.id.0); + let rate_id = u64::from(readout.id.0); + let frontiers = asap_physical_operators::physical_planner::enumerate_frontiers( + &dag, + &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), + &[root], + 128, + ) + .unwrap(); + assert!(frontiers.contains(&vec![])); + assert!(frontiers.contains(&vec![rate_id])); + assert!(frontiers.contains(&vec![root])); + assert!(!frontiers.contains(&vec![root, rate_id])); + assert!( + asap_physical_operators::physical_planner::enumerate_frontiers( + &dag, + &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), + &[root], + 1, + ) + .is_err() + ); + let candidates = compile_candidates( + &dag, + BTreeMap::from([(state_id, InputContract::bounded(input_schema))]), + &[root], + &[vec![], vec![rate_id], vec![root]], + ); + // Scoped cost fixtures select either precompute boundary. No readers are + // opened during candidate construction or selection. + for prefer_grouped in [false, true] { + let inventory = compile_candidates( + &dag, + BTreeMap::from([( + state_id, + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]), + &[root], + &[vec![999], vec![rate_id], vec![root]], + ); + assert!(inventory[0].is_err()); + let mut evaluated = 0; + let selected = select_candidate(inventory, |candidate| { + evaluated += 1; + let grouped = candidate.materialized_outputs.contains_key(&root); + Ok(Some(CandidateCost { + workload_scope: "reset-counter-workload".into(), + horizon_seconds: 300., + total_cost: if grouped == prefer_grouped { 1. } else { 100. }, + })) + }) + .unwrap(); + assert_eq!( + selected.candidate.materialized_outputs.contains_key(&root), + prefer_grouped + ); + assert_eq!(selected.cost.total_cost, 1.); + assert_eq!(evaluated, 2, "uncompilable candidates must never be priced"); + let candidate = selected.candidate; + let precompute = candidate.precompute.as_ref().unwrap(); + let stored = run( + precompute, + BTreeMap::from([(state_id, batch.clone())]), + Scope::Ingestion { + window_start_ms: -58_000, + window_end_ms: 2000, + revision: 1, + }, + ); + let output = run( + &candidate.query, + BTreeMap::from([(precompute.roots()[0], stored[0].clone())]), + Scope::Query { + evaluation_time_ms: 2000, + revision: 1, + }, + ); + assert!( + matches!(output[0].rows()[0][1], Value::Float64(value) if value == expected_rate_sum) + ); + } + let contracts = BTreeMap::from([( + state_id, + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + for frontier in [vec![rate_id, rate_id], vec![root, rate_id], vec![999]] { + assert!( + asap_physical_operators::physical_planner::compile_candidate( + &dag, + contracts.clone(), + &[root], + &frontier + ) + .is_err() + ); + } + let inventory = compile_candidates( + &dag, + contracts.clone(), + &[root], + &[vec![rate_id], vec![root]], + ); + let selected = select_candidate(inventory, |candidate| { + if candidate.materialized_outputs.contains_key(&root) { + return Ok(None); + } + Ok(Some(CandidateCost { + workload_scope: "same-workload".into(), + horizon_seconds: 300., + total_cost: 100., + })) + }) + .unwrap(); + assert!(selected + .candidate + .materialized_outputs + .contains_key(&rate_id)); + let inventory = compile_candidates(&dag, contracts, &[root], &[vec![rate_id], vec![root]]); + assert!( + select_candidate(inventory, |candidate| Ok(Some(CandidateCost { + workload_scope: "same-workload".into(), + horizon_seconds: if candidate.materialized_outputs.contains_key(&root) { + 60. + } else { + 300. + }, + total_cost: 1., + }))) + .is_err() + ); + let query_scope = Scope::Query { + evaluation_time_ms: 2000, + revision: 1, + }; + let maintenance_scope = Scope::Ingestion { + window_start_ms: -58_000, + window_end_ms: 2000, + revision: 1, + }; + let mut results = vec![]; + for candidate in candidates { + let candidate = candidate.unwrap(); + let inputs = if let Some(precompute) = &candidate.precompute { + let source = Operator::source(batch.schema().clone(), vec![batch.clone()]).unwrap(); + let invalid = precompute + .instantiate(BTreeMap::from([(state_id, Box::new(source) as Source<'_>)])) + .unwrap(); + let context = RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + assert!(invalid.execute(precompute.roots(), context).is_err()); + let stored = run( + precompute, + BTreeMap::from([(state_id, batch.clone())]), + maintenance_scope.clone(), + ); + assert_eq!(stored.len(), 1); + let boundary = precompute.roots()[0]; + assert_eq!( + candidate.materialized_outputs[&boundary].schema, + *stored[0].schema() + ); + BTreeMap::from([(boundary, stored[0].clone())]) + } else { + BTreeMap::from([(state_id, batch.clone())]) + }; + let output = run(&candidate.query, inputs, query_scope.clone()); + assert_eq!(output.len(), 1); + assert_eq!(output[0].rows().len(), 1); + assert!(matches!(&output[0].rows()[0][0], Value::Utf8(job) if job.as_ref() == "api")); + assert!( + matches!(output[0].rows()[0][1], Value::Float64(value) if value == expected_rate_sum) + ); + results.push( + output[0].rows()[0] + .iter() + .map(|value| value.key().unwrap()) + .collect::>(), + ); + } + assert_eq!(results[0], results[1]); + assert_eq!(results[1], results[2]); + let mut wrong_order = create_planner_accumulator(family, update, grouping).unwrap(); + for (i, value) in [200., 200., 100.].into_iter().enumerate() { + wrong_order.update_single(value, i as i64 * 1000); + } + let rate_of_sum = wrong_order + .into_accumulator() + .as_any() + .downcast_ref::() + .unwrap() + .readout(asap_physical_operators::Statistic::Rate, range_ms, None) + .unwrap() + .unwrap(); + assert_ne!( + expected_rate_sum, rate_of_sum, + "counter resets prohibit moving Sum before Rate" + ); +} + +/// Enumerated frontiers include both grouped-result and per-series readout +/// persistence; an explicit Rate-state input retains its original semantics. +#[test] +fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { + use asap_physical_operators::physical_planner::enumerate_frontiers; + let dag = grouped_rate(); + let state = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) + .unwrap(); + let inputs = BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + let roots = [u64::from(dag.root.0)]; + let frontiers = enumerate_frontiers(&dag, &inputs, &roots, 4096).unwrap(); + let candidates = compile_candidates(&dag, inputs.clone(), &roots, &frontiers) + .into_iter() + .collect::, _>>() + .unwrap(); + assert!(candidates.iter().any(|c| c.precompute.is_none())); + assert!(candidates + .iter() + .any(|c| c.materialized_outputs.contains_key(&roots[0]))); + assert!(candidates + .iter() + .any(|c| !c.materialized_outputs.is_empty() + && !c.materialized_outputs.contains_key(&roots[0]))); + assert!(enumerate_frontiers(&dag, &inputs, &roots, 1).is_err()); +} + +#[test] +fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { + let inventory = grouped_rate_space().enumerate_candidate_dags(4096).unwrap(); + let mut executed = 0; + for forest in inventory.candidates { + let root = &forest[0].1; + let dag = compile_post_asap_dag(root).unwrap(); + let Some(state) = dag.nodes.iter().find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) else { + continue; + }; + let boundary = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + .. + } + ) + }) + .map(|node| u64::from(node.id.0)) + .unwrap_or(u64::from(dag.root.0)); + let physical_candidates = compile_candidates( + &dag, + BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &[vec![], vec![boundary]], + ); + let (family, input, grouping) = match &state.payload { + PostAsapOperatorPayload::SummaryAgg { + family, + input, + grouping, + .. + } => (family, input, grouping), + _ => unreachable!(), + }; + let schema = Arc::new(state.output_schema.clone()); + let rows = ["a", "b"] + .into_iter() + .map(|instance| { + let mut accumulator = create_planner_accumulator(family, input, grouping).unwrap(); + for (timestamp, value) in [(1_000, 1.), (31_000, 31.), (59_000, 59.)] { + accumulator.update_single(value, timestamp); + } + let summary = Value::Summary { + family: family.clone(), + state: Arc::from(accumulator.into_accumulator()), + }; + schema + .fields + .iter() + .map(|field| match &field.dtype { + SummaryFamilyType::ExactAggregate(..) => summary.clone(), + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), + SummaryFamilyType::Plain(DataType::Utf8) + if field.name == "$promql_series_identity" => + { + Value::Utf8( + serde_json::to_string(&BTreeMap::from([ + ("job", "api"), + ("instance", instance), + ])) + .unwrap() + .into(), + ) + } + SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8("api".into()), + _ => panic!("unexpected input field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(schema, rows).unwrap(); + for physical in physical_candidates { + let physical = physical.unwrap(); + let inputs = if let Some(precompute) = &physical.precompute { + let source_id = precompute.input_contracts().next().unwrap().0; + let stored = run( + precompute, + BTreeMap::from([(source_id, batch.clone())]), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 60_000, + revision: 1, + }, + ); + assert_eq!(stored.len(), 1); + BTreeMap::from([(precompute.roots()[0], stored[0].clone())]) + } else { + BTreeMap::from([( + physical.query.input_contracts().next().unwrap().0, + batch.clone(), + )]) + }; + let output = run( + &physical.query, + inputs, + Scope::Query { + evaluation_time_ms: 60_000, + revision: 1, + }, + ); + assert_eq!(output.len(), 1); + assert_eq!(output[0].rows().len(), 1); + assert!(output[0] + .schema() + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + assert!( + output[0].rows()[0] + .iter() + .any(|value| matches!(value, Value::Float64(x) if (*x - 2.).abs() < 1e-12)), + "{:?}", + output[0].rows() + ); + executed += 1; + } + } + assert!( + executed >= 2, + "must execute both stored and query-time grouped Rate candidates: {executed}" + ); +} + +/// The per-frontier lowering used before compile-once cuts: each boundary +/// choice lowers the precompute and query DAGs from the logical DAG again. +fn recompiled_candidate( + dag: &PostAsapDag, + inputs: &BTreeMap, + roots: &[u64], + frontier: &[u64], +) -> Result { + use asap_physical_operators::plan::Emission; + if frontier.is_empty() { + return Ok(PhysicalCandidate { + precompute: None, + query: compile(dag, inputs.clone(), roots)?, + materialized_outputs: BTreeMap::new(), + }); + } + let precompute = compile(dag, inputs.clone(), frontier)?; + let mut materialized_outputs = BTreeMap::new(); + for &id in frontier { + let mut output = precompute.output_contract(id)?; + output.properties.emission = Emission::Unknown; + materialized_outputs.insert(id, output); + } + let mut query_inputs = inputs.clone(); + query_inputs.extend(materialized_outputs.clone()); + Ok(PhysicalCandidate { + precompute: Some(precompute), + query: compile(dag, query_inputs, roots)?, + materialized_outputs, + }) +} + +fn assert_cuts_match_recompilation( + dag: &PostAsapDag, + inputs: BTreeMap, + roots: &[u64], + min_frontiers: usize, +) { + let compiled = compile(dag, inputs.clone(), roots).unwrap(); + let frontiers = enumerate_frontiers(dag, &inputs, roots, 4096).unwrap(); + assert!(frontiers.len() >= min_frontiers, "{frontiers:?}"); + for frontier in &frontiers { + let cut = cut_candidate(&compiled, frontier).unwrap(); + let expected = recompiled_candidate(dag, &inputs, roots, frontier).unwrap(); + assert_eq!( + serde_json::to_vec(&cut).unwrap(), + serde_json::to_vec(&expected).unwrap(), + "{frontier:?}" + ); + } +} + +/// Every enumerated grouped Rate→Sum frontier (query-only, stored Rate, +/// stored Sum) cuts to exactly the candidate that per-frontier lowering builds. +#[test] +fn grouped_rate_cuts_equal_per_frontier_compilation() { + let dag = grouped_rate(); + let state = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .unwrap(); + let inputs = BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + assert_cuts_match_recompilation(&dag, inputs, &[u64::from(dag.root.0)], 3); +} + +/// Cuts of a DAG whose nodes lower to helper operators (current-series +/// population read by Sort→Limit) keep the same operator IDs as recompilation. +#[test] +fn population_topk_cuts_equal_per_frontier_compilation() { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("topk by(job)(1, m)".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(60_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let original = asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0); + let root = Rc::new( + asap_physical_operators::physical_planner::promql_rows::with_series_identity(&original) + .unwrap(), + ); + let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&root), + ) + .candidate(&root) + .unwrap(); + let dag = compile_post_asap_dag(&selected).unwrap(); + let raw = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .unwrap(); + let inputs = BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(Arc::new(raw.output_schema.clone())), + )]); + let roots = [u64::from(dag.root.0)]; + let compiled = compile(&dag, inputs.clone(), &roots).unwrap(); + // The root reads its population through a Sort helper numbered by the root. + let helper = u64::MAX - (roots[0] << 16); + assert_eq!(compiled.operator_name(helper), Some("Sort")); + assert!( + cut_candidate(&compiled, &[helper]).is_err(), + "helper operators are not Planner boundaries" + ); + assert_cuts_match_recompilation(&dag, inputs, &roots, 2); +} + +/// Cuts reject frontiers that recompilation rejects: duplicates, inputs, +/// unknown IDs, and an output shadowed by its descendant. +#[test] +fn cut_candidate_rejects_invalid_frontiers() { + let dag = grouped_rate(); + let state = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .unwrap(); + let readout = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator + } + ) + }) + .unwrap(); + let (state_id, rate_id, root) = ( + u64::from(state.id.0), + u64::from(readout.id.0), + u64::from(dag.root.0), + ); + let inputs = BTreeMap::from([( + state_id, + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + let compiled = compile(&dag, inputs.clone(), &[root]).unwrap(); + for frontier in [ + vec![rate_id, rate_id], + vec![state_id], + vec![999], + vec![root, rate_id], + ] { + assert!(cut_candidate(&compiled, &frontier).is_err(), "{frontier:?}"); + assert!(compile_candidate(&dag, inputs.clone(), &[root], &frontier).is_err()); + } +} diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs new file mode 100644 index 00000000..3bb3af6c --- /dev/null +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -0,0 +1,425 @@ +//! Persisted precompute graphs preserve group/window identity and execute state-to-state computation. +use asap_physical_operators::{ + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{precompute, CompiledPhysicalDag, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, + Statistic, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{ArithmeticOpKind, BinaryOpKind, ColumnRef, DataType, GroupKeys, Reduction}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +#[test] +fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let schema = |dtype| SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype, + nullable: false, + }], + time_index: None, + }; + let state_schema = schema(family.clone()); + let mut value_schema = schema(SummaryFamilyType::Plain(DataType::Float64)); + value_schema.fields.push(SummaryField { + name: "time".into(), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), + nullable: false, + }); + value_schema.time_index = Some(1); + for (weight, expected) in [ + (SummaryInputExpr::Column(ColumnRef::SampleValue), 60.), + (SummaryInputExpr::Constant(1.), 4.), + ] { + let nodes = vec![ + PostAsapDagNode { + id: PostAsapNodeId(0), + payload: PostAsapOperatorPayload::SummaryMerge, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: state_schema.clone(), + guarantee: None, + }, + PostAsapDagNode { + id: PostAsapNodeId(1), + payload: PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + }, + output_state: ExecutionDataState::INGESTION_ROWS, + output_schema: value_schema.clone(), + guarantee: None, + }, + PostAsapDagNode { + id: PostAsapNodeId(2), + payload: PostAsapOperatorPayload::Binary { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + }, + output_state: ExecutionDataState::INGESTION_ROWS, + output_schema: value_schema.clone(), + guarantee: None, + }, + PostAsapDagNode { + id: PostAsapNodeId(3), + payload: PostAsapOperatorPayload::SummaryAgg { + family: family.clone(), + input: SummaryUpdate { + weight, + ..SummaryUpdate::column(ColumnRef::SampleValue) + }, + reduction: Reduction::Reduce(GroupKeys::by(vec![])), + grouping: GroupingStrategy::PerSubpopulationInstance, + }, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: state_schema.clone(), + guarantee: None, + }, + ]; + let edges = [ + (0, 1, EdgeRole::Input), + (1, 2, EdgeRole::Left), + (1, 2, EdgeRole::Right), + (2, 3, EdgeRole::Input), + ] + .into_iter() + .map(|(producer, consumer, role)| PostAsapDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role, + intermediate_schema: nodes[producer as usize].output_schema.clone(), + data_state: nodes[producer as usize].output_state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }) + .collect(); + let dag = PostAsapDag { + nodes, + edges, + root: PostAsapNodeId(3), + }; + let mut invalid_grouping = dag.clone(); + let PostAsapOperatorPayload::SummaryAgg { reduction, .. } = + &mut invalid_grouping.nodes[3].payload + else { + unreachable!() + }; + *reduction = Reduction::Reduce(GroupKeys::by(vec![0])); + assert!( + precompute::compile(&invalid_grouping, &[0], &[3]).is_err(), + "numeric values cannot be reinterpreted as population labels" + ); + let program = precompute::compile(&dag, &[0], &[3]).unwrap(); + let program = + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + .unwrap(); + assert_eq!(program.input_contracts().count(), 1); + for revision in [1, 2] { + let rows = [ + ("a", 1000, 2.), + ("a", 2000, 4.), + ("b", 1000, 8.), + ("b", 2000, 16.), + ] + .into_iter() + .map(|(group, time, value)| { + let mut state = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + state.update_single(value, time); + vec![ + Value::Map( + vec![(Value::Utf8("instance".into()), Value::Utf8(group.into()))].into(), + ), + Value::Timestamp(time), + Value::Summary { + family: family.clone(), + state: Arc::from(state.into_accumulator()), + }, + ] + }) + .collect(); + let input = + Batch::try_new(precompute::population_schema(family.clone()), rows).unwrap(); + let sources = BTreeMap::from([( + 0, + Box::new(Operator::source(input.schema().clone(), vec![input]).unwrap()) + as Source<'_>, + )]); + let graph = program.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision, + }, + Limits::default(), + ) + .unwrap(); + let output = block_on(async { + let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let batch = stream.next().await.unwrap().unwrap(); + assert!(stream.next().await.is_none()); + batch + }); + assert_eq!(output.rows().len(), 1); + assert!(matches!(&output.rows()[0][0], Value::Map(labels) if labels.is_empty())); + assert!(matches!(&output.rows()[0][1], Value::Timestamp(2000))); + let Value::Summary { state, .. } = &output.rows()[0][2] else { + panic!("state expected") + }; + assert_eq!( + state + .as_any() + .downcast_ref::() + .unwrap() + .readout(Statistic::Sum, None, None) + .unwrap() + .unwrap(), + expected + ); + } + } +} + +fn logical_schema(family: SummaryFamilyType) -> SummarySchema { + SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: family, + nullable: false, + }], + time_index: None, + } +} +fn state_graph( + family: SummaryFamilyType, + target: Option, + merge: bool, +) -> CompiledPhysicalDag { + let mut nodes = vec![PostAsapDagNode { + id: PostAsapNodeId(0), + payload: PostAsapOperatorPayload::SummaryMerge, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: logical_schema(family.clone()), + guarantee: None, + }]; + if merge { + nodes.push(PostAsapDagNode { + id: PostAsapNodeId(1), + payload: PostAsapOperatorPayload::SummaryMerge, + ..nodes[0].clone() + }); + } + let read_id = nodes.len() as u32; + nodes.push(PostAsapDagNode { + id: PostAsapNodeId(read_id), + payload: PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + }, + output_state: ExecutionDataState::INGESTION_ROWS, + output_schema: logical_schema(SummaryFamilyType::Plain(DataType::Float64)), + guarantee: None, + }); + if let Some(target) = target { + nodes.push(PostAsapDagNode { + id: PostAsapNodeId(nodes.len() as u32), + payload: PostAsapOperatorPayload::SummaryAgg { + family: target.clone(), + input: SummaryUpdate::column(ColumnRef::SampleValue), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + }, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: logical_schema(target), + guarantee: None, + }); + } + let edges = (1..nodes.len()) + .map(|i| PostAsapDagEdge { + producer: nodes[i - 1].id, + consumer: nodes[i].id, + role: EdgeRole::Input, + intermediate_schema: nodes[i - 1].output_schema.clone(), + data_state: nodes[i - 1].output_state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }) + .collect(); + let root = nodes.last().unwrap().id; + precompute::compile( + &PostAsapDag { nodes, edges, root }, + &[0], + &[u64::from(root.0)], + ) + .unwrap() +} +fn native_run( + program: &CompiledPhysicalDag, + family: SummaryFamilyType, + states: Vec>, + context: RunContext, +) -> Result>, asap_physical_operators::Error> { + let program = + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + .unwrap(); + let rows = states + .into_iter() + .enumerate() + .map(|(i, state)| { + vec![ + Value::Map(vec![].into()), + Value::Timestamp((i as i64 + 1) * 1000), + Value::Summary { + family: family.clone(), + state, + }, + ] + }) + .collect(); + let input = Batch::try_new(precompute::population_schema(family), rows)?; + let graph = program.instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(input.schema().clone(), vec![input])?) as Source<'_>, + )]))?; + block_on(async { + let mut rows = Vec::new(); + let mut stream = graph.execute(program.roots(), context)?.remove(0); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok(rows) + }) +} +fn ingestion_context(limits: Limits) -> RunContext { + RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }, + limits, + ) + .unwrap() +} +fn sum_state(value: f64) -> Arc { + let mut state = asap_physical_operators::summary_kernels::exact::ExactAccumulator::new( + planner_types::post_asap::SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Sum, + planner_types::post_asap::ExactParams::Sum, + ), + false, + ) + .unwrap(); + state.update(None, value, 0); + Arc::new(state) +} + +// Only an explicit merge may collapse distinct pane updates before finalization. +#[test] +fn explicit_merge_changes_pane_cardinality() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + for (merge, expected) in [(false, vec![2., 7.]), (true, vec![9.])] { + let program = state_graph(family.clone(), None, merge); + let rows = native_run( + &program, + family.clone(), + vec![sum_state(2.), sum_state(7.)], + ingestion_context(Limits::default()), + ) + .unwrap(); + let values = rows + .iter() + .map(|row| match row[2] { + Value::Float64(v) => v, + _ => panic!("numeric readout expected"), + }) + .collect::>(); + assert_eq!(values, expected); + assert!(matches!(rows.last().unwrap()[1], Value::Timestamp(2000))); + } +} + +// Typed updates reject invalid domains before publishing any target state. +#[test] +fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { + let source = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let target = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::DDSketch, + SketchParams::DDSketch { alpha: 0.01 }, + ), + GroupingStrategy::default(), + ); + let program = state_graph(source.clone(), Some(target), false); + assert!(native_run( + &program, + source.clone(), + vec![sum_state(20.)], + ingestion_context(Limits::default()) + ) + .is_ok()); + for value in [-20., 0., f64::MAX, f64::NAN, f64::INFINITY] { + assert!(native_run( + &program, + source.clone(), + vec![sum_state(value)], + ingestion_context(Limits::default()) + ) + .is_err()); + } +} + +// An exact count must not silently lose units when exposed through Float64 rows. +#[test] +fn precompute_count_conversion_checks_precision() { + use asap_physical_operators::summary_kernels::exact::ExactAccumulator; + let family = SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count); + let program = state_graph(family.clone(), None, false); + for (count, valid) in [(3u64, true), ((1u64 << 53) + 1, false)] { + let mut state = + serde_json::to_value(ExactAccumulator::new(family.clone(), false).unwrap()).unwrap(); + state["scalar"]["Count"] = count.into(); + let state: ExactAccumulator = serde_json::from_value(state).unwrap(); + let result = native_run( + &program, + family.clone(), + vec![Arc::new(state)], + ingestion_context(Limits::default()), + ); + assert_eq!(result.is_ok(), valid); + } +} + +// Graph execution retains terminal cancellation and shared workspace limits. +#[test] +fn precompute_graph_enforces_cancellation_and_budget() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let program = state_graph(family.clone(), None, true); + let context = ingestion_context(Limits::default()); + context.cancel(); + let error = native_run(&program, family.clone(), vec![sum_state(1.)], context).unwrap_err(); + assert!(format!("{error:?}").contains("Cancelled")); + let error = native_run( + &program, + family, + vec![sum_state(1.)], + ingestion_context(Limits { + max_bytes: 1, + ..Limits::default() + }), + ) + .unwrap_err(); + assert!(format!("{error:?}").contains("MemoryLimit")); +} diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs new file mode 100644 index 00000000..24844b5e --- /dev/null +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -0,0 +1,270 @@ +//! Binary computation must be fully compiled before deployment binds values. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{compile_node, CompiledPhysicalDag, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Schema, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{ + BinaryOperator, ExecutionDataState, PostAsapDagNode, PostAsapNodeId, + PostAsapOperatorPayload, SummaryFamilyType, SummaryField, SummarySchema, + }, + pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +fn schema() -> Schema { + Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "labels".into(), + dtype: SummaryFamilyType::Plain(DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }), + nullable: false, + }, + SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }, + ], + time_index: None, + }) +} +fn row(name: &str, job: &str, value: f64) -> Vec { + vec![ + Value::Map( + vec![ + (Value::Utf8("__name__".into()), Value::Utf8(name.into())), + (Value::Utf8("job".into()), Value::Utf8(job.into())), + ] + .into(), + ), + Value::Float64(value), + ] +} +fn program() -> CompiledPhysicalDag { + let schema = schema(); + let node = PostAsapDagNode { + id: PostAsapNodeId(2), + payload: PostAsapOperatorPayload::Binary { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + vector_match: None, + checked_relative_division: true, + checked_finite_division: false, + }, + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*schema).clone(), + guarantee: None, + }; + let operator = compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(); + let graph = CompiledPhysicalDag::from_operators( + BTreeMap::from([ + (0, InputContract::bounded(schema.clone())), + (1, InputContract::bounded(schema)), + ]), + BTreeMap::from([(2, (vec![0, 1], operator))]), + vec![2], + ) + .unwrap(); + serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()).unwrap() +} +fn evaluate( + left: Vec>, + right: Vec>, +) -> Result>, asap_physical_operators::Error> { + let graph = program(); + let sources = [left, right] + .into_iter() + .enumerate() + .map(|(id, rows)| { + let batch = Batch::try_new(schema(), rows).unwrap(); + ( + id as u64, + Box::new(Operator::source(schema(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect(); + let bound = graph.instantiate(sources)?; + let ctx = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 0, + }, + Limits::default(), + )?; + block_on(async { + let mut stream = bound.execute(&[2], ctx)?.remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok(rows) + }) +} + +#[test] +fn compiled_binary_matches_series_and_preserves_checked_division() { + let rows = evaluate( + vec![row("left", "api", 6.), row("left", "unmatched", 8.)], + vec![row("right", "api", 2.)], + ) + .unwrap(); + assert_eq!( + serde_json::to_value(&rows).unwrap(), + serde_json::to_value(vec![vec![ + Value::Map(vec![(Value::Utf8("job".into()), Value::Utf8("api".into()))].into()), + Value::Float64(3.) + ]]) + .unwrap() + ); + assert!(evaluate(vec![row("a", "api", 1.)], vec![row("b", "api", 0.)]).is_err()); +} + +#[test] +fn duplicate_matching_identity_is_rejected() { + assert!(evaluate( + vec![row("a", "api", 1.)], + vec![row("b", "api", 2.), row("c", "api", 3.)] + ) + .is_err()); +} + +// Scalar broadcasting and comparison filtering keep the vector operand's value. +#[test] +fn scalar_broadcast_and_bool_comparison_are_distinct() { + use asap_physical_operators::physical_planner::promql_values; + use planner_types::pre_asap::CompareOpKind; + for return_bool in [false, true] { + let graph = promql_values::compile_binary( + &BinaryOperator { + kind: BinaryOpKind::Compare(CompareOpKind::Lt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + true, + false, + ) + .unwrap(); + let graph = + serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + .unwrap(); + let scalar = promql_values::scalar_schema(); + let vector = promql_values::vector_schema(); + let sources = BTreeMap::from([ + ( + 0, + Box::new( + Operator::source( + scalar.clone(), + vec![Batch::try_new(scalar, vec![vec![Value::Float64(2.)]]).unwrap()], + ) + .unwrap(), + ) as Source<'_>, + ), + ( + 1, + Box::new( + Operator::source( + vector.clone(), + vec![Batch::try_new( + vector, + vec![row("requests", "api", 4.), row("requests", "worker", 1.)], + ) + .unwrap()], + ) + .unwrap(), + ) as Source<'_>, + ), + ]); + let bound = graph.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + let result = block_on(async { + bound + .execute(&[2], context) + .unwrap() + .remove(0) + .next() + .await + .unwrap() + .unwrap() + }); + assert_eq!(result.rows().len(), if return_bool { 2 } else { 1 }); + assert!( + matches!(result.rows()[0].last(), Some(Value::Float64(v)) if *v == if return_bool { 1. } else { 4. }) + ); + let Value::Map(labels) = &result.rows()[0][0] else { + panic!("missing labels") + }; + assert_eq!( + labels + .iter() + .any(|(key, _)| matches!(key, Value::Utf8(s) if s.as_ref() == "__name__")), + !return_bool + ); + } +} + +// Terminal request controls retain their native error classification. +#[test] +fn binary_obeys_memory_and_cancellation() { + for cancel in [false, true] { + let graph = program(); + let sources = (0..2) + .map(|id| { + ( + id, + Box::new( + Operator::source( + schema(), + vec![Batch::try_new(schema(), vec![row("x", "api", 1.)]).unwrap()], + ) + .unwrap(), + ) as Source<'_>, + ) + }) + .collect(); + let bound = graph.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 0, + }, + Limits { + max_bytes: if cancel { 10000 } else { 1 }, + ..Limits::default() + }, + ) + .unwrap(); + if cancel { + context.cancel(); + } + let result = block_on(async { + match bound.execute(&[2], context) { + Err(error) => Err(error), + Ok(mut streams) => streams.remove(0).next().await.unwrap().map(|_| ()), + } + }); + assert!(matches!( + (cancel, result), + (true, Err(asap_physical_operators::Error::Cancelled)) + | (false, Err(asap_physical_operators::Error::MemoryLimit)) + )); + } +} diff --git a/crates/asap-physical-operators/tests/promql_fallback.rs b/crates/asap-physical-operators/tests/promql_fallback.rs new file mode 100644 index 00000000..d274495c --- /dev/null +++ b/crates/asap-physical-operators/tests/promql_fallback.rs @@ -0,0 +1,624 @@ +//! A retained PromQL subtree (`Fallback`) compiles from its typed expression. +//! The deployment supplies only its selector's raw series; expected values are +//! hand-computed with Prometheus semantics. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{compile, promql_fallback, promql_rows, CompiledPhysicalDag, InputContract}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{execution_data_state::lift_plain, *}, + pre_asap::QueryExpr, + types::AccuracyTarget, + workload::*, +}; +use std::{collections::BTreeMap, rc::Rc}; + +/// Bare selectors look back one ingestion interval: 60s. +fn parse(query: &str) -> QueryExpr { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(60_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0) +} + +fn lower(query: &str) -> QueryExpr { + promql_rows::with_series_identity(&parse(query)).unwrap() +} + +/// The whole query retained as one pre-ASAP node. +fn fallback_dag(expression: QueryExpr) -> PostAsapDag { + let schema = lift_plain(&expression.output_schema().unwrap()); + compile_post_asap_dag(&Rc::new(SummaryNode { + expr: SummaryExpr::KeepPreAsap(Rc::new(expression)), + schema, + guarantee: None, + })) + .unwrap() +} + +/// `(labels, seconds, value)`. `labels` is `k=v,...`, or a bare `job` value. +type Sample = (&'static str, i64, f64); + +fn labels(spec: &str) -> BTreeMap { + if !spec.contains('=') { + return BTreeMap::from([("job".into(), spec.into())]); + } + spec.split(',') + .map(|pair| { + let (k, v) = pair.split_once('=').unwrap(); + (k.to_string(), v.to_string()) + }) + .collect() +} + +/// The metric a selector reads. +fn metric(selector: &QueryExpr) -> String { + match selector { + QueryExpr::Scan { + source: planner_types::pre_asap::Source::TimeSeries { metric }, + .. + } => metric.clone(), + QueryExpr::TimeRange { child, .. } | QueryExpr::TimeShift { child, .. } => metric(child), + other => panic!("not a selector: {other:?}"), + } +} + +fn compile_query(query: &str) -> Result { + let expression = lower(query); + let dag = fallback_dag(expression.clone()); + let root = u64::from(dag.root.0); + let inputs = promql_fallback::raw_series(&expression) + .map_err(|e| e.to_string())? + .into_iter() + .enumerate() + .map(|(i, (_, schema))| { + ( + promql_fallback::raw_series_input(root, i), + InputContract::bounded(schema), + ) + }) + .collect(); + let program = compile(&dag, inputs, &[root]).map_err(|e| e.to_string())?; + Ok(serde_json::from_slice(&serde_json::to_vec(&program).unwrap()).unwrap()) +} + +/// Evaluate at `at` seconds over samples of each named metric; returns +/// `(output labels, timestamp ms, value)` rows in order. +#[allow(clippy::type_complexity)] +fn evaluate( + query: &str, + metrics: &[(&str, &[Sample])], + at: i64, +) -> Result, i64, f64)>, String> { + let expression = lower(query); + let program = compile_query(query)?; + let mut sources = BTreeMap::new(); + let selectors = promql_fallback::raw_series(&expression).unwrap(); + for (i, (selector, schema)) in selectors.into_iter().enumerate() { + let name = metric(&selector); + let rows = metrics + .iter() + .filter(|(m, _)| *m == name) + .flat_map(|(_, samples)| samples.iter()) + .map(|(spec, seconds, value)| { + let mut labels = labels(spec); + labels.insert("__name__".into(), name.clone()); + promql_rows::series_row(&schema, &labels, seconds * 1000, *value).unwrap() + }) + .collect(); + let batch = Batch::try_new(schema.clone(), rows).unwrap(); + sources.insert( + promql_fallback::raw_series_input(program.roots()[0], i), + Box::new(Operator::source(schema, vec![batch]).unwrap()) as _, + ); + } + let graph = program.instantiate(sources).map_err(|e| e.to_string())?; + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: at * 1000, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = graph + .execute(program.roots(), context) + .map_err(|e| e.to_string())? + .remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + let batch = batch.map_err(|e| e.to_string())?; + let schema = batch.schema().clone(); + for row in batch.rows() { + let mut labels = BTreeMap::new(); + let mut time = -1; + let mut value = None; + for (field, cell) in schema.fields.iter().zip(row) { + match (field.name.as_str(), cell) { + (promql_rows::SERIES_IDENTITY_COLUMN, Value::Utf8(id)) => { + labels = promql_rows::decode_series_identity(id).unwrap() + } + (_, Value::Utf8(_) | Value::Null) => {} + (_, Value::Timestamp(t)) => time = *t, + (_, Value::Float64(v)) => value = Some(*v), + (_, Value::Int64(v)) => value = Some(*v as f64), + other => return Err(format!("unexpected cell {other:?}")), + } + } + if !schema + .fields + .iter() + .any(|f| f.name == promql_rows::SERIES_IDENTITY_COLUMN) + { + for (field, cell) in schema.fields.iter().zip(row) { + if let Value::Utf8(label) = cell { + if !label.is_empty() { + labels.insert(field.name.clone(), label.to_string()); + } + } + } + } + rows.push((labels, time, value.ok_or("missing value")?)); + } + } + Ok(rows) + }) +} + +/// Evaluate at `at` seconds over metric `m`; returns `(job or "", timestamp ms, value)`. +fn run(query: &str, samples: &[Sample], at: i64) -> Result, String> { + Ok(evaluate(query, &[("m", samples)], at)? + .into_iter() + .map(|(labels, time, value)| (labels.get("job").cloned().unwrap_or_default(), time, value)) + .collect()) +} + +/// Output rows as `(k=v,... sorted, value)`, including any `__name__`. +fn labeled(query: &str, metrics: &[(&str, &[Sample])], at: i64) -> Vec<(String, f64)> { + let mut rows = evaluate(query, metrics, at) + .unwrap_or_else(|e| panic!("{query}: {e}")) + .into_iter() + .map(|(labels, _, value)| { + let spec = labels + .iter() + .map(|(k, v)| format!("{k}={v}")) + .collect::>() + .join(","); + (spec, value) + }) + .collect::>(); + rows.sort_by(|a, b| a.0.cmp(&b.0)); + rows +} + +fn values(query: &str, samples: &[Sample], at: i64) -> Vec<(String, f64)> { + run(query, samples, at) + .unwrap_or_else(|e| panic!("{query}: {e}")) + .into_iter() + .map(|(job, _, value)| (job, value)) + .collect() +} + +fn one(query: &str, samples: &[Sample], at: i64) -> f64 { + match values(query, samples, at).as_slice() { + [(_, value)] => *value, + other => panic!("{query}: expected one sample, got {other:?}"), + } +} + +const COUNTER: &[Sample] = &[ + ("a", 60, 10.), + ("a", 120, 20.), + ("a", 180, 5.), + ("a", 240, 15.), +]; + +// rate/increase correct the reset at 180s and extrapolate half an interval at +// most; delta treats the same samples as a gauge. +#[test] +fn range_functions_follow_prometheus_extrapolation_and_resets() { + // Reset-corrected increase is 25 over 180s of samples; 60s on each side extrapolates. + let increase = 25. * (180. + 60. + 60.) / 180.; + assert!((one("increase(m[5m])", COUNTER, 300) - increase).abs() < 1e-9); + assert!((one("rate(m[5m])", COUNTER, 300) - increase / 300.).abs() < 1e-12); + let delta = 5. * (180. + 60. + 60.) / 180.; + assert!((one("delta(m[5m])", COUNTER, 300) - delta).abs() < 1e-9); + // Fewer than two samples yield no rate. + assert!(values("rate(m[2m])", COUNTER, 300).is_empty()); + for (query, expected) in [ + ("sum_over_time(m[5m])", 50.), + ("avg_over_time(m[5m])", 12.5), + ("min_over_time(m[5m])", 5.), + ("max_over_time(m[5m])", 20.), + ("count_over_time(m[5m])", 4.), + ] { + assert_eq!(one(query, COUNTER, 300), expected, "{query}"); + } +} + +// Ranges are left-open: a sample at `t - range` is excluded, one at `t` is included. +#[test] +fn ranges_exclude_their_start_and_offsets_shift_them() { + let samples = &[("a", 60, 1.), ("a", 90, 1.), ("a", 120, 1.), ("a", 150, 1.)]; + assert_eq!(one("count_over_time(m[1m])", samples, 120), 2.); + // offset 1m reads (60s, 120s] at 180s; output keeps the evaluation time. + let rows = run("count_over_time(m[1m] offset 1m)", samples, 180).unwrap(); + assert_eq!(rows, vec![("a".into(), 180_000, 2.)]); +} + +// A bare selector takes the latest sample within the lookback; a stale marker +// hides the series rather than exposing an older value. +#[test] +fn instant_selection_uses_lookback_and_stale_markers() { + let stale = f64::from_bits(0x7ff0_0000_0000_0002); + let samples = &[("a", 0, 1.), ("a", 30, 2.), ("b", 30, 3.), ("b", 50, stale)]; + assert_eq!(values("m", samples, 60), vec![("a".into(), 2.)]); + // The lookback (30s, 90s] excludes the sample at 30s. + assert!(values("m", samples, 90).is_empty()); + // Range functions skip stale markers. + assert_eq!( + values("sum_over_time(m[1m])", samples, 60), + vec![("a".into(), 2.), ("b".into(), 3.)] + ); +} + +// NaN samples follow Prometheus: min/max skip them, sums propagate them. +#[test] +fn nan_samples() { + let samples = &[("a", 10, f64::NAN), ("a", 20, 3.), ("a", 30, 1.)]; + assert_eq!(one("max_over_time(m[1m])", samples, 60), 3.); + assert_eq!(one("min_over_time(m[1m])", samples, 60), 1.); + assert!(one("sum_over_time(m[1m])", samples, 60).is_nan()); +} + +// Aggregation over no series is an empty vector, not one zero or null row; +// sort_desc orders the selected series. +#[test] +fn cross_series_aggregates_and_empty_inputs() { + let samples = &[("a", 50, 1.), ("b", 40, 2.), ("b", 55, 4.)]; + assert_eq!(values("sum(m)", samples, 60), vec![(String::new(), 5.)]); + assert_eq!(values("count(m)", samples, 60), vec![(String::new(), 2.)]); + assert_eq!( + values("max by (job) (m)", samples, 60), + vec![("a".into(), 1.), ("b".into(), 4.)] + ); + assert_eq!( + values("sort_desc(m)", samples, 60), + vec![("b".into(), 4.), ("a".into(), 1.)] + ); + // topk by (job) keeps the top series of each job, not one overall. + let jobs = &[("a", 50, 1.), ("b", 50, 2.)]; + let mut top = values("topk by (job) (1, m)", jobs, 60); + top.sort_by(|x, y| x.0.cmp(&y.0)); + assert_eq!(top, vec![("a".into(), 1.), ("b".into(), 2.)]); + assert_eq!(values("topk(1, m)", jobs, 60), vec![("b".into(), 2.)]); + for query in ["sum(m)", "count(m)", "max(m)", "sum by (job) (rate(m[5m]))"] { + assert!(values(query, &[], 60).is_empty(), "{query}"); + } +} + +// scalar() is the single series' value and NaN otherwise; vector() needs no input. +#[test] +fn scalar_and_vector_bridges() { + assert_eq!(one("scalar(m)", &[("a", 50, 7.)], 60), 7.); + assert!(one("scalar(m)", &[("a", 50, 7.), ("b", 50, 8.)], 60).is_nan()); + assert!(one("scalar(m)", &[], 60).is_nan()); + assert_eq!( + run("vector(3)", &[], 60).unwrap(), + vec![(String::new(), 60_000, 3.)] + ); + assert_eq!( + values("2 - m", &[("a", 50, 7.)], 60), + vec![("a".into(), -5.)] + ); + assert_eq!( + values("m * 2", &[("a", 50, 7.)], 60), + vec![("a".into(), 14.)] + ); +} + +// Subquery steps are absolute multiples of the resolution in the left-open +// range; each step evaluates the operand, and the outer function reduces them. +#[test] +fn subqueries_evaluate_their_operand_on_the_aligned_grid() { + // Steps 60..300: selections 1, 7, 3, (none at 240s), 4. + let samples = &[ + ("a", 50, 1.), + ("a", 110, 7.), + ("a", 170, 3.), + ("a", 290, 4.), + ]; + assert_eq!(one("max_over_time(m[5m:1m])", samples, 300), 7.); + assert_eq!(one("count_over_time(m[5m:1m])", samples, 300), 4.); + // At 190s the steps are 60, 120, 180 (not 70, 130, 190): counts 1 + 2 + 2. + let samples = &[("a", 30, 1.), ("a", 90, 1.), ("a", 150, 1.), ("a", 185, 1.)]; + assert_eq!( + one("sum_over_time(count_over_time(m[2m])[3m:1m])", samples, 190), + 5. + ); + // offset 1m moves the grid to (-50s, 130s]: steps 0, 60, 120 count 0 + 1 + 2. + assert_eq!( + one( + "sum_over_time(count_over_time(m[2m])[3m:1m] offset 1m)", + samples, + 190 + ), + 3. + ); +} + +// Subquery work is bounded by the query: at most 100000 steps. +#[test] +fn dense_subquery_grids_are_rejected() { + assert!(compile_query("max_over_time(m[100s:1ms])").is_ok()); + assert!(compile_query("max_over_time(m[30d:1ms])").is_err()); +} + +// The deployment must supply the selector's raw rows under the documented slot +// with the exact selector schema; unsupported shapes stay rejected. +#[test] +fn raw_series_contract_is_explicit() { + let expression = lower("rate(m[5m])"); + let dag = fallback_dag(expression.clone()); + let root = u64::from(dag.root.0); + let [(selector, schema)] = promql_fallback::raw_series(&expression) + .unwrap() + .try_into() + .unwrap(); + assert!(matches!(selector, QueryExpr::TimeRange { .. })); + let missing = compile(&dag, BTreeMap::new(), &[root]).err().unwrap(); + assert!(missing.to_string().contains("raw series input")); + let mut wrong = (*schema).clone(); + wrong.fields.pop(); + let wrong = compile( + &dag, + BTreeMap::from([( + promql_fallback::raw_series_input(root, 0), + InputContract::bounded(std::sync::Arc::new(wrong)), + )]), + &[root], + ); + assert!(wrong.is_err()); + // A consumed bare selector is raw range rows for its consumer; it is not + // turned into instant selection. + let selector = lower("m"); + let schema = lift_plain(&selector.output_schema().unwrap()); + let node = |id, payload| PostAsapDagNode { + id: PostAsapNodeId(id), + payload, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: schema.clone(), + guarantee: None, + }; + let consumed = PostAsapDag { + nodes: vec![ + node( + 0, + PostAsapOperatorPayload::Fallback { + expression: selector.clone(), + }, + ), + node( + 1, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Limit { + n: 1, + offset: 0, + partition_by: Default::default(), + }, + }, + ), + ], + edges: vec![PostAsapDagEdge { + producer: PostAsapNodeId(0), + consumer: PostAsapNodeId(1), + role: EdgeRole::Input, + intermediate_schema: schema.clone(), + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }], + root: PostAsapNodeId(1), + }; + let raw = promql_fallback::raw_series(&selector).unwrap().remove(0).1; + assert!(compile( + &consumed, + BTreeMap::from([( + promql_fallback::raw_series_input(0, 0), + InputContract::bounded(raw) + )]), + &[1], + ) + .is_err()); + // Implicit subquery resolution belongs to the deployment's evaluation interval. + assert!(compile_query("max_over_time(m[5m:])").is_err()); +} + +// irate/idelta use the last two samples (irate corrects a reset to the last +// value); changes/resets count value changes and decreases; quantile_over_time +// interpolates; all skip stale markers. +#[test] +fn instant_and_counting_range_functions() { + // COUNTER in (0s, 300s]: 10, 20, 5, 15. + assert!((one("irate(m[5m])", COUNTER, 300) - 10. / 60.).abs() < 1e-12); + assert_eq!(one("idelta(m[5m])", COUNTER, 300), 10.); + // At 200s the last pair 20 -> 5 is a reset: irate uses 5 as the increase. + assert!((one("irate(m[5m])", COUNTER, 200) - 5. / 60.).abs() < 1e-12); + assert_eq!(one("idelta(m[5m])", COUNTER, 200), -15.); + assert!(values("irate(m[1m])", COUNTER, 300).is_empty()); + assert_eq!(one("changes(m[5m])", COUNTER, 300), 3.); + assert_eq!(one("resets(m[5m])", COUNTER, 300), 1.); + assert_eq!(one("changes(m[2m])", COUNTER, 300), 0.); + // NaN to NaN is not a change; any other transition involving NaN is. + let flat = &[ + ("a", 10, 1.), + ("a", 20, 1.), + ("a", 30, 2.), + ("a", 40, f64::NAN), + ("a", 50, f64::NAN), + ("a", 55, 1.), + ]; + assert_eq!(one("changes(m[1m])", flat, 60), 3.); + let stale = f64::from_bits(0x7ff0_0000_0000_0002); + let ended = &[("a", 240, 15.), ("a", 250, stale)]; + assert_eq!(one("last_over_time(m[5m])", ended, 300), 15.); + assert!(values("m", ended, 300).is_empty()); + // Sorted 5, 10, 15, 20: rank 1.5 and 0.75; outside [0, 1] is +-Inf. + assert_eq!(one("quantile_over_time(0.5, m[5m])", COUNTER, 300), 12.5); + assert_eq!(one("quantile_over_time(0.25, m[5m])", COUNTER, 300), 8.75); + assert_eq!( + one("quantile_over_time(2, m[5m])", COUNTER, 300), + f64::INFINITY + ); + assert_eq!( + one("quantile_over_time(-1, m[5m])", COUNTER, 300), + f64::NEG_INFINITY + ); +} + +// `@ ` evaluates the selector or subquery at `t`, minus any offset, and the +// result keeps the query's evaluation time. +#[test] +fn at_modifier_fixes_the_evaluation_instant() { + let samples = &[("a", 60, 1.), ("a", 120, 2.), ("a", 180, 3.)]; + assert_eq!( + run("m @ 120", samples, 1000).unwrap(), + vec![("a".into(), 1_000_000, 2.)] + ); + assert!(values("m", samples, 1000).is_empty()); + assert_eq!(one("count_over_time(m[2m] @ 180)", samples, 1000), 2.); + assert_eq!(one("m @ 180 offset 1m", samples, 1000), 2.); + // The subquery grid is (60s, 180s]: steps 120 and 180 select 2 and 3. + assert_eq!(one("max_over_time(m[2m:1m] @ 180)", samples, 1000), 3.); + assert_eq!( + one("sum_over_time(m[2m:1m] @ 180 offset 1m)", samples, 1000), + 3. + ); + // An inner @ pins every step to the same instant. + assert_eq!(one("sum_over_time((m @ 60)[2m:1m])", samples, 180), 2.); + // start() and end() depend on the range query, which is the deployment's. + assert!(compile_query("m @ start()").is_err()); +} + +const A: &[Sample] = &[("job=x", 50, 10.), ("job=y", 50, 20.), ("job=w", 50, 0.)]; +const B: &[Sample] = &[("job=x", 50, 2.), ("job=z", 50, 5.), ("job=w", 50, 0.)]; + +// Vector-vector arithmetic matches series one-to-one on label sets without +// the metric name, and the result drops the metric name. +#[test] +fn vector_arithmetic_matches_label_sets() { + let metrics = &[("a", A), ("b", B)]; + let quotient = labeled("a / b", metrics, 60); + assert_eq!(quotient.len(), 2); + assert_eq!(quotient[0].0, "job=w"); + assert!(quotient[0].1.is_nan(), "0 / 0 is NaN"); + assert_eq!(quotient[1], ("job=x".into(), 5.)); + // Each selector reads its own raw rows, even a repeated metric. + assert_eq!( + labeled("(a - b) * a", metrics, 60), + vec![("job=w".into(), 0.), ("job=x".into(), 80.)] + ); + assert_eq!( + labeled("sum by (job) (a) - sum by (job) (b)", metrics, 60), + vec![("job=w".into(), 0.), ("job=x".into(), 8.)] + ); + // Rates of two counters over their own windows. + let up: &[Sample] = &[("job=x", 0, 0.), ("job=x", 60, 60.)]; + let down: &[Sample] = &[("job=x", 0, 0.), ("job=x", 60, 30.)]; + assert_eq!( + labeled("rate(a[2m]) / rate(b[2m])", &[("a", up), ("b", down)], 60), + vec![("job=x".into(), 2.)] + ); +} + +// on() keeps only the listed labels and ignoring() drops them; a duplicate +// match group is an error unless the left duplicates never match. +#[test] +fn on_and_ignoring_select_the_matching_labels() { + let a: &[Sample] = &[("job=x,inst=1", 50, 10.)]; + let b: &[Sample] = &[("job=x,inst=2", 50, 4.)]; + let metrics = &[("a", a), ("b", b)]; + assert!(labeled("a - b", metrics, 60).is_empty()); + assert_eq!( + labeled("a - on(job) b", metrics, 60), + vec![("job=x".into(), 6.)] + ); + assert_eq!( + labeled("a - ignoring(inst) b", metrics, 60), + vec![("job=x".into(), 6.)] + ); + let pair: &[Sample] = &[("job=x,inst=1", 50, 1.), ("job=x,inst=2", 50, 2.)]; + let other: &[Sample] = &[("job=y", 50, 1.)]; + assert!(evaluate("a + on(job) b", &[("a", a), ("b", pair)], 60).is_err()); + assert!(evaluate("a + on(job) b", &[("a", pair), ("b", b)], 60).is_err()); + assert!(labeled("a + on(job) b", &[("a", pair), ("b", other)], 60).is_empty()); + assert!(promql_rows::with_series_identity(&parse("a + on(job) group_left b")).is_err()); +} + +// without() groups by every label except the listed ones and the metric name. +#[test] +fn without_grouping_drops_labels_and_the_name() { + let a: &[Sample] = &[ + ("job=x,inst=1", 50, 1.), + ("job=x,inst=2", 50, 2.), + ("job=y,inst=1", 50, 4.), + ]; + let metrics = &[("a", a)]; + assert_eq!( + labeled("sum without (inst) (a)", metrics, 60), + vec![("job=x".into(), 3.), ("job=y".into(), 4.)] + ); + assert_eq!( + labeled("count without (inst) (a)", metrics, 60), + vec![("job=x".into(), 2.), ("job=y".into(), 1.)] + ); + assert_eq!( + labeled("max without (job, inst) (a)", metrics, 60), + vec![(String::new(), 4.)] + ); + assert!(labeled("sum without (inst) (a)", &[], 60).is_empty()); +} + +// An empty label value is an absent label, and an empty side yields an empty +// result before any duplicate check, as in Prometheus. +#[test] +fn empty_labels_and_empty_sides_match_prometheus() { + let a: &[Sample] = &[("job=x,env=", 50, 3.)]; + let b: &[Sample] = &[("job=x", 50, 1.)]; + assert_eq!( + labeled("a + b", &[("a", a), ("b", b)], 60), + vec![("job=x".into(), 4.)] + ); + let pair: &[Sample] = &[("job=x,inst=1", 50, 1.), ("job=x,inst=2", 50, 2.)]; + assert!(labeled("a + on(job) b", &[("b", pair)], 60).is_empty()); + assert!(labeled("b + on(job) a", &[("b", pair)], 60).is_empty()); + // A non-literal scalar operand has no identity realization yet. + assert!(promql_rows::with_series_identity(&parse("a + scalar(b)")).is_err()); +} diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs new file mode 100644 index 00000000..886b7d00 --- /dev/null +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -0,0 +1,492 @@ +//! Compile, persist and rebind dynamic-label computation without deployment lowering. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{promql_values::*, CompiledPhysicalDag, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::pre_asap::{AggIntent, ColumnRef, GroupKeys}; +use std::collections::BTreeMap; + +fn row(labels: &[(&str, &str)], value: f64) -> Vec { + vec![ + Value::Map( + labels + .iter() + .map(|(k, v)| (Value::Utf8((*k).into()), Value::Utf8((*v).into()))) + .collect::>() + .into(), + ), + Value::Float64(value), + ] +} +fn run(graph: CompiledPhysicalDag, rows: Vec>) -> Vec> { + run_inputs(graph, vec![Batch::try_new(vector_schema(), rows).unwrap()]).unwrap() +} +fn run_inputs( + graph: CompiledPhysicalDag, + batches: Vec, +) -> Result>, asap_physical_operators::Error> { + let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + .unwrap(); + let sources = batches + .into_iter() + .enumerate() + .map(|(id, batch)| { + ( + id as u64, + Box::new(Operator::source(batch.schema().clone(), vec![batch]).unwrap()) + as Source<'_>, + ) + }) + .collect::>(); + let bound = graph.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = bound.execute(graph.roots(), context).unwrap().remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok(rows) + }) +} +fn equal_rows(actual: Vec>, expected: Vec>) { + let mut actual = actual + .into_iter() + .map(|r| serde_json::to_string(&r).unwrap()) + .collect::>(); + let mut expected = expected + .into_iter() + .map(|r| serde_json::to_string(&r).unwrap()) + .collect::>(); + actual.sort(); + expected.sort(); + assert_eq!(actual, expected); +} + +#[test] +fn grouping_preserves_unenumerated_labels_and_empty_label_semantics() { + let rows = vec![ + row(&[("__name__", "m"), ("instance", "a"), ("job", "api")], 1.), + row(&[("__name__", "m"), ("instance", "b"), ("job", "api")], 2.), + row(&[("instance", "c"), ("job", "")], 4.), + row(&[("instance", "d")], 8.), + ]; + equal_rows( + run( + compile_aggregate( + &AggIntent::Sum { col: None }, + &GroupKeys::without(vec![ColumnRef::Named("instance".into())]), + ) + .unwrap(), + rows.clone(), + ), + vec![row(&[("job", "api")], 3.), row(&[], 12.)], + ); + equal_rows( + run( + compile_aggregate( + &AggIntent::Count { + accuracy: planner_types::types::AccuracyTarget::Exact, + }, + &GroupKeys::by(vec![ColumnRef::Named("job".into())]), + ) + .unwrap(), + rows, + ), + vec![row(&[("job", "api")], 2.), row(&[], 2.)], + ); +} + +#[test] +fn ranking_and_grouped_limit_preserve_full_selected_series() { + let grouping = GroupKeys::by(vec![ColumnRef::Named("job".into())]); + let rows = vec![ + row(&[("instance", "a"), ("job", "api")], 1.), + row(&[("instance", "b"), ("job", "api")], 3.), + row(&[("instance", "c"), ("job", "worker")], 2.), + ]; + let sorted = run(compile_sort(true, &grouping).unwrap(), rows); + let selected = run(compile_limit(1, 0, &grouping).unwrap(), sorted); + equal_rows( + selected, + vec![ + row(&[("instance", "b"), ("job", "api")], 3.), + row(&[("instance", "c"), ("job", "worker")], 2.), + ], + ); + assert!(run(compile_limit(0, 0, &grouping).unwrap(), vec![row(&[], 1.)]).is_empty()); +} + +#[test] +fn empty_vector_aggregation_stays_empty() { + assert!(run( + compile_aggregate(&AggIntent::Sum { col: None }, &GroupKeys::default()).unwrap(), + vec![] + ) + .is_empty()); + let scalar = run(compile_vector_to_scalar().unwrap(), vec![]); + assert!(matches!(scalar[0][0],Value::Float64(v) if v.is_nan())); +} + +// One persisted temporal graph accepts different request windows and detects resets. +#[test] +fn temporal_graph_uses_bound_window_without_recompilation() { + let graph = compile_temporal(&AggIntent::Rate, false).unwrap(); + for start in [0, 60_000] { + let labels = row(&[("__name__", "counter"), ("job", "api")], 0.)[0].clone(); + let samples = [(0, 5.), (30_000, 1.), (60_000, 7.)]; + let rows = samples + .into_iter() + .map(|(time, value)| { + vec![ + labels.clone(), + Value::Timestamp(start + time), + Value::Float64(value), + Value::Timestamp(start), + Value::Timestamp(start + 60_000), + ] + }) + .collect(); + let output = run_inputs( + graph.clone(), + vec![Batch::try_new(matrix_schema(), rows).unwrap()], + ) + .unwrap(); + equal_rows(output, vec![row(&[("job", "api")], 7. / 60.)]); + } + let labels = row(&[("job", "api")], 0.)[0].clone(); + let rows = vec![ + vec![ + labels.clone(), + Value::Timestamp(0), + Value::Float64(1.), + Value::Timestamp(0), + Value::Timestamp(1000), + ], + vec![ + labels, + Value::Timestamp(1000), + Value::Float64(2.), + Value::Timestamp(0), + Value::Timestamp(2000), + ], + ]; + assert!(run_inputs(graph, vec![Batch::try_new(matrix_schema(), rows).unwrap()]).is_err()); +} + +// The quantile is an ordinary scalar input, and bucket labels are native computation. +#[test] +fn histogram_quantile_keeps_each_label_group() { + let graph = compile_histogram_quantile().unwrap(); + let buckets = vec![ + row(&[("job", "api"), ("le", "1")], 2.), + row(&[("job", "api"), ("le", "2")], 4.), + row(&[("job", "api"), ("le", "+Inf")], 4.), + ]; + let output = run_inputs( + graph, + vec![ + Batch::try_new(scalar_schema(), vec![vec![Value::Float64(0.75)]]).unwrap(), + Batch::try_new(vector_schema(), buckets).unwrap(), + ], + ) + .unwrap(); + equal_rows(output, vec![row(&[("job", "api")], 1.5)]); +} + +// Linking an ensemble preserves its shared producer and every selected operator. +#[test] +fn composed_ensemble_shares_a_producer_across_roots() { + use asap_physical_operators::{ + physical_planner::InputContract, + plan::{PhysicalOperator, PlanProperties}, + runtime::{Input, OutputStream}, + values::Schema, + }; + use planner_types::{ + post_asap::BinaryOperator, + pre_asap::{ArithmeticOpKind, BinaryOpKind}, + }; + struct Counted { + source: Operator, + starts: std::rc::Rc>, + } + impl PhysicalOperator for Counted { + fn name(&self) -> &str { + "CountedInput" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.source.schema() + } + fn output_bytes(&self, batch: &Batch) -> usize { + batch.bytes() + } + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + self.source.properties(inputs) + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, asap_physical_operators::Error> { + self.starts.set(self.starts.get() + 1); + self.source.start(inputs, context) + } + } + let aggregate = compile_aggregate( + &AggIntent::Sum { col: None }, + &GroupKeys::by(vec![ColumnRef::Named("job".into())]), + ) + .unwrap(); + let binary = compile_binary( + &BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_finite_division: false, + checked_relative_division: false, + }, + false, + false, + false, + ) + .unwrap(); + let graph = CompiledPhysicalDag::compose( + BTreeMap::from([(0, InputContract::bounded(vector_schema()))]), + BTreeMap::from([ + (10, (vec![0], aggregate)), + (20, (vec![10, 10], binary)), + (30, (vec![10], compile_negate(false).unwrap())), + ]), + vec![20, 30], + ) + .unwrap(); + let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + .unwrap(); + assert_eq!(graph.input_contracts().count(), 1); + let starts = std::rc::Rc::new(std::cell::Cell::new(0)); + for _ in 0..2 { + let input = Batch::try_new(vector_schema(), vec![row(&[("job", "api")], 3.)]).unwrap(); + let source = Counted { + source: Operator::source(vector_schema(), vec![input]).unwrap(), + starts: starts.clone(), + }; + let bound = graph + .instantiate(BTreeMap::from([(0, Box::new(source) as Source<'_>)])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_buffered_batches: 1, + ..Limits::default() + }, + ) + .unwrap(); + let results = + block_on(futures::future::join_all( + bound + .execute(graph.roots(), context) + .unwrap() + .into_iter() + .map(|mut stream| async move { + stream.next().await.unwrap().unwrap().rows().to_vec() + }), + )); + equal_rows(results[0].clone(), vec![row(&[("job", "api")], 6.)]); + equal_rows(results[1].clone(), vec![row(&[("job", "api")], -3.)]); + } + assert_eq!(starts.get(), 2); +} + +#[test] +fn compiled_constant_needs_no_deployment_source() { + let graph = compile_scalar(3.).unwrap(); + assert_eq!(graph.input_contracts().count(), 0); + let result = run_inputs(graph, vec![]).unwrap(); + assert!(matches!(result[0][0], Value::Float64(3.))); +} + +// Scalar broadcasting cannot silently create duplicate result identities when +// arithmetic or bool comparisons remove the metric name. +#[test] +fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { + use planner_types::{ + post_asap::BinaryOperator, + pre_asap::{ArithmeticOpKind, BinaryOpKind, CompareOpKind}, + }; + for left_scalar in [false, true] { + for names in [["a", "a"], ["a", "b"]] { + for (kind, return_bool) in [ + (BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), false), + (BinaryOpKind::Compare(CompareOpKind::Gt), true), + ] { + let graph = compile_binary( + &BinaryOperator { + kind, + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + left_scalar, + !left_scalar, + ) + .unwrap(); + let vector = Batch::try_new( + vector_schema(), + vec![ + row(&[("__name__", names[0]), ("job", "api")], 2.), + row(&[("__name__", names[1]), ("job", "api")], 3.), + ], + ) + .unwrap(); + let scalar = + Batch::try_new(scalar_schema(), vec![vec![Value::Float64(1.)]]).unwrap(); + let result = run_inputs( + graph, + if left_scalar { + vec![scalar, vector] + } else { + vec![vector, scalar] + }, + ); + assert!(result.is_err(), "duplicate output label sets were accepted"); + } + } + } + let graph = compile_binary( + &BinaryOperator { + kind: BinaryOpKind::Compare(CompareOpKind::Gt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + false, + false, + true, + ) + .unwrap(); + let rows = vec![ + row(&[("__name__", "a"), ("job", "api")], 2.), + row(&[("__name__", "b"), ("job", "api")], 3.), + ]; + equal_rows( + run_inputs( + graph, + vec![ + Batch::try_new(vector_schema(), rows.clone()).unwrap(), + Batch::try_new(scalar_schema(), vec![vec![Value::Float64(1.)]]).unwrap(), + ], + ) + .unwrap(), + rows, + ); +} + +// Persisted exact readout graphs, rather than the storage adapter, merge panes, +// finalize each population, and preserve the requested metric-name semantics. +#[test] +fn exact_state_readouts_recover_and_finalize_panes() { + use asap_physical_operators::factory::create_planner_accumulator; + use planner_types::post_asap::*; + use std::sync::Arc; + for (kind, params, expected) in [ + (ExactKind::Sum, ExactParams::Sum, 12.), + (ExactKind::Count, ExactParams::Count, 4.), + (ExactKind::Min, ExactParams::Min, 1.), + (ExactKind::Max, ExactParams::Max, 5.), + ] { + let family = SummaryFamilyType::ExactAggregate(kind, params); + for preserve in [false, true] { + let rows = [[1., 2.], [4., 5.]] + .into_iter() + .map(|samples| { + let mut state = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + for sample in samples { + state.update_single(sample, 0); + } + let labels = row(&[("__name__", "m"), ("instance", "a")], 0.).remove(0); + vec![ + labels, + Value::Summary { + family: family.clone(), + state: Arc::from(state.into_accumulator()), + }, + ] + }) + .collect(); + let output = run_inputs( + compile_exact_readout(family.clone(), 60_000, preserve).unwrap(), + vec![Batch::try_new(exact_state_schema(family.clone()).unwrap(), rows).unwrap()], + ) + .unwrap(); + let labels = if preserve { + vec![("__name__", "m"), ("instance", "a")] + } else { + vec![("instance", "a")] + }; + equal_rows(output, vec![row(&labels, expected)]); + } + } +} + +#[test] +fn recovered_exact_counter_uses_window_and_omits_insufficient_samples() { + use asap_physical_operators::factory::create_planner_accumulator; + use planner_types::post_asap::*; + use std::sync::Arc; + for (kind, params, expected) in [ + (ExactKind::Rate, ExactParams::Rate, 1.), + (ExactKind::Increase, ExactParams::Increase, 60.), + ] { + let family = SummaryFamilyType::ExactAggregate(kind, params); + let rows = [1, 2] + .into_iter() + .map(|count| { + let mut state = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + state.update_single(100., -50_000); + if count == 2 { + state.update_single(140., -10_000); + } + vec![ + row(&[("instance", if count == 1 { "one" } else { "two" })], 0.).remove(0), + Value::Summary { + family: family.clone(), + state: Arc::from(state.into_accumulator()), + }, + ] + }) + .collect(); + let output = run_inputs( + compile_exact_readout(family.clone(), 60_000, false).unwrap(), + vec![Batch::try_new(exact_state_schema(family).unwrap(), rows).unwrap()], + ) + .unwrap(); + equal_rows(output, vec![row(&[("instance", "two")], expected)]); + } +} diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs new file mode 100644 index 00000000..78e4c8c0 --- /dev/null +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -0,0 +1,386 @@ +//! Scan acceptance uses the public connector contract and Planner physical DAGs. +use asap_physical_operators::dag::{ + planner::bind_with_data_sources, + scan::{DataSources, MemorySource, RawSource}, + values::{Batch, Schema, Value}, + Error, Limits, OutputStream, RunContext, Scope, +}; +use futures::{executor::block_on, stream, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{Column, DataType, GroupKeys, Predicate, QueryExpr, Source}, +}; +use std::{ + collections::BTreeMap, + rc::Rc, + sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, + }, +}; + +fn fixture() -> (QueryExpr, Schema, Vec) { + let schema = + planner_types::pre_asap::Schema::new(vec![Column::new("value", DataType::Int64, true)]); + let output = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: true, + }], + time_index: None, + }); + let scan = QueryExpr::Scan { + source: Source::Table { + table_ref: "numbers".into(), + }, + predicates: vec![Predicate(Rc::new(QueryExpr::IsNotNull(Rc::new( + QueryExpr::Column(0), + ))))], + schema, + }; + let batches = vec![ + Batch::try_new( + output.clone(), + vec![vec![Value::Int64(3)], vec![Value::Null]], + ) + .unwrap(), + Batch::try_new( + output.clone(), + vec![vec![Value::Int64(9)], vec![Value::Int64(2)]], + ) + .unwrap(), + ]; + (scan, output, batches) +} +fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsapDag { + let node = |id, payload| PostAsapDagNode { + id: PostAsapNodeId(id), + payload, + output_state: state, + output_schema: (**schema).clone(), + guarantee: None, + }; + let edge = |producer, consumer| PostAsapDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role: EdgeRole::Input, + intermediate_schema: (**schema).clone(), + data_state: state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }; + PostAsapDag { + nodes: vec![ + node(0, PostAsapOperatorPayload::Fallback { expression: scan }), + node( + 1, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Sort { + keys: vec![planner_types::pre_asap::SortKey { + expr: QueryExpr::Column(0), + ascending: false, + nulls_first: false, + }], + partition_by: GroupKeys::by(vec![]), + }, + }, + ), + node( + 2, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Limit { + n: 2, + offset: 0, + partition_by: GroupKeys::by(vec![]), + }, + }, + ), + ], + edges: vec![edge(0, 1), edge(1, 2)], + root: PostAsapNodeId(2), + } +} +fn registry(source: Arc) -> DataSources { + let mut r = DataSources::default(); + r.register( + Source::Table { + table_ref: "numbers".into(), + }, + source, + ) + .unwrap(); + r +} +fn context() -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits::default(), + ) + .unwrap() +} + +// Raw-only execution filters nulls and ranks across batches at either phase. +#[test] +fn raw_scan_to_sort_limit_at_both_phases() { + let (scan, schema, batches) = fixture(); + let sources = registry(Arc::new( + MemorySource::new(schema.clone(), batches).unwrap(), + )); + for state in [ + ExecutionDataState::QUERY_ROWS, + ExecutionDataState::INGESTION_ROWS, + ] { + let dag = plan(scan.clone(), &schema, state); + let bound = bind_with_data_sources(&dag, BTreeMap::new(), &[2], &sources).unwrap(); + let ctx = if state == ExecutionDataState::QUERY_ROWS { + context() + } else { + RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1, + revision: 1, + }, + Limits::default(), + ) + .unwrap() + }; + let rows = block_on(async { + let mut output = bound.execute(&[2], ctx.clone()).unwrap().remove(0); + let mut rows = vec![]; + while let Some(batch) = output.next().await { + rows.extend(batch.unwrap().rows().iter().cloned()); + } + rows + }); + assert!( + matches!(rows.as_slice(), [a,b] if matches!(a.as_slice(), [Value::Int64(9)]) && matches!(b.as_slice(), [Value::Int64(3)])) + ); + assert_eq!(ctx.retained_bytes(), 0); + } +} +struct CountingSource { + schema: Schema, + opened: Arc, + fail: bool, +} +impl RawSource for CountingSource { + fn boundedness(&self) -> asap_physical_operators::plan::Boundedness { + asap_physical_operators::plan::Boundedness::Bounded + } + fn schema(&self) -> Schema { + self.schema.clone() + } + fn scan(&self, _: RunContext) -> Result, Error> { + self.opened.fetch_add(1, Ordering::SeqCst); + if self.fail { + return Err(Error::Operator("reader failed".into())); + } + Ok(stream::iter(vec![Batch::try_new( + self.schema.clone(), + vec![vec![Value::Int64(7)]], + )]) + .boxed_local()) + } +} +// Binding and cancellation do not perform I/O; fan-out opens one cursor per run. +#[test] +fn lazy_open_shared_producer_and_cancellation() { + let (scan, schema, _) = fixture(); + let opened = Arc::new(AtomicUsize::new(0)); + let sources = registry(Arc::new(CountingSource { + schema: schema.clone(), + opened: opened.clone(), + fail: false, + })); + let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let bound = bind_with_data_sources(&plan, BTreeMap::new(), &[0, 2], &sources).unwrap(); + let ctx = context(); + let streams = bound.execute(&[0, 2], ctx.clone()).unwrap(); + assert_eq!(opened.load(Ordering::SeqCst), 0); + ctx.cancel(); + drop(streams); + assert_eq!(opened.load(Ordering::SeqCst), 0); + for _ in 0..2 { + block_on(async { + let streams = bound.execute(&[0, 2], context()).unwrap(); + let all = + futures::future::join_all(streams.into_iter().map(|s| s.collect::>())).await; + assert!(all.iter().flatten().all(Result::is_ok)); + }); + } + assert_eq!(opened.load(Ordering::SeqCst), 2); +} +// Unavailable sources and unsupported predicates fail before opening any cursor. +#[test] +fn binding_errors_and_reader_errors_are_not_empty_results() { + let (mut scan, schema, _) = fixture(); + assert!(DataSources::default().bind(&scan).is_err()); + let opened = Arc::new(AtomicUsize::new(0)); + let sources = registry(Arc::new(CountingSource { + schema: schema.clone(), + opened: opened.clone(), + fail: true, + })); + if let QueryExpr::Scan { predicates, .. } = &mut scan { + predicates.push(Predicate(Rc::new(QueryExpr::Column(0)))); + } + assert!(sources.bind(&scan).is_err()); + assert_eq!(opened.load(Ordering::SeqCst), 0); + let (scan, _, _) = fixture(); + let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let bound = bind_with_data_sources(&plan, BTreeMap::new(), &[2], &sources).unwrap(); + block_on(async { + let mut stream = bound.execute(&[2], context()).unwrap().remove(0); + assert!(stream.next().await.unwrap().is_err()); + }); +} + +// Schema drift cannot enter the DAG, and connector batches obey execution limits. +#[test] +fn schema_drift_and_memory_limits_fail_the_scan() { + struct Drift { + expected: Schema, + batch: Batch, + } + impl RawSource for Drift { + fn schema(&self) -> Schema { + self.expected.clone() + } + fn scan(&self, _: RunContext) -> Result, Error> { + Ok(stream::once(async { Ok(self.batch.clone()) }).boxed_local()) + } + } + let (scan, schema, batches) = fixture(); + let mut different = (*schema).clone(); + different.fields[0].name = "wrong".into(); + let bad = Batch::try_new(Arc::new(different), vec![vec![Value::Int64(1)]]).unwrap(); + let sources = registry(Arc::new(Drift { + expected: schema.clone(), + batch: bad, + })); + let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + block_on(async { + let mut s = graph.execute(&[0], context()).unwrap().remove(0); + assert!(s.next().await.unwrap().is_err()); + }); + let sources = registry(Arc::new(MemorySource::new(schema, batches).unwrap())); + let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + let ctx = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits { + max_bytes: 1, + max_buffered_batches: 1, + }, + ) + .unwrap(); + block_on(async { + let mut s = graph.execute(&[0], ctx.clone()).unwrap().remove(0); + assert!(s.next().await.unwrap().is_err()); + }); + assert_eq!(ctx.retained_bytes(), 0); +} + +// An empty table is a valid empty scan; nullable comparisons retain only TRUE. +#[test] +fn empty_sources_and_three_valued_predicates() { + use planner_types::pre_asap::{CompareOpKind, ScalarValue}; + let (mut scan, schema, batches) = fixture(); + if let QueryExpr::Scan { + predicates, source, .. + } = &mut scan + { + *source = Source::TimeSeries { + metric: "samples".into(), + }; + *predicates = vec![Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Gt, + right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(2))), + }))]; + } + for (batches, expected) in [(vec![], 0), (batches, 2)] { + let mut sources = DataSources::default(); + sources + .register( + Source::TimeSeries { + metric: "samples".into(), + }, + Arc::new(MemorySource::new(schema.clone(), batches).unwrap()), + ) + .unwrap(); + let plan = plan(scan.clone(), &schema, ExecutionDataState::QUERY_ROWS); + let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + block_on(async { + let mut s = graph.execute(&[0], context()).unwrap().remove(0); + let mut count = 0; + while let Some(b) = s.next().await { + count += b.unwrap().rows().len(); + } + assert_eq!(count, expected); + }); + } +} + +// A physical candidate can be compiled once without readers and rebound per run. +#[test] +fn compile_without_readers_and_rebind_inputs() { + use asap_physical_operators::{ + operators::Operator, + physical_planner::{compile, InputContract, Source}, + }; + let (scan, schema, batches) = fixture(); + let dag = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let compiled = compile( + &dag, + BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), + &[2], + ) + .unwrap(); + assert_eq!(compiled.input_contracts().count(), 1); + for _ in 0..2 { + let sources = BTreeMap::from([( + 0, + Box::new(Operator::source(schema.clone(), batches.clone()).unwrap()) as Source<'_>, + )]); + let graph = compiled.instantiate(sources).unwrap(); + let mut outputs = graph.execute(compiled.roots(), context()).unwrap(); + let result = block_on(outputs.remove(0).collect::>()); + assert!(result.iter().all(Result::is_ok)); + assert_eq!( + result + .iter() + .map(|b| b.as_ref().unwrap().rows().len()) + .sum::(), + 2 + ); + } + assert!(compiled.instantiate(BTreeMap::new()).is_err()); +} + +// Input boundedness must be proved during compilation, before readers exist. +#[test] +fn compilation_rejects_unknown_boundedness_for_sort() { + use asap_physical_operators::{ + physical_planner::{compile, InputContract}, + plan::{Boundedness, Emission, PlanProperties}, + }; + let (scan, schema, _) = fixture(); + let dag = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let input = InputContract { + schema, + properties: PlanProperties { + boundedness: Boundedness::Unknown, + emission: Emission::Unknown, + }, + }; + assert!(compile(&dag, BTreeMap::from([(0, input)]), &[2]).is_err()); +} diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs new file mode 100644 index 00000000..a61a596c --- /dev/null +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -0,0 +1,161 @@ +//! Opaque state travels through a retained physical projection without scalar decoding. +use asap_physical_operators::{ + expressions::Expression, + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{compile, CompiledPhysicalDag, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{ColumnRef, DataType, ProjectItem, QueryExpr}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +// A Post-ASAP projection may reorder/rename summary columns; recovery must retain +// the family and pass through the same immutable state, without decoding the payload. +#[test] +fn post_asap_summary_projection_survives_recovery() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let schema = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }, + SummaryField { + name: "service".into(), + dtype: SummaryFamilyType::Plain(DataType::Utf8), + nullable: false, + }, + ], + time_index: None, + }); + let output = SummarySchema { + fields: vec![ + schema.fields[1].clone(), + SummaryField { + name: "renamed".into(), + ..schema.fields[0].clone() + }, + ], + time_index: None, + }; + let dag = PostAsapDag { + nodes: vec![ + PostAsapDagNode { + id: PostAsapNodeId(0), + payload: PostAsapOperatorPayload::SummaryMerge, + output_schema: (*schema).clone(), + output_state: ExecutionDataState::INGESTION_SUMMARY, + guarantee: None, + }, + PostAsapDagNode { + id: PostAsapNodeId(1), + payload: PostAsapOperatorPayload::Value { + operation: ValueOperation::Project { + cols: vec![1, 0] + .into_iter() + .map(|index| ProjectItem { + alias: None, + expr: QueryExpr::Column(index), + }) + .collect(), + qualifier: None, + }, + }, + output_schema: output.clone(), + output_state: ExecutionDataState::INGESTION_SUMMARY, + guarantee: None, + }, + ], + edges: vec![PostAsapDagEdge { + producer: PostAsapNodeId(0), + consumer: PostAsapNodeId(1), + role: EdgeRole::Input, + intermediate_schema: (*schema).clone(), + data_state: ExecutionDataState::INGESTION_SUMMARY, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }], + root: PostAsapNodeId(1), + }; + let program = compile( + &dag, + BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), + &[1], + ) + .unwrap(); + let encoded = serde_json::to_vec(&program).unwrap(); + let program = serde_json::from_slice::(&encoded).unwrap(); + let mut forged: serde_json::Value = serde_json::from_slice(&encoded).unwrap(); + forged["nodes"]["1"]["Operator"]["operator"]["output"]["fields"][1]["dtype"] = + serde_json::json!({"Plain": "float64"}); + assert!( + serde_json::from_slice::(&serde_json::to_vec(&forged).unwrap()) + .is_err() + ); + assert!(Operator::project( + schema.clone(), + vec![("invalid".into(), Expression::Column(2))] + ) + .is_err()); + assert!(Operator::project( + schema.clone(), + vec![( + "invalid".into(), + Expression::Negate(Box::new(Expression::Column(0))) + )] + ) + .is_err()); + assert_eq!(*program.output_contract(1).unwrap().schema, output); + let mut accumulator = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + accumulator.update_single(7., 1); + let state = Arc::from(accumulator.into_accumulator()); + let batch = Batch::try_new( + schema.clone(), + vec![vec![ + Value::Summary { + family, + state: Arc::clone(&state), + }, + Value::Utf8("api".into()), + ]], + ) + .unwrap(); + let graph = program + .instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, + )])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut output = graph.execute(&[1], context).unwrap().remove(0); + let batch = output.next().await.unwrap().unwrap(); + assert!(matches!(&batch.rows()[0][0], Value::Utf8(label) if label.as_ref() == "api")); + let Value::Summary { + state: projected, .. + } = &batch.rows()[0][1] + else { + panic!("missing summary") + }; + assert!(Arc::ptr_eq(&state, projected)); + assert!(output.next().await.is_none()); + }); +} diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs new file mode 100644 index 00000000..ef086b1b --- /dev/null +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -0,0 +1,1052 @@ +//! Planner output binds directly to the shared runtime at a declared rate-value frontier. +use asap_aware_mapping::{ + accuracy::{ + AccuracyEvidenceProvider, DefaultAccuracyModel, EqualSplitAllocator, PropagationStats, + }, + cost_model::DefaultCostModel, + Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG, +}; +use asap_physical_operators::dag::{ + operators::Operator, + planner::{compile, InputContract, Source}, + values::{Batch, Value}, + Limits, RunContext, Scope, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{DataType, QueryExpr}, + types::AccuracyTarget, +}; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; +struct Evidence; +impl AccuracyEvidenceProvider for Evidence { + fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + Some(1000) + } + fn propagation_stats( + &self, + op: &CompositionOperator, + _: &SummaryFamilyType, + _: Option<&SketchQuery>, + ) -> PropagationStats { + if matches!(op, CompositionOperator::TopKSelection) { + PropagationStats { + topk_selected_lower_bound: Some(101.), + topk_excluded_upper_bound: Some(100.), + topk_interval_failure_probability: Some(0.001), + ..Default::default() + } + } else { + Default::default() + } + } +} +// The evidence here exercises binding; it is not inferred from the sample data. +#[test] +fn planner_weighted_topk_binds_at_either_deployment_phase() { + assert_weighted_binding(&Evidence, SketchAlgorithm::CmsWithHeap); + assert_weighted_binding(&Evidence, SketchAlgorithm::CountSketchWithHeap); +} + +// Binding validates representation, while deployment owns evidence acceptance. +#[test] +fn physical_binding_does_not_impose_an_accuracy_acceptance_policy() { + assert_weighted_binding( + &asap_aware_mapping::accuracy::NoAccuracyEvidence, + SketchAlgorithm::CmsWithHeap, + ); + assert_weighted_binding( + &asap_aware_mapping::accuracy::NoAccuracyEvidence, + SketchAlgorithm::CountSketchWithHeap, + ); +} + +fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: SketchAlgorithm) { + let root = Rc::new( + lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.1), + ) + .unwrap(), + ); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + evidence, + ); + let plan = strategy + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .find_map(|candidate| match candidate.replacement { + Replacement::Summary(node) + if candidate.rationale.contains(&format!("{algorithm:?}")) => + { + Some(node) + } + _ => None, + }) + .unwrap(); + let dag = compile_post_asap_dag(&plan).unwrap(); + let build=dag.nodes.iter().find(|node|matches!(&node.payload,PostAsapOperatorPayload::SummaryAgg{family:SummaryFamilyType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); + let rate_id = dag + .edges + .iter() + .find(|edge| edge.consumer == build.id) + .unwrap() + .producer; + let rates = Arc::new( + dag.nodes + .iter() + .find(|node| node.id == rate_id) + .unwrap() + .output_schema + .clone(), + ); + let rows = [ + ("auth", "api", 0.125), + ("auth", "api", 0.25), + ("checkout", "api", 0.3125), + ("search", "api", 0.0625), + ("ingest", "batch", 100.), + ("export", "batch", 80.), + ("cleanup", "batch", 20.), + ] + .into_iter() + .map(|(service, job, value)| { + rates + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "job" => Value::Utf8(job.into()), + "value" => Value::Float64(value), + _ => match field.dtype { + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), + _ => panic!("unexpected rate column {field:?}"), + }, + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(rates.clone(), rows).unwrap(); + for (phase, scope) in [ + ( + ExecutionTiming::IngestionTime, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 60_000, + revision: 1, + }, + ), + ( + ExecutionTiming::QueryTime, + Scope::Query { + evaluation_time_ms: 60_000, + revision: 1, + }, + ), + ] { + let placed = dag + .with_execution_phases(&dag.nodes.iter().map(|node| (node.id, phase)).collect()) + .unwrap(); + let source = Box::new(Operator::source(rates.clone(), vec![batch.clone()]).unwrap()) + as Source<'static>; + let compiled = compile( + &placed, + BTreeMap::from([(rate_id.0 as u64, InputContract::bounded(rates.clone()))]), + &[dag.root.0 as u64], + ) + .unwrap(); + let graph = compiled + .instantiate(BTreeMap::from([(rate_id.0 as u64, source)])) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let output = block_on(async { + let mut output = Vec::new(); + let mut stream = graph + .execute(&[dag.root.0 as u64], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + output.extend(batch.unwrap().rows().iter().cloned()); + } + output + }); + assert_eq!(output.len(), 4); + let mut scores = output + .iter() + .map(|row| { + row.iter() + .find_map(|v| { + if let Value::Float64(v) = v { + Some(*v) + } else { + None + } + }) + .unwrap() + }) + .collect::>(); + scores.sort_by(f64::total_cmp); + assert_eq!(scores, vec![0.3125, 0.375, 80., 100.]); + } +} + +use asap_frontend_promql::lower_promql_workload; +use planner_types::workload::{ + AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence as WorkloadEvidence, + PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, + TimeSelection, +}; +pub fn lower_promql( + query: &str, + accuracy: AccuracyTarget, +) -> Result { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: WorkloadEvidence { + value: Some(DurationMs(1_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let mut lowered = lower_promql_workload(&workload, 0)?; + Ok(lowered.remove(0)) +} + +// The old untyped heap updater must not silently round a Planner rate update. +#[test] +fn rate_updates_cannot_enter_integer_heap_factory() { + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 272, + depth: 5, + heap_size: 100, + }, + ), + Default::default(), + ); + let input = SummaryUpdate { + item: Some(SummaryInputExpr::Column( + planner_types::pre_asap::ColumnRef::Named("service".into()), + )), + weight: SummaryInputExpr::Column(planner_types::pre_asap::ColumnRef::SampleValue), + weight_domain: WeightDomain::NonNegative { + proof: NonNegativeWeightProof::ResetAwareCounterDerivative, + }, + }; + assert!( + asap_physical_operators::factory::create_planner_accumulator( + &family, + &input, + &Default::default() + ) + .is_err() + ); +} + +/// A catalog-resolved per-series rate can feed a heap sketch directly, without +/// requiring an otherwise unnecessary grouped Sum between Rate and TopK. +#[test] +fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { + check_direct_rate_topk(false); +} + +// Unreferenced labels still distinguish series throughout Rate and heap readout. +#[test] +fn direct_rate_topk_preserves_dynamic_unreferenced_labels() { + check_direct_rate_topk(true); +} + +fn check_direct_rate_topk(dynamic: bool) { + use asap_physical_operators::physical_planner::promql_rows::{ + decode_series_identity, series_row, with_series_identity, SERIES_IDENTITY_COLUMN, + }; + let mut logical = + lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(); + fn resolve_catalog(node: &mut QueryExpr) { + match node { + QueryExpr::Aggregate { child, .. } | QueryExpr::TimeRange { child, .. } => { + resolve_catalog(Rc::make_mut(child)) + } + QueryExpr::Scan { schema, .. } => { + schema.closed = true; + schema + .columns + .push(planner_types::pre_asap::schema::Column::new( + "service", + DataType::Utf8, + false, + )); + } + _ => panic!("unexpected input shape: {node:?}"), + } + } + if dynamic { + logical = with_series_identity(&logical).unwrap(); + } else { + resolve_catalog(&mut logical); + } + let root = Rc::new(logical); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let candidates = strategy.replacements(&TargetSubDAG::new(&root)); + for algorithm in [ + SketchAlgorithm::CmsWithHeap, + SketchAlgorithm::CountSketchWithHeap, + ] { + let candidate = candidates + .iter() + .find_map(|candidate| match &candidate.replacement { + Replacement::Summary(node) + if candidate.rationale.contains(&format!("{algorithm:?}")) => + { + Some(node) + } + _ => None, + }) + .unwrap_or_else(|| panic!("missing {algorithm:?} over direct Rate")); + if dynamic { + let (source, ranked) = + asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking( + candidate, + ) + .unwrap(); + assert!(matches!( + source.expr, + SummaryExpr::ValueOperation { + operation: ValueOperation::FinalizeExactAccumulator, + .. + } + )); + assert_eq!(ranked.input_contracts().count(), 1); + let encoded = String::from_utf8(serde_json::to_vec(&ranked).unwrap()).unwrap(); + assert!(encoded.contains("KeyedSummaryBuild")); + assert!(encoded.contains("KeyedReadout")); + assert!( + !encoded.contains("\"Rate\""), + "Rate must be supplied by its exact stored-state readout" + ); + } + let dag = compile_post_asap_dag(candidate).unwrap(); + assert!(dag.nodes.iter().any(|node| matches!(&node.payload, + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); + let build = dag.nodes.iter().find(|node| matches!(&node.payload, + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); + let input_id = dag + .edges + .iter() + .find(|edge| edge.consumer == build.id) + .unwrap() + .producer; + let schema = Arc::new( + dag.nodes + .iter() + .find(|node| node.id == input_id) + .unwrap() + .output_schema + .clone(), + ); + let raw = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::TimeRange { .. } + } + ) + }) + .unwrap_or_else(|| panic!("no raw counter source: {dag:?}")); + let raw_schema = Arc::new(raw.output_schema.clone()); + let raw_compiled = compile( + &dag, + BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(raw_schema.clone()), + )]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + let bytes = serde_json::to_vec(&raw_compiled).unwrap(); + let raw_compiled = serde_json::from_slice::< + asap_physical_operators::physical_planner::CompiledPhysicalDag, + >(&bytes) + .unwrap(); + // Each evaluation receives a complete raw window. A reset, a stopped + // series and an expired leader must not retain last run's heap weights. + for (end, series, expected) in [ + ( + 60_000, + vec![ + ("auth", vec![10., 30., 50.]), + ("checkout", vec![10., 50., 90.]), + ("search", vec![10., 70., 130.]), + ], + vec![11. / 6., 8. / 3.], + ), + ( + 120_000, + vec![ + ("auth", vec![100., 10., 50.]), + ("checkout", vec![100., 100., 100.]), + ], + vec![0., 1.25], + ), + ] { + let mut raw_rows = Vec::new(); + for (service, samples) in series { + for (offset, value) in [10_000, 30_000, 50_000].into_iter().zip(samples) { + if dynamic { + raw_rows.push( + series_row( + &raw_schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("service".into(), service.into()), + ("unreferenced".into(), format!("{service}-extra")), + ]), + end - 60_000 + offset, + value, + ) + .unwrap(), + ); + continue; + } + raw_rows.push( + raw_schema + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "job" => Value::Utf8("api".into()), + "value" => Value::Float64(value), + "ts" => Value::Timestamp(end - 60_000 + offset), + _ => panic!("unexpected raw field"), + }) + .collect(), + ); + } + } + let raw_batch = Batch::try_new(raw_schema.clone(), raw_rows).unwrap(); + for scope in [ + Scope::Ingestion { + window_start_ms: end - 60_000, + window_end_ms: end, + revision: 1, + }, + Scope::Query { + evaluation_time_ms: end, + revision: 1, + }, + ] { + let source = Box::new( + Operator::source(raw_schema.clone(), vec![raw_batch.clone()]).unwrap(), + ) as Source<'static>; + let graph = raw_compiled + .instantiate(BTreeMap::from([(u64::from(raw.id.0), source)])) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let mut raw_scores = block_on(async { + let mut scores = Vec::new(); + let mut stream = graph + .execute(&[u64::from(dag.root.0)], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + let batch = batch.unwrap(); + for row in batch.rows() { + if dynamic { + let column = batch + .schema() + .fields + .iter() + .position(|field| field.name == SERIES_IDENTITY_COLUMN) + .unwrap(); + let Value::Utf8(encoded) = &row[column] else { + panic!("identity lost"); + }; + let labels = decode_series_identity(encoded).unwrap(); + assert_eq!(labels["job"], "api"); + assert_eq!( + labels["unreferenced"], + format!("{}-extra", labels["service"]) + ); + } + assert!(row.iter().any( + |value| matches!(value, Value::Timestamp(time) if *time == end) + )); + scores.extend(row.iter().filter_map(|value| match value { + Value::Float64(value) => Some(*value), + _ => None, + })); + } + } + scores + }); + raw_scores.sort_by(f64::total_cmp); + assert_eq!(raw_scores.len(), expected.len()); + for (actual, expected) in raw_scores.iter().zip(&expected) { + assert!( + (actual - expected).abs() < 1e-12, + "raw counter semantics must precede heap ranking: {raw_scores:?}" + ); + } + } + } + let compiled = compile( + &dag, + BTreeMap::from([( + u64::from(input_id.0), + InputContract::bounded(schema.clone()), + )]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + for (time, values, expected) in [ + ( + 60_000, + vec![("auth", 3.), ("checkout", 2.), ("search", 1.)], + vec![2., 3.], + ), + ( + 61_000, + vec![("auth", 0.), ("checkout", 2.), ("search", 4.)], + vec![2., 4.], + ), + (62_000, vec![("auth", 0.), ("checkout", 2.)], vec![0., 2.]), + ] { + let rows = values + .into_iter() + .map(|(service, value)| { + if dynamic { + return series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("service".into(), service.into()), + ]), + time, + value, + ) + .unwrap(); + } + schema + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "job" => Value::Utf8("api".into()), + "value" => Value::Float64(value), + "ts" => Value::Timestamp(time), + _ => panic!("unexpected rate field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(schema.clone(), rows).unwrap(); + for scope in [ + Scope::Query { + evaluation_time_ms: time, + revision: 1, + }, + Scope::Ingestion { + window_start_ms: time - 60_000, + window_end_ms: time, + revision: 1, + }, + ] { + let source = + Box::new(Operator::source(schema.clone(), vec![batch.clone()]).unwrap()) + as Source<'static>; + let graph = compiled + .instantiate(BTreeMap::from([(u64::from(input_id.0), source)])) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let mut scores = block_on(async { + let mut scores = vec![]; + let mut stream = graph + .execute(&[u64::from(dag.root.0)], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + let batch = batch.unwrap(); + for row in batch.rows() { + assert!(row.iter().any( + |value| matches!(value, Value::Timestamp(actual) if *actual == time) + )); + scores.push( + row.iter() + .find_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .unwrap(), + ); + } + } + scores + }); + scores.sort_by(f64::total_cmp); + assert_eq!( + scores, expected, + "heap snapshots must not accumulate across evaluations" + ); + } + } + } +} + +// Spatial ranking consumes one eligible instant vector. Signed values require +// CountSketch; a raw metric does not establish the non-negative CMS contract. +#[test] +fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { + use asap_physical_operators::physical_planner::promql_rows::{ + decode_series_identity, series_row, with_series_identity, SERIES_IDENTITY_COLUMN, + }; + let logical = lower_promql("topk by(job)(1, m)", AccuracyTarget::Epsilon(0.1)).unwrap(); + let root = Rc::new(with_series_identity(&logical).unwrap()); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let candidates = strategy + .current_series_topk_candidates(&root, &AccuracyTarget::Epsilon(0.1)) + .candidates; + assert!(!candidates + .iter() + .any(|c| c.rationale.contains("CmsWithHeap"))); + let selected = candidates + .iter() + .find_map(|candidate| match &candidate.replacement { + Replacement::Summary(node) if candidate.rationale.contains("CountSketchWithHeap") => { + Some(node) + } + _ => None, + }) + .expect("signed spatial TopK must expose CountSketch with heap"); + let dag = compile_post_asap_dag(selected).unwrap(); + let raw = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::TimeRange { .. } + } + ) + }) + .unwrap(); + let schema = Arc::new(raw.output_schema.clone()); + let program = compile( + &dag, + BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + let snapshot_program = + asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + selected, + ) + .unwrap(); + let encoded: serde_json::Value = + serde_json::from_slice(&serde_json::to_vec(&snapshot_program).unwrap()).unwrap(); + assert!(!encoded.to_string().contains("CurrentSeries")); + assert!(encoded.to_string().contains("KeyedSummaryBuild")); + assert!(encoded.to_string().contains("KeyedReadout")); + for (values, expected, score) in [ + ([100., 20.], "a", 100.), + ([1., 20.], "b", 20.), + ([-10., -2.], "b", -2.), + ] { + let rows = ["a", "b"] + .into_iter() + .zip(values) + .map(|(instance, value)| { + series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("unreferenced".into(), instance.into()), + ]), + 60_000, + value, + ) + .unwrap() + }) + .collect(); + let batch = Batch::try_new(schema.clone(), rows).unwrap(); + let graph = program + .instantiate(BTreeMap::from([( + u64::from(raw.id.0), + Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'_>, + )])) + .unwrap(); + block_on(async { + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 60_000, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut result = Vec::new(); + while let Some(batch) = stream.next().await { + let batch = batch.unwrap(); + let identity = batch + .schema() + .fields + .iter() + .position(|f| f.name == SERIES_IDENTITY_COLUMN) + .unwrap(); + let value = batch + .schema() + .fields + .iter() + .position(|f| f.name == "value") + .unwrap(); + for row in batch.rows() { + let Value::Utf8(labels) = &row[identity] else { + panic!() + }; + let Value::Float64(v) = row[value] else { + panic!() + }; + result.push(( + decode_series_identity(labels).unwrap()["unreferenced"].clone(), + v, + )); + } + } + assert_eq!(result, vec![(expected.into(), score)]); + }); + } +} + +/// Deployment-side lifecycle choice: every summary state of `candidate` is +/// continuously maintained, and the chosen lifecycles set execution timing. +fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDag { + use asap_aware_mapping::{ + cost_model::{Cost, CostModel}, + enumerate_summary_maintenance_lifecycles, CostRate, Horizon, + SummaryMaintenanceCapabilities, SummaryMaintenanceLifecycleCapabilities, + SummaryMaintenanceLifecycleCostInputs, WorkloadDemand, + }; + use planner_types::workload::{ + DataArrival, Rate, RepeatedDemand, RepeatingEntry, RepetitionInterval, + }; + struct Costed; + impl CostModel for Costed { + fn rank_candidates( + &self, + _: &planner_types::pre_asap::agg_intent::AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + fn summary_maintenance_lifecycle_cost_inputs( + &self, + _: &SummaryNode, + ) -> SummaryMaintenanceLifecycleCostInputs { + SummaryMaintenanceLifecycleCostInputs { + build_cost: Some(Cost(10.)), + maintenance_cost_per_update: Some(Cost(1.)), + summary_read_cost: Some(Cost(1.)), + retention_cost_rate: Some(CostRate(0.1)), + retirement_cost: Some(Cost(1.)), + } + } + fn summary_maintenance_capabilities( + &self, + _: &SummaryNode, + ) -> SummaryMaintenanceCapabilities { + SummaryMaintenanceCapabilities { + incremental_update: true, + merge: true, + delete: true, + } + } + } + const NOW_MS: u64 = 1_000_000; + let queries = QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: None, + repeating_queries: Some(vec![RepeatingEntry { + query: Query("topk by(job)(2, rate(m[1m]))".into()), + demand: RepeatedDemand::FixedInterval(RepetitionInterval(60_000)), + requirements: QueryRequirements::default(), + predictability: Predictability::Predictable { known_at: None }, + time_selection: TimeSelection::default(), + }]), + }; + let data = DataWorkload { + arrival: DataArrival::ContinuouslyIngesting, + ingestion_rate: WorkloadEvidence { + value: Some(Rate(1.)), + source: planner_types::workload::EvidenceSource::Observed, + observed_at_ms: Some(NOW_MS), + valid_for_ms: Some(60_000), + }, + ..Default::default() + }; + let lifecycles = enumerate_summary_maintenance_lifecycles( + Rc::clone(candidate), + WorkloadDemand::new_with_data(&queries, &data, &[0]), + NOW_MS, + Some(Horizon(100.)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &Costed, + ) + .unwrap(); + let choices = lifecycles + .deployments() + .iter() + .map(|deployment| { + ( + deployment.post_asap_node_id, + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + ) + }) + .collect::>(); + lifecycles + .select(&choices) + .unwrap() + .execution_timed_dag() + .unwrap() +} + +// A maintained heap over finalized per-series Rate is the fixed-window +// placement: lifecycle timing, not a separate candidate, puts it in precompute. +#[test] +fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { + use asap_physical_operators::physical_planner::{ + compile_candidate, promql_rows::with_series_identity, + }; + let root = Rc::new( + with_series_identity( + &lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(), + ) + .unwrap(), + ); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let candidates = strategy + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .filter_map(|candidate| match candidate.replacement { + Replacement::Summary(root) if candidate.rationale.contains("WithHeap") => Some(root), + _ => None, + }) + .collect::>(); + assert_eq!(candidates.len(), 2); + for root in candidates { + let dag = continuously_maintained_dag(&root); + let state = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) + .unwrap(); + let heap = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::Sketch(..), + .. + } + ) + }) + .unwrap(); + assert_eq!(heap.output_state.timing, ExecutionTiming::IngestionTime); + let physical = compile_candidate( + &dag, + BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &[u64::from(heap.id.0)], + ) + .unwrap(); + let exported = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_aggregation(&dag).unwrap(); + assert_eq!( + serde_json::to_vec(&exported).unwrap(), + serde_json::to_vec(&physical).unwrap() + ); + // Execute the selected split across a state serialization boundary. + // Each run builds fresh weights from that window's counters. + let execute = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDag, + input: Batch, + scope: Scope| { + let id = plan.input_contracts().next().unwrap().0; + let source = Box::new(Operator::source(input.schema().clone(), vec![input]).unwrap()) + as Source<'static>; + let graph = plan.instantiate(BTreeMap::from([(id, source)])).unwrap(); + block_on(async { + let mut stream = graph + .execute( + plan.roots(), + RunContext::new(scope, Limits::default()).unwrap(), + ) + .unwrap() + .remove(0); + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push((*batch.unwrap()).clone()); + } + assert_eq!(batches.len(), 1); + batches.remove(0) + }) + }; + let (family, input, grouping) = match &state.payload { + PostAsapOperatorPayload::SummaryAgg { + family, + input, + grouping, + .. + } => (family, input, grouping), + _ => unreachable!(), + }; + for (end, samples, leader) in [ + ( + 60_000, + [[0., 100., 200.], [0., 10., 20.], [0., 1., 2.]], + "a", + ), + ( + 120_000, + [[200., 200., 200.], [100., 0., 300.], [2., 3., 4.]], + "b", + ), + ] { + let schema = Arc::new(state.output_schema.clone()); + let rows = samples + .into_iter() + .zip(["a", "b", "c"]) + .map(|(samples, label)| { + let mut accumulator = + asap_physical_operators::factory::create_planner_accumulator( + family, input, grouping, + ) + .unwrap(); + for (offset, value) in [10_000, 30_000, 50_000].into_iter().zip(samples) { + accumulator.update_single(value, end - 60_000 + offset); + } + let summary = Value::Summary { + family: family.clone(), + state: Arc::from(accumulator.into_accumulator()), + }; + schema + .fields + .iter() + .map(|field| match &field.dtype { + SummaryFamilyType::ExactAggregate(..) => summary.clone(), + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(end), + SummaryFamilyType::Plain(DataType::Utf8) + if field.name == "$promql_series_identity" => + { + Value::Utf8( + serde_json::to_string(&BTreeMap::from([ + ("job", "api"), + ("instance", label), + ])) + .unwrap() + .into(), + ) + } + SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8("api".into()), + _ => panic!("unexpected state field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(schema, rows).unwrap(); + let precompute = physical.precompute.as_ref().unwrap(); + let heap = execute( + precompute, + batch, + Scope::Ingestion { + window_start_ms: end - 60_000, + window_end_ms: end, + revision: 1, + }, + ); + let result = execute( + &physical.query, + heap, + Scope::Query { + evaluation_time_ms: end, + revision: 1, + }, + ); + let identity = result + .schema() + .fields + .iter() + .position(|f| f.name == "$promql_series_identity") + .unwrap(); + let Value::Utf8(encoded) = &result.rows()[0][identity] else { + panic!() + }; + let labels: BTreeMap = serde_json::from_str(encoded).unwrap(); + assert_eq!(labels["instance"], leader); + assert_eq!(result.rows().len(), 2); + } + let precompute = + String::from_utf8(serde_json::to_vec(&physical.precompute.unwrap()).unwrap()).unwrap(); + assert!(precompute.contains("KeyedSummaryBuild")); + assert!(precompute.contains("Rate")); + assert!( + !String::from_utf8(serde_json::to_vec(&physical.query).unwrap()) + .unwrap() + .contains("KeyedSummaryBuild") + ); + } +} diff --git a/crates/frontend-promql/src/promql.rs b/crates/frontend-promql/src/promql.rs index c3df6c67..b71a8c1e 100644 --- a/crates/frontend-promql/src/promql.rs +++ b/crates/frontend-promql/src/promql.rs @@ -307,11 +307,23 @@ fn walk(expr: &Expr) -> Result { vector_match: None, }), }, - Expr::Subquery(sq) => Ok(Unresolved::PromqlSubquery { - range: sq.range, - resolution: sq.step, - child: Rc::new(walk(&sq.expr)?), - }), + Expr::Subquery(sq) => { + let subquery = Unresolved::PromqlSubquery { + range: sq.range, + resolution: sq.step, + child: Rc::new(walk(&sq.expr)?), + }; + // `offset`/`@` move the whole subquery, including its step grid. + let shift = time_shift(sq.offset.as_ref(), sq.at.as_ref())?; + Ok(if shift.is_identity() { + subquery + } else { + Unresolved::TimeShift { + shift, + child: Rc::new(subquery), + } + }) + } Expr::VectorSelector(vs) => { let (metric, matchers, shift) = vs_parts(vs)?; Ok(instant_source(metric, matchers, shift)) diff --git a/crates/frontend-promql/tests/promql_lowering.rs b/crates/frontend-promql/tests/promql_lowering.rs index 2dd56173..3c92b7ba 100644 --- a/crates/frontend-promql/tests/promql_lowering.rs +++ b/crates/frontend-promql/tests/promql_lowering.rs @@ -1202,3 +1202,20 @@ fn histogram_quantiles_rejects_an_out_of_range_quantile() { ); } } + +// A subquery's `offset`/`@` shift the whole subquery, so the tree keeps them. +#[test] +fn subquery_time_shift_is_retained() { + let QueryExpr::Aggregate { child, .. } = lower("max_over_time(m[5m:1m] offset 1m)") else { + panic!("expected a range function"); + }; + let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + panic!("subquery offset was dropped: {child:?}"); + }; + assert_eq!(shift.offset_ms, 60_000); + assert!(matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. })); + assert!(matches!( + lower("max_over_time(m[5m:1m] @ 100)"), + QueryExpr::Aggregate { child, .. } if matches!(child.as_ref(), QueryExpr::TimeShift { .. }) + )); +} diff --git a/crates/integration-tests/Cargo.toml b/crates/integration-tests/Cargo.toml index 7b0c0d3e..5a5de990 100644 --- a/crates/integration-tests/Cargo.toml +++ b/crates/integration-tests/Cargo.toml @@ -13,3 +13,6 @@ asap-aware-mapping = { path = "../asap-aware-mapping" } asap_sketchlib = { workspace = true } serde_json = "1" tokio = { version = "1", features = ["rt", "macros", "rt-multi-thread"] } + +asap-physical-operators = { path = "../asap-physical-operators" } +futures = "0.3" diff --git a/crates/integration-tests/tests/kll_pane_execution.rs b/crates/integration-tests/tests/kll_pane_execution.rs new file mode 100644 index 00000000..e818796f --- /dev/null +++ b/crates/integration-tests/tests/kll_pane_execution.rs @@ -0,0 +1,286 @@ +//! Maintenance -> stored pane state -> independently bound query execution. +mod physical_common; +use asap_physical_operators::{ + operators::{Operator, ReadoutQuery}, + physical_planner::{CompiledPhysicalDag, InputContract, Source}, + plan::{PhysicalDag, PhysicalOperator, PlanProperties}, + runtime::{Input, Limits, OutputStream, RunContext, Scope}, + summary_kernels::datasketches_kll::DatasketchesKLLAccumulator, + values::{Batch, Schema, Value}, + AggregateCore, Error, +}; +use asap_types::{ + post_asap::{ + SketchAlgorithm, SketchKind, SketchParams, SketchQuery, SummaryFamilyType, SummaryField, + SummarySchema, + }, + pre_asap::DataType, +}; +use futures::{executor::block_on, StreamExt}; +use std::{ + collections::BTreeMap, + sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, + }, +}; + +fn family(k: u32) -> SummaryFamilyType { + SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), + Default::default(), + ) +} +fn raw_schema() -> Schema { + Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }) +} +fn query_scope() -> Scope { + Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + } +} +fn fixture() -> (CompiledPhysicalDag, Operator, Schema) { + let raw = raw_schema(); + let build = Operator::summary_build(raw.clone(), family(200), 0, None, vec![]).unwrap(); + let state = build.schema(); + let maintenance = CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(raw))]), + BTreeMap::from([(1, (vec![0], build))]), + vec![1], + ) + .unwrap(); + let merge = Operator::summary_merge(state.clone(), 0, vec![]).unwrap(); + (maintenance, merge, state) +} +fn pane_state(maintenance: &CompiledPhysicalDag, pane: i64) -> Arc { + // Twenty samples in each (start,end] one-minute pane; k=200 avoids + // compaction so quantiles and sample counts have deterministic oracles. + let raw = raw_schema(); + let rows = (0..20) + .map(|i| vec![Value::Float64((pane * 20 + i) as f64)]) + .collect(); + let state = physical_common::execute( + maintenance, + BTreeMap::from([(0, Batch::try_new(raw, rows).unwrap())]), + Scope::Ingestion { + window_start_ms: pane * 60_000, + window_end_ms: (pane + 1) * 60_000, + revision: 1, + }, + ); + let Value::Summary { state, .. } = &state[0][0].rows()[0][0] else { + panic!("missing KLL") + }; + state.clone() +} +fn restore(schema: Schema, states: &[Arc]) -> Batch { + Batch::try_new( + schema, + states + .iter() + .map(|state| { + vec![Value::Summary { + family: family(200), + state: state.clone(), + }] + }) + .collect(), + ) + .unwrap() +} +fn readout(schema: Schema, q: f64) -> Operator { + Operator::readout(schema, 0, ReadoutQuery::Sketch(SketchQuery::Quantile { q })).unwrap() +} +struct CountStarts { + operator: Operator, + starts: Arc, +} +impl PhysicalOperator for CountStarts { + fn name(&self) -> &str { + self.operator.name() + } + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + self.operator.properties(inputs) + } + fn requires_bounded_input(&self) -> bool { + self.operator.requires_bounded_input() + } + fn input_schemas(&self) -> Vec { + self.operator.input_schemas() + } + fn output_schema(&self) -> Schema { + self.operator.output_schema() + } + fn output_bytes(&self, batch: &Batch) -> usize { + self.operator.output_bytes(batch) + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + self.starts.fetch_add(1, Ordering::SeqCst); + self.operator.start(inputs, context) + } +} + +/// Actual codec bytes survive destruction of maintenance state; one shared +/// native merge supplies p50, p99 and the population-count oracle per run. +#[test] +fn five_panes_roundtrip_and_shared_merge_runs_once() { + let (maintenance, merge, schema) = fixture(); + let panes: Vec<_> = (0..6).map(|pane| pane_state(&maintenance, pane)).collect(); + drop(maintenance); + let compiled = CompiledPhysicalDag::from_operators( + (0..5) + .map(|id| (id, InputContract::bounded(schema.clone()))) + .collect(), + BTreeMap::from([ + ( + 5, + ( + vec![0, 1, 2, 3, 4], + Operator::union(schema.clone(), 5).unwrap(), + ), + ), + (6, (vec![5], merge.clone())), + (7, (vec![6], readout(schema.clone(), 0.5))), + (8, (vec![6], readout(schema.clone(), 0.99))), + ]), + vec![6, 7, 8], + ) + .unwrap(); + let layout = asap_types::post_asap::PaneLayout { + pane_width_ms: 60_000, + pane_origin_ms: Some(0), + }; + assert!( + asap_types::post_asap::validate_pane_coverage( + &layout, + Some(330_000), + &asap_types::post_asap::WindowEdgeCoverage::PaneAligned + ) + .is_err(), + "moving window edges require residual computation" + ); + for offset in [0, 1] { + let restored = restore(schema.clone(), &panes[offset..offset + 5]); + let evaluation_time_ms = (5 + offset as i64) * 60_000; + asap_types::post_asap::validate_pane_coverage( + &layout, + Some(evaluation_time_ms), + &asap_types::post_asap::WindowEdgeCoverage::PaneAligned, + ) + .unwrap(); + let inputs: BTreeMap<_, _> = (0..5) + .map(|id| { + ( + id as u64, + restore(schema.clone(), &panes[offset + id..offset + id + 1]), + ) + }) + .collect(); + // A five-pane deployment cannot bind only four state slots. + let incomplete: BTreeMap<_, _> = inputs + .iter() + .take(4) + .map(|(&id, batch)| { + ( + id, + Box::new(Operator::source(schema.clone(), vec![batch.clone()]).unwrap()) + as Source<'_>, + ) + }) + .collect(); + assert!(compiled.instantiate(incomplete).is_err()); + let result = physical_common::execute( + &compiled, + inputs, + Scope::Query { + evaluation_time_ms, + revision: 1, + }, + ); + let Value::Summary { state, .. } = &result[0][0].rows()[0][0] else { + panic!("missing merged state") + }; + let kll = state + .as_any() + .downcast_ref::() + .unwrap(); + assert_eq!(kll.inner.count(), 100); + let value = |index: usize| match result[index][0].rows()[0][0] { + Value::Float64(value) => value, + _ => panic!("missing quantile"), + }; + assert!((value(1) - (50 + offset * 20) as f64).abs() <= 1.); + assert!((value(2) - (99 + offset * 20) as f64).abs() <= 1.); + let starts = Arc::new(AtomicUsize::new(0)); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(schema.clone(), vec![restored]).unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + CountStarts { + operator: merge.clone(), + starts: starts.clone(), + }, + ) + .unwrap(); + dag.add(2, vec![1], readout(schema.clone(), 0.5)).unwrap(); + dag.add(3, vec![1], readout(schema.clone(), 0.99)).unwrap(); + let outputs = block_on(futures::future::join_all( + dag.execute( + &[2, 3], + RunContext::new(query_scope(), Limits::default()).unwrap(), + ) + .unwrap() + .into_iter() + .map(|stream| stream.collect::>()), + )); + assert!(outputs + .iter() + .all(|output| output.len() == 1 && output[0].is_ok())); + assert_eq!(starts.load(Ordering::SeqCst), 1); + } +} + +/// Relabelled KLL parameters and missing bindings fail explicitly. +#[test] +fn panes_reject_parameters_schema_and_missing_binding() { + let (_maintenance, merge, schema) = fixture(); + let wrong = Value::Summary { + family: family(200), + state: Arc::new(DatasketchesKLLAccumulator::new(128)), + }; + assert!(Batch::try_new(schema.clone(), vec![vec![wrong]]).is_err()); + let compiled = CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema))]), + BTreeMap::from([(1, (vec![0], merge))]), + vec![1], + ) + .unwrap(); + assert!(compiled.instantiate(BTreeMap::new()).is_err()); + let raw = raw_schema(); + let source = Operator::source( + raw.clone(), + vec![Batch::try_new(raw, vec![vec![Value::Float64(1.)]]).unwrap()], + ) + .unwrap(); + assert!(compiled + .instantiate(BTreeMap::from([(0, Box::new(source) as Source<'_>)])) + .is_err()); +} diff --git a/crates/integration-tests/tests/physical_common/mod.rs b/crates/integration-tests/tests/physical_common/mod.rs new file mode 100644 index 00000000..aca4d4ef --- /dev/null +++ b/crates/integration-tests/tests/physical_common/mod.rs @@ -0,0 +1,42 @@ +use asap_physical_operators::{ + operators::Operator, + physical_planner::{CompiledPhysicalDag, Source}, + runtime::{Limits, RunContext, Scope}, + values::Batch, +}; +use futures::{executor::block_on, StreamExt}; +use std::collections::BTreeMap; + +pub fn execute( + plan: &CompiledPhysicalDag, + inputs: BTreeMap, + scope: Scope, +) -> Vec> { + let sources = inputs + .into_iter() + .map(|(id, batch)| { + ( + id, + Box::new(Operator::source(batch.schema().clone(), vec![batch]).unwrap()) + as Source<'_>, + ) + }) + .collect(); + let dag = plan.instantiate(sources).unwrap(); + block_on(async { + let streams = dag + .execute( + plan.roots(), + RunContext::new(scope, Limits::default()).unwrap(), + ) + .unwrap(); + futures::future::join_all(streams.into_iter().map(|mut stream| async move { + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push((*batch.unwrap()).clone()); + } + batches + })) + .await + }) +} diff --git a/crates/integration-tests/tests/precompute_raw_samples.rs b/crates/integration-tests/tests/precompute_raw_samples.rs new file mode 100644 index 00000000..ade52ec8 --- /dev/null +++ b/crates/integration-tests/tests/precompute_raw_samples.rs @@ -0,0 +1,610 @@ +//! Planner-selected summaries over raw samples compile as precompute graphs +//! and produce the same estimates as feeding their kernel sample by sample. +use std::{collections::BTreeMap, collections::BTreeSet, rc::Rc, sync::Arc}; + +use asap_aware_mapping::cost_model::DefaultCostModel; +use asap_aware_mapping::{ + search_workload, Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, + TargetSubDAG, +}; +use asap_integration_tests::fixtures::lower_promql; +use asap_physical_operators::{ + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{precompute, Source}, + runtime::{Limits, RunContext, Scope}, + summary_kernels::{exact::ExactAccumulator, weighted_frequency::WeightedFrequency}, + values::{Batch, Value}, + AggregateCore, KeyByLabelValues, Statistic, +}; +use asap_types::post_asap::{ + compile_post_asap_dag, EntityIdentity, ExactKind, PostAsapDag, PostAsapOperatorPayload, + SketchAlgorithm, SketchQuery, SummaryFamilyType, SummaryInputExpr, SummaryNode, SummaryUpdate, +}; +use asap_types::pre_asap::{expr_ir::ColumnRef, query_expr::Reduction}; +use asap_types::types::AccuracyTarget; +use futures::{executor::block_on, StreamExt}; + +type Series = BTreeMap; + +/// A series label set; `None` omits the label. An empty value is present +/// in the input but is not part of the series identity. +fn series(service: Option<&str>, instance: &str) -> Series { + [ + ("__name__", Some("m")), + ("service", service), + ("instance", Some(instance)), + ] + .into_iter() + .filter_map(|(k, v)| Some((k.to_owned(), v?.to_owned()))) + .collect() +} + +fn canonical(labels: &Series) -> Series { + labels + .iter() + .filter(|(_, v)| !v.is_empty()) + .map(|(k, v)| (k.clone(), v.clone())) + .collect() +} + +/// Every Planner candidate for `query`: the searched selection plus each +/// summary replacement of the root. +fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { + let root = Rc::new(lower_promql(query, accuracy).expect("lowering failed")); + let mut result = SketchAlgorithmStrategy::default_cost_model() + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .filter_map(|candidate| match candidate { + ReplacementSubDAG { + replacement: Replacement::Summary(node), + .. + } => Some(node), + _ => None, + }) + .collect::>(); + let space = search_workload(vec![("query", root)]); + if let Ok(Some(selected)) = space + .global_selection(&DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + { + result.push(selected); + } + result +} + +/// Raw-input summary nodes: `(dag, raw source id, summary id)`. +fn raw_summaries(dag: &PostAsapDag) -> Vec<(u64, u64)> { + dag.nodes + .iter() + .filter(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .filter_map(|node| { + let inputs = dag + .edges + .iter() + .filter(|edge| edge.consumer == node.id) + .collect::>(); + let [edge] = inputs.as_slice() else { + return None; + }; + let source = dag.nodes.iter().find(|n| n.id == edge.producer)?; + matches!(source.payload, PostAsapOperatorPayload::Fallback { .. }) + .then_some((u64::from(source.id.0), u64::from(node.id.0))) + }) + .collect() +} + +fn samples() -> Vec<(Series, i64, f64)> { + let mut rows = Vec::new(); + let series_set = [ + (Some("a"), "1"), + (Some("a"), "2"), + (Some("b"), "1"), + (None, "3"), + (Some("b"), ""), + ]; + for (index, (service, instance)) in series_set.iter().enumerate() { + for step in 1..=5i64 { + let value = (index as f64 + 1.0) * step as f64 + (step % 2) as f64; + rows.push((series(*service, instance), step * 1000, value)); + } + } + rows +} + +fn execute( + dag: &PostAsapDag, + source: u64, + root: u64, + rows: &[(Series, i64, f64)], +) -> Vec<(Series, Arc)> { + let program = precompute::compile(dag, &[source], &[root]).unwrap_or_else(|error| { + panic!( + "raw summary {root} does not compile: {error}; source {:?}", + dag.nodes + .iter() + .find(|n| u64::from(n.id.0) == source) + .map(|n| (&n.output_schema, &n.payload)) + ); + }); + let program = serde_json::from_slice::< + asap_physical_operators::physical_planner::CompiledPhysicalDag, + >(&serde_json::to_vec(&program).unwrap()) + .unwrap(); + let schema = precompute::raw_sample_schema(); + let batch = Batch::try_new( + schema.clone(), + rows.iter() + .map(|(labels, time, value)| precompute::raw_sample_row(labels, *time, *value)) + .collect(), + ) + .unwrap(); + let sources = BTreeMap::from([( + source, + Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, + )]); + let graph = program.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 6000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut result = Vec::new(); + while let Some(batch) = stream.next().await { + for row in batch.unwrap().rows() { + let [Value::Map(labels), Value::Timestamp(6000), Value::Summary { state, .. }] = + row.as_slice() + else { + panic!("unexpected population row {row:?}"); + }; + let labels = labels + .iter() + .map(|(k, v)| match (k, v) { + (Value::Utf8(k), Value::Utf8(v)) => (k.to_string(), v.to_string()), + _ => panic!("non-label population entry"), + }) + .collect(); + result.push((labels, state.clone())); + } + } + result + }) +} + +/// PromQL grouping of a canonical label set: `by` keeps the named labels, +/// `without` drops them and `__name__`. +fn population(reduction: &Reduction, dag_labels: &[String], labels: &Series) -> Series { + let labels = canonical(labels); + match reduction { + Reduction::PerEntity => labels, + Reduction::Reduce(keys) => labels + .into_iter() + .filter(|(k, _)| { + if keys.is_without() { + k != "__name__" && !dag_labels.contains(k) + } else { + dag_labels.contains(k) + } + }) + .collect(), + } +} + +/// Item identity used by a keyed summary: labels by name, the sample value, +/// or the canonical label-set identity. +fn item(expr: &SummaryInputExpr, labels: &Series, value: f64, out: &mut Vec) { + match expr { + SummaryInputExpr::Column(ColumnRef::SampleValue) => out.push(Value::Float64(value)), + SummaryInputExpr::Column(ColumnRef::Named(name)) if name == "value" => { + out.push(Value::Float64(value)) + } + SummaryInputExpr::Column(ColumnRef::Named(name)) => out.push(Value::Utf8( + canonical(labels) + .get(name) + .cloned() + .unwrap_or_default() + .into(), + )), + SummaryInputExpr::EntityIdentity(EntityIdentity::PromqlLabelSet { excluding }) => { + let mut identity = canonical(labels); + for column in excluding { + let ColumnRef::Named(name) = column else { + panic!("unsupported fixture exclusion {column:?}") + }; + identity.remove(name); + } + out.push(Value::Utf8( + serde_json::to_string(&identity).unwrap().into(), + )) + } + SummaryInputExpr::Tuple(items) => items.iter().for_each(|i| item(i, labels, value, out)), + other => panic!("unsupported fixture item {other:?}"), + } +} + +fn weight(update: &SummaryUpdate, value: f64) -> f64 { + match update.weight { + SummaryInputExpr::Constant(weight) => weight, + _ => value, + } +} + +/// Estimates that identify a state's content for comparison. +fn readouts(state: &dyn AggregateCore, family: &SummaryFamilyType) -> Vec { + if let Some(exact) = state.as_any().downcast_ref::() { + let SummaryFamilyType::ExactAggregate(kind, _) = family else { + unreachable!() + }; + let statistic = match kind { + ExactKind::Sum => Statistic::Sum, + ExactKind::Count => Statistic::Count, + ExactKind::Min => Statistic::Min, + ExactKind::Max => Statistic::Max, + ExactKind::Rate => Statistic::Rate, + ExactKind::Increase => Statistic::Increase, + other => panic!("unexpected exact kind {other:?}"), + }; + return vec![exact + .readout(statistic, None, None::<&KeyByLabelValues>) + .unwrap() + .unwrap()]; + } + let SummaryFamilyType::Sketch(kind, _) = family else { + panic!("sketch state for exact family") + }; + match kind.algorithm() { + SketchAlgorithm::Kll | SketchAlgorithm::DDSketch => [0.1, 0.5, 0.9] + .into_iter() + .map(|q| state.estimate(&SketchQuery::Quantile { q }).unwrap()) + .collect(), + SketchAlgorithm::Hll => vec![state.estimate(&SketchQuery::Cardinality).unwrap()], + other => panic!("unexpected unkeyed sketch {other:?}"), + } +} + +/// Compile one raw-input summary, execute it over `rows`, and compare each +/// population with its kernel fed sample by sample. Returns the family label, +/// or the family when it has no native state. +fn check( + query: &str, + dag: &PostAsapDag, + source: u64, + root: u64, + rows: &[(Series, i64, f64)], +) -> Result { + let node = dag + .nodes + .iter() + .find(|n| u64::from(n.id.0) == root) + .unwrap(); + let PostAsapOperatorPayload::SummaryAgg { + family, + input, + reduction, + grouping, + } = &node.payload + else { + unreachable!() + }; + let source_node = dag + .nodes + .iter() + .find(|n| u64::from(n.id.0) == source) + .unwrap(); + let keys = match reduction { + Reduction::Reduce(keys) => keys + .keys() + .iter() + .map(|i| source_node.output_schema.fields[*i].name.clone()) + .collect(), + Reduction::PerEntity => vec![], + }; + if asap_physical_operators::capability::validate_native_family(family).is_err() { + // Families without a native state (e.g. plain CMS, UnivMon) + // are outside precompute execution; their compile must fail. + assert!(precompute::compile(dag, &[source], &[root]).is_err()); + return Err(format!("{family:?}")); + } + let actual = execute(dag, source, root, rows); + let label = match family { + SummaryFamilyType::ExactAggregate(kind, _) => format!("{kind:?}"), + SummaryFamilyType::Sketch(kind, _) => format!("{:?}", kind.algorithm()), + other => format!("{other:?}"), + }; + if let SummaryFamilyType::Sketch(kind, _) = family { + if let (Some(keyed), false) = (&input.item, kind.algorithm() == &SketchAlgorithm::Hll) { + // Keyed heaps: every item's estimated weight is its exact + // total at this scale (no collisions in the fixture). + let mut expected = BTreeMap::>::new(); + for (labels, _, value) in rows { + let mut items = Vec::new(); + item(keyed, labels, *value, &mut items); + *expected + .entry(population(reduction, &keys, labels)) + .or_default() + .entry(format!("{items:?}")) + .or_default() += weight(input, *value); + } + assert_eq!(actual.len(), expected.len(), "{query}"); + for (labels, state) in &actual { + let heap = state.as_any().downcast_ref::().unwrap(); + let got = heap + .rows(usize::MAX >> 1) + .into_iter() + .map(|mut row| { + let Some(Value::Float64(score)) = row.pop() else { + panic!("heap score") + }; + (format!("{row:?}"), score) + }) + .collect::>(); + assert_eq!(&got, &expected[labels], "{query}"); + } + return Ok(label); + } + } + let mut expected = + BTreeMap::>::new(); + for (labels, time, value) in rows { + let updater = expected + .entry(population(reduction, &keys, labels)) + .or_insert_with(|| create_planner_accumulator(family, input, grouping).unwrap()); + let unit = input.item.is_some(); + updater.update_single(if unit { *value } else { weight(input, *value) }, *time); + } + assert_eq!(actual.len(), expected.len(), "{query}"); + for (labels, state) in actual { + let reference = expected[&labels].snapshot_accumulator(); + assert_eq!( + readouts(state.as_ref(), family), + readouts(reference.as_ref(), family), + "{query}: {labels:?}" + ); + } + Ok(label) +} + +// Every raw-input summary selected by Planner compiles over raw sample rows, +// and each population's estimates equal feeding its kernel sample by sample. +#[test] +fn raw_sample_summaries_compile_and_match_their_kernels() { + let exact = AccuracyTarget::Exact; + let sketch = AccuracyTarget::Epsilon(0.02); + let queries = [ + ("sum_over_time(m[5m])", &exact), + ("count_over_time(m[5m])", &exact), + ("min_over_time(m[5m])", &exact), + ("max_over_time(m[5m])", &exact), + ("rate(m[5m])", &exact), + ("increase(m[5m])", &exact), + ("sum by (service) (sum_over_time(m[5m]))", &exact), + ("sum by (service) (rate(m[5m]))", &exact), + ("topk(2, sum_over_time(m[5m]))", &exact), + ("quantile_over_time(0.9, m[5m])", &sketch), + ("sum by (service) (quantile_over_time(0.9, m[5m]))", &sketch), + ("quantile by (service) (0.9, m)", &sketch), + ("distinct_over_time(m[5m])", &sketch), + ("count(m)", &sketch), + ("topk(2, m)", &sketch), + ( + "topk(2, sum by (service) (count_over_time(m[5m])))", + &sketch, + ), + ("topk(2, sum_over_time(m[5m]))", &sketch), + ("topk by (service) (2, sum_over_time(m[5m]))", &sketch), + ]; + let rows = samples(); + let mut families = BTreeSet::new(); + let mut unsupported = BTreeSet::new(); + let mut checked = BTreeMap::new(); + for (query, accuracy) in queries { + for candidate in candidates(query, accuracy.clone()) { + let dag = compile_post_asap_dag(&candidate).unwrap(); + for (source, root) in raw_summaries(&dag) { + match check(query, &dag, source, root, &rows) { + Ok(family) => { + families.insert(family); + *checked.entry(query).or_insert(0) += 1; + } + Err(family) => { + unsupported.insert(family); + } + } + } + } + } + println!("checked {checked:?}; families {families:?}; without native state {unsupported:?}"); + for query in [ + "topk by (service) (2, sum_over_time(m[5m]))", + "quantile by (service) (0.9, m)", + "distinct_over_time(m[5m])", + ] { + assert!( + checked.contains_key(query), + "{query} has no checked raw summary" + ); + } + for family in [ + "Sum", + "Count", + "Min", + "Max", + "Rate", + "Increase", + "Kll", + "DDSketch", + "Hll", + "CountSketchWithHeap", + ] { + assert!( + families.contains(family), + "no {family} fixture: {families:?}" + ); + } +} + +/// Replace the raw summary of `sum by (service) (sum_over_time(m[5m]))` with +/// another update, keeping its raw input and reduction. +fn grouped_raw_summary(family: SummaryFamilyType, input: SummaryUpdate) -> (PostAsapDag, u64, u64) { + let candidate = candidates( + "sum by (service) (sum_over_time(m[5m]))", + AccuracyTarget::Exact, + ) + .pop() + .unwrap(); + let mut dag = compile_post_asap_dag(&candidate).unwrap(); + let (source, root) = raw_summaries(&dag)[0]; + let node = dag + .nodes + .iter_mut() + .find(|n| u64::from(n.id.0) == root) + .unwrap(); + let PostAsapOperatorPayload::SummaryAgg { + family: old, + input: update, + .. + } = &mut node.payload + else { + unreachable!() + }; + for field in &mut node.output_schema.fields { + if field.dtype == *old { + field.dtype = family.clone(); + } + } + *old = family; + *update = input; + let schema = node.output_schema.clone(); + for edge in dag + .edges + .iter_mut() + .filter(|e| u64::from(e.producer.0) == root) + { + edge.intermediate_schema = schema.clone(); + } + (dag, source, root) +} + +// Keyed heaps over raw samples resolve items from the series label set and +// estimate each item's exact total; invalid weight contracts do not compile. +#[test] +fn raw_sample_heaps_resolve_items_from_labels() { + use asap_types::post_asap::{ + GroupingStrategy::PerSubpopulationInstance, NonNegativeWeightProof, SketchKind, + SketchParams, WeightDomain, + }; + let heap = |algorithm, params| { + SummaryFamilyType::Sketch(SketchKind::new(algorithm, params), PerSubpopulationInstance) + }; + let cms = heap( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 64, + depth: 3, + heap_size: 8, + }, + ); + let count_sketch = heap( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 64, + depth: 3, + heap_size: 8, + }, + ); + let identity = SummaryUpdate { + item: Some(SummaryInputExpr::EntityIdentity( + EntityIdentity::PromqlLabelSet { excluding: vec![] }, + )), + weight: SummaryInputExpr::Constant(1.0), + weight_domain: WeightDomain::NonNegative { + proof: NonNegativeWeightProof::UnitCount, + }, + }; + let by_instance = SummaryUpdate { + item: Some(SummaryInputExpr::Tuple(vec![SummaryInputExpr::Column( + ColumnRef::Named("instance".into()), + )])), + weight: SummaryInputExpr::Column(ColumnRef::SampleValue), + weight_domain: WeightDomain::UnknownOrSigned, + }; + let rows = samples(); + for (family, input) in [ + (cms.clone(), identity.clone()), + (count_sketch.clone(), identity), + (count_sketch, by_instance.clone()), + ] { + let (dag, source, root) = grouped_raw_summary(family, input); + assert!(check("heap", &dag, source, root, &rows).is_ok()); + } + let signed_cms = grouped_raw_summary(cms.clone(), by_instance.clone()); + assert!(precompute::compile(&signed_cms.0, &[signed_cms.1], &[signed_cms.2]).is_err()); + let derivative = grouped_raw_summary( + cms, + SummaryUpdate { + weight_domain: WeightDomain::NonNegative { + proof: NonNegativeWeightProof::ResetAwareCounterDerivative, + }, + ..by_instance.clone() + }, + ); + assert!( + precompute::compile(&derivative.0, &[derivative.1], &[derivative.2]).is_err(), + "raw samples are cumulative counters, not their derivative" + ); + // A scan's time column is not a label; it cannot silently read as empty. + let time_item = grouped_raw_summary( + heap( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 64, + depth: 3, + heap_size: 8, + }, + ), + SummaryUpdate { + item: Some(SummaryInputExpr::Column(ColumnRef::Named("ts".into()))), + ..by_instance + }, + ); + assert!(precompute::compile(&time_item.0, &[time_item.1], &[time_item.2]).is_err()); +} + +// `without` grouping over raw samples drops the listed labels and `__name__`. +#[test] +fn raw_sample_without_grouping_drops_labels_and_name() { + use asap_types::pre_asap::query_expr::GroupKeys; + let family = + SummaryFamilyType::ExactAggregate(ExactKind::Sum, asap_types::post_asap::ExactParams::Sum); + let (mut dag, source, root) = + grouped_raw_summary(family, SummaryUpdate::column(ColumnRef::SampleValue)); + let service = dag + .nodes + .iter() + .find(|n| u64::from(n.id.0) == source) + .unwrap() + .output_schema + .fields + .iter() + .position(|f| f.name == "service") + .unwrap(); + let node = dag + .nodes + .iter_mut() + .find(|n| u64::from(n.id.0) == root) + .unwrap(); + let PostAsapOperatorPayload::SummaryAgg { reduction, .. } = &mut node.payload else { + unreachable!() + }; + *reduction = Reduction::Reduce(GroupKeys::without(vec![service])); + assert_eq!( + check("without", &dag, source, root, &samples()), + Ok("Sum".into()) + ); +} diff --git a/crates/integration-tests/tests/sql_to_physical.rs b/crates/integration-tests/tests/sql_to_physical.rs new file mode 100644 index 00000000..be96107e --- /dev/null +++ b/crates/integration-tests/tests/sql_to_physical.rs @@ -0,0 +1,165 @@ +//! SQL frontend, candidate selection, physical compilation and fresh-run execution. +use asap_aware_mapping::{search_workload, DefaultCostModel}; +use asap_frontend_sql::{lower_sql, SqlCatalog}; +use asap_physical_operators::{ + physical_planner::{compile, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + sources::{DataSources, MemorySource}, + values::{Batch, Value}, +}; +use asap_types::{ + post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, + pre_asap::{Column, DataType, QueryExpr, Schema}, + types::AccuracyTarget, +}; +use futures::StreamExt; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; + +/// SQL filtering and grouped aggregation survive logical/physical lowering; +/// rebinding the compiled DAG runs against new data rather than cached results. +#[tokio::test] +async fn sql_filter_grouped_sum_executes_and_rebinds() { + let catalog = SqlCatalog::new().with_table( + "metrics", + Schema::new(vec![ + Column::new("service", DataType::Utf8, false), + Column::new("value", DataType::Float64, true), + ]), + ); + for query in [ + "SELECT service, SUM(value) AS total FROM metrics WHERE value > 1 GROUP BY service", + "SELECT service, SUM(value) AS total FROM metrics GROUP BY service", + ] { + let logical = Rc::new( + lower_sql(query, &catalog, AccuracyTarget::Exact) + .await + .unwrap(), + ); + let space = search_workload(vec![("sql", logical)]); + let selected = space + .global_selection(&DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + let dag = compile_post_asap_dag(&selected).unwrap(); + let scan = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::Scan { .. } + } + ) + }) + .expect("raw SQL scan"); + let schema = Arc::new(scan.output_schema.clone()); + assert!(schema + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + let plan = compile( + &dag, + BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + for multiplier in [1., 2.] { + let rows = [ + ("api", Some(2.)), + ("api", Some(3.)), + ("api", None), + ("batch", Some(4.)), + ("batch", Some(1.)), + ] + .into_iter() + .map(|(service, value)| { + schema + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "value" => { + value.map_or(Value::Null, |value| Value::Float64(value * multiplier)) + } + _ => panic!("unexpected field {field:?}"), + }) + .collect() + }) + .collect(); + let PostAsapOperatorPayload::Fallback { expression } = &scan.payload else { + unreachable!() + }; + let QueryExpr::Scan { source, .. } = expression else { + unreachable!() + }; + let mut sources = DataSources::default(); + sources + .register( + source.clone(), + Arc::new( + MemorySource::new( + schema.clone(), + vec![Batch::try_new(schema.clone(), rows).unwrap()], + ) + .unwrap(), + ), + ) + .unwrap(); + let bound = plan + .instantiate(BTreeMap::from([( + u64::from(scan.id.0), + Box::new(sources.bind(expression).unwrap()) as Source<'_>, + )])) + .unwrap(); + let mut stream = bound + .execute( + plan.roots(), + RunContext::new( + Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(), + ) + .unwrap() + .remove(0); + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push(batch.unwrap()); + } + let mut actual: Vec<_> = batches + .iter() + .flat_map(|batch| batch.rows()) + .map(|row| { + let Value::Utf8(service) = &row[0] else { + panic!("missing service") + }; + let Value::Float64(value) = row[1] else { + panic!("missing sum") + }; + (service.to_string(), value) + }) + .collect(); + actual.sort_by(|a, b| a.0.cmp(&b.0)); + assert_eq!( + actual, + vec![ + ("api".into(), 5. * multiplier), + ( + "batch".into(), + 4. * multiplier + + if query.contains("WHERE") && multiplier == 1. { + 0. + } else { + multiplier + } + ) + ] + ); + } + } +} diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index 186dd04b..08e3e3b1 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -119,52 +119,7 @@ fn dashboard_workload() -> PlanningWorkload { #[test] fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() { let workload = dashboard_workload(); - workload.validate().unwrap(); - - let lowered = lower_promql_workload(&workload, 0) - .expect("valid PromQL workload") - .into_iter() - .next() - .expect("one normalized workload entry"); - let root = Rc::new(lowered); - let strategies = asap_aware_mapping::default_strategies_with(&FullyCostedRuntime); - let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); - let target = Rc::clone(&space.roots[0].1); - let capabilities = SummaryMaintenanceLifecycleCapabilities { - supports_ephemeral: true, - supports_prepared: false, - supports_shared: false, - supports_continuously_maintained: true, - }; - - let selection = global_selection_with_summary_maintenance_lifecycles( - &space, - WorkloadDemand { - workload: &workload.query_workload, - data_workload: workload.data_workload.as_ref(), - entry_indices: &[1], - }, - NOW_MS, - Some(Horizon(100.0)), - capabilities, - &FullyCostedRuntime, - ) - .unwrap(); - let plan = assemble_selected_dag_with_summary_maintenance_lifecycles( - &selection, - &target, - WorkloadDemand::new_with_data( - &workload.query_workload, - workload.data_workload.as_ref().unwrap(), - &[1], - ), - NOW_MS, - Some(Horizon(100.0)), - capabilities, - &FullyCostedRuntime, - ) - .unwrap() - .expect("selected summary plan"); + let plan = selected_plan(&workload); assert!(!plan.selected_raw_recompute); assert_eq!(plan.expected_reads, Some(100.0)); @@ -231,3 +186,779 @@ fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() "continuously_maintained" ); } + +fn selected_plan( + workload: &PlanningWorkload, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + selected_plan_with_model(workload, &FullyCostedRuntime) +} + +fn selected_plan_with_model( + workload: &PlanningWorkload, + model: &dyn CostModel, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + selected_plan_with_horizon(workload, model, Horizon(100.)) +} + +fn selected_plan_with_horizon( + workload: &PlanningWorkload, + model: &dyn CostModel, + horizon: Horizon, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + workload.validate().unwrap(); + + let lowered = lower_promql_workload(workload, 0) + .expect("valid PromQL workload") + .into_iter() + .next() + .expect("one normalized workload entry"); + selected_plan_for_lowered(workload, lowered, model, horizon) +} + +fn selected_plan_for_lowered( + workload: &PlanningWorkload, + lowered: asap_types::pre_asap::QueryExpr, + model: &dyn CostModel, + horizon: Horizon, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + let root = Rc::new(lowered); + let strategies = asap_aware_mapping::default_strategies_with(model); + let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); + let target = Rc::clone(&space.roots[0].1); + let capabilities = SummaryMaintenanceLifecycleCapabilities { + supports_ephemeral: true, + supports_prepared: false, + supports_shared: false, + supports_continuously_maintained: true, + }; + + let selection = global_selection_with_summary_maintenance_lifecycles( + &space, + WorkloadDemand { + workload: &workload.query_workload, + data_workload: workload.data_workload.as_ref(), + entry_indices: &[1], + }, + NOW_MS, + Some(horizon), + capabilities, + model, + ) + .unwrap(); + assemble_selected_dag_with_summary_maintenance_lifecycles( + &selection, + &target, + WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + NOW_MS, + Some(horizon), + capabilities, + model, + ) + .unwrap() + .expect("selected summary plan") +} + +mod physical_common; + +/// A selected continuous lifecycle supplies a materialization boundary; its +/// maintenance and query DAGs execute the selected KLL computation in fresh runs. +#[test] +fn continuous_lifecycle_compiles_and_executes_spatial_kll() { + use asap_physical_operators::{ + physical_planner::{compile_candidate, InputContract}, + runtime::Scope, + values::{Batch, Value}, + }; + use asap_types::{ + post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, + pre_asap::DataType, + }; + use std::{collections::BTreeMap, sync::Arc}; + let mut workload = dashboard_workload(); + workload.query_workload.query_batch.as_mut().unwrap()[0].query = + Query("quantile(0.99, latency)".into()); + workload.query_workload.repeating_queries.as_mut().unwrap()[0].query = + Query("quantile(0.99, latency)".into()); + let selected = selected_plan(&workload); + assert_eq!( + selected.deployments[0] + .summary_maintenance_lifecycle_guarantee + .as_ref() + .unwrap() + .summary_maintenance_lifecycle, + SummaryMaintenanceLifecycle::ContinuouslyMaintained + ); + let dag = compile_post_asap_dag(&selected.root).unwrap(); + let build = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .unwrap(); + let input = dag + .edges + .iter() + .find(|edge| edge.consumer == build.id) + .unwrap() + .producer; + let raw = dag.nodes.iter().find(|node| node.id == input).unwrap(); + let schema = Arc::new(raw.output_schema.clone()); + let candidate = compile_candidate( + &dag, + BTreeMap::from([(u64::from(input.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &[u64::from(build.id.0)], + ) + .unwrap(); + + // A continuous input without a finite pane boundary cannot implement this + // blocking builder. Retain lifecycle ownership in the candidate payload; + // only the legal bounded request candidate reaches workload pricing. + let mut unbounded = InputContract::bounded(schema.clone()); + unbounded.properties.boundedness = asap_physical_operators::plan::Boundedness::Unbounded; + let rejected = compile_candidate( + &dag, + BTreeMap::from([(u64::from(input.0), unbounded)]), + &[u64::from(dag.root.0)], + &[u64::from(build.id.0)], + ); + assert!(rejected.is_err()); + let request = compile_candidate( + &dag, + BTreeMap::from([(u64::from(input.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &[], + ) + .unwrap(); + let mut priced = 0; + let feedback = asap_physical_operators::physical_planner::select_candidate( + vec![ + rejected.map(|candidate| { + ( + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + candidate, + ) + }), + Ok((SummaryMaintenanceLifecycle::Ephemeral, request)), + ], + |_| { + priced += 1; + Ok(Some( + asap_physical_operators::physical_planner::CandidateCost { + workload_scope: "dashboard".into(), + horizon_seconds: 100., + total_cost: 1000., + }, + )) + }, + ) + .unwrap(); + assert_eq!(priced, 1); + assert_eq!(feedback.candidate.0, SummaryMaintenanceLifecycle::Ephemeral); + for revision in [1, 2] { + let rows = (1..=100) + .map(|value| { + schema + .fields + .iter() + .map(|field| match field.dtype { + SummaryFamilyType::Plain(DataType::Float64) => { + Value::Float64(f64::from(value)) + } + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), + _ => panic!("unexpected field {field:?}"), + }) + .collect() + }) + .collect(); + let raw_batch = Batch::try_new(schema.clone(), rows).unwrap(); + let direct = physical_common::execute( + &feedback.candidate.1.query, + BTreeMap::from([(u64::from(input.0), raw_batch.clone())]), + Scope::Query { + evaluation_time_ms: 300_000, + revision, + }, + ); + let state = physical_common::execute( + candidate.precompute.as_ref().unwrap(), + BTreeMap::from([(u64::from(input.0), raw_batch)]), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 300_000, + revision, + }, + ); + let result = physical_common::execute( + &candidate.query, + BTreeMap::from([(u64::from(build.id.0), state[0][0].clone())]), + Scope::Query { + evaluation_time_ms: 300_000, + revision, + }, + ); + let values: Vec<_> = result[0] + .iter() + .flat_map(|batch| batch.rows()) + .flat_map(|row| row.iter()) + .filter_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .collect(); + let direct_values: Vec<_> = direct[0] + .iter() + .flat_map(|batch| batch.rows()) + .flat_map(|row| row.iter()) + .filter_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .collect(); + assert_eq!( + values, direct_values, + "maintenance and request candidates preserve the same population" + ); + assert_eq!(values.len(), 1); + assert!( + (98. ..=100.).contains(&values[0]), + "p99 rank must reflect the supplied population" + ); + } +} + +fn quantile_workload(query: &str) -> PlanningWorkload { + let mut workload = dashboard_workload(); + workload.query_workload.query_batch.as_mut().unwrap()[0].query = Query(query.into()); + workload.query_workload.repeating_queries.as_mut().unwrap()[0].query = Query(query.into()); + workload +} + +/// Timed DAG for `query` after binding every summary state to `lifecycle`. +/// Grouped queries carry a physical series identity, as per-entity state needs. +fn lifecycle_timed_dag( + query: &str, + lifecycle: &SummaryMaintenanceLifecycle, +) -> (asap_types::post_asap::PostAsapDag, Vec) { + use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; + let workload = quantile_workload(query); + let mut lowered = lower_promql_workload(&workload, 0).unwrap().remove(0); + if query.contains(" by(") { + lowered = + asap_physical_operators::physical_planner::promql_rows::with_series_identity(&lowered) + .unwrap(); + } + let root = + selected_plan_for_lowered(&workload, lowered, &FullyCostedRuntime, Horizon(100.)).root; + let candidates = enumerate_summary_maintenance_lifecycles( + root, + WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + NOW_MS, + Some(Horizon(100.)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &FullyCostedRuntime, + ) + .unwrap(); + let choices: Vec<_> = candidates + .deployments() + .iter() + .map(|deployment| (deployment.post_asap_node_id, lifecycle.clone())) + .collect(); + let mut states: Vec<_> = choices.iter().map(|(id, _)| u64::from(id.0)).collect(); + states.sort_unstable(); + let dag = candidates + .select(&choices) + .unwrap() + .execution_timed_dag() + .unwrap(); + (dag, states) +} + +/// Compile inputs for a timed DAG: its raw source, available at either phase. +fn raw_inputs( + dag: &asap_types::post_asap::PostAsapDag, +) -> std::collections::BTreeMap { + let raw = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + asap_types::post_asap::PostAsapOperatorPayload::Fallback { .. } + ) + }) + .unwrap(); + std::collections::BTreeMap::from([( + u64::from(raw.id.0), + asap_physical_operators::physical_planner::InputContract::bounded(std::sync::Arc::new( + raw.output_schema.clone(), + )), + )]) +} + +/// For existing PromQL fixtures, Planner's own retained lifecycle selection +/// reproduces the timing that realization strategies assign today. +#[test] +fn planner_lifecycle_selection_reproduces_strategy_timing() { + for query in [ + "quantile_over_time(0.99, latency[5m])", + "quantile(0.99, latency)", + "sum by(job)(rate(m[1m]))", + ] { + let plan = selected_plan(&quantile_workload(query)); + assert!(!plan.selected_raw_recompute, "{query}"); + assert!(plan.deployments.iter().all(|deployment| { + deployment + .summary_maintenance_lifecycle_guarantee + .as_ref() + .is_some_and(|guarantee| { + guarantee.summary_maintenance_lifecycle + != SummaryMaintenanceLifecycle::Ephemeral + }) + })); + let strategy = asap_types::post_asap::compile_post_asap_dag(&plan.root).unwrap(); + assert_eq!(plan.execution_timed_dag().unwrap(), strategy, "{query}"); + } +} + +/// An explicitly chosen lifecycle reaches physical compilation through timing: +/// ContinuouslyMaintained puts the state in precompute, Ephemeral leaves +/// precompute empty and reads the raw source at query time; both answer alike. +#[test] +fn chosen_lifecycle_timing_decides_precompute_contents() { + use asap_physical_operators::{ + physical_planner::{compile_candidate, frontier_from_timing}, + runtime::Scope, + values::{Batch, Value}, + }; + use asap_types::{post_asap::SummaryFamilyType, pre_asap::DataType}; + use std::collections::BTreeMap; + + let mut answers = Vec::new(); + for lifecycle in [ + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + SummaryMaintenanceLifecycle::Ephemeral, + ] { + let (dag, states) = lifecycle_timed_dag("quantile(0.99, latency)", &lifecycle); + let [state] = states[..] else { + panic!("one summary state"); + }; + let inputs = raw_inputs(&dag); + let (&raw_id, contract) = inputs.iter().next().unwrap(); + let schema = contract.schema.clone(); + let frontier = frontier_from_timing(&dag).unwrap(); + let candidate = + compile_candidate(&dag, inputs, &[u64::from(dag.root.0)], &frontier).unwrap(); + let rows = (1..=100) + .map(|value| { + schema + .fields + .iter() + .map(|field| match field.dtype { + SummaryFamilyType::Plain(DataType::Float64) => { + Value::Float64(f64::from(value)) + } + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), + _ => panic!("unexpected field {field:?}"), + }) + .collect() + }) + .collect(); + let raw_batch = Batch::try_new(schema.clone(), rows).unwrap(); + let query_scope = Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + }; + let result = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { + assert!(frontier.is_empty()); + assert!(candidate.precompute.is_none()); + physical_common::execute( + &candidate.query, + BTreeMap::from([(raw_id, raw_batch)]), + query_scope, + ) + } else { + assert_eq!(frontier, [state]); + assert_eq!( + candidate + .materialized_outputs + .keys() + .copied() + .collect::>(), + [state] + ); + let stored = physical_common::execute( + candidate.precompute.as_ref().unwrap(), + BTreeMap::from([(raw_id, raw_batch)]), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 300_000, + revision: 1, + }, + ); + physical_common::execute( + &candidate.query, + BTreeMap::from([(state, stored[0][0].clone())]), + query_scope, + ) + }; + answers.push( + result[0] + .iter() + .flat_map(|batch| batch.rows()) + .flat_map(|row| row.iter()) + .filter_map(|value| match value { + Value::Float64(value) => Some(*value), + _ => None, + }) + .collect::>(), + ); + } + assert_eq!(answers[0], answers[1]); + assert_eq!(answers[0].len(), 1); +} + +/// One compilation, cut by each lifecycle assignment's timing, yields exactly +/// the candidate `compile_candidate` builds for that timed DAG: the retained +/// state is the frontier under ContinuouslyMaintained, and nothing under +/// Ephemeral. Covers the KLL quantile fixture and grouped Rate→Sum. +#[test] +fn lifecycle_timing_cuts_one_compilation() { + use asap_physical_operators::physical_planner::{ + compile, compile_candidate, cut_candidate, frontier_from_timing, + }; + for query in ["quantile(0.99, latency)", "sum by(job)(rate(m[1m]))"] { + let ephemeral = SummaryMaintenanceLifecycle::Ephemeral; + let (compiled_dag, _) = lifecycle_timed_dag(query, &ephemeral); + let inputs = raw_inputs(&compiled_dag); + let roots = [u64::from(compiled_dag.root.0)]; + let compiled = compile(&compiled_dag, inputs.clone(), &roots).unwrap(); + for lifecycle in [ + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + ephemeral, + ] { + let (dag, states) = lifecycle_timed_dag(query, &lifecycle); + let frontier = frontier_from_timing(&dag).unwrap(); + // Retained states read by a query-time consumer, or the root itself. + let query_time = |id: u64| { + dag.nodes.iter().any(|node| { + u64::from(node.id.0) == id + && node.output_state.timing + == asap_types::post_asap::ExecutionTiming::QueryTime + }) + }; + let expected_frontier = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { + vec![] + } else { + states + .iter() + .copied() + .filter(|state| { + *state == u64::from(dag.root.0) + || dag.edges.iter().any(|edge| { + u64::from(edge.producer.0) == *state + && query_time(u64::from(edge.consumer.0)) + }) + }) + .collect() + }; + assert_eq!(frontier, expected_frontier, "{query} {lifecycle:?}"); + let cut = cut_candidate(&compiled, &frontier).unwrap(); + let expected = compile_candidate(&dag, inputs.clone(), &roots, &frontier).unwrap(); + assert_eq!( + serde_json::to_vec(&cut).unwrap(), + serde_json::to_vec(&expected).unwrap(), + "{query} {lifecycle:?}" + ); + } + } +} + +/// A maintained current-series population is placed by its lifecycle choice: +/// ContinuouslyMaintained stores the population in precompute, Ephemeral +/// rebuilds it from the raw source at query time; both rank alike. +#[test] +fn chosen_population_lifecycle_decides_precompute_contents() { + use asap_aware_mapping::{ + enumerate_summary_maintenance_lifecycles, + maintained_population::MaintainedPopulationStrategy, + }; + use asap_physical_operators::{ + physical_planner::{ + compile_candidate, + promql_rows::{series_row, with_series_identity}, + InputContract, + }, + runtime::Scope, + values::{Batch, Value}, + }; + use asap_types::post_asap::{ + maintained_population::PopulationInput, PostAsapOperatorPayload, ValueOperation, + }; + use std::{collections::BTreeMap, sync::Arc}; + + let workload = quantile_workload("topk by(job)(1, m)"); + let root = Rc::new( + with_series_identity(&lower_promql_workload(&workload, 0).unwrap().remove(0)).unwrap(), + ); + let root = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) + .candidate(&root) + .unwrap(); + let mut answers = Vec::new(); + for lifecycle in [ + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + SummaryMaintenanceLifecycle::Ephemeral, + ] { + let candidates = enumerate_summary_maintenance_lifecycles( + Rc::clone(&root), + WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + NOW_MS, + Some(Horizon(100.)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &FullyCostedRuntime, + ) + .unwrap(); + let [deployment] = candidates.deployments() else { + panic!("one population state"); + }; + let id = deployment.post_asap_node_id; + let dag = candidates + .select(&[(id, lifecycle.clone())]) + .unwrap() + .execution_timed_dag() + .unwrap(); + let population = dag.nodes.iter().find(|node| node.id == id).unwrap(); + let PostAsapOperatorPayload::Value { + operation: ValueOperation::MaintainPopulation { population }, + } = &population.payload + else { + panic!("the deployment is the maintained population"); + }; + let PopulationInput::CurrentSeries(spec) = &population.input else { + panic!("current-series population"); + }; + let lookback = i64::try_from(spec.lookback_ms).unwrap(); + let raw = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .unwrap(); + let (raw_id, schema) = (u64::from(raw.id.0), Arc::new(raw.output_schema.clone())); + let frontier = + asap_physical_operators::physical_planner::frontier_from_timing(&dag).unwrap(); + let candidate = compile_candidate( + &dag, + BTreeMap::from([(raw_id, InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &frontier, + ) + .unwrap(); + let end = 60_000; + let rows = [("a", end - 1, 100.), ("a", end, 1.), ("b", end, 20.)] + .into_iter() + .map(|(instance, at, value)| { + series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("instance".into(), instance.into()), + ]), + at, + value, + ) + .unwrap() + }) + .collect(); + let raw_batch = Batch::try_new(schema.clone(), rows).unwrap(); + let query_scope = Scope::Query { + evaluation_time_ms: end, + revision: 1, + }; + let result = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { + assert!(frontier.is_empty()); + assert!(candidate.precompute.is_none()); + physical_common::execute( + &candidate.query, + BTreeMap::from([(raw_id, raw_batch)]), + query_scope, + ) + } else { + let state = u64::from(id.0); + assert_eq!(frontier, [state]); + let stored = physical_common::execute( + candidate.precompute.as_ref().unwrap(), + BTreeMap::from([(raw_id, raw_batch)]), + Scope::Ingestion { + window_start_ms: end - lookback, + window_end_ms: end, + revision: 1, + }, + ); + physical_common::execute( + &candidate.query, + BTreeMap::from([(state, stored[0][0].clone())]), + query_scope, + ) + }; + answers.push( + result[0] + .iter() + .flat_map(|batch| batch.rows()) + .flat_map(|row| row.iter()) + .filter_map(|value| match value { + Value::Float64(value) => Some(*value), + _ => None, + }) + .collect::>(), + ); + } + assert_eq!(answers[0], answers[1]); + assert_eq!(answers[0], [20.]); +} + +/// Grouped Rate→Sum is one inventory candidate: retaining the Sum state puts +/// Rate and Sum in precompute, while an `Ephemeral` Sum over a retained Rate +/// state leaves Sum in the query DAG. +#[test] +fn grouped_rate_sum_placement_is_a_lifecycle_choice() { + use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; + use asap_physical_operators::physical_planner::{compile_candidate, InputContract}; + use asap_types::post_asap::{ + ExactKind, PostAsapOperatorPayload, SummaryExpr, SummaryFamilyType, + }; + use std::{collections::BTreeMap, sync::Arc}; + + let workload = quantile_workload("sum by(job)(rate(m[1m]))"); + let root = Rc::new( + asap_physical_operators::physical_planner::promql_rows::with_series_identity( + &lower_promql_workload(&workload, 0).unwrap().remove(0), + ) + .unwrap(), + ); + let is_exact = |node: &SummaryNode, kind: ExactKind| { + matches!(&node.expr, SummaryExpr::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(k, _), .. + } if *k == kind) + }; + let inventory = asap_aware_mapping::search_workload(vec![("q", root)]) + .enumerate_candidate_dags(4096) + .unwrap(); + let candidates = inventory + .candidates + .into_iter() + .map(|mut forest| forest.remove(0).1) + .filter(|candidate| { + matches!(&candidate.expr, SummaryExpr::ValueOperation { child, .. } + if is_exact(child, ExactKind::Sum)) + }) + .collect::>(); + let [candidate] = candidates.as_slice() else { + panic!("one grouped Sum candidate, got {}", candidates.len()); + }; + let mut placements = Vec::new(); + for sum_lifecycle in [ + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + SummaryMaintenanceLifecycle::Ephemeral, + ] { + let lifecycles = enumerate_summary_maintenance_lifecycles( + Rc::clone(candidate), + WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + NOW_MS, + Some(Horizon(100.)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &FullyCostedRuntime, + ) + .unwrap(); + let choices = lifecycles + .deployments() + .iter() + .map(|deployment| { + let lifecycle = if is_exact(&deployment.summary, ExactKind::Sum) { + sum_lifecycle.clone() + } else { + SummaryMaintenanceLifecycle::ContinuouslyMaintained + }; + (deployment.post_asap_node_id, lifecycle) + }) + .collect::>(); + assert_eq!(choices.len(), 2, "Rate and Sum states"); + let dag = lifecycles + .select(&choices) + .unwrap() + .execution_timed_dag() + .unwrap(); + let raw = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .unwrap(); + let frontier = + asap_physical_operators::physical_planner::frontier_from_timing(&dag).unwrap(); + let [boundary] = frontier.as_slice() else { + panic!("one precompute output, got {frontier:?}"); + }; + let boundary = dag + .nodes + .iter() + .find(|node| u64::from(node.id.0) == *boundary) + .unwrap(); + let physical = compile_candidate( + &dag, + BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(Arc::new(raw.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &frontier, + ) + .unwrap(); + let json = |value| String::from_utf8(serde_json::to_vec(value).unwrap()).unwrap(); + placements.push(( + boundary.payload.clone(), + json(physical.precompute.as_ref().unwrap()), + json(&physical.query), + )); + } + let builds = |json: &str, kind: &str| { + json.contains(&format!( + r#"{{"SummaryBuild":{{"family":{{"ExactAggregate":["{kind}","{kind}"]}}"# + )) + }; + let [(retained, retained_pre, retained_query), (ephemeral, ephemeral_pre, ephemeral_query)] = + placements.as_slice() + else { + unreachable!() + }; + let state = |payload: &PostAsapOperatorPayload, kind: ExactKind| { + matches!(payload, PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(k, _), .. + } if *k == kind) + }; + assert!(state(retained, ExactKind::Sum)); + assert!(builds(retained_pre, "Rate") && builds(retained_pre, "Sum")); + assert!(!retained_query.contains("SummaryBuild")); + assert!(state(ephemeral, ExactKind::Rate)); + assert!(builds(ephemeral_pre, "Rate") && !builds(ephemeral_pre, "Sum")); + assert!(builds(ephemeral_query, "Sum")); +} diff --git a/crates/types/src/post_asap/execution_data_state.rs b/crates/types/src/post_asap/execution_data_state.rs index 43f2cd05..a5d87c44 100644 --- a/crates/types/src/post_asap/execution_data_state.rs +++ b/crates/types/src/post_asap/execution_data_state.rs @@ -483,14 +483,17 @@ fn visit( operation, timing, } => { + // Population timing is a lifecycle decision: a retained population + // is maintained at ingestion time, an ephemeral one is rebuilt + // from raw input per query. Its input and readout contracts are + // structural and hold either way. let valid_population = match operation { ValueOperation::MaintainPopulation { population } => { - *timing == ExecutionTiming::IngestionTime - && matches!(&child.expr, SummaryExpr::KeepPreAsap(input) if population.matches_input(input)) + matches!(&child.expr, SummaryExpr::KeepPreAsap(input) if population.matches_input(input)) } ValueOperation::ReadPopulation { readout } => { *timing == ExecutionTiming::QueryTime - && matches!(&child.expr, SummaryExpr::ValueOperation { operation: ValueOperation::MaintainPopulation { population }, timing: ExecutionTiming::IngestionTime, .. } if population.supports(readout)) + && matches!(&child.expr, SummaryExpr::ValueOperation { operation: ValueOperation::MaintainPopulation { population }, .. } if population.supports(readout)) } _ => true, }; @@ -507,13 +510,13 @@ fn visit( && s.primitive == DataPrimitive::SummaryState && (*timing == ExecutionTiming::QueryTime || s.timing == *timing) && is_exact_accumulator_state(&child.schema).is_ok(); + // A query-time readout may read a population retained at ingestion. let population_readout = matches!(operation, ValueOperation::ReadPopulation { .. }) && *timing == ExecutionTiming::QueryTime && matches!( &child.expr, SummaryExpr::ValueOperation { operation: ValueOperation::MaintainPopulation { .. }, - timing: ExecutionTiming::IngestionTime, .. } ); diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs index dec0107a..5b63e5a2 100644 --- a/crates/types/src/pre_asap/query_expr.rs +++ b/crates/types/src/pre_asap/query_expr.rs @@ -878,7 +878,8 @@ pub enum QueryExpr { /// unchanged. Wraps the shifted selector directly — `m offset 1h` → /// `TimeShift { Scan }`; a ranged selector `m[5m] offset 1h` → /// `TimeRange { 5m, TimeShift { Scan } }` (the range is taken at the shifted - /// time). Never carries the identity shift (the converter emits a bare + /// time). A shifted subquery wraps the `PromqlSubquery`, moving its step + /// grid. Never carries the identity shift (the converter emits a bare /// selector when neither modifier is present). TimeShift { shift: TimeShift, diff --git a/crates/types/src/pre_asap/schema.rs b/crates/types/src/pre_asap/schema.rs index fbcb9d67..fdac93de 100644 --- a/crates/types/src/pre_asap/schema.rs +++ b/crates/types/src/pre_asap/schema.rs @@ -158,6 +158,92 @@ pub struct Schema { /// label map. `$` cannot occur in a user PromQL label name. pub const PROMQL_SERIES_IDENTITY: &str = "$promql_series_identity"; +/// Resolve a PromQL root to rows carrying [`PROMQL_SERIES_IDENTITY`] before +/// candidate search. `closed` describes physical columns here: the final +/// column contains every dynamic source label. It does not assert that the +/// query's projected labels are the full label set. +/// +/// This realization supports explicit `by` grouping and per-series computation. +/// Operators that rewrite or implicitly match dynamic label sets require their +/// own realization; they must not accidentally treat the opaque identity as a +/// user label or silently discard it. +pub fn with_promql_series_identity(root: &super::QueryExpr) -> Result { + use super::{QueryExpr, Source}; + use std::rc::Rc; + fn scalar_literal(expression: &QueryExpr) -> Option { + match expression { + QueryExpr::PromqlScalarBridge(child) => scalar_literal(child), + QueryExpr::Literal(super::ScalarValue::Float64(value)) => Some(*value), + _ => None, + } + } + let mut root = root.clone(); + fn visit(node: &mut QueryExpr) -> Result<(), String> { + match node { + QueryExpr::Scan { + source: Source::TimeSeries { .. }, + schema, + .. + } => { + if schema + .columns + .iter() + .any(|column| column.name == PROMQL_SERIES_IDENTITY) + { + return Err("source already contains a physical series identity".into()); + } + if schema.closed { + return Err("dynamic series identity requires an open PromQL source".into()); + } + schema + .columns + .push(Column::new(PROMQL_SERIES_IDENTITY, DataType::Utf8, false)); + schema.closed = true; + Ok(()) + } + QueryExpr::TimeRange { child, .. } + | QueryExpr::Limit { child, .. } + | QueryExpr::TimeShift { child, .. } + | QueryExpr::PromqlSubquery { child, .. } + | QueryExpr::PromqlScalarFromVector(child) => visit(Rc::make_mut(child)), + // Constants read no series. + QueryExpr::PromqlScalarBridge(_) => Ok(()), + QueryExpr::PromqlVectorFromScalar(child) if scalar_literal(child).is_some() => Ok(()), + QueryExpr::BinaryOp { + op: super::BinaryOpKind::Arithmetic(_), + lhs, + rhs, + vector_match, + } if vector_match.as_ref().is_none_or(|m| m.grouping.is_none()) + && ![&*lhs, &*rhs].into_iter().any(|side| { + matches!( + side.as_ref(), + QueryExpr::PromqlScalarFromVector(_) | QueryExpr::EvalTimestamp + ) + }) => + { + visit(Rc::make_mut(lhs))?; + visit(Rc::make_mut(rhs)) + } + QueryExpr::Aggregate { child, .. } => visit(Rc::make_mut(child)), + QueryExpr::Sort { + child, + partition_by, + .. + } => { + if partition_by.is_without() { + return Err("dynamic without ranking requires label-set projection".into()); + } + visit(Rc::make_mut(child)) + } + _ => Err("operator has no dynamic series-identity realization".into()), + } + } + visit(&mut root)?; + root.output_schema().map_err(|error| error.to_string())?; + Ok(root) +} + impl Schema { pub fn has_promql_series_identity(&self) -> bool { self.closed diff --git a/docs/design_docs/architecture/input-output-workflow.md b/docs/design_docs/architecture/input-output-workflow.md index fb9a7f42..32c0b093 100644 --- a/docs/design_docs/architecture/input-output-workflow.md +++ b/docs/design_docs/architecture/input-output-workflow.md @@ -34,7 +34,9 @@ fields and [frontend dependencies](#frontend-specific-dependencies). DAG assembly](#selection-and-dag-assembly), and [summary-maintenance lifecycle](#summary-maintenance-lifecycle-aware-helper) APIs operate on this `PlanSpace`. These are alternative uses of the candidate space, not mandatory sequential -stages. `PlanSpace` itself has no selected summary-maintenance lifecycle. +stages. `PlanSpace` itself has no selected summary-maintenance lifecycle, and +its candidates do not choose precompute versus query-time placement: a chosen +lifecycle assignment sets each node's execution timing. The candidate DAGs are logical planning artifacts. ASAPPlanner does **not** produce a deployed executable plan; downstream systems bind physical operators, diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md new file mode 100644 index 00000000..81afc40b --- /dev/null +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -0,0 +1,586 @@ +# Physical Planning, Summary Maintenance, and Deployment + +## 1. Architecture + +A Post-ASAP computation is progressively realized through four layers: + +```mermaid +flowchart LR + L["Logical Post-ASAP DAG
What computation?"] + M["Summary Maintenance Lifecycle
How is state maintained?"] + P["Physical DAG(s)
How is it executed?"] + D["Deployment Plan / DAG
How is it instantiated?"] + + L -->|"Summary Maintenance
Candidate Generation"| M + M -->|"Physical Plan
Compiler"| P + P -->|"Deployment Plan
Compiler"| D +``` + +| Layer | Defines | +| --- | --- | +| **Logical Post-ASAP DAG** | Computation semantics | +| **Summary Maintenance Lifecycle** | Build, retention, reuse, and window strategy | +| **Physical DAG(s)** | Supported physical candidates, executable operators and typed input boundaries | +| **Deployment Plan / DAG** | Selected candidate, concrete data/state bindings and operational lifecycle | + +ASAPPlanner owns the first three layers and the shared physical operator +implementation library. Deployment systems such as ASAPQuery and asap-fusion +own deployment compilation and operation. The lifecycle is a planning contract +associated with the logical DAG, not a separate computation IR. + +The Logical Post-ASAP DAG is preceded by the Pre-ASAP DAG (`QueryExpr`), the +language-independent query semantics before summary selection. Both are +logical. Planning builds Post-ASAP `SummaryNode` trees; `compile_post_asap_dag` +exports the selected tree as a `PostAsapDag`, which is the Physical Plan +Compiler's input. Its per-node execution phase (ingestion or query time) is +decided by the selected summary maintenance lifecycle, as the layer contract +below states. + +### Layer contract + +1. **Logical Post-ASAP** (`PlanSpace`) decides what to compute: summary + families, readouts and sharing. It does not decide placement; timing that a + realization strategy writes while building a candidate is provisional. +2. **Summary maintenance lifecycle** (Planner) lists the lifecycle choices for + each unique retained state: every summary state (`SummaryAgg`) and every + maintained population that does not feed a summary state. + A chosen assignment determines every node's + `ExecutionTiming`, plus window framework and retention. + `SummaryMaintenanceLifecyclePlan::execution_timed_dag` applies it: a retained + (non-`Ephemeral`) state and all of its inputs run at ingestion time; + readouts, other consumers, and `Ephemeral` states not consumed by retained + state run at query time. A population that feeds a summary state is one of + that state's inputs and follows its timing. +3. **Physical compile** (Planner) reads timing: ingestion-time nodes form the + precompute DAG and the rest form the query DAG, joined by typed outputs. It + does not see raw ingestion, panes, storage or stored-state readout. +4. **Backend** chooses the lifecycle assignment with its own `CostModel`: + precompute CPU (`maintenance_cost_per_update`), sketch/summary store cost + (`retention_cost_rate`), query reads (`summary_read_cost`) and per-query + builds (`build_cost`, for `Ephemeral`), counting shared state once. + `Ephemeral` requires the deployment to supply the state's raw input as a + query-time source. + +### Candidate generation and deployment selection + +Planner exposes the supported, semantically legal **physical plan candidates**. +It does not discard a computation family or materialization placement merely +because a deployment-independent cost estimate prefers another candidate. +Logical candidates are an internal search stage, not the deployment handoff. + +```text +Query semantics + accuracy and lifecycle requirements + ↓ Planner +Supported Physical DAG candidates + typed inputs/outputs + requirements + ↓ backend +Binding feasibility + runtime statistics + resource limits + ERP + ↓ backend deployment compiler +Selected PrecomputePlan + QueryPlan + StoredOutputReferences +``` + +Planner owns operators, dependencies, sharing, and each candidate's +materialization frontier. The backend rejects candidates it cannot realize and +prices feasible candidates over a comparable workload and time horizon. It binds +the selected candidate; it does not lower the logical computation again, exchange +operators, or move an operator across the selected frontier. A missing quote is +not a zero-cost implementation. ERP evidence cannot authorize an illegal rewrite. + +The candidate inventory must identify its supported search scope and budget. +If a configured exhaustive enumeration exceeds its budget, planning fails +explicitly instead of selecting from an undisclosed partial inventory. Reports +separate unsupported compilation, deployment infeasibility, missing evidence, +and a feasible candidate that loses on cost. Absence is not a cost comparison. + +For `sum by(job)(rate(m[1m]))`, Rate remains per series before grouped Sum. +`PlanSpace` offers one such candidate, with a per-series Rate state and a grouped +Sum state. Its lifecycle assignment places it: a retained Sum state finalizes +Rate and builds Sum within a bounded precompute run; an `Ephemeral` Sum over a +retained Rate state leaves the Rate readout and Sum in the query DAG. Storing a +value requires its exact evaluation window, revision, readiness and serving +cadence to match the query contract. + +For instant-vector TopK, CMS/CountSketch with a candidate heap requires explicit +series identity and a supported latest-value input protocol. Appending historical +sample values does not preserve instant-vector semantics. Replacement, rank +decrease, expiry, grouping and the required approximation guarantee must be +validated before admitting that physical candidate. + +Planner's candidate space decides what to compute, not placement. For an +instant-vector PromQL TopK, Planner resolves rows that carry the complete series +identity and lists the current-series heap realizations per root with the other +candidates, unranked. Precompute or query-time placement of Rate and grouped Sum +is not a separate Planner candidate: the summary maintenance lifecycle assigns +each node's timing, and the physical compiler reads it. + +This is the target ownership contract. A backend path that still reconstructs +operators from logical candidates has not completed this integration. + +### Input semantics and summary semantics + +`source`, `filter`, `grouping` and `window` describe input-data semantics: +where records originate, which records qualify, how they are grouped and which +time interval applies. They are not a complete description of arbitrary summary +computation. In particular, the same four fields can summarize different value +expressions or produce different states. + +| Concern | Required semantic information | +| --- | --- | +| Input computation | Source identities and schemas, filters, joins/transforms and their order, or a reference to the canonical input sub-DAG | +| Values and grouping | Value expressions, item identities and weights where applicable, group keys and types, and operation-defined null/duplicate handling | +| Time | Time column and interpretation, interval bounds, evaluation alignment, and distinction between query range and maintained panes | +| Summary computation | Exact operation or sketch family, algorithm and parameters, and supported build/merge behavior | +| Output | State versus finalized value, output schema/type, and readout parameters when part of the output computation | + +For example, KLL over `latency_seconds` and KLL over `log(latency_seconds)` differ +even with identical source, filter, grouping and window. Likewise, weighted +frequency state needs both item and weight expressions. More complex inputs +must retain their computation DAG; four descriptive fields cannot replace it. + +The canonical selected computation is authoritative. These categories describe +what must be preserved, not a new flat IR or a second expression language. +Operator-defined behavior should be referenced through its canonical contract, +not independently configured in deployment metadata. Unsupported or unresolved +semantics cannot be treated as compatible. + +Logical planning defines the semantics; physical compilation realizes them as +operators and typed boundaries. Deployment binds concrete readers and state +records that satisfy those requirements. A stored summary definition records or +references the relevant semantics for compatibility checks. Matching a definition +alone does not establish actual window coverage, revision compatibility or +readiness; those require runtime checks. Physical location, encoding, scheduling +and retention are separate execution/deployment contracts. + +Persisted semantic identity, its wire format and any tenant or dataset binding +belong to the deployment. Planner provides the typed `PostAsapDag` that a +deployment canonicalizes; it does not define a stored-definition format. + +### Running example + +Suppose p50 and p99 are requested over the same latency samples in a five-minute window, +and one Planner candidate uses KLL with `k=200`. Assume query windows align with one-minute +pane boundaries and that the selected parameters satisfy the required guarantees. +Operator names below are illustrative; the example defines the design, not a +claim that the entire deployment integration is implemented. + +The data source identifies where samples come from. Filters, grouping and the +window determine which samples enter each summary. Here `pane_duration: 1m` +means each stored pane covers one minute; the query range is five minutes. +Neither duration specifies how often maintenance runs or how long state is kept. + +The example evolves through the architecture as follows: + +```text +1. Logical Post-ASAP DAG + +raw latency + ↓ +KLLBuild(k=200) + ↓ +KLLMerge + ┌─┴─────┐ + ↓ ↓ + p50 p99 + + │ + │ Summary Maintenance Candidate Generation + ▼ + +2. Summary Maintenance Lifecycle + +KLLBuild(k=200) + strategy = continuously maintain + window = 1-minute panes + reuse = p50 + p99 + query = merge panes covering requested aligned 5 minutes + + │ + │ Physical Plan Compiler + ▼ + +3. Physical DAGs + +Precompute DAG: +RawInput + ↓ +NativeKllBuild(k=200) + ↓ +KllStateOutput + +Query DAG: +InputSlot[5 panes] + ↓ +NativeKllMerge(k=200) + ┌─┴────────┐ + ↓ ↓ + NativeP50 NativeP99 + + │ + │ Deployment Plan Compiler + ▼ + +4. Deployment Plan / DAG + +Precompute: +OTLP latency source + ↓ +run KLL build over each complete 1-minute input pane + ↓ +store as latency-kll-1m/ + +Query: +resolve five latency-kll-1m states + ↓ +execute query Physical DAG + ↓ +return p50 / p99 +``` + +Each stage adds a different class of decision while preserving the preceding +contracts. Here, continuous maintenance means recurring production of pane state; +the bounded build DAG does not itself implement an unbounded streaming window. + +## 2. Logical Post-ASAP DAG → Summary Maintenance Lifecycle + +The **Logical Post-ASAP DAG** defines computation semantics: + +```text +Scan(latency) + ↓ +KLLBuild(k=200) + ↓ +KLLMerge + ┌─┴────────────┐ + ↓ ↓ +Quantile(.5) Quantile(.99) +``` + +It establishes that KLL with `k=200` is used and that the merge is shared by the +two readouts. It does not determine when KLL states are built or retained. + +**Summary Maintenance Candidate Generation** enumerates legal lifecycle choices +using workload demand, window/freshness requirements and supported physical +implementations. Backend selection uses runtime feasibility and cost after +physical compilation. The following example follows one candidate. + +Candidate generation and selection are separate steps. For every unique retained +state, enumeration reports each lifecycle (ephemeral, prepared, shared, +continuously maintained) as legal, with a Planner cost or explicitly unknown +cost, or as rejected with a reason. Planner does not remove a legal alternative +because its own estimate prefers another. A deployment prices the legal +alternatives over the whole workload, counting shared state once, and binds one +lifecycle per state. Binding checks that the choice is legal and that states on +one maintenance path share an evaluation schedule. An alternative whose cost is +unknown can be bound only when the deployment's cost model is authoritative for +complete-candidate cost; unknown cost is never treated as zero. It then yields the same +lifecycle guarantee and window framework the physical compiler consumes when +Planner selects. Planner's own cheapest-alternative selection remains available +for callers without deployment pricing. The window framework is decided for the +complete combination, not for one alternative in isolation. + +A maintained population (for example, the current series of `topk by(job)(1, m)`) +is retained state like a summary. Retaining it maintains the latest sample per +series at ingestion and leaves only the readout at query time. Choosing +`Ephemeral` rebuilds that snapshot from raw samples for each query, so the +deployment must supply the raw source at query time. The caller's `CostModel` +prices both through the same lifecycle hooks; a model without population +evidence leaves them unknown, and they are not selected. + +For the running example, assume it selects: + +```text +producer: KLLBuild(k=200) + +strategy: + continuously maintain + +window realization: + 1-minute panes + +query requirement: + combine panes covering the requested aligned 5-minute range + +reuse: + one merged state serves p50 and p99 +``` + +This produces the **Summary Maintenance Lifecycle**. + +The lifecycle specifies how the selected logical summary should be maintained, +but not its concrete operator implementation or storage location. + +Physical feasibility may feed back into selection. For example, if the required +pane-based maintenance cannot be implemented, this lifecycle candidate cannot be +selected. One-minute panes alone also cannot cover an arbitrarily phased query +window; that requires supported boundary handling or a different candidate. + +## 3. Summary Maintenance Lifecycle → Physical DAG + +The **Physical Plan Compiler** consumes both computation semantics and maintenance +requirements: + +```text +Logical Post-ASAP DAG (PostAsapDag) ++ Summary Maintenance Lifecycle ++ physical capabilities + ↓ +Physical Plan Compiler + ↓ +Physical DAG(s) +``` + +For the running example, the lifecycle creates two execution boundaries. + +These two halves are named as `PhysicalCandidate` names them, `precompute` +and `query`. *Maintenance* stays the lifecycle's word (section 2): it covers +how state is built, retained, reused and scheduled. A precompute DAG is the +physical object that a maintenance lifecycle compiles to, so reusing +*maintenance* for it collapses two layers that the crates keep apart: +`asap-aware-mapping::summary_maintenance_*` owns the lifecycle, and +`asap-physical-operators::physical_planner` owns the DAGs. + +### Precompute Physical DAG + +```text +RawInputSlot( + window = 1m, + bounded = true +) + ↓ +NativeKllBuild(k=200) + ↓ +KllStateOutput(k=200) +``` + +This DAG computes the state of one maintained one-minute pane. Its input +contract requires all input samples matching the source, filters and group within that pane; the deployment supplies that +bounded input from its source integration. + +### Query Physical DAG + +```text +InputSlot( + k = 200, + coverage = requested aligned 5m +) + ↓ +NativeKllMerge(k=200) + ┌─┴──────────────────┐ + ↓ ↓ +NativeQuantile(.50) NativeQuantile(.99) +``` + +The Physical Plan Compiler chooses `NativeKllBuild`, `NativeKllMerge`, and the +physical quantile implementations, validates state compatibility, and preserves +the shared merge. It also resolves expressions, schemas, ordered dependencies +and execution properties. + +The resulting Physical DAGs know that compatible KLL states are required, but +do not know where those states are stored. + +For example: + +```text +InputSlot +``` + +is physical, while: + +```text +s3://.../latency-kll/12:01 +``` + +is deployment-specific. Placement and scheduling also remain outside the Physical +DAG. If the required behavior cannot be realized, physical compilation fails. + +### Physical candidates include precompute computation + +Materialization frontiers are Planner decisions. A candidate records both the +precompute Physical DAG and the query Physical DAG, with typed outputs connecting +them. The deployment compiler binds those outputs; it does not move operators. +Lifecycle timing gives the frontier: ingestion-time nodes read by query-time +nodes. For `sum by(job)(rate(m[1m]))`, the two lifecycle choices of the single +logical candidate give: + +```text +Candidate A (Rate state retained, Sum Ephemeral): + precompute: counter samples → per-series Rate state + materialized output: per-series Rate states for window/evaluation/revision + query: stored Rate states → Rate readout → grouped Sum → result + +Candidate B (Rate and Sum states retained): + precompute: counter samples → per-series Rate → grouped Sum state + materialized output: grouped Sum states for window/evaluation/revision + query: stored grouped Sum states → Sum readout → result +``` + +Explicit frontiers passed to `compile_candidates` can also persist per-series +rate values; deriving that frontier from timing inside the physical planner is +not yet implemented. + +Both preserve reset-aware Rate before Sum. Summing raw counters before Rate is +not equivalent. The counter-state build may be another precompute DAG; typed +state inputs do not imply that a deployment can construct or bind those states. + +The shared library exposes `physical_planner::compile_candidates(...)` to lower +explicit frontier candidates to `PhysicalCandidate { precompute, query, +materialized_outputs }`. `select_candidate(...)` accepts deployment feasibility +and scoped complete-workload costs and chooses the lowest-cost feasible +candidate. Costs must describe the same workload and planning horizon; missing +feasibility is rejected before pricing. The optimizer supplies candidate +frontiers and cost evidence, including updates, retention, recurrence and sharing. +`enumerate_frontiers` constructs bounded, reachable antichain frontiers above explicit input boundaries, including query-only and fully precomputed results. It fails explicitly when the candidate budget is exceeded. Maintenance selection must still reject frontiers that violate window, freshness, or reuse requirements; deployment feasibility is checked before pricing. + +The lifecycle layer decides timing; physical compilation reads it. Lowering a +node does not depend on the frontier, so each query DAG is lowered once and +different lifecycle assignments are different cuts of that lowering. +`compile(dag, inputs, roots)` yields the complete `CompiledPhysicalDag`. +`frontier_from_timing(&timed_dag)` reads an assignment's timed DAG (from +`execution_timed_dag`) and returns its frontier: ingestion-time nodes read by +query-time nodes, or an ingestion-time root; a query-time node feeding an +ingestion-time node is rejected. `cut_candidate(&compiled, &frontier)` then +partitions the lowered operators: the frontier's ancestors form the precompute +DAG and the rest form the query DAG. Helper operators are numbered by their +Planner node (`u64::MAX - (node_id << 16) - index`), so a cut is byte-identical to +`compile_candidate` for that frontier. One exception: an ingestion-time +`Binary` lowers differently, so its timing must match at compile time. +`compile_candidate(s)` and `enumerate_frontiers` wrap the same path. Temporal +pane candidates remain a separate lowering. + +Physical compilation opens no readers. Bounded precompute outputs become typed +query inputs. Their source, filters, grouping, build window, evaluation time, readiness and +revision contracts must accompany the selected lifecycle and be checked during +deployment binding. Type compatibility alone does not establish reuse legality. + +The Planner integration test executes both candidates through the shared runtime +and reverses the selected frontier with two controlled cost fixtures. It also +rejects shadowed/duplicate boundaries and incomparable planning horizons. This +establishes Planner capability; it does not establish that ASAPQuery currently +supports persisting every scalar/result-output frontier. + +## 4. Physical DAG → Deployment Plan / DAG + +The **Deployment Plan Compiler** binds the Physical DAGs to the concrete deployment: + +```text +Physical DAGs ++ Summary Maintenance Lifecycle ++ deployment catalog/state ++ sources/materializations ++ operational policy + ↓ +Deployment Plan Compiler + ↓ +Deployment Plan / DAG +``` + +For the precompute DAG, it may produce: + +```text +Source: + RawInputSlot + → complete bounded panes from the OTLP latency source + +Schedule: + each 1-minute pane, once its completion requirements are met + +Execution: + RawInput → NativeKllBuild(k=200) + +Output: + KllStateOutput + → latency-kll-1m/ +``` + +For a query over `(12:00, 12:05]`, its input-binding rule resolves: + +```text +InputSlot[5 panes] + ├── latency-kll-1m/(12:00,12:01] + ├── latency-kll-1m/(12:01,12:02] + ├── latency-kll-1m/(12:02,12:03] + ├── latency-kll-1m/(12:03,12:04] + └── latency-kll-1m/(12:04,12:05] + ↓ + Query Physical DAG + ↓ + p50, p99 +``` + +The Deployment Plan Compiler establishes bindings and checks that their contracts +satisfy the physical inputs and selected lifecycle, including KLL parameters, +source, filters, grouping, window coverage and revision scope. The deployment engine +resolves request-specific states and checks their actual coverage, revisions and +readiness at execution time. A compiled plan cannot establish future readiness. + +The compiler does not replace `NativeKllMerge`, choose another sketch, or decide +to maintain different windows. Such changes require replanning. A Deployment +Plan / DAG is an operational instantiation, not another computation IR. + +## 5. Responsibility Boundary + +The complete example makes the ownership boundary explicit: + +| Stage | KLL example decision | +| --- | --- | +| **Logical Post-ASAP DAG** | Use `KLL(k=200)` with shared merge for p50/p99 | +| **Summary Maintenance Candidate Generation** | Maintain 1-minute panes and reuse them for aligned five-minute queries | +| **Summary Maintenance Lifecycle** | Record pane/window/freshness/reuse requirements and each node's execution timing | +| **Physical Plan Compiler** | Lower to native KLL build, merge, and readout operators | +| **Physical DAG** | Define precompute and query DAGs with typed input/output boundaries | +| **Deployment Plan Compiler** | Bind raw input and KLL state slots to concrete sources/materializations | +| **Deployment Plan / DAG** | Specify maintenance schedules, stored-pane resolution and query execution | + +```text +Logical: + "Use KLL for p50/p99." + +Lifecycle: + "Maintain reusable 1-minute KLL panes." + +Physical: + "Execute NativeKllBuild and + NativeKllMerge → {p50, p99}." + +Deployment: + "Read OTLP here, store panes here, + and bind these five panes for this aligned query." +``` + +The deployment engine executes the bound Physical DAGs through ASAPPlanner's +shared physical operator implementation library, `asap-physical-operators`, and +its DAG runtime. The merge executes once per run for both consumers. Execution +does not introduce additional planning decisions. + +The physical layer does not own raw ingestion, pane construction or geometry, +storage formats, or decoding persisted bytes into typed state. It compiles +computation over typed input contracts: summary build, merge (for example KLL +merge), sketch estimates and exact finalization. The deployment constructs panes, +reads and decodes stored state, and binds the typed values to input slots. +Compiled physical plans are Planner outputs and keep their own serialized form. + +Each maintained pane contributes its input samples once. A replacement snapshot +replaces that pane's state; query merging must not count both the old and new +snapshots as separate inputs. + +## 6. Executable acceptance coverage + +The tests cover optimizer-selected lifecycle execution alongside independent +operator/runtime fixtures: + +| Test | Contract exercised | +| --- | --- | +| `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled precompute/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | +| `summary_maintenance_lifecycle_e2e::chosen_lifecycle_timing_decides_precompute_contents` | PromQL workload → enumerated lifecycles → explicit choice → timed DAG → compiled candidate; ContinuouslyMaintained stores the state in precompute, Ephemeral leaves precompute empty and reads the raw source at query time; both return the same p99 | +| `summary_maintenance_lifecycle_e2e::lifecycle_timing_cuts_one_compilation` | KLL quantile and grouped Rate→Sum: one compilation cut by the ContinuouslyMaintained and Ephemeral timed DAGs equals `compile_candidate` for each; the frontier is the retained state or empty | + +| `summary_maintenance_lifecycle_e2e::chosen_population_lifecycle_decides_precompute_contents` | PromQL `topk by(job)` over a maintained population → explicit choice → timed DAG → compiled candidate; ContinuouslyMaintained stores the population in precompute, Ephemeral rebuilds it from raw samples at query time; both rank alike | +| `summary_maintenance_lifecycle_e2e::planner_lifecycle_selection_reproduces_strategy_timing` | For PromQL summary fixtures, the timed DAG from Planner's retained selection equals the DAG realization strategies produce | +| `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute precompute DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | +| `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | +| `precompute_candidates::grouped_rate_can_be_materialized_before_or_after_grouped_sum` | Cost changes select different legal precompute frontiers; both selected candidates execute with the same reset-sensitive result; uncompilable candidates are not priced | +| `sql_to_physical::sql_filter_grouped_sum_executes_and_rebinds` | SQL text → candidate search → physical compilation → shared Scan predicates and grouped summary execution; NULL samples are ignored and fresh bindings produce new results | + +Pane construction, pane timestamp checks, stored identity, revisions, readiness +and complete coverage of required input samples are deployment +responsibilities. Real storage and HTTP execution belong to +deployment-repository E2E tests. diff --git a/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md b/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md index 99501e97..06555965 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md +++ b/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md @@ -70,7 +70,7 @@ columns and multi-measure aggregates need additional rules. ```text KeepPreAsap(input) - -> MaintainPopulation { input, max_k, quantiles } [maintenance] + -> MaintainPopulation { input, max_k, quantiles } [lifecycle-timed] -> ReadPopulation { Quantile(q1) } [read] -> ReadPopulation { Quantile(q2) } [read] -> ReadPopulation { TopK(k1) } [read] @@ -115,8 +115,9 @@ because their source names or numeric values happen to agree. ## Validation, selection and execution responsibilities -Planner validates the declared input, maintenance/read phases and readout -compatibility. Its intended guarantee is exact membership and exact readout; +Planner validates the declared input, the query-time readout and readout +compatibility; the population's lifecycle decides whether it is maintained at +ingestion or rebuilt per query. Its intended guarantee is exact membership and exact readout; a physical implementation still must preserve the language's numeric and empty-input semantics. In particular, SQL global COUNT over an empty population returns a row with zero, while PromQL COUNT over an empty vector returns an empty vector. diff --git a/docs/design_docs/proposals/operator-sharing.md b/docs/design_docs/proposals/operator-sharing.md index 03b36cf5..3d075105 100644 --- a/docs/design_docs/proposals/operator-sharing.md +++ b/docs/design_docs/proposals/operator-sharing.md @@ -287,7 +287,7 @@ Per node kind: | `SummaryAgg` | from the child; ingestion time under `KeepPreAsap` | **set** by binding, as today's fallback: `IngestionTime`, or `QueryTime` over a query-time child | | `FinalizeExactAccumulator` | a stored field, set by the planner | **set** by the planner: the same position allows either time | | `SummaryEstimate` | query time, fixed by the kind | **derived** from the kind: query time | -| `MaintainPopulation` / `ReadPopulation` | a stored field, always ingestion / query time | **derived** from the kind: ingestion / query time | +| `MaintainPopulation` / `ReadPopulation` | a stored field: population timing set by its lifecycle; readout always query time | population: **set** by its lifecycle; readout: **derived**, query time | | unused variants | `SummaryMerge`: a stored field; `Join` / `Subtract` / `Delete`: ingestion time | unimplemented (§1.3) | Unlike a guarantee, a timing depends on the parents, so `derive_timings` needs the whole diff --git a/docs/develop_docs/README.md b/docs/develop_docs/README.md index 2593f3f5..817722ad 100644 --- a/docs/develop_docs/README.md +++ b/docs/develop_docs/README.md @@ -15,5 +15,6 @@ formats, evidence, and verification workflows. - [Metrics-observability corpora](metrics-observability-corpora.md) - [Physical handoff cost references](physical-handoff-costs.md), [storage operations](storage-operation-costs.md) - [Replacement explanations](replacement-explanations.md) +- [Physical compile coverage for deployment computation](physical-compile-coverage.md) - [Planner vocabulary migration (#427)](planner-vocabulary-migration.md) diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index bfeea034..a4119122 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -267,6 +267,28 @@ they are not necessarily a globally sortable physical-cost scalar. Unavailable cost alternatives may remain for explanation. Inspect eligibility and evidence before physical selection; do not treat their presence as deployment permission. +### Enumerate candidate DAGs per root + +```text +PlanSpace::enumerate_candidate_dags_for_root(&self, id: &Id, expansion_limit: usize) + -> Result, RealizationError> +``` + +Returns every distinct finalized DAG for one root, unranked; other roots' +choices are not multiplied in. Exceeding `expansion_limit` is an error, never a +partial inventory. + +For PromQL roots that carry a target, `search_workload_with_targets` also asks +each strategy's `ReplacementStrategy::propose_for_root`. `SketchAlgorithmStrategy` +answers an instant-vector TopK with current-series heap realizations over rows +carrying the complete series identity (`$promql_series_identity`). They are +finalized, deduplicated, and marked `ReplacementProvenance::RootPhysicalRealization`. +Callers do not apply `with_series_identity` themselves. Compile each with +`promql_rows::compile_current_series_readout`; other queries keep their previous +inventory. `global_selection` never commits these candidates; the backend +compiles and prices them. PlanSpace lists no placement variants: node timing +comes from the summary maintenance lifecycle. + ## Choose strategies and models ### Strategy options @@ -611,6 +633,8 @@ that prepared or retained shared state is supported. | `plan_summary_maintenance_lifecycles` | Assembled logical DAG root, `WorkloadDemand`, `now_ms`, optional horizon, runtime capabilities, cost model | `Result` for that fixed root; does not revisit all semantic candidates | | `global_selection_with_summary_maintenance_lifecycles` | `PlanSpace`, workload/root-entry associations, time, horizon, capabilities, cost model | Lifecycle-aware compatible selection/error, using eligible cost evidence | | `assemble_selected_dag_with_summary_maintenance_lifecycles` | Selection, target root and lifecycle context | Optional lifecycle plan/error; attaches state deployment decisions | +| `enumerate_summary_maintenance_lifecycles` | Same inputs as `plan_summary_maintenance_lifecycles` | `SummaryMaintenanceLifecycleCandidates`: per unique retained state, every alternative with its cost or rejection; nothing selected. `guarantee(&lifecycle)` gives the mode/schedule that alternative would carry | +| `SummaryMaintenanceLifecycleCandidates::select(choices)` | One `(PostAsapNodeId, SummaryMaintenanceLifecycle)` per state, copied from `deployments()` | The same `SummaryMaintenanceLifecyclePlan` Planner selection would produce for that combination, or `SummaryMaintenanceLifecycleChoiceError` when a choice is unknown, missing, duplicated, rejected, schedule-incompatible, or not completely estimable | Inspect `deployments`, their selected lifecycle/alternatives/rejections, `selected_raw_recompute`, and optional summary/raw costs. Success of a function @@ -622,6 +646,52 @@ lifecycle analysis after structural selection can evaluate the selected root, but does not make the earlier selection lifecycle-optimal. An application may consume ranked candidates and perform this comparison downstream instead. +A deployment that prices lifecycles itself calls +`enumerate_summary_maintenance_lifecycles`, prices the alternatives, and binds +its choice with `select`. A choice is accepted only if Planner could select it: +an alternative with `MissingCostEvidence` is accepted only when the cost model's +complete-candidate hook covers lifecycle costs. Window frameworks and totals come +from that hook, as in Planner selection. + +A lifecycle choice then fixes each physical placement through timing: a +continuously maintained state and its inputs run at ingestion time, while an +ephemeral one stays at query time. Compile each query's `PostAsapDag` once and +cut every chosen assignment from that result: + +```rust +use asap_physical_operators::physical_planner::{ + compile, cut_candidate, frontier_from_timing, +}; + +let compiled = compile(&dag, inputs, &roots)?; // each node lowered once +for plan in lifecycle_plans { + let frontier = frontier_from_timing(&plan.execution_timed_dag()?)?; + // Precompute/query DAGs split at `frontier`; no logical lowering. + let candidate = cut_candidate(&compiled, &frontier)?; + // Check feasibility and price `candidate`; bind the selected one as is. +} +``` + +The frontier is the set of ingestion-time nodes read by query-time nodes (or an +ingestion-time root). `frontier_from_timing` rejects a query-time node feeding +an ingestion-time node. `cut_candidate` returns exactly what +`compile_candidate(&dag, inputs, &roots, &frontier)` returns and rejects the +same invalid frontiers. If the DAG has an ingestion-time `Binary`, compile with +the same timing for that node, because it lowers differently. Temporal pane +candidates are a different lowering and still use +`compile_temporal_pane_candidate`. + +Retained states are `SummaryAgg` nodes and `MaintainPopulation` nodes that do +not feed a `SummaryAgg`; a population that does feed one is part of that +state's input. The lifecycle cost hooks (`summary_maintenance_capabilities`, +`summary_maintenance_lifecycle_cost_inputs_for_horizon`) and the complete-candidate +hook therefore also receive `MaintainPopulation` nodes. A model that does not +recognize one should return unknown costs, which keep its alternatives +unselected; a model that prices every node uniformly now also prices +populations, so population candidates can win lifecycle-aware selection. `SummaryMaintenanceLifecyclePlan::execution_timed_dag` times a +population as it times a summary state: retained at ingestion, `Ephemeral` at +query time from the raw source. + ## Optional whole-plan selection and DAG assembly ### What does global selection mean? diff --git a/docs/develop_docs/native-promql-inputs.md b/docs/develop_docs/native-promql-inputs.md new file mode 100644 index 00000000..aa1a52b6 --- /dev/null +++ b/docs/develop_docs/native-promql-inputs.md @@ -0,0 +1,55 @@ +# Native PromQL source rows + +Audience: source-adapter and physical-executor developers. + +A PromQL query only names some labels. Those columns cannot establish series +identity for Rate or TopK: two series with the same `job` may have different +unreferenced instance labels. + +`physical_planner::promql_rows::with_series_identity` resolves supported unary +PromQL computations to a bounded row representation before candidate search. +It appends `$promql_series_identity`, a non-null UTF-8 column containing the +canonical JSON encoding of the full label map. The name cannot collide with a +legal PromQL label. The resulting schema is closed over physical columns; the +label map remains dynamic and is not restricted to labels named in the query. +This realization rejects unsupported label rewriting, implicit vector matching, +and `without` operations rather than dropping hidden labels. + +Source adapters construct batches with `series_row`. Named label columns are +projections of the same complete identity; absent named labels project to empty +strings. `decode_series_identity` restores all labels on result conversion and +rejects noncanonical encodings. A query adapter must still apply the selected +operator's metric-name/result-label rules. Source selection, complete window +coverage and revision admission remain deployment responsibilities. + +Planner's maintained-population candidate recognizes this explicit identity +representation. Its TopK readout compiles automatically to `CurrentSeries`, +`Sort`, and `Limit`; deployment supplies the raw boundary or an already maintained +population boundary. Compilation does not open either source. + +The native `CurrentSeries` operator selects the latest sample per complete +identity in `(evaluation_time - lookback, evaluation_time]`. It removes stale +markers after selecting the latest sample, so an older value cannot reappear. +It rejects conflicting values at one series timestamp and emits the evaluation +timestamp. Each run builds a new snapshot; decreased values and expired series +cannot retain earlier heap weights. It reserves workspace and observes the +run's cancellation and byte budget. Precompute scopes must match the declared +lookback before any input is polled. + +CMS/CountSketch heap operators can consume this snapshot. CMS still requires +nonnegative weights; legal approximate TopK admission still requires the +Planner's accuracy/membership evidence. Executing a heap does not establish +that its result satisfies a query's accuracy requirements. + +Tests cover open-label Rate → CMS/CountSketch heaps, hidden-label round trips, +reset and zero-rate cases, snapshot replacement/decrease/expiry/staleness, +serialized physical recovery, and resource rejection. These are shared-library +tests, not proof of Backend candidate selection or durable deployment execution. + +Spatial heap candidates use the same complete series identity. Planner's +`current_series_topk_candidates` explores a CountSketch-with-heap realization +of canonical Sort/Limit under an explicit accuracy target. The physical graph +selects the latest eligible samples before building a fresh heap. A maintained +population boundary can supply that snapshot directly. Arbitrary signed metric +values do not authorize CMS; counter Rate's non-negative proof is separate. +These candidates still require membership/score evidence for deployment admission. diff --git a/docs/develop_docs/physical-compile-coverage.md b/docs/develop_docs/physical-compile-coverage.md new file mode 100644 index 00000000..31c69782 --- /dev/null +++ b/docs/develop_docs/physical-compile-coverage.md @@ -0,0 +1,166 @@ +# Physical compile coverage for deployment computation + +Audience: developers moving computation from ASAPQuery-backend into +`asap_physical_operators::physical_planner`. + +## Contract + +Logical selection decides what to compute. The maintenance lifecycle sets node +timing. `physical_planner::compile` turns a timed `PostAsapDag` into physical +operator DAGs. The backend owns ingestion, panes, storage, stored-state +readout, external exact engines, pricing/selection, and execution scheduling. + +A backend lowering is *covered* when `compile` accepts the corresponding +`PostAsapDag` node and produces operators with the same result. The backend +should then pass the timed DAG and its input contracts to `compile`. It should +not rebuild operator choices from PromQL text or construct operators itself. + +## Inventory + +Surveyed backend: `ASAPQuery-backend` branch `perf/788-startup-search`. +Planner base: `split/462-f-physical-planner` (#475). + +Status values: + +- **Supported**: `compile` or `compile_node` already covers this computation. +- **Partial**: some shapes are covered. The Notes column lists the gap. +- **Missing**: `compile` rejects this computation. +- **Backend**: not computation, or owned by the backend. + +| # | Backend site | Computation | Planner node | Status at #475 | Notes | +|---|---|---|---|---|---| +| 1 | `query_time.rs` `Lower::lower`, `compile_logical` | PromQL AST → `QueryTimeOperator` graph for a native query | `Fallback { QueryExpr }` subtrees plus value payloads | Missing | `compile` lowers `Fallback` only as a raw `Scan` source. | +| 2 | `QueryTimeOperator::Aggregate` (sum/min/max/avg/count) | Grouped value aggregation | `Value::Exact(Aggregate)`; `SummaryAgg{ExactAggregate, Reduce}` over finalized values | Supported | Also `promql_values::compile_aggregate`. | +| 3 | `QueryTimeOperator::Sort`, `Limit` (topk, sort, sort_desc) | Ordering and per-group limits | `Value::Sort`, `Value::Limit` | Supported | | +| 4 | `QueryTimeOperator::Binary`, `QueryPlanNode::Binary` (vector ⊗ scalar) | Arithmetic with a scalar operand | `Binary` whose operand is `Fallback{PromqlScalarBridge(Literal)}` | Missing | Query-time `Binary` accepts only label-map vector schemas. The literal node has no native binding. | +| 5 | `QueryTimeOperator::Binary` (vector ⊗ vector) | One-to-one label matching and arithmetic | `Binary` over grouped value rows | Missing | Only the ingestion-time `aligned_binary` and label-map `vector_binary` exist. | +| 6 | `binary_operator` CheckedDiv / FiniteDiv | Guarded division | `BinaryOperator` checked flags | Partial | Flags are evaluated, but only where rows 4/5 are covered. | +| 7 | `QueryTimeOperator::Binary` comparisons, `bool` | Filter or 0/1 comparison | `Binary{Compare}` | Missing | `Payload::Binary` does not carry `return_bool`. | +| 8 | `QueryTimeOperator::UnaryNegate` | Negation | `Binary{Mul}` by literal `-1` | Missing | The frontend emits `* -1`; same gap as row 4. | +| 9 | `QueryTimeOperator::VectorToScalar` | `scalar()` | `Fallback{PromqlScalarFromVector}` | Missing | Only `promql_values::compile_vector_to_scalar`. | +| 10 | `QueryTimeOperator::HistogramQuantile` | Bucket interpolation | `Fallback` / `AggIntent::HistogramQuantile` | Missing | Only `promql_values::compile_histogram_quantile`. | +| 11 | `QueryTimeOperator::Temporal` (rate, increase, `*_over_time`) | Per-series window functions | `SummaryAgg{PerEntity}` over `TimeRange(Scan)` | Partial | Supported with closed series identity. Not supported over `Fallback` matrices (`compile_temporal` only). | +| 12 | `logical_dag.rs` `Subquery`, `subquery_grid`, `expanded_inputs` | Re-evaluate the child on a step grid and assemble a matrix | `Fallback{PromqlSubquery}` | Missing | No Planner operator. | +| 13 | `QueryPlanNode::Scalar`, `DagCompiler::lower` scalar literal | Scalar constant | `Fallback{PromqlScalarBridge(Literal)}` | Missing | Only `promql_values::compile_scalar`. | +| 14 | `DagCompiler::lower` `ReduceSum`; `physical_values.rs` PerEntity projection | Sum over finalized values; per-entity identity | `SummaryAgg{ExactAggregate(Sum)}` | Supported | The backend builds an identity `Operator::project` itself for PerEntity. | +| 15 | `DagCompiler::lower` `ExactReadout`; `post_asap_readout.rs` ExactReadout | Finalize exact state (sum/count/min/max/rate/increase) | `Value::FinalizeExactAccumulator` | Partial | Count yields Int64 against a declared Float64 PromQL value. `compile` rejects it. | +| 16 | `post_asap_readout.rs` SummaryEstimate (`readout_bound`, `expand_item_rows`) | Sketch estimate per group; TopK item expansion | `SummaryEstimate` | Partial | The backend's label-map state layout and MetricsQL `__name__` rules have no Planner equivalent. `compile_exact_readout` has no sketch counterpart. | +| 17 | `post_asap_readout.rs` SummaryMerge (`merge_bound_states`) | Merge states by group | `SummaryMerge` | Supported | Union plus `summary_merge`. | +| 18 | `post_asap_readout.rs` counter range parameters | Counter lookback for rate/increase | `TimeRange` ancestor of finalization | Supported | Applied through `with_counter_lookback`. | +| 19 | `post_asap_readout.rs` `execute_value_fragment` | Per-timestamp binding of a value fragment | n/a | Backend | Evaluation scheduling. | +| 20 | `DagCompiler::lower` SummaryJoin / Subtract / Delete | Summary algebra | `SummaryJoin`, `SummarySubtract`, `SummaryDelete` | Missing | The backend also rejects these (`ExactFallback`). | +| 21 | `current_series.rs` Snapshot + TopK | Current-series ranking | `ReadPopulation{TopK}` | Supported | | +| 22 | `current_series.rs` Sum / Count / Average | Current-series aggregates | `ReadPopulation{Sum,Count,Average}` | Missing | `compile` accepts only TopK. | +| 23 | `current_series.rs` Quantile | Current-series quantile | `ReadPopulation{Quantile}` | Missing | No exact quantile reduction. | +| 24 | `raw_dag.rs` weight `Column` | Summary update from a sample/projected value | `SummaryAgg` | Supported | | +| 25 | `raw_dag.rs` weight `Constant` | Unit/constant-weight update | `SummaryAgg` | Missing | `compile_node` requires a column weight. | +| 26 | `raw_dag.rs` item `Column` / `Tuple` | Keyed update item | `SummaryAgg{item}` | Supported | `keyed_summary_build`. | +| 27 | `raw_dag.rs` item `EntityIdentity` | Series-identity item | `SummaryAgg{item}` | Missing | Needs the series-identity column. | +| 28 | `physical_values.rs` `compile`, `combine` | Translate `QueryTimeOperator` to `promql_values::*`; compose fragments | n/a | Supported | Exists only because of row 1. `CompiledPhysicalDag::compose` is Planner API. | +| 29 | `query_plan.rs` `compile_native_fragment` (Semi join, Exact aggregate, Sort, Limit, Filter) | Relational value ops | `RelationalJoin`, `Value::*` | Supported | Already calls `compile`. | +| 30 | `query_time.rs` `selected_query_time_nodes`, `selected_native_expression`, `selected_aggregate_operator` | Recover operator identity from original PromQL text | Payload variants (`ExactKind::Min`/`Max`, `AggIntent`) | Supported | Payloads already carry the identity. These witnesses are needed only while row 1 remains. | +| 31 | Scan, ExactSubquery, CandidateExactSubquery, CurrentSeries ingest, ReadMaterialization, ExternalExact | Storage reads and external engines | Input contracts | Backend | | + +Totals at #475: 11 Supported, 4 Partial, 14 Missing, 2 Backend. + +## Covered after this change + +| Row | Change | +|---|---| +| 4, 8, 13 | Query-time `Binary` folds a scalar-literal operand into a projection over grouped value rows. | +| 5 | Query-time `Binary` over grouped value rows performs an inner equi-join on equal label columns, then applies the operator. Per-series rows remain Partial. | +| 15 | Count finalization converts exactly to the declared Float64 value. | +| 22, 23 | `ReadPopulation` Sum/Count/Average/Quantile compile to grouped aggregation. `Reduction::Quantile` implements PromQL interpolation. | + +Totals after this change: 17 Supported, 4 Partial, 8 Missing, 2 Backend. + +## Covered by PromQL fallback compilation + +`compile` now lowers a `Fallback{QueryExpr}` node from its typed expression, +realized with `promql_rows::with_series_identity`. The deployment supplies the +raw rows of the `i`th selector returned by `promql_fallback::raw_series` at +`promql_fallback::raw_series_input(node, i)`, with that selector's schema. Each +selector has its own slot, even when two selectors read the same metric. The +node's own ID still names its output, so a deployment may instead supply the +whole result, for example from an external exact engine. A Fallback that reads +no selector, such as `vector(1)`, needs no input. + +`Operator::series_window` evaluates each series at the query time, or at each +subquery step, over the left-open window `(t - offset - range, t - offset]`. +Instant selection takes the latest sample and omits the series if that sample is +a stale marker. Range functions ignore stale markers. The raw rows must cover +every window the node evaluates; `raw_series` documents the subquery extent. +A subquery has at most 100000 steps. Output rows keep the full series identity; +the query adapter still applies PromQL's metric-name rules. A bare selector +consumed by another node, such as a per-entity summary, remains raw range rows +and is not compiled as instant selection. + +| Row | Change | +|---|---| +| 1 | Supported shapes: selectors with `offset`; `rate`, `increase`, `delta`, and `sum`/`avg`/`min`/`max`/`count_over_time`; `by` aggregation; `sort`; `topk`/`limit`; `scalar()`; `vector(literal)`; arithmetic with one literal. Now Partial. | +| 9 | `scalar()` compiles to `VectorToScalar`. | +| 11 | Range functions over the raw selector rows. | +| 12 | `f(sel[R:S])` and `f(g(sel[r])[R:S])`, with subquery `offset`, evaluate on the aligned step grid. The frontend now retains subquery `offset`/`@` as a `TimeShift`; it previously dropped them. Now Partial. | + +Totals after this change: 19 Supported, 5 Partial, 5 Missing, 2 Backend. + +## Covered by multi-selector fallback compilation + +| Row | Change | +|---|---| +| 1 | Vector-vector arithmetic with one-to-one matching, `on`, and `ignoring`. Each side is reduced to its matching labels, then matched on equal label sets. A duplicate match group is an error, as in Prometheus. The result drops `__name__`. `without` aggregation. `irate`, `idelta`, `changes`, `resets`, `last_over_time`, and exact `quantile_over_time`. `@ ` on selectors. Still Partial. | +| 11 | The range functions above, over raw selector rows. | +| 12 | `@ ` on subqueries anchors the step grid. An inner selector's `@` pins every step. Still Partial. | + +`@` fixes the instant a window ends at, before `offset`; the output keeps the +query's evaluation time. The raw rows must cover the window at that instant. +Vector matching and `without` rewrite the series identity, so their results +already lack `__name__`. Other results keep it; the query adapter still drops +it. + +Totals are unchanged: 19 Supported, 5 Partial, 5 Missing, 2 Backend. + +## Remaining + +In order of backend usage: + +1. Rows 1, 10, and 12, the remaining `Fallback` shapes: + - `histogram_quantile` (row 10). The compiler cannot recover the grouping + from today's IR, which is `Aggregate{Reduce(by ()), [HistogramQuantile{q}]}`. + It computes one quantile over all buckets and drops the output labels. The IR + must carry: + - The bucket label: the `le` column of the child's schema, named by the + intent, for example `HistogramQuantile{q, le: C}`. The frontend must + add `le` to the selector's schema even when no matcher names it. + - The grouping: `Reduce(without([le]))`, so that each histogram is + one set of series that differ only in `le`. An explicit + `sum by (x, le)` inside the argument still yields `without (le)` over + those rows. + - The output labels: every input label except `le` and `__name__`. The + `without` output schema already carries them, and the series identity + with `le` removed. `series_labels(Ignoring, [le])` computes the latter. + The operator then applies Prometheus `bucketQuantile`. It parses `le` + as a float, skips unparsable values, and requires a `+Inf` bucket, else + it returns NaN. It forces cumulative counts to be monotonic and returns + NaN for fewer than two buckets. For q < 0 it returns -Inf; for q > 1, + +Inf. + - Comparisons and set operators (`and`, `or`, `unless`); `group_left` and + `group_right`; arithmetic with a non-literal scalar, such as + `scalar(x)` or `time()`. + - Subquery operands other than one per-series function; implicit + subquery resolution, which is a deployment default. + - `@ start()` and `@ end()`, which need the range query's bounds in the + run scope. + - Other functions, such as `deriv`, `predict_linear`, + `stddev_over_time`, `absent`, `label_replace`, and math functions. + After these shapes are covered, the backend can delete rows 28 and 30. +2. Row 7: comparison filters and `bool` comparisons. This needs `return_bool` + in the `Binary` payload. `compile` currently rejects comparisons. +3. Row 5 for per-series rows in a `Binary` payload node: its grouped-row join + still rejects `$promql_series_identity`. It could reuse the Fallback's + `series_labels` and `series_binary` operators. +4. Rows 25 and 27: constant weights and `EntityIdentity` items for precompute + `SummaryAgg`. +5. Row 16: a label-map sketch-state readout, the counterpart of + `compile_exact_readout`, and MetricsQL `__name__` retention rules. +6. Row 20: summary join, subtract, and delete.