diff --git a/crates/devtools/src/bin/stage_pipeline.rs b/crates/devtools/src/bin/stage_pipeline.rs index 7546a620..63932cbd 100644 --- a/crates/devtools/src/bin/stage_pipeline.rs +++ b/crates/devtools/src/bin/stage_pipeline.rs @@ -46,7 +46,7 @@ use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataArrival, DataDistribution, DataWorkload, DurationMs, Evidence, EvidenceSource, LatencyRequirement, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRecurrence, QueryRequirements, QueryTimeScope, QueryWorkload, Rate, - RepeatedDemand, RepeatingEntry, RepetitionInterval, TimeSelection, + RepeatedDemand, RepeatingEntry, RepetitionInterval, RootDemand, TimeSelection, }; use serde_json::{json, Value}; @@ -115,15 +115,15 @@ fn stage_pipeline(workload: &PlanningWorkload, max_candidates: usize) -> Result< }) .collect::, _>>()?; let stage0 = export(&roots)?; - let targets: Vec<_> = workload + let demand: Vec = workload .query_workload .entries() - .map(|entry| Some(entry.requirements.accuracy.target())) + .map(|entry| RootDemand::from(&entry)) .collect(); let data = workload.data_workload.clone().unwrap_or_default(); let run = plan_stages( roots.into_iter().enumerate().collect(), - &targets, + &demand, &data, PlanningModels::builtin(), max_candidates.max(1), diff --git a/crates/integration-tests/tests/planner_layering_example1.rs b/crates/integration-tests/tests/planner_layering_example1.rs index 0d9811b7..9a303262 100644 --- a/crates/integration-tests/tests/planner_layering_example1.rs +++ b/crates/integration-tests/tests/planner_layering_example1.rs @@ -144,14 +144,14 @@ mod stages { workload: &PlanningWorkload, _logical: &LogicalCandidate, ) -> Vec { - let targets: Vec<_> = workload + let demand: Vec<_> = workload .query_workload .entries() - .map(|entry| Some(entry.requirements.accuracy.target())) + .map(|entry| asap_types::workload::RootDemand::from(&entry)) .collect(); let run = plan_stages( lower(workload).into_iter().enumerate().collect(), - &targets, + &demand, workload.data_workload.as_ref().expect("data workload"), PlanningModels::builtin(), MAX_ENUMERATED_CANDIDATES, @@ -211,15 +211,15 @@ mod stages { physical: &[PhysicalCandidate], models: PlanningModels<'_>, ) -> Selection { - let targets: Vec<_> = workload + let demand: Vec<_> = workload .query_workload .entries() - .map(|entry| Some(entry.requirements.accuracy.target())) + .map(|entry| asap_types::workload::RootDemand::from(&entry)) .collect(); let candidates: Vec<_> = physical.iter().map(|p| p.stage2.clone()).collect(); let selection = asap_plan_selection::stage3_select( &candidates, - &targets, + &demand, workload.data_workload.as_ref().expect("data workload"), models, ) @@ -995,6 +995,34 @@ fn stage3_selects_cheapest_valid() { } } +/// Per-second cost keeps Example 1's ranking: both panels repeat every +/// 10 s and everything runs at query time, so every candidate costs 0.1 × +/// its per-evaluation cost, and P58 still wins at 52.201 × 0.1 per second. +#[test] +fn stage3_per_second_cost_keeps_the_ranking() { + let (workload, _, physical) = pipeline(); + let per_second = stage3_select(&workload, &physical, PlanningModels::builtin()); + // Evaluated once per second, a candidate's cost is its per-evaluation cost. + let mut every_second = workload.clone(); + for entry in every_second + .query_workload + .repeating_queries + .iter_mut() + .flatten() + { + entry.demand = RepeatedDemand::FixedInterval(RepetitionInterval(1_000)); + } + let per_evaluation = stage3_select(&every_second, &physical, PlanningModels::builtin()); + assert_eq!(per_second.selected, "P58"); + assert_eq!(per_evaluation.selected, "P58"); + for (id, cost) in &per_second.costs { + let expected = 0.1 * per_evaluation.costs[id].total; + assert!((cost.total - expected).abs() <= 1e-9 * expected, "{id}"); + } + let best = per_second.costs["P58"].total; + assert!((best - 5.2201).abs() < 1e-3, "{best}"); +} + /// Every node is charged exactly once, so a shared input is costed once for /// both queries. Stage 3 prices valid candidates only (user decision). #[test] diff --git a/crates/plan-selection/src/lib.rs b/crates/plan-selection/src/lib.rs index 7d27323f..94930e86 100644 --- a/crates/plan-selection/src/lib.rs +++ b/crates/plan-selection/src/lib.rs @@ -16,11 +16,16 @@ //! cannot be built or priced is rejected with its reason; it does not fail //! the selection. //! -//! Prices come from [`crate::cost::analytical_cost::estimate_operator`] over edge +//! Cost is per second of wall time (`docs/design_docs/proposals/stage3-cost-model.md`): +//! an ingestion-time node over the ingestion rate, a query-time node per +//! evaluation times the evaluation rate of the roots reaching it ([`RootDemand`]), +//! plus memory for ingestion-time state that query time reads. Prices come +//! from [`crate::cost::analytical_cost::estimate_operator`] over edge //! statistics derived from the [`DataWorkload`] and a fixed default group -//! count; summary build and estimation are priced as rows × sketch depth and -//! rows read out. These numbers are illustrative, not calibrated. Latency -//! bounds and deployment capabilities are not checked yet. +//! count, weighted by [`Stage3Calibration`]; summary build and estimation are +//! priced as rows × sketch depth and rows read out. These numbers are +//! illustrative, not calibrated. Latency bounds and deployment capabilities +//! are not checked yet. //! //! [`select_plan`] chooses over Stage 1's sharing variants without building //! every combination: per variant, a dynamic program over target nesting (see @@ -58,8 +63,7 @@ use asap_types::ir::schema::{ FieldDataType, SketchAlgorithm, SketchParams, SketchStatistic, WeightDomain, }; use asap_types::ir::{ASAPOp, Operator, OperatorNode, QueryRoot}; -use asap_types::types::AccuracyTarget; -use asap_types::workload::DataWorkload; +use asap_types::workload::{DataWorkload, QueryRecurrence, RepeatedDemand, RootDemand}; use thiserror::Error; use crate::cost::analytical_cost::{ @@ -83,8 +87,10 @@ use asap_physical_optimizer::implementation::physical_candidates::{ stage2_physical, PhysicalCandidate, }; -pub const COST_UNIT: &str = "cpu_ms_per_workload_evaluation"; -pub const COST_SOURCE: &str = "analytical-cost-v1 (illustrative statistics)"; +/// Cost per second of wall time; one cost unit is one CPU-millisecond under +/// [`Stage3Calibration::ILLUSTRATIVE`]. See +/// `docs/design_docs/proposals/stage3-cost-model.md`. +pub const COST_PER_SECOND: &str = "cost_per_second"; /// Groups assumed for every `by (...)` reduction, absent group-count evidence. const DEFAULT_GROUP_COUNT: u64 = 100; @@ -100,6 +106,53 @@ static DEFAULT_COST_MODEL: DefaultCostModel = DefaultCostModel; static DEFAULT_ACCURACY_MODEL: DefaultAccuracyModel = DefaultAccuracyModel; static NO_ACCURACY_EVIDENCE: NoAccuracyEvidence = NoAccuracyEvidence; +/// Stage 3's price coefficients and amortization horizon. Values are +/// illustrative until calibrated from measurements. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct Stage3Calibration { + pub cost_per_cpu_op: f64, + pub cost_per_scan_byte: f64, + /// Price of state retained across evaluations, per byte per second. + pub cost_per_retained_byte_second: f64, + /// Seconds over which one-off and unknown recurrence is amortized. + pub horizon_s: f64, + pub version: &'static str, +} + +impl Stage3Calibration { + /// 1 ns of CPU per operation; 1 GB retained costs 1/8 vCPU + /// (125 CPU-ms per second); one-off work is amortized over 1 h. + pub const ILLUSTRATIVE: Self = Self { + cost_per_cpu_op: 1e-6, + cost_per_scan_byte: 1e-7, + cost_per_retained_byte_second: 1.25e-7, + horizon_s: 3_600.0, + version: "illustrative-v2", + }; + + fn validate(&self) -> Result<(), AnalyticalCostError> { + for (name, value) in [ + ("cost_per_cpu_op", self.cost_per_cpu_op), + ("cost_per_scan_byte", self.cost_per_scan_byte), + ( + "cost_per_retained_byte_second", + self.cost_per_retained_byte_second, + ), + ] { + if !value.is_finite() || value < 0.0 { + return Err(AnalyticalCostError::InvalidCalibration(name, value)); + } + } + if !self.horizon_s.is_finite() || self.horizon_s <= 0.0 { + return Err(AnalyticalCostError::InvalidCalibration( + "horizon_s", + self.horizon_s, + )); + } + Ok(()) + } +} + /// Planning logic, as opposed to the scoped facts it consumes: a model can have /// a built-in default, evidence about a particular deployment cannot. /// @@ -111,6 +164,7 @@ pub struct PlanningModels<'a> { pub cost: &'a dyn CostModel, pub accuracy: &'a dyn AccuracyModel, pub evidence: &'a dyn AccuracyEvidenceProvider, + pub calibration: Stage3Calibration, } impl<'a> PlanningModels<'a> { @@ -123,6 +177,7 @@ impl<'a> PlanningModels<'a> { cost, accuracy, evidence, + calibration: Stage3Calibration::ILLUSTRATIVE, } } @@ -134,6 +189,7 @@ impl<'a> PlanningModels<'a> { cost: &DEFAULT_COST_MODEL, accuracy: &DEFAULT_ACCURACY_MODEL, evidence: &NO_ACCURACY_EVIDENCE, + calibration: Stage3Calibration::ILLUSTRATIVE, } } @@ -151,6 +207,11 @@ impl<'a> PlanningModels<'a> { self.evidence = evidence; self } + + pub fn with_calibration(mut self, calibration: Stage3Calibration) -> Self { + self.calibration = calibration; + self + } } #[derive(Debug, Clone, PartialEq)] @@ -167,7 +228,8 @@ pub struct NodeCost { pub struct CandidateCost { pub total: f64, pub unit: &'static str, - pub source: &'static str, + /// The cost model and its calibration version. + pub source: String, pub per_node: BTreeMap, } @@ -220,11 +282,11 @@ pub enum SelectionError { } /// Reject candidates that miss a query's accuracy target or cannot be priced, -/// price the rest and select the cheapest (the first on ties). `targets[i]` -/// is the requirement of `candidate.roots[i]`; `None` imposes none. +/// price the rest and select the cheapest (the first on ties). `demand[i]` +/// is the demand of `candidate.roots[i]`. pub fn stage3_select( cands: &[PhysicalCandidate], - targets: &[Option], + demand: &[RootDemand], data: &DataWorkload, models: PlanningModels<'_>, ) -> Result { @@ -232,7 +294,7 @@ pub fn stage3_select( let mut rejected = Vec::new(); let mut best: Option<(&str, f64)> = None; for candidate in cands { - match assess(candidate, targets, data, &models) { + match assess(candidate, demand, data, &models) { Ok(cost) => { if best.is_none_or(|(_, total)| cost.total < total) { best = Some((&candidate.id, cost.total)); @@ -255,7 +317,7 @@ pub fn stage3_select( id: candidate.id.clone(), valid: true, reason: format!( - "costlier: {:.3} vs {:.3} {COST_UNIT}", + "costlier: {:.3} vs {:.3} {COST_PER_SECOND}", costs[&candidate.id].total, best_total ), }); @@ -272,21 +334,22 @@ pub fn stage3_select( /// Stage 3's checks and price for one candidate; `Err` is the rejection reason. fn assess( candidate: &PhysicalCandidate, - targets: &[Option], + demand: &[RootDemand], data: &DataWorkload, models: &PlanningModels<'_>, ) -> Result { - if candidate.roots.len() != targets.len() { + if candidate.roots.len() != demand.len() { return Err(format!( - "{} roots but {} accuracy targets", + "{} roots but {} root demands (accuracy targets)", candidate.roots.len(), - targets.len() + demand.len() )); } - if let Some(reason) = accuracy_violation(candidate, targets, models) { + if let Some(reason) = accuracy_violation(candidate, demand, models) { return Err(reason); } - price(&candidate.dag, data).map_err(|(node, error)| format!("node {node:?}: {error}")) + price(&candidate.dag, demand, data, &models.calibration) + .map_err(|(node, error)| format!("node {node:?}: {error}")) } /// One Stage 1 sharing variant as selection sees it: candidates of variant @@ -410,17 +473,17 @@ pub struct Enumeration { /// its reason. pub fn select_exhaustive( stage1: &[SharingVariant], - targets: &[Option], + demand: &[RootDemand], data: &DataWorkload, models: PlanningModels<'_>, max: usize, ) -> Result, SelectionError> { - exhaustive(&variants(stage1), targets, data, models, max) + exhaustive(&variants(stage1), demand, data, models, max) } fn exhaustive( variants: &[Variant<'_, Id>], - targets: &[Option], + demand: &[RootDemand], data: &DataWorkload, models: PlanningModels<'_>, max: usize, @@ -457,7 +520,7 @@ fn exhaustive( .iter() .filter_map(|c| c.physical.clone()) .collect(); - let selection = match stage3_select(&physical, targets, data, models) { + let selection = match stage3_select(&physical, demand, data, models) { Ok(mut selection) => { selection.rejected.extend(failed); selection @@ -499,7 +562,7 @@ const ADDITIVITY_TOLERANCE: f64 = 1e-9; /// runs once per variant and the cheapest result wins (the first on ties). pub fn select_plan( stage1: &[SharingVariant], - targets: &[Option], + demand: &[RootDemand], data: &DataWorkload, models: PlanningModels<'_>, ) -> Result, SelectionError> { @@ -511,9 +574,9 @@ pub fn select_plan( for variant in variants(stage1) { let evaluate = |choice: &[usize]| -> Result { let (_, candidate) = realize(variant, choice)?; - assess(&candidate, targets, data, &models).map(|cost| cost.total) + assess(&candidate, demand, data, &models).map(|cost| cost.total) }; - let plan = match select_variant(variant, targets, data, models, &evaluate) { + let plan = match select_variant(variant, demand, data, models, &evaluate) { Ok(plan) => plan, Err(SelectionError::NoValidCandidate(reasons)) => { failures.extend(reasons); @@ -561,7 +624,7 @@ fn costlier(id: &str, total: f64, best: f64) -> Rejection { Rejection { id: id.to_string(), valid: true, - reason: format!("costlier: {total:.3} vs {best:.3} {COST_UNIT}"), + reason: format!("costlier: {total:.3} vs {best:.3} {COST_PER_SECOND}"), } } @@ -593,7 +656,7 @@ fn costlier(id: &str, total: f64, best: f64) -> Rejection { /// guaranteed optimal. fn select_variant( variant: Variant<'_, Id>, - targets: &[Option], + demand: &[RootDemand], data: &DataWorkload, models: PlanningModels<'_>, evaluate: &dyn Fn(&[usize]) -> Result, @@ -611,7 +674,7 @@ fn select_variant( let base = match evaluate(&with(&[])) { Ok(base) => base, Err(reason) => { - return fallback(variant, targets, data, models, with(&[]), reason); + return fallback(variant, demand, data, models, with(&[]), reason); } }; let local: Vec>> = inventory @@ -686,18 +749,18 @@ fn select_variant( } } if let Some(reason) = coupling { - return fallback(variant, targets, data, models, choice, reason); + return fallback(variant, demand, data, models, choice, reason); } match finish( variant, - targets, + demand, data, &models, choice.clone(), SelectionMethod::TreeDp, ) { Ok(plan) => Ok(plan), - Err(reason) => fallback(variant, targets, data, models, choice, reason), + Err(reason) => fallback(variant, demand, data, models, choice, reason), } } @@ -760,14 +823,14 @@ fn best_choice( /// Build `choice` in `variant` and run Stage 3 on it. fn finish( variant: Variant<'_, Id>, - targets: &[Option], + demand: &[RootDemand], data: &DataWorkload, models: &PlanningModels<'_>, choice: Vec, method: SelectionMethod, ) -> Result, String> { let (logical, physical) = realize(variant, &choice)?; - let cost = assess(&physical, targets, data, models)?; + let cost = assess(&physical, demand, data, models)?; Ok(SelectedPlan { shared: variant.shared, choice, @@ -787,14 +850,14 @@ fn finish( /// `reason`. fn fallback( variant: Variant<'_, Id>, - targets: &[Option], + demand: &[RootDemand], data: &DataWorkload, models: PlanningModels<'_>, choice: Vec, reason: String, ) -> Result, SelectionError> { if combination_count(variant.inventory) <= MAX_ENUMERATED_CANDIDATES { - let enumeration = exhaustive(&[variant], targets, data, models, MAX_ENUMERATED_CANDIDATES)?; + let enumeration = exhaustive(&[variant], demand, data, models, MAX_ENUMERATED_CANDIDATES)?; let winner = enumeration .candidates .iter() @@ -806,7 +869,7 @@ fn fallback( .expect("the selected candidate was built"); let mut plan = finish( variant, - targets, + demand, data, &models, winner.choice.clone(), @@ -829,7 +892,7 @@ fn fallback( let method = SelectionMethod::TreeDpNotGuaranteedOptimal { reason: reason.clone(), }; - finish(variant, targets, data, &models, choice.clone(), method).map_err(|failure| { + finish(variant, demand, data, &models, choice.clone(), method).map_err(|failure| { SelectionError::NoValidCandidate(vec![Rejection { id: format!("P{}", variant.number(&choice)), valid: false, @@ -855,16 +918,16 @@ pub struct StagePipelineRun { /// that many candidates for display as well (0: none). pub fn plan_stages( roots: Vec<(Id, QueryRoot)>, - targets: &[Option], + demand: &[RootDemand], data: &DataWorkload, models: PlanningModels<'_>, display: usize, ) -> Result, SelectionError> { let stage1 = stage1_logical_candidates(roots)?; - let plan = select_plan(&stage1, targets, data, models)?; + let plan = select_plan(&stage1, demand, data, models)?; let enumeration = match display { 0 => None, - max => Some(select_exhaustive(&stage1, targets, data, models, max)?), + max => Some(select_exhaustive(&stage1, demand, data, models, max)?), }; Ok(StagePipelineRun { stage1, @@ -876,11 +939,13 @@ pub fn plan_stages( /// The first summary estimate that misses its query's target, as a reason. fn accuracy_violation( candidate: &PhysicalCandidate, - targets: &[Option], + demand: &[RootDemand], models: &PlanningModels<'_>, ) -> Option { - for (query, (root, target)) in candidate.roots.iter().zip(targets).enumerate() { - let Some(target) = target else { continue }; + for (query, (root, demand)) in candidate.roots.iter().zip(demand).enumerate() { + let Some(target) = &demand.accuracy else { + continue; + }; for node in OperatorNode::reachable(root) { let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, @@ -936,15 +1001,133 @@ fn family_name(family: &FieldDataType) -> String { /// Statistics the analytical model needs, derived once per workload. struct Shape { series: u64, - rows_per_ms: f64, + /// λ, rows ingested per second. + rows_per_second: f64, +} + +/// λ: the declared ingestion rate, else one sample per series per ingestion +/// interval, else the default. +fn ingestion_rate(data: &DataWorkload) -> Result { + let rate = match ( + data.ingestion_rate.value, + data.input_cardinality.value, + data.data_ingestion_interval.value, + ) { + (Some(rate), _, _) => rate.0, + (None, Some(series), Some(interval)) if interval.0 > 0 => { + series as f64 * 1_000.0 / interval.0 as f64 + } + _ => DEFAULT_ROWS_PER_SECOND, + }; + if rate.is_finite() && rate >= 0.0 { + Ok(rate) + } else { + Err(AnalyticalCostError::InvalidIngestionRate(rate)) + } +} + +/// Evaluations per second of a query-time node read by `roots`. Repeating +/// roots with equal intervals are evaluated together, so each interval +/// counts once. One-off, scheduled and unknown roots run as one batch: the +/// most invocations among them, amortized over `horizon_s`. +fn evaluation_rate<'d>( + roots: impl IntoIterator, + horizon_s: f64, +) -> Result { + let mut intervals = std::collections::BTreeSet::new(); + let mut estimated = 0.0; + let mut invocations = 0u64; + for demand in roots { + match &demand.recurrence { + QueryRecurrence::Repeated( + RepeatedDemand::FixedInterval(interval) + | RepeatedDemand::FixedIntervalAt { interval, .. }, + ) => { + intervals.insert(*interval); + } + QueryRecurrence::Repeated(RepeatedDemand::EstimatedRate(estimate)) => { + let rate = estimate.expected_rate.0; + if !rate.is_finite() || rate < 0.0 { + return Err(AnalyticalCostError::InvalidRecurrence); + } + estimated += rate; + } + QueryRecurrence::Repeated(RepeatedDemand::Scheduled(times)) => { + invocations = invocations.max(times.len() as u64); + } + QueryRecurrence::OneTime { + invocations: count, .. + } => invocations = invocations.max(*count), + QueryRecurrence::Unknown => invocations = invocations.max(1), + } + } + let repeated = evaluation_rate_of(intervals) + .map_err(|_| AnalyticalCostError::InvalidRecurrence)? + .map_or(0.0, |rate| rate.0); + Ok(repeated + estimated + invocations as f64 / horizon_s) +} + +/// For every node, the indices of the roots that reach it. +fn reaching_roots(dag: &PhysicalASAPDAG) -> HashMap> { + let mut reached: HashMap> = HashMap::new(); + for (index, &root) in dag.roots.iter().enumerate() { + let mut stack = vec![root]; + while let Some(id) = stack.pop() { + let roots = reached.entry(id).or_default(); + if roots.last() == Some(&index) { + continue; + } + roots.push(index); + stack.extend( + dag.edges + .iter() + .filter(|e| e.consumer == id) + .map(|e| e.producer), + ); + } + } + reached +} + +/// How far back the time ranges reading `id` reach, in milliseconds: each +/// range plus the offsets between it and `id`. `None` when no range reads it. +fn scan_extent_ms(dag: &PhysicalASAPDAG, id: PhysicalASAPNodeId, offset_ms: i64) -> Option { + dag.edges + .iter() + .filter(|e| e.producer == id) + .filter_map(|e| { + let consumer = dag.nodes.iter().find(|n| n.id == e.consumer)?; + match &consumer.payload { + Payload::Relational { + operator: NonASAPOpKind::TimeShift { shift }, + } => scan_extent_ms(dag, consumer.id, offset_ms.saturating_add(shift.offset_ms)), + Payload::Relational { + operator: NonASAPOpKind::TimeRange { range, .. }, + } => Some((range.as_millis() as u64).saturating_add(offset_ms.max(0) as u64)), + _ => None, + } + }) + .max() } -/// Price every node of `dag` once. Nodes are exported children first, so -/// each node's input statistics are known when it is reached. +/// Price every node of `dag` once, per second of wall time. Nodes are +/// exported children first, so each node's input statistics are known when +/// it is reached. An ingestion-time node is priced over one second of +/// ingested rows; a query-time node per evaluation, times its evaluation +/// rate. State an ingestion-time node keeps for query-time readers is also +/// charged per retained byte per second. fn price( dag: &PhysicalASAPDAG, + demand: &[RootDemand], data: &DataWorkload, + calibration: &Stage3Calibration, ) -> Result { + let first = dag + .roots + .first() + .copied() + .unwrap_or(asap_types::ir::export::LogicalASAPNodeId(0)); + calibration.validate().map_err(|error| (first, error))?; let series = data .input_cardinality .value @@ -952,22 +1135,22 @@ fn price( .max(1); let shape = Shape { series, - rows_per_ms: data - .ingestion_rate - .value - .map_or(DEFAULT_ROWS_PER_SECOND, |rate| rate.0) - / 1_000.0, + rows_per_second: ingestion_rate(data).map_err(|error| (first, error))?, }; - let calibration = ResourceCalibration { - cost_per_cpu_op: 1e-6, - cost_per_scan_byte: 1e-7, + let rows_per_ms = shape.rows_per_second / 1_000.0; + let resources = ResourceCalibration { + cost_per_cpu_op: calibration.cost_per_cpu_op, + cost_per_scan_byte: calibration.cost_per_scan_byte, + // Transient query-time memory is not priced. cost_per_retained_byte: 0.0, - version: "illustrative-v1".into(), + version: calibration.version.into(), }; + let reached = reaching_roots(dag); let nodes: HashMap<_, _> = dag.nodes.iter().map(|n| (n.id, n)).collect(); let mut output: HashMap = HashMap::new(); let mut per_node = BTreeMap::new(); for node in &dag.nodes { + let ingestion = !node.output_state.timing.is_query_time(); let inputs: Vec<_> = dag .edges .iter() @@ -989,30 +1172,28 @@ fn price( promql: None, }; let groups = |reduction: &Reduction| { - match reduction { + let groups = match reduction { Reduction::Reduce(keys) if keys.keys().is_empty() && !keys.is_without() => 1, Reduction::Reduce(keys) if !keys.is_without() => DEFAULT_GROUP_COUNT, _ => shape.series, + }; + // One second of ingested rows does not bound the groups a + // maintained state holds. + match ingestion { + true => groups, + false => groups.min(input.rows.max(1)), } - .min(input.rows.max(1)) }; let (out, estimate, detail) = match &node.payload { Payload::Relational { operator } => match operator { NonASAPOpKind::Scan { .. } => { - // A scan reads what its time range keeps. - let lookback = dag - .edges - .iter() - .filter(|e| e.producer == node.id) - .filter_map(|e| match &nodes[&e.consumer].payload { - Payload::Relational { - operator: NonASAPOpKind::TimeRange { range, .. }, - } => Some(range.as_millis() as u64), - _ => None, - }) - .max() - .unwrap_or(DEFAULT_LOOKBACK_MS); - let out = edge(((shape.rows_per_ms * lookback as f64).round() as u64).max(1)); + // At ingestion time, one second of arriving rows; at + // query time, as far back as the ranges reading it reach. + let span_ms = match ingestion { + true => 1_000, + false => scan_extent_ms(dag, node.id, 0).unwrap_or(DEFAULT_LOOKBACK_MS), + }; + let out = edge(((rows_per_ms * span_ms as f64).round() as u64).max(1)); let estimate = estimate_operator( PhysicalOperator::Scan, OperatorStatistics::Scan { @@ -1095,6 +1276,23 @@ fn price( ); (out, estimate, format!("limit to {} rows", out.rows)) } + // At query time a range keeps only its own span of a longer + // scan: a filter on the timestamp. + NonASAPOpKind::TimeRange { range, .. } if !ingestion => { + let rows = (rows_per_ms * range.as_millis() as f64).round() as u64; + let out = edge(input.rows.min(rows.max(1))); + let estimate = estimate_operator( + PhysicalOperator::Filter { + predicate_operations_per_row: 1, + }, + OperatorStatistics::Filter { edges: unary(out) }, + ); + ( + out, + estimate, + format!("time range {range:?}: pass {} rows", out.rows), + ) + } other => { let estimate = estimate_operator( PhysicalOperator::PassThrough, @@ -1162,8 +1360,29 @@ fn price( ), }; let cost = estimate - .and_then(|estimate| estimate.calibrated_cost(&calibration)) + .and_then(|estimate| estimate.calibrated_cost(&resources)) .map_err(|error| (node.id, error))?; + let (cost, detail) = if ingestion { + let read_at_query_time = dag.edges.iter().any(|e| { + e.producer == node.id && nodes[&e.consumer].output_state.timing.is_query_time() + }); + match read_at_query_time { + // The window being built and the completed one. + true => { + let retained = 2 * out.rows * state_bytes(node); + ( + cost + calibration.cost_per_retained_byte_second * retained as f64, + format!("{detail}; ingestion time, retains {retained} bytes"), + ) + } + false => (cost, format!("{detail}; ingestion time")), + } + } else { + let roots = reached.get(&node.id).into_iter().flatten(); + let rate = evaluation_rate(roots.filter_map(|&r| demand.get(r)), calibration.horizon_s) + .map_err(|error| (node.id, error))?; + (cost * rate, format!("{detail}; x {rate:.4} evaluations/s")) + }; output.insert(node.id, out); per_node.insert( node.id, @@ -1176,8 +1395,11 @@ fn price( } Ok(CandidateCost { total: per_node.values().map(|n| n.cost).sum(), - unit: COST_UNIT, - source: COST_SOURCE, + unit: COST_PER_SECOND, + source: format!( + "analytical-cost-v2 (illustrative statistics, calibration {})", + calibration.version + ), per_node, }) } @@ -1251,6 +1473,18 @@ fn row_bytes(schema: &Schema) -> u64 { .max(1) } +/// Bytes per group of the state `node` keeps: the summary's state, or the +/// row of an exact state. +fn state_bytes(node: &asap_types::ir::export::PhysicalASAPDAGNode) -> u64 { + match &node.payload { + Payload::SummaryAgg { + family: family @ FieldDataType::Sketch(..), + .. + } => summary_shape(family).1, + _ => row_bytes(&node.output_schema), + } +} + /// Update operations per input row and bytes per state. fn summary_shape(family: &FieldDataType) -> (u64, u64) { match family { @@ -1283,7 +1517,24 @@ mod tests { use crate::test_support::lower_promql; use asap_physical_optimizer::implementation::physical_candidates::stage2_physical; use asap_types::ir::QueryRoot; - use asap_types::workload::{Evidence, Rate}; + use asap_types::types::AccuracyTarget; + use asap_types::workload::{Evidence, Predictability, Rate, RepetitionInterval}; + + /// A root repeating every `interval_ms`. + fn repeating(accuracy: Option, interval_ms: u32) -> RootDemand { + RootDemand { + accuracy, + recurrence: QueryRecurrence::Repeated(RepeatedDemand::FixedInterval( + RepetitionInterval(interval_ms), + )), + predictability: Predictability::Unknown, + } + } + + /// A root repeating every 10 s, as Example 1's panels do. + fn every_10s(accuracy: Option) -> RootDemand { + repeating(accuracy, 10_000) + } fn data() -> DataWorkload { DataWorkload { @@ -1353,7 +1604,7 @@ mod tests { }; let selection = stage3_select( &candidates, - &[Some(strict)], + &[every_10s(Some(strict))], &data(), PlanningModels::builtin(), ) @@ -1384,7 +1635,7 @@ mod tests { }; let selection = stage3_select( &candidates(), - &[Some(target)], + &[every_10s(Some(target))], &data(), PlanningModels::builtin(), ) @@ -1413,8 +1664,8 @@ mod tests { .unwrap() } - fn no_targets(inventory: &LocalLogicalCandidates) -> Vec> { - vec![None; inventory.roots.len()] + fn no_targets(inventory: &LocalLogicalCandidates) -> Vec { + vec![every_10s(None); inventory.roots.len()] } /// Real Stage 1 → 3 cost plus a penalty whenever the two named targets @@ -1422,7 +1673,7 @@ mod tests { /// of the target reading it. fn coupled<'a>( inventory: &'a LocalLogicalCandidates, - targets: &'a [Option], + targets: &'a [RootDemand], data: &'a DataWorkload, (outer, inner): (usize, usize), ) -> impl Fn(&[usize]) -> Result + 'a { @@ -1651,7 +1902,7 @@ mod tests { }; let selection = stage3_select( &candidates, - &[Some(target)], + &[every_10s(Some(target))], &data(), PlanningModels::builtin(), ) @@ -1677,7 +1928,13 @@ mod tests { let rows: Vec = candidates() .iter() .map(|candidate| { - let cost = price(&candidate.dag, &data).unwrap(); + let cost = price( + &candidate.dag, + &[every_10s(None)], + &data, + &Stage3Calibration::ILLUSTRATIVE, + ) + .unwrap(); cost.per_node[&candidate.dag.roots[0]].rows }) .collect(); @@ -1696,7 +1953,7 @@ mod tests { }; let selection = stage3_select( &candidates, - &[Some(target)], + &[every_10s(Some(target))], &data(), PlanningModels::builtin(), ) @@ -1717,4 +1974,273 @@ mod tests { assert_eq!(costlier.len(), 1); assert_ne!(costlier[0].id, selection.selected); } + + /// `query` with its KLL alternative chosen, every state maintained at + /// `timing`, as a priced-ready physical DAG. + fn kll_dag(query: &str, ingestion: bool) -> PhysicalASAPDAG { + use asap_types::ir::{ + apply_materialization_timings, MaterializationAssignment, TimingMemo, + }; + let inventory = inventory(&[query]); + let choice: Vec<_> = inventory + .targets + .iter() + .map(|t| { + t.alternatives + .iter() + .position(|a| matches!(a, asap_logical_optimizer::Realization::Sketch(kind) if *kind.algorithm() == SketchAlgorithm::Kll)) + .unwrap_or(0) + }) + .collect(); + let roots: Vec<_> = + asap_logical_optimizer::pass1::logical_candidates::compose_logical_candidate( + &inventory, &choice, + ) + .unwrap() + .into_iter() + .map(|(_, root)| match root { + QueryRoot::Operator(node) => node, + QueryRoot::Scalar(_) => panic!("operator root"), + }) + .collect(); + let assignment = match ingestion { + true => MaterializationAssignment::all_ingestion_time(), + false => MaterializationAssignment::all_query_time(), + }; + let mut memo = TimingMemo::new(); + let timed: Vec<_> = roots + .iter() + .map(|root| apply_materialization_timings(root, &assignment, &mut memo).unwrap()) + .collect(); + asap_types::ir::export::compile_physical_asap_workload_with_node_ids(&timed) + .unwrap() + .dag + } + + fn node_of( + dag: &PhysicalASAPDAG, + pick: impl Fn(&Payload) -> bool, + ) -> &asap_types::ir::export::PhysicalASAPDAGNode { + dag.nodes.iter().find(|n| pick(&n.payload)).unwrap() + } + + fn is_scan(payload: &Payload) -> bool { + matches!( + payload, + Payload::Relational { + operator: NonASAPOpKind::Scan { .. } + } + ) + } + + fn is_build(payload: &Payload) -> bool { + matches!(payload, Payload::SummaryAgg { .. }) + } + + /// An ingestion-time node is priced over λ rows per second, whatever its + /// readers' evaluation rate; a query-time node scales with that rate. + #[test] + fn ingestion_time_node_is_priced_at_the_ingestion_rate() { + let dag = kll_dag("quantile_over_time(0.99, m[1m])", true); + let scan = node_of(&dag, is_scan); + assert!(!scan.output_state.timing.is_query_time()); + let at = |interval_ms| { + price( + &dag, + &[repeating(None, interval_ms)], + &data(), + &Stage3Calibration::ILLUSTRATIVE, + ) + .unwrap() + }; + let (fast, slow) = (at(1_000), at(10_000)); + // data() declares λ = 10 000 rows/s. + assert_eq!(fast.per_node[&scan.id].rows, 10_000); + let build = node_of(&dag, is_build).id; + for id in [scan.id, build] { + assert_eq!(fast.per_node[&id].cost, slow.per_node[&id].cost); + } + let estimate = dag.roots[0]; + assert!( + (fast.per_node[&estimate].cost - 10.0 * slow.per_node[&estimate].cost).abs() < 1e-12 + ); + } + + /// The memory term is w × retained bytes: two windows of every group's + /// state, for ingestion-time state that query time reads. + #[test] + fn memory_term_is_weight_times_retained_bytes() { + let dag = kll_dag("quantile_over_time(0.99, m[1m])", true); + let build = node_of(&dag, is_build); + let Payload::SummaryAgg { family, .. } = &build.payload else { + unreachable!() + }; + let cost = |w| { + let calibration = Stage3Calibration { + cost_per_retained_byte_second: w, + ..Stage3Calibration::ILLUSTRATIVE + }; + price(&dag, &[every_10s(None)], &data(), &calibration).unwrap() + }; + let (with, without) = (cost(1.25e-7), cost(0.0)); + // One state per series: data() declares 10 000. + let groups = with.per_node[&build.id].rows; + assert_eq!(groups, 10_000); + let retained = 2 * groups * summary_shape(family).1; + let term = with.per_node[&build.id].cost - without.per_node[&build.id].cost; + assert!((term - 1.25e-7 * retained as f64).abs() < 1e-12, "{term}"); + assert!((with.total - without.total - term).abs() < 1e-12); + // Query-time state is transient: no memory term. + let query_time = kll_dag("quantile_over_time(0.99, m[1m])", false); + let (a, b) = ( + price( + &query_time, + &[every_10s(None)], + &data(), + &Stage3Calibration::ILLUSTRATIVE, + ), + price( + &query_time, + &[every_10s(None)], + &data(), + &Stage3Calibration { + cost_per_retained_byte_second: 0.0, + ..Stage3Calibration::ILLUSTRATIVE + }, + ), + ); + assert_eq!(a.unwrap().total, b.unwrap().total); + } + + /// A node reached by two roots with equal intervals is evaluated once + /// per interval; different intervals add their rates. + #[test] + fn node_shared_by_roots_with_equal_intervals_is_charged_once() { + let root = lower_promql("sum by (job) (rate(m[1m]))", AccuracyTarget::Exact); + let one = stage2_physical("L", std::slice::from_ref(&root)) + .unwrap() + .dag; + let two = stage2_physical("L", &[root.clone(), root]).unwrap().dag; + let total = |dag: &PhysicalASAPDAG, demand: &[RootDemand]| { + price(dag, demand, &data(), &Stage3Calibration::ILLUSTRATIVE) + .unwrap() + .total + }; + let single = total(&one, &[every_10s(None)]); + assert_eq!(total(&two, &[every_10s(None), every_10s(None)]), single); + let mixed = total(&two, &[every_10s(None), repeating(None, 20_000)]); + assert!((mixed - 1.5 * single).abs() < 1e-12 * single.max(1.0)); + } + + /// One-off recurrence is amortized over the horizon H: + /// invocations × per-evaluation cost / H. + #[test] + fn one_off_recurrence_is_amortized_over_the_horizon() { + let root = lower_promql("sum by (job) (rate(m[1m]))", AccuracyTarget::Exact); + let dag = stage2_physical("L", &[root]).unwrap().dag; + let calibration = Stage3Calibration::ILLUSTRATIVE; + let total = |recurrence| { + let demand = RootDemand { + recurrence, + ..every_10s(None) + }; + price(&dag, &[demand], &data(), &calibration).unwrap().total + }; + // Once per second: the per-evaluation cost. + let per_evaluation = total(QueryRecurrence::Repeated(RepeatedDemand::FixedInterval( + RepetitionInterval(1_000), + ))); + let once = total(QueryRecurrence::OneTime { + invocations: 3, + execute_at: None, + }); + let expected = 3.0 * per_evaluation / calibration.horizon_s; + assert!( + (once - expected).abs() < 1e-12 * expected, + "{once} vs {expected}" + ); + assert_eq!( + total(QueryRecurrence::Unknown), + per_evaluation / calibration.horizon_s + ); + } + + /// Regression (#509 Example 3, Pattern A): a scan is priced over the + /// longest range plus offset reading it, and each range passes only its + /// own span, so sharing one 5-year scan is not costlier than separate + /// scans. + #[test] + fn shared_long_range_scan_is_not_costlier_than_separate_scans() { + let queries = [ + "quantile_over_time(0.99, latency_ms[5y])", + "quantile_over_time(0.99, latency_ms[1y])", + "quantile_over_time(0.99, latency_ms[1y] offset 1y)", + "quantile_over_time(0.99, latency_ms[1y] offset 2y)", + "quantile_over_time(0.99, latency_ms[3y] offset 2y)", + ]; + let roots = queries + .iter() + .enumerate() + .map(|(i, query)| { + let root = lower_promql(query, AccuracyTarget::Exact); + let root = + asap_types::ir::schema_support::with_promql_series_identity(&root).unwrap(); + (i, QueryRoot::Operator(root)) + }) + .collect(); + let stage1 = stage1_logical_candidates(roots).unwrap(); + let data = DataWorkload { + ingestion_rate: Evidence { + value: Some(Rate(1_000_000.0 / 15.0)), + ..Default::default() + }, + input_cardinality: Evidence { + value: Some(1_000_000), + ..Default::default() + }, + ..Default::default() + }; + let demand: Vec<_> = queries + .iter() + .map(|_| RootDemand { + recurrence: QueryRecurrence::OneTime { + invocations: 1, + execute_at: None, + }, + ..every_10s(None) + }) + .collect(); + let total = |shared: bool| { + let variant = *variants(&stage1) + .iter() + .find(|v| v.shared == shared) + .unwrap(); + let raw = vec![0; variant.inventory.targets.len()]; + let (_, candidate) = realize(variant, &raw).unwrap(); + let cost = assess(&candidate, &demand, &data, &PlanningModels::builtin()).unwrap(); + let scan_rows: Vec<_> = candidate + .dag + .nodes + .iter() + .filter(|n| is_scan(&n.payload)) + .map(|n| cost.per_node[&n.id].rows) + .collect(); + (cost.total, scan_rows) + }; + let (shared, shared_scans) = total(true); + let (separate, separate_scans) = total(false); + let year = 365.0 * 24.0 * 3_600.0 * 1_000_000.0 / 15.0; + assert_eq!(shared_scans.len(), 1); + assert!( + (shared_scans[0] as f64 / year - 5.0).abs() < 0.01, + "{shared_scans:?}" + ); + let mut years: Vec<_> = separate_scans + .iter() + .map(|&rows| (rows as f64 / year).round() as u64) + .collect(); + years.sort(); + assert_eq!(years, [1, 2, 3, 5, 5]); + assert!(shared <= separate, "shared {shared} vs separate {separate}"); + } } diff --git a/crates/planner/src/pass/stage_pipeline.rs b/crates/planner/src/pass/stage_pipeline.rs index f0f6d2b4..dd6cbf29 100644 --- a/crates/planner/src/pass/stage_pipeline.rs +++ b/crates/planner/src/pass/stage_pipeline.rs @@ -4,12 +4,12 @@ //! [`plan_stages`] runs them: Stage 1 lists each target's local alternatives //! (Pass 1) with and without identical sub-DAGs shared across queries (Pass //! 2's identical-expression rule), Stage 2 implements a candidate physically -//! (everything at query time), and Stage 3 checks accuracy, prices it and -//! chooses. Pass 2's other rules are not planned yet. +//! (everything at query time), and Stage 3 checks accuracy, prices it per +//! second from each entry's recurrence and chooses. Pass 2's other rules are not planned yet. use asap_types::ir::schema_support::with_promql_series_identity; use asap_types::ir::QueryRoot; -use asap_types::workload::QueryLanguage; +use asap_types::workload::{QueryLanguage, RootDemand}; use super::{OptimizationInput, OptimizationPass, OptimizeError, PlanOutput, QueryPlan}; use asap_plan_selection::plan_stages; @@ -46,12 +46,12 @@ impl OptimizationPass for StagePipeline { (index, QueryRoot::Operator(root)) }) .collect(); - let targets: Vec<_> = workload + let demand: Vec = workload .entries() - .map(|(entry, _)| Some(entry.requirements.accuracy.target())) + .map(|(entry, _)| RootDemand::from(&entry)) .collect(); let data = workload.data_workload().cloned().unwrap_or_default(); - let plan = plan_stages(roots, &targets, &data, input.models, 0)?.plan; + let plan = plan_stages(roots, &demand, &data, input.models, 0)?.plan; output.plans = plan .logical diff --git a/crates/planner/tests/e2e_plan.rs b/crates/planner/tests/e2e_plan.rs index ea2eb615..3a54cd30 100644 --- a/crates/planner/tests/e2e_plan.rs +++ b/crates/planner/tests/e2e_plan.rs @@ -124,7 +124,7 @@ async fn facade_plans_match_exhaustive_stage_pipeline_selection() { .await .expect("lowers"); roots.push((index, QueryRoot::Operator(expr))); - targets.push(Some(accuracy)); + targets.push(asap_types::workload::RootDemand::from(&entry)); } let inventory = stage1_logical_candidates(roots).expect("Stage 1"); let data = workload.data_workload.clone().unwrap_or_default(); diff --git a/crates/planner/tests/stage_pipeline_selection.rs b/crates/planner/tests/stage_pipeline_selection.rs index 8be40c30..6d2a9c0e 100644 --- a/crates/planner/tests/stage_pipeline_selection.rs +++ b/crates/planner/tests/stage_pipeline_selection.rs @@ -20,7 +20,7 @@ use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataArrival, DataDistribution, DataWorkload, DurationMs, Evidence, EvidenceSource, LatencyRequirement, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryTimeScope, QueryWorkload, Rate, RepeatedDemand, - RepeatingEntry, RepetitionInterval, SqlDialect, TimeSelection, + RepeatingEntry, RepetitionInterval, RootDemand, SqlDialect, TimeSelection, }; type Inventory = Vec>; @@ -132,11 +132,11 @@ fn promql_inventory(workload: &PlanningWorkload) -> Inventory { stage1_logical_candidates(roots).expect("Stage 1") } -fn targets(workload: &PlanningWorkload) -> Vec> { +fn targets(workload: &PlanningWorkload) -> Vec { workload .query_workload .entries() - .map(|entry| Some(entry.requirements.accuracy.target())) + .map(|entry| RootDemand::from(&entry)) .collect() } diff --git a/crates/types/src/workload/mod.rs b/crates/types/src/workload/mod.rs index 0289d2e3..6e493f9b 100644 --- a/crates/types/src/workload/mod.rs +++ b/crates/types/src/workload/mod.rs @@ -427,6 +427,26 @@ impl From<&RepeatingEntry> for QueryWorkloadEntry { } } +/// What planning needs of one query root beyond its DAG: Stage 3 checks +/// `accuracy` (`None` imposes no target) and prices by `recurrence`; +/// Stage 2's ingestion-time eligibility also reads `predictability`. +#[derive(Debug, Clone, PartialEq)] +pub struct RootDemand { + pub accuracy: Option, + pub recurrence: QueryRecurrence, + pub predictability: Predictability, +} + +impl From<&QueryWorkloadEntry> for RootDemand { + fn from(entry: &QueryWorkloadEntry) -> Self { + Self { + accuracy: Some(entry.requirements.accuracy.target()), + recurrence: entry.recurrence.clone(), + predictability: entry.predictability.clone(), + } + } +} + // ── Data workload ───────────────────────────────────────────────────────────── /// Whether the data queried by this workload is static, still arriving, or a diff --git a/docs/design_docs/proposals/README.md b/docs/design_docs/proposals/README.md index cb44fadb..c39d4f5d 100644 --- a/docs/design_docs/proposals/README.md +++ b/docs/design_docs/proposals/README.md @@ -11,3 +11,4 @@ extensions. A design document is not a promise of downstream runtime support. - [Operator sharing](operator-sharing.md) - [Decoupling operators from scalar expressions](decoupling_op_and_expr.md) - [ASAPPlanner layering](planner-layering.md) +- [Stage 3 cost model](stage3-cost-model.md) diff --git a/docs/design_docs/proposals/planner-layering-example1-acceptance.md b/docs/design_docs/proposals/planner-layering-example1-acceptance.md index dc4b83d7..2f110708 100644 --- a/docs/design_docs/proposals/planner-layering-example1-acceptance.md +++ b/docs/design_docs/proposals/planner-layering-example1-acceptance.md @@ -114,70 +114,70 @@ Generated from `tools/dag-viewer/examples/planner-layering-example1.json`. | Id | Label | Q1 choice | Q2 choice | Input | Stage 3 outcome | Reason | |---|---|---|---|---|---|---| -| P1 (L1) | Q1 exact · Q2 exact | rate: raw; sum: raw | top-k: exact (sort → limit); sum_over_time: raw | separate | valid, costlier | 86.401 vs 52.201 cpu ms | -| P2 (L2) | Q1 exact · Q2 exact (Sum acc) | rate: raw; sum: raw | top-k: exact (sort → limit); sum_over_time: Sum acc | separate | valid, costlier | 83.401 vs 52.201 cpu ms | +| P1 (L1) | Q1 exact · Q2 exact | rate: raw; sum: raw | top-k: exact (sort → limit); sum_over_time: raw | separate | valid, costlier | 8.640 vs 5.220 cost/s | +| P2 (L2) | Q1 exact · Q2 exact (Sum acc) | rate: raw; sum: raw | top-k: exact (sort → limit); sum_over_time: Sum acc | separate | valid, costlier | 8.340 vs 5.220 cost/s | | P3 (L3) | Q1 exact · Q2 CMS+heap | rate: raw; sum: raw | top-k: Count-Min + heap; sum_over_time: raw | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | | P4 (L4) | Q1 exact · Q2 CMS+heap (Sum acc) | rate: raw; sum: raw | top-k: Count-Min + heap; sum_over_time: Sum acc | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P5 (L5) | Q1 exact · Q2 CountSketch+heap | rate: raw; sum: raw | top-k: CountSketch + heap; sum_over_time: raw | separate | valid, costlier | 198.401 vs 52.201 cpu ms | -| P6 (L6) | Q1 exact · Q2 CountSketch+heap (Sum acc) | rate: raw; sum: raw | top-k: CountSketch + heap; sum_over_time: Sum acc | separate | valid, costlier | 195.401 vs 52.201 cpu ms | +| P5 (L5) | Q1 exact · Q2 CountSketch+heap | rate: raw; sum: raw | top-k: CountSketch + heap; sum_over_time: raw | separate | valid, costlier | 19.840 vs 5.220 cost/s | +| P6 (L6) | Q1 exact · Q2 CountSketch+heap (Sum acc) | rate: raw; sum: raw | top-k: CountSketch + heap; sum_over_time: Sum acc | separate | valid, costlier | 19.540 vs 5.220 cost/s | | P7 (L7) | Q1 exact · Q2 whole-expression CMS+heap | rate: raw; sum: raw | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P8 (L8) | Q1 exact · Q2 whole-expression CountSketch+heap | rate: raw; sum: raw | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | separate | valid, costlier | 568.401 vs 52.201 cpu ms | -| P9 (L9) | Q1 exact (Rate acc) · Q2 exact | rate: Rate acc; sum: raw | top-k: exact (sort → limit); sum_over_time: raw | separate | valid, costlier | 83.401 vs 52.201 cpu ms | -| P10 (L10) | Q1 exact (Rate acc) · Q2 exact (Sum acc) | rate: Rate acc; sum: raw | top-k: exact (sort → limit); sum_over_time: Sum acc | separate | valid, costlier | 80.401 vs 52.201 cpu ms | +| P8 (L8) | Q1 exact · Q2 whole-expression CountSketch+heap | rate: raw; sum: raw | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | separate | valid, costlier | 56.840 vs 5.220 cost/s | +| P9 (L9) | Q1 exact (Rate acc) · Q2 exact | rate: Rate acc; sum: raw | top-k: exact (sort → limit); sum_over_time: raw | separate | valid, costlier | 8.340 vs 5.220 cost/s | +| P10 (L10) | Q1 exact (Rate acc) · Q2 exact (Sum acc) | rate: Rate acc; sum: raw | top-k: exact (sort → limit); sum_over_time: Sum acc | separate | valid, costlier | 8.040 vs 5.220 cost/s | | P11 (L11) | Q1 exact (Rate acc) · Q2 CMS+heap | rate: Rate acc; sum: raw | top-k: Count-Min + heap; sum_over_time: raw | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | | P12 (L12) | Q1 exact (Rate acc) · Q2 CMS+heap (Sum acc) | rate: Rate acc; sum: raw | top-k: Count-Min + heap; sum_over_time: Sum acc | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P13 (L13) | Q1 exact (Rate acc) · Q2 CountSketch+heap | rate: Rate acc; sum: raw | top-k: CountSketch + heap; sum_over_time: raw | separate | valid, costlier | 195.401 vs 52.201 cpu ms | -| P14 (L14) | Q1 exact (Rate acc) · Q2 CountSketch+heap (Sum acc) | rate: Rate acc; sum: raw | top-k: CountSketch + heap; sum_over_time: Sum acc | separate | valid, costlier | 192.401 vs 52.201 cpu ms | +| P13 (L13) | Q1 exact (Rate acc) · Q2 CountSketch+heap | rate: Rate acc; sum: raw | top-k: CountSketch + heap; sum_over_time: raw | separate | valid, costlier | 19.540 vs 5.220 cost/s | +| P14 (L14) | Q1 exact (Rate acc) · Q2 CountSketch+heap (Sum acc) | rate: Rate acc; sum: raw | top-k: CountSketch + heap; sum_over_time: Sum acc | separate | valid, costlier | 19.240 vs 5.220 cost/s | | P15 (L15) | Q1 exact (Rate acc) · Q2 whole-expression CMS+heap | rate: Rate acc; sum: raw | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P16 (L16) | Q1 exact (Rate acc) · Q2 whole-expression CountSketch+heap | rate: Rate acc; sum: raw | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | separate | valid, costlier | 565.401 vs 52.201 cpu ms | -| P17 (L17) | Q1 exact (Sum acc) · Q2 exact | rate: raw; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: raw | separate | valid, costlier | 85.401 vs 52.201 cpu ms | -| P18 (L18) | Q1 exact (Sum acc) · Q2 exact (Sum acc) | rate: raw; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: Sum acc | separate | valid, costlier | 82.401 vs 52.201 cpu ms | +| P16 (L16) | Q1 exact (Rate acc) · Q2 whole-expression CountSketch+heap | rate: Rate acc; sum: raw | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | separate | valid, costlier | 56.540 vs 5.220 cost/s | +| P17 (L17) | Q1 exact (Sum acc) · Q2 exact | rate: raw; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: raw | separate | valid, costlier | 8.540 vs 5.220 cost/s | +| P18 (L18) | Q1 exact (Sum acc) · Q2 exact (Sum acc) | rate: raw; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: Sum acc | separate | valid, costlier | 8.240 vs 5.220 cost/s | | P19 (L19) | Q1 exact (Sum acc) · Q2 CMS+heap | rate: raw; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: raw | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | | P20 (L20) | Q1 exact (Sum acc) · Q2 CMS+heap (Sum acc) | rate: raw; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: Sum acc | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P21 (L21) | Q1 exact (Sum acc) · Q2 CountSketch+heap | rate: raw; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: raw | separate | valid, costlier | 197.401 vs 52.201 cpu ms | -| P22 (L22) | Q1 exact (Sum acc) · Q2 CountSketch+heap (Sum acc) | rate: raw; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: Sum acc | separate | valid, costlier | 194.401 vs 52.201 cpu ms | +| P21 (L21) | Q1 exact (Sum acc) · Q2 CountSketch+heap | rate: raw; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: raw | separate | valid, costlier | 19.740 vs 5.220 cost/s | +| P22 (L22) | Q1 exact (Sum acc) · Q2 CountSketch+heap (Sum acc) | rate: raw; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: Sum acc | separate | valid, costlier | 19.440 vs 5.220 cost/s | | P23 (L23) | Q1 exact (Sum acc) · Q2 whole-expression CMS+heap | rate: raw; sum: Sum acc | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P24 (L24) | Q1 exact (Sum acc) · Q2 whole-expression CountSketch+heap | rate: raw; sum: Sum acc | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | separate | valid, costlier | 567.401 vs 52.201 cpu ms | -| P25 (L25) | Q1 exact (Sum acc, Rate acc) · Q2 exact | rate: Rate acc; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: raw | separate | valid, costlier | 82.401 vs 52.201 cpu ms | -| P26 (L26) | Q1 exact (Sum acc, Rate acc) · Q2 exact (Sum acc) | rate: Rate acc; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: Sum acc | separate | valid, costlier | 79.401 vs 52.201 cpu ms | +| P24 (L24) | Q1 exact (Sum acc) · Q2 whole-expression CountSketch+heap | rate: raw; sum: Sum acc | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | separate | valid, costlier | 56.740 vs 5.220 cost/s | +| P25 (L25) | Q1 exact (Sum acc, Rate acc) · Q2 exact | rate: Rate acc; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: raw | separate | valid, costlier | 8.240 vs 5.220 cost/s | +| P26 (L26) | Q1 exact (Sum acc, Rate acc) · Q2 exact (Sum acc) | rate: Rate acc; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: Sum acc | separate | valid, costlier | 7.940 vs 5.220 cost/s | | P27 (L27) | Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap | rate: Rate acc; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: raw | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | | P28 (L28) | Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap (Sum acc) | rate: Rate acc; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: Sum acc | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P29 (L29) | Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap | rate: Rate acc; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: raw | separate | valid, costlier | 194.401 vs 52.201 cpu ms | -| P30 (L30) | Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap (Sum acc) | rate: Rate acc; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: Sum acc | separate | valid, costlier | 191.401 vs 52.201 cpu ms | +| P29 (L29) | Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap | rate: Rate acc; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: raw | separate | valid, costlier | 19.440 vs 5.220 cost/s | +| P30 (L30) | Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap (Sum acc) | rate: Rate acc; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: Sum acc | separate | valid, costlier | 19.140 vs 5.220 cost/s | | P31 (L31) | Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CMS+heap | rate: Rate acc; sum: Sum acc | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P32 (L32) | Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CountSketch+heap | rate: Rate acc; sum: Sum acc | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | separate | valid, costlier | 564.401 vs 52.201 cpu ms | -| P33 (L33) | Q1 exact · Q2 exact · shared input | rate: raw; sum: raw | top-k: exact (sort → limit); sum_over_time: raw | shared | valid, costlier | 59.201 vs 52.201 cpu ms | -| P34 (L34) | Q1 exact · Q2 exact (Sum acc) · shared input | rate: raw; sum: raw | top-k: exact (sort → limit); sum_over_time: Sum acc | shared | valid, costlier | 56.201 vs 52.201 cpu ms | +| P32 (L32) | Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CountSketch+heap | rate: Rate acc; sum: Sum acc | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | separate | valid, costlier | 56.440 vs 5.220 cost/s | +| P33 (L33) | Q1 exact · Q2 exact · shared input | rate: raw; sum: raw | top-k: exact (sort → limit); sum_over_time: raw | shared | valid, costlier | 5.920 vs 5.220 cost/s | +| P34 (L34) | Q1 exact · Q2 exact (Sum acc) · shared input | rate: raw; sum: raw | top-k: exact (sort → limit); sum_over_time: Sum acc | shared | valid, costlier | 5.620 vs 5.220 cost/s | | P35 (L35) | Q1 exact · Q2 CMS+heap · shared input | rate: raw; sum: raw | top-k: Count-Min + heap; sum_over_time: raw | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | | P36 (L36) | Q1 exact · Q2 CMS+heap (Sum acc) · shared input | rate: raw; sum: raw | top-k: Count-Min + heap; sum_over_time: Sum acc | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P37 (L37) | Q1 exact · Q2 CountSketch+heap · shared input | rate: raw; sum: raw | top-k: CountSketch + heap; sum_over_time: raw | shared | valid, costlier | 171.201 vs 52.201 cpu ms | -| P38 (L38) | Q1 exact · Q2 CountSketch+heap (Sum acc) · shared input | rate: raw; sum: raw | top-k: CountSketch + heap; sum_over_time: Sum acc | shared | valid, costlier | 168.201 vs 52.201 cpu ms | +| P37 (L37) | Q1 exact · Q2 CountSketch+heap · shared input | rate: raw; sum: raw | top-k: CountSketch + heap; sum_over_time: raw | shared | valid, costlier | 17.120 vs 5.220 cost/s | +| P38 (L38) | Q1 exact · Q2 CountSketch+heap (Sum acc) · shared input | rate: raw; sum: raw | top-k: CountSketch + heap; sum_over_time: Sum acc | shared | valid, costlier | 16.820 vs 5.220 cost/s | | P39 (L39) | Q1 exact · Q2 whole-expression CMS+heap · shared input | rate: raw; sum: raw | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P40 (L40) | Q1 exact · Q2 whole-expression CountSketch+heap · shared input | rate: raw; sum: raw | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | shared | valid, costlier | 541.201 vs 52.201 cpu ms | -| P41 (L41) | Q1 exact (Rate acc) · Q2 exact · shared input | rate: Rate acc; sum: raw | top-k: exact (sort → limit); sum_over_time: raw | shared | valid, costlier | 56.201 vs 52.201 cpu ms | -| P42 (L42) | Q1 exact (Rate acc) · Q2 exact (Sum acc) · shared input | rate: Rate acc; sum: raw | top-k: exact (sort → limit); sum_over_time: Sum acc | shared | valid, costlier | 53.201 vs 52.201 cpu ms | +| P40 (L40) | Q1 exact · Q2 whole-expression CountSketch+heap · shared input | rate: raw; sum: raw | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | shared | valid, costlier | 54.120 vs 5.220 cost/s | +| P41 (L41) | Q1 exact (Rate acc) · Q2 exact · shared input | rate: Rate acc; sum: raw | top-k: exact (sort → limit); sum_over_time: raw | shared | valid, costlier | 5.620 vs 5.220 cost/s | +| P42 (L42) | Q1 exact (Rate acc) · Q2 exact (Sum acc) · shared input | rate: Rate acc; sum: raw | top-k: exact (sort → limit); sum_over_time: Sum acc | shared | valid, costlier | 5.320 vs 5.220 cost/s | | P43 (L43) | Q1 exact (Rate acc) · Q2 CMS+heap · shared input | rate: Rate acc; sum: raw | top-k: Count-Min + heap; sum_over_time: raw | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | | P44 (L44) | Q1 exact (Rate acc) · Q2 CMS+heap (Sum acc) · shared input | rate: Rate acc; sum: raw | top-k: Count-Min + heap; sum_over_time: Sum acc | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P45 (L45) | Q1 exact (Rate acc) · Q2 CountSketch+heap · shared input | rate: Rate acc; sum: raw | top-k: CountSketch + heap; sum_over_time: raw | shared | valid, costlier | 168.201 vs 52.201 cpu ms | -| P46 (L46) | Q1 exact (Rate acc) · Q2 CountSketch+heap (Sum acc) · shared input | rate: Rate acc; sum: raw | top-k: CountSketch + heap; sum_over_time: Sum acc | shared | valid, costlier | 165.201 vs 52.201 cpu ms | +| P45 (L45) | Q1 exact (Rate acc) · Q2 CountSketch+heap · shared input | rate: Rate acc; sum: raw | top-k: CountSketch + heap; sum_over_time: raw | shared | valid, costlier | 16.820 vs 5.220 cost/s | +| P46 (L46) | Q1 exact (Rate acc) · Q2 CountSketch+heap (Sum acc) · shared input | rate: Rate acc; sum: raw | top-k: CountSketch + heap; sum_over_time: Sum acc | shared | valid, costlier | 16.520 vs 5.220 cost/s | | P47 (L47) | Q1 exact (Rate acc) · Q2 whole-expression CMS+heap · shared input | rate: Rate acc; sum: raw | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P48 (L48) | Q1 exact (Rate acc) · Q2 whole-expression CountSketch+heap · shared input | rate: Rate acc; sum: raw | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | shared | valid, costlier | 538.201 vs 52.201 cpu ms | -| P49 (L49) | Q1 exact (Sum acc) · Q2 exact · shared input | rate: raw; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: raw | shared | valid, costlier | 58.201 vs 52.201 cpu ms | -| P50 (L50) | Q1 exact (Sum acc) · Q2 exact (Sum acc) · shared input | rate: raw; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: Sum acc | shared | valid, costlier | 55.201 vs 52.201 cpu ms | +| P48 (L48) | Q1 exact (Rate acc) · Q2 whole-expression CountSketch+heap · shared input | rate: Rate acc; sum: raw | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | shared | valid, costlier | 53.820 vs 5.220 cost/s | +| P49 (L49) | Q1 exact (Sum acc) · Q2 exact · shared input | rate: raw; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: raw | shared | valid, costlier | 5.820 vs 5.220 cost/s | +| P50 (L50) | Q1 exact (Sum acc) · Q2 exact (Sum acc) · shared input | rate: raw; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: Sum acc | shared | valid, costlier | 5.520 vs 5.220 cost/s | | P51 (L51) | Q1 exact (Sum acc) · Q2 CMS+heap · shared input | rate: raw; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: raw | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | | P52 (L52) | Q1 exact (Sum acc) · Q2 CMS+heap (Sum acc) · shared input | rate: raw; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: Sum acc | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P53 (L53) | Q1 exact (Sum acc) · Q2 CountSketch+heap · shared input | rate: raw; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: raw | shared | valid, costlier | 170.201 vs 52.201 cpu ms | -| P54 (L54) | Q1 exact (Sum acc) · Q2 CountSketch+heap (Sum acc) · shared input | rate: raw; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: Sum acc | shared | valid, costlier | 167.201 vs 52.201 cpu ms | +| P53 (L53) | Q1 exact (Sum acc) · Q2 CountSketch+heap · shared input | rate: raw; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: raw | shared | valid, costlier | 17.020 vs 5.220 cost/s | +| P54 (L54) | Q1 exact (Sum acc) · Q2 CountSketch+heap (Sum acc) · shared input | rate: raw; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: Sum acc | shared | valid, costlier | 16.720 vs 5.220 cost/s | | P55 (L55) | Q1 exact (Sum acc) · Q2 whole-expression CMS+heap · shared input | rate: raw; sum: Sum acc | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P56 (L56) | Q1 exact (Sum acc) · Q2 whole-expression CountSketch+heap · shared input | rate: raw; sum: Sum acc | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | shared | valid, costlier | 540.201 vs 52.201 cpu ms | -| P57 (L57) | Q1 exact (Sum acc, Rate acc) · Q2 exact · shared input | rate: Rate acc; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: raw | shared | valid, costlier | 55.201 vs 52.201 cpu ms | -| P58 (L58) | Q1 exact (Sum acc, Rate acc) · Q2 exact (Sum acc) · shared input | rate: Rate acc; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: Sum acc | shared | **selected** | cheapest valid (52.201 cpu ms) | +| P56 (L56) | Q1 exact (Sum acc) · Q2 whole-expression CountSketch+heap · shared input | rate: raw; sum: Sum acc | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | shared | valid, costlier | 54.020 vs 5.220 cost/s | +| P57 (L57) | Q1 exact (Sum acc, Rate acc) · Q2 exact · shared input | rate: Rate acc; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: raw | shared | valid, costlier | 5.520 vs 5.220 cost/s | +| P58 (L58) | Q1 exact (Sum acc, Rate acc) · Q2 exact (Sum acc) · shared input | rate: Rate acc; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: Sum acc | shared | **selected** | cheapest valid (5.220 cost/s) | | P59 (L59) | Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap · shared input | rate: Rate acc; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: raw | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | | P60 (L60) | Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap (Sum acc) · shared input | rate: Rate acc; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: Sum acc | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P61 (L61) | Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap · shared input | rate: Rate acc; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: raw | shared | valid, costlier | 167.201 vs 52.201 cpu ms | -| P62 (L62) | Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap (Sum acc) · shared input | rate: Rate acc; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: Sum acc | shared | valid, costlier | 164.201 vs 52.201 cpu ms | +| P61 (L61) | Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap · shared input | rate: Rate acc; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: raw | shared | valid, costlier | 16.720 vs 5.220 cost/s | +| P62 (L62) | Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap (Sum acc) · shared input | rate: Rate acc; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: Sum acc | shared | valid, costlier | 16.420 vs 5.220 cost/s | | P63 (L63) | Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CMS+heap · shared input | rate: Rate acc; sum: Sum acc | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | -| P64 (L64) | Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CountSketch+heap · shared input | rate: Rate acc; sum: Sum acc | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | shared | valid, costlier | 537.201 vs 52.201 cpu ms | +| P64 (L64) | Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CountSketch+heap · shared input | rate: Rate acc; sum: Sum acc | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | shared | valid, costlier | 53.720 vs 5.220 cost/s | Count-Min + heap stays invalid, whole-expression or not: Q2 ranks `sum_over_time` of raw samples, and nothing in the workload declares @@ -202,13 +202,16 @@ Invariants: input" winner. It saves exactly one scan and one range node over the same choices with separate inputs. -Outcome with the built-in models: P58 (all exact, with exact accumulators for -Q1's rate and sum and Q2's sum_over_time, over the shared input) at 52.201 cpu -ms, against 79.401 for the same choices with separate inputs (P26). The scan -(23.2) and range (4.0) are priced once instead of twice. The cheapest -whole-expression CountSketch + heap plan, P64, costs 537.201: its sketch -updates 4,000,000 raw samples × (depth 125 + 1 heap update) = 504.0, against -19.001 for Q2's exact Sum accumulator (5.0) and sort + limit (14.001). The doc's other typical winners need +Outcome with the built-in models, per evaluation (CPU-ms; Stage 3 reports +these × 0.1 evaluations/s, as cost per second, see +[Stage 3 cost model](stage3-cost-model.md)): P58 (all exact, with exact +accumulators for Q1's rate and sum and Q2's sum_over_time, over the shared +input) at 52.201, or 5.220 per second, against 79.401 for the same choices +with separate inputs (P26). The scan (23.2) and range (4.0) are priced once +instead of twice. The cheapest whole-expression CountSketch + heap plan, P64, +costs 537.201: its sketch updates 4,000,000 raw samples × (depth 125 + 1 heap +update) = 504.0, against 19.001 for Q2's exact Sum accumulator (5.0) and sort ++ limit (14.001). The doc's other typical winners need window forms or materialization and are out of MVP scope. ## Ambiguities and MVP deviations diff --git a/docs/design_docs/proposals/planner-layering.md b/docs/design_docs/proposals/planner-layering.md index d9f56a38..82d70c1c 100644 --- a/docs/design_docs/proposals/planner-layering.md +++ b/docs/design_docs/proposals/planner-layering.md @@ -430,6 +430,8 @@ which is what lets one shared summary beat several cheaper independent ones: the cost of a shared summary is estimated once, with the demand of all its consumers. +The cost model's scope is in [Stage 3 Cost Model](stage3-cost-model.md). + ### 4. Execution Execution runs outside ASAPPlanner. The deployment runs the selected plan as diff --git a/docs/design_docs/proposals/stage3-cost-model.md b/docs/design_docs/proposals/stage3-cost-model.md new file mode 100644 index 00000000..84030a6b --- /dev/null +++ b/docs/design_docs/proposals/stage3-cost-model.md @@ -0,0 +1,202 @@ +# Stage 3 Cost Model + +> Status: partly implemented. Audience: planner designers and architects. +> Scope: the cost that #509 Stage 3 (plan selection) minimizes. +> Companion: [ASAPPlanner layering](planner-layering.md), Stage 3. + +## Goal + +Stage 3 selects the cheapest valid physical candidate for the whole workload. +Candidates differ in what they compute, and also in when they compute it: at +ingestion time, continuously, or at query time, on each evaluation. To compare +them, every candidate is priced in one unit, **cost per second of wall time**, +for the workload in steady state. + +## What Stage 3 prices + +| Priced | Not priced (yet) | +|---|---| +| CPU work of every operator, at ingestion time or at query time | Transient query-time memory | +| Bytes a scan reads | Storage tier and retention: disk versus memory, and for how long (S3, with panes) | +| Memory held across evaluations by ingestion-time state that query time reads | Network, parallelism and partitioning | +| | Latency bounds and deployment capabilities (separate checks, not cost) | +| | One-time setup work such as backfill (deferred, S5) | + +Accuracy is a check, not a cost. Stage 3 rejects candidates that miss a +target, then prices the rest. + +## Time basis + +The price of a node depends on when the node runs. + +**Ingestion-time node.** The node runs as data arrives. It is priced over one +second of ingested rows: + +```text +cost_ingest(n) = c_cpu · ops(n, λ rows) + c_scan · scan_bytes(n, λ rows) + memory(n) +``` + +λ is the ingestion rate in rows per second. It is +`DataWorkload.ingestion_rate` if declared, else `input_cardinality / +data_ingestion_interval` (one sample per series per interval), else the +default in the statistics table below. + +**Query-time node.** The node runs on each evaluation of a query that reads it. +It is priced per evaluation, times its evaluation rate `r(n)`: + +```text +cost_query(n) = r(n) · (c_cpu · ops(n, one evaluation) + c_scan · scan_bytes(n, one evaluation)) +``` + +A query-time scan reads as far back as the time ranges reading it reach: +the longest range plus its offset. Each range then passes only its own +span. For example, `x[1y] offset 2y` scans 3 years, and a 1-year range over a +shared 5-year scan passes 1 year of rows. + +**Workload cost** is the sum over the candidate's DAG nodes: +`cost(P) = Σ_n cost(n)`, in cost per second. + +## Recurrence and horizon + +`r(n)` comes from the recurrence of the roots (queries) whose DAG reaches `n`: + +* **Repeating roots** with interval `t` contribute `1 / t` evaluations per + second. Roots with **equal intervals count once**, because those queries are + evaluated together and read one result. Distinct intervals add: + `r = Σ_{distinct t} 1 / t`. +* **Estimated-rate roots** contribute their expected rate. +* **One-off, scheduled and unknown roots** run as one batch. Their work is + amortized over a horizon `H`: the most invocations among them, divided by + `H`. `Unknown` counts as one invocation. `H` is 1 h by default (S5). + +So a one-off query costs `invocations · per-evaluation cost / H`. That keeps a +one-off query comparable with a repeating one, and puts ingestion-time +maintenance for a one-off query at its true disadvantage: it runs every second +for an answer that is needed once per hour. + +## Memory term + +State that is retained across evaluations costs memory for as long as it is +kept: + +```text +memory(n) = w_mem · retained_bytes(n) +w_mem = 1.25e-7 cost / (byte · s) +``` + +**Derivation.** One cost unit is one CPU-millisecond. 1 GB of memory is +priced like 1/8 of a vCPU, the memory-to-core ratio of memory-optimized cloud +instances (8 GB per vCPU): 125 CPU-ms per second for 10⁹ bytes, or 1.25e-7 +per byte per second. + +**What counts as retained.** Only ingestion-time state that a query-time node +reads: it lives between evaluations. Retained bytes are + +```text +retained_bytes(n) = 2 · groups(n) · state_bytes(n) +``` + +where `state_bytes` is the summary's state size (for example +`8 · width · depth` for Count-Min) or the row size of an exact state. The +factor 2 covers the window being built plus the completed window that +queries read. Query-time state, built and discarded within one evaluation, is +transient and not priced. + +## Calibration + +Every coefficient lives in `Stage3Calibration`, carried by `PlanningModels` +and set with `PlanningModels::with_calibration`: + +| Field | Default (`ILLUSTRATIVE`, `illustrative-v2`) | Meaning | +|---|---|---| +| `cost_per_cpu_op` | 1e-6 | 1 ns of CPU per operation, in CPU-ms | +| `cost_per_scan_byte` | 1e-7 | per byte read by a scan | +| `cost_per_retained_byte_second` | 1.25e-7 | `w_mem` | +| `horizon_s` | 3600 | `H` | +| `version` | `illustrative-v2` | reported in each candidate's cost `source` | + +A deployment that prices memory differently, for example memory-rich +instances, lowers `cost_per_retained_byte_second`; one that plans for a daily +batch raises `horizon_s` to 86 400. + +## Default statistics + +When the workload does not declare a statistic, Stage 3 uses these defaults. +They are **illustrative**: they order candidates plausibly, but they are not +measured and the absolute costs mean little. + +| Statistic | Default | +|---|---| +| Series (`input_cardinality`) | 1 000 | +| λ, rows per second | 1 000 | +| Lookback of a scan read by no time range | 1 min | +| Groups of a `by (...)` reduction | 100 | +| Summary update operations per row | sketch depth (+1 with a heap) | +| Summary state bytes | `8 · width · depth` (+24 per heap entry); 1 KiB for other sketches | +| Row bytes | 8 per plain value, 16 per string | + +## Interaction with selection + +**Additivity.** Stage 3's dynamic program over target nesting +([#572](https://github.com/ProjectASAP/ASAPPlanner/issues/572)) needs cost to +be a sum over nodes, with a choice for one target changing only its own nodes. +The per-second cost keeps this: each node's price depends on its own +statistics, its timing, and the set of roots that reach it. The roots that +reach a node do not depend on how other targets are realized. The program +still verifies additivity for every nested pair and falls back to enumeration +when it does not hold. + +**Shared nodes are charged once.** One DAG node is one computation. A node +reached by several queries is priced once, at the evaluation rate of the +distinct intervals of those queries. That is, a shared node is materialized +at query time within the batch: computed once per evaluation and read by every +consumer. + +**Decision: "not materialized" is a DAG shape, not a cost rule.** #509 Stage 2 +lists "not materialized" as an option for a multi-consumer sub-DAG (Example 4, +A3): each consumer recomputes it. Stage 2 represents that option by duplicating +the node for each consuming query, so the duplicates are priced separately. The +cost model never discounts or multiplies a node by its number of consumers. + +## Worked example: Example 1 + +Example 1 has two panels, each repeating every 10 s, over 1 000 000 series +ingested every 15 s (λ = 66 667 rows/s). Every node runs at query time, since +Stage 2 does not choose ingestion time yet. Both roots have the same 10-s +interval, so every node, shared or not, has `r = 0.1`/s. + +P58, the selected plan (all exact, the input shared by both queries): + +| Node | Reached by | Work per evaluation | Per evaluation | r (/s) | Per second | +|---|---|---|---|---|---| +| Scan | Q1, Q2 | 4 000 000 samples (1 min × λ) | 23.2 | 0.1 | 2.32 | +| Time range 1 min | Q1, Q2 | 4 000 000 rows | 4.0 | 0.1 | 0.40 | +| Rate accumulator | Q1 | 4 000 000 rows into 1 000 000 states | 4.0 | 0.1 | 0.40 | +| Finalize | Q1 | 1 000 000 accumulators | 1.0 | 0.1 | 0.10 | +| Sum by `job` accumulator | Q1 | 1 000 000 rows into 100 states | 1.0 | 0.1 | 0.10 | +| Finalize | Q1 | 100 accumulators | 0.0001 | 0.1 | 0.00001 | +| Sum accumulator (`sum_over_time`) | Q2 | 4 000 000 rows into 1 000 000 states | 4.0 | 0.1 | 0.40 | +| Finalize | Q2 | 1 000 000 accumulators | 1.0 | 0.1 | 0.10 | +| Sort by `job` | Q2 | 1 000 000 rows in 100 partitions | 14.0 | 0.1 | 1.40 | +| Limit 10 per `job` | Q2 | 1 000 rows | 0.001 | 0.1 | 0.0001 | +| **Total** | | | **52.201** | | **5.220** | + +The runner-up is P42 at 5.320 per second; the same choices with separate +inputs (P26) cost 7.940, because the scan and range are charged twice. + +Before per-second pricing, P58 cost 52.201 CPU-ms per workload evaluation. Now +it costs 52.201 × 0.1 = 5.220 per second. Every other candidate scales by the +same factor, so the ranking is unchanged. With uniform recurrence and no +ingestion-time nodes the new unit is a rescaling. It changes the ranking once +recurrences differ, or once Stage 2 offers ingestion-time candidates that pay +λ-rate maintenance and memory instead of per-evaluation rebuilds. + +## Out of scope for now + +* **Storage tier and retention (S3).** Retained state is priced as memory. + Choosing disk versus memory and how long to keep state comes with panes. +* **Calibration from measurements.** The coefficients and default statistics + are illustrative. Fitting them to measured operator costs, and pricing with + observed statistics such as group counts, is future work. +* **Backfill.** Building ingestion-time state over existing data when a query + is installed is a one-time cost and is not priced (S5). diff --git a/tools/dag-viewer/examples/planner-layering-example1.json b/tools/dag-viewer/examples/planner-layering-example1.json index 94d8bcea..850f2c37 100644 --- a/tools/dag-viewer/examples/planner-layering-example1.json +++ b/tools/dag-viewer/examples/planner-layering-example1.json @@ -137227,1754 +137227,1754 @@ "P1": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "4": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "5": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "6": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "7": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "8": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 86.40100000000001, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 8.640099999999999, + "unit": "cost_per_second" }, "P10": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "10": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "5": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "7": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "8": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "9": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 80.40100000000001, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 8.0401, + "unit": "cost_per_second" }, "P13": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "5": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "7": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "8": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "9": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 195.401, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 19.540100000000002, + "unit": "cost_per_second" }, "P14": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "10": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "5": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "7": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "8": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "9": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 192.401, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 19.2401, + "unit": "cost_per_second" }, "P16": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "5": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "7": { - "cost": 504.0, - "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + "cost": 50.400000000000006, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126; x 0.1000 evaluations/s" }, "8": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 565.401, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 56.54010000000001, + "unit": "cost_per_second" }, "P17": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "4": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "5": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "7": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "8": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "9": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 85.40110000000001, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 8.540109999999999, + "unit": "cost_per_second" }, "P18": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "10": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "4": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "5": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "7": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "8": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "9": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 82.40110000000001, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 8.24011, + "unit": "cost_per_second" }, "P2": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "4": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "5": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "7": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "8": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "9": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 83.40100000000001, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 8.3401, + "unit": "cost_per_second" }, "P21": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "4": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "5": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "7": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "8": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "9": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 197.4011, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 19.74011, + "unit": "cost_per_second" }, "P22": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "10": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "4": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "5": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "7": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "8": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "9": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 194.4011, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 19.44011, + "unit": "cost_per_second" }, "P24": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "4": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "5": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "7": { - "cost": 504.0, - "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + "cost": 50.400000000000006, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126; x 0.1000 evaluations/s" }, "8": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 567.4011, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 56.74011000000001, + "unit": "cost_per_second" }, "P25": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "10": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "5": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "6": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "7": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "8": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "9": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 82.40110000000001, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 8.24011, + "unit": "cost_per_second" }, "P26": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "10": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "11": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "5": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "6": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "7": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "8": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "9": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 79.40110000000001, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 7.940110000000001, + "unit": "cost_per_second" }, "P29": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "10": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "5": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "6": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "7": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "8": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "9": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 194.4011, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 19.44011, + "unit": "cost_per_second" }, "P30": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "10": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "11": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "5": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "6": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "7": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "8": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "9": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 191.4011, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 19.140110000000004, + "unit": "cost_per_second" }, "P32": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "5": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "6": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "7": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "8": { - "cost": 504.0, - "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + "cost": 50.400000000000006, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126; x 0.1000 evaluations/s" }, "9": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 564.4011, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 56.44011000000001, + "unit": "cost_per_second" }, "P33": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "4": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "5": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "6": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 59.201, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 5.9201, + "unit": "cost_per_second" }, "P34": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "4": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "5": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "6": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "7": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 56.201, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 5.6201, + "unit": "cost_per_second" }, "P37": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "4": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "5": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "6": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 171.201, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 17.1201, + "unit": "cost_per_second" }, "P38": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "4": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "5": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "6": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "7": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 168.201, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 16.8201, + "unit": "cost_per_second" }, "P40": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "4": { - "cost": 504.0, - "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + "cost": 50.400000000000006, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126; x 0.1000 evaluations/s" }, "5": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 541.201, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 54.12010000000001, + "unit": "cost_per_second" }, "P41": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "5": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "6": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "7": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 56.201, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 5.6201, + "unit": "cost_per_second" }, "P42": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "5": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "6": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "7": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "8": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 53.201, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 5.3201, + "unit": "cost_per_second" }, "P45": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "5": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "6": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "7": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 168.201, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 16.8201, + "unit": "cost_per_second" }, "P46": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "5": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "6": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "7": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "8": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 165.201, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 16.520100000000003, + "unit": "cost_per_second" }, "P48": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "5": { - "cost": 504.0, - "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + "cost": 50.400000000000006, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126; x 0.1000 evaluations/s" }, "6": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 538.201, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 53.82010000000001, + "unit": "cost_per_second" }, "P49": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "4": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "5": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "6": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "7": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 58.201100000000004, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 5.82011, + "unit": "cost_per_second" }, "P5": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "4": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "5": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "6": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "7": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "8": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 198.401, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 19.8401, + "unit": "cost_per_second" }, "P50": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "4": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "5": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "6": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "7": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "8": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 55.201100000000004, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 5.52011, + "unit": "cost_per_second" }, "P53": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "4": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "5": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "6": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "7": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 170.20110000000003, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 17.02011, + "unit": "cost_per_second" }, "P54": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "4": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "5": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "6": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "7": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "8": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 167.20110000000003, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 16.720110000000002, + "unit": "cost_per_second" }, "P56": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "4": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "5": { - "cost": 504.0, - "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + "cost": 50.400000000000006, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126; x 0.1000 evaluations/s" }, "6": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 540.2011, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 54.02011000000001, + "unit": "cost_per_second" }, "P57": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "5": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "6": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "7": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "8": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 55.201100000000004, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 5.52011, + "unit": "cost_per_second" }, "P58": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "5": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "7": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "8": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "9": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 52.201100000000004, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 5.22011, + "unit": "cost_per_second" }, "P6": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "4": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "5": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "7": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "8": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "9": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 195.401, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 19.5401, + "unit": "cost_per_second" }, "P61": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "5": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "6": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "7": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "8": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 167.20110000000003, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 16.720110000000002, + "unit": "cost_per_second" }, "P62": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "5": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "7": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "8": { - "cost": 126.0, - "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + "cost": 12.600000000000001, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126; x 0.1000 evaluations/s" }, "9": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 164.20110000000003, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 16.42011, + "unit": "cost_per_second" }, "P64": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 1.0, - "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + "cost": 0.1, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1; x 0.1000 evaluations/s" }, "5": { - "cost": 0.00009999999999999999, - "detail": "finalize 100 accumulators" + "cost": 9.999999999999999e-6, + "detail": "finalize 100 accumulators; x 0.1000 evaluations/s" }, "6": { - "cost": 504.0, - "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + "cost": 50.400000000000006, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126; x 0.1000 evaluations/s" }, "7": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 537.2011, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 53.720110000000005, + "unit": "cost_per_second" }, "P8": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "3": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "4": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "5": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "6": { - "cost": 504.0, - "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + "cost": 50.400000000000006, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126; x 0.1000 evaluations/s" }, "7": { - "cost": 0.001, - "detail": "estimate 1000 rows from 100 states" + "cost": 0.0001, + "detail": "estimate 1000 rows from 100 states; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 568.401, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 56.84010000000001, + "unit": "cost_per_second" }, "P9": { "per_node": { "0": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "1": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "2": { - "cost": 4.0, - "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + "cost": 0.4, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1; x 0.1000 evaluations/s" }, "3": { - "cost": 1.0, - "detail": "finalize 1000000 accumulators" + "cost": 0.1, + "detail": "finalize 1000000 accumulators; x 0.1000 evaluations/s" }, "4": { - "cost": 2.0, - "detail": "hash aggregate 1000000 rows into 100 groups" + "cost": 0.2, + "detail": "hash aggregate 1000000 rows into 100 groups; x 0.1000 evaluations/s" }, "5": { - "cost": 23.2, - "detail": "scan 4000000 samples" + "cost": 2.32, + "detail": "scan 4000000 samples; x 0.1000 evaluations/s" }, "6": { - "cost": 4.0, - "detail": "time range 60s: pass 4000000 rows" + "cost": 0.4, + "detail": "time range 60s: pass 4000000 rows; x 0.1000 evaluations/s" }, "7": { - "cost": 8.0, - "detail": "hash aggregate 4000000 rows into 1000000 groups" + "cost": 0.8, + "detail": "hash aggregate 4000000 rows into 1000000 groups; x 0.1000 evaluations/s" }, "8": { - "cost": 14.0, - "detail": "sort 1000000 rows in 100 partitions" + "cost": 1.4000000000000001, + "detail": "sort 1000000 rows in 100 partitions; x 0.1000 evaluations/s" }, "9": { - "cost": 0.001, - "detail": "limit to 1000 rows" + "cost": 0.0001, + "detail": "limit to 1000 rows; x 0.1000 evaluations/s" } }, - "source": "analytical-cost-v1 (illustrative statistics)", - "total": 83.40100000000001, - "unit": "cpu_ms_per_workload_evaluation" + "source": "analytical-cost-v2 (illustrative statistics, calibration illustrative-v2)", + "total": 8.3401, + "unit": "cost_per_second" } }, "rejected": [ @@ -139100,197 +139100,197 @@ }, { "id": "P1", - "reason": "costlier: 86.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 8.640 vs 5.220 cost_per_second", "valid": true }, { "id": "P2", - "reason": "costlier: 83.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 8.340 vs 5.220 cost_per_second", "valid": true }, { "id": "P5", - "reason": "costlier: 198.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 19.840 vs 5.220 cost_per_second", "valid": true }, { "id": "P6", - "reason": "costlier: 195.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 19.540 vs 5.220 cost_per_second", "valid": true }, { "id": "P8", - "reason": "costlier: 568.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 56.840 vs 5.220 cost_per_second", "valid": true }, { "id": "P9", - "reason": "costlier: 83.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 8.340 vs 5.220 cost_per_second", "valid": true }, { "id": "P10", - "reason": "costlier: 80.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 8.040 vs 5.220 cost_per_second", "valid": true }, { "id": "P13", - "reason": "costlier: 195.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 19.540 vs 5.220 cost_per_second", "valid": true }, { "id": "P14", - "reason": "costlier: 192.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 19.240 vs 5.220 cost_per_second", "valid": true }, { "id": "P16", - "reason": "costlier: 565.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 56.540 vs 5.220 cost_per_second", "valid": true }, { "id": "P17", - "reason": "costlier: 85.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 8.540 vs 5.220 cost_per_second", "valid": true }, { "id": "P18", - "reason": "costlier: 82.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 8.240 vs 5.220 cost_per_second", "valid": true }, { "id": "P21", - "reason": "costlier: 197.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 19.740 vs 5.220 cost_per_second", "valid": true }, { "id": "P22", - "reason": "costlier: 194.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 19.440 vs 5.220 cost_per_second", "valid": true }, { "id": "P24", - "reason": "costlier: 567.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 56.740 vs 5.220 cost_per_second", "valid": true }, { "id": "P25", - "reason": "costlier: 82.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 8.240 vs 5.220 cost_per_second", "valid": true }, { "id": "P26", - "reason": "costlier: 79.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 7.940 vs 5.220 cost_per_second", "valid": true }, { "id": "P29", - "reason": "costlier: 194.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 19.440 vs 5.220 cost_per_second", "valid": true }, { "id": "P30", - "reason": "costlier: 191.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 19.140 vs 5.220 cost_per_second", "valid": true }, { "id": "P32", - "reason": "costlier: 564.401 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 56.440 vs 5.220 cost_per_second", "valid": true }, { "id": "P33", - "reason": "costlier: 59.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 5.920 vs 5.220 cost_per_second", "valid": true }, { "id": "P34", - "reason": "costlier: 56.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 5.620 vs 5.220 cost_per_second", "valid": true }, { "id": "P37", - "reason": "costlier: 171.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 17.120 vs 5.220 cost_per_second", "valid": true }, { "id": "P38", - "reason": "costlier: 168.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 16.820 vs 5.220 cost_per_second", "valid": true }, { "id": "P40", - "reason": "costlier: 541.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 54.120 vs 5.220 cost_per_second", "valid": true }, { "id": "P41", - "reason": "costlier: 56.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 5.620 vs 5.220 cost_per_second", "valid": true }, { "id": "P42", - "reason": "costlier: 53.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 5.320 vs 5.220 cost_per_second", "valid": true }, { "id": "P45", - "reason": "costlier: 168.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 16.820 vs 5.220 cost_per_second", "valid": true }, { "id": "P46", - "reason": "costlier: 165.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 16.520 vs 5.220 cost_per_second", "valid": true }, { "id": "P48", - "reason": "costlier: 538.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 53.820 vs 5.220 cost_per_second", "valid": true }, { "id": "P49", - "reason": "costlier: 58.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 5.820 vs 5.220 cost_per_second", "valid": true }, { "id": "P50", - "reason": "costlier: 55.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 5.520 vs 5.220 cost_per_second", "valid": true }, { "id": "P53", - "reason": "costlier: 170.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 17.020 vs 5.220 cost_per_second", "valid": true }, { "id": "P54", - "reason": "costlier: 167.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 16.720 vs 5.220 cost_per_second", "valid": true }, { "id": "P56", - "reason": "costlier: 540.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 54.020 vs 5.220 cost_per_second", "valid": true }, { "id": "P57", - "reason": "costlier: 55.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 5.520 vs 5.220 cost_per_second", "valid": true }, { "id": "P61", - "reason": "costlier: 167.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 16.720 vs 5.220 cost_per_second", "valid": true }, { "id": "P62", - "reason": "costlier: 164.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 16.420 vs 5.220 cost_per_second", "valid": true }, { "id": "P64", - "reason": "costlier: 537.201 vs 52.201 cpu_ms_per_workload_evaluation", + "reason": "costlier: 53.720 vs 5.220 cost_per_second", "valid": true } ],