Repository navigation
fix(planner): apply shared cleanup thresholds #786
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
base: main
Are you sure you want to change the base?
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -2,35 +2,46 @@ use asap_types::aggregation_reference::AggregationReference; | |
| use asap_types::inference_config::InferenceConfig; | ||
| use asap_types::query_config::QueryConfig; | ||
| use asap_types::streaming_config::StreamingConfig; | ||
| use std::collections::HashMap; | ||
|
|
||
| use super::solution::{OptimizerSolution, QueryMethod}; | ||
|
|
||
| /// Translate an `OptimizerSolution` into the deployment artifacts consumed by | ||
| /// Arroyo and the query engine. | ||
| /// | ||
| pub fn translate(solution: &OptimizerSolution) -> (StreamingConfig, InferenceConfig) { | ||
| let streaming_config = build_streaming_config(solution); | ||
| let inference_config = build_inference_config(solution); | ||
| let streaming_config = build_streaming_config(solution, &inference_config); | ||
| (streaming_config, inference_config) | ||
| } | ||
|
|
||
| fn build_streaming_config(solution: &OptimizerSolution) -> StreamingConfig { | ||
| // Deployed configs map directly to AggregationConfigs — the types are the same. | ||
| StreamingConfig::new(solution.deployed_configs().clone()) | ||
| fn build_streaming_config( | ||
| solution: &OptimizerSolution, | ||
| inference_config: &InferenceConfig, | ||
| ) -> StreamingConfig { | ||
| let mut configs = solution.deployed_configs().clone(); | ||
| for (aggregation_id, threshold) in read_count_thresholds(inference_config) { | ||
| let config = configs | ||
| .get_mut(&aggregation_id) | ||
| .expect("every assigned aggregation must be deployed"); | ||
| config.num_aggregates_to_retain = None; | ||
| config.read_count_threshold = Some(threshold); | ||
| } | ||
| StreamingConfig::new(configs) | ||
| } | ||
|
|
||
| fn build_inference_config(solution: &OptimizerSolution) -> InferenceConfig { | ||
| use asap_types::enums::{CleanupPolicy, QueryLanguage}; | ||
|
|
||
| let mut inference = InferenceConfig::new(QueryLanguage::promql, CleanupPolicy::NoCleanup); | ||
| let mut inference = InferenceConfig::new(QueryLanguage::promql, CleanupPolicy::ReadBased); | ||
|
|
||
| for assignment in &solution.assignments { | ||
| let aggregation_id = assignment.aggregation_id; | ||
| let retain = retention_count_for_assignment(&assignment.query_method); | ||
| let agg_ref = AggregationReference::new(aggregation_id, Some(retain)); | ||
| let cleanup_count = cleanup_count_for_assignment(&assignment.query_method); | ||
| let agg_ref = | ||
| AggregationReference::with_read_count_threshold(aggregation_id, Some(cleanup_count)); | ||
| let key_ref = assignment.key_aggregation_id.map(|key_id| { | ||
| let key_retain = solution.deployed_configs()[&key_id].num_aggregates_to_retain; | ||
| AggregationReference::new(key_id, key_retain) | ||
| AggregationReference::with_read_count_threshold(key_id, Some(cleanup_count)) | ||
|
Contributor
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. The key tracker gets the value aggregation's threshold, which doesn't match how often its panes are read. Example: W = 120s, S = 60s, range = 240s gives a key threshold of 2. Each key pane is deleted after 2 of the 4 runs that need it. Later runs then build the key set from missing deltas, and keys drop out of the query output. |
||
| }); | ||
|
|
||
| for query_string in &assignment.item.query_strings { | ||
|
|
@@ -47,9 +58,26 @@ fn build_inference_config(solution: &OptimizerSolution) -> InferenceConfig { | |
| inference | ||
| } | ||
|
|
||
| /// For a Merge assignment, the number of retained windows to configure in the | ||
| /// inference config (num_aggregates_to_retain on the AggregationReference). | ||
| pub fn retention_count_for_assignment(query_method: &QueryMethod) -> u64 { | ||
| fn read_count_thresholds(inference_config: &InferenceConfig) -> HashMap<u64, u64> { | ||
| let mut thresholds = HashMap::new(); | ||
| for query_config in &inference_config.query_configs { | ||
| for aggregation in &query_config.aggregations { | ||
| let Some(cleanup_count) = aggregation.read_count_threshold else { | ||
| continue; | ||
| }; | ||
| let threshold = thresholds | ||
| .entry(aggregation.aggregation_id) | ||
| .or_insert(0_u64); | ||
| *threshold = threshold | ||
| .checked_add(cleanup_count) | ||
| .expect("read-count threshold overflowed"); | ||
| } | ||
| } | ||
| thresholds | ||
| } | ||
|
|
||
| /// Number of aggregate reads an assignment needs before its windows may be cleaned up. | ||
| pub fn cleanup_count_for_assignment(query_method: &QueryMethod) -> u64 { | ||
| match query_method { | ||
| QueryMethod::Direct => 1, | ||
| QueryMethod::Merge { num_windows } => *num_windows, | ||
|
Contributor
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Memory grows without limit when queries run less often than the window slides.
In both cases those windows are never deleted. |
||
|
|
@@ -77,3 +105,116 @@ impl TranslationSummary { | |
| } | ||
| } | ||
| } | ||
|
|
||
| #[cfg(test)] | ||
| mod tests { | ||
| use std::collections::HashMap; | ||
|
|
||
| use asap_types::aggregation_config::AggregationConfig; | ||
| use asap_types::enums::{CleanupPolicy, WindowType}; | ||
| use asap_types::query_requirements::QueryRequirements; | ||
| use promql_utilities::data_model::KeyByLabelNames; | ||
| use promql_utilities::query_logics::enums::{AggregationType, Statistic}; | ||
|
|
||
| use super::super::solution::{AQEAssignment, OptimizerItem}; | ||
| use super::*; | ||
|
|
||
| fn assignment(aggregation_id: u64, query_method: QueryMethod) -> AQEAssignment { | ||
| AQEAssignment { | ||
| item: OptimizerItem { | ||
| requirements: QueryRequirements { | ||
| metric: "metric".into(), | ||
| statistics: vec![Statistic::Sum], | ||
| data_range_ms: 60_000, | ||
| grouping_labels: KeyByLabelNames::empty(), | ||
| spatial_filter_normalized: String::new(), | ||
| topk_count_events: None, | ||
| topk_by_labels: None, | ||
| }, | ||
| query_strings: vec!["sum(metric)".into()], | ||
| query_frequency_hz: 1.0, | ||
| t_repeat_ms: 60_000, | ||
| accuracy_sla: 0.01, | ||
| latency_sla: 1.0, | ||
| }, | ||
| aggregation_id, | ||
| key_aggregation_id: None, | ||
| query_method, | ||
| estimated_query_cost_per_sec: 0.0, | ||
| } | ||
| } | ||
|
|
||
| #[test] | ||
| fn translation_uses_read_based_cleanup_and_sums_shared_thresholds() { | ||
| let mut solution = OptimizerSolution::empty(); | ||
| let aggregation_id = solution.register_config(AggregationConfig::new( | ||
| 0, | ||
| AggregationType::Sum, | ||
| "sum".into(), | ||
| HashMap::new(), | ||
| KeyByLabelNames::empty(), | ||
| KeyByLabelNames::empty(), | ||
| KeyByLabelNames::empty(), | ||
| String::new(), | ||
| 60_000, | ||
| 60_000, | ||
| WindowType::Tumbling, | ||
| String::new(), | ||
| "metric".into(), | ||
| None, | ||
| None, | ||
| None, | ||
| None, | ||
| )); | ||
| solution.assignments = vec![ | ||
| assignment(aggregation_id, QueryMethod::Merge { num_windows: 3 }), | ||
| assignment(aggregation_id, QueryMethod::Direct), | ||
| ]; | ||
|
|
||
| let (streaming, inference) = translate(&solution); | ||
|
|
||
| assert_eq!(inference.cleanup_policy, CleanupPolicy::ReadBased); | ||
| assert_eq!(streaming[aggregation_id].read_count_threshold, Some(4)); | ||
| assert!(inference.query_configs.iter().all(|query_config| { | ||
| query_config.aggregations.iter().all(|aggregation| { | ||
| aggregation.num_aggregates_to_retain.is_none() | ||
| && aggregation.read_count_threshold.is_some() | ||
| }) | ||
| })); | ||
| } | ||
|
|
||
| #[test] | ||
| fn translation_counts_each_query_string_in_a_shared_threshold() { | ||
| let mut solution = OptimizerSolution::empty(); | ||
| let aggregation_id = solution.register_config(AggregationConfig::new( | ||
| 0, | ||
| AggregationType::Sum, | ||
| "sum".into(), | ||
| HashMap::new(), | ||
| KeyByLabelNames::empty(), | ||
| KeyByLabelNames::empty(), | ||
| KeyByLabelNames::empty(), | ||
| String::new(), | ||
| 60_000, | ||
| 60_000, | ||
| WindowType::Tumbling, | ||
| String::new(), | ||
| "metric".into(), | ||
| None, | ||
| None, | ||
| None, | ||
| None, | ||
| )); | ||
| let mut shared_assignment = | ||
| assignment(aggregation_id, QueryMethod::Merge { num_windows: 3 }); | ||
| shared_assignment | ||
| .item | ||
| .query_strings | ||
| .push("sum by (job) (metric)".into()); | ||
| solution.assignments = vec![shared_assignment]; | ||
|
|
||
| let (streaming, _) = translate(&solution); | ||
|
|
||
| assert_eq!(streaming[aggregation_id].read_count_threshold, Some(6)); | ||
| } | ||
| } | ||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
Subtract windows are deleted while queries still need them (this comment is about
Subtract => 2at line 89, which takes effect because of the switch toReadBasedhere).Subtract => 2assumes the engine reads only two prefix-sum checkpoints. The engine has no subtract path (IncrementalMergeris listed as future work inwindow_merger), so Subtract runs as a normal range query that reads all n windows on every run.Example:
sum_over_time(m[5m])with t_repeat = 60s and scrape = 60s. This picks tumbling W = 60s with n = 5, and Sum is subtractable, so the method is Subtract with threshold 2. Each window is needed by 5 consecutive runs but is deleted after the 2nd read. Runs 3–5 merge only about 2 of the 5 windows, so the sum comes back too low and no error is raised.