diff --git a/.github/workflows/python.yml b/.github/workflows/python.yml index d185ecf3..4a00b872 100644 --- a/.github/workflows/python.yml +++ b/.github/workflows/python.yml @@ -39,6 +39,7 @@ jobs: utilities: ${{ steps.filter.outputs.utilities }} prometheus_exporters: ${{ steps.filter.outputs.prometheus_exporters }} execution_utilities: ${{ steps.filter.outputs.execution_utilities }} + dataset_analysis: ${{ steps.filter.outputs.dataset_analysis }} steps: - uses: actions/checkout@v4 - uses: dorny/paths-filter@v2 @@ -57,6 +58,8 @@ jobs: execution_utilities: - 'asap-tools/execution-utilities/**' - 'asap-common/dependencies/py/**' + dataset_analysis: + - 'asap-tools/dataset-analysis/**' test-prometheus-client: needs: detect-changes @@ -137,6 +140,37 @@ jobs: working-directory: asap-tools/experiments run: python -m unittest discover -s tests -p 'test_*.py' -v + test-dataset-analysis: + needs: detect-changes + if: needs.detect-changes.outputs.dataset_analysis == 'true' + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - name: Set up Python + uses: actions/setup-python@v4 + with: + python-version: '3.11' + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install black==24.8.0 flake8==6.1.0 isort==5.13.2 mypy==1.14.1 types-PyYAML + pip install -r asap-tools/dataset-analysis/requirements.txt + - name: Check formatting with Black + working-directory: asap-tools/dataset-analysis + run: black --check --diff . + - name: Check import sorting with isort + working-directory: asap-tools/dataset-analysis + run: isort --check-only --diff --settings-file .isort.cfg . + - name: Lint with flake8 + working-directory: asap-tools/dataset-analysis + run: flake8 . --config=../../.flake8 --count --show-source --statistics + - name: Type check with mypy + working-directory: asap-tools/dataset-analysis + run: mypy . --config-file=.mypy.ini + - name: Run tests + working-directory: asap-tools/dataset-analysis + run: python -m unittest discover -s tests -p 'test_*.py' -v + test-prometheus-exporters: needs: detect-changes if: needs.detect-changes.outputs.prometheus_exporters == 'true' diff --git a/asap-tools/dataset-analysis/.gitignore b/asap-tools/dataset-analysis/.gitignore new file mode 100644 index 00000000..89f9ac04 --- /dev/null +++ b/asap-tools/dataset-analysis/.gitignore @@ -0,0 +1 @@ +out/ diff --git a/asap-tools/dataset-analysis/.isort.cfg b/asap-tools/dataset-analysis/.isort.cfg new file mode 100644 index 00000000..b9fb3f3e --- /dev/null +++ b/asap-tools/dataset-analysis/.isort.cfg @@ -0,0 +1,2 @@ +[settings] +profile=black diff --git a/asap-tools/dataset-analysis/.mypy.ini b/asap-tools/dataset-analysis/.mypy.ini new file mode 100644 index 00000000..8cb50b6d --- /dev/null +++ b/asap-tools/dataset-analysis/.mypy.ini @@ -0,0 +1,4 @@ +[mypy] +files = "**/*.py" +ignore_missing_imports = True +disable_error_code = import-untyped diff --git a/asap-tools/dataset-analysis/README.md b/asap-tools/dataset-analysis/README.md new file mode 100644 index 00000000..ded14b96 --- /dev/null +++ b/asap-tools/dataset-analysis/README.md @@ -0,0 +1,208 @@ +# Dataset skew analysis + +Measures how skewed real observability traces are, so sketch benchmarks can +be run over a realistic range of skew rather than a single guessed value. + +Two tasks: + +1. **Label query sets per dataset.** `queries/*.yaml` lists PromQL queries + over each trace, each in an instant form (`sum by (user) (cpu_rate)`) + and/or a range form (`sum by (user) (sum_over_time(cpu_rate[5m]))`), + with the table, group-by labels and value column it reads. +2. **Lower / MLE / upper skew per distribution.** `fit_skew.py` fits + - a discrete Zipf exponent θ to the rank-frequency of every key query's + group-by key (weighted by row count and by value sum), and + - a continuous power-law tail exponent α to every value query's column, + + once on all data and once per evaluation time, the way Prometheus would + evaluate the query every `step_s` seconds. + +## Running + +```bash +pip install -r requirements.txt +./fetch_data.sh /path/to/trace-data # ~63 GB; re-run to resume +python fit_skew.py --data-root /path/to/trace-data +``` + +This writes `results/skew_summary.csv` (committed) and plots to `out/` +(gitignored). Useful flags: + +- `--queries queries/google_2011.yaml`: run a subset of datasets. +- `--max-rows 300000`: rows read per file (BOOM: time steps per series) + for a quick smoke run. +- `--no-plots`, `--out DIR`, `--summary PATH`. +- `--min-eval-rows` (default 200) and `--min-eval-keys` (default 2): + evaluations below either threshold are skipped when computing the bounds. +- `--workers N`: process pool size (default: all cores). Files and fits run + in parallel. + +A full run over the fetched data takes about 2 hours with 24 workers on a +56-core machine (peak RSS of the main process about 69 GB). Alibaba archives are streamed with `tarfile`, not extracted. + +Tests: `python -m unittest discover -s tests -p 'test_*.py'`. + +## Datasets + +| Dataset | Files | Tables | Step | Ranges | +|---|---|---|---|---| +| Google ClusterData 2011-2 | `task_usage` and `task_events` parts 0..119 of 500 (about 7 days), all 500 `job_events` parts, `schema.csv` | `task_usage` joined with task and job attributes | 5 min | instant, 5m, 1h | +| Alibaba microservices v2022 | `MCRRTUpdate_0..119` (first 6 hours) | `MSRTMCR` | 1 min | instant, 5m, 1h | +| | `CallGraph_0..119` (first 6 hours) | `CallGraph` | 1 min | 1m, 5m, 1h (events, no instant) | +| | `MSMetricsUpdate_0..47`, `NodeMetricsUpdate_0..1` (first day) | `MSMetrics`, `NodeMetrics` | 1 min | instant, 5m, 1h | +| Datadog BOOM | `dataset_taxonomy.json` and 20 multivariate series | per-series `target` | none | 20 equal chunks per series | + +Citations: + +```bibtex +@online{google-traces-2011, + title={{Google ClusterData 2011 traces}}, + url={{https://github.com/google/cluster-data/blob/master/ClusterData2011_2.md}}, + year={2025} +} + +@inproceedings{luo2022Prediction, + title={The Power of Prediction: Microservice Auto Scaling via Workload Learning}, + author={Luo, Shutian and Xu, Huanle and Ye, Kejiang and Xu, Guoyao and Zhang, Liping and Yang, Guodong and Xu, Chengzhong}, + booktitle={Proceedings of the ACM Symposium on Cloud Computing}, + year={2022} +} + +@misc{cohen2025timedifferentobservabilityperspective, + title={This Time is Different: An Observability Perspective on Time Series Foundation Models}, + author={Ben Cohen and Emaad Khwaja and Youssef Doubli and Salahidine Lemaachi and Chris Lettieri and Charles Masson and Hugo Miccinilli and Elise Ramé and Qiqi Ren and Afshin Rostamizadeh and Jean Ogier du Terrail and Anna-Monica Toon and Kan Wang and Stephan Xie and Zongzhe Xu and Viktoriya Zhukova and David Asker and Ameet Talwalkar and Othmane Abou-Amal}, + year={2025}, + eprint={2505.14766}, + archivePrefix={arXiv}, + primaryClass={cs.LG}, + url={https://arxiv.org/abs/2505.14766} +} +``` + +## Definitions + +- **Key θ**: sort the per-key weights descending and fit the exponent `s` of + a discrete Zipf over ranks `1..K` by maximum likelihood + (`scipy.optimize.minimize_scalar`, bounded to [0, 5]). The weight is + either the row count per key (`weight=count`) or the sum of the value per + key with negatives clipped to 0 (`weight=value`). +- **Value α**: continuous power law fitted with `powerlaw.Fit` (pinned to + 1.5), with α the exponent of the pdf `p(x) ∝ x^-α` for `x ≥ xmin`. + Non-positive values are dropped first, and each fit uses a uniform + subsample of at most 100,000 values. `xmin` minimizes the KS distance over + a grid of 50 quantiles from p50 to p99.9, restricted to candidates that + keep at least 100 values in the tail. α therefore describes a tail between + the top half and roughly the top 0.1% of the values; `tail_frac` says + which. An evaluation needs at least 200 positive values to be fittable. +- **mle**: the estimate on all data pooled. +- **lower / upper**: min / max over the set {every per-evaluation estimate that + passes the thresholds, mle}, so `lower <= mle <= upper` always. The pooled + and per-evaluation fits have no fixed order: pooling adds rare keys to the tail + and accumulates mass on persistent heavy keys, which usually steepens the + pooled rank-frequency (most CallGraph queries), but it can also flatten it. +- **Evaluation**: each table has a `step_s` (its sampling period). A row + at time `t` belongs to step `ceil(t / step_s)`, so the evaluation at step + `w` (time `w * step_s`) sees rows in `((w - 1) * step_s, w * step_s]`. + Every query is evaluated at every step, as in a Prometheus range query + with that step: + - **range** (`5m`, `1h`, ...; a multiple of `step_s`): the rows of the + last `range / step_s` steps, i.e. a sliding window, not disjoint + windows. Only evaluations whose whole range lies in the data are used. + Key queries sum the per-step per-key aggregates; value queries merge + per-step uniform samples (at most 100,000 each) into a uniform sample + of the range. + - **instant**: one row per series (the table's `series_key`), its latest + sample at or before the evaluation time and at most 5 minutes old (the + Prometheus lookback; one step if `step_s` is longer). A series that + stops is still seen for up to 5 minutes. Needs files that are + consecutive time chunks; the run fails if a series' samples interleave + across files. Event tables (CallGraph) have no series and only range + forms. + + mle pools every row the query form sees: all rows for range forms, and + every (series, evaluation) pair for the instant form, so `rows_total` of + an instant row counts a series once per evaluation it is visible in. + BOOM has no timestamps in this analysis and keeps its 20 equal chunks per + series (`range` is empty). +- **tail_class**: the power law is compared with an exponential and a + lognormal (`distribution_compare`, likelihood ratio `R` and p-value `p`; + significant means `p < 0.1`). `light` if the power law does not + significantly beat the exponential (`R > 0`); otherwise `power_law` if it + also significantly beats the lognormal, `lognormal` if it is significantly + worse than the lognormal (`R < 0`), else `heavy_inconclusive`. α is always + reported. Power law and lognormal are often indistinguishable: a lognormal + with large σ is nearly straight on a log-log plot over several decades, so + the likelihood-ratio test usually cannot separate them without far more + tail data than a trace provides (Clauset, Shalizi and Newman, "Power-law + distributions in empirical data", SIAM Review 2009). Even an exact + Pareto(α=2) sample of 20,000 values comes out `heavy_inconclusive` here. + +## Output: `results/skew_summary.csv` + +One row per (query, kind, weight, range). The sketch-bench saturation study +reads the worst case per query: `worst_theta_cms` (key rows: the lowest θ of +the weight a per-key counter sketch sees, `cms_weight`), `worst_K`, +`min_N` and `max_N` for the key cardinality and stream size an evaluation +sees, and `worst_alpha_rank` / `worst_alpha_memory` for value rows, plus the +`target_*` accuracy it has to reach. For +value rows it should use only rows whose `tail_class` is not `light`: a light tail decays at least exponentially, so its α is just the +slope of whatever sliver of the tail the fit picked and does not describe a +power-law regime. + +| Column | Meaning | +|---|---| +| `dataset`, `query_id`, `promql` | query identity (a query with both kinds gets a `keys` and a `values` row) | +| `kind` | `keys` (θ) or `values` (α) | +| `weight` | `count` or `value` for key rows, empty for value rows | +| `range`, `range_s` | `instant` or the range duration (`range_s` empty for instant and BOOM) | +| `step_s` | evaluation step of the table | +| `K_total` | distinct keys with positive weight over the whole sample (BOOM: variates) | +| `rows_total` | rows with non-null keys over the whole sample (values: finite values; instant: series-evaluations) | +| `K_win_min`, `K_win_median`, `K_win_max` | key rows: distinct keys per evaluation | +| `rows_win_min`, `rows_win_median`, `rows_win_max` | rows per evaluation (value rows: positive values; empty for BOOM) | +| `n_evals` | evaluations used for the bounds (BOOM: variate-chunk fits) | +| `lower`, `mle`, `upper` | θ or α as defined above | +| `worst_theta_cms` | key rows whose weight is the query's `cms_weight`: `lower` | +| `worst_K`, `min_N`, `max_N` | `K_win_max`, `rows_win_min`, `rows_win_max` | +| `worst_alpha_rank`, `worst_alpha_memory` | value rows: `upper` (steepest tail, hardest for rank error) and `lower` (heaviest tail, largest value range) | +| `target_are_top100`, `target_precision_at_k`, `target_hll_rel_err`, `target_rank_err` | accuracy targets (defaults in `fit_skew.py`, overridden by a query's `targets`) | +| `top1_share`, `theta_ls` | key rows: share of the largest key, negated log-log least-squares slope | +| `dropped_frac` | value rows: fraction of finite values that were ≤ 0 | +| `xmin`, `ks_d`, `tail_frac` | value rows: fitted `xmin`, KS distance of the tail, and fraction of the fitted sample at or above `xmin` | +| `R_lognormal`, `p_lognormal`, `R_exponential`, `p_exponential` | log-likelihood ratio (power law vs alternative) and its p-value | +| `tail_class` | see Definitions (BOOM: majority class over variates) | +| `ok_frac` | BOOM rows: share of variates whose `tail_class` is not `light` | + +Plots in `out/`: `____rank___.png` (log-log +rank-frequency with the lower/mle/upper θ lines) and +`____ccdf__.png` (empirical CCDF with the fitted α). + +## Caveats + +- **θ depends on the query form.** An instant query sees one sample per + series, so its count-weighted θ measures how series spread over keys; a + range query sees every row in its range. Short ranges see fewer keys with + noisier counts; long ones pool more keys. Use the row whose `range` + matches the query. + +- **BOOM** strips tags and z-scores each variate, so there is no key θ and + the raw value scale is lost. α is fitted per variate on `x - min(x)` + (zeros dropped), and each variate gets its own `tail_class`. The series + row reports the majority class and `ok_frac`, the share of variates that + are not `light`. lower/mle/upper are medians over the non-light variates of + each variate's lower, mle and upper, and the diagnostics (`xmin`, `R`, + `p`, ...) are medians over the same variates; all are empty if every + variate is light. +- **Google join rule**: `task_usage` rows get `user`, `priority` and + `scheduling_class` from the last non-null value for the same + `(job_id, task_index)` over all 120 fetched `task_events` parts, and `logical_job_name` from the last non-null + `job_events` value for the same `job_id` (both ordered by event time). +- **Small counts bias θ upward**: θ is fitted to the *sorted observed* + counts, so the long tail of keys seen once or twice is flatter than the + true law and the order statistics exaggerate the head. Evaluations with few + rows per key (short ranges, high-cardinality keys) are most affected, which + widens `upper`. +- Google `task_usage` rows measure 5-minute intervals and are stamped at + their `end_time`. CallGraph has malformed rows + (extra fields), which are skipped and counted in the log, and `rt` values + of `None`, which are read as missing. diff --git a/asap-tools/dataset-analysis/fetch_data.sh b/asap-tools/dataset-analysis/fetch_data.sh new file mode 100755 index 00000000..1512ed73 --- /dev/null +++ b/asap-tools/dataset-analysis/fetch_data.sh @@ -0,0 +1,79 @@ +#!/usr/bin/env bash +# Download the trace subsets analyzed by fit_skew.py into DATA_ROOT. +# Files that already exist are skipped, so the script can be re-run to resume. +set -euo pipefail + +if [ "$#" -ne 1 ]; then + echo "usage: $0 DATA_ROOT" >&2 + exit 1 +fi +DATA_ROOT=$1 + +GOOGLE_URL=https://storage.googleapis.com/clusterdata-2011-2 +ALIBABA_URL=https://aliopentrace.oss-cn-beijing.aliyuncs.com/v2022MicroservicesTraces +BOOM_URL=https://huggingface.co/datasets/Datadog/BOOM/resolve/main + +# task_usage parts cover about 1.4 hours each; 120 parts = about 7 days. +# task_events uses the same parts; job_events is read in full for job names. +GOOGLE_TASK_PARTS=120 +GOOGLE_JOB_EVENT_PARTS=500 +# CallGraph and MCRRTUpdate shards cover 3 minutes each: 120 shards = 6 hours. +ALIBABA_RPC_SHARDS=120 +# MSMetricsUpdate shards cover 30 minutes, NodeMetricsUpdate 12 hours: 1 day. +ALIBABA_MS_SHARDS=48 +ALIBABA_NODE_SHARDS=2 + +BOOM_SERIES=( + ds-2187-H ds-2394-D ds-1135-5T ds-1833-D ds-2806-D + ds-2573-D ds-2222-30T ds-1914-30T ds-2577-H ds-2212-D + ds-2782-H ds-1400-10S ds-2650-D ds-1316-10S ds-1558-5T + ds-1972-D ds-1840-D ds-671-10S ds-1524-5T ds-899-T +) +BOOM_SERIES_FILES=(data-00000-of-00001.arrow dataset_info.json state.json) + +# fetch URL DEST: download to a temp file, then rename so partial files never look complete. +fetch() { + local url=$1 dest=$2 + if [ -s "$dest" ]; then + return + fi + mkdir -p "$(dirname "$dest")" + echo "fetch $url" + curl -fL --retry 5 --retry-delay 5 -C - -o "$dest.part" "$url" + mv "$dest.part" "$dest" +} + +google_part() { + printf 'part-%05d-of-00500.csv.gz' "$1" +} + +G=$DATA_ROOT/google-2011 +fetch "$GOOGLE_URL/schema.csv" "$G/schema.csv" +for ((i = 0; i < GOOGLE_TASK_PARTS; i++)); do + for t in task_usage task_events; do + fetch "$GOOGLE_URL/$t/$(google_part "$i")" "$G/$t/$(google_part "$i")" + done +done +for ((i = 0; i < GOOGLE_JOB_EVENT_PARTS; i++)); do + fetch "$GOOGLE_URL/job_events/$(google_part "$i")" "$G/job_events/$(google_part "$i")" +done + +A=$DATA_ROOT/alibaba-v2022 +for ((i = 0; i < ALIBABA_NODE_SHARDS; i++)); do + fetch "$ALIBABA_URL/NodeMetricsUpdate/NodeMetricsUpdate_$i.tar.gz" "$A/NodeMetricsUpdate_$i.tar.gz" +done +for ((i = 0; i < ALIBABA_MS_SHARDS; i++)); do + fetch "$ALIBABA_URL/MSMetricsUpdate/MSMetricsUpdate_$i.tar.gz" "$A/MSMetricsUpdate_$i.tar.gz" +done +for ((i = 0; i < ALIBABA_RPC_SHARDS; i++)); do + fetch "$ALIBABA_URL/CallGraph/CallGraph_$i.tar.gz" "$A/CallGraph_$i.tar.gz" + fetch "$ALIBABA_URL/MCRRTUpdate/MCRRTUpdate_$i.tar.gz" "$A/MCRRTUpdate_$i.tar.gz" +done + +B=$DATA_ROOT/boom +fetch "$BOOM_URL/dataset_taxonomy.json" "$B/dataset_taxonomy.json" +for series in "${BOOM_SERIES[@]}"; do + for f in "${BOOM_SERIES_FILES[@]}"; do + fetch "$BOOM_URL/$series/$f" "$B/$series/$f" + done +done diff --git a/asap-tools/dataset-analysis/fit_skew.py b/asap-tools/dataset-analysis/fit_skew.py new file mode 100644 index 00000000..181812db --- /dev/null +++ b/asap-tools/dataset-analysis/fit_skew.py @@ -0,0 +1,1274 @@ +#!/usr/bin/env python3 +"""Fit label skew (Zipf theta) and value skew (power-law alpha) for trace query sets. + +Key queries fit a discrete Zipf exponent to the rank-frequency of the group-by +key; value queries fit a continuous power law to the queried column. Every +query is evaluated at each step of its table, as an instant query (latest +sample per series) and/or as range queries over the last S seconds: lower/upper +are the min/max over evaluation times and mle is the fit on all data. +""" + +import argparse +import glob +import io +import logging +import os +import re +import tarfile +import time +import warnings +from contextlib import redirect_stdout +from multiprocessing import Pool +from pathlib import Path +from typing import Any, Dict, Iterator, List, Optional, Sequence, Set, Tuple + +import numpy as np +import pandas as pd +import powerlaw +import pyarrow as pa +import pyarrow.csv as pacsv +import yaml +from matplotlib.figure import Figure +from numpy.typing import ArrayLike +from scipy import sparse +from scipy.optimize import minimize_scalar + +SCRIPT_DIR = Path(__file__).resolve().parent +DEFAULT_QUERIES = sorted(str(p) for p in (SCRIPT_DIR / "queries").glob("*.yaml")) +DEFAULT_SUMMARY = SCRIPT_DIR / "results" / "skew_summary.csv" +DEFAULT_OUT = SCRIPT_DIR / "out" + +ZIPF_THETA_BOUNDS = (0.0, 5.0) +# Each power-law fit uses a uniform subsample of at most this many values. +MAX_FIT_SAMPLES = 100_000 +# xmin is chosen by KS distance over this many quantiles of the sample, and only +# where at least MIN_TAIL_SAMPLES values remain in the tail. +XMIN_GRID_SIZE = 50 +XMIN_GRID_QUANTILES = (0.5, 0.999) +MIN_TAIL_SAMPLES = 100 +SAMPLE_SEED = 0 +DEFAULT_MIN_EVAL_ROWS = 200 +DEFAULT_MIN_EVAL_KEYS = 2 +# A likelihood-ratio comparison is significant when its p is below this. +COMPARE_P_THRESHOLD = 0.1 +TAIL_LIGHT = "light" +TAIL_POWER_LAW = "power_law" +TAIL_LOGNORMAL = "lognormal" +TAIL_HEAVY_INCONCLUSIVE = "heavy_inconclusive" +POWER_LAW_ALTERNATIVES = ("lognormal", "exponential") + +INSTANT = "instant" +# An instant query sees a series' latest sample at most this old (Prometheus +# default), or one sampling period if that is longer. +PROMETHEUS_LOOKBACK_S = 300 +DURATION_UNITS_S = {"s": 1, "m": 60, "h": 3600, "d": 86400} +# Range key sums are materialized for this many evaluation times at a time. +EVAL_CHUNK = 32 +# Accuracy each sketch must reach on a query; a query's `targets` overrides. +DEFAULT_TARGETS = { + "are_top100": 0.05, # CMS / CountSketch mean relative error of the top 100 keys + "precision_at_k": 0.95, # top-k precision@k + "hll_rel_err": 0.02, # HLL relative cardinality error + "rank_err": 0.01, # KLL / DDSketch mean rank error +} + +CSV_BLOCK_BYTES = 64 << 20 +CSV_NULL_VALUES = ["", "None", "NULL", "NaN", "nan"] +LABEL_TYPE = pa.dictionary(pa.int32(), pa.string()) + +# Step index w holds timestamps in ((w - 1) * step_s, w * step_s], so the +# evaluation at t = w * step_s over range S sums steps w - S / step_s + 1 .. w. +STEP_COL = "_step" +TIME_COL = "_time_s" +# Label tuples are hashed to uint64; MISSING_KEY marks a null label. +KEY_COL = "_key" +SERIES_COL = "_series" +NEXT_COL = "_next_step" +FILE_COL = "_file" +MISSING_KEY = np.uint64(0) +COUNT_COL = "count" +VALUE_SUM_COL = "value_sum" +KEY_WEIGHTS = {"count", "value"} +QUERY_KINDS = {"keys", "values"} + +# Enough digits to keep row counts exact. +SUMMARY_FLOAT_FORMAT = "%.10g" +SUMMARY_COLUMNS = [ + "dataset", + "query_id", + "promql", + "kind", + "weight", + "range", + "range_s", + "step_s", + "K_total", + "rows_total", + "K_win_min", + "K_win_median", + "K_win_max", + "rows_win_min", + "rows_win_median", + "rows_win_max", + "n_evals", + "lower", + "mle", + "upper", + "worst_theta_cms", + "worst_K", + "min_N", + "max_N", + "worst_alpha_rank", + "worst_alpha_memory", + *(f"target_{name}" for name in DEFAULT_TARGETS), + "top1_share", + "theta_ls", + "dropped_frac", + "xmin", + "ks_d", + "tail_frac", + "R_lognormal", + "p_lognormal", + "R_exponential", + "p_exponential", + "tail_class", + "ok_frac", +] + +# Plot colors: observed data in neutral ink, the three estimates as one blue ramp. +OBSERVED_COLOR = "#8a8984" +BOUND_COLORS = {"lower": "#86b6ef", "mle": "#2a78d6", "upper": "#104281"} + +log = logging.getLogger("fit_skew") + + +# ---------------------------------------------------------------- config + + +def load_config(path: str) -> Dict[str, Any]: + with open(path) as f: + cfg = yaml.safe_load(f) + validate_config(cfg) + return cfg + + +def duration_secs(token: str) -> int: + """'5m' -> 300.""" + match = re.fullmatch(r"(\d+)([smhd])", token) + if match is None: + raise ValueError(f"bad duration {token!r}, expected e.g. 30s, 5m, 1h") + return int(match.group(1)) * DURATION_UNITS_S[match.group(2)] + + +def range_secs(token: str) -> float: + """Range duration in seconds; NaN for instant.""" + return np.nan if token == INSTANT else duration_secs(token) + + +def range_steps(token: str, step_s: int) -> int: + """Steps an evaluation covers: one for instant (per-evaluation aggregates).""" + return 1 if token == INSTANT else duration_secs(token) // step_s + + +def lookback_steps(table: Dict[str, Any]) -> int: + lookback_s = max(PROMETHEUS_LOOKBACK_S, table["step_s"]) + return -(-lookback_s // table["step_s"]) + + +def validate_ranges(q: Dict[str, Any], table: Dict[str, Any], where: str) -> None: + if not q.get("range"): + raise ValueError(f"{where}: range must list instant and/or durations") + for token in q["range"]: + if token == INSTANT: + if "series_key" not in table: + raise ValueError(f"{where}: instant needs a table series_key") + if "promql" not in q: + raise ValueError(f"{where}: instant needs promql") + continue + secs = duration_secs(token) + if secs % table["step_s"]: + raise ValueError(f"{where}: range {token} is not a multiple of step_s") + if "{range}" not in q.get("promql_range", ""): + raise ValueError(f"{where}: range queries need promql_range with {{range}}") + + +def validate_query(q: Dict[str, Any], table: Dict[str, Any], where: str) -> None: + if q["kind"] not in QUERY_KINDS: + raise ValueError(f"{where}: kind must be one of {sorted(QUERY_KINDS)}") + for col in q["group_by"]: + if col not in table["label_columns"]: + raise ValueError(f"{where}: {col!r} is not a label column") + value = q.get("value") + if value is not None and value not in table["value_columns"]: + raise ValueError(f"{where}: {value!r} is not a value column") + unknown = set(q.get("targets", {})) - set(DEFAULT_TARGETS) + if unknown: + raise ValueError(f"{where}: unknown targets {sorted(unknown)}") + if table.get("format") != "boom_arrow": + validate_ranges(q, table, where) + if q["kind"] == "values": + if value is None: + raise ValueError(f"{where}: value queries need a value column") + return + if not q["group_by"]: + raise ValueError(f"{where}: key queries need group_by") + weights = set(q["weights"]) + if not weights or not weights <= KEY_WEIGHTS: + raise ValueError(f"{where}: weights must be a subset of {KEY_WEIGHTS}") + if "value" in weights and value is None: + raise ValueError(f"{where}: weight 'value' needs a value column") + if cms_weight(q) not in weights: + raise ValueError(f"{where}: cms_weight must be one of the weights") + + +def validate_config(cfg: Dict[str, Any]) -> None: + for name, table in cfg["tables"].items(): + where = f"{cfg['dataset']}/{name}" + if table.get("format") == "boom_arrow": + continue + if not isinstance(table.get("step_s"), int) or table["step_s"] <= 0: + raise ValueError(f"{where}: step_s must be a positive integer") + if not set(table.get("series_key", [])) <= set(table["label_columns"]): + raise ValueError(f"{where}: series_key must be label columns") + for q in cfg["queries"]: + where = f"{cfg['dataset']}/{q['id']}" + table = cfg["tables"].get(q["table"]) + if table is None: + raise ValueError(f"{where}: unknown table {q['table']!r}") + validate_query(q, table, where) + + +def cms_weight(q: Dict[str, Any]) -> str: + """Weight a per-key counter sketch sees: count for count-by, value for sum-by.""" + return q.get("cms_weight", "count") + + +def query_targets(q: Dict[str, Any]) -> Dict[str, float]: + targets = {**DEFAULT_TARGETS, **q.get("targets", {})} + return {f"target_{name}": value for name, value in targets.items()} + + +def range_promql(q: Dict[str, Any], token: str) -> str: + if token == INSTANT: + return q["promql"] + return q["promql_range"].format(range=token) + + +def expand_files(data_root: str, patterns: Sequence[str]) -> List[str]: + paths = [] + for pattern in patterns: + matches = sorted(glob.glob(os.path.join(data_root, pattern))) + if not matches: + raise FileNotFoundError(f"no files match {pattern} under {data_root}") + paths.extend(matches) + return paths + + +# ---------------------------------------------------------------- reading + + +def google_column_names(schema_path: str, table: str) -> List[str]: + """Column names for a headerless Google table, e.g. 'CPU rate' -> 'cpu_rate'.""" + schema = pd.read_csv(schema_path) + rows = schema[schema["file pattern"].str.startswith(table + "/")] + if rows.empty: + raise ValueError(f"table {table!r} not found in {schema_path}") + contents = rows.sort_values("field number")["content"] + return [c.strip().lower().replace(" ", "_") for c in contents] + + +def csv_batches( + source: Any, + columns: Sequence[str], + float_columns: Set[str], + column_names: Optional[List[str]], + bad_rows: List[str], +) -> Iterator[pd.DataFrame]: + """Stream a CSV as DataFrames. Float columns are float64, the rest categorical.""" + + def skip_row(row: Any) -> str: + bad_rows.append(row.text) + return "skip" + + types = {c: pa.float64() if c in float_columns else LABEL_TYPE for c in columns} + reader = pacsv.open_csv( + source, + read_options=pacsv.ReadOptions( + column_names=column_names, block_size=CSV_BLOCK_BYTES + ), + parse_options=pacsv.ParseOptions(invalid_row_handler=skip_row), + convert_options=pacsv.ConvertOptions( + include_columns=list(columns), + column_types=types, + null_values=CSV_NULL_VALUES, + strings_can_be_null=True, + ), + ) + for batch in reader: + yield batch.to_pandas() + + +def table_frames( + data_root: str, + table: Dict[str, Any], + path: str, + columns: Sequence[str], + float_columns: Set[str], + bad_rows: List[str], +) -> Iterator[pd.DataFrame]: + fmt = table["format"] + if fmt == "google_csv": + # The table name is the directory name, as in the schema's file patterns. + schema = os.path.join(data_root, table["schema"]) + names = google_column_names(schema, Path(path).parent.name) + yield from csv_batches(path, columns, float_columns, names, bad_rows) + elif fmt == "alibaba_tar": + # Stream members straight out of the archive instead of extracting. + with tarfile.open(path, mode="r|gz") as archive: + for member in archive: + f = archive.extractfile(member) + if f is not None: + yield from csv_batches(f, columns, float_columns, None, bad_rows) + else: + raise ValueError(f"unsupported table format {fmt!r}") + + +def load_join( + data_root: str, table: Dict[str, Any], join: Dict[str, Any] +) -> pd.DataFrame: + """Last non-null value of each join column per key, ordered by sort_by.""" + columns = join["keys"] + [join["sort_by"]] + join["columns"] + bad_rows: List[str] = [] + frames = [ + frame + for path in expand_files(data_root, join["files"]) + for frame in table_frames( + data_root, table, path, columns, {join["sort_by"]}, bad_rows + ) + ] + df = pd.concat(frames, ignore_index=True) + df = df.astype({c: object for c in join["keys"] + join["columns"]}) + df = df.sort_values(join["sort_by"], kind="stable") + return df.groupby(join["keys"])[join["columns"]].last() + + +def read_boom_series(path: str) -> np.ndarray: + """Target of a BOOM series as a (variates, time) array.""" + with pa.memory_map(path) as src: + table = pa.ipc.open_stream(src).read_all() + if table.num_rows != 1: + raise ValueError(f"{path}: expected one row, found {table.num_rows}") + return np.atleast_2d(np.array(table.column("target")[0].as_py(), dtype=float)) + + +# ---------------------------------------------------------------- aggregation + + +def query_key(q: Dict[str, Any]) -> Tuple[str, str]: + return q["id"], q["kind"] + + +def has_range(q: Dict[str, Any]) -> bool: + return any(token != INSTANT for token in q["range"]) + + +def label_hash(frame: pd.DataFrame, cols: Sequence[str]) -> np.ndarray: + """uint64 hash of each row's label tuple, equal across batches and files.""" + return pd.util.hash_pandas_object(frame[list(cols)], index=False).to_numpy() + + +def key_hash(frame: pd.DataFrame, cols: Sequence[str]) -> np.ndarray: + """label_hash, with MISSING_KEY where any label is null (such rows are dropped).""" + hashes = label_hash(frame, cols) + hashes[frame[list(cols)].isna().any(axis=1).to_numpy()] = MISSING_KEY + return hashes + + +def group_col(q: Dict[str, Any]) -> str: + return f"{KEY_COL}:{','.join(q['group_by'])}" + + +def merge_key_parts(parts: List[pd.DataFrame]) -> pd.DataFrame: + """Sum per-(step, key) counts and value sums.""" + return ( + pd.concat(parts, ignore_index=True) + .groupby([STEP_COL, KEY_COL], sort=False) + .sum() + .reset_index() + ) + + +def key_part(frame: pd.DataFrame, q: Dict[str, Any]) -> pd.DataFrame: + """Row count and clipped value sum per (step, key); frame carries KEY_COL.""" + frame = frame[frame[KEY_COL] != MISSING_KEY] + keys = [STEP_COL, KEY_COL] + if "value" in q["weights"]: + clipped = frame.assign(**{VALUE_SUM_COL: frame[q["value"]].clip(lower=0)}) + part = clipped.groupby(keys, sort=False).agg( + **{ + COUNT_COL: (VALUE_SUM_COL, "size"), + VALUE_SUM_COL: (VALUE_SUM_COL, "sum"), + } + ) + else: + part = frame.groupby(keys, sort=False).size().to_frame(COUNT_COL) + return part.reset_index() + + +def add_values(acc: Dict[str, Any], steps: np.ndarray, values: np.ndarray) -> None: + finite = np.isfinite(values) + positive = values > 0 + acc["n_finite"] += int(finite.sum()) + for w in np.unique(steps[positive]): + acc["steps"].setdefault(int(w), []).append(values[positive & (steps == w)]) + + +def value_sample(x: np.ndarray) -> Tuple[int, np.ndarray]: + """(len(x), up to MAX_FIT_SAMPLES of x in random order): any prefix of the + sample is a uniform sample of x, which merge_samples relies on.""" + rng = np.random.default_rng(SAMPLE_SEED) + return len(x), x[rng.permutation(len(x))[:MAX_FIT_SAMPLES]] + + +def merge_samples(parts: Sequence[Tuple[int, np.ndarray]]) -> Tuple[int, np.ndarray]: + """Uniform sample of the union of value_sample parts: each part gives a + multivariate-hypergeometric share of its prefix.""" + counts = np.array([n for n, _ in parts], dtype=np.int64) + total = int(counts.sum()) + rng = np.random.default_rng(SAMPLE_SEED) + take = rng.multivariate_hypergeometric(counts, min(total, MAX_FIT_SAMPLES)) + merged = np.concatenate([x[:k] for (_, x), k in zip(parts, take)]) + return total, rng.permutation(merged) + + +def sample_steps(acc: Dict[str, Any]) -> Dict[str, Any]: + """Replace each step's value arrays by one value_sample.""" + steps = {w: value_sample(np.concatenate(xs)) for w, xs in acc["steps"].items()} + return {"n_finite": acc["n_finite"], "steps": steps} + + +def merge_step_samples(accs: Sequence[Dict[str, Any]]) -> Dict[str, Any]: + parts: Dict[int, List[Tuple[int, np.ndarray]]] = {} + for acc in accs: + for w, sample in acc["steps"].items(): + parts.setdefault(w, []).append(sample) + return { + "n_finite": sum(acc["n_finite"] for acc in accs), + "steps": {w: merge_samples(ps) for w, ps in parts.items()}, + } + + +def latest_per_step(rows: pd.DataFrame) -> pd.DataFrame: + """Each series' last sample in each step.""" + rows = rows.sort_values(TIME_COL, kind="stable") + return rows.drop_duplicates([SERIES_COL, STEP_COL], keep="last") + + +def latest_part( + frame: pd.DataFrame, series_key: List[str], queries: List[Dict[str, Any]] +) -> pd.DataFrame: + """Per (series, step) latest sample with the group keys and values the + instant queries need.""" + cols: Dict[str, Any] = { + SERIES_COL: label_hash(frame, series_key), + STEP_COL: frame[STEP_COL].to_numpy(), + TIME_COL: frame[TIME_COL].to_numpy(), + } + for q in queries: + if q["kind"] == "keys": + cols[group_col(q)] = key_hash(frame, q["group_by"]) + if q.get("value"): + cols[q["value"]] = frame[q["value"]].to_numpy(dtype=float) + return latest_per_step(pd.DataFrame(cols)) + + +def next_step_of_series(rows: pd.DataFrame) -> Tuple[np.ndarray, np.ndarray]: + """For rows sorted by (series, step): the series' next step (NaN if none) + and whether the row is the series' first.""" + series = rows[SERIES_COL].to_numpy() + steps = rows[STEP_COL].to_numpy() + same = series[1:] == series[:-1] + has_next = np.append(same, False) + next_step = np.where(has_next, np.append(steps[1:], 0), np.nan) + return next_step, ~np.append(False, same) + + +def split_instant( + rows: pd.DataFrame, queries: List[Dict[str, Any]], lookback: int +) -> Tuple[Dict[Tuple[str, str], Any], pd.DataFrame]: + """Instant aggregates of one file's latest samples, except each series' + first and last sample in the file, which are returned for resolve_boundaries + because another file may hold the same step or the next sample.""" + rows = latest_per_step(rows).sort_values([SERIES_COL, STEP_COL], kind="stable") + next_step, first = next_step_of_series(rows) + rows = rows.assign(**{NEXT_COL: next_step}) + interior = ~first & ~np.isnan(next_step) + return instant_parts(rows[interior], queries, lookback), rows[~interior] + + +def check_file_order(rows: pd.DataFrame) -> None: + """split_instant takes a series' next sample from the same file, so a + series' steps in one file must not interleave with its steps in another.""" + spans = ( + rows.groupby([SERIES_COL, FILE_COL])[STEP_COL] + .agg(["min", "max"]) + .reset_index() + .sort_values([SERIES_COL, "min", "max"]) + ) + series = spans[SERIES_COL].to_numpy() + lo, hi = spans["min"].to_numpy(), spans["max"].to_numpy() + overlap = (series[1:] == series[:-1]) & (lo[1:] < hi[:-1]) + if overlap.any(): + raise ValueError( + f"{int(overlap.sum())} series have samples interleaved across files; " + "instant evaluation needs files that are consecutive time chunks" + ) + + +def resolve_boundaries( + boundary: Sequence[pd.DataFrame], queries: List[Dict[str, Any]], lookback: int +) -> Dict[Tuple[str, str], Any]: + """Instant aggregates of the per-file first/last samples: keep the latest + sample per (series, step) over files, and take its next step as the + nearer of the in-file next step and the next boundary sample.""" + rows = pd.concat( + [b.assign(**{FILE_COL: i}) for i, b in enumerate(boundary)], ignore_index=True + ) + check_file_order(rows) + rows = rows.sort_values([SERIES_COL, STEP_COL, TIME_COL], kind="stable") + in_file_next = rows.groupby([SERIES_COL, STEP_COL])[NEXT_COL].transform("min") + rows = rows.assign(**{NEXT_COL: in_file_next}) + rows = rows.drop_duplicates([SERIES_COL, STEP_COL], keep="last") + next_step, _ = next_step_of_series(rows) + rows = rows.assign(**{NEXT_COL: np.fmin(rows[NEXT_COL].to_numpy(), next_step)}) + return instant_parts(rows, queries, lookback) + + +def instant_parts( + rows: pd.DataFrame, queries: List[Dict[str, Any]], lookback: int +) -> Dict[Tuple[str, str], Any]: + """Per-evaluation aggregates of latest samples. A sample at step w is its + series' latest for evaluations w .. min(next step, w + lookback) - 1, e.g. + with lookback 5 a series that stops after step 3 counts at steps 3..7.""" + start = rows[STEP_COL].to_numpy() + end = np.fmin(rows[NEXT_COL].to_numpy(), start + lookback).astype(np.int64) + reps = end - start + idx = np.repeat(np.arange(len(rows)), reps) + offsets = np.arange(len(idx)) - np.repeat(np.cumsum(reps) - reps, reps) + evals = rows.iloc[idx].assign(**{STEP_COL: start[idx] + offsets}) + out: Dict[Tuple[str, str], Any] = {} + for q in queries: + if q["kind"] == "keys": + out[query_key(q)] = key_part( + evals.rename(columns={group_col(q): KEY_COL}), q + ) + else: + acc: Dict[str, Any] = {"n_finite": 0, "steps": {}} + add_values( + acc, evals[STEP_COL].to_numpy(), evals[q["value"]].to_numpy(float) + ) + out[query_key(q)] = sample_steps(acc) + return out + + +def aggregate_file(task: Tuple[Any, ...]) -> Dict[str, Any]: + """Per-step key aggregates and value samples (range queries) and instant + aggregates (instant queries) for one file.""" + data_root, table, path, queries, joins, max_rows = task + time_col = table["time_column"] + range_queries = [q for q in queries if has_range(q)] + instant_queries = [q for q in queries if INSTANT in q["range"]] + value_cols = {q["value"] for q in queries if q.get("value")} + join_cols = [c for j in table.get("joins", []) for c in j["columns"]] + needed = {time_col} | value_cols + needed |= {c for q in queries for c in q["group_by"] if c not in join_cols} + needed |= {c for j in table.get("joins", []) for c in j["keys"]} + if instant_queries: + needed |= set(table["series_key"]) + + key_parts: Dict[Tuple[str, str], List[pd.DataFrame]] = {} + values: Dict[Tuple[str, str], Dict[str, Any]] = {} + for q in range_queries: + if q["kind"] == "keys": + key_parts[query_key(q)] = [] + else: + values[query_key(q)] = {"n_finite": 0, "steps": {}} + latest: List[pd.DataFrame] = [] + rows_read = 0 + step_span = (np.inf, -np.inf) + bad_rows: List[str] = [] + frames = table_frames( + data_root, table, path, sorted(needed), value_cols | {time_col}, bad_rows + ) + for frame in frames: + if max_rows is not None: + frame = frame.iloc[: max_rows - rows_read] + rows_read += len(frame) + secs = frame[time_col] * table["time_unit_secs"] + keep = secs.notna() + frame = frame[keep].copy() + frame[TIME_COL] = secs[keep] + frame[STEP_COL] = np.ceil(secs[keep] / table["step_s"]).astype(np.int64) + if len(frame): + step_span = ( + min(step_span[0], frame[STEP_COL].min()), + max(step_span[1], frame[STEP_COL].max()), + ) + for join, lookup in joins: + frame = frame.astype({c: object for c in join["keys"]}) + frame = frame.join(lookup, on=join["keys"]) + for q in range_queries: + if q["kind"] == "keys": + keyed = frame.assign(**{KEY_COL: key_hash(frame, q["group_by"])}) + key_parts[query_key(q)].append(key_part(keyed, q)) + else: + add_values( + values[query_key(q)], + frame[STEP_COL].to_numpy(), + frame[q["value"]].to_numpy(dtype=float), + ) + if instant_queries: + latest.append(latest_part(frame, table["series_key"], instant_queries)) + if max_rows is not None and rows_read >= max_rows: + break + out: Dict[str, Any] = { + "keys": {k: merge_key_parts(parts) for k, parts in key_parts.items()}, + "values": {k: sample_steps(acc) for k, acc in values.items()}, + "span": step_span, + "rows_read": rows_read, + "bad_rows": len(bad_rows), + } + if instant_queries: + out["instant"], out["boundary"] = split_instant( + pd.concat(latest, ignore_index=True), instant_queries, lookback_steps(table) + ) + return out + + +# ---------------------------------------------------------------- fitting + + +def zipf_mle(weights: ArrayLike) -> float: + """Discrete Zipf exponent over ranks 1..K maximizing the likelihood of the + sorted weights (counts, or value sums used as fractional counts).""" + w = np.sort(np.asarray(weights, dtype=float))[::-1] + w = w[w > 0] + if len(w) < 2: + return float("nan") + log_rank = np.log(np.arange(1, len(w) + 1)) + total, weighted_log_rank = w.sum(), (w * log_rank).sum() + + def neg_log_likelihood(s: float) -> float: + return s * weighted_log_rank + total * np.log(np.exp(-s * log_rank).sum()) + + res = minimize_scalar( + neg_log_likelihood, bounds=ZIPF_THETA_BOUNDS, method="bounded" + ) + return float(res.x) + + +def loglog_slope(weights: ArrayLike) -> float: + """Negated least-squares slope of log(weight) vs log(rank).""" + w = np.sort(np.asarray(weights, dtype=float))[::-1] + w = w[w > 0] + if len(w) < 2: + return float("nan") + return float(-np.polyfit(np.log(np.arange(1, len(w) + 1)), np.log(w), 1)[0]) + + +def window_bounds( + window_estimates: Sequence[float], pooled: float +) -> Tuple[float, float, int]: + """(lower, upper, n_evals): min and max over the finite per-window + estimates together with the pooled estimate, so lower <= pooled <= upper.""" + windows = [e for e in window_estimates if np.isfinite(e)] + candidates = windows + ([pooled] if np.isfinite(pooled) else []) + if not candidates: + return float("nan"), float("nan"), 0 + return min(candidates), max(candidates), len(windows) + + +def spread(prefix: str, per_window: Sequence[float]) -> Dict[str, float]: + """{prefix}_min, _median and _max over per-window values.""" + if not len(per_window): + return {} + return { + f"{prefix}_min": float(np.min(per_window)), + f"{prefix}_median": float(np.median(per_window)), + f"{prefix}_max": float(np.max(per_window)), + } + + +def subsample(x: np.ndarray) -> np.ndarray: + if len(x) <= MAX_FIT_SAMPLES: + return x + rng = np.random.default_rng(SAMPLE_SEED) + return x[rng.choice(len(x), MAX_FIT_SAMPLES, replace=False)] + + +def significant(comparison: Tuple[float, float], sign: int) -> bool: + """Whether (R, p) favors the power law (sign=1) or the alternative (sign=-1).""" + ratio, p = comparison + return ratio * sign > 0 and p < COMPARE_P_THRESHOLD + + +def tail_class(comparisons: Dict[str, Tuple[float, float]]) -> str: + """light unless the power law significantly beats the exponential; then + power_law or lognormal by whichever significantly wins, else inconclusive.""" + if not significant(comparisons["exponential"], 1): + return TAIL_LIGHT + if significant(comparisons["lognormal"], 1): + return TAIL_POWER_LAW + if significant(comparisons["lognormal"], -1): + return TAIL_LOGNORMAL + return TAIL_HEAVY_INCONCLUSIVE + + +def xmin_candidates(x_sorted: np.ndarray) -> np.ndarray: + """Quantile grid of xmin values that leave at least MIN_TAIL_SAMPLES in the tail.""" + levels = np.linspace(*XMIN_GRID_QUANTILES, XMIN_GRID_SIZE) + grid = np.unique(np.quantile(x_sorted, levels)) + return grid[grid <= x_sorted[-MIN_TAIL_SAMPLES]] + + +def fit_power_law(x: np.ndarray, compare: bool) -> Dict[str, Any]: + """Continuous power-law fit of positive values with xmin minimizing the KS + distance over a quantile grid; with compare, also the likelihood ratio + against each alternative distribution.""" + result: Dict[str, Any] = {"alpha": np.nan, "xmin": np.nan, "ks_d": np.nan} + if len(x) < MIN_TAIL_SAMPLES: + return result + x = np.sort(x) + candidates = xmin_candidates(x) + if not len(candidates): + return result + # powerlaw prints xmin search progress unconditionally. + with warnings.catch_warnings(), np.errstate(all="ignore"), redirect_stdout( + io.StringIO() + ): + warnings.simplefilter("ignore") + fits = [powerlaw.Fit(x, xmin=xmin, verbose=False) for xmin in candidates] + distances = np.array([f.power_law.D for f in fits], dtype=float) + if np.all(np.isnan(distances)): + return result + fit = fits[int(np.nanargmin(distances))] + result.update( + alpha=fit.power_law.alpha, + xmin=fit.xmin, + ks_d=fit.power_law.D, + tail_frac=np.mean(x >= fit.xmin), + ) + if not compare or not np.isfinite(result["alpha"]): + return result + comparisons = {} + for alt in POWER_LAW_ALTERNATIVES: + ratio, p = fit.distribution_compare("power_law", alt) + result[f"R_{alt}"], result[f"p_{alt}"] = ratio, p + comparisons[alt] = (ratio, p) + result["tail_class"] = tail_class(comparisons) + return result + + +def median_or_nan(values: Sequence[float]) -> float: + finite = [v for v in values if v is not None and np.isfinite(v)] + return float(np.median(finite)) if finite else float("nan") + + +# ---------------------------------------------------------------- plots + + +def plot_rank_frequency( + weights_desc: np.ndarray, thetas: Dict[str, float], title: str, path: Path +) -> None: + ranks = np.arange(1, len(weights_desc) + 1) + fig = Figure(figsize=(6, 4)) + ax = fig.subplots() + ax.loglog(ranks, weights_desc, ".", ms=3, color=OBSERVED_COLOR, label="observed") + for name, theta in thetas.items(): + if np.isfinite(theta): + ref = weights_desc.sum() * ranks**-theta / np.sum(ranks**-theta) + style = "-" if name == "mle" else "--" + ax.loglog( + ranks, + ref, + style, + lw=2, + color=BOUND_COLORS[name], + label=f"{name} θ={theta:.2f}", + ) + ax.set(xlabel="rank", ylabel="weight", title=title) + ax.legend() + fig.savefig(path, dpi=120, bbox_inches="tight") + + +def plot_ccdf(x: np.ndarray, fit: Dict[str, Any], title: str, path: Path) -> None: + xs = np.sort(x) + ccdf = 1.0 - np.arange(len(xs)) / len(xs) + fig = Figure(figsize=(6, 4)) + ax = fig.subplots() + ax.loglog(xs, ccdf, ".", ms=3, color=OBSERVED_COLOR, label="observed") + alpha, xmin = fit["alpha"], fit["xmin"] + if np.isfinite(alpha): + tail = xs[xs >= xmin] + ref = np.mean(xs >= xmin) * (tail / xmin) ** (1.0 - alpha) + label = f"power law α={alpha:.2f}, xmin={xmin:.3g}" + ax.loglog(tail, ref, "-", lw=2, color=BOUND_COLORS["mle"], label=label) + ax.set(xlabel="value", ylabel="P(X ≥ x)", title=title) + ax.legend() + fig.savefig(path, dpi=120, bbox_inches="tight") + + +def plot_path(plot_dir: Path, dataset: str, name: str) -> Path: + return plot_dir / f"{dataset}__{name}.png" + + +# ---------------------------------------------------------------- summaries + + +def rolling_key_sums( + agg: pd.DataFrame, columns: Sequence[str], n_steps: int, span: Tuple[int, int] +) -> Iterator[Dict[str, np.ndarray]]: + """For each evaluation step t in [first + n_steps - 1, last], the per-key + sums of `columns` over steps t - n_steps + 1 .. t (keys present only). + Computed as a banded 0/1 matrix times the sparse step x key matrix.""" + first, last = span + agg = agg[(agg[STEP_COL] >= first) & (agg[STEP_COL] <= last)] + n_total = last - first + 1 + n_evals = n_total - n_steps + 1 + if n_evals <= 0 or agg.empty: + return + _, key_idx = np.unique(agg[KEY_COL].to_numpy(), return_inverse=True) + coords = (agg[STEP_COL].to_numpy() - first, key_idx) + shape = (n_total, int(key_idx.max()) + 1) + mats = { + c: sparse.csr_matrix((agg[c].to_numpy(dtype=float), coords), shape=shape) + for c in columns + } + # Row e covers steps e .. e + n_steps - 1, i.e. evaluation first + e + n_steps - 1. + band_rows = np.repeat(np.arange(n_evals), n_steps) + band_cols = band_rows + np.tile(np.arange(n_steps), n_evals) + band = sparse.csr_matrix( + (np.ones(len(band_rows)), (band_rows, band_cols)), shape=(n_evals, n_total) + ) + for start in range(0, n_evals, EVAL_CHUNK): + chunk = { + c: (band[start : start + EVAL_CHUNK] @ m).tocsr() for c, m in mats.items() + } + for i in range(min(EVAL_CHUNK, n_evals - start)): + yield {c: m.data[m.indptr[i] : m.indptr[i + 1]] for c, m in chunk.items()} + + +def summarize_keys( + dataset: str, + q: Dict[str, Any], + token: str, + step_s: int, + agg: pd.DataFrame, + span: Tuple[int, int], + min_rows: int, + min_keys: int, + plot_dir: Optional[Path], +) -> List[Dict[str, Any]]: + """One row per weight for one range; agg holds per-step key aggregates + (range queries) or per-evaluation ones (instant, one step each).""" + n_steps = range_steps(token, step_s) + rows = int(agg[COUNT_COL].sum()) + if rows == 0: + raise ValueError(f"{dataset}/{q['id']}: no rows with non-null group keys") + columns = {w: COUNT_COL if w == "count" else VALUE_SUM_COL for w in q["weights"]} + per_key = {} + for weight, col in columns.items(): + totals = agg.groupby(KEY_COL)[col].sum().to_numpy() + per_key[weight] = np.sort(totals[totals > 0])[::-1] + pooled = {weight: zipf_mle(w) for weight, w in per_key.items()} + estimates: Dict[str, List[float]] = {weight: [] for weight in columns} + keys_per_eval, rows_per_eval = [], [] + for sums in rolling_key_sums(agg, sorted(set(columns.values())), n_steps, span): + counts = sums[COUNT_COL] + n_rows = counts.sum() + if n_rows == 0: + # No data in range (a gap in the trace), as summarize_values skips. + continue + keys_per_eval.append(int(np.sum(counts > 0))) + rows_per_eval.append(n_rows) + for weight, col in columns.items(): + w = sums[col] + if n_rows >= min_rows and np.sum(w > 0) >= min_keys: + estimates[weight].append(zipf_mle(w)) + eval_stats = { + **spread("K_win", keys_per_eval), + **spread("rows_win", rows_per_eval), + } + out = [] + for weight in columns: + lower, upper, n_evals = window_bounds(estimates[weight], pooled[weight]) + keys = per_key[weight] + out.append( + { + "dataset": dataset, + "query_id": q["id"], + "promql": range_promql(q, token), + "kind": "keys", + "weight": weight, + "range": token, + "range_s": range_secs(token), + "step_s": step_s, + "K_total": len(keys), + "rows_total": rows, + **eval_stats, + "n_evals": n_evals, + "lower": lower, + "mle": pooled[weight], + "upper": upper, + "worst_theta_cms": lower if weight == cms_weight(q) else np.nan, + "worst_K": eval_stats.get("K_win_max", np.nan), + "min_N": eval_stats.get("rows_win_min", np.nan), + "max_N": eval_stats.get("rows_win_max", np.nan), + **query_targets(q), + "top1_share": keys[0] / keys.sum() if len(keys) else np.nan, + "theta_ls": loglog_slope(keys), + } + ) + if plot_dir is not None and len(keys): + plot_rank_frequency( + keys, + {"lower": lower, "mle": pooled[weight], "upper": upper}, + f"{dataset}: {range_promql(q, token)} [{weight}]", + plot_path(plot_dir, dataset, f"{q['id']}__rank_{weight}__{token}"), + ) + return out + + +def summarize_values( + dataset: str, + q: Dict[str, Any], + token: str, + step_s: int, + acc: Dict[str, Any], + span: Tuple[int, int], + pool: Any, + min_rows: int, + plot_dir: Optional[Path], +) -> Dict[str, Any]: + """Row for one range; acc holds a value_sample per step (range queries) + or per evaluation (instant). Each evaluation merges its steps' samples.""" + n_steps = range_steps(token, step_s) + steps = acc["steps"] + if not steps: + raise ValueError(f"{dataset}/{q['id']}: no positive values") + n_positive, mle_sample = merge_samples(list(steps.values())) + jobs = [(mle_sample, True)] + rows_per_eval = [] + first, last = span + for t in range(first + n_steps - 1, last + 1): + parts = [steps[s] for s in range(t - n_steps + 1, t + 1) if s in steps] + if not parts: + continue + n_t, sample = merge_samples(parts) + rows_per_eval.append(n_t) + if n_t >= min_rows: + jobs.append((sample, False)) + fits = pool.starmap(fit_power_law, jobs) + mle_fit = fits[0] + if plot_dir is not None: + plot_ccdf( + mle_sample, + mle_fit, + f"{dataset}: {range_promql(q, token)}", + plot_path(plot_dir, dataset, f"{q['id']}__ccdf__{token}"), + ) + lower, upper, n_evals = window_bounds( + [f["alpha"] for f in fits[1:]], mle_fit["alpha"] + ) + eval_stats = spread("rows_win", rows_per_eval) + return { + "dataset": dataset, + "query_id": q["id"], + "promql": range_promql(q, token), + "kind": "values", + "weight": "", + "range": token, + "range_s": range_secs(token), + "step_s": step_s, + "rows_total": acc["n_finite"], + **eval_stats, + "n_evals": n_evals, + "lower": lower, + "mle": mle_fit["alpha"], + "upper": upper, + "min_N": eval_stats.get("rows_win_min", np.nan), + "max_N": eval_stats.get("rows_win_max", np.nan), + "worst_alpha_rank": upper, + "worst_alpha_memory": lower, + **query_targets(q), + "dropped_frac": 1.0 - n_positive / acc["n_finite"], + **{k: v for k, v in mle_fit.items() if k != "alpha"}, + } + + +def analyze_table( + cfg: Dict[str, Any], + table: Dict[str, Any], + queries: List[Dict[str, Any]], + data_root: str, + pool: Any, + args: argparse.Namespace, + plot_dir: Optional[Path], +) -> List[Dict[str, Any]]: + # Load each join once here rather than in every file task. + joins = [(j, load_join(data_root, table, j)) for j in table.get("joins", [])] + tasks = [ + (data_root, table, path, queries, joins, args.max_rows) + for path in expand_files(data_root, table["files"]) + ] + partials = pool.map(aggregate_file, tasks) + span = ( + int(min(p["span"][0] for p in partials)), + int(max(p["span"][1] for p in partials)), + ) + log.info( + "read %d rows (%d malformed rows skipped) from %d files, steps %d..%d", + sum(p["rows_read"] for p in partials), + sum(p["bad_rows"] for p in partials), + len(partials), + *span, + ) + instant_queries = [q for q in queries if INSTANT in q["range"]] + boundary: Dict[Tuple[str, str], Any] = {} + if instant_queries: + boundary = resolve_boundaries( + [p.pop("boundary") for p in partials], + instant_queries, + lookback_steps(table), + ) + out: List[Dict[str, Any]] = [] + for q in queries: + key = query_key(q) + merge: Any = merge_key_parts if q["kind"] == "keys" else merge_step_samples + # Pop parts as they are merged to free memory early. + per_step: Any = ( + merge([p[q["kind"]].pop(key) for p in partials]) if has_range(q) else None + ) + for token in q["range"]: + agg: Any + if token == INSTANT: + agg = merge([p["instant"].pop(key) for p in partials] + [boundary[key]]) + else: + agg = per_step + if q["kind"] == "keys": + out.extend( + summarize_keys( + cfg["dataset"], + q, + token, + table["step_s"], + agg, + span, + args.min_eval_rows, + args.min_eval_keys, + plot_dir, + ) + ) + else: + out.append( + summarize_values( + cfg["dataset"], + q, + token, + table["step_s"], + agg, + span, + pool, + args.min_eval_rows, + plot_dir, + ) + ) + log.info("%s %s %s %s done", cfg["dataset"], q["id"], q["kind"], token) + return out + + +def boom_fit_jobs( + shifted: np.ndarray, n_chunks: int, min_rows: int +) -> Tuple[List[Tuple[np.ndarray, bool]], List[int]]: + """Fit jobs for each variate (full series with comparisons, then each + chunk) and the variate each job belongs to.""" + jobs: List[Tuple[np.ndarray, bool]] = [] + owners: List[int] = [] + for v, y in enumerate(shifted): + jobs.append((subsample(y[y > 0]), True)) + owners.append(v) + for chunk in np.array_split(y, n_chunks): + chunk = chunk[chunk > 0] + if len(chunk) >= min_rows: + jobs.append((subsample(chunk), False)) + owners.append(v) + return jobs, owners + + +def summarize_boom_series( + dataset: str, + q: Dict[str, Any], + series: str, + shifted: np.ndarray, + jobs: List[Tuple[np.ndarray, bool]], + owners: List[int], + fits: List[Dict[str, Any]], + plot_dir: Optional[Path], +) -> Dict[str, Any]: + """Majority tail class over the variates; alpha and diagnostics are medians + over the non-light variates, empty if every variate is light.""" + full: Dict[int, Dict[str, Any]] = {} + samples: Dict[int, np.ndarray] = {} + chunk_alphas: Dict[int, List[float]] = {v: [] for v in range(len(shifted))} + for (x, is_full), v, fit in zip(jobs, owners, fits): + if is_full: + full[v], samples[v] = fit, x + else: + chunk_alphas[v].append(fit["alpha"]) + classes = [f["tail_class"] for f in full.values() if "tail_class" in f] + chosen = [ + v for v, f in full.items() if f.get("tail_class", TAIL_LIGHT) != TAIL_LIGHT + ] + finite = int(np.isfinite(shifted).sum()) + row: Dict[str, Any] = { + "dataset": dataset, + "query_id": f"{q['id']}[{series}]", + "promql": q["promql"], + "kind": "values", + "weight": "", + "K_total": len(shifted), + "rows_total": finite, + "dropped_frac": 1.0 - np.sum(shifted > 0) / finite, + "tail_class": max(sorted(set(classes)), key=classes.count) if classes else "", + "ok_frac": len(chosen) / len(classes) if classes else np.nan, + "n_evals": 0, + **query_targets(q), + } + if chosen: + bounds = [window_bounds(chunk_alphas[v], full[v]["alpha"]) for v in chosen] + row.update( + n_evals=sum(b[2] for b in bounds), + lower=median_or_nan([b[0] for b in bounds]), + mle=median_or_nan([full[v]["alpha"] for v in chosen]), + upper=median_or_nan([b[1] for b in bounds]), + ) + row.update(worst_alpha_rank=row["upper"], worst_alpha_memory=row["lower"]) + for col in ("xmin", "ks_d", "tail_frac") + tuple( + f"{k}_{alt}" for alt in POWER_LAW_ALTERNATIVES for k in ("R", "p") + ): + row[col] = median_or_nan([full[v].get(col, np.nan) for v in chosen]) + if plot_dir is not None and chosen: + # Show the chosen variate whose alpha is closest to their median. + median_alpha = median_or_nan([full[v]["alpha"] for v in chosen]) + v = min(chosen, key=lambda v: abs(full[v]["alpha"] - median_alpha)) + plot_ccdf( + samples[v], + full[v], + f"boom {series}: variate {v} of {len(shifted)}", + plot_path(plot_dir, dataset, f"{series}__ccdf"), + ) + return row + + +def analyze_boom( + cfg: Dict[str, Any], + table: Dict[str, Any], + q: Dict[str, Any], + data_root: str, + pool: Any, + args: argparse.Namespace, + plot_dir: Optional[Path], +) -> List[Dict[str, Any]]: + """Per-variate tail fits on x - min(x), summarized per series.""" + out = [] + for path in expand_files(data_root, table["files"]): + variates = read_boom_series(path) + if args.max_rows is not None: + variates = variates[:, : args.max_rows] + shifted = variates - np.nanmin(variates, axis=1, keepdims=True) + jobs, owners = boom_fit_jobs(shifted, cfg["n_chunks"], args.min_eval_rows) + fits = pool.starmap(fit_power_law, jobs) + series = Path(path).parent.name + out.append( + summarize_boom_series( + cfg["dataset"], q, series, shifted, jobs, owners, fits, plot_dir + ) + ) + log.info("boom %s done", series) + return out + + +def analyze_dataset( + cfg: Dict[str, Any], + data_root: str, + pool: Any, + args: argparse.Namespace, + plot_dir: Optional[Path], +) -> List[Dict[str, Any]]: + out: List[Dict[str, Any]] = [] + for name, table in cfg["tables"].items(): + queries = [q for q in cfg["queries"] if q["table"] == name] + if not queries: + continue + log.info("%s: table %s, %d queries", cfg["dataset"], name, len(queries)) + if table["format"] == "boom_arrow": + for q in queries: + out.extend(analyze_boom(cfg, table, q, data_root, pool, args, plot_dir)) + else: + out.extend( + analyze_table(cfg, table, queries, data_root, pool, args, plot_dir) + ) + return out + + +def parse_args(argv: Optional[Sequence[str]] = None) -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--data-root", required=True, help="fetch_data.sh DATA_ROOT") + parser.add_argument("--queries", nargs="+", default=DEFAULT_QUERIES) + parser.add_argument("--out", type=Path, default=DEFAULT_OUT, help="plot dir") + parser.add_argument("--summary", type=Path, default=DEFAULT_SUMMARY) + parser.add_argument("--no-plots", action="store_true") + parser.add_argument( + "--max-rows", + type=int, + default=None, + help="rows read per file (BOOM: time steps per series), for smoke runs", + ) + parser.add_argument( + "--min-eval-rows", + type=int, + default=DEFAULT_MIN_EVAL_ROWS, + help="skip evaluations with fewer rows (values: fewer positive values)", + ) + parser.add_argument( + "--min-eval-keys", + type=int, + default=DEFAULT_MIN_EVAL_KEYS, + help="skip key-query evaluations with fewer distinct keys", + ) + parser.add_argument("--workers", type=int, default=os.cpu_count()) + return parser.parse_args(argv) + + +def main() -> None: + logging.basicConfig(level=logging.INFO, format="%(asctime)s %(message)s") + args = parse_args() + configs = [load_config(path) for path in args.queries] + plot_dir = None if args.no_plots else args.out + if plot_dir is not None: + plot_dir.mkdir(parents=True, exist_ok=True) + start = time.time() + rows: List[Dict[str, Any]] = [] + with Pool(args.workers) as pool: + for cfg in configs: + rows.extend(analyze_dataset(cfg, args.data_root, pool, args, plot_dir)) + args.summary.parent.mkdir(parents=True, exist_ok=True) + summary = pd.DataFrame(rows).reindex(columns=SUMMARY_COLUMNS) + summary.to_csv(args.summary, index=False, float_format=SUMMARY_FLOAT_FORMAT) + log.info( + "wrote %s (%d rows) in %.0fs", args.summary, len(rows), time.time() - start + ) + + +if __name__ == "__main__": + main() diff --git a/asap-tools/dataset-analysis/queries/alibaba_v2022.yaml b/asap-tools/dataset-analysis/queries/alibaba_v2022.yaml new file mode 100644 index 00000000..7ce8a085 --- /dev/null +++ b/asap-tools/dataset-analysis/queries/alibaba_v2022.yaml @@ -0,0 +1,186 @@ +# Alibaba cluster-trace-microservices-v2022: first 6 hours of CallGraph and +# MCRRTUpdate, first day of MSMetricsUpdate and NodeMetricsUpdate. +# +# Tables: step_s is the sampling period and the evaluation step; series_key +# (metric tables only) is the label set identifying one series. +# Queries: range lists `instant` and/or durations; promql is the instant form +# and promql_range the range form with {range} substituted. cms_weight is the +# key weight a per-key counter sketch sees (default count; value for sum-by). +# targets overrides the default accuracy targets in fit_skew.py. +dataset: alibaba_v2022 +tables: + MSRTMCR: + format: alibaba_tar + files: ["alibaba-v2022/MCRRTUpdate_*.tar.gz"] + time_column: timestamp + time_unit_secs: 1.0e-3 + step_s: 60 + series_key: [msname, msinstanceid, nodeid] + label_columns: [msname, msinstanceid, nodeid] + value_columns: [providerrpc_rt, providerrpc_mcr] + CallGraph: + format: alibaba_tar + files: ["alibaba-v2022/CallGraph_*.tar.gz"] + time_column: timestamp + time_unit_secs: 1.0e-3 + step_s: 60 + label_columns: [traceid, service, rpc_id, rpctype, um, uminstanceid, interface, dm, dminstanceid] + value_columns: [rt] + MSMetrics: + format: alibaba_tar + files: ["alibaba-v2022/MSMetricsUpdate_*.tar.gz"] + time_column: timestamp + time_unit_secs: 1.0e-3 + step_s: 60 + series_key: [msname, msinstanceid, nodeid] + label_columns: [msname, msinstanceid, nodeid] + value_columns: [cpu_utilization, memory_utilization] + NodeMetrics: + format: alibaba_tar + files: ["alibaba-v2022/NodeMetricsUpdate_*.tar.gz"] + time_column: timestamp + time_unit_secs: 1.0e-3 + step_s: 60 + series_key: [nodeid] + label_columns: [nodeid] + value_columns: [cpu_utilization, memory_utilization] +queries: + - id: mcr_by_msname + promql: sum by (msname) (providerrpc_mcr) + promql_range: sum by (msname) (sum_over_time(providerrpc_mcr[{range}])) + range: [instant, 5m, 1h] + table: MSRTMCR + kind: keys + group_by: [msname] + value: providerrpc_mcr + weights: [count, value] + cms_weight: value + - id: mcr_by_nodeid + promql: sum by (nodeid) (providerrpc_mcr) + promql_range: sum by (nodeid) (sum_over_time(providerrpc_mcr[{range}])) + range: [instant, 5m, 1h] + table: MSRTMCR + kind: keys + group_by: [nodeid] + value: providerrpc_mcr + weights: [count, value] + cms_weight: value + - id: rt_p99_by_msname + promql: quantile by (msname) (0.99, providerrpc_rt) + promql_range: quantile by (msname) (0.99, quantile_over_time(0.99, providerrpc_rt[{range}])) + range: [instant, 5m, 1h] + table: MSRTMCR + kind: keys + group_by: [msname] + value: providerrpc_rt + weights: [count, value] + - id: rt_p99_by_msname + promql: quantile(0.99, providerrpc_rt) + promql_range: quantile_over_time(0.99, providerrpc_rt[{range}]) + range: [instant, 5m, 1h] + table: MSRTMCR + kind: values + group_by: [] + value: providerrpc_rt + # CallGraph rows are events (one per call), so only range queries apply. + - id: calls_by_rpctype + promql_range: sum by (rpctype) (count_over_time(rt[{range}])) + range: [1m, 5m, 1h] + table: CallGraph + kind: keys + group_by: [rpctype] + weights: [count] + - id: calls_by_service + promql_range: sum by (service) (count_over_time(rt[{range}])) + range: [1m, 5m, 1h] + table: CallGraph + kind: keys + group_by: [service] + weights: [count] + - id: calls_by_um + promql_range: sum by (um) (count_over_time(rt[{range}])) + range: [1m, 5m, 1h] + table: CallGraph + kind: keys + group_by: [um] + weights: [count] + - id: calls_by_dm + promql_range: sum by (dm) (count_over_time(rt[{range}])) + range: [1m, 5m, 1h] + table: CallGraph + kind: keys + group_by: [dm] + weights: [count] + - id: calls_by_interface + promql_range: sum by (interface) (count_over_time(rt[{range}])) + range: [1m, 5m, 1h] + table: CallGraph + kind: keys + group_by: [interface] + weights: [count] + - id: calls_by_um_dm + promql_range: sum by (um, dm) (count_over_time(rt[{range}])) + range: [1m, 5m, 1h] + table: CallGraph + kind: keys + group_by: [um, dm] + weights: [count] + - id: calls_by_service_um_dm + promql_range: sum by (service, um, dm) (count_over_time(rt[{range}])) + range: [1m, 5m, 1h] + table: CallGraph + kind: keys + group_by: [service, um, dm] + weights: [count] + - id: rt_p99_by_service + promql_range: quantile by (service) (0.99, quantile_over_time(0.99, rt[{range}])) + range: [1m, 5m, 1h] + table: CallGraph + kind: keys + group_by: [service] + value: rt + weights: [count, value] + - id: rt_p99_by_service + promql_range: quantile_over_time(0.99, rt[{range}]) + range: [1m, 5m, 1h] + table: CallGraph + kind: values + group_by: [] + value: rt + - id: ms_cpu_by_msname + promql: sum by (msname) (cpu_utilization) + promql_range: sum by (msname) (sum_over_time(cpu_utilization[{range}])) + range: [instant, 5m, 1h] + table: MSMetrics + kind: keys + group_by: [msname] + value: cpu_utilization + weights: [count, value] + cms_weight: value + # Baseline: one key per node, expected to be close to uniform. + - id: node_cpu_by_nodeid + promql: sum by (nodeid) (cpu_utilization) + promql_range: sum by (nodeid) (sum_over_time(cpu_utilization[{range}])) + range: [instant, 5m, 1h] + table: NodeMetrics + kind: keys + group_by: [nodeid] + value: cpu_utilization + weights: [count, value] + cms_weight: value + - id: ms_cpu_p99 + promql: quantile(0.99, cpu_utilization) + promql_range: quantile_over_time(0.99, cpu_utilization[{range}]) + range: [instant, 5m, 1h] + table: MSMetrics + kind: values + group_by: [] + value: cpu_utilization + - id: node_cpu_p99 + promql: quantile(0.99, cpu_utilization) + promql_range: quantile_over_time(0.99, cpu_utilization[{range}]) + range: [instant, 5m, 1h] + table: NodeMetrics + kind: values + group_by: [] + value: cpu_utilization diff --git a/asap-tools/dataset-analysis/queries/boom.yaml b/asap-tools/dataset-analysis/queries/boom.yaml new file mode 100644 index 00000000..3797194d --- /dev/null +++ b/asap-tools/dataset-analysis/queries/boom.yaml @@ -0,0 +1,38 @@ +# Datadog BOOM: 20 multivariate series. Tags are stripped and each variate is +# z-scored, so only the value tail is fitted, per variate on x - min(x). +dataset: boom +# Each series is split into this many equal-length chunks for the bounds. +n_chunks: 20 +tables: + series: + format: boom_arrow + files: + - boom/ds-2187-H/data-00000-of-00001.arrow + - boom/ds-2394-D/data-00000-of-00001.arrow + - boom/ds-1135-5T/data-00000-of-00001.arrow + - boom/ds-1833-D/data-00000-of-00001.arrow + - boom/ds-2806-D/data-00000-of-00001.arrow + - boom/ds-2573-D/data-00000-of-00001.arrow + - boom/ds-2222-30T/data-00000-of-00001.arrow + - boom/ds-1914-30T/data-00000-of-00001.arrow + - boom/ds-2577-H/data-00000-of-00001.arrow + - boom/ds-2212-D/data-00000-of-00001.arrow + - boom/ds-2782-H/data-00000-of-00001.arrow + - boom/ds-1400-10S/data-00000-of-00001.arrow + - boom/ds-2650-D/data-00000-of-00001.arrow + - boom/ds-1316-10S/data-00000-of-00001.arrow + - boom/ds-1558-5T/data-00000-of-00001.arrow + - boom/ds-1972-D/data-00000-of-00001.arrow + - boom/ds-1840-D/data-00000-of-00001.arrow + - boom/ds-671-10S/data-00000-of-00001.arrow + - boom/ds-1524-5T/data-00000-of-00001.arrow + - boom/ds-899-T/data-00000-of-00001.arrow + label_columns: [] + value_columns: [target] +queries: + - id: target_tail + promql: quantile(0.99, target) + table: series + kind: values + group_by: [] + value: target diff --git a/asap-tools/dataset-analysis/queries/google_2011.yaml b/asap-tools/dataset-analysis/queries/google_2011.yaml new file mode 100644 index 00000000..e2ba46bf --- /dev/null +++ b/asap-tools/dataset-analysis/queries/google_2011.yaml @@ -0,0 +1,113 @@ +# Google ClusterData 2011-2: task_usage parts 0-119 (about 7 days) joined with +# task and job attributes. Fields are described in alibaba_v2022.yaml. +dataset: google_2011 +tables: + task_usage: + format: google_csv + schema: google-2011/schema.csv + files: ["google-2011/task_usage/part-*-of-00500.csv.gz"] + # Each row measures a 5-minute interval; it is stamped at its end time. + time_column: end_time + time_unit_secs: 1.0e-6 + step_s: 300 + series_key: [job_id, task_index, machine_id] + # Each join keeps the last event per key (sorted by sort_by) and left-joins it. + joins: + - files: ["google-2011/task_events/part-*-of-00500.csv.gz"] + keys: [job_id, task_index] + sort_by: time + columns: [user, priority, scheduling_class] + - files: [google-2011/job_events/part-*-of-00500.csv.gz] + keys: [job_id] + sort_by: time + columns: [logical_job_name] + label_columns: [job_id, task_index, machine_id, user, priority, scheduling_class, logical_job_name] + value_columns: [cpu_rate, canonical_memory_usage] +queries: + - id: cpu_by_user + promql: sum by (user) (cpu_rate) + promql_range: sum by (user) (sum_over_time(cpu_rate[{range}])) + range: [instant, 5m, 1h] + table: task_usage + kind: keys + group_by: [user] + value: cpu_rate + weights: [count, value] + cms_weight: value + - id: cpu_by_user_priority + promql: sum by (user, priority) (cpu_rate) + promql_range: sum by (user, priority) (sum_over_time(cpu_rate[{range}])) + range: [instant, 5m, 1h] + table: task_usage + kind: keys + group_by: [user, priority] + value: cpu_rate + weights: [count, value] + cms_weight: value + - id: cpu_by_priority + promql: sum by (priority) (cpu_rate) + promql_range: sum by (priority) (sum_over_time(cpu_rate[{range}])) + range: [instant, 5m, 1h] + table: task_usage + kind: keys + group_by: [priority] + value: cpu_rate + weights: [count, value] + cms_weight: value + - id: cpu_by_job_id + promql: sum by (job_id) (cpu_rate) + promql_range: sum by (job_id) (sum_over_time(cpu_rate[{range}])) + range: [instant, 5m, 1h] + table: task_usage + kind: keys + group_by: [job_id] + value: cpu_rate + weights: [count, value] + cms_weight: value + - id: cpu_by_logical_job_name + promql: sum by (logical_job_name) (cpu_rate) + promql_range: sum by (logical_job_name) (sum_over_time(cpu_rate[{range}])) + range: [instant, 5m, 1h] + table: task_usage + kind: keys + group_by: [logical_job_name] + value: cpu_rate + weights: [count, value] + cms_weight: value + - id: cpu_by_scheduling_class + promql: sum by (scheduling_class) (cpu_rate) + promql_range: sum by (scheduling_class) (sum_over_time(cpu_rate[{range}])) + range: [instant, 5m, 1h] + table: task_usage + kind: keys + group_by: [scheduling_class] + value: cpu_rate + weights: [count, value] + cms_weight: value + # Baseline: one key per machine, expected to be close to uniform. + - id: cpu_by_machine_id + promql: sum by (machine_id) (cpu_rate) + promql_range: sum by (machine_id) (sum_over_time(cpu_rate[{range}])) + range: [instant, 5m, 1h] + table: task_usage + kind: keys + group_by: [machine_id] + value: cpu_rate + weights: [count, value] + cms_weight: value + - id: cpu_p99 + promql: quantile(0.99, cpu_rate) + promql_range: quantile_over_time(0.99, cpu_rate[{range}]) + range: [instant, 5m, 1h] + table: task_usage + kind: values + group_by: [] + value: cpu_rate + - id: memory_p99 + promql: quantile(0.99, canonical_memory_usage) + promql_range: quantile_over_time(0.99, canonical_memory_usage[{range}]) + range: [instant, 5m, 1h] + table: task_usage + kind: values + group_by: [] + value: canonical_memory_usage diff --git a/asap-tools/dataset-analysis/requirements.txt b/asap-tools/dataset-analysis/requirements.txt new file mode 100644 index 00000000..2faa653d --- /dev/null +++ b/asap-tools/dataset-analysis/requirements.txt @@ -0,0 +1,8 @@ +matplotlib==3.10.9 +numpy==2.2.6 +pandas==2.3.3 +# 2.x caps alpha at 3 by default and falls back to slow numerical fits. +powerlaw==1.5 +pyarrow==25.0.1 +PyYAML==6.0.3 +scipy==1.15.3 diff --git a/asap-tools/dataset-analysis/results/skew_summary.csv b/asap-tools/dataset-analysis/results/skew_summary.csv new file mode 100644 index 00000000..3b27643c --- /dev/null +++ b/asap-tools/dataset-analysis/results/skew_summary.csv @@ -0,0 +1,138 @@ +dataset,query_id,promql,kind,weight,range,range_s,step_s,K_total,rows_total,K_win_min,K_win_median,K_win_max,rows_win_min,rows_win_median,rows_win_max,n_evals,lower,mle,upper,worst_theta_cms,worst_K,min_N,max_N,worst_alpha_rank,worst_alpha_memory,target_are_top100,target_precision_at_k,target_hll_rel_err,target_rank_err,top1_share,theta_ls,dropped_frac,xmin,ks_d,tail_frac,R_lognormal,p_lognormal,R_exponential,p_exponential,tail_class,ok_frac +alibaba_v2022,mcr_by_msname,sum by (msname) (providerrpc_mcr),keys,count,instant,,60,26147,188487731,25889,25958,26019,511654,517405.5,521344,360,0.825078553,0.8257464425,0.8262605267,,26019,511654,521344,,,0.05,0.95,0.02,0.01,0.005095726894,1.349104357,,,,,,,,,, +alibaba_v2022,mcr_by_msname,sum by (msname) (providerrpc_mcr),keys,value,instant,,60,18730,188487731,25889,25958,26019,511654,517405.5,521344,360,0.9838294103,1.002188064,1.011853042,0.9838294103,26019,511654,521344,,,0.05,0.95,0.02,0.01,0.03789945576,3.805505342,,,,,,,,,, +alibaba_v2022,mcr_by_msname,sum by (msname) (sum_over_time(providerrpc_mcr[5m])),keys,count,5m,300,60,26147,867646875,25889,25958,26019,8899080,12088243.5,20040209,356,1.328445287,1.443313195,1.664745,,26019,8899080,20040209,,,0.05,0.95,0.02,0.01,0.4545719547,1.37444187,,,,,,,,,, +alibaba_v2022,mcr_by_msname,sum by (msname) (sum_over_time(providerrpc_mcr[5m])),keys,value,5m,300,60,18730,867646875,25889,25958,26019,8899080,12088243.5,20040209,356,1.058947546,1.101430346,1.181235381,1.058947546,26019,8899080,20040209,,,0.05,0.95,0.02,0.01,0.1498488827,3.811405825,,,,,,,,,, +alibaba_v2022,mcr_by_msname,sum by (msname) (sum_over_time(providerrpc_mcr[1h])),keys,count,1h,3600,60,26147,867646875,26028,26051,26094,125327081,139361791,166100193,301,1.386390161,1.443313195,1.499657657,,26094,125327081,166100193,,,0.05,0.95,0.02,0.01,0.4545719547,1.37444187,,,,,,,,,, +alibaba_v2022,mcr_by_msname,sum by (msname) (sum_over_time(providerrpc_mcr[1h])),keys,value,1h,3600,60,18730,867646875,26028,26051,26094,125327081,139361791,166100193,301,1.080487406,1.101430346,1.13802497,1.080487406,26094,125327081,166100193,,,0.05,0.95,0.02,0.01,0.1498488827,3.811405825,,,,,,,,,, +alibaba_v2022,mcr_by_nodeid,sum by (nodeid) (providerrpc_mcr),keys,count,instant,,60,30825,188487731,30686,30726,30783,511654,517405.5,521344,360,0.3328414645,0.3344975088,0.3344975088,,30783,511654,521344,,,0.05,0.95,0.02,0.01,0.0001346559793,0.5851563748,,,,,,,,,, +alibaba_v2022,mcr_by_nodeid,sum by (nodeid) (providerrpc_mcr),keys,value,instant,,60,28348,188487731,30686,30726,30783,511654,517405.5,521344,360,0.4702333583,0.4945285204,0.5382725596,0.4702333583,30783,511654,521344,,,0.05,0.95,0.02,0.01,0.0006887262111,0.9051588471,,,,,,,,,, +alibaba_v2022,mcr_by_nodeid,sum by (nodeid) (sum_over_time(providerrpc_mcr[5m])),keys,count,5m,300,60,30825,867646875,30686,30724,30783,8899080,12088243.5,20040209,356,1.063438048,1.097553229,1.175793283,,30783,8899080,20040209,,,0.05,0.95,0.02,0.01,0.02685695145,0.6757970814,,,,,,,,,, +alibaba_v2022,mcr_by_nodeid,sum by (nodeid) (sum_over_time(providerrpc_mcr[5m])),keys,value,5m,300,60,28348,867646875,30686,30724,30783,8899080,12088243.5,20040209,356,0.6840111645,0.7429289939,0.8586025258,0.6840111645,30783,8899080,20040209,,,0.05,0.95,0.02,0.01,0.007415134693,0.9461668197,,,,,,,,,, +alibaba_v2022,mcr_by_nodeid,sum by (nodeid) (sum_over_time(providerrpc_mcr[1h])),keys,count,1h,3600,60,30825,867646875,30705,30744,30799,125327081,139361791,166100193,301,1.083318419,1.097553229,1.112654503,,30799,125327081,166100193,,,0.05,0.95,0.02,0.01,0.02685695145,0.6757970814,,,,,,,,,, +alibaba_v2022,mcr_by_nodeid,sum by (nodeid) (sum_over_time(providerrpc_mcr[1h])),keys,value,1h,3600,60,28348,867646875,30705,30744,30799,125327081,139361791,166100193,301,0.7146426913,0.7429289939,0.7950597902,0.7146426913,30799,125327081,166100193,,,0.05,0.95,0.02,0.01,0.007415134693,0.9461668197,,,,,,,,,, +alibaba_v2022,rt_p99_by_msname,"quantile by (msname) (0.99, providerrpc_rt)",keys,count,instant,,60,26147,188487731,25889,25958,26019,511654,517405.5,521344,360,0.825078553,0.8257464425,0.8262605267,0.825078553,26019,511654,521344,,,0.05,0.95,0.02,0.01,0.005095726894,1.349104357,,,,,,,,,, +alibaba_v2022,rt_p99_by_msname,"quantile by (msname) (0.99, providerrpc_rt)",keys,value,instant,,60,18721,188487731,25889,25958,26019,511654,517405.5,521344,360,0.9704918171,0.9962931964,1.113884125,,26019,511654,521344,,,0.05,0.95,0.02,0.01,0.07232418839,2.43660772,,,,,,,,,, +alibaba_v2022,rt_p99_by_msname,"quantile by (msname) (0.99, quantile_over_time(0.99, providerrpc_rt[5m]))",keys,count,5m,300,60,26147,867646875,25889,25958,26019,8899080,12088243.5,20040209,356,1.328445287,1.443313195,1.664745,1.328445287,26019,8899080,20040209,,,0.05,0.95,0.02,0.01,0.4545719547,1.37444187,,,,,,,,,, +alibaba_v2022,rt_p99_by_msname,"quantile by (msname) (0.99, quantile_over_time(0.99, providerrpc_rt[5m]))",keys,value,5m,300,60,18721,867646875,25889,25958,26019,8899080,12088243.5,20040209,356,1.685770501,1.970047387,2.341149975,,26019,8899080,20040209,,,0.05,0.95,0.02,0.01,0.8659870115,2.444903781,,,,,,,,,, +alibaba_v2022,rt_p99_by_msname,"quantile by (msname) (0.99, quantile_over_time(0.99, providerrpc_rt[1h]))",keys,count,1h,3600,60,26147,867646875,26028,26051,26094,125327081,139361791,166100193,301,1.386390161,1.443313195,1.499657657,1.386390161,26094,125327081,166100193,,,0.05,0.95,0.02,0.01,0.4545719547,1.37444187,,,,,,,,,, +alibaba_v2022,rt_p99_by_msname,"quantile by (msname) (0.99, quantile_over_time(0.99, providerrpc_rt[1h]))",keys,value,1h,3600,60,18721,867646875,26028,26051,26094,125327081,139361791,166100193,301,1.845706718,1.970047387,2.084916758,,26094,125327081,166100193,,,0.05,0.95,0.02,0.01,0.8659870115,2.444903781,,,,,,,,,, +alibaba_v2022,rt_p99_by_msname,"quantile(0.99, providerrpc_rt)",values,,instant,,60,,188487731,,,,371197,381090,401372,360,2.327272661,2.697295887,2.917634682,,,371197,401372,2.917634682,2.327272661,0.05,0.95,0.02,0.01,,,0.2611442174,152.5468286,0.02861841946,0.06211,-0.4219459635,0.5136902768,4951.538918,0.003338755863,heavy_inconclusive, +alibaba_v2022,rt_p99_by_msname,"quantile_over_time(0.99, providerrpc_rt[5m])",values,,5m,300,60,,867646875,,,,7751463,10923938.5,18912878,356,2.152247068,2.725233624,12.67166397,,,7751463,18912878,12.67166397,2.152247068,0.05,0.95,0.02,0.01,,,0.09788567037,653.418334,0.0512169711,0.001,-0.3066099099,0.615141357,10.22728826,0.09405626913,heavy_inconclusive, +alibaba_v2022,rt_p99_by_msname,"quantile_over_time(0.99, providerrpc_rt[1h])",values,,1h,3600,60,,867646875,,,,110947221,125117066,152748684,301,2.521861465,2.725233624,9.27532118,,,110947221,152748684,9.27532118,2.521861465,0.05,0.95,0.02,0.01,,,0.09788567037,653.418334,0.0512169711,0.001,-0.3066099099,0.615141357,10.22728826,0.09405626913,heavy_inconclusive, +alibaba_v2022,calls_by_rpctype,sum by (rpctype) (count_over_time(rt[1m])),keys,count,1m,60,60,6,1295335556,6,6,6,64,3564367,5401423,360,1.242292116,1.310265913,1.41486647,1.242292116,6,64,5401423,,,0.05,0.95,0.02,0.01,0.4283451021,2.257448052,,,,,,,,,, +alibaba_v2022,calls_by_rpctype,sum by (rpctype) (count_over_time(rt[5m])),keys,count,5m,300,60,6,1295335556,6,6,6,13888129,17749395,26594881,357,1.272106881,1.310265913,1.391208048,1.272106881,6,13888129,26594881,,,0.05,0.95,0.02,0.01,0.4283451021,2.257448052,,,,,,,,,, +alibaba_v2022,calls_by_rpctype,sum by (rpctype) (count_over_time(rt[1h])),keys,count,1h,3600,60,6,1295335556,6,6,6,185233013,204992179,260437245,302,1.298551553,1.310265913,1.337741437,1.298551553,6,185233013,260437245,,,0.05,0.95,0.02,0.01,0.4283451021,2.257448052,,,,,,,,,, +alibaba_v2022,calls_by_service,sum by (service) (count_over_time(rt[1m])),keys,count,1m,60,60,2769176,1295335556,30,40120,43871,64,3564367,5401423,360,0.9793470382,1.099337547,1.099337547,0.9793470382,43871,64,5401423,,,0.05,0.95,0.02,0.01,0.0630261268,1.274170165,,,,,,,,,, +alibaba_v2022,calls_by_service,sum by (service) (count_over_time(rt[5m])),keys,count,5m,300,60,2769176,1295335556,102106,119394,136321,13888129,17749395,26594881,357,1.021367206,1.099337547,1.099337547,1.021367206,136321,13888129,26594881,,,0.05,0.95,0.02,0.01,0.0630261268,1.274170165,,,,,,,,,, +alibaba_v2022,calls_by_service,sum by (service) (count_over_time(rt[1h])),keys,count,1h,3600,60,2769176,1295335556,559964,639248,785894,185233013,204992179,260437245,302,1.071178148,1.099337547,1.099337547,1.071178148,785894,185233013,260437245,,,0.05,0.95,0.02,0.01,0.0630261268,1.274170165,,,,,,,,,, +alibaba_v2022,calls_by_um,sum by (um) (count_over_time(rt[1m])),keys,count,1m,60,60,22016,1295335556,27,6081,8501,64,3564367,5401423,360,1.093784502,1.185355196,1.208596563,1.093784502,8501,64,5401423,,,0.05,0.95,0.02,0.01,0.1535391521,3.145206929,,,,,,,,,, +alibaba_v2022,calls_by_um,sum by (um) (count_over_time(rt[5m])),keys,count,5m,300,60,22016,1295335556,7830,9168,12512,13888129,17749395,26594881,357,1.113014015,1.185355196,1.216728802,1.113014015,12512,13888129,26594881,,,0.05,0.95,0.02,0.01,0.1535391521,3.145206929,,,,,,,,,, +alibaba_v2022,calls_by_um,sum by (um) (count_over_time(rt[1h])),keys,count,1h,3600,60,22016,1295335556,15346,15828,18110,185233013,204992179,260437245,302,1.147118456,1.185355196,1.208155885,1.147118456,18110,185233013,260437245,,,0.05,0.95,0.02,0.01,0.1535391521,3.145206929,,,,,,,,,, +alibaba_v2022,calls_by_dm,sum by (dm) (count_over_time(rt[1m])),keys,count,1m,60,60,35550,1295335556,35,10420,14067,64,3564367,5401423,360,1.096641576,1.157928672,1.157928672,1.096641576,14067,64,5401423,,,0.05,0.95,0.02,0.01,0.07973378135,3.024759942,,,,,,,,,, +alibaba_v2022,calls_by_dm,sum by (dm) (count_over_time(rt[5m])),keys,count,5m,300,60,35550,1295335556,13705,15785,20479,13888129,17749395,26594881,357,1.112039296,1.157928672,1.160544083,1.112039296,20479,13888129,26594881,,,0.05,0.95,0.02,0.01,0.07973378135,3.024759942,,,,,,,,,, +alibaba_v2022,calls_by_dm,sum by (dm) (count_over_time(rt[1h])),keys,count,1h,3600,60,35550,1295335556,25442,26156.5,29583,185233013,204992179,260437245,302,1.132430313,1.157928672,1.16981668,1.132430313,29583,185233013,260437245,,,0.05,0.95,0.02,0.01,0.07973378135,3.024759942,,,,,,,,,, +alibaba_v2022,calls_by_interface,sum by (interface) (count_over_time(rt[1m])),keys,count,1m,60,60,8655921,1295335556,46,93480,139177,64,3564367,5401423,360,0.9891497351,1.097910713,1.097910713,0.9891497351,139177,64,5401423,,,0.05,0.95,0.02,0.01,0.02572182771,0.8643899176,,,,,,,,,, +alibaba_v2022,calls_by_interface,sum by (interface) (count_over_time(rt[5m])),keys,count,5m,300,60,8655921,1295335556,218334,263943,392354,13888129,17749395,26594881,357,1.021672522,1.097910713,1.097910713,1.021672522,392354,13888129,26594881,,,0.05,0.95,0.02,0.01,0.02572182771,0.8643899176,,,,,,,,,, +alibaba_v2022,calls_by_interface,sum by (interface) (count_over_time(rt[1h])),keys,count,1h,3600,60,8655921,1295335556,1506702,1717590,2131776,185233013,204992179,260437245,302,1.070025164,1.097910713,1.097910713,1.070025164,2131776,185233013,260437245,,,0.05,0.95,0.02,0.01,0.02572182771,0.8643899176,,,,,,,,,, +alibaba_v2022,calls_by_um_dm,"sum by (um, dm) (count_over_time(rt[1m]))",keys,count,1m,60,60,165895,1295335556,47,30289,47939,64,3564367,5401423,360,0.9766069583,1.070642064,1.070642064,0.9766069583,47939,64,5401423,,,0.05,0.95,0.02,0.01,0.04140612504,2.721022211,,,,,,,,,, +alibaba_v2022,calls_by_um_dm,"sum by (um, dm) (count_over_time(rt[5m]))",keys,count,5m,300,60,165895,1295335556,40953,51206,78195,13888129,17749395,26594881,357,0.9959293238,1.070642064,1.072961931,0.9959293238,78195,13888129,26594881,,,0.05,0.95,0.02,0.01,0.04140612504,2.721022211,,,,,,,,,, +alibaba_v2022,calls_by_um_dm,"sum by (um, dm) (count_over_time(rt[1h]))",keys,count,1h,3600,60,165895,1295335556,98924,103769,127687,185233013,204992179,260437245,302,1.029614959,1.070642064,1.084624121,1.029614959,127687,185233013,260437245,,,0.05,0.95,0.02,0.01,0.04140612504,2.721022211,,,,,,,,,, +alibaba_v2022,calls_by_service_um_dm,"sum by (service, um, dm) (count_over_time(rt[1m]))",keys,count,1m,60,60,7963812,1295335556,50,159173,239493,64,3564367,5401423,360,0.8917234928,1.022270823,1.022270823,0.8917234928,239493,64,5401423,,,0.05,0.95,0.02,0.01,0.01313775872,1.385250269,,,,,,,,,, +alibaba_v2022,calls_by_service_um_dm,"sum by (service, um, dm) (count_over_time(rt[5m]))",keys,count,5m,300,60,7963812,1295335556,319216,390377,582262,13888129,17749395,26594881,357,0.9320852478,1.022270823,1.022270823,0.9320852478,582262,13888129,26594881,,,0.05,0.95,0.02,0.01,0.01313775872,1.385250269,,,,,,,,,, +alibaba_v2022,calls_by_service_um_dm,"sum by (service, um, dm) (count_over_time(rt[1h]))",keys,count,1h,3600,60,7963812,1295335556,1713602,1986192.5,2221976,185233013,204992179,260437245,302,0.9867997632,1.022270823,1.022270823,0.9867997632,2221976,185233013,260437245,,,0.05,0.95,0.02,0.01,0.01313775872,1.385250269,,,,,,,,,, +alibaba_v2022,rt_p99_by_service,"quantile by (service) (0.99, quantile_over_time(0.99, rt[1m]))",keys,count,1m,60,60,2769176,1295335556,30,40120,43871,64,3564367,5401423,360,0.9793470382,1.099337547,1.099337547,0.9793470382,43871,64,5401423,,,0.05,0.95,0.02,0.01,0.0630261268,1.274170165,,,,,,,,,, +alibaba_v2022,rt_p99_by_service,"quantile by (service) (0.99, quantile_over_time(0.99, rt[1m]))",keys,value,1m,60,60,2717104,1295335556,30,40120,43871,64,3564367,5401423,360,1.096890188,1.179024529,1.606016772,,43871,64,5401423,,,0.05,0.95,0.02,0.01,0.2835197699,1.778216939,,,,,,,,,, +alibaba_v2022,rt_p99_by_service,"quantile by (service) (0.99, quantile_over_time(0.99, rt[5m]))",keys,count,5m,300,60,2769176,1295335556,102106,119394,136321,13888129,17749395,26594881,357,1.021367206,1.099337547,1.099337547,1.021367206,136321,13888129,26594881,,,0.05,0.95,0.02,0.01,0.0630261268,1.274170165,,,,,,,,,, +alibaba_v2022,rt_p99_by_service,"quantile by (service) (0.99, quantile_over_time(0.99, rt[5m]))",keys,value,5m,300,60,2717104,1295335556,102106,119394,136321,13888129,17749395,26594881,357,1.115950417,1.179024529,1.28406221,,136321,13888129,26594881,,,0.05,0.95,0.02,0.01,0.2835197699,1.778216939,,,,,,,,,, +alibaba_v2022,rt_p99_by_service,"quantile by (service) (0.99, quantile_over_time(0.99, rt[1h]))",keys,count,1h,3600,60,2769176,1295335556,559964,639248,785894,185233013,204992179,260437245,302,1.071178148,1.099337547,1.099337547,1.071178148,785894,185233013,260437245,,,0.05,0.95,0.02,0.01,0.0630261268,1.274170165,,,,,,,,,, +alibaba_v2022,rt_p99_by_service,"quantile by (service) (0.99, quantile_over_time(0.99, rt[1h]))",keys,value,1h,3600,60,2717104,1295335556,559964,639248,785894,185233013,204992179,260437245,302,1.159047439,1.179024529,1.199256386,,785894,185233013,260437245,,,0.05,0.95,0.02,0.01,0.2835197699,1.778216939,,,,,,,,,, +alibaba_v2022,rt_p99_by_service,"quantile_over_time(0.99, rt[1m])",values,,1m,60,60,,1270987041,,,,43,2097984,3311360,360,1.891474819,2.029339494,2.46113073,,,43,3311360,2.46113073,1.891474819,0.05,0.95,0.02,0.01,,,0.3996959565,17,0.03031811014,0.09527,-0.6373337002,0.6539354026,13183.39466,1.699185041e-56,heavy_inconclusive, +alibaba_v2022,rt_p99_by_service,"quantile_over_time(0.99, rt[5m])",values,,5m,300,60,,1270987041,,,,7881752,10442357,16264263,357,1.957358515,2.029339494,2.429250439,,,7881752,16264263,2.429250439,1.957358515,0.05,0.95,0.02,0.01,,,0.3996959565,17,0.03031811014,0.09527,-0.6373337002,0.6539354026,13183.39466,1.699185041e-56,heavy_inconclusive, +alibaba_v2022,rt_p99_by_service,"quantile_over_time(0.99, rt[1h])",values,,1h,3600,60,,1270987041,,,,106821507,119913902,157226182,302,2.007578773,2.029339494,2.297062459,,,106821507,157226182,2.297062459,2.007578773,0.05,0.95,0.02,0.01,,,0.3996959565,17,0.03031811014,0.09527,-0.6373337002,0.6539354026,13183.39466,1.699185041e-56,heavy_inconclusive, +alibaba_v2022,ms_cpu_by_msname,sum by (msname) (cpu_utilization),keys,count,instant,,60,28260,677928534,28143,28159,28171,465931,470224,471603,1440,0.8251239317,0.8257044743,0.8263779532,,28171,465931,471603,,,0.05,0.95,0.02,0.01,0.01642566353,1.278838704,,,,,,,,,, +alibaba_v2022,ms_cpu_by_msname,sum by (msname) (cpu_utilization),keys,value,instant,,60,28259,677928534,28143,28159,28171,465931,470224,471603,1440,0.8708920197,0.8894428844,0.9061510436,0.8708920197,28171,465931,471603,,,0.05,0.95,0.02,0.01,0.01023773901,1.713144009,,,,,,,,,, +alibaba_v2022,ms_cpu_by_msname,sum by (msname) (sum_over_time(cpu_utilization[5m])),keys,count,5m,300,60,28260,675424445,28145,28159,28171,2328999,2348472.5,2355461,1436,0.8250307623,0.8256900161,0.8263540837,,28171,2328999,2355461,,,0.05,0.95,0.02,0.01,0.01642157177,1.279481397,,,,,,,,,, +alibaba_v2022,ms_cpu_by_msname,sum by (msname) (sum_over_time(cpu_utilization[5m])),keys,value,5m,300,60,28259,675424445,28145,28159,28171,2328999,2348472.5,2355461,1436,0.8715262892,0.8893327869,0.9055036603,0.8715262892,28171,2328999,2355461,,,0.05,0.95,0.02,0.01,0.01026331486,1.71275427,,,,,,,,,, +alibaba_v2022,ms_cpu_by_msname,sum by (msname) (sum_over_time(cpu_utilization[1h])),keys,count,1h,3600,60,28260,675424445,28158,28173,28184,27957228,28183500,28254558,1381,0.8252466649,0.8256900161,0.8263148174,,28184,27957228,28254558,,,0.05,0.95,0.02,0.01,0.01642157177,1.279481397,,,,,,,,,, +alibaba_v2022,ms_cpu_by_msname,sum by (msname) (sum_over_time(cpu_utilization[1h])),keys,value,1h,3600,60,28259,675424445,28158,28173,28184,27957228,28183500,28254558,1381,0.8731117944,0.8893327869,0.9040277038,0.8731117944,28184,27957228,28254558,,,0.05,0.95,0.02,0.01,0.01026331486,1.71275427,,,,,,,,,, +alibaba_v2022,ms_cpu_p99,"quantile(0.99, cpu_utilization)",values,,instant,,60,,677928534,,,,465917,470208.5,471589,1440,3.01821605,6.167318107,10.1539801,,,465917,471589,10.1539801,3.01821605,0.05,0.95,0.02,0.01,,,3.964134662e-05,0.5721030984,0.06483467076,0.03156,-139.6838326,1.843295234e-28,-102.9144993,2.130182551e-79,light, +alibaba_v2022,ms_cpu_p99,"quantile_over_time(0.99, cpu_utilization[5m])",values,,5m,300,60,,675424445,,,,2328934,2348401.5,2355395,1436,3.030257924,6.355302868,10.0474611,,,2328934,2355395,10.0474611,3.030257924,0.05,0.95,0.02,0.01,,,3.036461021e-05,0.5741396769,0.05959687717,0.03156,-132.3740985,4.881332668e-27,-97.32785636,1.956182046e-75,light, +alibaba_v2022,ms_cpu_p99,"quantile_over_time(0.99, cpu_utilization[1h])",values,,1h,3600,60,,675424445,,,,27956442,28182559,28253774,1381,3.09158951,6.355302868,7.052208747,,,27956442,28253774,7.052208747,3.09158951,0.05,0.95,0.02,0.01,,,3.036461021e-05,0.5741396769,0.05959687717,0.03156,-132.3740985,4.881332668e-27,-97.32785636,1.956182046e-75,light, +alibaba_v2022,node_cpu_by_nodeid,sum by (nodeid) (cpu_utilization),keys,count,instant,,60,43382,58923660,42235,42333,42957,42235,42333,42957,1383,3.673105566e-06,0.02052130964,0.02052130964,,42957,42235,42957,,,0.05,0.95,0.02,0.01,2.353893156e-05,0.06043580555,,,,,,,,,, +alibaba_v2022,node_cpu_by_nodeid,sum by (nodeid) (cpu_utilization),keys,value,instant,,60,43381,58923660,42235,42333,42957,42235,42333,42957,1383,0.2864927327,0.2864927327,0.3554983787,0.2864927327,42957,42235,42957,,,0.05,0.95,0.02,0.01,9.71328786e-05,0.3768673766,,,,,,,,,, +alibaba_v2022,node_cpu_by_nodeid,sum by (nodeid) (sum_over_time(cpu_utilization[5m])),keys,count,5m,300,60,43382,58460197,42235,42333,42957,42235,211242,214454,1379,3.673105566e-06,0.02250972512,0.02250972512,,42957,42235,214454,,,0.05,0.95,0.02,0.01,2.358869916e-05,0.07484077042,,,,,,,,,, +alibaba_v2022,node_cpu_by_nodeid,sum by (nodeid) (sum_over_time(cpu_utilization[5m])),keys,value,5m,300,60,43381,58460197,42235,42333,42957,42235,211242,214454,1379,0.2852500855,0.2865167137,0.3417391972,0.2852500855,42957,42235,214454,,,0.05,0.95,0.02,0.01,9.702091431e-05,0.3892546546,,,,,,,,,, +alibaba_v2022,node_cpu_by_nodeid,sum by (nodeid) (sum_over_time(cpu_utilization[1h])),keys,count,1h,3600,60,43382,58460197,42235,42424,43027,42235,2534282,2570715,1379,3.673105566e-06,0.02250972512,0.02250972512,,43027,42235,2570715,,,0.05,0.95,0.02,0.01,2.358869916e-05,0.07484077042,,,,,,,,,, +alibaba_v2022,node_cpu_by_nodeid,sum by (nodeid) (sum_over_time(cpu_utilization[1h])),keys,value,1h,3600,60,43381,58460197,42235,42424,43027,42235,2534282,2570715,1379,0.2800713326,0.2865167137,0.3302192401,0.2800713326,43027,42235,2570715,,,0.05,0.95,0.02,0.01,9.702091431e-05,0.3892546546,,,,,,,,,, +alibaba_v2022,node_cpu_p99,"quantile(0.99, cpu_utilization)",values,,instant,,60,,58923660,,,,42233,42331,42955,1383,4.226872374,5.481659723,6.443423228,,,42233,42955,6.443423228,4.226872374,0.05,0.95,0.02,0.01,,,4.979324095e-05,0.3420009998,0.01588971021,0.07229,-3.277517211,0.07361697995,115.7009555,1.534003298e-14,lognormal, +alibaba_v2022,node_cpu_p99,"quantile_over_time(0.99, cpu_utilization[5m])",values,,5m,300,60,,58460197,,,,42233,211232,214444,1379,4.271156891,5.401433245,6.368754462,,,42233,214444,6.368754462,4.271156891,0.05,0.95,0.02,0.01,,,4.945245053e-05,0.330743608,0.01881783954,0.08247,-4.380360956,0.0419334074,139.060049,1.566767944e-16,lognormal, +alibaba_v2022,node_cpu_p99,"quantile_over_time(0.99, cpu_utilization[1h])",values,,1h,3600,60,,58460197,,,,42233,2534156,2570595,1379,4.365719214,5.401433245,6.246543807,,,42233,2570595,6.246543807,4.365719214,0.05,0.95,0.02,0.01,,,4.945245053e-05,0.330743608,0.01881783954,0.08247,-4.380360956,0.0419334074,139.060049,1.566767944e-16,lognormal, +boom,target_tail[ds-2187-H],"quantile(0.99, target)",values,,,,,100,523100,,,,,,,900,1.673862396,2.879820136,5.449343474,,,,,5.449343474,1.673862396,0.05,0.95,0.02,0.01,,,0.0001911680367,1.13650366,0.05719670041,0.1334608031,-0.7518535826,0.1985446789,81.95794065,4.297158362e-06,light,0.45 +boom,target_tail[ds-2394-D],"quantile(0.99, target)",values,,,,,73,10001,,,,,,,0,,,,,,,,,,0.05,0.95,0.02,0.01,,,0.1858814119,,,,,,,,, +boom,target_tail[ds-1135-5T],"quantile(0.99, target)",values,,,,,26,425984,,,,,,,63,2.072418608,2.555273335,4.464465828,,,,,4.464465828,2.072418608,0.05,0.95,0.02,0.01,,,0.5493680514,1.309216454,0.09571546402,0.3780193237,-3.202171683,0.08890763899,233.8257673,9.588002711e-10,light,0.3571428571 +boom,target_tail[ds-1833-D],"quantile(0.99, target)",values,,,,,17,3689,,,,,,,0,,,,,,,,,,0.05,0.95,0.02,0.01,,,0.2396313364,,,,,,,,, +boom,target_tail[ds-2806-D],"quantile(0.99, target)",values,,,,,100,13700,,,,,,,0,,,,,,,,,,0.05,0.95,0.02,0.01,,,0.7003649635,,,,,,,,, +boom,target_tail[ds-2573-D],"quantile(0.99, target)",values,,,,,45,9765,,,,,,,0,7.824807058,7.824807058,7.824807058,,,,,7.824807058,7.824807058,0.05,0.95,0.02,0.01,,,0.007987711214,2.832573324,0.2237516935,0.4814814815,-0.1549932255,0.1811684879,8.450514888,8.137699192e-05,light,0.2444444444 +boom,target_tail[ds-2222-30T],"quantile(0.99, target)",values,,,,,100,1045600,,,,,,,800,5.121075907,27.15946506,45.43533407,,,,,45.43533407,5.121075907,0.05,0.95,0.02,0.01,,,0.5705183627,3.907679107,0.138594883,0.03610214265,-0.3643459076,0.2354562876,16.72715255,3.110465898e-15,heavy_inconclusive,0.97 +boom,target_tail[ds-1914-30T],"quantile(0.99, target)",values,,,,,64,669504,,,,,,,0,,,,,,,,,,0.05,0.95,0.02,0.01,,,0.9991770027,,,,,,,,, +boom,target_tail[ds-2577-H],"quantile(0.99, target)",values,,,,,11,57541,,,,,,,40,2.357316003,3.318605295,7.114686774,,,,,7.114686774,2.357316003,0.05,0.95,0.02,0.01,,,0.0001911680367,1.451908794,0.04583397605,0.4337476099,-3.619220504,0.2878918275,168.5626824,1.8816055e-20,light,0.1818181818 +boom,target_tail[ds-2212-D],"quantile(0.99, target)",values,,,,,42,9114,,,,,,,0,2.097505042,2.097505042,2.097505042,,,,,2.097505042,2.097505042,0.05,0.95,0.02,0.01,,,0.2055080097,0.1728252252,0.07861559668,0.4741809117,-0.6104251813,0.3802005494,57.44497486,0.003636112868,light,0.3636363636 +boom,target_tail[ds-2782-H],"quantile(0.99, target)",values,,,,,20,16500,,,,,,,0,,,,,,,,,,0.05,0.95,0.02,0.01,,,0.9923636364,,,,,,,,, +boom,target_tail[ds-1400-10S],"quantile(0.99, target)",values,,,,,75,1228800,,,,,,,680,3.665809149,5.927304101,8.608042368,,,,,8.608042368,3.665809149,0.05,0.95,0.02,0.01,,,0.0001399739583,3.22594058,0.03461878105,0.1945003968,-0.5020486991,0.1052433657,67.48976464,1.414326313e-05,light,0.4533333333 +boom,target_tail[ds-2650-D],"quantile(0.99, target)",values,,,,,56,12152,,,,,,,0,2.372256328,2.372256328,2.372256328,,,,,2.372256328,2.372256328,0.05,0.95,0.02,0.01,,,0.004772876893,0.3060346595,0.07025218904,0.4814814815,-0.4345545346,0.5532457248,23.54371772,0.03707674656,light,0.25 +boom,target_tail[ds-1316-10S],"quantile(0.99, target)",values,,,,,40,655360,,,,,,,800,1.394454939,2.841176959,3.782745604,,,,,3.782745604,1.394454939,0.05,0.95,0.02,0.01,,,0.1220932007,2.273231677,0.03804312837,0.03161851893,-1.526906431,0.1625683967,48.84256082,4.793863342e-08,heavy_inconclusive,1 +boom,target_tail[ds-1558-5T],"quantile(0.99, target)",values,,,,,97,1589248,,,,,,,1195,2.859011495,4.965505023,14.93467462,,,,,14.93467462,2.859011495,0.05,0.95,0.02,0.01,,,0.1971838253,2.030408032,0.08454367574,0.2353085011,0.01446142949,0.1477685349,143.8265644,7.494706033e-08,heavy_inconclusive,0.6701030928 +boom,target_tail[ds-1972-D],"quantile(0.99, target)",values,,,,,39,5343,,,,,,,0,,,,,,,,,,0.05,0.95,0.02,0.01,,,0.05577390979,,,,,,,,, +boom,target_tail[ds-1840-D],"quantile(0.99, target)",values,,,,,100,21700,,,,,,,0,48.70890868,48.70890868,48.70890868,,,,,48.70890868,48.70890868,0.05,0.95,0.02,0.01,,,0.00465437788,5.808935183,0.1729496632,0.5046296296,-0.1558493932,0.6911382477,15.25220643,2.851033848e-05,light,0.11 +boom,target_tail[ds-671-10S],"quantile(0.99, target)",values,,,,,100,1638400,,,,,,,0,,,,,,,,,,0.05,0.95,0.02,0.01,,,6.103515625e-05,,,,,,,,light,0 +boom,target_tail[ds-1524-5T],"quantile(0.99, target)",values,,,,,93,1267683,,,,,,,18,3.310389413,3.310389413,3.692673948,,,,,3.692673948,3.310389413,0.05,0.95,0.02,0.01,,,0.944850566,1.984224252,0.2043641455,0.523993808,-0.7337541742,0.1428173664,51.73262359,2.004766276e-06,heavy_inconclusive,0.6470588235 +boom,target_tail[ds-899-T],"quantile(0.99, target)",values,,,,,12,196608,,,,,,,144,2.379500335,4.776016519,4.776016519,,,,,4.776016519,2.379500335,0.05,0.95,0.02,0.01,,,0.4203440348,3.316427802,0.0820839487,0.05264234598,-0.6434653813,0.1335383104,64.54346235,4.612421122e-07,heavy_inconclusive,1 +google_2011,cpu_by_user,sum by (user) (cpu_rate),keys,count,instant,,300,686,254393256,480,519,538,100546,126200,212533,2005,1.11463296,1.148419293,1.413295054,,538,100546,212533,,,0.05,0.95,0.02,0.01,0.07735343817,2.565160441,,,,,,,,,, +google_2011,cpu_by_user,sum by (user) (cpu_rate),keys,value,instant,,300,656,254393256,480,519,538,100546,126200,212533,2005,1.193602067,1.244075542,1.482299462,1.193602067,538,100546,212533,,,0.05,0.95,0.02,0.01,0.1261323528,3.723521859,,,,,,,,,, +google_2011,cpu_by_user,sum by (user) (sum_over_time(cpu_rate[5m])),keys,count,5m,300,300,686,288357812,480,519,538,105373,137958,379702,2005,1.101644423,1.146256644,1.675767301,,538,105373,379702,,,0.05,0.95,0.02,0.01,0.07984469656,2.530528113,,,,,,,,,, +google_2011,cpu_by_user,sum by (user) (sum_over_time(cpu_rate[5m])),keys,value,5m,300,300,665,288357812,480,519,538,105373,137958,379702,2005,1.188664169,1.235004601,1.459116669,1.188664169,538,105373,379702,,,0.05,0.95,0.02,0.01,0.1217389859,3.698174511,,,,,,,,,, +google_2011,cpu_by_user,sum by (user) (sum_over_time(cpu_rate[1h])),keys,count,1h,3600,300,686,288357812,507,528,548,1310788,1663612,4179680,1994,1.103070785,1.146256644,1.596141868,,548,1310788,4179680,,,0.05,0.95,0.02,0.01,0.07984469656,2.530528113,,,,,,,,,, +google_2011,cpu_by_user,sum by (user) (sum_over_time(cpu_rate[1h])),keys,value,1h,3600,300,665,288357812,507,528,548,1310788,1663612,4179680,1994,1.198936994,1.235004601,1.357764519,1.198936994,548,1310788,4179680,,,0.05,0.95,0.02,0.01,0.1217389859,3.698174511,,,,,,,,,, +google_2011,cpu_by_user_priority,"sum by (user, priority) (cpu_rate)",keys,count,instant,,300,1356,254393256,788,838,880,100546,126200,212533,2005,1.133137437,1.15789991,1.405238475,,880,100546,212533,,,0.05,0.95,0.02,0.01,0.07636852606,2.71359643,,,,,,,,,, +google_2011,cpu_by_user_priority,"sum by (user, priority) (cpu_rate)",keys,value,instant,,300,1294,254393256,788,838,880,100546,126200,212533,2005,1.220156306,1.266536663,1.486342061,1.220156306,880,100546,212533,,,0.05,0.95,0.02,0.01,0.1259520722,3.694609415,,,,,,,,,, +google_2011,cpu_by_user_priority,"sum by (user, priority) (sum_over_time(cpu_rate[5m]))",keys,count,5m,300,300,1356,288357812,788,838,880,105373,137958,379702,2005,1.123516501,1.15419467,1.65624668,,880,105373,379702,,,0.05,0.95,0.02,0.01,0.07783297718,2.613510728,,,,,,,,,, +google_2011,cpu_by_user_priority,"sum by (user, priority) (sum_over_time(cpu_rate[5m]))",keys,value,5m,300,300,1326,288357812,788,838,880,105373,137958,379702,2005,1.215459225,1.258025912,1.466602293,1.215459225,880,105373,379702,,,0.05,0.95,0.02,0.01,0.121567009,3.607655725,,,,,,,,,, +google_2011,cpu_by_user_priority,"sum by (user, priority) (sum_over_time(cpu_rate[1h]))",keys,count,1h,3600,300,1356,288357812,836,875,918,1310788,1663612,4179680,1994,1.125128087,1.15419467,1.580642019,,918,1310788,4179680,,,0.05,0.95,0.02,0.01,0.07783297718,2.613510728,,,,,,,,,, +google_2011,cpu_by_user_priority,"sum by (user, priority) (sum_over_time(cpu_rate[1h]))",keys,value,1h,3600,300,1326,288357812,836,875,918,1310788,1663612,4179680,1994,1.225676891,1.258025912,1.376433035,1.225676891,918,1310788,4179680,,,0.05,0.95,0.02,0.01,0.121567009,3.607655725,,,,,,,,,, +google_2011,cpu_by_priority,sum by (priority) (cpu_rate),keys,count,instant,,300,11,254393256,7,9,10,100546,126200,212533,2005,1.041417612,1.417093333,1.693290433,,10,100546,212533,,,0.05,0.95,0.02,0.01,0.3956743531,4.248710668,,,,,,,,,, +google_2011,cpu_by_priority,sum by (priority) (cpu_rate),keys,value,instant,,300,11,254393256,7,9,10,100546,126200,212533,2005,1.467698146,1.878339954,2.361124179,1.467698146,10,100546,212533,,,0.05,0.95,0.02,0.01,0.6262266935,4.720397597,,,,,,,,,, +google_2011,cpu_by_priority,sum by (priority) (sum_over_time(cpu_rate[5m])),keys,count,5m,300,300,11,288357812,7,9,10,105373,137958,379702,2005,0.9875708441,1.371566014,2.184801397,,10,105373,379702,,,0.05,0.95,0.02,0.01,0.3625884913,4.062229564,,,,,,,,,, +google_2011,cpu_by_priority,sum by (priority) (sum_over_time(cpu_rate[5m])),keys,value,5m,300,300,11,288357812,7,9,10,105373,137958,379702,2005,1.36454469,1.828458458,2.314749802,1.36454469,10,105373,379702,,,0.05,0.95,0.02,0.01,0.6034462731,4.306203284,,,,,,,,,, +google_2011,cpu_by_priority,sum by (priority) (sum_over_time(cpu_rate[1h])),keys,count,1h,3600,300,11,288357812,7,9,11,1310788,1663612,4179680,1994,1.032055526,1.371566014,2.061975815,,11,1310788,4179680,,,0.05,0.95,0.02,0.01,0.3625884913,4.062229564,,,,,,,,,, +google_2011,cpu_by_priority,sum by (priority) (sum_over_time(cpu_rate[1h])),keys,value,1h,3600,300,11,288357812,7,9,11,1310788,1663612,4179680,1994,1.446809389,1.828458458,2.17016232,1.446809389,11,1310788,4179680,,,0.05,0.95,0.02,0.01,0.6034462731,4.306203284,,,,,,,,,, +google_2011,cpu_by_job_id,sum by (job_id) (cpu_rate),keys,count,instant,,300,149904,254393256,3736,4073,4329,100546,126200,212533,2005,1.018094917,1.111840032,1.287598425,,4329,100546,212533,,,0.05,0.95,0.02,0.01,0.03853351757,2.264320313,,,,,,,,,, +google_2011,cpu_by_job_id,sum by (job_id) (cpu_rate),keys,value,instant,,300,84343,254393256,3736,4073,4329,100546,126200,212533,2005,1.105104651,1.151146338,1.33690346,1.105104651,4329,100546,212533,,,0.05,0.95,0.02,0.01,0.04569306047,3.285347569,,,,,,,,,, +google_2011,cpu_by_job_id,sum by (job_id) (sum_over_time(cpu_rate[5m])),keys,count,5m,300,300,149904,288357812,3736,4073,4329,105373,137958,379702,2005,1.017149621,1.102080327,1.503255855,,4329,105373,379702,,,0.05,0.95,0.02,0.01,0.05772714769,2.06447159,,,,,,,,,, +google_2011,cpu_by_job_id,sum by (job_id) (sum_over_time(cpu_rate[5m])),keys,value,5m,300,300,148594,288357812,3736,4073,4329,105373,137958,379702,2005,1.103163853,1.153824387,1.327898976,1.103163853,4329,105373,379702,,,0.05,0.95,0.02,0.01,0.0430493823,2.930785652,,,,,,,,,, +google_2011,cpu_by_job_id,sum by (job_id) (sum_over_time(cpu_rate[1h])),keys,count,1h,3600,300,149904,288357812,4370,4862,5682,1310788,1663612,4179680,1994,1.007087334,1.102080327,1.424299342,,5682,1310788,4179680,,,0.05,0.95,0.02,0.01,0.05772714769,2.06447159,,,,,,,,,, +google_2011,cpu_by_job_id,sum by (job_id) (sum_over_time(cpu_rate[1h])),keys,value,1h,3600,300,148594,288357812,4370,4862,5682,1310788,1663612,4179680,1994,1.11335546,1.153824387,1.233694396,1.11335546,5682,1310788,4179680,,,0.05,0.95,0.02,0.01,0.0430493823,2.930785652,,,,,,,,,, +google_2011,cpu_by_logical_job_name,sum by (logical_job_name) (cpu_rate),keys,count,instant,,300,17796,254393256,3314,3654,3869,100546,126200,212533,2005,1.02248213,1.089261918,1.289781084,,3869,100546,212533,,,0.05,0.95,0.02,0.01,0.0415626073,3.000008214,,,,,,,,,, +google_2011,cpu_by_logical_job_name,sum by (logical_job_name) (cpu_rate),keys,value,instant,,300,15463,254393256,3314,3654,3869,100546,126200,212533,2005,1.101552401,1.157491198,1.336405436,1.101552401,3869,100546,212533,,,0.05,0.95,0.02,0.01,0.0657728459,3.747414902,,,,,,,,,, +google_2011,cpu_by_logical_job_name,sum by (logical_job_name) (sum_over_time(cpu_rate[5m])),keys,count,5m,300,300,17796,288357812,3314,3654,3869,105373,137958,379702,2005,1.021586211,1.086252759,1.514068393,,3869,105373,379702,,,0.05,0.95,0.02,0.01,0.05772714769,2.749505785,,,,,,,,,, +google_2011,cpu_by_logical_job_name,sum by (logical_job_name) (sum_over_time(cpu_rate[5m])),keys,value,5m,300,300,17686,288357812,3314,3654,3869,105373,137958,379702,2005,1.099326069,1.155715922,1.327207337,1.099326069,3869,105373,379702,,,0.05,0.95,0.02,0.01,0.0619674051,3.460309727,,,,,,,,,, +google_2011,cpu_by_logical_job_name,sum by (logical_job_name) (sum_over_time(cpu_rate[1h])),keys,count,1h,3600,300,17796,288357812,3731,4062,4535,1310788,1663612,4179680,1994,1.008349617,1.086252759,1.43851047,,4535,1310788,4179680,,,0.05,0.95,0.02,0.01,0.05772714769,2.749505785,,,,,,,,,, +google_2011,cpu_by_logical_job_name,sum by (logical_job_name) (sum_over_time(cpu_rate[1h])),keys,value,1h,3600,300,17686,288357812,3731,4062,4535,1310788,1663612,4179680,1994,1.102961084,1.155715922,1.231968728,1.102961084,4535,1310788,4179680,,,0.05,0.95,0.02,0.01,0.0619674051,3.460309727,,,,,,,,,, +google_2011,cpu_by_scheduling_class,sum by (scheduling_class) (cpu_rate),keys,count,instant,,300,4,254393256,4,4,4,100546,126200,212533,2005,0.1342594637,0.2571920412,1.259619862,,4,100546,212533,,,0.05,0.95,0.02,0.01,0.2933255707,0.2655223102,,,,,,,,,, +google_2011,cpu_by_scheduling_class,sum by (scheduling_class) (cpu_rate),keys,value,instant,,300,4,254393256,4,4,4,100546,126200,212533,2005,0.134651854,0.5570643761,1.118978684,0.134651854,4,100546,212533,,,0.05,0.95,0.02,0.01,0.352956055,0.5956980856,,,,,,,,,, +google_2011,cpu_by_scheduling_class,sum by (scheduling_class) (sum_over_time(cpu_rate[5m])),keys,count,5m,300,300,4,288357812,4,4,4,105373,137958,379702,2005,0.1279176611,0.3979267591,2.03910581,,4,105373,379702,,,0.05,0.95,0.02,0.01,0.3291290267,0.4087825709,,,,,,,,,, +google_2011,cpu_by_scheduling_class,sum by (scheduling_class) (sum_over_time(cpu_rate[5m])),keys,value,5m,300,300,4,288357812,4,4,4,105373,137958,379702,2005,0.09835771313,0.474743457,1.072402634,0.09835771313,4,105373,379702,,,0.05,0.95,0.02,0.01,0.3377085349,0.5064505245,,,,,,,,,, +google_2011,cpu_by_scheduling_class,sum by (scheduling_class) (sum_over_time(cpu_rate[1h])),keys,count,1h,3600,300,4,288357812,4,4,4,1310788,1663612,4179680,1994,0.1312536567,0.3979267591,1.873488766,,4,1310788,4179680,,,0.05,0.95,0.02,0.01,0.3291290267,0.4087825709,,,,,,,,,, +google_2011,cpu_by_scheduling_class,sum by (scheduling_class) (sum_over_time(cpu_rate[1h])),keys,value,1h,3600,300,4,288357812,4,4,4,1310788,1663612,4179680,1994,0.2164418871,0.474743457,0.866047153,0.2164418871,4,1310788,4179680,,,0.05,0.95,0.02,0.01,0.3377085349,0.5064505245,,,,,,,,,, +google_2011,cpu_by_machine_id,sum by (machine_id) (cpu_rate),keys,count,instant,,300,12555,254393256,12450,12494,12513,100546,126200,212533,2005,0.2002784015,0.2002784015,0.3641132007,,12513,100546,212533,,,0.05,0.95,0.02,0.01,0.0002062279513,0.2345071156,,,,,,,,,, +google_2011,cpu_by_machine_id,sum by (machine_id) (cpu_rate),keys,value,instant,,300,12555,254393256,12450,12494,12513,100546,126200,212533,2005,0.2760861204,0.2760861204,0.4809930696,0.2760861204,12513,100546,212533,,,0.05,0.95,0.02,0.01,0.0002508010797,0.3495270925,,,,,,,,,, +google_2011,cpu_by_machine_id,sum by (machine_id) (sum_over_time(cpu_rate[5m])),keys,count,5m,300,300,12555,288357812,12450,12494,12513,105373,137958,379702,2005,0.1923715173,0.1923715173,0.4945129903,,12513,105373,379702,,,0.05,0.95,0.02,0.01,0.001366878176,0.2201118062,,,,,,,,,, +google_2011,cpu_by_machine_id,sum by (machine_id) (sum_over_time(cpu_rate[5m])),keys,value,5m,300,300,12555,288357812,12450,12494,12513,105373,137958,379702,2005,0.264742317,0.264742317,0.4631528705,0.264742317,12513,105373,379702,,,0.05,0.95,0.02,0.01,0.0002457267992,0.3330981966,,,,,,,,,, +google_2011,cpu_by_machine_id,sum by (machine_id) (sum_over_time(cpu_rate[1h])),keys,count,1h,3600,300,12555,288357812,12464,12499,12517,1310788,1663612,4179680,1994,0.1923715173,0.1923715173,0.4151173444,,12517,1310788,4179680,,,0.05,0.95,0.02,0.01,0.001366878176,0.2201118062,,,,,,,,,, +google_2011,cpu_by_machine_id,sum by (machine_id) (sum_over_time(cpu_rate[1h])),keys,value,1h,3600,300,12555,288357812,12464,12499,12517,1310788,1663612,4179680,1994,0.2554549929,0.264742317,0.3833340129,0.2554549929,12517,1310788,4179680,,,0.05,0.95,0.02,0.01,0.0002457267992,0.3330981966,,,,,,,,,, +google_2011,cpu_p99,"quantile(0.99, cpu_rate)",values,,instant,,300,,254393256,,,,91155,110621,138139,2005,2.251147561,2.886080738,21.48065083,,,91155,138139,21.48065083,2.251147561,0.05,0.95,0.02,0.01,,,0.1204071935,0.04236,0.02938791912,0.18458,-36.31165363,1.412366445e-10,1571.498664,1.004172398e-168,lognormal, +google_2011,cpu_p99,"quantile_over_time(0.99, cpu_rate[5m])",values,,5m,300,300,,288357812,,,,95686,120112,277263,2005,2.275868657,2.915934134,20.27861172,,,95686,277263,20.27861172,2.275868657,0.05,0.95,0.02,0.01,,,0.1328455773,0.04242,0.02767439641,0.17435,-26.37966239,8.600621204e-08,1545.518962,3.118557454e-133,lognormal, +google_2011,cpu_p99,"quantile_over_time(0.99, cpu_rate[1h])",values,,1h,3600,300,,288357812,,,,1200321,1434201.5,3021637,1994,2.320001561,2.915934134,16.29186648,,,1200321,3021637,16.29186648,2.320001561,0.05,0.95,0.02,0.01,,,0.1328455773,0.04242,0.02767439641,0.17435,-26.37966239,8.600621204e-08,1545.518962,3.118557454e-133,lognormal, +google_2011,memory_p99,"quantile(0.99, canonical_memory_usage)",values,,instant,,300,,254393256,,,,91239,113206,136461,2005,1.702567801,4.302722325,17.21791254,,,91239,136461,17.21791254,1.702567801,0.05,0.95,0.02,0.01,,,0.110901973,0.1195,0.1178449732,0.04182,-129.8619128,2.92376938e-16,-130.6899204,1.249790439e-13,light, +google_2011,memory_p99,"quantile_over_time(0.99, canonical_memory_usage[5m])",values,,5m,300,300,,288357812,,,,94938,120433,225145,2005,1.644191691,2.215617498,17.45458006,,,94938,225145,17.45458006,1.644191691,0.05,0.95,0.02,0.01,,,0.1503744105,0.0242,0.128213286,0.2863,-3016.902833,0,-2853.875617,0,light, +google_2011,memory_p99,"quantile_over_time(0.99, canonical_memory_usage[1h])",values,,1h,3600,300,,288357812,,,,1188516,1445212.5,2446753,1994,1.732888435,2.215617498,16.09107673,,,1188516,2446753,16.09107673,1.732888435,0.05,0.95,0.02,0.01,,,0.1503744105,0.0242,0.128213286,0.2863,-3016.902833,0,-2853.875617,0,light, diff --git a/asap-tools/dataset-analysis/tests/test_fit_skew.py b/asap-tools/dataset-analysis/tests/test_fit_skew.py new file mode 100644 index 00000000..ee6cd898 --- /dev/null +++ b/asap-tools/dataset-analysis/tests/test_fit_skew.py @@ -0,0 +1,618 @@ +"""Tests for the skew fits, window bounds, config validation and readers.""" + +import copy +import io +import tarfile +import tempfile +import unittest +from multiprocessing.pool import ThreadPool +from pathlib import Path + +import numpy as np +import pandas as pd + +import fit_skew + +ZIPF_K = 1000 +ZIPF_SAMPLES = 2_000_000 +ZIPF_TOLERANCE = 0.05 +PARETO_SAMPLES = 20_000 +PARETO_REL_TOLERANCE = 0.05 + + +def zipf_counts(theta: float, rng: np.random.Generator) -> np.ndarray: + ranks = np.arange(1, ZIPF_K + 1) + probs = ranks**-theta / np.sum(ranks**-theta) + return rng.multinomial(ZIPF_SAMPLES, probs) + + +class ZipfFitTest(unittest.TestCase): + def test_recovers_theta(self): + rng = np.random.default_rng(1) + for theta in (0.8, 1.2): + with self.subTest(theta=theta): + counts = zipf_counts(theta, rng) + # Rank order is recovered by sorting; key order must not matter. + rng.shuffle(counts) + self.assertAlmostEqual( + fit_skew.zipf_mle(counts), theta, delta=ZIPF_TOLERANCE + ) + + def test_uniform_is_zero(self): + self.assertAlmostEqual(fit_skew.zipf_mle(np.full(100, 50.0)), 0.0, places=3) + + def test_zero_weights_ignored(self): + counts = zipf_counts(1.0, np.random.default_rng(2)).astype(float) + padded = np.concatenate([counts, np.zeros(500)]) + self.assertAlmostEqual( + fit_skew.zipf_mle(padded), fit_skew.zipf_mle(counts), places=6 + ) + + def test_fewer_than_two_keys_is_nan(self): + self.assertTrue(np.isnan(fit_skew.zipf_mle([10.0]))) + self.assertTrue(np.isnan(fit_skew.zipf_mle([10.0, 0.0]))) + self.assertTrue(np.isnan(fit_skew.loglog_slope([]))) + + +class PowerLawFitTest(unittest.TestCase): + def test_recovers_pareto_alpha(self): + rng = np.random.default_rng(3) + for shape in (1.0, 2.0, 4.0): + alpha = shape + 1.0 # pdf exponent of a Pareto with this shape + with self.subTest(alpha=alpha): + x = rng.pareto(shape, PARETO_SAMPLES) + 1.0 + fit = fit_skew.fit_power_law(x, compare=False) + self.assertAlmostEqual( + fit["alpha"], alpha, delta=PARETO_REL_TOLERANCE * alpha + ) + tail = np.sum(x >= fit["xmin"]) + self.assertGreaterEqual(tail, fit_skew.MIN_TAIL_SAMPLES) + + def test_tail_class_synthetic(self): + exponential = np.random.default_rng(11).exponential(1.0, 20_000) + self.assertEqual( + fit_skew.fit_power_law(exponential, compare=True)["tail_class"], + fit_skew.TAIL_LIGHT, + ) + # The KS-chosen tail of a lognormal(sigma=1) is its top few percent, + # where it decays too fast for the power law to beat an exponential, + # so it is classed light; only heavier lognormal tails reach + # lognormal or heavy_inconclusive. + lognormal = np.random.default_rng(12).lognormal(0.0, 1.0, 20_000) + fit = fit_skew.fit_power_law(lognormal, compare=True) + self.assertEqual(fit["tail_class"], fit_skew.TAIL_LIGHT) + self.assertTrue(np.isfinite(fit["alpha"])) + # A lognormal with large sigma mimics a power-law tail, so an exact + # Pareto(alpha=2) beats the exponential but not the lognormal. + pareto = np.random.default_rng(13).pareto(1.0, 20_000) + 1.0 + fit = fit_skew.fit_power_law(pareto, compare=True) + self.assertIn( + fit["tail_class"], + (fit_skew.TAIL_POWER_LAW, fit_skew.TAIL_HEAVY_INCONCLUSIVE), + ) + + def test_tail_class_rule(self): + def classify(lognormal, exponential): + return fit_skew.tail_class( + {"lognormal": lognormal, "exponential": exponential} + ) + + beats_exp = (5.0, 0.01) + self.assertEqual(classify((2.0, 0.01), (5.0, 0.2)), fit_skew.TAIL_LIGHT) + self.assertEqual(classify((2.0, 0.01), (-5.0, 0.01)), fit_skew.TAIL_LIGHT) + self.assertEqual(classify((2.0, 0.01), beats_exp), fit_skew.TAIL_POWER_LAW) + self.assertEqual(classify((-2.0, 0.01), beats_exp), fit_skew.TAIL_LOGNORMAL) + self.assertEqual( + classify((-2.0, 0.5), beats_exp), fit_skew.TAIL_HEAVY_INCONCLUSIVE + ) + self.assertEqual( + classify((2.0, 0.5), beats_exp), fit_skew.TAIL_HEAVY_INCONCLUSIVE + ) + + def test_xmin_grid(self): + x = np.sort(np.random.default_rng(9).pareto(1.5, PARETO_SAMPLES) + 1.0) + grid = fit_skew.xmin_candidates(x) + self.assertLessEqual(len(grid), fit_skew.XMIN_GRID_SIZE) + self.assertGreaterEqual(grid[0], np.quantile(x, 0.5)) + self.assertGreaterEqual(np.sum(x >= grid[-1]), fit_skew.MIN_TAIL_SAMPLES) + fit = fit_skew.fit_power_law(x, compare=False) + self.assertIn(fit["xmin"], grid) + + def test_xmin_grid_small_sample(self): + rng = np.random.default_rng(10) + # 300 values: p99.9 leaves one value, so the grid stops at 100 tail values. + x = np.sort(rng.pareto(1.5, 300) + 1.0) + grid = fit_skew.xmin_candidates(x) + self.assertGreaterEqual(np.sum(x >= grid[-1]), fit_skew.MIN_TAIL_SAMPLES) + # 150 values: 100 tail values would reach below the median, so no fit. + x = np.sort(rng.pareto(1.5, 150) + 1.0) + self.assertEqual(len(fit_skew.xmin_candidates(x)), 0) + self.assertTrue(np.isnan(fit_skew.fit_power_law(x, compare=True)["alpha"])) + + def test_too_few_samples_is_nan(self): + fit = fit_skew.fit_power_law( + np.arange(1.0, fit_skew.MIN_TAIL_SAMPLES), compare=True + ) + self.assertTrue(np.isnan(fit["alpha"])) + self.assertNotIn("tail_class", fit) + + +def key_step_frames(thetas, rng): + """Per-step key count frames, one Zipf distribution per step.""" + return [ + pd.DataFrame( + { + fit_skew.STEP_COL: step, + fit_skew.KEY_COL: np.arange(1, ZIPF_K + 1, dtype=np.uint64), + fit_skew.COUNT_COL: zipf_counts(theta, rng), + } + ) + for step, theta in enumerate(thetas) + ] + + +COUNT_QUERY = { + "id": "q", + "promql": "count by (k) (x)", + "promql_range": "sum by (k) (count_over_time(x[{range}]))", + "group_by": ["k"], + "weights": ["count"], +} +VALUE_QUERY = { + "id": "v", + "promql": "quantile(0.99, x)", + "promql_range": "quantile_over_time(0.99, x[{range}])", + "value": "x", +} + + +def summarize_counts(frames, token, span, min_keys=2): + agg = fit_skew.merge_key_parts(frames) + return fit_skew.summarize_keys( + "test", COUNT_QUERY, token, 60, agg, span, 1, min_keys, None + ) + + +class WindowBoundsTest(unittest.TestCase): + def test_min_max_count(self): + self.assertEqual(fit_skew.window_bounds([1.1, 0.7, 1.4], 1.0), (0.7, 1.4, 3)) + + def test_pooled_extends_bounds(self): + # The pooled fit need not lie between the window fits; it widens them. + self.assertEqual(fit_skew.window_bounds([1.1, 1.2], 1.5), (1.1, 1.5, 2)) + self.assertEqual(fit_skew.window_bounds([1.1, 1.2], 0.9), (0.9, 1.2, 2)) + + def test_nan_windows_skipped(self): + self.assertEqual( + fit_skew.window_bounds([np.nan, 0.9, np.nan, 1.2], 1.0), (0.9, 1.2, 2) + ) + + def test_no_windows_uses_pooled(self): + for estimates in ([], [np.nan]): + self.assertEqual(fit_skew.window_bounds(estimates, 1.3), (1.3, 1.3, 0)) + + def test_nothing_finite(self): + lower, upper, n = fit_skew.window_bounds([np.nan], np.nan) + self.assertTrue(np.isnan(lower) and np.isnan(upper)) + self.assertEqual(n, 0) + + +class RangeEvaluationTest(unittest.TestCase): + def test_skips_small_evaluations(self): + frames = key_step_frames((0.8, 1.2), np.random.default_rng(5)) + # Step 2 has too few keys to be fitted. + frames.append( + pd.DataFrame( + { + fit_skew.STEP_COL: 2, + fit_skew.KEY_COL: np.array([1, 2], dtype=np.uint64), + fit_skew.COUNT_COL: [1e6, 1], + } + ) + ) + (row,) = summarize_counts(frames, "1m", (0, 2), min_keys=10) + self.assertEqual(row["n_evals"], 2) + self.assertEqual(row["range_s"], 60) + self.assertEqual(row["promql"], "sum by (k) (count_over_time(x[1m]))") + self.assertAlmostEqual(row["lower"], 0.8, delta=ZIPF_TOLERANCE) + self.assertAlmostEqual(row["upper"], 1.2, delta=ZIPF_TOLERANCE) + self.assertTrue(row["lower"] <= row["mle"] <= row["upper"]) + self.assertEqual(row["K_total"], ZIPF_K) + + def test_sliding_ranges(self): + # A 2-step range is evaluated at every step once it is full (3 times + # over 4 steps), not over disjoint windows. + frames = key_step_frames((0.8, 1.2, 0.8, 1.2), np.random.default_rng(6)) + (one,) = summarize_counts(frames, "1m", (0, 3)) + (two,) = summarize_counts(frames, "2m", (0, 3)) + self.assertEqual((one["n_evals"], two["n_evals"]), (4, 3)) + self.assertEqual(one["mle"], two["mle"]) + # Merging a 0.8 and a 1.2 step gives something in between. + self.assertGreater(two["lower"], one["lower"]) + self.assertLess(two["upper"], one["upper"]) + + def test_per_evaluation_stats(self): + # Steps 0..3 with 3, 1, 2, 2 keys; key 1 appears in all. + frame = pd.DataFrame( + { + fit_skew.STEP_COL: [0, 0, 0, 1, 2, 2, 3, 3], + fit_skew.KEY_COL: np.array([1, 2, 3, 1, 1, 2, 1, 4], dtype=np.uint64), + fit_skew.COUNT_COL: [5, 1, 1, 7, 2, 2, 1, 9], + } + ) + (one,) = summarize_counts([frame], "1m", (0, 3)) + (two,) = summarize_counts([frame], "2m", (0, 3)) + self.assertEqual((one["K_total"], one["rows_total"]), (4, 28)) + self.assertEqual( + (one["K_win_min"], one["K_win_median"], one["K_win_max"]), (1, 2, 3) + ) + self.assertEqual( + (one["rows_win_min"], one["rows_win_median"], one["rows_win_max"]), + (4, 7, 10), + ) + # Ranges {0,1}, {1,2}, {2,3}: keys {1,2,3}, {1,2}, {1,2,4}; rows 14, 11, 14. + self.assertEqual((two["K_win_min"], two["K_win_max"]), (2, 3)) + self.assertEqual((two["rows_win_min"], two["rows_win_max"]), (11, 14)) + self.assertLessEqual(set(one), set(fit_skew.SUMMARY_COLUMNS)) + + def test_empty_steps_inside_span(self): + # Steps 1 and 2 have no rows; a 1-step range there is not evaluated. + frame = pd.DataFrame( + { + fit_skew.STEP_COL: [0, 0, 3, 3], + fit_skew.KEY_COL: np.array([1, 2, 1, 2], dtype=np.uint64), + fit_skew.COUNT_COL: [3, 1, 3, 1], + } + ) + (row,) = summarize_counts([frame], "1m", (0, 3)) + self.assertEqual(row["n_evals"], 2) + # Empty evaluations are gaps, not evaluations that saw zero rows. + self.assertEqual((row["rows_win_min"], row["K_win_min"]), (4, 2)) + + def test_merge_samples_is_uniform(self): + # Two parts, 10x apart in size, merged into a sample capped at + # MAX_FIT_SAMPLES: each part's share follows its size. + big = fit_skew.value_sample(np.zeros(10 * fit_skew.MAX_FIT_SAMPLES)) + small = fit_skew.value_sample(np.ones(fit_skew.MAX_FIT_SAMPLES)) + total, merged = fit_skew.merge_samples([big, small]) + self.assertEqual(total, 11 * fit_skew.MAX_FIT_SAMPLES) + self.assertEqual(len(merged), fit_skew.MAX_FIT_SAMPLES) + self.assertAlmostEqual(merged.mean(), 1 / 11, delta=0.01) + + def test_summarize_values_ranges(self): + rng = np.random.default_rng(7) + acc = { + "n_finite": 4000, + "steps": { + w: fit_skew.value_sample(rng.pareto(1.5, 1000) + 1.0) for w in range(4) + }, + } + rows = [] + # One thread: redirect_stdout in fit_power_law is process-wide. + with ThreadPool(1) as pool: + for token in ("1m", "4m"): + rows.append( + fit_skew.summarize_values( + "test", VALUE_QUERY, token, 60, acc, (0, 3), pool, 1, None + ) + ) + self.assertEqual([r["n_evals"] for r in rows], [4, 1]) + self.assertEqual([r["rows_win_median"] for r in rows], [1000, 4000]) + self.assertNotIn("K_win_median", rows[0]) + self.assertLessEqual(set(rows[0]), set(fit_skew.SUMMARY_COLUMNS)) + for row in rows: + self.assertTrue(row["lower"] <= row["mle"] <= row["upper"]) + + +def latest_rows(samples): + """(series, time_s) pairs -> latest_part-style rows with step = ceil(t / 60).""" + series, times = zip(*samples) + times = np.array(times, dtype=float) + return pd.DataFrame( + { + fit_skew.SERIES_COL: np.array(series, dtype=np.uint64), + fit_skew.STEP_COL: np.ceil(times / 60).astype(np.int64), + fit_skew.TIME_COL: times, + "_key:k": np.array(series, dtype=np.uint64), + } + ) + + +INSTANT_QUERY = {**COUNT_QUERY, "kind": "keys", "range": ["instant"]} +INSTANT_KEY = fit_skew.query_key(INSTANT_QUERY) + + +def instant_counts(parts): + """{(step, key): count} from instant_parts outputs.""" + agg = fit_skew.merge_key_parts([p[INSTANT_KEY] for p in parts]) + return { + (int(s), int(k)): int(c) + for s, k, c in agg[ + [fit_skew.STEP_COL, fit_skew.KEY_COL, fit_skew.COUNT_COL] + ].itertuples(index=False) + } + + +def split_files(files, lookback): + """instant parts computed per file then merged via resolve_boundaries.""" + parts, boundary = [], [] + for samples in files: + inner, edge = fit_skew.split_instant( + latest_rows(samples), [INSTANT_QUERY], lookback + ) + parts.append(inner) + boundary.append(edge) + parts.append(fit_skew.resolve_boundaries(boundary, [INSTANT_QUERY], lookback)) + return instant_counts(parts) + + +class InstantEvaluationTest(unittest.TestCase): + def test_lookback_carries_last_sample(self): + # Series 1 is sampled at steps 1 and 2 then stops: with a 3-step + # lookback it is still seen at steps 3 and 4, not at 5. + counts = split_files([[(1, 60), (1, 120)]], lookback=3) + self.assertEqual(sorted(counts), [(1, 1), (2, 1), (3, 1), (4, 1)]) + + def test_gap_shorter_than_lookback(self): + # Samples at steps 1 and 3: step 2 still sees the step-1 sample once. + counts = split_files([[(1, 60), (1, 180)]], lookback=5) + self.assertEqual(counts[(2, 1)], 1) + self.assertEqual(counts[(3, 1)], 1) + + def test_latest_sample_per_step(self): + # Two samples in step 1 count once. + counts = split_files([[(1, 30), (1, 55)]], lookback=1) + self.assertEqual(counts, {(1, 1): 1}) + + def test_split_across_files_matches_one_file(self): + rng = np.random.default_rng(8) + samples = [ + (int(series), float(t)) + for series in range(1, 6) + for t in np.sort(rng.choice(np.arange(1, 1200), 15, replace=False)) + ] + samples.sort(key=lambda st: st[1]) + whole = split_files([samples], lookback=5) + # Consecutive time chunks, cut inside a step so step 10 spans both files. + cut = [s for s in samples if s[1] <= 570], [s for s in samples if s[1] > 570] + self.assertEqual(split_files(list(cut), lookback=5), whole) + + def test_interleaved_files_fail(self): + files = [[(1, 60), (1, 300)], [(1, 180)]] + with self.assertRaises(ValueError): + split_files(files, lookback=5) + + def test_lookback_steps(self): + self.assertEqual(fit_skew.lookback_steps({"step_s": 60}), 5) + # A sampling period longer than 5 minutes is one step. + self.assertEqual(fit_skew.lookback_steps({"step_s": 600}), 1) + + +def boom_inputs(classes): + """Fake per-variate full fits (alpha = index + 2) with the given classes.""" + shifted = np.ones((len(classes), 10)) + jobs = [(np.ones(10), True) for _ in classes] + fits = [ + { + "alpha": v + 2.0, + "xmin": 1.0, + "ks_d": 0.01, + "tail_frac": 0.1, + "R_lognormal": -1.0, + "p_lognormal": 0.5, + "R_exponential": -2.0 if cls == fit_skew.TAIL_LIGHT else 4.0, + "p_exponential": 0.01, + "tail_class": cls, + } + for v, cls in enumerate(classes) + ] + return shifted, jobs, list(range(len(classes))), fits + + +LIGHT = fit_skew.TAIL_LIGHT +HEAVY = fit_skew.TAIL_HEAVY_INCONCLUSIVE + + +class BoomSummaryTest(unittest.TestCase): + def summarize(self, classes): + q = {"id": "t", "promql": "quantile(0.99, target)"} + return fit_skew.summarize_boom_series( + "boom", q, "s", *boom_inputs(classes), None + ) + + def test_alpha_over_non_light_variates(self): + row = self.summarize([HEAVY, LIGHT, HEAVY, fit_skew.TAIL_POWER_LAW]) + self.assertEqual(row["tail_class"], HEAVY) + self.assertEqual(row["ok_frac"], 0.75) + # Non-light variates 0, 2, 3 have alpha 2, 4, 5. + self.assertEqual(row["mle"], 4.0) + self.assertTrue(row["lower"] <= row["mle"] <= row["upper"]) + self.assertEqual( + (row["worst_alpha_memory"], row["worst_alpha_rank"]), + (row["lower"], row["upper"]), + ) + # Every summary field must be a CSV column, or it is silently dropped. + self.assertLessEqual(set(row), set(fit_skew.SUMMARY_COLUMNS)) + # R/p come from the same non-light variates. + self.assertEqual(row["R_exponential"], 4.0) + + def test_majority_light_still_reports_heavy_alpha(self): + row = self.summarize([HEAVY, LIGHT, LIGHT, LIGHT]) + self.assertEqual(row["tail_class"], LIGHT) + self.assertEqual(row["ok_frac"], 0.25) + self.assertEqual(row["mle"], 2.0) + self.assertEqual(row["R_exponential"], 4.0) + + def test_all_light_has_no_alpha(self): + row = self.summarize([LIGHT, LIGHT]) + self.assertEqual(row["tail_class"], LIGHT) + self.assertEqual(row["ok_frac"], 0.0) + self.assertNotIn("mle", row) + self.assertTrue(np.isnan(row["R_exponential"])) + + def test_no_compared_variates(self): + q = {"id": "t", "promql": "quantile(0.99, target)"} + shifted, jobs, owners, _ = boom_inputs([LIGHT]) + fits = [{"alpha": np.nan, "xmin": np.nan, "ks_d": np.nan}] + row = fit_skew.summarize_boom_series( + "boom", q, "s", shifted, jobs, owners, fits, None + ) + self.assertEqual(row["tail_class"], "") + self.assertTrue(np.isnan(row["ok_frac"])) + + +VALID_CONFIG = { + "dataset": "d", + "tables": { + "t": { + "label_columns": ["a", "b"], + "value_columns": ["v"], + "step_s": 60, + "series_key": ["a", "b"], + }, + }, + "queries": [ + { + "id": "keys", + "table": "t", + "kind": "keys", + "group_by": ["a"], + "value": "v", + "weights": ["count", "value"], + "promql": "count by (a) (v)", + "promql_range": "sum by (a) (count_over_time(v[{range}]))", + "range": ["instant", "5m"], + }, + { + "id": "vals", + "table": "t", + "kind": "values", + "group_by": [], + "value": "v", + "promql": "quantile(0.99, v)", + "range": ["instant"], + }, + ], +} + + +class ValidateConfigTest(unittest.TestCase): + def test_valid(self): + fit_skew.validate_config(VALID_CONFIG) + + def test_invalid(self): + cases = { + "unknown table": ("table", "missing"), + "unknown label": ("group_by", ["c"]), + "unknown value": ("value", "w"), + "bad kind": ("kind", "other"), + "bad weight": ("weights", ["sum"]), + "no weights": ("weights", []), + "no group_by": ("group_by", []), + } + for name, (field, value) in cases.items(): + with self.subTest(name): + cfg = copy.deepcopy(VALID_CONFIG) + cfg["queries"][0][field] = value + with self.assertRaises(ValueError): + fit_skew.validate_config(cfg) + + def test_bad_ranges(self): + cases = { + "no range": [], + "bad duration": ["5x"], + "not a multiple of step_s": ["90s"], + } + for name, ranges in cases.items(): + with self.subTest(name): + cfg = copy.deepcopy(VALID_CONFIG) + cfg["queries"][0]["range"] = ranges + with self.assertRaises(ValueError): + fit_skew.validate_config(cfg) + + def test_range_needs_promql_range(self): + cfg = copy.deepcopy(VALID_CONFIG) + cfg["queries"][1]["range"] = ["5m"] + with self.assertRaises(ValueError): + fit_skew.validate_config(cfg) + + def test_instant_needs_series_key(self): + cfg = copy.deepcopy(VALID_CONFIG) + del cfg["tables"]["t"]["series_key"] + with self.assertRaises(ValueError): + fit_skew.validate_config(cfg) + + def test_bad_step(self): + for step in (None, 0, 1.5): + with self.subTest(step=step): + cfg = copy.deepcopy(VALID_CONFIG) + cfg["tables"]["t"]["step_s"] = step + with self.assertRaises(ValueError): + fit_skew.validate_config(cfg) + + def test_value_weight_needs_value(self): + cfg = copy.deepcopy(VALID_CONFIG) + del cfg["queries"][0]["value"] + with self.assertRaises(ValueError): + fit_skew.validate_config(cfg) + + def test_values_query_needs_value(self): + cfg = copy.deepcopy(VALID_CONFIG) + del cfg["queries"][1]["value"] + with self.assertRaises(ValueError): + fit_skew.validate_config(cfg) + + def test_missing_files_fail(self): + with tempfile.TemporaryDirectory() as root: + with self.assertRaises(FileNotFoundError): + fit_skew.expand_files(root, ["nothing-*.csv"]) + + +class ReaderTest(unittest.TestCase): + def test_alibaba_tar_streaming(self): + # CallGraph shards contain rows with extra fields and rt == "None". + csv = ( + "timestamp,um,dm,rt\n" + "1000,MS_1,MS_2,3.0\n" + "2000,MS_1,MS_3,None\n" + "3000,MS_1,MS_2,4.0,extra\n" + "61000,MS_2,MS_3,5.0\n" + ).encode() + with tempfile.TemporaryDirectory() as root: + path = Path(root) / "t.tar.gz" + with tarfile.open(path, "w:gz") as archive: + info = tarfile.TarInfo("t.csv") + info.size = len(csv) + archive.addfile(info, io.BytesIO(csv)) + table = {"format": "alibaba_tar"} + bad: list = [] + frames = list( + fit_skew.table_frames( + root, table, str(path), ["timestamp", "um", "rt"], {"rt"}, bad + ) + ) + df = frames[0] + self.assertEqual(len(bad), 1) + self.assertEqual(list(df["um"].astype(str)), ["MS_1", "MS_1", "MS_2"]) + self.assertTrue(np.isnan(df["rt"][1])) + + def test_google_column_names(self): + schema = ( + "file pattern,field number,content,format,mandatory\n" + "task_usage/part-?????-of-?????.csv.gz,2,CPU rate,FLOAT,NO\n" + "task_usage/part-?????-of-?????.csv.gz,1,start time,INTEGER,YES\n" + "job_events/part-?????-of-?????.csv.gz,1,time,INTEGER,YES\n" + ) + with tempfile.TemporaryDirectory() as root: + path = Path(root) / "schema.csv" + path.write_text(schema) + self.assertEqual( + fit_skew.google_column_names(str(path), "task_usage"), + ["start_time", "cpu_rate"], + ) + with self.assertRaises(ValueError): + fit_skew.google_column_names(str(path), "machine_events") + + +if __name__ == "__main__": + unittest.main()