From a63140a685958dedaf5ecaa3ef2f8a2fec03769c Mon Sep 17 00:00:00 2001 From: jjscripts Date: Mon, 21 Sep 2026 15:22:39 +0800 Subject: [PATCH] Return NaN for non-finite input instead of a made-up entropy One NaN or inf in the input used to produce a plausible-looking number: permutation.compute counted NaN as an ordinal pattern (argsort ranks it last), spectral.compute returned -0.0 because a NaN spectrum filters to nothing, sample.compute returned a large finite value, and shannon.compute raised. rolling inherited this, so one gap silently corrupted every window that contained it. Every measure's kernel is now wrapped by _core.nan_on_non_finite, so compute, normalized, rolling and delta return NaN for exactly the windows that hold a non-finite value and leave the rest of the series unchanged. multiscale, transfer, spectral.normalized and divergence.kl/js get the same guard. Zero entropy is also returned as 0.0, never -0.0. Found while building the training early-warning study (github.com/Par-python/training-early-warning). Co-Authored-By: Claude Opus 5 --- CHANGELOG.md | 10 +++ README.md | 5 ++ entroscope/_core.py | 19 ++++++ entroscope/approximate.py | 1 + entroscope/differential.py | 1 + entroscope/divergence.py | 9 +++ entroscope/multiscale.py | 3 +- entroscope/permutation.py | 3 +- entroscope/sample.py | 1 + entroscope/shannon.py | 5 +- entroscope/spectral.py | 5 +- entroscope/transfer.py | 8 ++- tests/test_missing_values.py | 122 +++++++++++++++++++++++++++++++++++ 13 files changed, 184 insertions(+), 8 deletions(-) create mode 100644 tests/test_missing_values.py diff --git a/CHANGELOG.md b/CHANGELOG.md index e26f9b0..85fd256 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,16 @@ ## Unreleased +- **Fixed (changes results):** missing and infinite values no longer produce + plausible-looking numbers. Previously one NaN made `permutation.compute` + return a finite value, `spectral.compute` return `-0.0`, `sample.compute` + return a large finite value, and `shannon.compute` raise. Now every measure + (`compute`, `normalized`, `multiscale`, `transfer`, `divergence.kl`/`js`) + returns NaN when its input holds NaN or ±inf, and `rolling`/`delta` are NaN + for exactly the windows that contain one. Reported while building the + training early-warning study. +- **Fixed:** zero entropy is returned as `0.0`, never `-0.0` (constant or + perfectly ordered input, a single region in `shannon.geographic`). - CI: move the remaining GitHub Actions off the deprecated Node 20 runtime (`upload-artifact` and `download-artifact` to v7, `upload-pages-artifact` and `deploy-pages` to v5). diff --git a/README.md b/README.md index aed03e5..09b59aa 100644 --- a/README.md +++ b/README.md @@ -83,6 +83,11 @@ Every measure exposes the same methods, so switching measures is a one-word chan Shannon additionally provides `geographic(df, col=...)` for spatial distributions (e.g. search interest by region). Multiscale provides `compute` and `plot`. +**Missing values.** Any NaN or ±inf in the input makes the result NaN, never a +made-up number. `rolling` and `delta` are NaN only for the windows that contain +the gap, so the rest of the series is unaffected. `rolling` is causal: the value +at position t uses positions t−window+1 through t. + The two-input measures take a pair of series. `transfer.compute(x, y)` (plus `rolling`, `delta`, `plot`) estimates how much `x`'s past tells you about `y`'s future; `divergence.kl(p, q)` and `divergence.js(p, q)` (plus `plot`) compare two diff --git a/entroscope/_core.py b/entroscope/_core.py index 1da6dca..5eb48ac 100644 --- a/entroscope/_core.py +++ b/entroscope/_core.py @@ -10,6 +10,8 @@ non-interactive backend should set ``MPLBACKEND=Agg`` in their environment. """ +import functools + import matplotlib.pyplot as plt import numpy as np import pandas as pd @@ -30,6 +32,23 @@ def _is_polars_series(x): return cls.__name__ == "Series" and cls.__module__.startswith("polars") +def nan_on_non_finite(kernel): + """Make `kernel` return NaN when its input contains NaN or +/-inf. + + Without this, kernels turn missing values into plausible-looking numbers (argsort + ranks NaN last, a NaN spectrum sums to NaN, ...). Applied to each measure's + `_kernel`, it also makes `rolling` NaN for exactly the windows that hold a gap. + """ + + @functools.wraps(kernel) + def wrapper(values, *args, **kwargs): + if not np.isfinite(np.asarray(values, dtype=float)).all(): + return float("nan") + return kernel(values, *args, **kwargs) + + return wrapper + + def as_array(x): """Coerce input to a 1-D float ndarray, returning (array, index_or_None). diff --git a/entroscope/approximate.py b/entroscope/approximate.py index 9527bc6..a71d014 100644 --- a/entroscope/approximate.py +++ b/entroscope/approximate.py @@ -16,6 +16,7 @@ def _phi(values, m, tol): return float(np.mean(np.log(counts))) +@_core.nan_on_non_finite def _kernel(values, m=2, r=0.2): """Approximate entropy: phi(m) - phi(m+1).""" if r <= 0: diff --git a/entroscope/differential.py b/entroscope/differential.py index 22065f8..61e873e 100644 --- a/entroscope/differential.py +++ b/entroscope/differential.py @@ -6,6 +6,7 @@ from . import _core +@_core.nan_on_non_finite def _kernel(values, dist="normal"): """Differential entropy (nats). dist in {'normal', 'kde'}.""" values = np.asarray(values, dtype=float) diff --git a/entroscope/divergence.py b/entroscope/divergence.py index 3c34246..8e1c442 100644 --- a/entroscope/divergence.py +++ b/entroscope/divergence.py @@ -40,6 +40,11 @@ def _binned_probs(p, q, bins): _LN2 = np.log(2.0) +def _all_finite(*arrays): + """False if any sample holds NaN or +/-inf; kl/js then return NaN.""" + return all(np.isfinite(a).all() for a in arrays) + + def _kl_bits(pp, qq): """KL(pp || qq) in bits, with epsilon smoothing so the result stays finite.""" pp = pp + _EPS @@ -57,6 +62,8 @@ def kl(p, q, bins=10): """ pa, _ = _core.as_array(p) qa, _ = _core.as_array(q) + if not _all_finite(pa, qa): + return float("nan") pp, qq = _binned_probs(pa, qa, bins) return _kl_bits(pp, qq) @@ -68,6 +75,8 @@ def js(p, q, bins=10): """ pa, _ = _core.as_array(p) qa, _ = _core.as_array(q) + if not _all_finite(pa, qa): + return float("nan") pp, qq = _binned_probs(pa, qa, bins) m = 0.5 * (pp + qq) return 0.5 * _kl_bits(pp, m) + 0.5 * _kl_bits(qq, m) diff --git a/entroscope/multiscale.py b/entroscope/multiscale.py index 17f6d70..a6e4975 100644 --- a/entroscope/multiscale.py +++ b/entroscope/multiscale.py @@ -30,6 +30,7 @@ def compute(series, scales=range(1, 10), method="sample", m=2, r=0.2): if m < 1: raise ValueError("m must be >= 1") arr, _ = _core.as_array(series) + finite = bool(np.isfinite(arr).all()) tol = r * np.std(arr) result = {} for scale in scales: @@ -39,7 +40,7 @@ def compute(series, scales=range(1, 10), method="sample", m=2, r=0.2): grained = _coarse_grain(arr, scale) if len(grained) <= m + 1: # sample entropy needs n > m+1 continue - result[int(scale)] = sample._sampen(grained, m, tol) + result[int(scale)] = sample._sampen(grained, m, tol) if finite else float("nan") return result diff --git a/entroscope/permutation.py b/entroscope/permutation.py index ad8015a..6dfa318 100644 --- a/entroscope/permutation.py +++ b/entroscope/permutation.py @@ -9,6 +9,7 @@ from .utils import normalize +@_core.nan_on_non_finite def _kernel(values, order=3, delay=1): """Permutation entropy (base 2) of ordinal patterns of length `order`.""" if order < 2: @@ -28,7 +29,7 @@ def _kernel(values, order=3, delay=1): counts[perm_index[pattern]] += 1 total = counts.sum() p = counts[counts > 0] / total - return float(-np.sum(p * np.log2(p))) + return float(-np.sum(p * np.log2(p))) + 0.0 # + 0.0 turns -0.0 into 0.0 def _check_window(window, order, delay): diff --git a/entroscope/sample.py b/entroscope/sample.py index bf66184..dba7af0 100644 --- a/entroscope/sample.py +++ b/entroscope/sample.py @@ -32,6 +32,7 @@ def _sampen(values, m, tol): return float(-np.log(a / b)) +@_core.nan_on_non_finite def _kernel(values, m=2, r=0.2): """Sample entropy: -ln(A/B) of length-(m+1) vs length-m matches.""" if r <= 0: diff --git a/entroscope/shannon.py b/entroscope/shannon.py index 5e4b2dd..dc4a7e7 100644 --- a/entroscope/shannon.py +++ b/entroscope/shannon.py @@ -6,6 +6,7 @@ from .utils import normalize +@_core.nan_on_non_finite def _kernel(values, bins=10): """Shannon entropy (base 2) of `values` histogrammed into `bins`.""" if bins <= 0: @@ -16,7 +17,7 @@ def _kernel(values, bins=10): if total == 0: return 0.0 p = counts[counts > 0] / total - return float(-np.sum(p * np.log2(p))) + return float(-np.sum(p * np.log2(p))) + 0.0 # + 0.0 turns -0.0 into 0.0 def compute(series, bins=10): @@ -44,7 +45,7 @@ def geographic(region_df, col="interest"): if total == 0: return 0.0 p = values[values > 0] / total - return float(-np.sum(p * np.log2(p))) + return float(-np.sum(p * np.log2(p))) + 0.0 # + 0.0 turns -0.0 into 0.0 def plot(series, window=20, bins=10, title=None): diff --git a/entroscope/spectral.py b/entroscope/spectral.py index f244792..d960950 100644 --- a/entroscope/spectral.py +++ b/entroscope/spectral.py @@ -15,6 +15,7 @@ def _psd(values, sf): return freqs, psd +@_core.nan_on_non_finite def _kernel(values, sf=1.0): """Spectral entropy (base 2) of the normalized power spectrum.""" _, psd = _psd(values, sf) @@ -22,7 +23,7 @@ def _kernel(values, sf=1.0): if total == 0: return 0.0 p = psd[psd > 0] / total - return float(-np.sum(p * np.log2(p))) + return float(-np.sum(p * np.log2(p))) + 0.0 # + 0.0 turns -0.0 into 0.0 def compute(series, sf=1.0): @@ -41,6 +42,8 @@ def delta(series, window=50, sf=1.0): def normalized(series, sf=1.0): """Entropy scaled to [0, 1] by log2(number of frequency bins).""" arr, _ = _core.as_array(series) + if not np.isfinite(arr).all(): + return float("nan") _, psd_vals = _psd(arr, sf) n_bins = int(np.count_nonzero(psd_vals > 0)) if n_bins <= 1: diff --git a/entroscope/transfer.py b/entroscope/transfer.py index 1d3b97f..bfa0956 100644 --- a/entroscope/transfer.py +++ b/entroscope/transfer.py @@ -36,13 +36,15 @@ def _guard(n_samples, k, method): def _estimate(xa, ya, k, lag, method, bins): + if method not in ("ksg", "binned"): + raise ValueError(f"unknown method {method!r}; use 'ksg' or 'binned'") + if not (np.isfinite(xa).all() and np.isfinite(ya).all()): + return float("nan") yf, yp, xp = est.embed(xa, ya, lag=lag) _guard(len(yf), k, method) if method == "ksg": return est.te_ksg(yf, yp, xp, k=k) - if method == "binned": - return est.te_binned(yf, yp, xp, bins=bins) - raise ValueError(f"unknown method {method!r}; use 'ksg' or 'binned'") + return est.te_binned(yf, yp, xp, bins=bins) def compute(x, y, *, k=4, lag=1, method="ksg", bins=6): diff --git a/tests/test_missing_values.py b/tests/test_missing_values.py new file mode 100644 index 0000000..dac4540 --- /dev/null +++ b/tests/test_missing_values.py @@ -0,0 +1,122 @@ +"""Non-finite input (NaN, +/-inf) yields NaN, consistently, instead of a plausible number. + +Rolling results are NaN only for the windows that contain a non-finite value. +""" + +import math + +import numpy as np +import pandas as pd +import pytest + +from entroscope import ( + approximate, + differential, + divergence, + multiscale, + permutation, + sample, + shannon, + spectral, + transfer, +) + +CLEAN = np.random.RandomState(0).randn(120) +BAD_AT = 50 +WINDOW = 20 + + +def _with(value): + x = CLEAN.copy() + x[BAD_AT] = value + return x + + +SINGLE = [ + pytest.param(shannon.compute, id="shannon"), + pytest.param(permutation.compute, id="permutation"), + pytest.param(spectral.compute, id="spectral"), + pytest.param(sample.compute, id="sample"), + pytest.param(approximate.compute, id="approximate"), + pytest.param(lambda x: differential.compute(x, dist="normal"), id="differential-normal"), + pytest.param(lambda x: differential.compute(x, dist="kde"), id="differential-kde"), + pytest.param(shannon.normalized, id="shannon-normalized"), + pytest.param(permutation.normalized, id="permutation-normalized"), + pytest.param(spectral.normalized, id="spectral-normalized"), +] + +ROLLING = [shannon, permutation, spectral, sample, approximate, differential] + + +@pytest.mark.parametrize("bad", [np.nan, np.inf, -np.inf], ids=["nan", "inf", "-inf"]) +@pytest.mark.parametrize("fn", SINGLE) +def test_compute_with_non_finite_is_nan(fn, bad): + assert math.isnan(fn(_with(bad))) + + +@pytest.mark.parametrize("fn", SINGLE) +def test_compute_all_nan_is_nan(fn): + assert math.isnan(fn(np.full(64, np.nan))) + + +@pytest.mark.parametrize("measure", ROLLING, ids=lambda m: m.__name__.split(".")[-1]) +def test_rolling_nan_only_in_windows_containing_the_gap(measure): + clean = measure.rolling(pd.Series(CLEAN), window=WINDOW) + gappy = measure.rolling(pd.Series(_with(np.nan)), window=WINDOW) + # Window ending at t covers t-WINDOW+1..t, so it contains BAD_AT for these ends. + affected = np.arange(BAD_AT, BAD_AT + WINDOW) + assert gappy.iloc[affected].isna().all() + unaffected = np.setdiff1d(np.arange(WINDOW - 1, len(CLEAN)), affected) + pd.testing.assert_series_equal(gappy.iloc[unaffected], clean.iloc[unaffected]) + + +def test_delta_nan_around_the_gap(): + d = shannon.delta(_with(np.nan), window=WINDOW) + assert np.isnan(d[BAD_AT : BAD_AT + WINDOW + 1]).all() + assert np.isfinite(d[WINDOW:BAD_AT]).all() + + +def test_multiscale_non_finite_is_nan_at_every_scale(): + clean = multiscale.compute(CLEAN, scales=range(1, 5)) + gappy = multiscale.compute(_with(np.nan), scales=range(1, 5)) + assert gappy.keys() == clean.keys() + assert all(math.isnan(v) for v in gappy.values()) + + +@pytest.mark.parametrize("method", ["ksg", "binned"]) +@pytest.mark.parametrize("which", ["x", "y"]) +def test_transfer_non_finite_is_nan(method, which): + x, y = CLEAN, np.roll(CLEAN, 1) + if which == "x": + x = _with(np.nan) + else: + y = _with(np.nan) + assert math.isnan(transfer.compute(x, y, method=method)) + + +def test_transfer_rolling_nan_only_near_the_gap(): + x, y = _with(np.nan), np.roll(CLEAN, 1) + roll = transfer.rolling(x, y, window=60) + assert np.isnan(roll[BAD_AT : BAD_AT + 60]).all() + assert np.isfinite(roll[BAD_AT + 60 :]).all() + + +@pytest.mark.parametrize("fn", [divergence.kl, divergence.js], ids=["kl", "js"]) +@pytest.mark.parametrize("bad", [np.nan, np.inf]) +def test_divergence_non_finite_is_nan(fn, bad): + assert math.isnan(fn(_with(bad), CLEAN)) + assert math.isnan(fn(CLEAN, _with(bad))) + + +@pytest.mark.parametrize( + "value", + [ + pytest.param(lambda: shannon.compute(np.ones(10)), id="shannon-constant"), + pytest.param(lambda: permutation.compute(np.arange(10.0)), id="permutation-ordered"), + pytest.param(lambda: shannon.geographic({"interest": [5.0]}), id="geographic-one-region"), + ], +) +def test_zero_entropy_is_positive_zero(value): + result = value() + assert result == 0.0 + assert math.copysign(1.0, result) == 1.0, "got -0.0"