entroscope 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. {entroscope-0.2.0 → entroscope-0.3.0}/PKG-INFO +75 -11
  2. {entroscope-0.2.0 → entroscope-0.3.0}/README.md +61 -9
  3. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/__init__.py +12 -12
  4. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/_core.py +35 -8
  5. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/_transfer_estimators.py +1 -0
  6. entroscope-0.3.0/entroscope/features.py +96 -0
  7. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/multiscale.py +13 -3
  8. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/sample.py +22 -15
  9. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/spectral.py +2 -2
  10. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/transfer.py +4 -7
  11. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/utils/knn.py +8 -13
  12. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/utils/plot.py +4 -3
  13. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope.egg-info/PKG-INFO +75 -11
  14. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope.egg-info/SOURCES.txt +4 -0
  15. entroscope-0.3.0/entroscope.egg-info/requires.txt +25 -0
  16. {entroscope-0.2.0 → entroscope-0.3.0}/pyproject.toml +8 -3
  17. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_approximate.py +2 -1
  18. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_backend.py +1 -0
  19. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_consistency.py +2 -1
  20. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_core.py +1 -0
  21. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_differential.py +2 -1
  22. entroscope-0.3.0/tests/test_integrations.py +124 -0
  23. entroscope-0.3.0/tests/test_knn.py +32 -0
  24. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_multiscale.py +2 -1
  25. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_permutation.py +2 -1
  26. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_plot.py +2 -1
  27. entroscope-0.3.0/tests/test_reference.py +115 -0
  28. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_sample.py +2 -1
  29. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_shannon.py +2 -1
  30. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_spectral.py +2 -1
  31. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_transfer.py +1 -0
  32. entroscope-0.2.0/entroscope.egg-info/requires.txt +0 -9
  33. {entroscope-0.2.0 → entroscope-0.3.0}/LICENSE +0 -0
  34. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/approximate.py +0 -0
  35. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/differential.py +0 -0
  36. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/divergence.py +0 -0
  37. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/permutation.py +0 -0
  38. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/shannon.py +0 -0
  39. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/utils/__init__.py +0 -0
  40. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/utils/normalize.py +0 -0
  41. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/utils/windows.py +0 -0
  42. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope.egg-info/dependency_links.txt +0 -0
  43. {entroscope-0.2.0 → entroscope-0.3.0}/entroscope.egg-info/top_level.txt +0 -0
  44. {entroscope-0.2.0 → entroscope-0.3.0}/setup.cfg +0 -0
  45. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_divergence.py +0 -0
  46. {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_examples.py +0 -0
@@ -1,10 +1,10 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: entroscope
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: The definitive entropy toolkit for time series data
5
5
  Author: entroscope contributors
6
6
  License: MIT
7
- Project-URL: Homepage, https://github.com/entroscope/entroscope
7
+ Project-URL: Homepage, https://github.com/Par-python/entroscope
8
8
  Keywords: entropy,time-series,shannon,permutation,spectral
9
9
  Classifier: Development Status :: 4 - Beta
10
10
  Classifier: Intended Audience :: Science/Research
@@ -27,21 +27,41 @@ Requires-Dist: numpy
27
27
  Requires-Dist: pandas
28
28
  Requires-Dist: scipy
29
29
  Requires-Dist: matplotlib
30
+ Provides-Extra: sklearn
31
+ Requires-Dist: scikit-learn; extra == "sklearn"
32
+ Provides-Extra: polars
33
+ Requires-Dist: polars; extra == "polars"
34
+ Provides-Extra: reference
35
+ Requires-Dist: antropy; extra == "reference"
36
+ Requires-Dist: EntropyHub; extra == "reference"
37
+ Provides-Extra: docs
38
+ Requires-Dist: mkdocs-material; extra == "docs"
39
+ Requires-Dist: mkdocs<2; extra == "docs"
30
40
  Provides-Extra: dev
31
41
  Requires-Dist: pytest; extra == "dev"
32
42
  Requires-Dist: pytest-cov; extra == "dev"
33
- Requires-Dist: ruff; extra == "dev"
43
+ Requires-Dist: ruff==0.16.7; extra == "dev"
44
+ Requires-Dist: scikit-learn; extra == "dev"
45
+ Requires-Dist: polars; extra == "dev"
34
46
  Dynamic: license-file
35
47
 
36
48
  # entroscope
37
49
 
38
- [![PyPI version](https://img.shields.io/pypi/v/entroscope.svg)](https://pypi.org/project/entroscope/)
39
- [![Python versions](https://img.shields.io/pypi/pyversions/entroscope.svg)](https://pypi.org/project/entroscope/)
40
- [![CI](https://github.com/Par-python/entroscope/actions/workflows/ci.yml/badge.svg)](https://github.com/Par-python/entroscope/actions/workflows/ci.yml)
41
- [![License: MIT](https://img.shields.io/badge/license-MIT-green.svg)](https://opensource.org/licenses/MIT)
50
+ [![CI](https://img.shields.io/github/actions/workflow/status/Par-python/entroscope/ci.yml?style=flat-square&logo=githubactions&logoColor=white&label=CI&labelColor=24292e)](https://github.com/Par-python/entroscope/actions/workflows/ci.yml)
51
+ [![PyPI](https://img.shields.io/pypi/v/entroscope.svg?style=flat-square&logo=pypi&logoColor=white&labelColor=24292e&color=blue)](https://pypi.org/project/entroscope/)
52
+ [![Downloads](https://img.shields.io/pepy/dt/entroscope.svg?style=flat-square&logo=python&logoColor=white&labelColor=24292e&color=blue)](https://pepy.tech/project/entroscope)
53
+ [![Python](https://img.shields.io/pypi/pyversions/entroscope.svg?style=flat-square&logo=python&logoColor=white&labelColor=24292e&color=blue)](https://pypi.org/project/entroscope/)
54
+ [![Stars](https://img.shields.io/github/stars/Par-python/entroscope.svg?style=flat-square&logo=github&logoColor=white&labelColor=24292e&color=yellow)](https://github.com/Par-python/entroscope/stargazers)
55
+ [![License](https://img.shields.io/badge/license-MIT-blue.svg?style=flat-square&logo=opensourceinitiative&logoColor=white&labelColor=24292e)](https://opensource.org/licenses/MIT)
42
56
 
43
- **The definitive entropy toolkit for time series data.** Seven entropy measures,
44
- one consistent interface, working directly on pandas Series and numpy arrays.
57
+ **The definitive entropy toolkit for time series data.** Nine entropy measures,
58
+ one consistent interface, working directly on pandas and polars Series and numpy
59
+ arrays. Results are [validated against antropy, EntropyHub and scipy](#validated-results).
60
+
61
+ <picture>
62
+ <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/Par-python/entroscope/master/docs/assets/entropy-drop-dark.png">
63
+ <img alt="Two stacked charts. Top: a synthetic signal that is random noise until step 180, then a regular cycle. Bottom: its rolling spectral entropy, high during the noise and falling sharply shortly after step 180." src="https://raw.githubusercontent.com/Par-python/entroscope/master/docs/assets/entropy-drop-light.png">
64
+ </picture>
45
65
 
46
66
  It started in [NextOnMenu](https://github.com/Par-python/nextonmenu): a falling
47
67
  Shannon entropy of a food's regional search interest turned out to be an early
@@ -60,7 +80,7 @@ from entroscope import shannon
60
80
 
61
81
  s = pd.Series([10, 20, 15, 80, 90, 85, 88, 92])
62
82
 
63
- shannon.compute(s) # 0.73 (a single entropy value)
83
+ shannon.compute(s) # 1.75 (a single entropy value, in bits)
64
84
  shannon.rolling(s, window=20) # rolling entropy over time (a Series)
65
85
  shannon.delta(s, window=20) # rate of change of entropy
66
86
  shannon.normalized(s) # entropy scaled to [0, 1]
@@ -81,7 +101,7 @@ the standard environment variable:
81
101
  export MPLBACKEND=Agg # or, in a Dockerfile: ENV MPLBACKEND=Agg
82
102
  ```
83
103
 
84
- ## The seven measures
104
+ ## The nine measures
85
105
 
86
106
  | Measure | Import | Captures |
87
107
  | ---------------- | ------------------------- | ------------------------------------------------ |
@@ -92,6 +112,8 @@ export MPLBACKEND=Agg # or, in a Dockerfile: ENV MPLBACKEND=Agg
92
112
  | **Spectral** | `entroscope.spectral` | Spread of the power spectrum (frequency domain) |
93
113
  | **Differential** | `entroscope.differential` | Continuous entropy via a fitted distribution |
94
114
  | **Multiscale** | `entroscope.multiscale` | Sample entropy across coarse-grained time scales |
115
+ | **Transfer** | `entroscope.transfer` | Directional information flow X → Y (KSG/binned) |
116
+ | **Divergence** | `entroscope.divergence` | KL and Jensen-Shannon distance between samples |
95
117
 
96
118
  ## One consistent API
97
119
 
@@ -108,6 +130,11 @@ Every measure exposes the same methods, so switching measures is a one-word chan
108
130
  Shannon additionally provides `geographic(df, col=...)` for spatial distributions
109
131
  (e.g. search interest by region). Multiscale provides `compute` and `plot`.
110
132
 
133
+ The two-input measures take a pair of series. `transfer.compute(x, y)` (plus
134
+ `rolling`, `delta`, `plot`) estimates how much `x`'s past tells you about `y`'s
135
+ future; `divergence.kl(p, q)` and `divergence.js(p, q)` (plus `plot`) compare two
136
+ samples' distributions over shared bins.
137
+
111
138
  ## Visualization
112
139
 
113
140
  ```python
@@ -126,6 +153,43 @@ plot.drop_events(s, measure="shannon", window=20, threshold=0.4)
126
153
  All plot functions return a `matplotlib.figure.Figure` and never call
127
154
  `plt.show()`, so they're safe in scripts, notebooks, and CI alike.
128
155
 
156
+ ## Integrations
157
+
158
+ **polars**: pass a polars Series anywhere a pandas Series works; rolling and
159
+ delta results come back as a polars Series with the same name.
160
+
161
+ **scikit-learn**: `EntropyFeatures` turns time-series windows into entropy
162
+ features inside a pipeline:
163
+
164
+ ```python
165
+ from sklearn.ensemble import RandomForestClassifier
166
+ from sklearn.pipeline import make_pipeline
167
+ from entroscope.features import EntropyFeatures
168
+
169
+ # windows: shape (n_windows, window_length); one row per window
170
+ model = make_pipeline(EntropyFeatures(), RandomForestClassifier())
171
+ model.fit(windows, labels)
172
+ ```
173
+
174
+ Install the optional dependencies with `pip install "entroscope[sklearn]"` or
175
+ `"entroscope[polars]"`. See the [integrations guide](docs/integrations.md).
176
+
177
+ ## Validated results
178
+
179
+ Every measure is checked against an independent implementation, on every CI run:
180
+
181
+ | Measure | Checked against |
182
+ | -------------------------------- | ------------------------------------------------- |
183
+ | sample, approximate, permutation | antropy and EntropyHub: exact match (< 1e-10) |
184
+ | spectral | antropy: exact match |
185
+ | multiscale | EntropyHub `MSEn`: exact match |
186
+ | shannon, differential (normal) | scipy: exact match |
187
+ | transfer | analytic Gaussian closed form and Kraskov (2004) |
188
+
189
+ Details, including two definitions corrected along the way, are in
190
+ [docs/validation.md](docs/validation.md). Run the checks yourself with
191
+ `pip install -e ".[dev,reference]" && pytest tests/test_reference.py`.
192
+
129
193
  ## Real-world examples
130
194
 
131
195
  Runnable scripts live in [`examples/`](examples/); worked write-ups are in
@@ -1,12 +1,20 @@
1
1
  # entroscope
2
2
 
3
- [![PyPI version](https://img.shields.io/pypi/v/entroscope.svg)](https://pypi.org/project/entroscope/)
4
- [![Python versions](https://img.shields.io/pypi/pyversions/entroscope.svg)](https://pypi.org/project/entroscope/)
5
- [![CI](https://github.com/Par-python/entroscope/actions/workflows/ci.yml/badge.svg)](https://github.com/Par-python/entroscope/actions/workflows/ci.yml)
6
- [![License: MIT](https://img.shields.io/badge/license-MIT-green.svg)](https://opensource.org/licenses/MIT)
7
-
8
- **The definitive entropy toolkit for time series data.** Seven entropy measures,
9
- one consistent interface, working directly on pandas Series and numpy arrays.
3
+ [![CI](https://img.shields.io/github/actions/workflow/status/Par-python/entroscope/ci.yml?style=flat-square&logo=githubactions&logoColor=white&label=CI&labelColor=24292e)](https://github.com/Par-python/entroscope/actions/workflows/ci.yml)
4
+ [![PyPI](https://img.shields.io/pypi/v/entroscope.svg?style=flat-square&logo=pypi&logoColor=white&labelColor=24292e&color=blue)](https://pypi.org/project/entroscope/)
5
+ [![Downloads](https://img.shields.io/pepy/dt/entroscope.svg?style=flat-square&logo=python&logoColor=white&labelColor=24292e&color=blue)](https://pepy.tech/project/entroscope)
6
+ [![Python](https://img.shields.io/pypi/pyversions/entroscope.svg?style=flat-square&logo=python&logoColor=white&labelColor=24292e&color=blue)](https://pypi.org/project/entroscope/)
7
+ [![Stars](https://img.shields.io/github/stars/Par-python/entroscope.svg?style=flat-square&logo=github&logoColor=white&labelColor=24292e&color=yellow)](https://github.com/Par-python/entroscope/stargazers)
8
+ [![License](https://img.shields.io/badge/license-MIT-blue.svg?style=flat-square&logo=opensourceinitiative&logoColor=white&labelColor=24292e)](https://opensource.org/licenses/MIT)
9
+
10
+ **The definitive entropy toolkit for time series data.** Nine entropy measures,
11
+ one consistent interface, working directly on pandas and polars Series and numpy
12
+ arrays. Results are [validated against antropy, EntropyHub and scipy](#validated-results).
13
+
14
+ <picture>
15
+ <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/Par-python/entroscope/master/docs/assets/entropy-drop-dark.png">
16
+ <img alt="Two stacked charts. Top: a synthetic signal that is random noise until step 180, then a regular cycle. Bottom: its rolling spectral entropy, high during the noise and falling sharply shortly after step 180." src="https://raw.githubusercontent.com/Par-python/entroscope/master/docs/assets/entropy-drop-light.png">
17
+ </picture>
10
18
 
11
19
  It started in [NextOnMenu](https://github.com/Par-python/nextonmenu): a falling
12
20
  Shannon entropy of a food's regional search interest turned out to be an early
@@ -25,7 +33,7 @@ from entroscope import shannon
25
33
 
26
34
  s = pd.Series([10, 20, 15, 80, 90, 85, 88, 92])
27
35
 
28
- shannon.compute(s) # 0.73 (a single entropy value)
36
+ shannon.compute(s) # 1.75 (a single entropy value, in bits)
29
37
  shannon.rolling(s, window=20) # rolling entropy over time (a Series)
30
38
  shannon.delta(s, window=20) # rate of change of entropy
31
39
  shannon.normalized(s) # entropy scaled to [0, 1]
@@ -46,7 +54,7 @@ the standard environment variable:
46
54
  export MPLBACKEND=Agg # or, in a Dockerfile: ENV MPLBACKEND=Agg
47
55
  ```
48
56
 
49
- ## The seven measures
57
+ ## The nine measures
50
58
 
51
59
  | Measure | Import | Captures |
52
60
  | ---------------- | ------------------------- | ------------------------------------------------ |
@@ -57,6 +65,8 @@ export MPLBACKEND=Agg # or, in a Dockerfile: ENV MPLBACKEND=Agg
57
65
  | **Spectral** | `entroscope.spectral` | Spread of the power spectrum (frequency domain) |
58
66
  | **Differential** | `entroscope.differential` | Continuous entropy via a fitted distribution |
59
67
  | **Multiscale** | `entroscope.multiscale` | Sample entropy across coarse-grained time scales |
68
+ | **Transfer** | `entroscope.transfer` | Directional information flow X → Y (KSG/binned) |
69
+ | **Divergence** | `entroscope.divergence` | KL and Jensen-Shannon distance between samples |
60
70
 
61
71
  ## One consistent API
62
72
 
@@ -73,6 +83,11 @@ Every measure exposes the same methods, so switching measures is a one-word chan
73
83
  Shannon additionally provides `geographic(df, col=...)` for spatial distributions
74
84
  (e.g. search interest by region). Multiscale provides `compute` and `plot`.
75
85
 
86
+ The two-input measures take a pair of series. `transfer.compute(x, y)` (plus
87
+ `rolling`, `delta`, `plot`) estimates how much `x`'s past tells you about `y`'s
88
+ future; `divergence.kl(p, q)` and `divergence.js(p, q)` (plus `plot`) compare two
89
+ samples' distributions over shared bins.
90
+
76
91
  ## Visualization
77
92
 
78
93
  ```python
@@ -91,6 +106,43 @@ plot.drop_events(s, measure="shannon", window=20, threshold=0.4)
91
106
  All plot functions return a `matplotlib.figure.Figure` and never call
92
107
  `plt.show()`, so they're safe in scripts, notebooks, and CI alike.
93
108
 
109
+ ## Integrations
110
+
111
+ **polars**: pass a polars Series anywhere a pandas Series works; rolling and
112
+ delta results come back as a polars Series with the same name.
113
+
114
+ **scikit-learn**: `EntropyFeatures` turns time-series windows into entropy
115
+ features inside a pipeline:
116
+
117
+ ```python
118
+ from sklearn.ensemble import RandomForestClassifier
119
+ from sklearn.pipeline import make_pipeline
120
+ from entroscope.features import EntropyFeatures
121
+
122
+ # windows: shape (n_windows, window_length); one row per window
123
+ model = make_pipeline(EntropyFeatures(), RandomForestClassifier())
124
+ model.fit(windows, labels)
125
+ ```
126
+
127
+ Install the optional dependencies with `pip install "entroscope[sklearn]"` or
128
+ `"entroscope[polars]"`. See the [integrations guide](docs/integrations.md).
129
+
130
+ ## Validated results
131
+
132
+ Every measure is checked against an independent implementation, on every CI run:
133
+
134
+ | Measure | Checked against |
135
+ | -------------------------------- | ------------------------------------------------- |
136
+ | sample, approximate, permutation | antropy and EntropyHub: exact match (< 1e-10) |
137
+ | spectral | antropy: exact match |
138
+ | multiscale | EntropyHub `MSEn`: exact match |
139
+ | shannon, differential (normal) | scipy: exact match |
140
+ | transfer | analytic Gaussian closed form and Kraskov (2004) |
141
+
142
+ Details, including two definitions corrected along the way, are in
143
+ [docs/validation.md](docs/validation.md). Run the checks yourself with
144
+ `pip install -e ".[dev,reference]" && pytest tests/test_reference.py`.
145
+
94
146
  ## Real-world examples
95
147
 
96
148
  Runnable scripts live in [`examples/`](examples/); worked write-ups are in
@@ -1,28 +1,28 @@
1
1
  """entroscope: the definitive entropy toolkit for time series data."""
2
2
 
3
3
  from . import (
4
- shannon,
5
- permutation,
6
- spectral,
7
- sample,
8
4
  approximate,
9
5
  differential,
6
+ divergence,
10
7
  multiscale,
8
+ permutation,
9
+ sample,
10
+ shannon,
11
+ spectral,
11
12
  transfer,
12
- divergence,
13
13
  )
14
14
  from .utils import plot
15
15
 
16
- __version__ = "0.2.0"
16
+ __version__ = "0.3.0"
17
17
  __all__ = [
18
- "shannon",
19
- "permutation",
20
- "spectral",
21
- "sample",
22
18
  "approximate",
23
19
  "differential",
24
- "multiscale",
25
- "transfer",
26
20
  "divergence",
21
+ "multiscale",
22
+ "permutation",
27
23
  "plot",
24
+ "sample",
25
+ "shannon",
26
+ "spectral",
27
+ "transfer",
28
28
  ]
@@ -17,11 +17,31 @@ import pandas as pd
17
17
  from .utils.windows import sliding_windows
18
18
 
19
19
 
20
+ class _PolarsName:
21
+ """Stand-in 'index' for polars input, which has no index — only a name."""
22
+
23
+ def __init__(self, name):
24
+ self.name = name
25
+
26
+
27
+ def _is_polars_series(x):
28
+ # Checked by module name so polars stays an optional dependency.
29
+ cls = type(x)
30
+ return cls.__name__ == "Series" and cls.__module__.startswith("polars")
31
+
32
+
20
33
  def as_array(x):
21
- """Coerce input to a 1-D float ndarray, returning (array, index_or_None)."""
34
+ """Coerce input to a 1-D float ndarray, returning (array, index_or_None).
35
+
36
+ pandas input returns its index; polars input returns a `_PolarsName` so
37
+ `wrap` can rebuild a polars Series; anything else returns None.
38
+ """
22
39
  if isinstance(x, pd.Series):
23
40
  index = x.index
24
41
  arr = x.to_numpy(dtype=float)
42
+ elif _is_polars_series(x):
43
+ index = _PolarsName(x.name)
44
+ arr = np.asarray(x.to_numpy(), dtype=float)
25
45
  else:
26
46
  index = None
27
47
  arr = np.asarray(x, dtype=float)
@@ -33,8 +53,12 @@ def as_array(x):
33
53
 
34
54
 
35
55
  def wrap(values, index):
36
- """Wrap a result array as a Series (if index given) or return the ndarray."""
56
+ """Wrap a result array to match the input type: pandas, polars, or ndarray."""
37
57
  values = np.asarray(values, dtype=float)
58
+ if isinstance(index, _PolarsName):
59
+ import polars as pl
60
+
61
+ return pl.Series(index.name, values)
38
62
  if index is not None:
39
63
  return pd.Series(values, index=index)
40
64
  return values
@@ -60,11 +84,14 @@ def rolling(x, window, kernel, **params):
60
84
 
61
85
  def delta(x, window, kernel, **params):
62
86
  """First difference of the rolling entropy."""
63
- roll = rolling(x, window, kernel, **params)
64
- if isinstance(roll, pd.Series):
65
- return roll.diff()
66
- out = np.full_like(roll, np.nan)
67
- out[1:] = np.diff(roll)
87
+ arr, index = as_array(x)
88
+ return wrap(first_difference(rolling(arr, window, kernel, **params)), index)
89
+
90
+
91
+ def first_difference(values):
92
+ """`values[t] - values[t-1]`, with NaN at position 0 (like `Series.diff`)."""
93
+ out = np.full(len(values), np.nan)
94
+ out[1:] = np.diff(values)
68
95
  return out
69
96
 
70
97
 
@@ -75,7 +102,7 @@ def make_plot(x, window, kernel, *, title=None, ylabel="entropy", **params):
75
102
  if isinstance(roll, pd.Series):
76
103
  ax.plot(roll.index, roll.to_numpy())
77
104
  else:
78
- ax.plot(range(len(roll)), roll)
105
+ ax.plot(range(len(roll)), np.asarray(roll))
79
106
  ax.set_title(title or f"Rolling {ylabel} (window={window})")
80
107
  ax.set_xlabel("position")
81
108
  ax.set_ylabel(ylabel)
@@ -13,6 +13,7 @@ embedding test.
13
13
 
14
14
  import numpy as np
15
15
  from scipy.special import digamma
16
+
16
17
  from .utils import knn
17
18
 
18
19
 
@@ -0,0 +1,96 @@
1
+ """scikit-learn integration — entropy values as features for ML pipelines.
2
+
3
+ Each row of ``X`` is one time-series window; each output column is one entropy
4
+ measure of that window::
5
+
6
+ from sklearn.pipeline import make_pipeline
7
+ from sklearn.ensemble import RandomForestClassifier
8
+ from entroscope.features import EntropyFeatures
9
+
10
+ model = make_pipeline(EntropyFeatures(), RandomForestClassifier())
11
+ model.fit(windows, labels) # windows: (n_windows, window_length)
12
+
13
+ Requires scikit-learn (``pip install "entroscope[sklearn]"``). It is imported
14
+ here rather than in ``entroscope/__init__.py`` so the core library stays free
15
+ of the dependency.
16
+ """
17
+
18
+ import numpy as np
19
+
20
+ try:
21
+ from sklearn.base import BaseEstimator, TransformerMixin
22
+ from sklearn.utils.validation import check_array, check_is_fitted
23
+ except ImportError as exc: # pragma: no cover - exercised only without sklearn
24
+ raise ImportError(
25
+ "entroscope.features needs scikit-learn: pip install 'entroscope[sklearn]'"
26
+ ) from exc
27
+
28
+ from . import approximate, differential, permutation, sample, shannon, spectral
29
+
30
+ _MEASURES = {
31
+ "shannon": shannon.compute,
32
+ "permutation": permutation.compute,
33
+ "spectral": spectral.compute,
34
+ "sample": sample.compute,
35
+ "approximate": approximate.compute,
36
+ "differential": differential.compute,
37
+ }
38
+
39
+ DEFAULT_MEASURES = ("shannon", "permutation", "spectral", "sample", "approximate")
40
+
41
+
42
+ class EntropyFeatures(TransformerMixin, BaseEstimator):
43
+ """Transform time-series windows into one entropy feature per measure.
44
+
45
+ Parameters
46
+ ----------
47
+ measures : sequence of str
48
+ Measures to compute, in output-column order. Any of ``shannon``,
49
+ ``permutation``, ``spectral``, ``sample``, ``approximate``,
50
+ ``differential``.
51
+ params : dict, optional
52
+ Per-measure keyword arguments, e.g. ``{"sample": {"m": 2, "r": 0.15}}``.
53
+
54
+ Stateless: ``fit`` only validates input and records its width. Supports
55
+ ``set_output(transform="pandas")`` for DataFrame output with named columns.
56
+ """
57
+
58
+ def __init__(self, measures=DEFAULT_MEASURES, params=None):
59
+ self.measures = measures
60
+ self.params = params
61
+
62
+ def _check_config(self):
63
+ unknown = [m for m in self.measures if m not in _MEASURES]
64
+ if unknown or not len(self.measures):
65
+ raise ValueError(
66
+ f"unknown or empty measures {unknown or list(self.measures)!r}; "
67
+ f"choose from {sorted(_MEASURES)}"
68
+ )
69
+ extra = set(self.params or {}) - set(self.measures)
70
+ if extra:
71
+ raise ValueError(f"params given for measures not in `measures`: {sorted(extra)}")
72
+
73
+ def fit(self, X, y=None):
74
+ self._check_config()
75
+ X = check_array(X, dtype=float)
76
+ self.n_features_in_ = X.shape[1]
77
+ return self
78
+
79
+ def transform(self, X):
80
+ check_is_fitted(self, "n_features_in_")
81
+ X = check_array(X, dtype=float)
82
+ if X.shape[1] != self.n_features_in_:
83
+ raise ValueError(
84
+ f"X has {X.shape[1]} features, but EntropyFeatures is expecting "
85
+ f"{self.n_features_in_} features as input."
86
+ )
87
+ params = self.params or {}
88
+ out = np.empty((X.shape[0], len(self.measures)))
89
+ for j, name in enumerate(self.measures):
90
+ fn, kwargs = _MEASURES[name], params.get(name, {})
91
+ out[:, j] = [fn(row, **kwargs) for row in X]
92
+ return out
93
+
94
+ def get_feature_names_out(self, input_features=None):
95
+ check_is_fitted(self, "n_features_in_")
96
+ return np.asarray([f"{name}_entropy" for name in self.measures], dtype=object)
@@ -1,6 +1,7 @@
1
1
  """Multiscale entropy — sample entropy across coarse-grained time scales."""
2
2
 
3
3
  import matplotlib.pyplot as plt
4
+ import numpy as np
4
5
 
5
6
  from . import _core, sample
6
7
 
@@ -12,24 +13,33 @@ def _coarse_grain(values, scale):
12
13
  return trimmed.reshape(n, scale).mean(axis=1)
13
14
 
14
15
 
15
- def compute(series, scales=range(1, 10), method="sample"):
16
+ def compute(series, scales=range(1, 10), method="sample", m=2, r=0.2):
16
17
  """Return {scale: entropy} by coarse-graining then applying `method`.
17
18
 
19
+ Following Costa et al. (2002), the tolerance is fixed at ``r * std`` of the
20
+ ORIGINAL series and reused at every scale, so white noise loses entropy as
21
+ the scale grows while 1/f-like signals keep it.
22
+
18
23
  Scales that coarse-grain the series below sample entropy's minimum
19
24
  length are skipped (omitted from the result).
20
25
  """
21
26
  if method != "sample":
22
27
  raise ValueError("only method='sample' is supported")
28
+ if r <= 0:
29
+ raise ValueError("r must be positive")
30
+ if m < 1:
31
+ raise ValueError("m must be >= 1")
23
32
  arr, _ = _core.as_array(series)
33
+ tol = r * np.std(arr)
24
34
  result = {}
25
35
  for scale in scales:
26
36
  if scale == 1:
27
37
  grained = arr
28
38
  else:
29
39
  grained = _coarse_grain(arr, scale)
30
- if len(grained) < 4: # sample entropy needs n > m+1 (m=2 default)
40
+ if len(grained) <= m + 1: # sample entropy needs n > m+1
31
41
  continue
32
- result[int(scale)] = sample.compute(grained)
42
+ result[int(scale)] = sample._sampen(grained, m, tol)
33
43
  return result
34
44
 
35
45
 
@@ -5,10 +5,13 @@ import numpy as np
5
5
  from . import _core
6
6
 
7
7
 
8
- def _count_matches(values, m, tol):
9
- """Count template-vector pairs (length m) within Chebyshev distance `tol`."""
10
- n = len(values)
11
- templates = np.array([values[i : i + m] for i in range(n - m + 1)])
8
+ def _count_matches(values, m, tol, n_templates):
9
+ """Count template-vector pairs (length m) within Chebyshev distance `tol`.
10
+
11
+ Only the first `n_templates` templates are used, so the length-m and
12
+ length-(m+1) counts are taken over the same starting points.
13
+ """
14
+ templates = np.array([values[i : i + m] for i in range(n_templates)])
12
15
  count = 0
13
16
  for i in range(len(templates) - 1):
14
17
  dist = np.max(np.abs(templates[i + 1 :] - templates[i]), axis=1)
@@ -16,6 +19,19 @@ def _count_matches(values, m, tol):
16
19
  return count
17
20
 
18
21
 
22
+ def _sampen(values, m, tol):
23
+ """Sample entropy with an absolute tolerance (Richman & Moorman, 2000)."""
24
+ n = len(values)
25
+ if tol == 0:
26
+ return 0.0 # constant signal: perfectly regular
27
+ b = _count_matches(values, m, tol, n - m)
28
+ a = _count_matches(values, m + 1, tol, n - m)
29
+ if b == 0 or a == 0:
30
+ # no regularity detected; return a large-but-finite ceiling
31
+ return float(np.log((n - m) * (n - m - 1)))
32
+ return float(-np.log(a / b))
33
+
34
+
19
35
  def _kernel(values, m=2, r=0.2):
20
36
  """Sample entropy: -ln(A/B) of length-(m+1) vs length-m matches."""
21
37
  if r <= 0:
@@ -23,18 +39,9 @@ def _kernel(values, m=2, r=0.2):
23
39
  if m < 1:
24
40
  raise ValueError("m must be >= 1")
25
41
  values = np.asarray(values, dtype=float)
26
- n = len(values)
27
- if n <= m + 1:
42
+ if len(values) <= m + 1:
28
43
  raise ValueError("series too short for given m")
29
- tol = r * np.std(values)
30
- if tol == 0:
31
- return 0.0 # constant signal: perfectly regular
32
- b = _count_matches(values, m, tol)
33
- a = _count_matches(values, m + 1, tol)
34
- if b == 0 or a == 0:
35
- # no regularity detected; return a large-but-finite ceiling
36
- return float(np.log((n - m) * (n - m - 1)))
37
- return float(-np.log(a / b))
44
+ return _sampen(values, m, r * np.std(values))
38
45
 
39
46
 
40
47
  def compute(series, m=2, r=0.2):
@@ -41,8 +41,8 @@ def delta(series, window=50, sf=1.0):
41
41
  def normalized(series, sf=1.0):
42
42
  """Entropy scaled to [0, 1] by log2(number of frequency bins)."""
43
43
  arr, _ = _core.as_array(series)
44
- freqs, _psd_vals = _psd(arr, sf)
45
- n_bins = int(np.count_nonzero(_psd_vals > 0))
44
+ _, psd_vals = _psd(arr, sf)
45
+ n_bins = int(np.count_nonzero(psd_vals > 0))
46
46
  if n_bins <= 1:
47
47
  return 0.0
48
48
  return normalize.by_max(_kernel(arr, sf=sf), np.log2(n_bins))
@@ -73,12 +73,9 @@ def rolling(x, y, window=120, *, k=4, lag=1, method="ksg", bins=6):
73
73
 
74
74
  def delta(x, y, window=120, *, k=4, lag=1, method="ksg", bins=6):
75
75
  """First difference of the rolling transfer entropy."""
76
- roll = rolling(x, y, window, k=k, lag=lag, method=method, bins=bins)
77
- if isinstance(roll, pd.Series):
78
- return roll.diff()
79
- out = np.full_like(roll, np.nan)
80
- out[1:] = np.diff(roll)
81
- return out
76
+ xa, ya, yindex = _coerce_pair(x, y)
77
+ roll = rolling(xa, ya, window, k=k, lag=lag, method=method, bins=bins)
78
+ return _core.wrap(_core.first_difference(roll), yindex)
82
79
 
83
80
 
84
81
  def plot(x, y, window=120, *, k=4, lag=1, method="ksg", bins=6, title=None):
@@ -88,7 +85,7 @@ def plot(x, y, window=120, *, k=4, lag=1, method="ksg", bins=6, title=None):
88
85
  if isinstance(roll, pd.Series):
89
86
  ax.plot(roll.index, roll.to_numpy())
90
87
  else:
91
- ax.plot(range(len(roll)), roll)
88
+ ax.plot(range(len(roll)), np.asarray(roll))
92
89
  ax.set_title(title or f"Rolling transfer entropy (window={window})")
93
90
  ax.set_xlabel("position")
94
91
  ax.set_ylabel("transfer entropy (bits)")
@@ -25,16 +25,11 @@ def count_within_radius(points, radii):
25
25
  points = np.asarray(points, dtype=float)
26
26
  radii = np.asarray(radii, dtype=float)
27
27
  tree = cKDTree(points)
28
- counts = np.empty(len(points), dtype=int)
29
- for i, (p, r) in enumerate(zip(points, radii)):
30
- # ball_point with p=inf, count neighbors strictly inside r, minus self.
31
- idx = tree.query_ball_point(p, r=r, p=np.inf)
32
- # exclude self and any point exactly at radius r (strict inequality).
33
- c = 0
34
- for j in idx:
35
- if j == i:
36
- continue
37
- if np.max(np.abs(points[j] - p)) < r:
38
- c += 1
39
- counts[i] = c
40
- return counts
28
+ # query_ball_point counts distance <= r; shrinking each radius to the next
29
+ # float below it gives the strict inequality. Subtract 1 for the point itself.
30
+ inner = np.nextafter(radii, 0)
31
+ counts = np.asarray(
32
+ tree.query_ball_point(points, r=inner, p=np.inf, return_length=True), dtype=int
33
+ )
34
+ # Nothing is strictly within a zero radius, but duplicates sit at distance 0.
35
+ return np.where(radii > 0, counts - 1, 0)