entroscope 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {entroscope-0.2.0 → entroscope-0.3.0}/PKG-INFO +75 -11
- {entroscope-0.2.0 → entroscope-0.3.0}/README.md +61 -9
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/__init__.py +12 -12
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/_core.py +35 -8
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/_transfer_estimators.py +1 -0
- entroscope-0.3.0/entroscope/features.py +96 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/multiscale.py +13 -3
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/sample.py +22 -15
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/spectral.py +2 -2
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/transfer.py +4 -7
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/utils/knn.py +8 -13
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/utils/plot.py +4 -3
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope.egg-info/PKG-INFO +75 -11
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope.egg-info/SOURCES.txt +4 -0
- entroscope-0.3.0/entroscope.egg-info/requires.txt +25 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/pyproject.toml +8 -3
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_approximate.py +2 -1
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_backend.py +1 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_consistency.py +2 -1
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_core.py +1 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_differential.py +2 -1
- entroscope-0.3.0/tests/test_integrations.py +124 -0
- entroscope-0.3.0/tests/test_knn.py +32 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_multiscale.py +2 -1
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_permutation.py +2 -1
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_plot.py +2 -1
- entroscope-0.3.0/tests/test_reference.py +115 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_sample.py +2 -1
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_shannon.py +2 -1
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_spectral.py +2 -1
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_transfer.py +1 -0
- entroscope-0.2.0/entroscope.egg-info/requires.txt +0 -9
- {entroscope-0.2.0 → entroscope-0.3.0}/LICENSE +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/approximate.py +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/differential.py +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/divergence.py +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/permutation.py +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/shannon.py +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/utils/__init__.py +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/utils/normalize.py +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope/utils/windows.py +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope.egg-info/dependency_links.txt +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/entroscope.egg-info/top_level.txt +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/setup.cfg +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_divergence.py +0 -0
- {entroscope-0.2.0 → entroscope-0.3.0}/tests/test_examples.py +0 -0
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: entroscope
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: The definitive entropy toolkit for time series data
|
|
5
5
|
Author: entroscope contributors
|
|
6
6
|
License: MIT
|
|
7
|
-
Project-URL: Homepage, https://github.com/
|
|
7
|
+
Project-URL: Homepage, https://github.com/Par-python/entroscope
|
|
8
8
|
Keywords: entropy,time-series,shannon,permutation,spectral
|
|
9
9
|
Classifier: Development Status :: 4 - Beta
|
|
10
10
|
Classifier: Intended Audience :: Science/Research
|
|
@@ -27,21 +27,41 @@ Requires-Dist: numpy
|
|
|
27
27
|
Requires-Dist: pandas
|
|
28
28
|
Requires-Dist: scipy
|
|
29
29
|
Requires-Dist: matplotlib
|
|
30
|
+
Provides-Extra: sklearn
|
|
31
|
+
Requires-Dist: scikit-learn; extra == "sklearn"
|
|
32
|
+
Provides-Extra: polars
|
|
33
|
+
Requires-Dist: polars; extra == "polars"
|
|
34
|
+
Provides-Extra: reference
|
|
35
|
+
Requires-Dist: antropy; extra == "reference"
|
|
36
|
+
Requires-Dist: EntropyHub; extra == "reference"
|
|
37
|
+
Provides-Extra: docs
|
|
38
|
+
Requires-Dist: mkdocs-material; extra == "docs"
|
|
39
|
+
Requires-Dist: mkdocs<2; extra == "docs"
|
|
30
40
|
Provides-Extra: dev
|
|
31
41
|
Requires-Dist: pytest; extra == "dev"
|
|
32
42
|
Requires-Dist: pytest-cov; extra == "dev"
|
|
33
|
-
Requires-Dist: ruff; extra == "dev"
|
|
43
|
+
Requires-Dist: ruff==0.16.7; extra == "dev"
|
|
44
|
+
Requires-Dist: scikit-learn; extra == "dev"
|
|
45
|
+
Requires-Dist: polars; extra == "dev"
|
|
34
46
|
Dynamic: license-file
|
|
35
47
|
|
|
36
48
|
# entroscope
|
|
37
49
|
|
|
38
|
-
[](https://github.com/Par-python/entroscope/actions/workflows/ci.yml)
|
|
51
|
+
[](https://pypi.org/project/entroscope/)
|
|
52
|
+
[](https://pepy.tech/project/entroscope)
|
|
53
|
+
[](https://pypi.org/project/entroscope/)
|
|
54
|
+
[](https://github.com/Par-python/entroscope/stargazers)
|
|
55
|
+
[](https://opensource.org/licenses/MIT)
|
|
42
56
|
|
|
43
|
-
**The definitive entropy toolkit for time series data.**
|
|
44
|
-
one consistent interface, working directly on pandas Series and numpy
|
|
57
|
+
**The definitive entropy toolkit for time series data.** Nine entropy measures,
|
|
58
|
+
one consistent interface, working directly on pandas and polars Series and numpy
|
|
59
|
+
arrays. Results are [validated against antropy, EntropyHub and scipy](#validated-results).
|
|
60
|
+
|
|
61
|
+
<picture>
|
|
62
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/Par-python/entroscope/master/docs/assets/entropy-drop-dark.png">
|
|
63
|
+
<img alt="Two stacked charts. Top: a synthetic signal that is random noise until step 180, then a regular cycle. Bottom: its rolling spectral entropy, high during the noise and falling sharply shortly after step 180." src="https://raw.githubusercontent.com/Par-python/entroscope/master/docs/assets/entropy-drop-light.png">
|
|
64
|
+
</picture>
|
|
45
65
|
|
|
46
66
|
It started in [NextOnMenu](https://github.com/Par-python/nextonmenu): a falling
|
|
47
67
|
Shannon entropy of a food's regional search interest turned out to be an early
|
|
@@ -60,7 +80,7 @@ from entroscope import shannon
|
|
|
60
80
|
|
|
61
81
|
s = pd.Series([10, 20, 15, 80, 90, 85, 88, 92])
|
|
62
82
|
|
|
63
|
-
shannon.compute(s) #
|
|
83
|
+
shannon.compute(s) # 1.75 (a single entropy value, in bits)
|
|
64
84
|
shannon.rolling(s, window=20) # rolling entropy over time (a Series)
|
|
65
85
|
shannon.delta(s, window=20) # rate of change of entropy
|
|
66
86
|
shannon.normalized(s) # entropy scaled to [0, 1]
|
|
@@ -81,7 +101,7 @@ the standard environment variable:
|
|
|
81
101
|
export MPLBACKEND=Agg # or, in a Dockerfile: ENV MPLBACKEND=Agg
|
|
82
102
|
```
|
|
83
103
|
|
|
84
|
-
## The
|
|
104
|
+
## The nine measures
|
|
85
105
|
|
|
86
106
|
| Measure | Import | Captures |
|
|
87
107
|
| ---------------- | ------------------------- | ------------------------------------------------ |
|
|
@@ -92,6 +112,8 @@ export MPLBACKEND=Agg # or, in a Dockerfile: ENV MPLBACKEND=Agg
|
|
|
92
112
|
| **Spectral** | `entroscope.spectral` | Spread of the power spectrum (frequency domain) |
|
|
93
113
|
| **Differential** | `entroscope.differential` | Continuous entropy via a fitted distribution |
|
|
94
114
|
| **Multiscale** | `entroscope.multiscale` | Sample entropy across coarse-grained time scales |
|
|
115
|
+
| **Transfer** | `entroscope.transfer` | Directional information flow X → Y (KSG/binned) |
|
|
116
|
+
| **Divergence** | `entroscope.divergence` | KL and Jensen-Shannon distance between samples |
|
|
95
117
|
|
|
96
118
|
## One consistent API
|
|
97
119
|
|
|
@@ -108,6 +130,11 @@ Every measure exposes the same methods, so switching measures is a one-word chan
|
|
|
108
130
|
Shannon additionally provides `geographic(df, col=...)` for spatial distributions
|
|
109
131
|
(e.g. search interest by region). Multiscale provides `compute` and `plot`.
|
|
110
132
|
|
|
133
|
+
The two-input measures take a pair of series. `transfer.compute(x, y)` (plus
|
|
134
|
+
`rolling`, `delta`, `plot`) estimates how much `x`'s past tells you about `y`'s
|
|
135
|
+
future; `divergence.kl(p, q)` and `divergence.js(p, q)` (plus `plot`) compare two
|
|
136
|
+
samples' distributions over shared bins.
|
|
137
|
+
|
|
111
138
|
## Visualization
|
|
112
139
|
|
|
113
140
|
```python
|
|
@@ -126,6 +153,43 @@ plot.drop_events(s, measure="shannon", window=20, threshold=0.4)
|
|
|
126
153
|
All plot functions return a `matplotlib.figure.Figure` and never call
|
|
127
154
|
`plt.show()`, so they're safe in scripts, notebooks, and CI alike.
|
|
128
155
|
|
|
156
|
+
## Integrations
|
|
157
|
+
|
|
158
|
+
**polars**: pass a polars Series anywhere a pandas Series works; rolling and
|
|
159
|
+
delta results come back as a polars Series with the same name.
|
|
160
|
+
|
|
161
|
+
**scikit-learn**: `EntropyFeatures` turns time-series windows into entropy
|
|
162
|
+
features inside a pipeline:
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
166
|
+
from sklearn.pipeline import make_pipeline
|
|
167
|
+
from entroscope.features import EntropyFeatures
|
|
168
|
+
|
|
169
|
+
# windows: shape (n_windows, window_length); one row per window
|
|
170
|
+
model = make_pipeline(EntropyFeatures(), RandomForestClassifier())
|
|
171
|
+
model.fit(windows, labels)
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
Install the optional dependencies with `pip install "entroscope[sklearn]"` or
|
|
175
|
+
`"entroscope[polars]"`. See the [integrations guide](docs/integrations.md).
|
|
176
|
+
|
|
177
|
+
## Validated results
|
|
178
|
+
|
|
179
|
+
Every measure is checked against an independent implementation, on every CI run:
|
|
180
|
+
|
|
181
|
+
| Measure | Checked against |
|
|
182
|
+
| -------------------------------- | ------------------------------------------------- |
|
|
183
|
+
| sample, approximate, permutation | antropy and EntropyHub: exact match (< 1e-10) |
|
|
184
|
+
| spectral | antropy: exact match |
|
|
185
|
+
| multiscale | EntropyHub `MSEn`: exact match |
|
|
186
|
+
| shannon, differential (normal) | scipy: exact match |
|
|
187
|
+
| transfer | analytic Gaussian closed form and Kraskov (2004) |
|
|
188
|
+
|
|
189
|
+
Details, including two definitions corrected along the way, are in
|
|
190
|
+
[docs/validation.md](docs/validation.md). Run the checks yourself with
|
|
191
|
+
`pip install -e ".[dev,reference]" && pytest tests/test_reference.py`.
|
|
192
|
+
|
|
129
193
|
## Real-world examples
|
|
130
194
|
|
|
131
195
|
Runnable scripts live in [`examples/`](examples/); worked write-ups are in
|
|
@@ -1,12 +1,20 @@
|
|
|
1
1
|
# entroscope
|
|
2
2
|
|
|
3
|
-
[](https://github.com/Par-python/entroscope/actions/workflows/ci.yml)
|
|
4
|
+
[](https://pypi.org/project/entroscope/)
|
|
5
|
+
[](https://pepy.tech/project/entroscope)
|
|
6
|
+
[](https://pypi.org/project/entroscope/)
|
|
7
|
+
[](https://github.com/Par-python/entroscope/stargazers)
|
|
8
|
+
[](https://opensource.org/licenses/MIT)
|
|
9
|
+
|
|
10
|
+
**The definitive entropy toolkit for time series data.** Nine entropy measures,
|
|
11
|
+
one consistent interface, working directly on pandas and polars Series and numpy
|
|
12
|
+
arrays. Results are [validated against antropy, EntropyHub and scipy](#validated-results).
|
|
13
|
+
|
|
14
|
+
<picture>
|
|
15
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/Par-python/entroscope/master/docs/assets/entropy-drop-dark.png">
|
|
16
|
+
<img alt="Two stacked charts. Top: a synthetic signal that is random noise until step 180, then a regular cycle. Bottom: its rolling spectral entropy, high during the noise and falling sharply shortly after step 180." src="https://raw.githubusercontent.com/Par-python/entroscope/master/docs/assets/entropy-drop-light.png">
|
|
17
|
+
</picture>
|
|
10
18
|
|
|
11
19
|
It started in [NextOnMenu](https://github.com/Par-python/nextonmenu): a falling
|
|
12
20
|
Shannon entropy of a food's regional search interest turned out to be an early
|
|
@@ -25,7 +33,7 @@ from entroscope import shannon
|
|
|
25
33
|
|
|
26
34
|
s = pd.Series([10, 20, 15, 80, 90, 85, 88, 92])
|
|
27
35
|
|
|
28
|
-
shannon.compute(s) #
|
|
36
|
+
shannon.compute(s) # 1.75 (a single entropy value, in bits)
|
|
29
37
|
shannon.rolling(s, window=20) # rolling entropy over time (a Series)
|
|
30
38
|
shannon.delta(s, window=20) # rate of change of entropy
|
|
31
39
|
shannon.normalized(s) # entropy scaled to [0, 1]
|
|
@@ -46,7 +54,7 @@ the standard environment variable:
|
|
|
46
54
|
export MPLBACKEND=Agg # or, in a Dockerfile: ENV MPLBACKEND=Agg
|
|
47
55
|
```
|
|
48
56
|
|
|
49
|
-
## The
|
|
57
|
+
## The nine measures
|
|
50
58
|
|
|
51
59
|
| Measure | Import | Captures |
|
|
52
60
|
| ---------------- | ------------------------- | ------------------------------------------------ |
|
|
@@ -57,6 +65,8 @@ export MPLBACKEND=Agg # or, in a Dockerfile: ENV MPLBACKEND=Agg
|
|
|
57
65
|
| **Spectral** | `entroscope.spectral` | Spread of the power spectrum (frequency domain) |
|
|
58
66
|
| **Differential** | `entroscope.differential` | Continuous entropy via a fitted distribution |
|
|
59
67
|
| **Multiscale** | `entroscope.multiscale` | Sample entropy across coarse-grained time scales |
|
|
68
|
+
| **Transfer** | `entroscope.transfer` | Directional information flow X → Y (KSG/binned) |
|
|
69
|
+
| **Divergence** | `entroscope.divergence` | KL and Jensen-Shannon distance between samples |
|
|
60
70
|
|
|
61
71
|
## One consistent API
|
|
62
72
|
|
|
@@ -73,6 +83,11 @@ Every measure exposes the same methods, so switching measures is a one-word chan
|
|
|
73
83
|
Shannon additionally provides `geographic(df, col=...)` for spatial distributions
|
|
74
84
|
(e.g. search interest by region). Multiscale provides `compute` and `plot`.
|
|
75
85
|
|
|
86
|
+
The two-input measures take a pair of series. `transfer.compute(x, y)` (plus
|
|
87
|
+
`rolling`, `delta`, `plot`) estimates how much `x`'s past tells you about `y`'s
|
|
88
|
+
future; `divergence.kl(p, q)` and `divergence.js(p, q)` (plus `plot`) compare two
|
|
89
|
+
samples' distributions over shared bins.
|
|
90
|
+
|
|
76
91
|
## Visualization
|
|
77
92
|
|
|
78
93
|
```python
|
|
@@ -91,6 +106,43 @@ plot.drop_events(s, measure="shannon", window=20, threshold=0.4)
|
|
|
91
106
|
All plot functions return a `matplotlib.figure.Figure` and never call
|
|
92
107
|
`plt.show()`, so they're safe in scripts, notebooks, and CI alike.
|
|
93
108
|
|
|
109
|
+
## Integrations
|
|
110
|
+
|
|
111
|
+
**polars**: pass a polars Series anywhere a pandas Series works; rolling and
|
|
112
|
+
delta results come back as a polars Series with the same name.
|
|
113
|
+
|
|
114
|
+
**scikit-learn**: `EntropyFeatures` turns time-series windows into entropy
|
|
115
|
+
features inside a pipeline:
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
119
|
+
from sklearn.pipeline import make_pipeline
|
|
120
|
+
from entroscope.features import EntropyFeatures
|
|
121
|
+
|
|
122
|
+
# windows: shape (n_windows, window_length); one row per window
|
|
123
|
+
model = make_pipeline(EntropyFeatures(), RandomForestClassifier())
|
|
124
|
+
model.fit(windows, labels)
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
Install the optional dependencies with `pip install "entroscope[sklearn]"` or
|
|
128
|
+
`"entroscope[polars]"`. See the [integrations guide](docs/integrations.md).
|
|
129
|
+
|
|
130
|
+
## Validated results
|
|
131
|
+
|
|
132
|
+
Every measure is checked against an independent implementation, on every CI run:
|
|
133
|
+
|
|
134
|
+
| Measure | Checked against |
|
|
135
|
+
| -------------------------------- | ------------------------------------------------- |
|
|
136
|
+
| sample, approximate, permutation | antropy and EntropyHub: exact match (< 1e-10) |
|
|
137
|
+
| spectral | antropy: exact match |
|
|
138
|
+
| multiscale | EntropyHub `MSEn`: exact match |
|
|
139
|
+
| shannon, differential (normal) | scipy: exact match |
|
|
140
|
+
| transfer | analytic Gaussian closed form and Kraskov (2004) |
|
|
141
|
+
|
|
142
|
+
Details, including two definitions corrected along the way, are in
|
|
143
|
+
[docs/validation.md](docs/validation.md). Run the checks yourself with
|
|
144
|
+
`pip install -e ".[dev,reference]" && pytest tests/test_reference.py`.
|
|
145
|
+
|
|
94
146
|
## Real-world examples
|
|
95
147
|
|
|
96
148
|
Runnable scripts live in [`examples/`](examples/); worked write-ups are in
|
|
@@ -1,28 +1,28 @@
|
|
|
1
1
|
"""entroscope: the definitive entropy toolkit for time series data."""
|
|
2
2
|
|
|
3
3
|
from . import (
|
|
4
|
-
shannon,
|
|
5
|
-
permutation,
|
|
6
|
-
spectral,
|
|
7
|
-
sample,
|
|
8
4
|
approximate,
|
|
9
5
|
differential,
|
|
6
|
+
divergence,
|
|
10
7
|
multiscale,
|
|
8
|
+
permutation,
|
|
9
|
+
sample,
|
|
10
|
+
shannon,
|
|
11
|
+
spectral,
|
|
11
12
|
transfer,
|
|
12
|
-
divergence,
|
|
13
13
|
)
|
|
14
14
|
from .utils import plot
|
|
15
15
|
|
|
16
|
-
__version__ = "0.
|
|
16
|
+
__version__ = "0.3.0"
|
|
17
17
|
__all__ = [
|
|
18
|
-
"shannon",
|
|
19
|
-
"permutation",
|
|
20
|
-
"spectral",
|
|
21
|
-
"sample",
|
|
22
18
|
"approximate",
|
|
23
19
|
"differential",
|
|
24
|
-
"multiscale",
|
|
25
|
-
"transfer",
|
|
26
20
|
"divergence",
|
|
21
|
+
"multiscale",
|
|
22
|
+
"permutation",
|
|
27
23
|
"plot",
|
|
24
|
+
"sample",
|
|
25
|
+
"shannon",
|
|
26
|
+
"spectral",
|
|
27
|
+
"transfer",
|
|
28
28
|
]
|
|
@@ -17,11 +17,31 @@ import pandas as pd
|
|
|
17
17
|
from .utils.windows import sliding_windows
|
|
18
18
|
|
|
19
19
|
|
|
20
|
+
class _PolarsName:
|
|
21
|
+
"""Stand-in 'index' for polars input, which has no index — only a name."""
|
|
22
|
+
|
|
23
|
+
def __init__(self, name):
|
|
24
|
+
self.name = name
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _is_polars_series(x):
|
|
28
|
+
# Checked by module name so polars stays an optional dependency.
|
|
29
|
+
cls = type(x)
|
|
30
|
+
return cls.__name__ == "Series" and cls.__module__.startswith("polars")
|
|
31
|
+
|
|
32
|
+
|
|
20
33
|
def as_array(x):
|
|
21
|
-
"""Coerce input to a 1-D float ndarray, returning (array, index_or_None).
|
|
34
|
+
"""Coerce input to a 1-D float ndarray, returning (array, index_or_None).
|
|
35
|
+
|
|
36
|
+
pandas input returns its index; polars input returns a `_PolarsName` so
|
|
37
|
+
`wrap` can rebuild a polars Series; anything else returns None.
|
|
38
|
+
"""
|
|
22
39
|
if isinstance(x, pd.Series):
|
|
23
40
|
index = x.index
|
|
24
41
|
arr = x.to_numpy(dtype=float)
|
|
42
|
+
elif _is_polars_series(x):
|
|
43
|
+
index = _PolarsName(x.name)
|
|
44
|
+
arr = np.asarray(x.to_numpy(), dtype=float)
|
|
25
45
|
else:
|
|
26
46
|
index = None
|
|
27
47
|
arr = np.asarray(x, dtype=float)
|
|
@@ -33,8 +53,12 @@ def as_array(x):
|
|
|
33
53
|
|
|
34
54
|
|
|
35
55
|
def wrap(values, index):
|
|
36
|
-
"""Wrap a result array
|
|
56
|
+
"""Wrap a result array to match the input type: pandas, polars, or ndarray."""
|
|
37
57
|
values = np.asarray(values, dtype=float)
|
|
58
|
+
if isinstance(index, _PolarsName):
|
|
59
|
+
import polars as pl
|
|
60
|
+
|
|
61
|
+
return pl.Series(index.name, values)
|
|
38
62
|
if index is not None:
|
|
39
63
|
return pd.Series(values, index=index)
|
|
40
64
|
return values
|
|
@@ -60,11 +84,14 @@ def rolling(x, window, kernel, **params):
|
|
|
60
84
|
|
|
61
85
|
def delta(x, window, kernel, **params):
|
|
62
86
|
"""First difference of the rolling entropy."""
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
87
|
+
arr, index = as_array(x)
|
|
88
|
+
return wrap(first_difference(rolling(arr, window, kernel, **params)), index)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def first_difference(values):
|
|
92
|
+
"""`values[t] - values[t-1]`, with NaN at position 0 (like `Series.diff`)."""
|
|
93
|
+
out = np.full(len(values), np.nan)
|
|
94
|
+
out[1:] = np.diff(values)
|
|
68
95
|
return out
|
|
69
96
|
|
|
70
97
|
|
|
@@ -75,7 +102,7 @@ def make_plot(x, window, kernel, *, title=None, ylabel="entropy", **params):
|
|
|
75
102
|
if isinstance(roll, pd.Series):
|
|
76
103
|
ax.plot(roll.index, roll.to_numpy())
|
|
77
104
|
else:
|
|
78
|
-
ax.plot(range(len(roll)), roll)
|
|
105
|
+
ax.plot(range(len(roll)), np.asarray(roll))
|
|
79
106
|
ax.set_title(title or f"Rolling {ylabel} (window={window})")
|
|
80
107
|
ax.set_xlabel("position")
|
|
81
108
|
ax.set_ylabel(ylabel)
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""scikit-learn integration — entropy values as features for ML pipelines.
|
|
2
|
+
|
|
3
|
+
Each row of ``X`` is one time-series window; each output column is one entropy
|
|
4
|
+
measure of that window::
|
|
5
|
+
|
|
6
|
+
from sklearn.pipeline import make_pipeline
|
|
7
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
8
|
+
from entroscope.features import EntropyFeatures
|
|
9
|
+
|
|
10
|
+
model = make_pipeline(EntropyFeatures(), RandomForestClassifier())
|
|
11
|
+
model.fit(windows, labels) # windows: (n_windows, window_length)
|
|
12
|
+
|
|
13
|
+
Requires scikit-learn (``pip install "entroscope[sklearn]"``). It is imported
|
|
14
|
+
here rather than in ``entroscope/__init__.py`` so the core library stays free
|
|
15
|
+
of the dependency.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import numpy as np
|
|
19
|
+
|
|
20
|
+
try:
|
|
21
|
+
from sklearn.base import BaseEstimator, TransformerMixin
|
|
22
|
+
from sklearn.utils.validation import check_array, check_is_fitted
|
|
23
|
+
except ImportError as exc: # pragma: no cover - exercised only without sklearn
|
|
24
|
+
raise ImportError(
|
|
25
|
+
"entroscope.features needs scikit-learn: pip install 'entroscope[sklearn]'"
|
|
26
|
+
) from exc
|
|
27
|
+
|
|
28
|
+
from . import approximate, differential, permutation, sample, shannon, spectral
|
|
29
|
+
|
|
30
|
+
_MEASURES = {
|
|
31
|
+
"shannon": shannon.compute,
|
|
32
|
+
"permutation": permutation.compute,
|
|
33
|
+
"spectral": spectral.compute,
|
|
34
|
+
"sample": sample.compute,
|
|
35
|
+
"approximate": approximate.compute,
|
|
36
|
+
"differential": differential.compute,
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
DEFAULT_MEASURES = ("shannon", "permutation", "spectral", "sample", "approximate")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class EntropyFeatures(TransformerMixin, BaseEstimator):
|
|
43
|
+
"""Transform time-series windows into one entropy feature per measure.
|
|
44
|
+
|
|
45
|
+
Parameters
|
|
46
|
+
----------
|
|
47
|
+
measures : sequence of str
|
|
48
|
+
Measures to compute, in output-column order. Any of ``shannon``,
|
|
49
|
+
``permutation``, ``spectral``, ``sample``, ``approximate``,
|
|
50
|
+
``differential``.
|
|
51
|
+
params : dict, optional
|
|
52
|
+
Per-measure keyword arguments, e.g. ``{"sample": {"m": 2, "r": 0.15}}``.
|
|
53
|
+
|
|
54
|
+
Stateless: ``fit`` only validates input and records its width. Supports
|
|
55
|
+
``set_output(transform="pandas")`` for DataFrame output with named columns.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
def __init__(self, measures=DEFAULT_MEASURES, params=None):
|
|
59
|
+
self.measures = measures
|
|
60
|
+
self.params = params
|
|
61
|
+
|
|
62
|
+
def _check_config(self):
|
|
63
|
+
unknown = [m for m in self.measures if m not in _MEASURES]
|
|
64
|
+
if unknown or not len(self.measures):
|
|
65
|
+
raise ValueError(
|
|
66
|
+
f"unknown or empty measures {unknown or list(self.measures)!r}; "
|
|
67
|
+
f"choose from {sorted(_MEASURES)}"
|
|
68
|
+
)
|
|
69
|
+
extra = set(self.params or {}) - set(self.measures)
|
|
70
|
+
if extra:
|
|
71
|
+
raise ValueError(f"params given for measures not in `measures`: {sorted(extra)}")
|
|
72
|
+
|
|
73
|
+
def fit(self, X, y=None):
|
|
74
|
+
self._check_config()
|
|
75
|
+
X = check_array(X, dtype=float)
|
|
76
|
+
self.n_features_in_ = X.shape[1]
|
|
77
|
+
return self
|
|
78
|
+
|
|
79
|
+
def transform(self, X):
|
|
80
|
+
check_is_fitted(self, "n_features_in_")
|
|
81
|
+
X = check_array(X, dtype=float)
|
|
82
|
+
if X.shape[1] != self.n_features_in_:
|
|
83
|
+
raise ValueError(
|
|
84
|
+
f"X has {X.shape[1]} features, but EntropyFeatures is expecting "
|
|
85
|
+
f"{self.n_features_in_} features as input."
|
|
86
|
+
)
|
|
87
|
+
params = self.params or {}
|
|
88
|
+
out = np.empty((X.shape[0], len(self.measures)))
|
|
89
|
+
for j, name in enumerate(self.measures):
|
|
90
|
+
fn, kwargs = _MEASURES[name], params.get(name, {})
|
|
91
|
+
out[:, j] = [fn(row, **kwargs) for row in X]
|
|
92
|
+
return out
|
|
93
|
+
|
|
94
|
+
def get_feature_names_out(self, input_features=None):
|
|
95
|
+
check_is_fitted(self, "n_features_in_")
|
|
96
|
+
return np.asarray([f"{name}_entropy" for name in self.measures], dtype=object)
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
"""Multiscale entropy — sample entropy across coarse-grained time scales."""
|
|
2
2
|
|
|
3
3
|
import matplotlib.pyplot as plt
|
|
4
|
+
import numpy as np
|
|
4
5
|
|
|
5
6
|
from . import _core, sample
|
|
6
7
|
|
|
@@ -12,24 +13,33 @@ def _coarse_grain(values, scale):
|
|
|
12
13
|
return trimmed.reshape(n, scale).mean(axis=1)
|
|
13
14
|
|
|
14
15
|
|
|
15
|
-
def compute(series, scales=range(1, 10), method="sample"):
|
|
16
|
+
def compute(series, scales=range(1, 10), method="sample", m=2, r=0.2):
|
|
16
17
|
"""Return {scale: entropy} by coarse-graining then applying `method`.
|
|
17
18
|
|
|
19
|
+
Following Costa et al. (2002), the tolerance is fixed at ``r * std`` of the
|
|
20
|
+
ORIGINAL series and reused at every scale, so white noise loses entropy as
|
|
21
|
+
the scale grows while 1/f-like signals keep it.
|
|
22
|
+
|
|
18
23
|
Scales that coarse-grain the series below sample entropy's minimum
|
|
19
24
|
length are skipped (omitted from the result).
|
|
20
25
|
"""
|
|
21
26
|
if method != "sample":
|
|
22
27
|
raise ValueError("only method='sample' is supported")
|
|
28
|
+
if r <= 0:
|
|
29
|
+
raise ValueError("r must be positive")
|
|
30
|
+
if m < 1:
|
|
31
|
+
raise ValueError("m must be >= 1")
|
|
23
32
|
arr, _ = _core.as_array(series)
|
|
33
|
+
tol = r * np.std(arr)
|
|
24
34
|
result = {}
|
|
25
35
|
for scale in scales:
|
|
26
36
|
if scale == 1:
|
|
27
37
|
grained = arr
|
|
28
38
|
else:
|
|
29
39
|
grained = _coarse_grain(arr, scale)
|
|
30
|
-
if len(grained)
|
|
40
|
+
if len(grained) <= m + 1: # sample entropy needs n > m+1
|
|
31
41
|
continue
|
|
32
|
-
result[int(scale)] = sample.
|
|
42
|
+
result[int(scale)] = sample._sampen(grained, m, tol)
|
|
33
43
|
return result
|
|
34
44
|
|
|
35
45
|
|
|
@@ -5,10 +5,13 @@ import numpy as np
|
|
|
5
5
|
from . import _core
|
|
6
6
|
|
|
7
7
|
|
|
8
|
-
def _count_matches(values, m, tol):
|
|
9
|
-
"""Count template-vector pairs (length m) within Chebyshev distance `tol`.
|
|
10
|
-
|
|
11
|
-
|
|
8
|
+
def _count_matches(values, m, tol, n_templates):
|
|
9
|
+
"""Count template-vector pairs (length m) within Chebyshev distance `tol`.
|
|
10
|
+
|
|
11
|
+
Only the first `n_templates` templates are used, so the length-m and
|
|
12
|
+
length-(m+1) counts are taken over the same starting points.
|
|
13
|
+
"""
|
|
14
|
+
templates = np.array([values[i : i + m] for i in range(n_templates)])
|
|
12
15
|
count = 0
|
|
13
16
|
for i in range(len(templates) - 1):
|
|
14
17
|
dist = np.max(np.abs(templates[i + 1 :] - templates[i]), axis=1)
|
|
@@ -16,6 +19,19 @@ def _count_matches(values, m, tol):
|
|
|
16
19
|
return count
|
|
17
20
|
|
|
18
21
|
|
|
22
|
+
def _sampen(values, m, tol):
|
|
23
|
+
"""Sample entropy with an absolute tolerance (Richman & Moorman, 2000)."""
|
|
24
|
+
n = len(values)
|
|
25
|
+
if tol == 0:
|
|
26
|
+
return 0.0 # constant signal: perfectly regular
|
|
27
|
+
b = _count_matches(values, m, tol, n - m)
|
|
28
|
+
a = _count_matches(values, m + 1, tol, n - m)
|
|
29
|
+
if b == 0 or a == 0:
|
|
30
|
+
# no regularity detected; return a large-but-finite ceiling
|
|
31
|
+
return float(np.log((n - m) * (n - m - 1)))
|
|
32
|
+
return float(-np.log(a / b))
|
|
33
|
+
|
|
34
|
+
|
|
19
35
|
def _kernel(values, m=2, r=0.2):
|
|
20
36
|
"""Sample entropy: -ln(A/B) of length-(m+1) vs length-m matches."""
|
|
21
37
|
if r <= 0:
|
|
@@ -23,18 +39,9 @@ def _kernel(values, m=2, r=0.2):
|
|
|
23
39
|
if m < 1:
|
|
24
40
|
raise ValueError("m must be >= 1")
|
|
25
41
|
values = np.asarray(values, dtype=float)
|
|
26
|
-
|
|
27
|
-
if n <= m + 1:
|
|
42
|
+
if len(values) <= m + 1:
|
|
28
43
|
raise ValueError("series too short for given m")
|
|
29
|
-
|
|
30
|
-
if tol == 0:
|
|
31
|
-
return 0.0 # constant signal: perfectly regular
|
|
32
|
-
b = _count_matches(values, m, tol)
|
|
33
|
-
a = _count_matches(values, m + 1, tol)
|
|
34
|
-
if b == 0 or a == 0:
|
|
35
|
-
# no regularity detected; return a large-but-finite ceiling
|
|
36
|
-
return float(np.log((n - m) * (n - m - 1)))
|
|
37
|
-
return float(-np.log(a / b))
|
|
44
|
+
return _sampen(values, m, r * np.std(values))
|
|
38
45
|
|
|
39
46
|
|
|
40
47
|
def compute(series, m=2, r=0.2):
|
|
@@ -41,8 +41,8 @@ def delta(series, window=50, sf=1.0):
|
|
|
41
41
|
def normalized(series, sf=1.0):
|
|
42
42
|
"""Entropy scaled to [0, 1] by log2(number of frequency bins)."""
|
|
43
43
|
arr, _ = _core.as_array(series)
|
|
44
|
-
|
|
45
|
-
n_bins = int(np.count_nonzero(
|
|
44
|
+
_, psd_vals = _psd(arr, sf)
|
|
45
|
+
n_bins = int(np.count_nonzero(psd_vals > 0))
|
|
46
46
|
if n_bins <= 1:
|
|
47
47
|
return 0.0
|
|
48
48
|
return normalize.by_max(_kernel(arr, sf=sf), np.log2(n_bins))
|
|
@@ -73,12 +73,9 @@ def rolling(x, y, window=120, *, k=4, lag=1, method="ksg", bins=6):
|
|
|
73
73
|
|
|
74
74
|
def delta(x, y, window=120, *, k=4, lag=1, method="ksg", bins=6):
|
|
75
75
|
"""First difference of the rolling transfer entropy."""
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
out = np.full_like(roll, np.nan)
|
|
80
|
-
out[1:] = np.diff(roll)
|
|
81
|
-
return out
|
|
76
|
+
xa, ya, yindex = _coerce_pair(x, y)
|
|
77
|
+
roll = rolling(xa, ya, window, k=k, lag=lag, method=method, bins=bins)
|
|
78
|
+
return _core.wrap(_core.first_difference(roll), yindex)
|
|
82
79
|
|
|
83
80
|
|
|
84
81
|
def plot(x, y, window=120, *, k=4, lag=1, method="ksg", bins=6, title=None):
|
|
@@ -88,7 +85,7 @@ def plot(x, y, window=120, *, k=4, lag=1, method="ksg", bins=6, title=None):
|
|
|
88
85
|
if isinstance(roll, pd.Series):
|
|
89
86
|
ax.plot(roll.index, roll.to_numpy())
|
|
90
87
|
else:
|
|
91
|
-
ax.plot(range(len(roll)), roll)
|
|
88
|
+
ax.plot(range(len(roll)), np.asarray(roll))
|
|
92
89
|
ax.set_title(title or f"Rolling transfer entropy (window={window})")
|
|
93
90
|
ax.set_xlabel("position")
|
|
94
91
|
ax.set_ylabel("transfer entropy (bits)")
|
|
@@ -25,16 +25,11 @@ def count_within_radius(points, radii):
|
|
|
25
25
|
points = np.asarray(points, dtype=float)
|
|
26
26
|
radii = np.asarray(radii, dtype=float)
|
|
27
27
|
tree = cKDTree(points)
|
|
28
|
-
counts
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
continue
|
|
37
|
-
if np.max(np.abs(points[j] - p)) < r:
|
|
38
|
-
c += 1
|
|
39
|
-
counts[i] = c
|
|
40
|
-
return counts
|
|
28
|
+
# query_ball_point counts distance <= r; shrinking each radius to the next
|
|
29
|
+
# float below it gives the strict inequality. Subtract 1 for the point itself.
|
|
30
|
+
inner = np.nextafter(radii, 0)
|
|
31
|
+
counts = np.asarray(
|
|
32
|
+
tree.query_ball_point(points, r=inner, p=np.inf, return_length=True), dtype=int
|
|
33
|
+
)
|
|
34
|
+
# Nothing is strictly within a zero radius, but duplicates sit at distance 0.
|
|
35
|
+
return np.where(radii > 0, counts - 1, 0)
|