pytrendclust 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pytrendclust-0.1.0/LICENSE +21 -0
- pytrendclust-0.1.0/PKG-INFO +158 -0
- pytrendclust-0.1.0/README.md +125 -0
- pytrendclust-0.1.0/pyproject.toml +112 -0
- pytrendclust-0.1.0/src/pytrendclust/__init__.py +37 -0
- pytrendclust-0.1.0/src/pytrendclust/_cost.py +70 -0
- pytrendclust-0.1.0/src/pytrendclust/_dp.py +80 -0
- pytrendclust-0.1.0/src/pytrendclust/clusterer.py +310 -0
- pytrendclust-0.1.0/src/pytrendclust/penalty.py +74 -0
- pytrendclust-0.1.0/src/pytrendclust/py.typed +0 -0
- pytrendclust-0.1.0/src/pytrendclust/result.py +303 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Andrey Ferubko
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pytrendclust
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Trend clustering of time series with an adaptive number of clusters: split a series into rising, falling and flat runs with an exponentially growing penalty on change points
|
|
5
|
+
Keywords: time-series,clustering,trend,change-point,segmentation,piecewise-linear,regression
|
|
6
|
+
Author: Andrey Ferubko
|
|
7
|
+
Author-email: Andrey Ferubko <ferubko1999@yandex.ru>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Mathematics
|
|
23
|
+
Classifier: Typing :: Typed
|
|
24
|
+
Requires-Dist: numpy>=1.26
|
|
25
|
+
Requires-Dist: pandas>=2.1 ; extra == 'pandas'
|
|
26
|
+
Requires-Python: >=3.11
|
|
27
|
+
Project-URL: Homepage, https://github.com/89605502155/pytrendclust
|
|
28
|
+
Project-URL: Repository, https://github.com/89605502155/pytrendclust
|
|
29
|
+
Project-URL: Documentation, https://github.com/89605502155/pytrendclust#readme
|
|
30
|
+
Project-URL: Issues, https://github.com/89605502155/pytrendclust/issues
|
|
31
|
+
Provides-Extra: pandas
|
|
32
|
+
Description-Content-Type: text/markdown
|
|
33
|
+
|
|
34
|
+
# pytrendclust
|
|
35
|
+
|
|
36
|
+
[](https://pypi.org/project/pytrendclust/)
|
|
37
|
+
[](https://pypi.org/project/pytrendclust/)
|
|
38
|
+
[](https://github.com/89605502155/pytrendclust/blob/main/LICENSE)
|
|
39
|
+
|
|
40
|
+
Source code: <https://github.com/89605502155/pytrendclust>
|
|
41
|
+
|
|
42
|
+
Trend clustering of time series with an **adaptive number of clusters**.
|
|
43
|
+
|
|
44
|
+
`pytrendclust` splits a series into consecutive runs that rise, fall or stay
|
|
45
|
+
flat. You do not pass the number of clusters. It is chosen by a penalised
|
|
46
|
+
least-squares criterion in which every extra change point costs exponentially
|
|
47
|
+
more than the previous one. Real changes of direction are found, and noise is
|
|
48
|
+
not mistaken for a trend.
|
|
49
|
+
|
|
50
|
+
Typical use: keep only the most recent regime of a series as the training set
|
|
51
|
+
for a regression model. For example, a series of RAM consumption first falls
|
|
52
|
+
and then grows. Fitting a line to the whole history mixes two regimes; fitting
|
|
53
|
+
it to the last cluster does not.
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
from pytrendclust import TrendClusterer
|
|
57
|
+
|
|
58
|
+
y = [10, 9, 8, 7, 6, 5, 4, 5, 6, 7]
|
|
59
|
+
|
|
60
|
+
result = TrendClusterer().fit(y).result_
|
|
61
|
+
print(result.summary())
|
|
62
|
+
# 10 points, 1 change point(s), noise std 0.27
|
|
63
|
+
# #0: [0, 7) 7 pts down slope -1
|
|
64
|
+
# #1: [7, 10) 3 pts up slope +1
|
|
65
|
+
|
|
66
|
+
t_recent, y_recent = result.select() # points of the last (newest) cluster
|
|
67
|
+
result.export("result.json") # or "points.csv"
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
A "picket fence" `5, 6, 5, 6, ...` is one flat cluster, not ten tiny trends:
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
>>> from pytrendclust import cluster_trends
|
|
74
|
+
>>> cluster_trends([5, 6] * 5).summary()
|
|
75
|
+
'10 points, 0 change point(s), noise std 1.211\n #0: [0, 10) 10 pts flat slope +0.0303'
|
|
76
|
+
>>> cluster_trends([5, 6] * 5, penalty="none").n_change_points # no regularisation
|
|
77
|
+
2
|
|
78
|
+
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Installation
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
pip install pytrendclust # numpy only
|
|
85
|
+
pip install "pytrendclust[pandas]" # + DataFrame export
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## Method
|
|
89
|
+
|
|
90
|
+
For every possible number of change points `K`, a dynamic programme finds the
|
|
91
|
+
exact partition into `K + 1` runs with the smallest total squared error of
|
|
92
|
+
per-run least-squares lines, `SSE(K)`. The clusterer then picks
|
|
93
|
+
|
|
94
|
+
```
|
|
95
|
+
K* = argmin_K SSE(K) + sigma² · pen(K)
|
|
96
|
+
pen(K) = c · (exp(g·K) − 1) / (exp(g) − 1)
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
* `c` (`penalty_scale`): price of the first change point in noise variances.
|
|
100
|
+
The default is `5·ln(n)`.
|
|
101
|
+
* `g` (`penalty_growth`, default 0.5): each next change point costs `e^g`
|
|
102
|
+
times more than the previous one. `g = 0` gives the classic linear penalty.
|
|
103
|
+
* `sigma` (`noise_std`): noise level. By default it is estimated robustly
|
|
104
|
+
from second differences of the series, which cancel any linear trend. This
|
|
105
|
+
makes the result independent of units: bytes and gigabytes give the same
|
|
106
|
+
clusters.
|
|
107
|
+
|
|
108
|
+
The search over `K` stops as soon as the penalty alone exceeds the best value
|
|
109
|
+
found. The exponential penalty makes this happen after a few steps. The
|
|
110
|
+
solution is the global optimum of the criterion; the test suite checks it
|
|
111
|
+
against brute-force enumeration.
|
|
112
|
+
|
|
113
|
+
Each cluster is labelled `up`, `down` or `flat` with a t-test on its slope:
|
|
114
|
+
`|slope| / se(slope) < flat_threshold` (default 2) means `flat`.
|
|
115
|
+
|
|
116
|
+
## API
|
|
117
|
+
|
|
118
|
+
| Object | Purpose |
|
|
119
|
+
|---|---|
|
|
120
|
+
| `TrendClusterer(**params)` | set parameters; `fit(y, t=None)`, `fit_predict`, `get_params`, `set_params` |
|
|
121
|
+
| `cluster_trends(y, t=None, **params)` | one-call shortcut returning the result |
|
|
122
|
+
| `TrendClusteringResult` | `segments`, `change_points`, `labels`, `trends`, `fitted`, `last_segment`, `select()`, `indices()`, `criterion` |
|
|
123
|
+
| export | `to_dict()`, `to_json(path)`, `to_csv(path)`, `to_dataframe()`, `export(path)`; `from_dict`, `from_json` |
|
|
124
|
+
|
|
125
|
+
Parameters: `penalty` (`"exponential"`, `"linear"`, `"none"`, or a callable
|
|
126
|
+
`f(K, n, sigma²)`), `penalty_scale`, `penalty_growth`, `min_size` (default 3),
|
|
127
|
+
`max_change_points`, `noise_std`, `flat_threshold`.
|
|
128
|
+
|
|
129
|
+
`result.criterion` lists `SSE(K)`, the penalty and their sum for every `K` the
|
|
130
|
+
search evaluated. Use it to justify the chosen number of clusters.
|
|
131
|
+
|
|
132
|
+
Complexity is `O(K · n²)` time and `O(K · n)` memory: about 0.4 s for 1000
|
|
133
|
+
points and 6 s for 5000.
|
|
134
|
+
|
|
135
|
+
## Кратко по-русски
|
|
136
|
+
|
|
137
|
+
Трендовая кластеризация временного ряда с адаптивным числом классов. Ряд
|
|
138
|
+
делится на участки роста, спада и плато. Число точек разладки выбирается
|
|
139
|
+
автоматически: к ошибке модели добавляется штраф, растущий экспоненциально с
|
|
140
|
+
каждой новой точкой. Так шумовые колебания («заборчик») не принимаются за смену
|
|
141
|
+
тренда. `result.select()` возвращает точки последнего, самого свежего участка.
|
|
142
|
+
На них обучается регрессия.
|
|
143
|
+
|
|
144
|
+
## Development
|
|
145
|
+
|
|
146
|
+
```bash
|
|
147
|
+
git clone https://github.com/89605502155/pytrendclust.git
|
|
148
|
+
cd pytrendclust
|
|
149
|
+
uv sync --all-extras
|
|
150
|
+
uv run pytest
|
|
151
|
+
uv run ruff check src tests && uv run ruff format --check src tests
|
|
152
|
+
uv run mypy
|
|
153
|
+
uv build
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
## License
|
|
157
|
+
|
|
158
|
+
MIT, see [LICENSE](https://github.com/89605502155/pytrendclust/blob/main/LICENSE).
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
# pytrendclust
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/pytrendclust/)
|
|
4
|
+
[](https://pypi.org/project/pytrendclust/)
|
|
5
|
+
[](https://github.com/89605502155/pytrendclust/blob/main/LICENSE)
|
|
6
|
+
|
|
7
|
+
Source code: <https://github.com/89605502155/pytrendclust>
|
|
8
|
+
|
|
9
|
+
Trend clustering of time series with an **adaptive number of clusters**.
|
|
10
|
+
|
|
11
|
+
`pytrendclust` splits a series into consecutive runs that rise, fall or stay
|
|
12
|
+
flat. You do not pass the number of clusters. It is chosen by a penalised
|
|
13
|
+
least-squares criterion in which every extra change point costs exponentially
|
|
14
|
+
more than the previous one. Real changes of direction are found, and noise is
|
|
15
|
+
not mistaken for a trend.
|
|
16
|
+
|
|
17
|
+
Typical use: keep only the most recent regime of a series as the training set
|
|
18
|
+
for a regression model. For example, a series of RAM consumption first falls
|
|
19
|
+
and then grows. Fitting a line to the whole history mixes two regimes; fitting
|
|
20
|
+
it to the last cluster does not.
|
|
21
|
+
|
|
22
|
+
```python
|
|
23
|
+
from pytrendclust import TrendClusterer
|
|
24
|
+
|
|
25
|
+
y = [10, 9, 8, 7, 6, 5, 4, 5, 6, 7]
|
|
26
|
+
|
|
27
|
+
result = TrendClusterer().fit(y).result_
|
|
28
|
+
print(result.summary())
|
|
29
|
+
# 10 points, 1 change point(s), noise std 0.27
|
|
30
|
+
# #0: [0, 7) 7 pts down slope -1
|
|
31
|
+
# #1: [7, 10) 3 pts up slope +1
|
|
32
|
+
|
|
33
|
+
t_recent, y_recent = result.select() # points of the last (newest) cluster
|
|
34
|
+
result.export("result.json") # or "points.csv"
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
A "picket fence" `5, 6, 5, 6, ...` is one flat cluster, not ten tiny trends:
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
>>> from pytrendclust import cluster_trends
|
|
41
|
+
>>> cluster_trends([5, 6] * 5).summary()
|
|
42
|
+
'10 points, 0 change point(s), noise std 1.211\n #0: [0, 10) 10 pts flat slope +0.0303'
|
|
43
|
+
>>> cluster_trends([5, 6] * 5, penalty="none").n_change_points # no regularisation
|
|
44
|
+
2
|
|
45
|
+
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
## Installation
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
pip install pytrendclust # numpy only
|
|
52
|
+
pip install "pytrendclust[pandas]" # + DataFrame export
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
## Method
|
|
56
|
+
|
|
57
|
+
For every possible number of change points `K`, a dynamic programme finds the
|
|
58
|
+
exact partition into `K + 1` runs with the smallest total squared error of
|
|
59
|
+
per-run least-squares lines, `SSE(K)`. The clusterer then picks
|
|
60
|
+
|
|
61
|
+
```
|
|
62
|
+
K* = argmin_K SSE(K) + sigma² · pen(K)
|
|
63
|
+
pen(K) = c · (exp(g·K) − 1) / (exp(g) − 1)
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
* `c` (`penalty_scale`): price of the first change point in noise variances.
|
|
67
|
+
The default is `5·ln(n)`.
|
|
68
|
+
* `g` (`penalty_growth`, default 0.5): each next change point costs `e^g`
|
|
69
|
+
times more than the previous one. `g = 0` gives the classic linear penalty.
|
|
70
|
+
* `sigma` (`noise_std`): noise level. By default it is estimated robustly
|
|
71
|
+
from second differences of the series, which cancel any linear trend. This
|
|
72
|
+
makes the result independent of units: bytes and gigabytes give the same
|
|
73
|
+
clusters.
|
|
74
|
+
|
|
75
|
+
The search over `K` stops as soon as the penalty alone exceeds the best value
|
|
76
|
+
found. The exponential penalty makes this happen after a few steps. The
|
|
77
|
+
solution is the global optimum of the criterion; the test suite checks it
|
|
78
|
+
against brute-force enumeration.
|
|
79
|
+
|
|
80
|
+
Each cluster is labelled `up`, `down` or `flat` with a t-test on its slope:
|
|
81
|
+
`|slope| / se(slope) < flat_threshold` (default 2) means `flat`.
|
|
82
|
+
|
|
83
|
+
## API
|
|
84
|
+
|
|
85
|
+
| Object | Purpose |
|
|
86
|
+
|---|---|
|
|
87
|
+
| `TrendClusterer(**params)` | set parameters; `fit(y, t=None)`, `fit_predict`, `get_params`, `set_params` |
|
|
88
|
+
| `cluster_trends(y, t=None, **params)` | one-call shortcut returning the result |
|
|
89
|
+
| `TrendClusteringResult` | `segments`, `change_points`, `labels`, `trends`, `fitted`, `last_segment`, `select()`, `indices()`, `criterion` |
|
|
90
|
+
| export | `to_dict()`, `to_json(path)`, `to_csv(path)`, `to_dataframe()`, `export(path)`; `from_dict`, `from_json` |
|
|
91
|
+
|
|
92
|
+
Parameters: `penalty` (`"exponential"`, `"linear"`, `"none"`, or a callable
|
|
93
|
+
`f(K, n, sigma²)`), `penalty_scale`, `penalty_growth`, `min_size` (default 3),
|
|
94
|
+
`max_change_points`, `noise_std`, `flat_threshold`.
|
|
95
|
+
|
|
96
|
+
`result.criterion` lists `SSE(K)`, the penalty and their sum for every `K` the
|
|
97
|
+
search evaluated. Use it to justify the chosen number of clusters.
|
|
98
|
+
|
|
99
|
+
Complexity is `O(K · n²)` time and `O(K · n)` memory: about 0.4 s for 1000
|
|
100
|
+
points and 6 s for 5000.
|
|
101
|
+
|
|
102
|
+
## Кратко по-русски
|
|
103
|
+
|
|
104
|
+
Трендовая кластеризация временного ряда с адаптивным числом классов. Ряд
|
|
105
|
+
делится на участки роста, спада и плато. Число точек разладки выбирается
|
|
106
|
+
автоматически: к ошибке модели добавляется штраф, растущий экспоненциально с
|
|
107
|
+
каждой новой точкой. Так шумовые колебания («заборчик») не принимаются за смену
|
|
108
|
+
тренда. `result.select()` возвращает точки последнего, самого свежего участка.
|
|
109
|
+
На них обучается регрессия.
|
|
110
|
+
|
|
111
|
+
## Development
|
|
112
|
+
|
|
113
|
+
```bash
|
|
114
|
+
git clone https://github.com/89605502155/pytrendclust.git
|
|
115
|
+
cd pytrendclust
|
|
116
|
+
uv sync --all-extras
|
|
117
|
+
uv run pytest
|
|
118
|
+
uv run ruff check src tests && uv run ruff format --check src tests
|
|
119
|
+
uv run mypy
|
|
120
|
+
uv build
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
## License
|
|
124
|
+
|
|
125
|
+
MIT, see [LICENSE](https://github.com/89605502155/pytrendclust/blob/main/LICENSE).
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "pytrendclust"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Trend clustering of time series with an adaptive number of clusters: split a series into rising, falling and flat runs with an exponentially growing penalty on change points"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
authors = [
|
|
7
|
+
{ name = "Andrey Ferubko", email = "ferubko1999@yandex.ru" }
|
|
8
|
+
]
|
|
9
|
+
requires-python = ">=3.11"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
keywords = [
|
|
13
|
+
"time-series",
|
|
14
|
+
"clustering",
|
|
15
|
+
"trend",
|
|
16
|
+
"change-point",
|
|
17
|
+
"segmentation",
|
|
18
|
+
"piecewise-linear",
|
|
19
|
+
"regression",
|
|
20
|
+
]
|
|
21
|
+
classifiers = [
|
|
22
|
+
"Development Status :: 4 - Beta",
|
|
23
|
+
"Intended Audience :: Developers",
|
|
24
|
+
"Intended Audience :: Science/Research",
|
|
25
|
+
"Operating System :: OS Independent",
|
|
26
|
+
"Programming Language :: Python :: 3",
|
|
27
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
28
|
+
"Programming Language :: Python :: 3.11",
|
|
29
|
+
"Programming Language :: Python :: 3.12",
|
|
30
|
+
"Programming Language :: Python :: 3.13",
|
|
31
|
+
"Programming Language :: Python :: 3.14",
|
|
32
|
+
"Topic :: Scientific/Engineering",
|
|
33
|
+
"Topic :: Scientific/Engineering :: Information Analysis",
|
|
34
|
+
"Topic :: Scientific/Engineering :: Mathematics",
|
|
35
|
+
"Typing :: Typed",
|
|
36
|
+
]
|
|
37
|
+
dependencies = [
|
|
38
|
+
"numpy>=1.26",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
[project.optional-dependencies]
|
|
42
|
+
pandas = [
|
|
43
|
+
"pandas>=2.1",
|
|
44
|
+
]
|
|
45
|
+
|
|
46
|
+
[project.urls]
|
|
47
|
+
Homepage = "https://github.com/89605502155/pytrendclust"
|
|
48
|
+
Repository = "https://github.com/89605502155/pytrendclust"
|
|
49
|
+
Documentation = "https://github.com/89605502155/pytrendclust#readme"
|
|
50
|
+
Issues = "https://github.com/89605502155/pytrendclust/issues"
|
|
51
|
+
|
|
52
|
+
[build-system]
|
|
53
|
+
requires = ["uv_build>=0.11.9,<0.12.0"]
|
|
54
|
+
build-backend = "uv_build"
|
|
55
|
+
|
|
56
|
+
[dependency-groups]
|
|
57
|
+
dev = [
|
|
58
|
+
"hypothesis>=6.168.5",
|
|
59
|
+
"mypy>=2.4.0",
|
|
60
|
+
"pytest>=9.1.1",
|
|
61
|
+
"pytest-cov>=7.1.0",
|
|
62
|
+
"ruff>=0.16.10",
|
|
63
|
+
]
|
|
64
|
+
|
|
65
|
+
[tool.ruff]
|
|
66
|
+
line-length = 100
|
|
67
|
+
target-version = "py311"
|
|
68
|
+
src = ["src", "tests"]
|
|
69
|
+
|
|
70
|
+
[tool.ruff.lint]
|
|
71
|
+
select = ["E", "W", "F", "I", "N", "UP", "B", "A", "C4", "EM", "ISC", "PIE", "PT", "RET", "SIM",
|
|
72
|
+
"TID", "T20", "PERF", "PL", "RUF", "S"]
|
|
73
|
+
ignore = [
|
|
74
|
+
"PLR0913", # too many arguments: the constructor mirrors the documented parameters
|
|
75
|
+
"PLR2004", # magic values
|
|
76
|
+
"PLC0415", # imports inside functions keep optional dependencies optional
|
|
77
|
+
]
|
|
78
|
+
|
|
79
|
+
[tool.ruff.lint.per-file-ignores]
|
|
80
|
+
"tests/**" = ["S101", "PLR2004", "PLC0415", "N802", "S311", "RUF001", "RUF003"]
|
|
81
|
+
"examples/**" = ["T20", "S101", "RUF001", "RUF003"]
|
|
82
|
+
|
|
83
|
+
[tool.ruff.lint.isort]
|
|
84
|
+
known-first-party = ["pytrendclust"]
|
|
85
|
+
|
|
86
|
+
[tool.ruff.format]
|
|
87
|
+
docstring-code-format = true
|
|
88
|
+
|
|
89
|
+
[tool.pytest.ini_options]
|
|
90
|
+
testpaths = ["tests", "src", "README.md"]
|
|
91
|
+
addopts = ["-ra", "--strict-markers", "--strict-config", "--doctest-modules", "--doctest-glob=README.md", "--import-mode=importlib"]
|
|
92
|
+
doctest_optionflags = ["NORMALIZE_WHITESPACE", "ELLIPSIS"]
|
|
93
|
+
filterwarnings = ["error"]
|
|
94
|
+
|
|
95
|
+
[tool.coverage.run]
|
|
96
|
+
source = ["pytrendclust"]
|
|
97
|
+
branch = true
|
|
98
|
+
|
|
99
|
+
[tool.coverage.report]
|
|
100
|
+
show_missing = true
|
|
101
|
+
skip_covered = true
|
|
102
|
+
exclude_also = ["if TYPE_CHECKING:"]
|
|
103
|
+
|
|
104
|
+
[tool.mypy]
|
|
105
|
+
python_version = "3.11"
|
|
106
|
+
files = ["src/pytrendclust"]
|
|
107
|
+
strict = true
|
|
108
|
+
warn_unused_ignores = true
|
|
109
|
+
|
|
110
|
+
[[tool.mypy.overrides]]
|
|
111
|
+
module = ["pandas", "pandas.*"]
|
|
112
|
+
ignore_missing_imports = true
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Trend clustering of time series with an adaptive number of clusters.
|
|
2
|
+
|
|
3
|
+
Splits a series into consecutive rising, falling and flat runs, choosing the
|
|
4
|
+
number of change points by a penalised least-squares criterion whose penalty
|
|
5
|
+
grows exponentially with every extra change point.
|
|
6
|
+
|
|
7
|
+
>>> from pytrendclust import cluster_trends
|
|
8
|
+
>>> result = cluster_trends([10, 9, 8, 7, 6, 5, 4, 5, 6, 7])
|
|
9
|
+
>>> result.last_segment.trend.value, result.indices().tolist()
|
|
10
|
+
('up', [7, 8, 9])
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from pytrendclust.clusterer import TrendClusterer, cluster_trends
|
|
14
|
+
from pytrendclust.penalty import (
|
|
15
|
+
PenaltyFunc,
|
|
16
|
+
default_scale,
|
|
17
|
+
estimate_noise_std,
|
|
18
|
+
exponential_penalty,
|
|
19
|
+
linear_penalty,
|
|
20
|
+
)
|
|
21
|
+
from pytrendclust.result import CriterionRow, Segment, Trend, TrendClusteringResult
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"CriterionRow",
|
|
25
|
+
"PenaltyFunc",
|
|
26
|
+
"Segment",
|
|
27
|
+
"Trend",
|
|
28
|
+
"TrendClusterer",
|
|
29
|
+
"TrendClusteringResult",
|
|
30
|
+
"cluster_trends",
|
|
31
|
+
"default_scale",
|
|
32
|
+
"estimate_noise_std",
|
|
33
|
+
"exponential_penalty",
|
|
34
|
+
"linear_penalty",
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Least-squares cost of fitting a straight line to a contiguous run of points.
|
|
2
|
+
|
|
3
|
+
Prefix sums make the residual sum of squares (SSE) of any run ``[start, stop)``
|
|
4
|
+
an O(1) computation, which is what keeps the dynamic programme in
|
|
5
|
+
:mod:`pytrendclust._dp` affordable.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
from numpy.typing import NDArray
|
|
14
|
+
|
|
15
|
+
FloatArray = NDArray[np.float64]
|
|
16
|
+
IntArray = NDArray[np.int64]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True, slots=True)
|
|
20
|
+
class LineFit:
|
|
21
|
+
"""Ordinary least-squares line ``y = intercept + slope * t`` over one run."""
|
|
22
|
+
|
|
23
|
+
slope: float
|
|
24
|
+
intercept: float
|
|
25
|
+
sse: float
|
|
26
|
+
#: Centred sum of squares of ``t``; zero when the run has a single point.
|
|
27
|
+
stt: float
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def fit_line(t: FloatArray, y: FloatArray) -> LineFit:
|
|
31
|
+
"""Fit a straight line to ``(t, y)`` by ordinary least squares."""
|
|
32
|
+
t_mean = float(t.mean())
|
|
33
|
+
y_mean = float(y.mean())
|
|
34
|
+
dt = t - t_mean
|
|
35
|
+
dy = y - y_mean
|
|
36
|
+
stt = float(dt @ dt)
|
|
37
|
+
slope = float(dt @ dy) / stt if stt > 0.0 else 0.0
|
|
38
|
+
intercept = y_mean - slope * t_mean
|
|
39
|
+
residual = dy - slope * dt
|
|
40
|
+
return LineFit(slope=slope, intercept=intercept, sse=float(residual @ residual), stt=stt)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class LinearCost:
|
|
44
|
+
"""SSE of the best straight line on any run ``[start, stop)`` of a series."""
|
|
45
|
+
|
|
46
|
+
def __init__(self, t: FloatArray, y: FloatArray) -> None:
|
|
47
|
+
# Centring both axes keeps the prefix-sum differences well conditioned.
|
|
48
|
+
tc = t - t.mean()
|
|
49
|
+
yc = y - y.mean()
|
|
50
|
+
zero = np.zeros(1)
|
|
51
|
+
self._n = np.arange(len(y) + 1, dtype=np.float64)
|
|
52
|
+
self._st = np.concatenate([zero, np.cumsum(tc)])
|
|
53
|
+
self._sy = np.concatenate([zero, np.cumsum(yc)])
|
|
54
|
+
self._stt = np.concatenate([zero, np.cumsum(tc * tc)])
|
|
55
|
+
self._sty = np.concatenate([zero, np.cumsum(tc * yc)])
|
|
56
|
+
self._syy = np.concatenate([zero, np.cumsum(yc * yc)])
|
|
57
|
+
#: Total sum of squares of the series around its mean — the cost scale.
|
|
58
|
+
self.total_ss = float(self._syy[-1])
|
|
59
|
+
|
|
60
|
+
def sse(self, starts: IntArray, stop: int) -> FloatArray:
|
|
61
|
+
"""SSE of the runs ``[s, stop)`` for every ``s`` in ``starts`` (vectorised)."""
|
|
62
|
+
n = self._n[stop] - self._n[starts]
|
|
63
|
+
st = self._st[stop] - self._st[starts]
|
|
64
|
+
sy = self._sy[stop] - self._sy[starts]
|
|
65
|
+
ctt = (self._stt[stop] - self._stt[starts]) - st * st / n
|
|
66
|
+
cty = (self._sty[stop] - self._sty[starts]) - st * sy / n
|
|
67
|
+
cyy = (self._syy[stop] - self._syy[starts]) - sy * sy / n
|
|
68
|
+
explained = np.divide(cty * cty, ctt, out=np.zeros_like(ctt), where=ctt > 0.0)
|
|
69
|
+
sse: FloatArray = np.maximum(cyy - explained, 0.0)
|
|
70
|
+
return sse
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""Exact penalised segmentation by dynamic programming over the number of segments.
|
|
2
|
+
|
|
3
|
+
For every number of change points ``K`` the programme finds the partition of the
|
|
4
|
+
series into ``K + 1`` runs with the smallest total SSE (the "segment
|
|
5
|
+
neighbourhood" recursion). The penalty may be any non-decreasing function of
|
|
6
|
+
``K`` — in particular an exponential one, which an additive per-segment
|
|
7
|
+
recursion such as PELT cannot express.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from collections.abc import Callable
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
|
|
15
|
+
import numpy as np
|
|
16
|
+
|
|
17
|
+
from pytrendclust._cost import IntArray, LinearCost
|
|
18
|
+
from pytrendclust.result import CriterionRow
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True, slots=True)
|
|
22
|
+
class Segmentation:
|
|
23
|
+
change_points: tuple[int, ...]
|
|
24
|
+
criterion: tuple[CriterionRow, ...]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def segment(
|
|
28
|
+
cost: LinearCost,
|
|
29
|
+
n: int,
|
|
30
|
+
min_size: int,
|
|
31
|
+
max_change_points: int,
|
|
32
|
+
penalty: Callable[[int], float],
|
|
33
|
+
) -> Segmentation:
|
|
34
|
+
"""Return the change points minimising ``SSE(K) + penalty(K)``.
|
|
35
|
+
|
|
36
|
+
Change points are indices ``c`` such that a new segment starts at ``c``.
|
|
37
|
+
The search over ``K`` stops early once ``penalty(K)`` alone exceeds the best
|
|
38
|
+
objective seen so far: SSE is non-negative and the penalty non-decreasing,
|
|
39
|
+
so no larger ``K`` can win.
|
|
40
|
+
"""
|
|
41
|
+
inf = np.inf
|
|
42
|
+
all_stops = np.arange(n + 1)
|
|
43
|
+
# best[j]: minimal SSE of splitting the prefix [0, j) into k + 1 segments.
|
|
44
|
+
best = np.full(n + 1, inf)
|
|
45
|
+
best[min_size:] = [cost.sse(np.array([0]), j)[0] for j in all_stops[min_size:]]
|
|
46
|
+
back: list[IntArray] = []
|
|
47
|
+
tie_tol = 1e-10 * cost.total_ss
|
|
48
|
+
|
|
49
|
+
criterion = [CriterionRow(0, float(best[n]), penalty(0))]
|
|
50
|
+
best_k, best_obj = 0, criterion[0].objective
|
|
51
|
+
|
|
52
|
+
for k in range(1, max_change_points + 1):
|
|
53
|
+
pen = penalty(k)
|
|
54
|
+
if pen >= best_obj:
|
|
55
|
+
break
|
|
56
|
+
new = np.full(n + 1, inf)
|
|
57
|
+
arg = np.zeros(n + 1, dtype=np.int64)
|
|
58
|
+
for j in range((k + 1) * min_size, n + 1):
|
|
59
|
+
starts = np.arange(k * min_size, j - min_size + 1)
|
|
60
|
+
total = best[starts] + cost.sse(starts, j)
|
|
61
|
+
# On ties take the latest start: a point lying on both lines (the
|
|
62
|
+
# vertex of a V) stays with the older regime, keeping the newest
|
|
63
|
+
# segment free of ambiguous points.
|
|
64
|
+
i = int(np.flatnonzero(total <= total.min() + tie_tol)[-1])
|
|
65
|
+
new[j] = total[i]
|
|
66
|
+
arg[j] = starts[i]
|
|
67
|
+
best = new
|
|
68
|
+
back.append(arg)
|
|
69
|
+
point = CriterionRow(k, float(best[n]), pen)
|
|
70
|
+
criterion.append(point)
|
|
71
|
+
# Strict improvement beyond rounding noise: ties go to fewer change points.
|
|
72
|
+
if point.objective < best_obj - tie_tol:
|
|
73
|
+
best_k, best_obj = k, point.objective
|
|
74
|
+
|
|
75
|
+
change_points: list[int] = []
|
|
76
|
+
j = n
|
|
77
|
+
for k in range(best_k, 0, -1):
|
|
78
|
+
j = int(back[k - 1][j])
|
|
79
|
+
change_points.append(j)
|
|
80
|
+
return Segmentation(tuple(reversed(change_points)), tuple(criterion))
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
"""Trend clustering of a time series with an adaptive number of clusters."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import itertools
|
|
6
|
+
import math
|
|
7
|
+
from typing import Any, Literal
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
from numpy.typing import ArrayLike, NDArray
|
|
11
|
+
|
|
12
|
+
from pytrendclust._cost import LinearCost, fit_line
|
|
13
|
+
from pytrendclust._dp import segment
|
|
14
|
+
from pytrendclust.penalty import (
|
|
15
|
+
PenaltyFunc,
|
|
16
|
+
default_scale,
|
|
17
|
+
estimate_noise_std,
|
|
18
|
+
exponential_penalty,
|
|
19
|
+
linear_penalty,
|
|
20
|
+
)
|
|
21
|
+
from pytrendclust.result import Segment, Trend, TrendClusteringResult
|
|
22
|
+
|
|
23
|
+
PenaltyKind = Literal["exponential", "linear", "none"]
|
|
24
|
+
|
|
25
|
+
# Floor of the noise scale relative to the series' spread: on a perfectly clean
|
|
26
|
+
# series the noise estimate is 0, and a zero penalty would make every
|
|
27
|
+
# segmentation with zero SSE a tie.
|
|
28
|
+
_NOISE_FLOOR = 1e-6
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class TrendClusterer:
|
|
32
|
+
"""Split a time series into consecutive trend clusters (rising, falling, flat).
|
|
33
|
+
|
|
34
|
+
Each cluster is a contiguous run of points with its own least-squares line.
|
|
35
|
+
The number of clusters is not given in advance: the clusterer minimises
|
|
36
|
+
|
|
37
|
+
SSE(K) + sigma**2 * penalty(K)
|
|
38
|
+
|
|
39
|
+
over the number of change points ``K``, where ``SSE(K)`` is the best total
|
|
40
|
+
squared error of a piecewise-linear fit with ``K`` change points and
|
|
41
|
+
``sigma`` is the noise level. The default exponential penalty makes every
|
|
42
|
+
extra change point dearer than the previous one, so noise such as a
|
|
43
|
+
"picket fence" (up, down, up, down, ...) is recognised as a single flat
|
|
44
|
+
cluster instead of ten tiny trends.
|
|
45
|
+
|
|
46
|
+
Parameters
|
|
47
|
+
----------
|
|
48
|
+
penalty:
|
|
49
|
+
``"exponential"`` (default) — the ``K``-th change point costs
|
|
50
|
+
``penalty_scale * exp(penalty_growth * (K - 1))`` noise variances;
|
|
51
|
+
``"linear"`` — every change point costs ``penalty_scale``;
|
|
52
|
+
``"none"`` — no regularisation (only ``max_change_points`` limits K);
|
|
53
|
+
or a callable ``f(n_change_points, n_samples, noise_variance)`` returning
|
|
54
|
+
the penalty in the units of SSE, non-decreasing in ``n_change_points``.
|
|
55
|
+
penalty_scale:
|
|
56
|
+
Price of the first change point in noise variances. Defaults to
|
|
57
|
+
``5 * ln(n)`` (see :func:`~pytrendclust.default_scale`). Larger means
|
|
58
|
+
fewer change points.
|
|
59
|
+
penalty_growth:
|
|
60
|
+
Exponential growth rate of the price of every next change point.
|
|
61
|
+
min_size:
|
|
62
|
+
Minimum number of points in a cluster; at least 2 (a line needs two
|
|
63
|
+
points, and three are needed before a line can be wrong).
|
|
64
|
+
max_change_points:
|
|
65
|
+
Hard upper bound on ``K``; ``None`` means only the penalty limits it.
|
|
66
|
+
noise_std:
|
|
67
|
+
Noise standard deviation in the units of the series. ``None`` estimates
|
|
68
|
+
it robustly from second differences of the series.
|
|
69
|
+
flat_threshold:
|
|
70
|
+
A cluster is ``flat`` when ``|slope| / standard_error(slope)`` is below
|
|
71
|
+
this value (2 is roughly the 95 % significance level).
|
|
72
|
+
|
|
73
|
+
Examples
|
|
74
|
+
--------
|
|
75
|
+
>>> y = [10, 9, 8, 7, 6, 5, 4, 5, 6, 7]
|
|
76
|
+
>>> result = TrendClusterer().fit(y).result_
|
|
77
|
+
>>> result.change_points
|
|
78
|
+
[7]
|
|
79
|
+
>>> [s.trend.value for s in result.segments]
|
|
80
|
+
['down', 'up']
|
|
81
|
+
>>> result.select()[1]
|
|
82
|
+
array([5., 6., 7.])
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
def __init__(
|
|
86
|
+
self,
|
|
87
|
+
*,
|
|
88
|
+
penalty: PenaltyKind | PenaltyFunc = "exponential",
|
|
89
|
+
penalty_scale: float | None = None,
|
|
90
|
+
penalty_growth: float = 0.5,
|
|
91
|
+
min_size: int = 3,
|
|
92
|
+
max_change_points: int | None = None,
|
|
93
|
+
noise_std: float | None = None,
|
|
94
|
+
flat_threshold: float = 2.0,
|
|
95
|
+
) -> None:
|
|
96
|
+
self.penalty = penalty
|
|
97
|
+
self.penalty_scale = penalty_scale
|
|
98
|
+
self.penalty_growth = penalty_growth
|
|
99
|
+
self.min_size = min_size
|
|
100
|
+
self.max_change_points = max_change_points
|
|
101
|
+
self.noise_std = noise_std
|
|
102
|
+
self.flat_threshold = flat_threshold
|
|
103
|
+
self._validate()
|
|
104
|
+
self.result_: TrendClusteringResult | None = None
|
|
105
|
+
|
|
106
|
+
# -- parameters --------------------------------------------------------
|
|
107
|
+
|
|
108
|
+
_PARAM_NAMES = (
|
|
109
|
+
"penalty",
|
|
110
|
+
"penalty_scale",
|
|
111
|
+
"penalty_growth",
|
|
112
|
+
"min_size",
|
|
113
|
+
"max_change_points",
|
|
114
|
+
"noise_std",
|
|
115
|
+
"flat_threshold",
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
def get_params(self) -> dict[str, Any]:
|
|
119
|
+
"""Current parameters, as accepted by the constructor and :meth:`set_params`."""
|
|
120
|
+
return {name: getattr(self, name) for name in self._PARAM_NAMES}
|
|
121
|
+
|
|
122
|
+
def set_params(self, **params: Any) -> TrendClusterer:
|
|
123
|
+
"""Change parameters in place; returns ``self`` for chaining."""
|
|
124
|
+
unknown = set(params) - set(self._PARAM_NAMES)
|
|
125
|
+
if unknown:
|
|
126
|
+
msg = f"unknown parameter(s): {', '.join(sorted(unknown))}"
|
|
127
|
+
raise ValueError(msg)
|
|
128
|
+
old = self.get_params()
|
|
129
|
+
for name, value in params.items():
|
|
130
|
+
setattr(self, name, value)
|
|
131
|
+
try:
|
|
132
|
+
self._validate()
|
|
133
|
+
except (TypeError, ValueError):
|
|
134
|
+
for name, value in old.items():
|
|
135
|
+
setattr(self, name, value)
|
|
136
|
+
raise
|
|
137
|
+
return self
|
|
138
|
+
|
|
139
|
+
def __repr__(self) -> str:
|
|
140
|
+
args = ", ".join(f"{k}={v!r}" for k, v in self.get_params().items())
|
|
141
|
+
return f"{type(self).__name__}({args})"
|
|
142
|
+
|
|
143
|
+
def _validate(self) -> None:
|
|
144
|
+
if not callable(self.penalty) and self.penalty not in ("exponential", "linear", "none"):
|
|
145
|
+
msg = (
|
|
146
|
+
"penalty must be 'exponential', 'linear', 'none' or a callable, "
|
|
147
|
+
f"got {self.penalty!r}"
|
|
148
|
+
)
|
|
149
|
+
raise ValueError(msg)
|
|
150
|
+
if self.penalty_scale is not None and not (
|
|
151
|
+
math.isfinite(self.penalty_scale) and self.penalty_scale >= 0
|
|
152
|
+
):
|
|
153
|
+
msg = f"penalty_scale must be a finite number >= 0, got {self.penalty_scale!r}"
|
|
154
|
+
raise ValueError(msg)
|
|
155
|
+
if not (math.isfinite(self.penalty_growth) and self.penalty_growth >= 0):
|
|
156
|
+
msg = f"penalty_growth must be a finite number >= 0, got {self.penalty_growth!r}"
|
|
157
|
+
raise ValueError(msg)
|
|
158
|
+
if isinstance(self.min_size, bool) or not isinstance(self.min_size, int):
|
|
159
|
+
msg = f"min_size must be an int, got {self.min_size!r}"
|
|
160
|
+
raise TypeError(msg)
|
|
161
|
+
if self.min_size < 2:
|
|
162
|
+
msg = f"min_size must be >= 2, got {self.min_size}"
|
|
163
|
+
raise ValueError(msg)
|
|
164
|
+
if self.max_change_points is not None and (
|
|
165
|
+
isinstance(self.max_change_points, bool)
|
|
166
|
+
or not isinstance(self.max_change_points, int)
|
|
167
|
+
or self.max_change_points < 0
|
|
168
|
+
):
|
|
169
|
+
msg = f"max_change_points must be None or an int >= 0, got {self.max_change_points!r}"
|
|
170
|
+
raise ValueError(msg)
|
|
171
|
+
if self.noise_std is not None and not (
|
|
172
|
+
math.isfinite(self.noise_std) and self.noise_std >= 0
|
|
173
|
+
):
|
|
174
|
+
msg = f"noise_std must be None or a finite number >= 0, got {self.noise_std!r}"
|
|
175
|
+
raise ValueError(msg)
|
|
176
|
+
if not (math.isfinite(self.flat_threshold) and self.flat_threshold >= 0):
|
|
177
|
+
msg = f"flat_threshold must be a finite number >= 0, got {self.flat_threshold!r}"
|
|
178
|
+
raise ValueError(msg)
|
|
179
|
+
|
|
180
|
+
# -- fitting -----------------------------------------------------------
|
|
181
|
+
|
|
182
|
+
def fit(self, y: ArrayLike, t: ArrayLike | None = None) -> TrendClusterer:
|
|
183
|
+
"""Cluster the series ``y`` observed at times ``t`` (default ``0, 1, 2, ...``).
|
|
184
|
+
|
|
185
|
+
The result is stored in :attr:`result_`; returns ``self``.
|
|
186
|
+
"""
|
|
187
|
+
y_arr, t_arr = _as_series(y, t)
|
|
188
|
+
n = len(y_arr)
|
|
189
|
+
|
|
190
|
+
sigma = self._noise_std(y_arr)
|
|
191
|
+
sigma2 = sigma * sigma
|
|
192
|
+
penalty = self._penalty_in_sse_units(n, sigma2)
|
|
193
|
+
largest_k = n // self.min_size - 1
|
|
194
|
+
if self.max_change_points is not None:
|
|
195
|
+
largest_k = min(largest_k, self.max_change_points)
|
|
196
|
+
|
|
197
|
+
seg = segment(LinearCost(t_arr, y_arr), n, self.min_size, max(largest_k, 0), penalty)
|
|
198
|
+
|
|
199
|
+
bounds = [0, *seg.change_points, n]
|
|
200
|
+
segments = tuple(
|
|
201
|
+
self._describe(index=i, start=start, stop=stop, t=t_arr, y=y_arr, sigma=sigma)
|
|
202
|
+
for i, (start, stop) in enumerate(itertools.pairwise(bounds))
|
|
203
|
+
)
|
|
204
|
+
self.result_ = TrendClusteringResult(
|
|
205
|
+
t=t_arr,
|
|
206
|
+
y=y_arr,
|
|
207
|
+
segments=segments,
|
|
208
|
+
noise_std=sigma,
|
|
209
|
+
criterion=seg.criterion,
|
|
210
|
+
params=self._exportable_params(),
|
|
211
|
+
)
|
|
212
|
+
return self
|
|
213
|
+
|
|
214
|
+
def fit_predict(self, y: ArrayLike, t: ArrayLike | None = None) -> NDArray[np.int64]:
|
|
215
|
+
"""Fit and return the cluster (segment) index of every point."""
|
|
216
|
+
result = self.fit(y, t).result_
|
|
217
|
+
assert result is not None # noqa: S101 - set by fit()
|
|
218
|
+
return result.labels
|
|
219
|
+
|
|
220
|
+
def _noise_std(self, y: NDArray[np.float64]) -> float:
|
|
221
|
+
sigma = estimate_noise_std(y) if self.noise_std is None else float(self.noise_std)
|
|
222
|
+
spread = float(np.ptp(y)) if len(y) else 0.0
|
|
223
|
+
return max(sigma, _NOISE_FLOOR * spread)
|
|
224
|
+
|
|
225
|
+
def _penalty_in_sse_units(self, n: int, sigma2: float) -> Any:
|
|
226
|
+
if callable(self.penalty):
|
|
227
|
+
func = self.penalty
|
|
228
|
+
return lambda k: float(func(k, n, sigma2))
|
|
229
|
+
if self.penalty == "none":
|
|
230
|
+
return lambda k: 0.0
|
|
231
|
+
scale = default_scale(n) if self.penalty_scale is None else self.penalty_scale
|
|
232
|
+
if self.penalty == "linear":
|
|
233
|
+
return lambda k: sigma2 * linear_penalty(k, scale)
|
|
234
|
+
growth = self.penalty_growth
|
|
235
|
+
return lambda k: sigma2 * exponential_penalty(k, scale, growth)
|
|
236
|
+
|
|
237
|
+
def _describe(
|
|
238
|
+
self,
|
|
239
|
+
*,
|
|
240
|
+
index: int,
|
|
241
|
+
start: int,
|
|
242
|
+
stop: int,
|
|
243
|
+
t: NDArray[np.float64],
|
|
244
|
+
y: NDArray[np.float64],
|
|
245
|
+
sigma: float,
|
|
246
|
+
) -> Segment:
|
|
247
|
+
line = fit_line(t[start:stop], y[start:stop])
|
|
248
|
+
if line.stt == 0.0 or line.slope == 0.0:
|
|
249
|
+
t_stat = 0.0
|
|
250
|
+
elif sigma == 0.0:
|
|
251
|
+
t_stat = math.inf
|
|
252
|
+
else:
|
|
253
|
+
t_stat = abs(line.slope) * math.sqrt(line.stt) / sigma
|
|
254
|
+
if t_stat < self.flat_threshold:
|
|
255
|
+
trend = Trend.FLAT
|
|
256
|
+
else:
|
|
257
|
+
trend = Trend.UP if line.slope > 0 else Trend.DOWN
|
|
258
|
+
return Segment(
|
|
259
|
+
index=index,
|
|
260
|
+
start=start,
|
|
261
|
+
stop=stop,
|
|
262
|
+
trend=trend,
|
|
263
|
+
slope=line.slope,
|
|
264
|
+
intercept=line.intercept,
|
|
265
|
+
sse=line.sse,
|
|
266
|
+
t_statistic=t_stat,
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
def _exportable_params(self) -> dict[str, Any]:
|
|
270
|
+
params = self.get_params()
|
|
271
|
+
if callable(params["penalty"]):
|
|
272
|
+
params["penalty"] = getattr(params["penalty"], "__name__", repr(params["penalty"]))
|
|
273
|
+
return params
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def cluster_trends(
|
|
277
|
+
y: ArrayLike, t: ArrayLike | None = None, **params: Any
|
|
278
|
+
) -> TrendClusteringResult:
|
|
279
|
+
"""One-call shortcut: ``TrendClusterer(**params).fit(y, t).result_``."""
|
|
280
|
+
result = TrendClusterer(**params).fit(y, t).result_
|
|
281
|
+
assert result is not None # noqa: S101 - set by fit()
|
|
282
|
+
return result
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _as_series(
|
|
286
|
+
y: ArrayLike, t: ArrayLike | None
|
|
287
|
+
) -> tuple[NDArray[np.float64], NDArray[np.float64]]:
|
|
288
|
+
y_arr = np.array(y, dtype=np.float64)
|
|
289
|
+
if y_arr.ndim != 1:
|
|
290
|
+
msg = f"y must be one-dimensional, got shape {y_arr.shape}"
|
|
291
|
+
raise ValueError(msg)
|
|
292
|
+
if len(y_arr) == 0:
|
|
293
|
+
msg = "y is empty"
|
|
294
|
+
raise ValueError(msg)
|
|
295
|
+
if not np.all(np.isfinite(y_arr)):
|
|
296
|
+
msg = "y contains NaN or infinite values"
|
|
297
|
+
raise ValueError(msg)
|
|
298
|
+
if t is None:
|
|
299
|
+
return y_arr, np.arange(len(y_arr), dtype=np.float64)
|
|
300
|
+
t_arr = np.array(t, dtype=np.float64)
|
|
301
|
+
if t_arr.shape != y_arr.shape:
|
|
302
|
+
msg = f"t and y must have the same shape, got {t_arr.shape} and {y_arr.shape}"
|
|
303
|
+
raise ValueError(msg)
|
|
304
|
+
if not np.all(np.isfinite(t_arr)):
|
|
305
|
+
msg = "t contains NaN or infinite values"
|
|
306
|
+
raise ValueError(msg)
|
|
307
|
+
if np.any(np.diff(t_arr) <= 0):
|
|
308
|
+
msg = "t must be strictly increasing"
|
|
309
|
+
raise ValueError(msg)
|
|
310
|
+
return y_arr, t_arr
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Penalties on the number of change points and the noise estimate they are scaled by.
|
|
2
|
+
|
|
3
|
+
A penalty is measured in units of the noise variance ``sigma**2``: SSE grows with
|
|
4
|
+
the square of the series' units, so a penalty in the same units makes the
|
|
5
|
+
trade-off independent of whether memory is counted in bytes or gigabytes.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import math
|
|
11
|
+
from collections.abc import Callable
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
from numpy.typing import NDArray
|
|
15
|
+
|
|
16
|
+
#: ``penalty(n_change_points, n_samples, noise_variance) -> float``.
|
|
17
|
+
#: A custom penalty must be non-decreasing in ``n_change_points``.
|
|
18
|
+
PenaltyFunc = Callable[[int, int, float], float]
|
|
19
|
+
|
|
20
|
+
# Consistency constant turning a median absolute deviation into a standard
|
|
21
|
+
# deviation for Gaussian noise.
|
|
22
|
+
_MAD_TO_STD = 1.482602218505602
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def default_scale(n_samples: int) -> float:
|
|
26
|
+
"""Default price of the first change point, in noise variances: ``5 * ln(n)``.
|
|
27
|
+
|
|
28
|
+
Every extra segment costs three parameters — slope, intercept and the
|
|
29
|
+
position of its start — for which BIC would charge ``3 * ln(n)``. BIC is
|
|
30
|
+
too liberal here: the start position is chosen as the best of ``n``
|
|
31
|
+
candidates, so on pure noise it finds a spurious short trend in ~7 % of
|
|
32
|
+
100-point series, against ~0.5 % with ``5 * ln(n)`` at the same detection
|
|
33
|
+
rate of real breaks.
|
|
34
|
+
"""
|
|
35
|
+
return 5.0 * math.log(max(n_samples, 2))
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def exponential_penalty(n_change_points: int, scale: float, growth: float) -> float:
|
|
39
|
+
"""``scale * (exp(growth * K) - 1) / (exp(growth) - 1)``, in noise variances.
|
|
40
|
+
|
|
41
|
+
The first change point costs exactly ``scale``; every next one costs
|
|
42
|
+
``exp(growth)`` times more than the previous, so the model grows ever more
|
|
43
|
+
reluctant to add change points. ``growth = 0`` degenerates to the linear
|
|
44
|
+
penalty ``scale * K``.
|
|
45
|
+
"""
|
|
46
|
+
if n_change_points == 0:
|
|
47
|
+
return 0.0
|
|
48
|
+
if growth == 0.0:
|
|
49
|
+
return scale * n_change_points
|
|
50
|
+
return scale * math.expm1(growth * n_change_points) / math.expm1(growth)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def linear_penalty(n_change_points: int, scale: float) -> float:
|
|
54
|
+
"""Classic constant price per change point: ``scale * K``, in noise variances."""
|
|
55
|
+
return scale * n_change_points
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def estimate_noise_std(y: NDArray[np.float64]) -> float:
|
|
59
|
+
"""Robust estimate of the noise standard deviation around a piecewise-linear trend.
|
|
60
|
+
|
|
61
|
+
Second differences cancel any linear trend, leaving ``eps[i] - 2 eps[i+1] +
|
|
62
|
+
eps[i+2]`` with variance ``6 sigma**2``; a few kinks between segments are
|
|
63
|
+
outliers that the median absolute deviation ignores. Falls back to the plain
|
|
64
|
+
standard deviation of the second differences when more than half of them
|
|
65
|
+
coincide (for example, on a quantised series), and returns 0 for series too
|
|
66
|
+
short to tell.
|
|
67
|
+
"""
|
|
68
|
+
if len(y) < 4:
|
|
69
|
+
return 0.0
|
|
70
|
+
d2 = np.diff(y, n=2)
|
|
71
|
+
mad = float(np.median(np.abs(d2 - np.median(d2))))
|
|
72
|
+
if mad > 0.0:
|
|
73
|
+
return _MAD_TO_STD * mad / math.sqrt(6.0)
|
|
74
|
+
return float(d2.std()) / math.sqrt(6.0)
|
|
File without changes
|
|
@@ -0,0 +1,303 @@
|
|
|
1
|
+
"""Result of a trend clustering and its export to dict, JSON, CSV and pandas."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import io
|
|
7
|
+
import json
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
from enum import StrEnum
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import TYPE_CHECKING, Any
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
from numpy.typing import NDArray
|
|
15
|
+
|
|
16
|
+
if TYPE_CHECKING:
|
|
17
|
+
import pandas as pd
|
|
18
|
+
|
|
19
|
+
FORMAT_VERSION = 1
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class Trend(StrEnum):
|
|
23
|
+
"""Direction of a segment's linear trend."""
|
|
24
|
+
|
|
25
|
+
UP = "up"
|
|
26
|
+
DOWN = "down"
|
|
27
|
+
FLAT = "flat"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True, slots=True)
|
|
31
|
+
class Segment:
|
|
32
|
+
"""One trend cluster: a contiguous run of points ``[start, stop)`` and its line."""
|
|
33
|
+
|
|
34
|
+
index: int
|
|
35
|
+
start: int
|
|
36
|
+
stop: int
|
|
37
|
+
trend: Trend
|
|
38
|
+
slope: float
|
|
39
|
+
intercept: float
|
|
40
|
+
sse: float
|
|
41
|
+
#: ``|slope| / standard_error(slope)``; ``inf`` for a noise-free series.
|
|
42
|
+
t_statistic: float
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def n_points(self) -> int:
|
|
46
|
+
return self.stop - self.start
|
|
47
|
+
|
|
48
|
+
def predict(self, t: Any) -> NDArray[np.float64]:
|
|
49
|
+
"""Value of the segment's line at time(s) ``t`` — also beyond the segment."""
|
|
50
|
+
return self.intercept + self.slope * np.asarray(t, dtype=np.float64)
|
|
51
|
+
|
|
52
|
+
def to_dict(self) -> dict[str, Any]:
|
|
53
|
+
return {
|
|
54
|
+
"index": self.index,
|
|
55
|
+
"start": self.start,
|
|
56
|
+
"stop": self.stop,
|
|
57
|
+
"n_points": self.n_points,
|
|
58
|
+
"trend": self.trend.value,
|
|
59
|
+
"slope": self.slope,
|
|
60
|
+
"intercept": self.intercept,
|
|
61
|
+
"sse": self.sse,
|
|
62
|
+
"t_statistic": _json_float(self.t_statistic),
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
@classmethod
|
|
66
|
+
def from_dict(cls, data: dict[str, Any]) -> Segment:
|
|
67
|
+
return cls(
|
|
68
|
+
index=int(data["index"]),
|
|
69
|
+
start=int(data["start"]),
|
|
70
|
+
stop=int(data["stop"]),
|
|
71
|
+
trend=Trend(data["trend"]),
|
|
72
|
+
slope=float(data["slope"]),
|
|
73
|
+
intercept=float(data["intercept"]),
|
|
74
|
+
sse=float(data["sse"]),
|
|
75
|
+
t_statistic=float(data["t_statistic"]),
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
@dataclass(frozen=True, slots=True)
|
|
80
|
+
class CriterionRow:
|
|
81
|
+
"""Penalised criterion for one candidate number of change points."""
|
|
82
|
+
|
|
83
|
+
n_change_points: int
|
|
84
|
+
sse: float
|
|
85
|
+
penalty: float
|
|
86
|
+
|
|
87
|
+
@property
|
|
88
|
+
def objective(self) -> float:
|
|
89
|
+
return self.sse + self.penalty
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@dataclass(frozen=True)
|
|
93
|
+
class TrendClusteringResult:
|
|
94
|
+
"""Which point belongs to which trend cluster, plus everything needed to audit it.
|
|
95
|
+
|
|
96
|
+
``segments`` are ordered in time and cover the series without gaps;
|
|
97
|
+
``segments[-1]`` is the most recent regime — the one to train a regression on.
|
|
98
|
+
"""
|
|
99
|
+
|
|
100
|
+
t: NDArray[np.float64]
|
|
101
|
+
y: NDArray[np.float64]
|
|
102
|
+
segments: tuple[Segment, ...]
|
|
103
|
+
noise_std: float
|
|
104
|
+
#: Criterion for every number of change points the search evaluated.
|
|
105
|
+
criterion: tuple[CriterionRow, ...]
|
|
106
|
+
#: Parameters of the :class:`~pytrendclust.TrendClusterer` that produced this result.
|
|
107
|
+
params: dict[str, Any] = field(default_factory=dict)
|
|
108
|
+
|
|
109
|
+
# -- structure ---------------------------------------------------------
|
|
110
|
+
|
|
111
|
+
@property
|
|
112
|
+
def n_points(self) -> int:
|
|
113
|
+
return len(self.y)
|
|
114
|
+
|
|
115
|
+
@property
|
|
116
|
+
def n_segments(self) -> int:
|
|
117
|
+
return len(self.segments)
|
|
118
|
+
|
|
119
|
+
@property
|
|
120
|
+
def n_change_points(self) -> int:
|
|
121
|
+
return len(self.segments) - 1
|
|
122
|
+
|
|
123
|
+
@property
|
|
124
|
+
def change_points(self) -> list[int]:
|
|
125
|
+
"""Indices where a new segment starts (the first segment's start, 0, excluded)."""
|
|
126
|
+
return [s.start for s in self.segments[1:]]
|
|
127
|
+
|
|
128
|
+
@property
|
|
129
|
+
def labels(self) -> NDArray[np.int64]:
|
|
130
|
+
"""Segment index of every point."""
|
|
131
|
+
return np.repeat(
|
|
132
|
+
np.arange(self.n_segments, dtype=np.int64), [s.n_points for s in self.segments]
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
@property
|
|
136
|
+
def trends(self) -> list[Trend]:
|
|
137
|
+
"""Trend of the segment every point belongs to."""
|
|
138
|
+
return [s.trend for s in self.segments for _ in range(s.n_points)]
|
|
139
|
+
|
|
140
|
+
@property
|
|
141
|
+
def fitted(self) -> NDArray[np.float64]:
|
|
142
|
+
"""Piecewise-linear fit: every point's value on its own segment's line."""
|
|
143
|
+
return np.concatenate([s.predict(self.t[s.start : s.stop]) for s in self.segments])
|
|
144
|
+
|
|
145
|
+
@property
|
|
146
|
+
def last_segment(self) -> Segment:
|
|
147
|
+
"""The most recent regime of the series."""
|
|
148
|
+
return self.segments[-1]
|
|
149
|
+
|
|
150
|
+
@property
|
|
151
|
+
def objective(self) -> float:
|
|
152
|
+
"""Value of the penalised criterion at the chosen segmentation."""
|
|
153
|
+
chosen = self.n_change_points
|
|
154
|
+
return next(r.objective for r in self.criterion if r.n_change_points == chosen)
|
|
155
|
+
|
|
156
|
+
# -- selection ---------------------------------------------------------
|
|
157
|
+
|
|
158
|
+
def indices(self, segment: int = -1) -> NDArray[np.int64]:
|
|
159
|
+
"""Indices of the points of one segment; the last one by default."""
|
|
160
|
+
s = self.segments[segment]
|
|
161
|
+
return np.arange(s.start, s.stop, dtype=np.int64)
|
|
162
|
+
|
|
163
|
+
def select(self, segment: int = -1) -> tuple[NDArray[np.float64], NDArray[np.float64]]:
|
|
164
|
+
"""``(t, y)`` of one segment's points; the last (most recent) one by default."""
|
|
165
|
+
s = self.segments[segment]
|
|
166
|
+
return self.t[s.start : s.stop].copy(), self.y[s.start : s.stop].copy()
|
|
167
|
+
|
|
168
|
+
# -- export ------------------------------------------------------------
|
|
169
|
+
|
|
170
|
+
def to_dict(self, *, include_points: bool = True) -> dict[str, Any]:
|
|
171
|
+
"""Plain-Python representation, round-trippable through :meth:`from_dict`."""
|
|
172
|
+
data: dict[str, Any] = {
|
|
173
|
+
"format_version": FORMAT_VERSION,
|
|
174
|
+
"n_points": self.n_points,
|
|
175
|
+
"n_segments": self.n_segments,
|
|
176
|
+
"n_change_points": self.n_change_points,
|
|
177
|
+
"change_points": self.change_points,
|
|
178
|
+
"noise_std": self.noise_std,
|
|
179
|
+
"objective": self.objective,
|
|
180
|
+
"params": self.params,
|
|
181
|
+
"segments": [s.to_dict() for s in self.segments],
|
|
182
|
+
"criterion": [
|
|
183
|
+
{
|
|
184
|
+
"n_change_points": r.n_change_points,
|
|
185
|
+
"sse": r.sse,
|
|
186
|
+
"penalty": r.penalty,
|
|
187
|
+
"objective": r.objective,
|
|
188
|
+
}
|
|
189
|
+
for r in self.criterion
|
|
190
|
+
],
|
|
191
|
+
}
|
|
192
|
+
if include_points:
|
|
193
|
+
data["points"] = {"t": self.t.tolist(), "y": self.y.tolist()}
|
|
194
|
+
return data
|
|
195
|
+
|
|
196
|
+
@classmethod
|
|
197
|
+
def from_dict(cls, data: dict[str, Any]) -> TrendClusteringResult:
|
|
198
|
+
"""Rebuild a result exported with ``to_dict(include_points=True)``."""
|
|
199
|
+
if data.get("format_version") != FORMAT_VERSION:
|
|
200
|
+
msg = f"unsupported format_version: {data.get('format_version')!r}"
|
|
201
|
+
raise ValueError(msg)
|
|
202
|
+
if "points" not in data:
|
|
203
|
+
msg = "the export has no points; export it with include_points=True"
|
|
204
|
+
raise ValueError(msg)
|
|
205
|
+
return cls(
|
|
206
|
+
t=np.asarray(data["points"]["t"], dtype=np.float64),
|
|
207
|
+
y=np.asarray(data["points"]["y"], dtype=np.float64),
|
|
208
|
+
segments=tuple(Segment.from_dict(s) for s in data["segments"]),
|
|
209
|
+
noise_std=float(data["noise_std"]),
|
|
210
|
+
criterion=tuple(
|
|
211
|
+
CriterionRow(int(r["n_change_points"]), float(r["sse"]), float(r["penalty"]))
|
|
212
|
+
for r in data["criterion"]
|
|
213
|
+
),
|
|
214
|
+
params=dict(data["params"]),
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
def to_json(self, path: str | Path | None = None, *, indent: int | None = 2) -> str:
|
|
218
|
+
"""Serialise to JSON; also write it to ``path`` when given."""
|
|
219
|
+
text = json.dumps(self.to_dict(), ensure_ascii=False, indent=indent)
|
|
220
|
+
if path is not None:
|
|
221
|
+
Path(path).write_text(text, encoding="utf-8")
|
|
222
|
+
return text
|
|
223
|
+
|
|
224
|
+
@classmethod
|
|
225
|
+
def from_json(cls, source: str | Path) -> TrendClusteringResult:
|
|
226
|
+
"""Load a result from a JSON file path or a JSON string."""
|
|
227
|
+
text = str(source)
|
|
228
|
+
if not text.lstrip().startswith("{"):
|
|
229
|
+
text = Path(source).read_text(encoding="utf-8")
|
|
230
|
+
return cls.from_dict(json.loads(text))
|
|
231
|
+
|
|
232
|
+
def point_table(self) -> list[dict[str, Any]]:
|
|
233
|
+
"""One row per point: index, t, y, segment, trend, fitted value, residual."""
|
|
234
|
+
labels = self.labels
|
|
235
|
+
trends = self.trends
|
|
236
|
+
fitted = self.fitted
|
|
237
|
+
return [
|
|
238
|
+
{
|
|
239
|
+
"index": i,
|
|
240
|
+
"t": float(self.t[i]),
|
|
241
|
+
"y": float(self.y[i]),
|
|
242
|
+
"segment": int(labels[i]),
|
|
243
|
+
"trend": trends[i].value,
|
|
244
|
+
"fitted": float(fitted[i]),
|
|
245
|
+
"residual": float(self.y[i] - fitted[i]),
|
|
246
|
+
"is_last_segment": bool(labels[i] == self.n_segments - 1),
|
|
247
|
+
}
|
|
248
|
+
for i in range(self.n_points)
|
|
249
|
+
]
|
|
250
|
+
|
|
251
|
+
def to_csv(self, path: str | Path | None = None) -> str:
|
|
252
|
+
"""Per-point table as CSV; also write it to ``path`` when given."""
|
|
253
|
+
rows = self.point_table()
|
|
254
|
+
buffer = io.StringIO()
|
|
255
|
+
writer = csv.DictWriter(buffer, fieldnames=list(rows[0]), lineterminator="\n")
|
|
256
|
+
writer.writeheader()
|
|
257
|
+
writer.writerows(rows)
|
|
258
|
+
text = buffer.getvalue()
|
|
259
|
+
if path is not None:
|
|
260
|
+
Path(path).write_text(text, encoding="utf-8")
|
|
261
|
+
return text
|
|
262
|
+
|
|
263
|
+
def to_dataframe(self) -> pd.DataFrame:
|
|
264
|
+
"""Per-point table as a :class:`pandas.DataFrame` (needs ``pytrendclust[pandas]``)."""
|
|
265
|
+
try:
|
|
266
|
+
import pandas as pd
|
|
267
|
+
except ImportError as exc:
|
|
268
|
+
msg = "to_dataframe() needs pandas: pip install 'pytrendclust[pandas]'"
|
|
269
|
+
raise ImportError(msg) from exc
|
|
270
|
+
return pd.DataFrame(self.point_table())
|
|
271
|
+
|
|
272
|
+
def export(self, path: str | Path) -> None:
|
|
273
|
+
"""Write the result to ``path``; the format follows the suffix (``.json``/``.csv``)."""
|
|
274
|
+
suffix = Path(path).suffix.lower()
|
|
275
|
+
if suffix == ".json":
|
|
276
|
+
self.to_json(path)
|
|
277
|
+
elif suffix == ".csv":
|
|
278
|
+
self.to_csv(path)
|
|
279
|
+
else:
|
|
280
|
+
msg = f"unsupported export format {suffix!r}; use .json or .csv"
|
|
281
|
+
raise ValueError(msg)
|
|
282
|
+
|
|
283
|
+
def summary(self) -> str:
|
|
284
|
+
"""Human-readable one-line-per-segment description."""
|
|
285
|
+
lines = [
|
|
286
|
+
(
|
|
287
|
+
f"{self.n_points} points, {self.n_change_points} change point(s), "
|
|
288
|
+
f"noise std {self.noise_std:.4g}"
|
|
289
|
+
)
|
|
290
|
+
]
|
|
291
|
+
lines.extend(
|
|
292
|
+
f" #{s.index}: [{s.start}, {s.stop}) {s.n_points:>4} pts "
|
|
293
|
+
f"{s.trend.value:<4} slope {s.slope:+.4g}"
|
|
294
|
+
for s in self.segments
|
|
295
|
+
)
|
|
296
|
+
return "\n".join(lines)
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def _json_float(value: float) -> float | str:
|
|
300
|
+
"""JSON has no infinity; spell it as a string that ``float()`` parses back."""
|
|
301
|
+
if np.isinf(value):
|
|
302
|
+
return "inf" if value > 0 else "-inf"
|
|
303
|
+
return value
|