pytrendclust 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Andrey Ferubko
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,158 @@
1
+ Metadata-Version: 2.4
2
+ Name: pytrendclust
3
+ Version: 0.1.0
4
+ Summary: Trend clustering of time series with an adaptive number of clusters: split a series into rising, falling and flat runs with an exponentially growing penalty on change points
5
+ Keywords: time-series,clustering,trend,change-point,segmentation,piecewise-linear,regression
6
+ Author: Andrey Ferubko
7
+ Author-email: Andrey Ferubko <ferubko1999@yandex.ru>
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: Operating System :: OS Independent
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3 :: Only
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Programming Language :: Python :: 3.14
20
+ Classifier: Topic :: Scientific/Engineering
21
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
22
+ Classifier: Topic :: Scientific/Engineering :: Mathematics
23
+ Classifier: Typing :: Typed
24
+ Requires-Dist: numpy>=1.26
25
+ Requires-Dist: pandas>=2.1 ; extra == 'pandas'
26
+ Requires-Python: >=3.11
27
+ Project-URL: Homepage, https://github.com/89605502155/pytrendclust
28
+ Project-URL: Repository, https://github.com/89605502155/pytrendclust
29
+ Project-URL: Documentation, https://github.com/89605502155/pytrendclust#readme
30
+ Project-URL: Issues, https://github.com/89605502155/pytrendclust/issues
31
+ Provides-Extra: pandas
32
+ Description-Content-Type: text/markdown
33
+
34
+ # pytrendclust
35
+
36
+ [![PyPI](https://img.shields.io/pypi/v/pytrendclust)](https://pypi.org/project/pytrendclust/)
37
+ [![Python](https://img.shields.io/pypi/pyversions/pytrendclust)](https://pypi.org/project/pytrendclust/)
38
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue)](https://github.com/89605502155/pytrendclust/blob/main/LICENSE)
39
+
40
+ Source code: <https://github.com/89605502155/pytrendclust>
41
+
42
+ Trend clustering of time series with an **adaptive number of clusters**.
43
+
44
+ `pytrendclust` splits a series into consecutive runs that rise, fall or stay
45
+ flat. You do not pass the number of clusters. It is chosen by a penalised
46
+ least-squares criterion in which every extra change point costs exponentially
47
+ more than the previous one. Real changes of direction are found, and noise is
48
+ not mistaken for a trend.
49
+
50
+ Typical use: keep only the most recent regime of a series as the training set
51
+ for a regression model. For example, a series of RAM consumption first falls
52
+ and then grows. Fitting a line to the whole history mixes two regimes; fitting
53
+ it to the last cluster does not.
54
+
55
+ ```python
56
+ from pytrendclust import TrendClusterer
57
+
58
+ y = [10, 9, 8, 7, 6, 5, 4, 5, 6, 7]
59
+
60
+ result = TrendClusterer().fit(y).result_
61
+ print(result.summary())
62
+ # 10 points, 1 change point(s), noise std 0.27
63
+ # #0: [0, 7) 7 pts down slope -1
64
+ # #1: [7, 10) 3 pts up slope +1
65
+
66
+ t_recent, y_recent = result.select() # points of the last (newest) cluster
67
+ result.export("result.json") # or "points.csv"
68
+ ```
69
+
70
+ A "picket fence" `5, 6, 5, 6, ...` is one flat cluster, not ten tiny trends:
71
+
72
+ ```python
73
+ >>> from pytrendclust import cluster_trends
74
+ >>> cluster_trends([5, 6] * 5).summary()
75
+ '10 points, 0 change point(s), noise std 1.211\n #0: [0, 10) 10 pts flat slope +0.0303'
76
+ >>> cluster_trends([5, 6] * 5, penalty="none").n_change_points # no regularisation
77
+ 2
78
+
79
+ ```
80
+
81
+ ## Installation
82
+
83
+ ```bash
84
+ pip install pytrendclust # numpy only
85
+ pip install "pytrendclust[pandas]" # + DataFrame export
86
+ ```
87
+
88
+ ## Method
89
+
90
+ For every possible number of change points `K`, a dynamic programme finds the
91
+ exact partition into `K + 1` runs with the smallest total squared error of
92
+ per-run least-squares lines, `SSE(K)`. The clusterer then picks
93
+
94
+ ```
95
+ K* = argmin_K SSE(K) + sigma² · pen(K)
96
+ pen(K) = c · (exp(g·K) − 1) / (exp(g) − 1)
97
+ ```
98
+
99
+ * `c` (`penalty_scale`): price of the first change point in noise variances.
100
+ The default is `5·ln(n)`.
101
+ * `g` (`penalty_growth`, default 0.5): each next change point costs `e^g`
102
+ times more than the previous one. `g = 0` gives the classic linear penalty.
103
+ * `sigma` (`noise_std`): noise level. By default it is estimated robustly
104
+ from second differences of the series, which cancel any linear trend. This
105
+ makes the result independent of units: bytes and gigabytes give the same
106
+ clusters.
107
+
108
+ The search over `K` stops as soon as the penalty alone exceeds the best value
109
+ found. The exponential penalty makes this happen after a few steps. The
110
+ solution is the global optimum of the criterion; the test suite checks it
111
+ against brute-force enumeration.
112
+
113
+ Each cluster is labelled `up`, `down` or `flat` with a t-test on its slope:
114
+ `|slope| / se(slope) < flat_threshold` (default 2) means `flat`.
115
+
116
+ ## API
117
+
118
+ | Object | Purpose |
119
+ |---|---|
120
+ | `TrendClusterer(**params)` | set parameters; `fit(y, t=None)`, `fit_predict`, `get_params`, `set_params` |
121
+ | `cluster_trends(y, t=None, **params)` | one-call shortcut returning the result |
122
+ | `TrendClusteringResult` | `segments`, `change_points`, `labels`, `trends`, `fitted`, `last_segment`, `select()`, `indices()`, `criterion` |
123
+ | export | `to_dict()`, `to_json(path)`, `to_csv(path)`, `to_dataframe()`, `export(path)`; `from_dict`, `from_json` |
124
+
125
+ Parameters: `penalty` (`"exponential"`, `"linear"`, `"none"`, or a callable
126
+ `f(K, n, sigma²)`), `penalty_scale`, `penalty_growth`, `min_size` (default 3),
127
+ `max_change_points`, `noise_std`, `flat_threshold`.
128
+
129
+ `result.criterion` lists `SSE(K)`, the penalty and their sum for every `K` the
130
+ search evaluated. Use it to justify the chosen number of clusters.
131
+
132
+ Complexity is `O(K · n²)` time and `O(K · n)` memory: about 0.4 s for 1000
133
+ points and 6 s for 5000.
134
+
135
+ ## Кратко по-русски
136
+
137
+ Трендовая кластеризация временного ряда с адаптивным числом классов. Ряд
138
+ делится на участки роста, спада и плато. Число точек разладки выбирается
139
+ автоматически: к ошибке модели добавляется штраф, растущий экспоненциально с
140
+ каждой новой точкой. Так шумовые колебания («заборчик») не принимаются за смену
141
+ тренда. `result.select()` возвращает точки последнего, самого свежего участка.
142
+ На них обучается регрессия.
143
+
144
+ ## Development
145
+
146
+ ```bash
147
+ git clone https://github.com/89605502155/pytrendclust.git
148
+ cd pytrendclust
149
+ uv sync --all-extras
150
+ uv run pytest
151
+ uv run ruff check src tests && uv run ruff format --check src tests
152
+ uv run mypy
153
+ uv build
154
+ ```
155
+
156
+ ## License
157
+
158
+ MIT, see [LICENSE](https://github.com/89605502155/pytrendclust/blob/main/LICENSE).
@@ -0,0 +1,125 @@
1
+ # pytrendclust
2
+
3
+ [![PyPI](https://img.shields.io/pypi/v/pytrendclust)](https://pypi.org/project/pytrendclust/)
4
+ [![Python](https://img.shields.io/pypi/pyversions/pytrendclust)](https://pypi.org/project/pytrendclust/)
5
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue)](https://github.com/89605502155/pytrendclust/blob/main/LICENSE)
6
+
7
+ Source code: <https://github.com/89605502155/pytrendclust>
8
+
9
+ Trend clustering of time series with an **adaptive number of clusters**.
10
+
11
+ `pytrendclust` splits a series into consecutive runs that rise, fall or stay
12
+ flat. You do not pass the number of clusters. It is chosen by a penalised
13
+ least-squares criterion in which every extra change point costs exponentially
14
+ more than the previous one. Real changes of direction are found, and noise is
15
+ not mistaken for a trend.
16
+
17
+ Typical use: keep only the most recent regime of a series as the training set
18
+ for a regression model. For example, a series of RAM consumption first falls
19
+ and then grows. Fitting a line to the whole history mixes two regimes; fitting
20
+ it to the last cluster does not.
21
+
22
+ ```python
23
+ from pytrendclust import TrendClusterer
24
+
25
+ y = [10, 9, 8, 7, 6, 5, 4, 5, 6, 7]
26
+
27
+ result = TrendClusterer().fit(y).result_
28
+ print(result.summary())
29
+ # 10 points, 1 change point(s), noise std 0.27
30
+ # #0: [0, 7) 7 pts down slope -1
31
+ # #1: [7, 10) 3 pts up slope +1
32
+
33
+ t_recent, y_recent = result.select() # points of the last (newest) cluster
34
+ result.export("result.json") # or "points.csv"
35
+ ```
36
+
37
+ A "picket fence" `5, 6, 5, 6, ...` is one flat cluster, not ten tiny trends:
38
+
39
+ ```python
40
+ >>> from pytrendclust import cluster_trends
41
+ >>> cluster_trends([5, 6] * 5).summary()
42
+ '10 points, 0 change point(s), noise std 1.211\n #0: [0, 10) 10 pts flat slope +0.0303'
43
+ >>> cluster_trends([5, 6] * 5, penalty="none").n_change_points # no regularisation
44
+ 2
45
+
46
+ ```
47
+
48
+ ## Installation
49
+
50
+ ```bash
51
+ pip install pytrendclust # numpy only
52
+ pip install "pytrendclust[pandas]" # + DataFrame export
53
+ ```
54
+
55
+ ## Method
56
+
57
+ For every possible number of change points `K`, a dynamic programme finds the
58
+ exact partition into `K + 1` runs with the smallest total squared error of
59
+ per-run least-squares lines, `SSE(K)`. The clusterer then picks
60
+
61
+ ```
62
+ K* = argmin_K SSE(K) + sigma² · pen(K)
63
+ pen(K) = c · (exp(g·K) − 1) / (exp(g) − 1)
64
+ ```
65
+
66
+ * `c` (`penalty_scale`): price of the first change point in noise variances.
67
+ The default is `5·ln(n)`.
68
+ * `g` (`penalty_growth`, default 0.5): each next change point costs `e^g`
69
+ times more than the previous one. `g = 0` gives the classic linear penalty.
70
+ * `sigma` (`noise_std`): noise level. By default it is estimated robustly
71
+ from second differences of the series, which cancel any linear trend. This
72
+ makes the result independent of units: bytes and gigabytes give the same
73
+ clusters.
74
+
75
+ The search over `K` stops as soon as the penalty alone exceeds the best value
76
+ found. The exponential penalty makes this happen after a few steps. The
77
+ solution is the global optimum of the criterion; the test suite checks it
78
+ against brute-force enumeration.
79
+
80
+ Each cluster is labelled `up`, `down` or `flat` with a t-test on its slope:
81
+ `|slope| / se(slope) < flat_threshold` (default 2) means `flat`.
82
+
83
+ ## API
84
+
85
+ | Object | Purpose |
86
+ |---|---|
87
+ | `TrendClusterer(**params)` | set parameters; `fit(y, t=None)`, `fit_predict`, `get_params`, `set_params` |
88
+ | `cluster_trends(y, t=None, **params)` | one-call shortcut returning the result |
89
+ | `TrendClusteringResult` | `segments`, `change_points`, `labels`, `trends`, `fitted`, `last_segment`, `select()`, `indices()`, `criterion` |
90
+ | export | `to_dict()`, `to_json(path)`, `to_csv(path)`, `to_dataframe()`, `export(path)`; `from_dict`, `from_json` |
91
+
92
+ Parameters: `penalty` (`"exponential"`, `"linear"`, `"none"`, or a callable
93
+ `f(K, n, sigma²)`), `penalty_scale`, `penalty_growth`, `min_size` (default 3),
94
+ `max_change_points`, `noise_std`, `flat_threshold`.
95
+
96
+ `result.criterion` lists `SSE(K)`, the penalty and their sum for every `K` the
97
+ search evaluated. Use it to justify the chosen number of clusters.
98
+
99
+ Complexity is `O(K · n²)` time and `O(K · n)` memory: about 0.4 s for 1000
100
+ points and 6 s for 5000.
101
+
102
+ ## Кратко по-русски
103
+
104
+ Трендовая кластеризация временного ряда с адаптивным числом классов. Ряд
105
+ делится на участки роста, спада и плато. Число точек разладки выбирается
106
+ автоматически: к ошибке модели добавляется штраф, растущий экспоненциально с
107
+ каждой новой точкой. Так шумовые колебания («заборчик») не принимаются за смену
108
+ тренда. `result.select()` возвращает точки последнего, самого свежего участка.
109
+ На них обучается регрессия.
110
+
111
+ ## Development
112
+
113
+ ```bash
114
+ git clone https://github.com/89605502155/pytrendclust.git
115
+ cd pytrendclust
116
+ uv sync --all-extras
117
+ uv run pytest
118
+ uv run ruff check src tests && uv run ruff format --check src tests
119
+ uv run mypy
120
+ uv build
121
+ ```
122
+
123
+ ## License
124
+
125
+ MIT, see [LICENSE](https://github.com/89605502155/pytrendclust/blob/main/LICENSE).
@@ -0,0 +1,112 @@
1
+ [project]
2
+ name = "pytrendclust"
3
+ version = "0.1.0"
4
+ description = "Trend clustering of time series with an adaptive number of clusters: split a series into rising, falling and flat runs with an exponentially growing penalty on change points"
5
+ readme = "README.md"
6
+ authors = [
7
+ { name = "Andrey Ferubko", email = "ferubko1999@yandex.ru" }
8
+ ]
9
+ requires-python = ">=3.11"
10
+ license = "MIT"
11
+ license-files = ["LICENSE"]
12
+ keywords = [
13
+ "time-series",
14
+ "clustering",
15
+ "trend",
16
+ "change-point",
17
+ "segmentation",
18
+ "piecewise-linear",
19
+ "regression",
20
+ ]
21
+ classifiers = [
22
+ "Development Status :: 4 - Beta",
23
+ "Intended Audience :: Developers",
24
+ "Intended Audience :: Science/Research",
25
+ "Operating System :: OS Independent",
26
+ "Programming Language :: Python :: 3",
27
+ "Programming Language :: Python :: 3 :: Only",
28
+ "Programming Language :: Python :: 3.11",
29
+ "Programming Language :: Python :: 3.12",
30
+ "Programming Language :: Python :: 3.13",
31
+ "Programming Language :: Python :: 3.14",
32
+ "Topic :: Scientific/Engineering",
33
+ "Topic :: Scientific/Engineering :: Information Analysis",
34
+ "Topic :: Scientific/Engineering :: Mathematics",
35
+ "Typing :: Typed",
36
+ ]
37
+ dependencies = [
38
+ "numpy>=1.26",
39
+ ]
40
+
41
+ [project.optional-dependencies]
42
+ pandas = [
43
+ "pandas>=2.1",
44
+ ]
45
+
46
+ [project.urls]
47
+ Homepage = "https://github.com/89605502155/pytrendclust"
48
+ Repository = "https://github.com/89605502155/pytrendclust"
49
+ Documentation = "https://github.com/89605502155/pytrendclust#readme"
50
+ Issues = "https://github.com/89605502155/pytrendclust/issues"
51
+
52
+ [build-system]
53
+ requires = ["uv_build>=0.11.9,<0.12.0"]
54
+ build-backend = "uv_build"
55
+
56
+ [dependency-groups]
57
+ dev = [
58
+ "hypothesis>=6.168.5",
59
+ "mypy>=2.4.0",
60
+ "pytest>=9.1.1",
61
+ "pytest-cov>=7.1.0",
62
+ "ruff>=0.16.10",
63
+ ]
64
+
65
+ [tool.ruff]
66
+ line-length = 100
67
+ target-version = "py311"
68
+ src = ["src", "tests"]
69
+
70
+ [tool.ruff.lint]
71
+ select = ["E", "W", "F", "I", "N", "UP", "B", "A", "C4", "EM", "ISC", "PIE", "PT", "RET", "SIM",
72
+ "TID", "T20", "PERF", "PL", "RUF", "S"]
73
+ ignore = [
74
+ "PLR0913", # too many arguments: the constructor mirrors the documented parameters
75
+ "PLR2004", # magic values
76
+ "PLC0415", # imports inside functions keep optional dependencies optional
77
+ ]
78
+
79
+ [tool.ruff.lint.per-file-ignores]
80
+ "tests/**" = ["S101", "PLR2004", "PLC0415", "N802", "S311", "RUF001", "RUF003"]
81
+ "examples/**" = ["T20", "S101", "RUF001", "RUF003"]
82
+
83
+ [tool.ruff.lint.isort]
84
+ known-first-party = ["pytrendclust"]
85
+
86
+ [tool.ruff.format]
87
+ docstring-code-format = true
88
+
89
+ [tool.pytest.ini_options]
90
+ testpaths = ["tests", "src", "README.md"]
91
+ addopts = ["-ra", "--strict-markers", "--strict-config", "--doctest-modules", "--doctest-glob=README.md", "--import-mode=importlib"]
92
+ doctest_optionflags = ["NORMALIZE_WHITESPACE", "ELLIPSIS"]
93
+ filterwarnings = ["error"]
94
+
95
+ [tool.coverage.run]
96
+ source = ["pytrendclust"]
97
+ branch = true
98
+
99
+ [tool.coverage.report]
100
+ show_missing = true
101
+ skip_covered = true
102
+ exclude_also = ["if TYPE_CHECKING:"]
103
+
104
+ [tool.mypy]
105
+ python_version = "3.11"
106
+ files = ["src/pytrendclust"]
107
+ strict = true
108
+ warn_unused_ignores = true
109
+
110
+ [[tool.mypy.overrides]]
111
+ module = ["pandas", "pandas.*"]
112
+ ignore_missing_imports = true
@@ -0,0 +1,37 @@
1
+ """Trend clustering of time series with an adaptive number of clusters.
2
+
3
+ Splits a series into consecutive rising, falling and flat runs, choosing the
4
+ number of change points by a penalised least-squares criterion whose penalty
5
+ grows exponentially with every extra change point.
6
+
7
+ >>> from pytrendclust import cluster_trends
8
+ >>> result = cluster_trends([10, 9, 8, 7, 6, 5, 4, 5, 6, 7])
9
+ >>> result.last_segment.trend.value, result.indices().tolist()
10
+ ('up', [7, 8, 9])
11
+ """
12
+
13
+ from pytrendclust.clusterer import TrendClusterer, cluster_trends
14
+ from pytrendclust.penalty import (
15
+ PenaltyFunc,
16
+ default_scale,
17
+ estimate_noise_std,
18
+ exponential_penalty,
19
+ linear_penalty,
20
+ )
21
+ from pytrendclust.result import CriterionRow, Segment, Trend, TrendClusteringResult
22
+
23
+ __all__ = [
24
+ "CriterionRow",
25
+ "PenaltyFunc",
26
+ "Segment",
27
+ "Trend",
28
+ "TrendClusterer",
29
+ "TrendClusteringResult",
30
+ "cluster_trends",
31
+ "default_scale",
32
+ "estimate_noise_std",
33
+ "exponential_penalty",
34
+ "linear_penalty",
35
+ ]
36
+
37
+ __version__ = "0.1.0"
@@ -0,0 +1,70 @@
1
+ """Least-squares cost of fitting a straight line to a contiguous run of points.
2
+
3
+ Prefix sums make the residual sum of squares (SSE) of any run ``[start, stop)``
4
+ an O(1) computation, which is what keeps the dynamic programme in
5
+ :mod:`pytrendclust._dp` affordable.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass
11
+
12
+ import numpy as np
13
+ from numpy.typing import NDArray
14
+
15
+ FloatArray = NDArray[np.float64]
16
+ IntArray = NDArray[np.int64]
17
+
18
+
19
+ @dataclass(frozen=True, slots=True)
20
+ class LineFit:
21
+ """Ordinary least-squares line ``y = intercept + slope * t`` over one run."""
22
+
23
+ slope: float
24
+ intercept: float
25
+ sse: float
26
+ #: Centred sum of squares of ``t``; zero when the run has a single point.
27
+ stt: float
28
+
29
+
30
+ def fit_line(t: FloatArray, y: FloatArray) -> LineFit:
31
+ """Fit a straight line to ``(t, y)`` by ordinary least squares."""
32
+ t_mean = float(t.mean())
33
+ y_mean = float(y.mean())
34
+ dt = t - t_mean
35
+ dy = y - y_mean
36
+ stt = float(dt @ dt)
37
+ slope = float(dt @ dy) / stt if stt > 0.0 else 0.0
38
+ intercept = y_mean - slope * t_mean
39
+ residual = dy - slope * dt
40
+ return LineFit(slope=slope, intercept=intercept, sse=float(residual @ residual), stt=stt)
41
+
42
+
43
+ class LinearCost:
44
+ """SSE of the best straight line on any run ``[start, stop)`` of a series."""
45
+
46
+ def __init__(self, t: FloatArray, y: FloatArray) -> None:
47
+ # Centring both axes keeps the prefix-sum differences well conditioned.
48
+ tc = t - t.mean()
49
+ yc = y - y.mean()
50
+ zero = np.zeros(1)
51
+ self._n = np.arange(len(y) + 1, dtype=np.float64)
52
+ self._st = np.concatenate([zero, np.cumsum(tc)])
53
+ self._sy = np.concatenate([zero, np.cumsum(yc)])
54
+ self._stt = np.concatenate([zero, np.cumsum(tc * tc)])
55
+ self._sty = np.concatenate([zero, np.cumsum(tc * yc)])
56
+ self._syy = np.concatenate([zero, np.cumsum(yc * yc)])
57
+ #: Total sum of squares of the series around its mean — the cost scale.
58
+ self.total_ss = float(self._syy[-1])
59
+
60
+ def sse(self, starts: IntArray, stop: int) -> FloatArray:
61
+ """SSE of the runs ``[s, stop)`` for every ``s`` in ``starts`` (vectorised)."""
62
+ n = self._n[stop] - self._n[starts]
63
+ st = self._st[stop] - self._st[starts]
64
+ sy = self._sy[stop] - self._sy[starts]
65
+ ctt = (self._stt[stop] - self._stt[starts]) - st * st / n
66
+ cty = (self._sty[stop] - self._sty[starts]) - st * sy / n
67
+ cyy = (self._syy[stop] - self._syy[starts]) - sy * sy / n
68
+ explained = np.divide(cty * cty, ctt, out=np.zeros_like(ctt), where=ctt > 0.0)
69
+ sse: FloatArray = np.maximum(cyy - explained, 0.0)
70
+ return sse
@@ -0,0 +1,80 @@
1
+ """Exact penalised segmentation by dynamic programming over the number of segments.
2
+
3
+ For every number of change points ``K`` the programme finds the partition of the
4
+ series into ``K + 1`` runs with the smallest total SSE (the "segment
5
+ neighbourhood" recursion). The penalty may be any non-decreasing function of
6
+ ``K`` — in particular an exponential one, which an additive per-segment
7
+ recursion such as PELT cannot express.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from collections.abc import Callable
13
+ from dataclasses import dataclass
14
+
15
+ import numpy as np
16
+
17
+ from pytrendclust._cost import IntArray, LinearCost
18
+ from pytrendclust.result import CriterionRow
19
+
20
+
21
+ @dataclass(frozen=True, slots=True)
22
+ class Segmentation:
23
+ change_points: tuple[int, ...]
24
+ criterion: tuple[CriterionRow, ...]
25
+
26
+
27
+ def segment(
28
+ cost: LinearCost,
29
+ n: int,
30
+ min_size: int,
31
+ max_change_points: int,
32
+ penalty: Callable[[int], float],
33
+ ) -> Segmentation:
34
+ """Return the change points minimising ``SSE(K) + penalty(K)``.
35
+
36
+ Change points are indices ``c`` such that a new segment starts at ``c``.
37
+ The search over ``K`` stops early once ``penalty(K)`` alone exceeds the best
38
+ objective seen so far: SSE is non-negative and the penalty non-decreasing,
39
+ so no larger ``K`` can win.
40
+ """
41
+ inf = np.inf
42
+ all_stops = np.arange(n + 1)
43
+ # best[j]: minimal SSE of splitting the prefix [0, j) into k + 1 segments.
44
+ best = np.full(n + 1, inf)
45
+ best[min_size:] = [cost.sse(np.array([0]), j)[0] for j in all_stops[min_size:]]
46
+ back: list[IntArray] = []
47
+ tie_tol = 1e-10 * cost.total_ss
48
+
49
+ criterion = [CriterionRow(0, float(best[n]), penalty(0))]
50
+ best_k, best_obj = 0, criterion[0].objective
51
+
52
+ for k in range(1, max_change_points + 1):
53
+ pen = penalty(k)
54
+ if pen >= best_obj:
55
+ break
56
+ new = np.full(n + 1, inf)
57
+ arg = np.zeros(n + 1, dtype=np.int64)
58
+ for j in range((k + 1) * min_size, n + 1):
59
+ starts = np.arange(k * min_size, j - min_size + 1)
60
+ total = best[starts] + cost.sse(starts, j)
61
+ # On ties take the latest start: a point lying on both lines (the
62
+ # vertex of a V) stays with the older regime, keeping the newest
63
+ # segment free of ambiguous points.
64
+ i = int(np.flatnonzero(total <= total.min() + tie_tol)[-1])
65
+ new[j] = total[i]
66
+ arg[j] = starts[i]
67
+ best = new
68
+ back.append(arg)
69
+ point = CriterionRow(k, float(best[n]), pen)
70
+ criterion.append(point)
71
+ # Strict improvement beyond rounding noise: ties go to fewer change points.
72
+ if point.objective < best_obj - tie_tol:
73
+ best_k, best_obj = k, point.objective
74
+
75
+ change_points: list[int] = []
76
+ j = n
77
+ for k in range(best_k, 0, -1):
78
+ j = int(back[k - 1][j])
79
+ change_points.append(j)
80
+ return Segmentation(tuple(reversed(change_points)), tuple(criterion))
@@ -0,0 +1,310 @@
1
+ """Trend clustering of a time series with an adaptive number of clusters."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import itertools
6
+ import math
7
+ from typing import Any, Literal
8
+
9
+ import numpy as np
10
+ from numpy.typing import ArrayLike, NDArray
11
+
12
+ from pytrendclust._cost import LinearCost, fit_line
13
+ from pytrendclust._dp import segment
14
+ from pytrendclust.penalty import (
15
+ PenaltyFunc,
16
+ default_scale,
17
+ estimate_noise_std,
18
+ exponential_penalty,
19
+ linear_penalty,
20
+ )
21
+ from pytrendclust.result import Segment, Trend, TrendClusteringResult
22
+
23
+ PenaltyKind = Literal["exponential", "linear", "none"]
24
+
25
+ # Floor of the noise scale relative to the series' spread: on a perfectly clean
26
+ # series the noise estimate is 0, and a zero penalty would make every
27
+ # segmentation with zero SSE a tie.
28
+ _NOISE_FLOOR = 1e-6
29
+
30
+
31
+ class TrendClusterer:
32
+ """Split a time series into consecutive trend clusters (rising, falling, flat).
33
+
34
+ Each cluster is a contiguous run of points with its own least-squares line.
35
+ The number of clusters is not given in advance: the clusterer minimises
36
+
37
+ SSE(K) + sigma**2 * penalty(K)
38
+
39
+ over the number of change points ``K``, where ``SSE(K)`` is the best total
40
+ squared error of a piecewise-linear fit with ``K`` change points and
41
+ ``sigma`` is the noise level. The default exponential penalty makes every
42
+ extra change point dearer than the previous one, so noise such as a
43
+ "picket fence" (up, down, up, down, ...) is recognised as a single flat
44
+ cluster instead of ten tiny trends.
45
+
46
+ Parameters
47
+ ----------
48
+ penalty:
49
+ ``"exponential"`` (default) — the ``K``-th change point costs
50
+ ``penalty_scale * exp(penalty_growth * (K - 1))`` noise variances;
51
+ ``"linear"`` — every change point costs ``penalty_scale``;
52
+ ``"none"`` — no regularisation (only ``max_change_points`` limits K);
53
+ or a callable ``f(n_change_points, n_samples, noise_variance)`` returning
54
+ the penalty in the units of SSE, non-decreasing in ``n_change_points``.
55
+ penalty_scale:
56
+ Price of the first change point in noise variances. Defaults to
57
+ ``5 * ln(n)`` (see :func:`~pytrendclust.default_scale`). Larger means
58
+ fewer change points.
59
+ penalty_growth:
60
+ Exponential growth rate of the price of every next change point.
61
+ min_size:
62
+ Minimum number of points in a cluster; at least 2 (a line needs two
63
+ points, and three are needed before a line can be wrong).
64
+ max_change_points:
65
+ Hard upper bound on ``K``; ``None`` means only the penalty limits it.
66
+ noise_std:
67
+ Noise standard deviation in the units of the series. ``None`` estimates
68
+ it robustly from second differences of the series.
69
+ flat_threshold:
70
+ A cluster is ``flat`` when ``|slope| / standard_error(slope)`` is below
71
+ this value (2 is roughly the 95 % significance level).
72
+
73
+ Examples
74
+ --------
75
+ >>> y = [10, 9, 8, 7, 6, 5, 4, 5, 6, 7]
76
+ >>> result = TrendClusterer().fit(y).result_
77
+ >>> result.change_points
78
+ [7]
79
+ >>> [s.trend.value for s in result.segments]
80
+ ['down', 'up']
81
+ >>> result.select()[1]
82
+ array([5., 6., 7.])
83
+ """
84
+
85
+ def __init__(
86
+ self,
87
+ *,
88
+ penalty: PenaltyKind | PenaltyFunc = "exponential",
89
+ penalty_scale: float | None = None,
90
+ penalty_growth: float = 0.5,
91
+ min_size: int = 3,
92
+ max_change_points: int | None = None,
93
+ noise_std: float | None = None,
94
+ flat_threshold: float = 2.0,
95
+ ) -> None:
96
+ self.penalty = penalty
97
+ self.penalty_scale = penalty_scale
98
+ self.penalty_growth = penalty_growth
99
+ self.min_size = min_size
100
+ self.max_change_points = max_change_points
101
+ self.noise_std = noise_std
102
+ self.flat_threshold = flat_threshold
103
+ self._validate()
104
+ self.result_: TrendClusteringResult | None = None
105
+
106
+ # -- parameters --------------------------------------------------------
107
+
108
+ _PARAM_NAMES = (
109
+ "penalty",
110
+ "penalty_scale",
111
+ "penalty_growth",
112
+ "min_size",
113
+ "max_change_points",
114
+ "noise_std",
115
+ "flat_threshold",
116
+ )
117
+
118
+ def get_params(self) -> dict[str, Any]:
119
+ """Current parameters, as accepted by the constructor and :meth:`set_params`."""
120
+ return {name: getattr(self, name) for name in self._PARAM_NAMES}
121
+
122
+ def set_params(self, **params: Any) -> TrendClusterer:
123
+ """Change parameters in place; returns ``self`` for chaining."""
124
+ unknown = set(params) - set(self._PARAM_NAMES)
125
+ if unknown:
126
+ msg = f"unknown parameter(s): {', '.join(sorted(unknown))}"
127
+ raise ValueError(msg)
128
+ old = self.get_params()
129
+ for name, value in params.items():
130
+ setattr(self, name, value)
131
+ try:
132
+ self._validate()
133
+ except (TypeError, ValueError):
134
+ for name, value in old.items():
135
+ setattr(self, name, value)
136
+ raise
137
+ return self
138
+
139
+ def __repr__(self) -> str:
140
+ args = ", ".join(f"{k}={v!r}" for k, v in self.get_params().items())
141
+ return f"{type(self).__name__}({args})"
142
+
143
+ def _validate(self) -> None:
144
+ if not callable(self.penalty) and self.penalty not in ("exponential", "linear", "none"):
145
+ msg = (
146
+ "penalty must be 'exponential', 'linear', 'none' or a callable, "
147
+ f"got {self.penalty!r}"
148
+ )
149
+ raise ValueError(msg)
150
+ if self.penalty_scale is not None and not (
151
+ math.isfinite(self.penalty_scale) and self.penalty_scale >= 0
152
+ ):
153
+ msg = f"penalty_scale must be a finite number >= 0, got {self.penalty_scale!r}"
154
+ raise ValueError(msg)
155
+ if not (math.isfinite(self.penalty_growth) and self.penalty_growth >= 0):
156
+ msg = f"penalty_growth must be a finite number >= 0, got {self.penalty_growth!r}"
157
+ raise ValueError(msg)
158
+ if isinstance(self.min_size, bool) or not isinstance(self.min_size, int):
159
+ msg = f"min_size must be an int, got {self.min_size!r}"
160
+ raise TypeError(msg)
161
+ if self.min_size < 2:
162
+ msg = f"min_size must be >= 2, got {self.min_size}"
163
+ raise ValueError(msg)
164
+ if self.max_change_points is not None and (
165
+ isinstance(self.max_change_points, bool)
166
+ or not isinstance(self.max_change_points, int)
167
+ or self.max_change_points < 0
168
+ ):
169
+ msg = f"max_change_points must be None or an int >= 0, got {self.max_change_points!r}"
170
+ raise ValueError(msg)
171
+ if self.noise_std is not None and not (
172
+ math.isfinite(self.noise_std) and self.noise_std >= 0
173
+ ):
174
+ msg = f"noise_std must be None or a finite number >= 0, got {self.noise_std!r}"
175
+ raise ValueError(msg)
176
+ if not (math.isfinite(self.flat_threshold) and self.flat_threshold >= 0):
177
+ msg = f"flat_threshold must be a finite number >= 0, got {self.flat_threshold!r}"
178
+ raise ValueError(msg)
179
+
180
+ # -- fitting -----------------------------------------------------------
181
+
182
+ def fit(self, y: ArrayLike, t: ArrayLike | None = None) -> TrendClusterer:
183
+ """Cluster the series ``y`` observed at times ``t`` (default ``0, 1, 2, ...``).
184
+
185
+ The result is stored in :attr:`result_`; returns ``self``.
186
+ """
187
+ y_arr, t_arr = _as_series(y, t)
188
+ n = len(y_arr)
189
+
190
+ sigma = self._noise_std(y_arr)
191
+ sigma2 = sigma * sigma
192
+ penalty = self._penalty_in_sse_units(n, sigma2)
193
+ largest_k = n // self.min_size - 1
194
+ if self.max_change_points is not None:
195
+ largest_k = min(largest_k, self.max_change_points)
196
+
197
+ seg = segment(LinearCost(t_arr, y_arr), n, self.min_size, max(largest_k, 0), penalty)
198
+
199
+ bounds = [0, *seg.change_points, n]
200
+ segments = tuple(
201
+ self._describe(index=i, start=start, stop=stop, t=t_arr, y=y_arr, sigma=sigma)
202
+ for i, (start, stop) in enumerate(itertools.pairwise(bounds))
203
+ )
204
+ self.result_ = TrendClusteringResult(
205
+ t=t_arr,
206
+ y=y_arr,
207
+ segments=segments,
208
+ noise_std=sigma,
209
+ criterion=seg.criterion,
210
+ params=self._exportable_params(),
211
+ )
212
+ return self
213
+
214
+ def fit_predict(self, y: ArrayLike, t: ArrayLike | None = None) -> NDArray[np.int64]:
215
+ """Fit and return the cluster (segment) index of every point."""
216
+ result = self.fit(y, t).result_
217
+ assert result is not None # noqa: S101 - set by fit()
218
+ return result.labels
219
+
220
+ def _noise_std(self, y: NDArray[np.float64]) -> float:
221
+ sigma = estimate_noise_std(y) if self.noise_std is None else float(self.noise_std)
222
+ spread = float(np.ptp(y)) if len(y) else 0.0
223
+ return max(sigma, _NOISE_FLOOR * spread)
224
+
225
+ def _penalty_in_sse_units(self, n: int, sigma2: float) -> Any:
226
+ if callable(self.penalty):
227
+ func = self.penalty
228
+ return lambda k: float(func(k, n, sigma2))
229
+ if self.penalty == "none":
230
+ return lambda k: 0.0
231
+ scale = default_scale(n) if self.penalty_scale is None else self.penalty_scale
232
+ if self.penalty == "linear":
233
+ return lambda k: sigma2 * linear_penalty(k, scale)
234
+ growth = self.penalty_growth
235
+ return lambda k: sigma2 * exponential_penalty(k, scale, growth)
236
+
237
+ def _describe(
238
+ self,
239
+ *,
240
+ index: int,
241
+ start: int,
242
+ stop: int,
243
+ t: NDArray[np.float64],
244
+ y: NDArray[np.float64],
245
+ sigma: float,
246
+ ) -> Segment:
247
+ line = fit_line(t[start:stop], y[start:stop])
248
+ if line.stt == 0.0 or line.slope == 0.0:
249
+ t_stat = 0.0
250
+ elif sigma == 0.0:
251
+ t_stat = math.inf
252
+ else:
253
+ t_stat = abs(line.slope) * math.sqrt(line.stt) / sigma
254
+ if t_stat < self.flat_threshold:
255
+ trend = Trend.FLAT
256
+ else:
257
+ trend = Trend.UP if line.slope > 0 else Trend.DOWN
258
+ return Segment(
259
+ index=index,
260
+ start=start,
261
+ stop=stop,
262
+ trend=trend,
263
+ slope=line.slope,
264
+ intercept=line.intercept,
265
+ sse=line.sse,
266
+ t_statistic=t_stat,
267
+ )
268
+
269
+ def _exportable_params(self) -> dict[str, Any]:
270
+ params = self.get_params()
271
+ if callable(params["penalty"]):
272
+ params["penalty"] = getattr(params["penalty"], "__name__", repr(params["penalty"]))
273
+ return params
274
+
275
+
276
+ def cluster_trends(
277
+ y: ArrayLike, t: ArrayLike | None = None, **params: Any
278
+ ) -> TrendClusteringResult:
279
+ """One-call shortcut: ``TrendClusterer(**params).fit(y, t).result_``."""
280
+ result = TrendClusterer(**params).fit(y, t).result_
281
+ assert result is not None # noqa: S101 - set by fit()
282
+ return result
283
+
284
+
285
+ def _as_series(
286
+ y: ArrayLike, t: ArrayLike | None
287
+ ) -> tuple[NDArray[np.float64], NDArray[np.float64]]:
288
+ y_arr = np.array(y, dtype=np.float64)
289
+ if y_arr.ndim != 1:
290
+ msg = f"y must be one-dimensional, got shape {y_arr.shape}"
291
+ raise ValueError(msg)
292
+ if len(y_arr) == 0:
293
+ msg = "y is empty"
294
+ raise ValueError(msg)
295
+ if not np.all(np.isfinite(y_arr)):
296
+ msg = "y contains NaN or infinite values"
297
+ raise ValueError(msg)
298
+ if t is None:
299
+ return y_arr, np.arange(len(y_arr), dtype=np.float64)
300
+ t_arr = np.array(t, dtype=np.float64)
301
+ if t_arr.shape != y_arr.shape:
302
+ msg = f"t and y must have the same shape, got {t_arr.shape} and {y_arr.shape}"
303
+ raise ValueError(msg)
304
+ if not np.all(np.isfinite(t_arr)):
305
+ msg = "t contains NaN or infinite values"
306
+ raise ValueError(msg)
307
+ if np.any(np.diff(t_arr) <= 0):
308
+ msg = "t must be strictly increasing"
309
+ raise ValueError(msg)
310
+ return y_arr, t_arr
@@ -0,0 +1,74 @@
1
+ """Penalties on the number of change points and the noise estimate they are scaled by.
2
+
3
+ A penalty is measured in units of the noise variance ``sigma**2``: SSE grows with
4
+ the square of the series' units, so a penalty in the same units makes the
5
+ trade-off independent of whether memory is counted in bytes or gigabytes.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import math
11
+ from collections.abc import Callable
12
+
13
+ import numpy as np
14
+ from numpy.typing import NDArray
15
+
16
+ #: ``penalty(n_change_points, n_samples, noise_variance) -> float``.
17
+ #: A custom penalty must be non-decreasing in ``n_change_points``.
18
+ PenaltyFunc = Callable[[int, int, float], float]
19
+
20
+ # Consistency constant turning a median absolute deviation into a standard
21
+ # deviation for Gaussian noise.
22
+ _MAD_TO_STD = 1.482602218505602
23
+
24
+
25
+ def default_scale(n_samples: int) -> float:
26
+ """Default price of the first change point, in noise variances: ``5 * ln(n)``.
27
+
28
+ Every extra segment costs three parameters — slope, intercept and the
29
+ position of its start — for which BIC would charge ``3 * ln(n)``. BIC is
30
+ too liberal here: the start position is chosen as the best of ``n``
31
+ candidates, so on pure noise it finds a spurious short trend in ~7 % of
32
+ 100-point series, against ~0.5 % with ``5 * ln(n)`` at the same detection
33
+ rate of real breaks.
34
+ """
35
+ return 5.0 * math.log(max(n_samples, 2))
36
+
37
+
38
+ def exponential_penalty(n_change_points: int, scale: float, growth: float) -> float:
39
+ """``scale * (exp(growth * K) - 1) / (exp(growth) - 1)``, in noise variances.
40
+
41
+ The first change point costs exactly ``scale``; every next one costs
42
+ ``exp(growth)`` times more than the previous, so the model grows ever more
43
+ reluctant to add change points. ``growth = 0`` degenerates to the linear
44
+ penalty ``scale * K``.
45
+ """
46
+ if n_change_points == 0:
47
+ return 0.0
48
+ if growth == 0.0:
49
+ return scale * n_change_points
50
+ return scale * math.expm1(growth * n_change_points) / math.expm1(growth)
51
+
52
+
53
+ def linear_penalty(n_change_points: int, scale: float) -> float:
54
+ """Classic constant price per change point: ``scale * K``, in noise variances."""
55
+ return scale * n_change_points
56
+
57
+
58
+ def estimate_noise_std(y: NDArray[np.float64]) -> float:
59
+ """Robust estimate of the noise standard deviation around a piecewise-linear trend.
60
+
61
+ Second differences cancel any linear trend, leaving ``eps[i] - 2 eps[i+1] +
62
+ eps[i+2]`` with variance ``6 sigma**2``; a few kinks between segments are
63
+ outliers that the median absolute deviation ignores. Falls back to the plain
64
+ standard deviation of the second differences when more than half of them
65
+ coincide (for example, on a quantised series), and returns 0 for series too
66
+ short to tell.
67
+ """
68
+ if len(y) < 4:
69
+ return 0.0
70
+ d2 = np.diff(y, n=2)
71
+ mad = float(np.median(np.abs(d2 - np.median(d2))))
72
+ if mad > 0.0:
73
+ return _MAD_TO_STD * mad / math.sqrt(6.0)
74
+ return float(d2.std()) / math.sqrt(6.0)
File without changes
@@ -0,0 +1,303 @@
1
+ """Result of a trend clustering and its export to dict, JSON, CSV and pandas."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import csv
6
+ import io
7
+ import json
8
+ from dataclasses import dataclass, field
9
+ from enum import StrEnum
10
+ from pathlib import Path
11
+ from typing import TYPE_CHECKING, Any
12
+
13
+ import numpy as np
14
+ from numpy.typing import NDArray
15
+
16
+ if TYPE_CHECKING:
17
+ import pandas as pd
18
+
19
+ FORMAT_VERSION = 1
20
+
21
+
22
+ class Trend(StrEnum):
23
+ """Direction of a segment's linear trend."""
24
+
25
+ UP = "up"
26
+ DOWN = "down"
27
+ FLAT = "flat"
28
+
29
+
30
+ @dataclass(frozen=True, slots=True)
31
+ class Segment:
32
+ """One trend cluster: a contiguous run of points ``[start, stop)`` and its line."""
33
+
34
+ index: int
35
+ start: int
36
+ stop: int
37
+ trend: Trend
38
+ slope: float
39
+ intercept: float
40
+ sse: float
41
+ #: ``|slope| / standard_error(slope)``; ``inf`` for a noise-free series.
42
+ t_statistic: float
43
+
44
+ @property
45
+ def n_points(self) -> int:
46
+ return self.stop - self.start
47
+
48
+ def predict(self, t: Any) -> NDArray[np.float64]:
49
+ """Value of the segment's line at time(s) ``t`` — also beyond the segment."""
50
+ return self.intercept + self.slope * np.asarray(t, dtype=np.float64)
51
+
52
+ def to_dict(self) -> dict[str, Any]:
53
+ return {
54
+ "index": self.index,
55
+ "start": self.start,
56
+ "stop": self.stop,
57
+ "n_points": self.n_points,
58
+ "trend": self.trend.value,
59
+ "slope": self.slope,
60
+ "intercept": self.intercept,
61
+ "sse": self.sse,
62
+ "t_statistic": _json_float(self.t_statistic),
63
+ }
64
+
65
+ @classmethod
66
+ def from_dict(cls, data: dict[str, Any]) -> Segment:
67
+ return cls(
68
+ index=int(data["index"]),
69
+ start=int(data["start"]),
70
+ stop=int(data["stop"]),
71
+ trend=Trend(data["trend"]),
72
+ slope=float(data["slope"]),
73
+ intercept=float(data["intercept"]),
74
+ sse=float(data["sse"]),
75
+ t_statistic=float(data["t_statistic"]),
76
+ )
77
+
78
+
79
+ @dataclass(frozen=True, slots=True)
80
+ class CriterionRow:
81
+ """Penalised criterion for one candidate number of change points."""
82
+
83
+ n_change_points: int
84
+ sse: float
85
+ penalty: float
86
+
87
+ @property
88
+ def objective(self) -> float:
89
+ return self.sse + self.penalty
90
+
91
+
92
+ @dataclass(frozen=True)
93
+ class TrendClusteringResult:
94
+ """Which point belongs to which trend cluster, plus everything needed to audit it.
95
+
96
+ ``segments`` are ordered in time and cover the series without gaps;
97
+ ``segments[-1]`` is the most recent regime — the one to train a regression on.
98
+ """
99
+
100
+ t: NDArray[np.float64]
101
+ y: NDArray[np.float64]
102
+ segments: tuple[Segment, ...]
103
+ noise_std: float
104
+ #: Criterion for every number of change points the search evaluated.
105
+ criterion: tuple[CriterionRow, ...]
106
+ #: Parameters of the :class:`~pytrendclust.TrendClusterer` that produced this result.
107
+ params: dict[str, Any] = field(default_factory=dict)
108
+
109
+ # -- structure ---------------------------------------------------------
110
+
111
+ @property
112
+ def n_points(self) -> int:
113
+ return len(self.y)
114
+
115
+ @property
116
+ def n_segments(self) -> int:
117
+ return len(self.segments)
118
+
119
+ @property
120
+ def n_change_points(self) -> int:
121
+ return len(self.segments) - 1
122
+
123
+ @property
124
+ def change_points(self) -> list[int]:
125
+ """Indices where a new segment starts (the first segment's start, 0, excluded)."""
126
+ return [s.start for s in self.segments[1:]]
127
+
128
+ @property
129
+ def labels(self) -> NDArray[np.int64]:
130
+ """Segment index of every point."""
131
+ return np.repeat(
132
+ np.arange(self.n_segments, dtype=np.int64), [s.n_points for s in self.segments]
133
+ )
134
+
135
+ @property
136
+ def trends(self) -> list[Trend]:
137
+ """Trend of the segment every point belongs to."""
138
+ return [s.trend for s in self.segments for _ in range(s.n_points)]
139
+
140
+ @property
141
+ def fitted(self) -> NDArray[np.float64]:
142
+ """Piecewise-linear fit: every point's value on its own segment's line."""
143
+ return np.concatenate([s.predict(self.t[s.start : s.stop]) for s in self.segments])
144
+
145
+ @property
146
+ def last_segment(self) -> Segment:
147
+ """The most recent regime of the series."""
148
+ return self.segments[-1]
149
+
150
+ @property
151
+ def objective(self) -> float:
152
+ """Value of the penalised criterion at the chosen segmentation."""
153
+ chosen = self.n_change_points
154
+ return next(r.objective for r in self.criterion if r.n_change_points == chosen)
155
+
156
+ # -- selection ---------------------------------------------------------
157
+
158
+ def indices(self, segment: int = -1) -> NDArray[np.int64]:
159
+ """Indices of the points of one segment; the last one by default."""
160
+ s = self.segments[segment]
161
+ return np.arange(s.start, s.stop, dtype=np.int64)
162
+
163
+ def select(self, segment: int = -1) -> tuple[NDArray[np.float64], NDArray[np.float64]]:
164
+ """``(t, y)`` of one segment's points; the last (most recent) one by default."""
165
+ s = self.segments[segment]
166
+ return self.t[s.start : s.stop].copy(), self.y[s.start : s.stop].copy()
167
+
168
+ # -- export ------------------------------------------------------------
169
+
170
+ def to_dict(self, *, include_points: bool = True) -> dict[str, Any]:
171
+ """Plain-Python representation, round-trippable through :meth:`from_dict`."""
172
+ data: dict[str, Any] = {
173
+ "format_version": FORMAT_VERSION,
174
+ "n_points": self.n_points,
175
+ "n_segments": self.n_segments,
176
+ "n_change_points": self.n_change_points,
177
+ "change_points": self.change_points,
178
+ "noise_std": self.noise_std,
179
+ "objective": self.objective,
180
+ "params": self.params,
181
+ "segments": [s.to_dict() for s in self.segments],
182
+ "criterion": [
183
+ {
184
+ "n_change_points": r.n_change_points,
185
+ "sse": r.sse,
186
+ "penalty": r.penalty,
187
+ "objective": r.objective,
188
+ }
189
+ for r in self.criterion
190
+ ],
191
+ }
192
+ if include_points:
193
+ data["points"] = {"t": self.t.tolist(), "y": self.y.tolist()}
194
+ return data
195
+
196
+ @classmethod
197
+ def from_dict(cls, data: dict[str, Any]) -> TrendClusteringResult:
198
+ """Rebuild a result exported with ``to_dict(include_points=True)``."""
199
+ if data.get("format_version") != FORMAT_VERSION:
200
+ msg = f"unsupported format_version: {data.get('format_version')!r}"
201
+ raise ValueError(msg)
202
+ if "points" not in data:
203
+ msg = "the export has no points; export it with include_points=True"
204
+ raise ValueError(msg)
205
+ return cls(
206
+ t=np.asarray(data["points"]["t"], dtype=np.float64),
207
+ y=np.asarray(data["points"]["y"], dtype=np.float64),
208
+ segments=tuple(Segment.from_dict(s) for s in data["segments"]),
209
+ noise_std=float(data["noise_std"]),
210
+ criterion=tuple(
211
+ CriterionRow(int(r["n_change_points"]), float(r["sse"]), float(r["penalty"]))
212
+ for r in data["criterion"]
213
+ ),
214
+ params=dict(data["params"]),
215
+ )
216
+
217
+ def to_json(self, path: str | Path | None = None, *, indent: int | None = 2) -> str:
218
+ """Serialise to JSON; also write it to ``path`` when given."""
219
+ text = json.dumps(self.to_dict(), ensure_ascii=False, indent=indent)
220
+ if path is not None:
221
+ Path(path).write_text(text, encoding="utf-8")
222
+ return text
223
+
224
+ @classmethod
225
+ def from_json(cls, source: str | Path) -> TrendClusteringResult:
226
+ """Load a result from a JSON file path or a JSON string."""
227
+ text = str(source)
228
+ if not text.lstrip().startswith("{"):
229
+ text = Path(source).read_text(encoding="utf-8")
230
+ return cls.from_dict(json.loads(text))
231
+
232
+ def point_table(self) -> list[dict[str, Any]]:
233
+ """One row per point: index, t, y, segment, trend, fitted value, residual."""
234
+ labels = self.labels
235
+ trends = self.trends
236
+ fitted = self.fitted
237
+ return [
238
+ {
239
+ "index": i,
240
+ "t": float(self.t[i]),
241
+ "y": float(self.y[i]),
242
+ "segment": int(labels[i]),
243
+ "trend": trends[i].value,
244
+ "fitted": float(fitted[i]),
245
+ "residual": float(self.y[i] - fitted[i]),
246
+ "is_last_segment": bool(labels[i] == self.n_segments - 1),
247
+ }
248
+ for i in range(self.n_points)
249
+ ]
250
+
251
+ def to_csv(self, path: str | Path | None = None) -> str:
252
+ """Per-point table as CSV; also write it to ``path`` when given."""
253
+ rows = self.point_table()
254
+ buffer = io.StringIO()
255
+ writer = csv.DictWriter(buffer, fieldnames=list(rows[0]), lineterminator="\n")
256
+ writer.writeheader()
257
+ writer.writerows(rows)
258
+ text = buffer.getvalue()
259
+ if path is not None:
260
+ Path(path).write_text(text, encoding="utf-8")
261
+ return text
262
+
263
+ def to_dataframe(self) -> pd.DataFrame:
264
+ """Per-point table as a :class:`pandas.DataFrame` (needs ``pytrendclust[pandas]``)."""
265
+ try:
266
+ import pandas as pd
267
+ except ImportError as exc:
268
+ msg = "to_dataframe() needs pandas: pip install 'pytrendclust[pandas]'"
269
+ raise ImportError(msg) from exc
270
+ return pd.DataFrame(self.point_table())
271
+
272
+ def export(self, path: str | Path) -> None:
273
+ """Write the result to ``path``; the format follows the suffix (``.json``/``.csv``)."""
274
+ suffix = Path(path).suffix.lower()
275
+ if suffix == ".json":
276
+ self.to_json(path)
277
+ elif suffix == ".csv":
278
+ self.to_csv(path)
279
+ else:
280
+ msg = f"unsupported export format {suffix!r}; use .json or .csv"
281
+ raise ValueError(msg)
282
+
283
+ def summary(self) -> str:
284
+ """Human-readable one-line-per-segment description."""
285
+ lines = [
286
+ (
287
+ f"{self.n_points} points, {self.n_change_points} change point(s), "
288
+ f"noise std {self.noise_std:.4g}"
289
+ )
290
+ ]
291
+ lines.extend(
292
+ f" #{s.index}: [{s.start}, {s.stop}) {s.n_points:>4} pts "
293
+ f"{s.trend.value:<4} slope {s.slope:+.4g}"
294
+ for s in self.segments
295
+ )
296
+ return "\n".join(lines)
297
+
298
+
299
+ def _json_float(value: float) -> float | str:
300
+ """JSON has no infinity; spell it as a string that ``float()`` parses back."""
301
+ if np.isinf(value):
302
+ return "inf" if value > 0 else "-inf"
303
+ return value