timeseries-anomaly 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- timeseries_anomaly-0.1.0/.gitignore +30 -0
- timeseries_anomaly-0.1.0/LICENSE +21 -0
- timeseries_anomaly-0.1.0/PKG-INFO +188 -0
- timeseries_anomaly-0.1.0/README.md +161 -0
- timeseries_anomaly-0.1.0/pyproject.toml +54 -0
- timeseries_anomaly-0.1.0/src/timeseries_anomaly/__init__.py +29 -0
- timeseries_anomaly-0.1.0/src/timeseries_anomaly/__main__.py +7 -0
- timeseries_anomaly-0.1.0/src/timeseries_anomaly/_core.py +477 -0
- timeseries_anomaly-0.1.0/src/timeseries_anomaly/_io.py +335 -0
- timeseries_anomaly-0.1.0/src/timeseries_anomaly/_methods.py +643 -0
- timeseries_anomaly-0.1.0/src/timeseries_anomaly/_result.py +301 -0
- timeseries_anomaly-0.1.0/src/timeseries_anomaly/_robust.py +210 -0
- timeseries_anomaly-0.1.0/src/timeseries_anomaly/cli.py +127 -0
- timeseries_anomaly-0.1.0/tests/test_cli.py +162 -0
- timeseries_anomaly-0.1.0/tests/test_detect.py +224 -0
- timeseries_anomaly-0.1.0/tests/test_detector.py +113 -0
- timeseries_anomaly-0.1.0/tests/test_edge_cases.py +203 -0
- timeseries_anomaly-0.1.0/tests/test_regressions.py +335 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Virtual environments and scratch space used while building. Never committed:
|
|
2
|
+
# they are large, machine-specific, and rebuilt from pyproject.toml anyway.
|
|
3
|
+
_venvs/
|
|
4
|
+
_proof/
|
|
5
|
+
.venv*/
|
|
6
|
+
|
|
7
|
+
# Build output. Wheels are built by the release workflow from this source, so a
|
|
8
|
+
# wheel in git could silently differ from the code beside it.
|
|
9
|
+
dist/
|
|
10
|
+
build/
|
|
11
|
+
*.egg-info/
|
|
12
|
+
src/*.egg-info/
|
|
13
|
+
|
|
14
|
+
# Python noise
|
|
15
|
+
__pycache__/
|
|
16
|
+
*.py[cod]
|
|
17
|
+
.pytest_cache/
|
|
18
|
+
.mypy_cache/
|
|
19
|
+
.ruff_cache/
|
|
20
|
+
|
|
21
|
+
# Local verification state, not source
|
|
22
|
+
verify_results.json
|
|
23
|
+
publish-log.txt
|
|
24
|
+
published.json
|
|
25
|
+
|
|
26
|
+
# Credentials. None of these belong here, and this line is the backstop.
|
|
27
|
+
.pypirc
|
|
28
|
+
.env
|
|
29
|
+
*.pem
|
|
30
|
+
*.key
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Pranay Mahendrakar
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: timeseries-anomaly
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Find anomalies in any time series or IoT signal with one call, no model training required
|
|
5
|
+
Project-URL: Homepage, https://pypi.org/project/timeseries-anomaly/
|
|
6
|
+
Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
|
|
7
|
+
Author: Pranay Mahendrakar
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: anomaly-detection,iot,monitoring,outlier-detection,pandas,seasonality,sensor,time-series
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Requires-Python: >=3.9
|
|
19
|
+
Requires-Dist: numpy>=1.23
|
|
20
|
+
Requires-Dist: pandas>=1.5
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: pyarrow>=12; extra == 'dev'
|
|
23
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
24
|
+
Provides-Extra: parquet
|
|
25
|
+
Requires-Dist: pyarrow>=12; extra == 'parquet'
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# timeseries-anomaly
|
|
29
|
+
|
|
30
|
+
Find the points in a time series or IoT signal that do not belong, with one call and no model training.
|
|
31
|
+
|
|
32
|
+
## Install
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
pip install timeseries-anomaly
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Reading `.parquet` files needs `pip install timeseries-anomaly[parquet]`.
|
|
39
|
+
|
|
40
|
+
## Quickstart
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
import timeseries_anomaly
|
|
44
|
+
|
|
45
|
+
readings = [10, 11, 10, 12, 11, 10, 11, 60, 10, 11, 12, 10, 11, 10, 12]
|
|
46
|
+
result = timeseries_anomaly.detect(readings)
|
|
47
|
+
print(result.summary())
|
|
48
|
+
print(result.anomalies)
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
```text
|
|
52
|
+
timeseries-anomaly: 1 anomaly in 15 points (6.7%)
|
|
53
|
+
series : value
|
|
54
|
+
method : zscore (asked for 'auto')
|
|
55
|
+
baseline : level 11, robust sigma 1.4826 (mad)
|
|
56
|
+
threshold : 3 robust sigmas from expected
|
|
57
|
+
flagged :
|
|
58
|
+
index 7: value 60, expected 11, 33.05 sigmas
|
|
59
|
+
warnings:
|
|
60
|
+
- method 'auto' chose 'zscore' because the series has only 15 points
|
|
61
|
+
[7]
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
It also takes a `pandas.Series` (with or without a `DatetimeIndex`), a `DataFrame`,
|
|
65
|
+
a numpy array, or a path to a `.csv` / `.parquet` file:
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
result = timeseries_anomaly.detect("sensor_log.csv", value="temperature", time="recorded_at")
|
|
69
|
+
result.to_frame().to_csv("scored.csv", index=False)
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## What it does
|
|
73
|
+
|
|
74
|
+
- Compares every point against what it should have been, and reports the gap in
|
|
75
|
+
**robust sigmas** - median and MAD based, so a handful of extreme points cannot
|
|
76
|
+
hide the rest.
|
|
77
|
+
- Picks the method for you (`method="auto"`), or runs every applicable method and
|
|
78
|
+
flags only what a strict majority agree on (`method="all"`).
|
|
79
|
+
- Handles seasonal signals: give it `seasonality=24`, or let it infer the cycle
|
|
80
|
+
from a `DatetimeIndex` or from the shape of the series itself.
|
|
81
|
+
- Never raises on awkward data. A constant series, an empty series, a single
|
|
82
|
+
point, missing values, unsorted or duplicated timestamps are all handled, and
|
|
83
|
+
whatever had to be adjusted is recorded in `result.warnings`.
|
|
84
|
+
- Never invents anomalies out of rounding error. When a series matches its
|
|
85
|
+
baseline exactly - a clean cycle, a steady ramp, a flat line - the spread
|
|
86
|
+
estimate steps down from MAD to standard deviation and then to reporting
|
|
87
|
+
nothing at all, instead of dividing by almost zero.
|
|
88
|
+
- Never reports a trend as a fault. A signal that is simply going somewhere - an
|
|
89
|
+
energy meter, a packet counter, an odometer - has its slope taken out before it
|
|
90
|
+
is judged, so every method measures the wobble around the climb rather than the
|
|
91
|
+
climb itself, and a spike on a steep slope is reported without dragging its
|
|
92
|
+
neighbours in with it.
|
|
93
|
+
- Covers the first and last reading like any other. Every point is compared
|
|
94
|
+
against its neighbours and never against itself, so an anomaly on the newest
|
|
95
|
+
sample - usually the one that matters - is reported under the default settings.
|
|
96
|
+
- Learns a baseline once and reuses it for later batches, for streaming use.
|
|
97
|
+
- Deterministic: the same input always gives the same answer, with no sampling
|
|
98
|
+
anywhere and no peeking at future points when scoring a stream.
|
|
99
|
+
|
|
100
|
+
### Methods
|
|
101
|
+
|
|
102
|
+
| `method` | Baseline each point is compared against |
|
|
103
|
+
|------------|------------------------------------------------------------------------------|
|
|
104
|
+
| `zscore` | the median of the whole series, spread from the MAD (modified z-score) |
|
|
105
|
+
| `iqr` | the same flat median, spread from the interquartile range |
|
|
106
|
+
| `rolling` | the median of the neighbours either side of it, window `max(7, n // 20)` |
|
|
107
|
+
| `seasonal` | a median trend plus the median of the other points sharing its phase |
|
|
108
|
+
| `ewma` | the exponential moving average of the points before it |
|
|
109
|
+
| `auto` | `rolling` for 30 or more points, otherwise `zscore` |
|
|
110
|
+
| `all` | every applicable method above, voting; anomaly when a strict majority agree |
|
|
111
|
+
|
|
112
|
+
Every method leaves the point it is explaining out of the baseline explaining it,
|
|
113
|
+
and takes the series' own slope out first. So a steadily rising signal does not
|
|
114
|
+
report itself as anomalous, an anomaly on the very first or very last reading is
|
|
115
|
+
reported like any other, and the spread each point is measured against is the
|
|
116
|
+
noise in the data rather than the shape of the fit.
|
|
117
|
+
|
|
118
|
+
`all` is the conservative choice: it only reports what more than half of the
|
|
119
|
+
applicable methods agree on. On a strongly trending series the two flat-baseline
|
|
120
|
+
methods (`zscore` and `iqr`) have nothing to say and can outvote the two that do,
|
|
121
|
+
in which case nothing is reported and `result.warnings` says that something was
|
|
122
|
+
flagged and overruled. Use `rolling` or `ewma` directly on that kind of signal.
|
|
123
|
+
|
|
124
|
+
`seasonal` infers its cycle from a `DatetimeIndex`, or from the shape of the
|
|
125
|
+
series, in `O(n log n)`; passing `seasonality=` explicitly is still a little
|
|
126
|
+
faster and removes any doubt about which cycle was used.
|
|
127
|
+
|
|
128
|
+
## API
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
detect(data, *, value=None, time=None, method="auto", sensitivity=3.0,
|
|
132
|
+
seasonality=None) -> AnomalyResult
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
- **data** - `pandas.Series`, `DataFrame`, list/array of numbers, or a `.csv` / `.parquet` path.
|
|
136
|
+
- **value**, **time** - column names, when `data` is a table. Guessed when omitted.
|
|
137
|
+
- **method** - one of the table above.
|
|
138
|
+
- **sensitivity** - the threshold in robust sigmas. Higher means fewer anomalies.
|
|
139
|
+
- **seasonality** - points per cycle for `seasonal`; inferred when omitted.
|
|
140
|
+
|
|
141
|
+
`AnomalyResult`:
|
|
142
|
+
|
|
143
|
+
| Member | What it gives you |
|
|
144
|
+
|--------|-------------------|
|
|
145
|
+
| `.anomalies` | `list[int]` positional indices of the flagged points |
|
|
146
|
+
| `.mask` | numpy bool array aligned to the input |
|
|
147
|
+
| `.scores` | numpy float array, robust sigmas from expected |
|
|
148
|
+
| `.expected` | numpy float array, the baseline each point was compared against |
|
|
149
|
+
| `.n_anomalies` / `.rate` | count, and the share of all points flagged |
|
|
150
|
+
| `.method_used` | the method that actually ran, after `auto` and any fallback |
|
|
151
|
+
| `.to_frame()` | `DataFrame(time, value, expected, score, is_anomaly)`, keeping your index |
|
|
152
|
+
| `.summary()` | the plain-text report above |
|
|
153
|
+
| `.to_dict()` | JSON-safe dict of everything, including `warnings` |
|
|
154
|
+
| `.plot_data()` | time-ordered lists ready for any plotting library |
|
|
155
|
+
| `.top(n)` | the worst points first, flagged or not |
|
|
156
|
+
| `.warnings` | what had to be adjusted: fallbacks, sorting, missing values |
|
|
157
|
+
|
|
158
|
+
`.mask`, `.scores`, `.expected` and `.values` all line up with the input exactly as
|
|
159
|
+
you passed it, even when the series had to be sorted by timestamp first.
|
|
160
|
+
|
|
161
|
+
`Detector` takes the same keyword options and keeps the baseline between batches:
|
|
162
|
+
|
|
163
|
+
```python
|
|
164
|
+
from timeseries_anomaly import Detector
|
|
165
|
+
|
|
166
|
+
detector = Detector(method="rolling", sensitivity=4.0).fit(history)
|
|
167
|
+
result = detector.score(new_batch) # same baseline, same threshold
|
|
168
|
+
print(detector.level_, detector.scale_, detector.method_)
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
`Detector.detect(data)` is the one-shot form, identical to `detect()`.
|
|
172
|
+
|
|
173
|
+
## CLI
|
|
174
|
+
|
|
175
|
+
```bash
|
|
176
|
+
timeseries-anomaly readings.csv
|
|
177
|
+
timeseries-anomaly readings.csv --value temperature --time recorded_at
|
|
178
|
+
timeseries-anomaly readings.csv --method seasonal --seasonality 24 --sensitivity 4
|
|
179
|
+
timeseries-anomaly readings.csv --json > anomalies.json
|
|
180
|
+
timeseries-anomaly readings.csv --output scored.csv
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
`--help` lists every option. `--json` prints `to_dict()`, `--output` writes the
|
|
184
|
+
scored table as `.csv` or `.parquet`.
|
|
185
|
+
|
|
186
|
+
## License
|
|
187
|
+
|
|
188
|
+
MIT
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
# timeseries-anomaly
|
|
2
|
+
|
|
3
|
+
Find the points in a time series or IoT signal that do not belong, with one call and no model training.
|
|
4
|
+
|
|
5
|
+
## Install
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install timeseries-anomaly
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
Reading `.parquet` files needs `pip install timeseries-anomaly[parquet]`.
|
|
12
|
+
|
|
13
|
+
## Quickstart
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
import timeseries_anomaly
|
|
17
|
+
|
|
18
|
+
readings = [10, 11, 10, 12, 11, 10, 11, 60, 10, 11, 12, 10, 11, 10, 12]
|
|
19
|
+
result = timeseries_anomaly.detect(readings)
|
|
20
|
+
print(result.summary())
|
|
21
|
+
print(result.anomalies)
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
```text
|
|
25
|
+
timeseries-anomaly: 1 anomaly in 15 points (6.7%)
|
|
26
|
+
series : value
|
|
27
|
+
method : zscore (asked for 'auto')
|
|
28
|
+
baseline : level 11, robust sigma 1.4826 (mad)
|
|
29
|
+
threshold : 3 robust sigmas from expected
|
|
30
|
+
flagged :
|
|
31
|
+
index 7: value 60, expected 11, 33.05 sigmas
|
|
32
|
+
warnings:
|
|
33
|
+
- method 'auto' chose 'zscore' because the series has only 15 points
|
|
34
|
+
[7]
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
It also takes a `pandas.Series` (with or without a `DatetimeIndex`), a `DataFrame`,
|
|
38
|
+
a numpy array, or a path to a `.csv` / `.parquet` file:
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
result = timeseries_anomaly.detect("sensor_log.csv", value="temperature", time="recorded_at")
|
|
42
|
+
result.to_frame().to_csv("scored.csv", index=False)
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## What it does
|
|
46
|
+
|
|
47
|
+
- Compares every point against what it should have been, and reports the gap in
|
|
48
|
+
**robust sigmas** - median and MAD based, so a handful of extreme points cannot
|
|
49
|
+
hide the rest.
|
|
50
|
+
- Picks the method for you (`method="auto"`), or runs every applicable method and
|
|
51
|
+
flags only what a strict majority agree on (`method="all"`).
|
|
52
|
+
- Handles seasonal signals: give it `seasonality=24`, or let it infer the cycle
|
|
53
|
+
from a `DatetimeIndex` or from the shape of the series itself.
|
|
54
|
+
- Never raises on awkward data. A constant series, an empty series, a single
|
|
55
|
+
point, missing values, unsorted or duplicated timestamps are all handled, and
|
|
56
|
+
whatever had to be adjusted is recorded in `result.warnings`.
|
|
57
|
+
- Never invents anomalies out of rounding error. When a series matches its
|
|
58
|
+
baseline exactly - a clean cycle, a steady ramp, a flat line - the spread
|
|
59
|
+
estimate steps down from MAD to standard deviation and then to reporting
|
|
60
|
+
nothing at all, instead of dividing by almost zero.
|
|
61
|
+
- Never reports a trend as a fault. A signal that is simply going somewhere - an
|
|
62
|
+
energy meter, a packet counter, an odometer - has its slope taken out before it
|
|
63
|
+
is judged, so every method measures the wobble around the climb rather than the
|
|
64
|
+
climb itself, and a spike on a steep slope is reported without dragging its
|
|
65
|
+
neighbours in with it.
|
|
66
|
+
- Covers the first and last reading like any other. Every point is compared
|
|
67
|
+
against its neighbours and never against itself, so an anomaly on the newest
|
|
68
|
+
sample - usually the one that matters - is reported under the default settings.
|
|
69
|
+
- Learns a baseline once and reuses it for later batches, for streaming use.
|
|
70
|
+
- Deterministic: the same input always gives the same answer, with no sampling
|
|
71
|
+
anywhere and no peeking at future points when scoring a stream.
|
|
72
|
+
|
|
73
|
+
### Methods
|
|
74
|
+
|
|
75
|
+
| `method` | Baseline each point is compared against |
|
|
76
|
+
|------------|------------------------------------------------------------------------------|
|
|
77
|
+
| `zscore` | the median of the whole series, spread from the MAD (modified z-score) |
|
|
78
|
+
| `iqr` | the same flat median, spread from the interquartile range |
|
|
79
|
+
| `rolling` | the median of the neighbours either side of it, window `max(7, n // 20)` |
|
|
80
|
+
| `seasonal` | a median trend plus the median of the other points sharing its phase |
|
|
81
|
+
| `ewma` | the exponential moving average of the points before it |
|
|
82
|
+
| `auto` | `rolling` for 30 or more points, otherwise `zscore` |
|
|
83
|
+
| `all` | every applicable method above, voting; anomaly when a strict majority agree |
|
|
84
|
+
|
|
85
|
+
Every method leaves the point it is explaining out of the baseline explaining it,
|
|
86
|
+
and takes the series' own slope out first. So a steadily rising signal does not
|
|
87
|
+
report itself as anomalous, an anomaly on the very first or very last reading is
|
|
88
|
+
reported like any other, and the spread each point is measured against is the
|
|
89
|
+
noise in the data rather than the shape of the fit.
|
|
90
|
+
|
|
91
|
+
`all` is the conservative choice: it only reports what more than half of the
|
|
92
|
+
applicable methods agree on. On a strongly trending series the two flat-baseline
|
|
93
|
+
methods (`zscore` and `iqr`) have nothing to say and can outvote the two that do,
|
|
94
|
+
in which case nothing is reported and `result.warnings` says that something was
|
|
95
|
+
flagged and overruled. Use `rolling` or `ewma` directly on that kind of signal.
|
|
96
|
+
|
|
97
|
+
`seasonal` infers its cycle from a `DatetimeIndex`, or from the shape of the
|
|
98
|
+
series, in `O(n log n)`; passing `seasonality=` explicitly is still a little
|
|
99
|
+
faster and removes any doubt about which cycle was used.
|
|
100
|
+
|
|
101
|
+
## API
|
|
102
|
+
|
|
103
|
+
```python
|
|
104
|
+
detect(data, *, value=None, time=None, method="auto", sensitivity=3.0,
|
|
105
|
+
seasonality=None) -> AnomalyResult
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
- **data** - `pandas.Series`, `DataFrame`, list/array of numbers, or a `.csv` / `.parquet` path.
|
|
109
|
+
- **value**, **time** - column names, when `data` is a table. Guessed when omitted.
|
|
110
|
+
- **method** - one of the table above.
|
|
111
|
+
- **sensitivity** - the threshold in robust sigmas. Higher means fewer anomalies.
|
|
112
|
+
- **seasonality** - points per cycle for `seasonal`; inferred when omitted.
|
|
113
|
+
|
|
114
|
+
`AnomalyResult`:
|
|
115
|
+
|
|
116
|
+
| Member | What it gives you |
|
|
117
|
+
|--------|-------------------|
|
|
118
|
+
| `.anomalies` | `list[int]` positional indices of the flagged points |
|
|
119
|
+
| `.mask` | numpy bool array aligned to the input |
|
|
120
|
+
| `.scores` | numpy float array, robust sigmas from expected |
|
|
121
|
+
| `.expected` | numpy float array, the baseline each point was compared against |
|
|
122
|
+
| `.n_anomalies` / `.rate` | count, and the share of all points flagged |
|
|
123
|
+
| `.method_used` | the method that actually ran, after `auto` and any fallback |
|
|
124
|
+
| `.to_frame()` | `DataFrame(time, value, expected, score, is_anomaly)`, keeping your index |
|
|
125
|
+
| `.summary()` | the plain-text report above |
|
|
126
|
+
| `.to_dict()` | JSON-safe dict of everything, including `warnings` |
|
|
127
|
+
| `.plot_data()` | time-ordered lists ready for any plotting library |
|
|
128
|
+
| `.top(n)` | the worst points first, flagged or not |
|
|
129
|
+
| `.warnings` | what had to be adjusted: fallbacks, sorting, missing values |
|
|
130
|
+
|
|
131
|
+
`.mask`, `.scores`, `.expected` and `.values` all line up with the input exactly as
|
|
132
|
+
you passed it, even when the series had to be sorted by timestamp first.
|
|
133
|
+
|
|
134
|
+
`Detector` takes the same keyword options and keeps the baseline between batches:
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
from timeseries_anomaly import Detector
|
|
138
|
+
|
|
139
|
+
detector = Detector(method="rolling", sensitivity=4.0).fit(history)
|
|
140
|
+
result = detector.score(new_batch) # same baseline, same threshold
|
|
141
|
+
print(detector.level_, detector.scale_, detector.method_)
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
`Detector.detect(data)` is the one-shot form, identical to `detect()`.
|
|
145
|
+
|
|
146
|
+
## CLI
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
timeseries-anomaly readings.csv
|
|
150
|
+
timeseries-anomaly readings.csv --value temperature --time recorded_at
|
|
151
|
+
timeseries-anomaly readings.csv --method seasonal --seasonality 24 --sensitivity 4
|
|
152
|
+
timeseries-anomaly readings.csv --json > anomalies.json
|
|
153
|
+
timeseries-anomaly readings.csv --output scored.csv
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
`--help` lists every option. `--json` prints `to_dict()`, `--output` writes the
|
|
157
|
+
scored table as `.csv` or `.parquet`.
|
|
158
|
+
|
|
159
|
+
## License
|
|
160
|
+
|
|
161
|
+
MIT
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.27"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "timeseries-anomaly"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Find anomalies in any time series or IoT signal with one call, no model training required"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Pranay Mahendrakar" }]
|
|
14
|
+
keywords = [
|
|
15
|
+
"anomaly-detection",
|
|
16
|
+
"time-series",
|
|
17
|
+
"outlier-detection",
|
|
18
|
+
"iot",
|
|
19
|
+
"sensor",
|
|
20
|
+
"monitoring",
|
|
21
|
+
"seasonality",
|
|
22
|
+
"pandas",
|
|
23
|
+
]
|
|
24
|
+
classifiers = [
|
|
25
|
+
"Development Status :: 4 - Beta",
|
|
26
|
+
"Intended Audience :: Developers",
|
|
27
|
+
"Intended Audience :: Science/Research",
|
|
28
|
+
"Programming Language :: Python :: 3",
|
|
29
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
30
|
+
"Operating System :: OS Independent",
|
|
31
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
32
|
+
]
|
|
33
|
+
dependencies = [
|
|
34
|
+
"pandas>=1.5",
|
|
35
|
+
"numpy>=1.23",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[project.optional-dependencies]
|
|
39
|
+
# heavy or niche deps go here, never in `dependencies`
|
|
40
|
+
parquet = ["pyarrow>=12"]
|
|
41
|
+
dev = ["pytest>=7", "pyarrow>=12"]
|
|
42
|
+
|
|
43
|
+
[project.scripts]
|
|
44
|
+
timeseries-anomaly = "timeseries_anomaly.cli:main"
|
|
45
|
+
|
|
46
|
+
[project.urls]
|
|
47
|
+
Homepage = "https://pypi.org/project/timeseries-anomaly/"
|
|
48
|
+
Author = "https://pypi.org/user/pranaymahendrakar/"
|
|
49
|
+
|
|
50
|
+
[tool.hatch.build.targets.wheel]
|
|
51
|
+
packages = ["src/timeseries_anomaly"]
|
|
52
|
+
|
|
53
|
+
[tool.pytest.ini_options]
|
|
54
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""timeseries-anomaly: find anomalies in any time series or IoT signal with one call.
|
|
2
|
+
|
|
3
|
+
No training, no model files, no configuration. Point it at numbers and it tells you
|
|
4
|
+
which ones do not belong, what they should have been, and how far off they were::
|
|
5
|
+
|
|
6
|
+
import timeseries_anomaly
|
|
7
|
+
|
|
8
|
+
result = timeseries_anomaly.detect(readings)
|
|
9
|
+
print(result.summary())
|
|
10
|
+
print(result.anomalies)
|
|
11
|
+
|
|
12
|
+
For data that arrives in batches, learn the baseline once and reuse it::
|
|
13
|
+
|
|
14
|
+
detector = timeseries_anomaly.Detector().fit(history)
|
|
15
|
+
result = detector.score(new_batch)
|
|
16
|
+
"""
|
|
17
|
+
from ._core import Detector, detect
|
|
18
|
+
from ._methods import METHODS
|
|
19
|
+
from ._result import AnomalyResult
|
|
20
|
+
|
|
21
|
+
__version__ = "0.1.0"
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"AnomalyResult",
|
|
25
|
+
"Detector",
|
|
26
|
+
"METHODS",
|
|
27
|
+
"detect",
|
|
28
|
+
"__version__",
|
|
29
|
+
]
|