timeseries-anomaly 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,30 @@
1
+ # Virtual environments and scratch space used while building. Never committed:
2
+ # they are large, machine-specific, and rebuilt from pyproject.toml anyway.
3
+ _venvs/
4
+ _proof/
5
+ .venv*/
6
+
7
+ # Build output. Wheels are built by the release workflow from this source, so a
8
+ # wheel in git could silently differ from the code beside it.
9
+ dist/
10
+ build/
11
+ *.egg-info/
12
+ src/*.egg-info/
13
+
14
+ # Python noise
15
+ __pycache__/
16
+ *.py[cod]
17
+ .pytest_cache/
18
+ .mypy_cache/
19
+ .ruff_cache/
20
+
21
+ # Local verification state, not source
22
+ verify_results.json
23
+ publish-log.txt
24
+ published.json
25
+
26
+ # Credentials. None of these belong here, and this line is the backstop.
27
+ .pypirc
28
+ .env
29
+ *.pem
30
+ *.key
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Pranay Mahendrakar
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,188 @@
1
+ Metadata-Version: 2.5
2
+ Name: timeseries-anomaly
3
+ Version: 0.1.0
4
+ Summary: Find anomalies in any time series or IoT signal with one call, no model training required
5
+ Project-URL: Homepage, https://pypi.org/project/timeseries-anomaly/
6
+ Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
7
+ Author: Pranay Mahendrakar
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: anomaly-detection,iot,monitoring,outlier-detection,pandas,seasonality,sensor,time-series
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
18
+ Requires-Python: >=3.9
19
+ Requires-Dist: numpy>=1.23
20
+ Requires-Dist: pandas>=1.5
21
+ Provides-Extra: dev
22
+ Requires-Dist: pyarrow>=12; extra == 'dev'
23
+ Requires-Dist: pytest>=7; extra == 'dev'
24
+ Provides-Extra: parquet
25
+ Requires-Dist: pyarrow>=12; extra == 'parquet'
26
+ Description-Content-Type: text/markdown
27
+
28
+ # timeseries-anomaly
29
+
30
+ Find the points in a time series or IoT signal that do not belong, with one call and no model training.
31
+
32
+ ## Install
33
+
34
+ ```bash
35
+ pip install timeseries-anomaly
36
+ ```
37
+
38
+ Reading `.parquet` files needs `pip install timeseries-anomaly[parquet]`.
39
+
40
+ ## Quickstart
41
+
42
+ ```python
43
+ import timeseries_anomaly
44
+
45
+ readings = [10, 11, 10, 12, 11, 10, 11, 60, 10, 11, 12, 10, 11, 10, 12]
46
+ result = timeseries_anomaly.detect(readings)
47
+ print(result.summary())
48
+ print(result.anomalies)
49
+ ```
50
+
51
+ ```text
52
+ timeseries-anomaly: 1 anomaly in 15 points (6.7%)
53
+ series : value
54
+ method : zscore (asked for 'auto')
55
+ baseline : level 11, robust sigma 1.4826 (mad)
56
+ threshold : 3 robust sigmas from expected
57
+ flagged :
58
+ index 7: value 60, expected 11, 33.05 sigmas
59
+ warnings:
60
+ - method 'auto' chose 'zscore' because the series has only 15 points
61
+ [7]
62
+ ```
63
+
64
+ It also takes a `pandas.Series` (with or without a `DatetimeIndex`), a `DataFrame`,
65
+ a numpy array, or a path to a `.csv` / `.parquet` file:
66
+
67
+ ```python
68
+ result = timeseries_anomaly.detect("sensor_log.csv", value="temperature", time="recorded_at")
69
+ result.to_frame().to_csv("scored.csv", index=False)
70
+ ```
71
+
72
+ ## What it does
73
+
74
+ - Compares every point against what it should have been, and reports the gap in
75
+ **robust sigmas** - median and MAD based, so a handful of extreme points cannot
76
+ hide the rest.
77
+ - Picks the method for you (`method="auto"`), or runs every applicable method and
78
+ flags only what a strict majority agree on (`method="all"`).
79
+ - Handles seasonal signals: give it `seasonality=24`, or let it infer the cycle
80
+ from a `DatetimeIndex` or from the shape of the series itself.
81
+ - Never raises on awkward data. A constant series, an empty series, a single
82
+ point, missing values, unsorted or duplicated timestamps are all handled, and
83
+ whatever had to be adjusted is recorded in `result.warnings`.
84
+ - Never invents anomalies out of rounding error. When a series matches its
85
+ baseline exactly - a clean cycle, a steady ramp, a flat line - the spread
86
+ estimate steps down from MAD to standard deviation and then to reporting
87
+ nothing at all, instead of dividing by almost zero.
88
+ - Never reports a trend as a fault. A signal that is simply going somewhere - an
89
+ energy meter, a packet counter, an odometer - has its slope taken out before it
90
+ is judged, so every method measures the wobble around the climb rather than the
91
+ climb itself, and a spike on a steep slope is reported without dragging its
92
+ neighbours in with it.
93
+ - Covers the first and last reading like any other. Every point is compared
94
+ against its neighbours and never against itself, so an anomaly on the newest
95
+ sample - usually the one that matters - is reported under the default settings.
96
+ - Learns a baseline once and reuses it for later batches, for streaming use.
97
+ - Deterministic: the same input always gives the same answer, with no sampling
98
+ anywhere and no peeking at future points when scoring a stream.
99
+
100
+ ### Methods
101
+
102
+ | `method` | Baseline each point is compared against |
103
+ |------------|------------------------------------------------------------------------------|
104
+ | `zscore` | the median of the whole series, spread from the MAD (modified z-score) |
105
+ | `iqr` | the same flat median, spread from the interquartile range |
106
+ | `rolling` | the median of the neighbours either side of it, window `max(7, n // 20)` |
107
+ | `seasonal` | a median trend plus the median of the other points sharing its phase |
108
+ | `ewma` | the exponential moving average of the points before it |
109
+ | `auto` | `rolling` for 30 or more points, otherwise `zscore` |
110
+ | `all` | every applicable method above, voting; anomaly when a strict majority agree |
111
+
112
+ Every method leaves the point it is explaining out of the baseline explaining it,
113
+ and takes the series' own slope out first. So a steadily rising signal does not
114
+ report itself as anomalous, an anomaly on the very first or very last reading is
115
+ reported like any other, and the spread each point is measured against is the
116
+ noise in the data rather than the shape of the fit.
117
+
118
+ `all` is the conservative choice: it only reports what more than half of the
119
+ applicable methods agree on. On a strongly trending series the two flat-baseline
120
+ methods (`zscore` and `iqr`) have nothing to say and can outvote the two that do,
121
+ in which case nothing is reported and `result.warnings` says that something was
122
+ flagged and overruled. Use `rolling` or `ewma` directly on that kind of signal.
123
+
124
+ `seasonal` infers its cycle from a `DatetimeIndex`, or from the shape of the
125
+ series, in `O(n log n)`; passing `seasonality=` explicitly is still a little
126
+ faster and removes any doubt about which cycle was used.
127
+
128
+ ## API
129
+
130
+ ```python
131
+ detect(data, *, value=None, time=None, method="auto", sensitivity=3.0,
132
+ seasonality=None) -> AnomalyResult
133
+ ```
134
+
135
+ - **data** - `pandas.Series`, `DataFrame`, list/array of numbers, or a `.csv` / `.parquet` path.
136
+ - **value**, **time** - column names, when `data` is a table. Guessed when omitted.
137
+ - **method** - one of the table above.
138
+ - **sensitivity** - the threshold in robust sigmas. Higher means fewer anomalies.
139
+ - **seasonality** - points per cycle for `seasonal`; inferred when omitted.
140
+
141
+ `AnomalyResult`:
142
+
143
+ | Member | What it gives you |
144
+ |--------|-------------------|
145
+ | `.anomalies` | `list[int]` positional indices of the flagged points |
146
+ | `.mask` | numpy bool array aligned to the input |
147
+ | `.scores` | numpy float array, robust sigmas from expected |
148
+ | `.expected` | numpy float array, the baseline each point was compared against |
149
+ | `.n_anomalies` / `.rate` | count, and the share of all points flagged |
150
+ | `.method_used` | the method that actually ran, after `auto` and any fallback |
151
+ | `.to_frame()` | `DataFrame(time, value, expected, score, is_anomaly)`, keeping your index |
152
+ | `.summary()` | the plain-text report above |
153
+ | `.to_dict()` | JSON-safe dict of everything, including `warnings` |
154
+ | `.plot_data()` | time-ordered lists ready for any plotting library |
155
+ | `.top(n)` | the worst points first, flagged or not |
156
+ | `.warnings` | what had to be adjusted: fallbacks, sorting, missing values |
157
+
158
+ `.mask`, `.scores`, `.expected` and `.values` all line up with the input exactly as
159
+ you passed it, even when the series had to be sorted by timestamp first.
160
+
161
+ `Detector` takes the same keyword options and keeps the baseline between batches:
162
+
163
+ ```python
164
+ from timeseries_anomaly import Detector
165
+
166
+ detector = Detector(method="rolling", sensitivity=4.0).fit(history)
167
+ result = detector.score(new_batch) # same baseline, same threshold
168
+ print(detector.level_, detector.scale_, detector.method_)
169
+ ```
170
+
171
+ `Detector.detect(data)` is the one-shot form, identical to `detect()`.
172
+
173
+ ## CLI
174
+
175
+ ```bash
176
+ timeseries-anomaly readings.csv
177
+ timeseries-anomaly readings.csv --value temperature --time recorded_at
178
+ timeseries-anomaly readings.csv --method seasonal --seasonality 24 --sensitivity 4
179
+ timeseries-anomaly readings.csv --json > anomalies.json
180
+ timeseries-anomaly readings.csv --output scored.csv
181
+ ```
182
+
183
+ `--help` lists every option. `--json` prints `to_dict()`, `--output` writes the
184
+ scored table as `.csv` or `.parquet`.
185
+
186
+ ## License
187
+
188
+ MIT
@@ -0,0 +1,161 @@
1
+ # timeseries-anomaly
2
+
3
+ Find the points in a time series or IoT signal that do not belong, with one call and no model training.
4
+
5
+ ## Install
6
+
7
+ ```bash
8
+ pip install timeseries-anomaly
9
+ ```
10
+
11
+ Reading `.parquet` files needs `pip install timeseries-anomaly[parquet]`.
12
+
13
+ ## Quickstart
14
+
15
+ ```python
16
+ import timeseries_anomaly
17
+
18
+ readings = [10, 11, 10, 12, 11, 10, 11, 60, 10, 11, 12, 10, 11, 10, 12]
19
+ result = timeseries_anomaly.detect(readings)
20
+ print(result.summary())
21
+ print(result.anomalies)
22
+ ```
23
+
24
+ ```text
25
+ timeseries-anomaly: 1 anomaly in 15 points (6.7%)
26
+ series : value
27
+ method : zscore (asked for 'auto')
28
+ baseline : level 11, robust sigma 1.4826 (mad)
29
+ threshold : 3 robust sigmas from expected
30
+ flagged :
31
+ index 7: value 60, expected 11, 33.05 sigmas
32
+ warnings:
33
+ - method 'auto' chose 'zscore' because the series has only 15 points
34
+ [7]
35
+ ```
36
+
37
+ It also takes a `pandas.Series` (with or without a `DatetimeIndex`), a `DataFrame`,
38
+ a numpy array, or a path to a `.csv` / `.parquet` file:
39
+
40
+ ```python
41
+ result = timeseries_anomaly.detect("sensor_log.csv", value="temperature", time="recorded_at")
42
+ result.to_frame().to_csv("scored.csv", index=False)
43
+ ```
44
+
45
+ ## What it does
46
+
47
+ - Compares every point against what it should have been, and reports the gap in
48
+ **robust sigmas** - median and MAD based, so a handful of extreme points cannot
49
+ hide the rest.
50
+ - Picks the method for you (`method="auto"`), or runs every applicable method and
51
+ flags only what a strict majority agree on (`method="all"`).
52
+ - Handles seasonal signals: give it `seasonality=24`, or let it infer the cycle
53
+ from a `DatetimeIndex` or from the shape of the series itself.
54
+ - Never raises on awkward data. A constant series, an empty series, a single
55
+ point, missing values, unsorted or duplicated timestamps are all handled, and
56
+ whatever had to be adjusted is recorded in `result.warnings`.
57
+ - Never invents anomalies out of rounding error. When a series matches its
58
+ baseline exactly - a clean cycle, a steady ramp, a flat line - the spread
59
+ estimate steps down from MAD to standard deviation and then to reporting
60
+ nothing at all, instead of dividing by almost zero.
61
+ - Never reports a trend as a fault. A signal that is simply going somewhere - an
62
+ energy meter, a packet counter, an odometer - has its slope taken out before it
63
+ is judged, so every method measures the wobble around the climb rather than the
64
+ climb itself, and a spike on a steep slope is reported without dragging its
65
+ neighbours in with it.
66
+ - Covers the first and last reading like any other. Every point is compared
67
+ against its neighbours and never against itself, so an anomaly on the newest
68
+ sample - usually the one that matters - is reported under the default settings.
69
+ - Learns a baseline once and reuses it for later batches, for streaming use.
70
+ - Deterministic: the same input always gives the same answer, with no sampling
71
+ anywhere and no peeking at future points when scoring a stream.
72
+
73
+ ### Methods
74
+
75
+ | `method` | Baseline each point is compared against |
76
+ |------------|------------------------------------------------------------------------------|
77
+ | `zscore` | the median of the whole series, spread from the MAD (modified z-score) |
78
+ | `iqr` | the same flat median, spread from the interquartile range |
79
+ | `rolling` | the median of the neighbours either side of it, window `max(7, n // 20)` |
80
+ | `seasonal` | a median trend plus the median of the other points sharing its phase |
81
+ | `ewma` | the exponential moving average of the points before it |
82
+ | `auto` | `rolling` for 30 or more points, otherwise `zscore` |
83
+ | `all` | every applicable method above, voting; anomaly when a strict majority agree |
84
+
85
+ Every method leaves the point it is explaining out of the baseline explaining it,
86
+ and takes the series' own slope out first. So a steadily rising signal does not
87
+ report itself as anomalous, an anomaly on the very first or very last reading is
88
+ reported like any other, and the spread each point is measured against is the
89
+ noise in the data rather than the shape of the fit.
90
+
91
+ `all` is the conservative choice: it only reports what more than half of the
92
+ applicable methods agree on. On a strongly trending series the two flat-baseline
93
+ methods (`zscore` and `iqr`) have nothing to say and can outvote the two that do,
94
+ in which case nothing is reported and `result.warnings` says that something was
95
+ flagged and overruled. Use `rolling` or `ewma` directly on that kind of signal.
96
+
97
+ `seasonal` infers its cycle from a `DatetimeIndex`, or from the shape of the
98
+ series, in `O(n log n)`; passing `seasonality=` explicitly is still a little
99
+ faster and removes any doubt about which cycle was used.
100
+
101
+ ## API
102
+
103
+ ```python
104
+ detect(data, *, value=None, time=None, method="auto", sensitivity=3.0,
105
+ seasonality=None) -> AnomalyResult
106
+ ```
107
+
108
+ - **data** - `pandas.Series`, `DataFrame`, list/array of numbers, or a `.csv` / `.parquet` path.
109
+ - **value**, **time** - column names, when `data` is a table. Guessed when omitted.
110
+ - **method** - one of the table above.
111
+ - **sensitivity** - the threshold in robust sigmas. Higher means fewer anomalies.
112
+ - **seasonality** - points per cycle for `seasonal`; inferred when omitted.
113
+
114
+ `AnomalyResult`:
115
+
116
+ | Member | What it gives you |
117
+ |--------|-------------------|
118
+ | `.anomalies` | `list[int]` positional indices of the flagged points |
119
+ | `.mask` | numpy bool array aligned to the input |
120
+ | `.scores` | numpy float array, robust sigmas from expected |
121
+ | `.expected` | numpy float array, the baseline each point was compared against |
122
+ | `.n_anomalies` / `.rate` | count, and the share of all points flagged |
123
+ | `.method_used` | the method that actually ran, after `auto` and any fallback |
124
+ | `.to_frame()` | `DataFrame(time, value, expected, score, is_anomaly)`, keeping your index |
125
+ | `.summary()` | the plain-text report above |
126
+ | `.to_dict()` | JSON-safe dict of everything, including `warnings` |
127
+ | `.plot_data()` | time-ordered lists ready for any plotting library |
128
+ | `.top(n)` | the worst points first, flagged or not |
129
+ | `.warnings` | what had to be adjusted: fallbacks, sorting, missing values |
130
+
131
+ `.mask`, `.scores`, `.expected` and `.values` all line up with the input exactly as
132
+ you passed it, even when the series had to be sorted by timestamp first.
133
+
134
+ `Detector` takes the same keyword options and keeps the baseline between batches:
135
+
136
+ ```python
137
+ from timeseries_anomaly import Detector
138
+
139
+ detector = Detector(method="rolling", sensitivity=4.0).fit(history)
140
+ result = detector.score(new_batch) # same baseline, same threshold
141
+ print(detector.level_, detector.scale_, detector.method_)
142
+ ```
143
+
144
+ `Detector.detect(data)` is the one-shot form, identical to `detect()`.
145
+
146
+ ## CLI
147
+
148
+ ```bash
149
+ timeseries-anomaly readings.csv
150
+ timeseries-anomaly readings.csv --value temperature --time recorded_at
151
+ timeseries-anomaly readings.csv --method seasonal --seasonality 24 --sensitivity 4
152
+ timeseries-anomaly readings.csv --json > anomalies.json
153
+ timeseries-anomaly readings.csv --output scored.csv
154
+ ```
155
+
156
+ `--help` lists every option. `--json` prints `to_dict()`, `--output` writes the
157
+ scored table as `.csv` or `.parquet`.
158
+
159
+ ## License
160
+
161
+ MIT
@@ -0,0 +1,54 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.27"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "timeseries-anomaly"
7
+ version = "0.1.0"
8
+ description = "Find anomalies in any time series or IoT signal with one call, no model training required"
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Pranay Mahendrakar" }]
14
+ keywords = [
15
+ "anomaly-detection",
16
+ "time-series",
17
+ "outlier-detection",
18
+ "iot",
19
+ "sensor",
20
+ "monitoring",
21
+ "seasonality",
22
+ "pandas",
23
+ ]
24
+ classifiers = [
25
+ "Development Status :: 4 - Beta",
26
+ "Intended Audience :: Developers",
27
+ "Intended Audience :: Science/Research",
28
+ "Programming Language :: Python :: 3",
29
+ "Programming Language :: Python :: 3 :: Only",
30
+ "Operating System :: OS Independent",
31
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
32
+ ]
33
+ dependencies = [
34
+ "pandas>=1.5",
35
+ "numpy>=1.23",
36
+ ]
37
+
38
+ [project.optional-dependencies]
39
+ # heavy or niche deps go here, never in `dependencies`
40
+ parquet = ["pyarrow>=12"]
41
+ dev = ["pytest>=7", "pyarrow>=12"]
42
+
43
+ [project.scripts]
44
+ timeseries-anomaly = "timeseries_anomaly.cli:main"
45
+
46
+ [project.urls]
47
+ Homepage = "https://pypi.org/project/timeseries-anomaly/"
48
+ Author = "https://pypi.org/user/pranaymahendrakar/"
49
+
50
+ [tool.hatch.build.targets.wheel]
51
+ packages = ["src/timeseries_anomaly"]
52
+
53
+ [tool.pytest.ini_options]
54
+ testpaths = ["tests"]
@@ -0,0 +1,29 @@
1
+ """timeseries-anomaly: find anomalies in any time series or IoT signal with one call.
2
+
3
+ No training, no model files, no configuration. Point it at numbers and it tells you
4
+ which ones do not belong, what they should have been, and how far off they were::
5
+
6
+ import timeseries_anomaly
7
+
8
+ result = timeseries_anomaly.detect(readings)
9
+ print(result.summary())
10
+ print(result.anomalies)
11
+
12
+ For data that arrives in batches, learn the baseline once and reuse it::
13
+
14
+ detector = timeseries_anomaly.Detector().fit(history)
15
+ result = detector.score(new_batch)
16
+ """
17
+ from ._core import Detector, detect
18
+ from ._methods import METHODS
19
+ from ._result import AnomalyResult
20
+
21
+ __version__ = "0.1.0"
22
+
23
+ __all__ = [
24
+ "AnomalyResult",
25
+ "Detector",
26
+ "METHODS",
27
+ "detect",
28
+ "__version__",
29
+ ]
@@ -0,0 +1,7 @@
1
+ """Allow ``python -m timeseries_anomaly``."""
2
+ import sys
3
+
4
+ from .cli import main
5
+
6
+ if __name__ == "__main__":
7
+ sys.exit(main())