sensorlint 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sensorlint/__init__.py ADDED
@@ -0,0 +1,162 @@
1
+ """sensorlint: assertions for sensor-data pipelines that abort instead of warning.
2
+
3
+ Sensor pipelines do not usually fail by crashing. They fail by producing
4
+ believable numbers. A truncated payload becomes a short file. Raw converter
5
+ counts become a physical measurement off by a constant factor. A swapped pair
6
+ of channels becomes a trend on the wrong location. A stage that processed zero
7
+ records becomes a green build.
8
+
9
+ Every check here raises. There is no warning mode, because a warning in a
10
+ batch job is a log line nobody reads.
11
+
12
+ Quickstart::
13
+
14
+ import numpy as np
15
+ from sensorlint import assert_expected_length, assert_not_clipped, safe_gunzip
16
+
17
+ payload = safe_gunzip(raw_bytes) # refuses to return partial data
18
+ samples = np.frombuffer(payload, dtype=np.int16)
19
+ assert_expected_length(samples, 20480)
20
+ assert_not_clipped(samples, adc_max=32767, leading_samples=2048)
21
+
22
+ Every ``assert_*`` has a ``check_*`` twin that returns a
23
+ :class:`~sensorlint.result.CheckResult` instead of raising, so results can be
24
+ aggregated across a fleet with :mod:`sensorlint.report`.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ from sensorlint.checks import (
30
+ DecodeProbe,
31
+ RawCountsVerdict,
32
+ RunLengthReport,
33
+ StageCounter,
34
+ TranspositionVerdict,
35
+ assert_channels_not_transposed,
36
+ assert_coverage,
37
+ assert_decoded_fully,
38
+ assert_expected_length,
39
+ assert_nonzero_kept,
40
+ assert_not_clipped,
41
+ assert_not_stale,
42
+ assert_sample_rate,
43
+ assert_scale_declared,
44
+ autodetect_rails,
45
+ check_channel_transposition,
46
+ check_coverage,
47
+ check_decoded_fully,
48
+ check_expected_length,
49
+ check_nonzero_kept,
50
+ check_not_clipped,
51
+ check_not_stale,
52
+ check_sample_rate,
53
+ check_scale_declared,
54
+ coverage_histogram,
55
+ detect_channel_transposition,
56
+ estimate_expected_ratio_db,
57
+ estimate_sample_rate,
58
+ longest_frozen_run,
59
+ longest_interpolated_run,
60
+ looks_like_raw_counts,
61
+ measure_length,
62
+ nonzero_kept,
63
+ pipeline_stage,
64
+ probe_compressed,
65
+ safe_gunzip,
66
+ sniff_container,
67
+ transposition_run_lengths,
68
+ )
69
+ from sensorlint.errors import (
70
+ ClippingError,
71
+ CoverageError,
72
+ DecodeError,
73
+ EmptyStageError,
74
+ LengthError,
75
+ SampleRateError,
76
+ ScaleError,
77
+ SensorLintError,
78
+ StalenessError,
79
+ TranspositionError,
80
+ )
81
+ from sensorlint.report import (
82
+ ReportSummary,
83
+ format_report,
84
+ format_table,
85
+ summarize,
86
+ worst_offenders,
87
+ )
88
+ from sensorlint.result import CheckResult, Severity
89
+
90
+ __version__ = "0.1.0"
91
+
92
+ __all__ = [
93
+ # result and reporting
94
+ "CheckResult",
95
+ "ReportSummary",
96
+ "Severity",
97
+ "format_report",
98
+ "format_table",
99
+ "summarize",
100
+ "worst_offenders",
101
+ # errors
102
+ "ClippingError",
103
+ "CoverageError",
104
+ "DecodeError",
105
+ "EmptyStageError",
106
+ "LengthError",
107
+ "SampleRateError",
108
+ "ScaleError",
109
+ "SensorLintError",
110
+ "StalenessError",
111
+ "TranspositionError",
112
+ # supporting types
113
+ "DecodeProbe",
114
+ "RawCountsVerdict",
115
+ "RunLengthReport",
116
+ "StageCounter",
117
+ "TranspositionVerdict",
118
+ # 1. decode
119
+ "assert_decoded_fully",
120
+ "check_decoded_fully",
121
+ "probe_compressed",
122
+ "safe_gunzip",
123
+ "sniff_container",
124
+ # 2. length
125
+ "assert_expected_length",
126
+ "check_expected_length",
127
+ "measure_length",
128
+ # 3. clipping
129
+ "assert_not_clipped",
130
+ "autodetect_rails",
131
+ "check_not_clipped",
132
+ # 4. scale
133
+ "assert_scale_declared",
134
+ "check_scale_declared",
135
+ "looks_like_raw_counts",
136
+ # 5. staleness
137
+ "assert_not_stale",
138
+ "check_not_stale",
139
+ "longest_frozen_run",
140
+ "longest_interpolated_run",
141
+ # 6. sample rate
142
+ "assert_sample_rate",
143
+ "check_sample_rate",
144
+ "estimate_sample_rate",
145
+ # 7. transposition
146
+ "assert_channels_not_transposed",
147
+ "check_channel_transposition",
148
+ "detect_channel_transposition",
149
+ "estimate_expected_ratio_db",
150
+ "transposition_run_lengths",
151
+ # 8. coverage
152
+ "assert_coverage",
153
+ "check_coverage",
154
+ "coverage_histogram",
155
+ # 9. nonzero kept
156
+ "assert_nonzero_kept",
157
+ "check_nonzero_kept",
158
+ "nonzero_kept",
159
+ "pipeline_stage",
160
+ # version
161
+ "__version__",
162
+ ]
sensorlint/__main__.py ADDED
@@ -0,0 +1,377 @@
1
+ """Command-line entry point: ``python -m sensorlint check <file>``.
2
+
3
+ Deliberately small. The CLI exists so a check can be dropped into a shell
4
+ pipeline or a CI step without writing Python, and so that the exit code
5
+ carries the verdict. It runs whichever checks the flags you passed make
6
+ applicable, prints a report, and exits non-zero if anything failed.
7
+
8
+ If a flag is not supplied, the corresponding check is skipped and the report
9
+ says so. It does not guess thresholds on your behalf.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import argparse
15
+ import json
16
+ import sys
17
+ from pathlib import Path
18
+ from typing import Any, List, Optional, Sequence, Tuple
19
+
20
+ import numpy as np
21
+
22
+ from sensorlint import __version__
23
+ from sensorlint.checks.clipping import check_not_clipped
24
+ from sensorlint.checks.length import check_expected_length
25
+ from sensorlint.checks.sample_rate import check_sample_rate
26
+ from sensorlint.checks.scale import check_scale_declared
27
+ from sensorlint.checks.staleness import check_not_stale
28
+ from sensorlint.report import format_report, summarize
29
+ from sensorlint.result import CheckResult, Severity
30
+
31
+ __all__ = ["build_parser", "load_series", "main", "run_checks"]
32
+
33
+ _EXIT_OK = 0
34
+ _EXIT_FAILED = 1
35
+ _EXIT_USAGE = 2
36
+
37
+
38
+ def load_series(
39
+ path: Path,
40
+ *,
41
+ column: int = 0,
42
+ timestamp_column: Optional[int] = None,
43
+ delimiter: str = ",",
44
+ skip_header: int = 0,
45
+ ) -> Tuple[Any, Optional[Any]]:
46
+ """Load ``(values, timestamps)`` from a ``.npy`` or delimited text file.
47
+
48
+ Args:
49
+ path: File to read. ``.npy`` is loaded with :func:`numpy.load`;
50
+ anything else is parsed as delimited text.
51
+ column: Which column holds the sample values, for 2-D input.
52
+ timestamp_column: Which column holds timestamps, if any.
53
+ delimiter: Field separator for text files.
54
+ skip_header: Header rows to skip in text files.
55
+
56
+ Returns:
57
+ ``(values, timestamps)`` where ``timestamps`` is ``None`` unless
58
+ ``timestamp_column`` was given.
59
+
60
+ Raises:
61
+ ValueError: if the file cannot be interpreted as a numeric series,
62
+ or the requested columns do not exist.
63
+ """
64
+ if path.suffix.lower() == ".npy":
65
+ data = np.load(path, allow_pickle=False)
66
+ else:
67
+ data = np.genfromtxt(
68
+ path, delimiter=delimiter, skip_header=skip_header, dtype=np.float64
69
+ )
70
+
71
+ array = np.asarray(data)
72
+ if array.ndim == 0:
73
+ raise ValueError(f"{path} holds a scalar, not a series")
74
+
75
+ if array.ndim == 1:
76
+ if timestamp_column is not None:
77
+ raise ValueError(
78
+ f"{path} has one column; --timestamp-column needs a 2-D file"
79
+ )
80
+ if column != 0:
81
+ raise ValueError(f"{path} has one column; --column {column} is out of range")
82
+ return array, None
83
+
84
+ if array.ndim > 2:
85
+ raise ValueError(f"{path} has {array.ndim} dimensions; expected 1 or 2")
86
+
87
+ n_cols = array.shape[1]
88
+ if not 0 <= column < n_cols:
89
+ raise ValueError(f"--column {column} is out of range for {n_cols} column(s)")
90
+ values = array[:, column]
91
+
92
+ timestamps = None
93
+ if timestamp_column is not None:
94
+ if not 0 <= timestamp_column < n_cols:
95
+ raise ValueError(
96
+ f"--timestamp-column {timestamp_column} is out of range for "
97
+ f"{n_cols} column(s)"
98
+ )
99
+ timestamps = array[:, timestamp_column]
100
+
101
+ return values, timestamps
102
+
103
+
104
+ def run_checks(args: argparse.Namespace) -> List[CheckResult]:
105
+ """Run every check the supplied flags make applicable.
106
+
107
+ Returns:
108
+ The results, in the order the checks ran.
109
+ """
110
+ path = Path(args.path)
111
+ values, timestamps = load_series(
112
+ path,
113
+ column=args.column,
114
+ timestamp_column=args.timestamp_column,
115
+ delimiter=args.delimiter,
116
+ skip_header=args.skip_header,
117
+ )
118
+ target = path.name
119
+ results: List[CheckResult] = []
120
+
121
+ if args.expected_length is not None:
122
+ results.append(
123
+ check_expected_length(
124
+ values,
125
+ args.expected_length,
126
+ args.length_tolerance,
127
+ target=target,
128
+ )
129
+ )
130
+
131
+ if args.adc_max is not None or args.adc_min is not None or args.autodetect_rails:
132
+ results.append(
133
+ check_not_clipped(
134
+ values,
135
+ adc_max=args.adc_max,
136
+ adc_min=args.adc_min,
137
+ max_fraction=args.max_clipped_fraction,
138
+ leading_samples=args.leading_samples,
139
+ leading_max_fraction=args.leading_max_fraction,
140
+ target=target,
141
+ )
142
+ )
143
+
144
+ if not args.no_scale_check:
145
+ results.append(
146
+ check_scale_declared(
147
+ values, args.scale, args.unit, target=target
148
+ )
149
+ )
150
+
151
+ if not args.no_stale_check:
152
+ results.append(
153
+ check_not_stale(
154
+ values,
155
+ max_frozen_run=args.max_frozen_run,
156
+ max_interpolated_run=args.max_interpolated_run,
157
+ target=target,
158
+ )
159
+ )
160
+
161
+ if args.fs is not None:
162
+ if timestamps is None:
163
+ # No timestamp column: synthesise the times the claimed rate
164
+ # implies. That makes the rate check trivially true, so it is
165
+ # skipped rather than faked.
166
+ results.append(
167
+ CheckResult(
168
+ check="sample_rate",
169
+ passed=True,
170
+ severity=Severity.WARNING,
171
+ message=(
172
+ f"skipped: --fs {args.fs} given but no --timestamp-column, "
173
+ "so there are no observed intervals to compare against"
174
+ ),
175
+ details={"skipped": True, "claimed_fs": args.fs},
176
+ target=target,
177
+ )
178
+ )
179
+ else:
180
+ results.append(
181
+ check_sample_rate(
182
+ timestamps,
183
+ args.fs,
184
+ tolerance_pct=args.fs_tolerance_pct,
185
+ allow_gaps=args.allow_gaps,
186
+ unit=args.time_unit,
187
+ target=target,
188
+ )
189
+ )
190
+
191
+ return results
192
+
193
+
194
+ def build_parser() -> argparse.ArgumentParser:
195
+ """Construct the argument parser."""
196
+ parser = argparse.ArgumentParser(
197
+ prog="sensorlint",
198
+ description=(
199
+ "Assertions for sensor-data pipelines. Every check aborts rather "
200
+ "than warns."
201
+ ),
202
+ )
203
+ parser.add_argument("--version", action="version", version=f"sensorlint {__version__}")
204
+ sub = parser.add_subparsers(dest="command", required=True)
205
+
206
+ check = sub.add_parser(
207
+ "check",
208
+ help="run the applicable checks against a .npy or delimited text file",
209
+ description=(
210
+ "Runs the checks your flags make applicable and exits non-zero if any "
211
+ "fail. Checks with no corresponding flag are skipped rather than run "
212
+ "with a guessed threshold."
213
+ ),
214
+ )
215
+ check.add_argument("path", help="path to a .npy or delimited text file")
216
+
217
+ load = check.add_argument_group("loading")
218
+ load.add_argument(
219
+ "--column", type=int, default=0, help="column holding sample values (default 0)"
220
+ )
221
+ load.add_argument(
222
+ "--timestamp-column",
223
+ type=int,
224
+ default=None,
225
+ help="column holding timestamps; required for --fs to do anything",
226
+ )
227
+ load.add_argument("--delimiter", default=",", help="text field separator (default ,)")
228
+ load.add_argument(
229
+ "--skip-header", type=int, default=0, help="header rows to skip in text files"
230
+ )
231
+ load.add_argument(
232
+ "--time-unit",
233
+ default="s",
234
+ choices=["s", "ms", "us", "ns"],
235
+ help="unit of the timestamp column (default s)",
236
+ )
237
+
238
+ length = check.add_argument_group("length")
239
+ length.add_argument(
240
+ "--expected-length", type=int, default=None, help="required sample count"
241
+ )
242
+ length.add_argument(
243
+ "--length-tolerance",
244
+ type=int,
245
+ default=0,
246
+ help="allowed absolute deviation in samples (default 0)",
247
+ )
248
+
249
+ clip = check.add_argument_group("clipping")
250
+ clip.add_argument("--adc-max", type=float, default=None, help="positive rail value")
251
+ clip.add_argument("--adc-min", type=float, default=None, help="negative rail value")
252
+ clip.add_argument(
253
+ "--autodetect-rails",
254
+ action="store_true",
255
+ help="run the clipping check with rails inferred from repeated extremes",
256
+ )
257
+ clip.add_argument(
258
+ "--max-clipped-fraction",
259
+ type=float,
260
+ default=0.001,
261
+ help="whole-array railed fraction allowed (default 0.001)",
262
+ )
263
+ clip.add_argument(
264
+ "--leading-samples",
265
+ type=int,
266
+ default=None,
267
+ help="also test this many leading samples separately",
268
+ )
269
+ clip.add_argument(
270
+ "--leading-max-fraction",
271
+ type=float,
272
+ default=None,
273
+ help="railed fraction allowed in the leading window",
274
+ )
275
+
276
+ scale = check.add_argument_group("scale")
277
+ scale.add_argument("--scale", type=float, default=None, help="counts-to-physical factor")
278
+ scale.add_argument("--unit", default=None, help="physical unit, e.g. g or mm/s")
279
+ scale.add_argument(
280
+ "--no-scale-check", action="store_true", help="skip the scale-declared check"
281
+ )
282
+
283
+ stale = check.add_argument_group("staleness")
284
+ stale.add_argument(
285
+ "--max-frozen-run", type=int, default=10, help="longest repeated-value run allowed"
286
+ )
287
+ stale.add_argument(
288
+ "--max-interpolated-run",
289
+ type=int,
290
+ default=12,
291
+ help="longest constant-slope run allowed",
292
+ )
293
+ stale.add_argument(
294
+ "--no-stale-check", action="store_true", help="skip the staleness check"
295
+ )
296
+
297
+ rate = check.add_argument_group("sample rate")
298
+ rate.add_argument("--fs", type=float, default=None, help="claimed sampling rate in Hz")
299
+ rate.add_argument(
300
+ "--fs-tolerance-pct",
301
+ type=float,
302
+ default=1.0,
303
+ help="allowed rate error, percent (default 1.0)",
304
+ )
305
+ rate.add_argument(
306
+ "--allow-gaps", action="store_true", help="tolerate gaps in the timestamp series"
307
+ )
308
+
309
+ out = check.add_argument_group("output")
310
+ out.add_argument("--json", action="store_true", help="emit JSON instead of a table")
311
+ out.add_argument(
312
+ "--quiet", action="store_true", help="print nothing; rely on the exit code"
313
+ )
314
+
315
+ return parser
316
+
317
+
318
+ def main(argv: Optional[Sequence[str]] = None) -> int:
319
+ """Run the CLI.
320
+
321
+ Returns:
322
+ ``0`` when every applicable check passed, ``1`` when any failed,
323
+ ``2`` on a usage or loading error.
324
+ """
325
+ parser = build_parser()
326
+ args = parser.parse_args(argv)
327
+
328
+ try:
329
+ results = run_checks(args)
330
+ except (OSError, ValueError, TypeError) as exc:
331
+ print(f"sensorlint: {exc}", file=sys.stderr)
332
+ return _EXIT_USAGE
333
+
334
+ summary = summarize(results)
335
+
336
+ if args.json:
337
+ payload = {
338
+ "summary": summary.to_dict(),
339
+ "results": [r.to_dict() for r in results],
340
+ }
341
+ print(json.dumps(payload, indent=2, default=_json_default))
342
+ elif not args.quiet:
343
+ print(format_report(results, title=f"sensorlint {Path(args.path).name}"))
344
+ # A skipped check counts as a pass in the table, which would let it
345
+ # hide inside a green run. Say so explicitly instead.
346
+ skipped = [r for r in results if r.details.get("skipped")]
347
+ if skipped:
348
+ print("")
349
+ print("skipped:")
350
+ for item in skipped:
351
+ print(f" {item.check}: {item.message.removeprefix('skipped: ')}")
352
+
353
+ if not results and not args.quiet and not args.json:
354
+ print(
355
+ "\nno checks ran: pass at least one of --expected-length, --adc-max, "
356
+ "--autodetect-rails or --fs",
357
+ file=sys.stderr,
358
+ )
359
+
360
+ return _EXIT_OK if summary.clean else _EXIT_FAILED
361
+
362
+
363
+ def _json_default(value: Any) -> Any:
364
+ """Coerce numpy scalars so ``json.dumps`` can serialise check details."""
365
+ if isinstance(value, (np.integer,)):
366
+ return int(value)
367
+ if isinstance(value, (np.floating,)):
368
+ return float(value)
369
+ if isinstance(value, (np.bool_,)):
370
+ return bool(value)
371
+ if isinstance(value, np.ndarray):
372
+ return value.tolist()
373
+ return str(value)
374
+
375
+
376
+ if __name__ == "__main__": # pragma: no cover
377
+ raise SystemExit(main())