ntpstats 2.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ntpstats/__init__.py ADDED
@@ -0,0 +1,11 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Copyright (c) 2012-2026 Thiago de Freitas <thiagodefreitas@gmail.com>
3
+ """ntpstats: NTP / network-time offset and stability analysis toolkit."""
4
+
5
+ __version__ = "2.5.0"
6
+ __author__ = "Thiago de Freitas"
7
+ __email__ = "thiagodefreitas@gmail.com"
8
+ __license__ = "MIT"
9
+
10
+ from .parsers import load, load_one # noqa: E402,F401
11
+ from .series import TimeSeries # noqa: E402,F401
ntpstats/__main__.py ADDED
@@ -0,0 +1,7 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Copyright (c) 2012-2026 Thiago de Freitas <thiagodefreitas@gmail.com>
3
+ import sys
4
+
5
+ from .cli import main
6
+
7
+ sys.exit(main())
ntpstats/analysis.py ADDED
@@ -0,0 +1,164 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Copyright (c) 2012-2026 Thiago de Freitas <thiagodefreitas@gmail.com>
3
+ """Descriptive statistics, detrending and outlier handling for offset series."""
4
+
5
+ from __future__ import annotations
6
+
7
+ from typing import Dict, Optional
8
+
9
+ import numpy as np
10
+
11
+ from .series import TimeSeries
12
+
13
+ PERCENTILES = (1, 5, 50, 95, 99)
14
+
15
+
16
+ def detrend(t: np.ndarray, x: np.ndarray, kind: Optional[str] = "linear") -> np.ndarray:
17
+ """Remove a least-squares polynomial trend.
18
+
19
+ ``linear`` removes a constant frequency offset; ``quadratic`` also
20
+ removes a linear frequency drift (aging). Time is centred first so the
21
+ fit stays well conditioned with POSIX timestamps.
22
+ """
23
+ if not kind or kind == "none":
24
+ return np.asarray(x, dtype=float)
25
+ deg = {"mean": 0, "linear": 1, "quadratic": 2}[kind]
26
+ t = np.asarray(t, dtype=float)
27
+ x = np.asarray(x, dtype=float)
28
+ if t.size <= deg:
29
+ return x - np.mean(x)
30
+ tc = t - t.mean()
31
+ coef = np.polyfit(tc, x, deg)
32
+ return x - np.polyval(coef, tc)
33
+
34
+
35
+ def linear_fit(t, x):
36
+ """Least-squares slope (s/s) and intercept of offset vs time."""
37
+ t = np.asarray(t, dtype=float)
38
+ tc = t - t.mean()
39
+ slope, icpt = np.polyfit(tc, np.asarray(x, dtype=float), 1)
40
+ return float(slope), float(icpt)
41
+
42
+
43
+ def theil_sen_slope(t, x, max_pairs: int = 200_000, seed: int = 0) -> float:
44
+ """Robust (median of pairwise slopes) frequency estimate, s/s.
45
+
46
+ Insensitive to up to ~29 % outliers, unlike least squares. Pairs are
47
+ randomly subsampled for long series.
48
+ """
49
+ t = np.asarray(t, dtype=float)
50
+ x = np.asarray(x, dtype=float)
51
+ n = t.size
52
+ if n < 2:
53
+ return float("nan")
54
+ if n * (n - 1) // 2 <= max_pairs:
55
+ i, j = np.triu_indices(n, 1)
56
+ else:
57
+ rng = np.random.default_rng(seed)
58
+ i = rng.integers(0, n, max_pairs)
59
+ j = rng.integers(0, n, max_pairs)
60
+ dt = t[j] - t[i]
61
+ ok = dt != 0
62
+ return float(np.median((x[j] - x[i])[ok] / dt[ok]))
63
+
64
+
65
+ def mad_outliers(x: np.ndarray, k: float = 5.0, t: Optional[np.ndarray] = None) -> np.ndarray:
66
+ """Boolean mask of outliers: |x - median| > k * 1.4826 * MAD.
67
+
68
+ If ``t`` is given the test is applied to linearly detrended data so a
69
+ frequency offset is not mistaken for outliers.
70
+ """
71
+ x = np.asarray(x, dtype=float)
72
+ r = detrend(t, x, "linear") if t is not None and x.size > 2 else x
73
+ med = np.median(r)
74
+ mad = 1.4826 * np.median(np.abs(r - med))
75
+ if mad == 0:
76
+ return np.zeros(x.shape, dtype=bool)
77
+ return np.abs(r - med) > k * mad
78
+
79
+
80
+ def remove_outliers(series: TimeSeries, k: float = 5.0) -> TimeSeries:
81
+ mask = mad_outliers(series.offset, k, series.t)
82
+ out = series.select(~mask)
83
+ out.meta["outliers_removed"] = int(mask.sum())
84
+ return out
85
+
86
+
87
+ def summary(series: TimeSeries) -> Dict[str, object]:
88
+ """Summary statistics in the spirit of NTPsec's ``ntpviz`` report."""
89
+ x = series.offset
90
+ n = len(series)
91
+ out: Dict[str, object] = {
92
+ "name": series.name,
93
+ "format": series.source_format,
94
+ "samples": n,
95
+ }
96
+ if n == 0:
97
+ return out
98
+ out.update(
99
+ {
100
+ "start": float(series.t[0]),
101
+ "end": float(series.t[-1]),
102
+ "span_s": series.span,
103
+ "mean": float(np.mean(x)),
104
+ "std": float(np.std(x, ddof=1)) if n > 1 else 0.0,
105
+ "rms": float(np.sqrt(np.mean(x * x))),
106
+ "min": float(np.min(x)),
107
+ "max": float(np.max(x)),
108
+ "mean_abs": float(np.mean(np.abs(x))),
109
+ }
110
+ )
111
+ pct = np.percentile(x, PERCENTILES)
112
+ out["percentiles"] = {f"p{p}": float(v) for p, v in zip(PERCENTILES, pct)}
113
+ out["range_90"] = float(pct[3] - pct[1]) # p95 - p5
114
+ out["range_98"] = float(pct[4] - pct[0]) # p99 - p1
115
+ if n > 2:
116
+ out["median_interval_s"] = series.median_interval()
117
+ out["regularity"] = series.regularity()
118
+ slope, _ = linear_fit(series.t, x)
119
+ out["offset_slope_ppm"] = slope * 1e6
120
+ out["offset_slope_robust_ppm"] = theil_sen_slope(series.t, x) * 1e6
121
+ resid = detrend(series.t, x, "linear")
122
+ out["residual_rms"] = float(np.sqrt(np.mean(resid ** 2)))
123
+ out["outliers_5mad"] = int(mad_outliers(x, 5.0, series.t).sum())
124
+ d = series.intervals()
125
+ med = series.median_interval()
126
+ out["gaps"] = int(np.sum(d > 3 * med))
127
+ out["longest_gap_s"] = float(d.max())
128
+ for key, col in series.extra.items():
129
+ if col.size and np.isfinite(col).any():
130
+ out[f"{key}_median"] = float(np.nanmedian(col))
131
+ return out
132
+
133
+
134
+ def format_seconds(v: Optional[float]) -> str:
135
+ """Pretty-print a time value with an SI prefix (e.g. 12.3 µs)."""
136
+ if v is None or not np.isfinite(v):
137
+ return "n/a"
138
+ a = abs(v)
139
+ for scale, unit in ((1, "s"), (1e-3, "ms"), (1e-6, "µs"), (1e-9, "ns")):
140
+ if a >= scale:
141
+ return f"{v / scale:.3g} {unit}"
142
+ return f"{v / 1e-12:.3g} ps"
143
+
144
+
145
+ def compare(estimate: TimeSeries, reference: TimeSeries) -> Dict[str, float]:
146
+ """Score ``estimate`` against a ``reference`` (e.g. simulated truth, a
147
+ GNSS/PPS-disciplined clock or a better server).
148
+
149
+ The reference is linearly interpolated onto the estimate's times over
150
+ their common span. Returns error statistics of ``estimate - reference``.
151
+ """
152
+ r = reference.sorted()
153
+ e = estimate.sorted().between(r.t[0], r.t[-1])
154
+ if len(e) < 2:
155
+ raise ValueError("series do not overlap")
156
+ err = e.offset - np.interp(e.t, r.t, r.offset)
157
+ return {
158
+ "samples": int(err.size),
159
+ "bias": float(np.mean(err)),
160
+ "rms": float(np.sqrt(np.mean(err ** 2))),
161
+ "std": float(np.std(err)),
162
+ "max_abs": float(np.max(np.abs(err))),
163
+ "p95_abs": float(np.percentile(np.abs(err), 95)),
164
+ }
ntpstats/bench.py ADDED
@@ -0,0 +1,157 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Copyright (c) 2012-2026 Thiago de Freitas <thiagodefreitas@gmail.com>
3
+ """Benchmark synchronisation estimators against simulated ground truth.
4
+
5
+ rows = run_bench(load_scenarios(["internet", "examples/scenarios/route-change.toml"]),
6
+ estimators=["kalman-dw", "regression", "rfc5905"], seeds=range(1, 6))
7
+ table = summarize(rows)
8
+
9
+ Scenarios are preset names (:data:`ntpstats.simulate.PRESETS`) or TOML/JSON
10
+ files (see :func:`ntpstats.simulate.scenario_from_dict`). Single-server
11
+ estimators are scored on the first server; multi-server estimators get all.
12
+
13
+ Metrics of the estimation error ``e = estimate - truth``: RMS, bias (mean),
14
+ p95 and max of |e|, MTIE of e over 1 h windows, and wall-clock runtime.
15
+ The first ``warmup`` seconds (default 30 min) are excluded, as usual when
16
+ comparing synchronisation algorithms, so start-up transients do not dominate.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import csv
22
+ import io
23
+ import json
24
+ import os
25
+ import time
26
+ from dataclasses import replace
27
+ from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
28
+
29
+ import numpy as np
30
+
31
+ from . import estimators as est_mod
32
+ from .simulate import PRESETS, Scenario, scenario_from_dict, simulate_multi
33
+
34
+ METRICS = ("rms", "bias", "p95_abs", "max_abs", "mtie_1h", "runtime_s")
35
+
36
+
37
+ def _load_file(path: str) -> dict:
38
+ with open(path, "rb") as fh:
39
+ raw = fh.read()
40
+ if path.endswith(".json"):
41
+ return json.loads(raw)
42
+ import importlib
43
+
44
+ try:
45
+ toml: Any = importlib.import_module("tomllib") # Python 3.11+
46
+ except ModuleNotFoundError: # pragma: no cover
47
+ try:
48
+ toml = importlib.import_module("tomli")
49
+ except ModuleNotFoundError as exc:
50
+ raise SystemExit("TOML scenarios need Python 3.11+ or `pip install tomli`") from exc
51
+ return toml.loads(raw.decode("utf-8"))
52
+
53
+
54
+ def load_scenarios(specs: Iterable[str]) -> List[Tuple[str, Scenario]]:
55
+ out = []
56
+ for spec in specs:
57
+ if spec in PRESETS:
58
+ out.append((spec, replace(PRESETS[spec], name=spec)))
59
+ elif os.path.exists(spec):
60
+ d = _load_file(spec)
61
+ name = d.get("name") or os.path.splitext(os.path.basename(spec))[0]
62
+ out.append((name, scenario_from_dict(dict(d, name=name))))
63
+ else:
64
+ raise ValueError(f"unknown scenario {spec!r} (presets: {', '.join(PRESETS)})")
65
+ return out
66
+
67
+
68
+ def _mtie(t: np.ndarray, e: np.ndarray, window: float) -> float:
69
+ if t.size < 2:
70
+ return float("nan")
71
+ j = np.searchsorted(t, t + window, side="right")
72
+ best = 0.0
73
+ for i in range(t.size):
74
+ seg = e[i: max(j[i], i + 1)]
75
+ best = max(best, float(seg.max() - seg.min()))
76
+ if j[i] >= t.size:
77
+ break
78
+ return best
79
+
80
+
81
+ def score(estimate, truth, warmup: float = 0.0) -> dict:
82
+ t = estimate.t
83
+ ok = (t >= truth.t[0] + warmup) & (t <= truth.t[-1]) & np.isfinite(estimate.offset)
84
+ t = t[ok]
85
+ e = estimate.offset[ok] - np.interp(t, truth.t, truth.offset)
86
+ if e.size == 0:
87
+ return {k: float("nan") for k in METRICS[:-1]}
88
+ return {
89
+ "rms": float(np.sqrt(np.mean(e * e))),
90
+ "bias": float(np.mean(e)),
91
+ "p95_abs": float(np.percentile(np.abs(e), 95)),
92
+ "max_abs": float(np.max(np.abs(e))),
93
+ "mtie_1h": _mtie(t, e, 3600.0),
94
+ }
95
+
96
+
97
+ def run_bench(scenarios: Sequence[Tuple[str, Scenario]], estimators: Optional[Sequence[str]] = None,
98
+ seeds: Iterable[int] = (1, 2, 3), duration: Optional[float] = None, progress=None,
99
+ warmup: float = 1800.0) -> List[dict]:
100
+ names = list(estimators) if estimators else sorted(est_mod.available())
101
+ ests = [est_mod.get(n) for n in names]
102
+ rows = []
103
+ for sc_name, sc in scenarios:
104
+ for seed in seeds:
105
+ s2 = replace(sc, seed=int(seed))
106
+ if duration:
107
+ s2 = replace(s2, duration=float(duration))
108
+ meas, truth = simulate_multi(s2, name=sc_name)
109
+ for e in ests:
110
+ if e.multi is False and not meas:
111
+ continue
112
+ t0 = time.perf_counter()
113
+ try:
114
+ out = est_mod.run(e, meas)
115
+ m = score(out, truth, warmup)
116
+ except Exception as exc: # a failing plugin must not abort the benchmark
117
+ m = {k: float("nan") for k in METRICS[:-1]}
118
+ m["error"] = f"{type(exc).__name__}: {exc}"
119
+ m["runtime_s"] = time.perf_counter() - t0
120
+ row = {"scenario": sc_name, "seed": int(seed), "estimator": e.name, "multi": e.multi,
121
+ "servers": len(meas), **m}
122
+ rows.append(row)
123
+ if progress:
124
+ progress(row)
125
+ return rows
126
+
127
+
128
+ def summarize(rows: Sequence[dict]) -> List[dict]:
129
+ """Mean and standard deviation of each metric over seeds."""
130
+ groups: Dict[tuple, List[dict]] = {}
131
+ for r in rows:
132
+ groups.setdefault((r["scenario"], r["estimator"]), []).append(r)
133
+ out = []
134
+ for (sc, est), rs in groups.items():
135
+ d = {"scenario": sc, "estimator": est, "runs": len(rs)}
136
+ for k in METRICS:
137
+ v = np.array([r[k] for r in rs], dtype=float)
138
+ d[k] = float(np.nanmean(v)) if np.isfinite(v).any() else float("nan")
139
+ d[k + "_std"] = float(np.nanstd(v)) if np.isfinite(v).sum() > 1 else 0.0
140
+ out.append(d)
141
+ out.sort(key=lambda d: (d["scenario"], d["rms"] if np.isfinite(d["rms"]) else np.inf))
142
+ return out
143
+
144
+
145
+ def to_csv(rows: Sequence[dict]) -> str:
146
+ if not rows:
147
+ return ""
148
+ keys = list(rows[0].keys())
149
+ for r in rows:
150
+ for k in r:
151
+ if k not in keys:
152
+ keys.append(k)
153
+ buf = io.StringIO()
154
+ w = csv.DictWriter(buf, fieldnames=keys)
155
+ w.writeheader()
156
+ w.writerows(rows)
157
+ return buf.getvalue()