sensor-modeling 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sensor_modeling/__init__.py +45 -0
- sensor_modeling/alerts/__init__.py +26 -0
- sensor_modeling/alerts/alert.py +532 -0
- sensor_modeling/analysis/__init__.py +43 -0
- sensor_modeling/analysis/_frame.py +19 -0
- sensor_modeling/analysis/behavioral_analysis.py +57 -0
- sensor_modeling/analysis/behavioral_metrics.py +66 -0
- sensor_modeling/analysis/comparison.py +164 -0
- sensor_modeling/analysis/dependency_network.py +408 -0
- sensor_modeling/analysis/granger_causality.py +314 -0
- sensor_modeling/analysis/pipeline.py +168 -0
- sensor_modeling/analysis/reporting.py +109 -0
- sensor_modeling/baseline/__init__.py +30 -0
- sensor_modeling/baseline/adaptive.py +520 -0
- sensor_modeling/baseline/features.py +224 -0
- sensor_modeling/change_point/__init__.py +13 -0
- sensor_modeling/change_point/_validation.py +31 -0
- sensor_modeling/change_point/adaptive_normalization.py +55 -0
- sensor_modeling/change_point/embedding_cpd.py +60 -0
- sensor_modeling/change_point/energy_efficient.py +57 -0
- sensor_modeling/change_point/genetic_optimization.py +65 -0
- sensor_modeling/cli.py +416 -0
- sensor_modeling/context/__init__.py +33 -0
- sensor_modeling/context/occupancy.py +529 -0
- sensor_modeling/data/__init__.py +5 -0
- sensor_modeling/data/loaders.py +146 -0
- sensor_modeling/data/preprocessing.py +83 -0
- sensor_modeling/data/synthetic.py +121 -0
- sensor_modeling/data/validation.py +81 -0
- sensor_modeling/evaluation/__init__.py +92 -0
- sensor_modeling/evaluation/ablation.py +303 -0
- sensor_modeling/evaluation/attribution.py +474 -0
- sensor_modeling/evaluation/detection.py +297 -0
- sensor_modeling/evaluation/metrics.py +541 -0
- sensor_modeling/evaluation/provenance.py +309 -0
- sensor_modeling/examples/__init__.py +1 -0
- sensor_modeling/examples/demos/__init__.py +1 -0
- sensor_modeling/examples/demos/ambient_pipeline_demo.py +418 -0
- sensor_modeling/examples/demos/bernoulli_ar_demo.py +356 -0
- sensor_modeling/examples/demos/cpd_ar_demo.py +25 -0
- sensor_modeling/examples/demos/cpd_benchmark.py +42 -0
- sensor_modeling/examples/demos/hmm_granger_demo.py +30 -0
- sensor_modeling/examples/demos/nhpp_pelt_demo.py +80 -0
- sensor_modeling/examples/tutorials/__init__.py +1 -0
- sensor_modeling/fusion/__init__.py +46 -0
- sensor_modeling/fusion/defaults.py +296 -0
- sensor_modeling/fusion/emissions.py +339 -0
- sensor_modeling/fusion/estimate.py +375 -0
- sensor_modeling/fusion/filter.py +323 -0
- sensor_modeling/health/__init__.py +31 -0
- sensor_modeling/health/monitor.py +590 -0
- sensor_modeling/health/status.py +74 -0
- sensor_modeling/hmm/__init__.py +15 -0
- sensor_modeling/hmm/adaptive_hmm.py +22 -0
- sensor_modeling/hmm/base.py +134 -0
- sensor_modeling/hmm/circadian_hmm.py +22 -0
- sensor_modeling/hmm/heterogeneous_hmm.py +22 -0
- sensor_modeling/hmm/hierarchical_hmm.py +35 -0
- sensor_modeling/hmm/scaled_dirichlet_hmm.py +23 -0
- sensor_modeling/interop/__init__.py +57 -0
- sensor_modeling/interop/fhir.py +418 -0
- sensor_modeling/interop/privacy.py +308 -0
- sensor_modeling/models/__init__.py +12 -0
- sensor_modeling/models/bernoulli_ar/__init__.py +6 -0
- sensor_modeling/models/bernoulli_ar/base_model.py +569 -0
- sensor_modeling/models/bernoulli_ar/multivariate_model.py +411 -0
- sensor_modeling/models/change_point_detection/__init__.py +10 -0
- sensor_modeling/models/change_point_detection/deep.py +65 -0
- sensor_modeling/models/change_point_detection/pelt.py +159 -0
- sensor_modeling/models/nhpp_pelt/__init__.py +5 -0
- sensor_modeling/models/nhpp_pelt/bspline.py +96 -0
- sensor_modeling/models/nhpp_pelt/cli.py +243 -0
- sensor_modeling/models/nhpp_pelt/diagnostics.py +234 -0
- sensor_modeling/models/nhpp_pelt/io.py +58 -0
- sensor_modeling/models/nhpp_pelt/model.py +408 -0
- sensor_modeling/models/nhpp_pelt/optimizer.py +142 -0
- sensor_modeling/models/nhpp_pelt/plotting.py +218 -0
- sensor_modeling/models/nhpp_pelt/quad.py +72 -0
- sensor_modeling/models/nhpp_pelt/regularization.py +121 -0
- sensor_modeling/models/nhpp_pelt/utils.py +174 -0
- sensor_modeling/observations/__init__.py +59 -0
- sensor_modeling/observations/adapters.py +195 -0
- sensor_modeling/observations/ingest.py +269 -0
- sensor_modeling/observations/observation.py +270 -0
- sensor_modeling/observations/registry.py +262 -0
- sensor_modeling/observations/stream.py +342 -0
- sensor_modeling/observations/types.py +107 -0
- sensor_modeling/observations/units.py +117 -0
- sensor_modeling/online/__init__.py +36 -0
- sensor_modeling/online/benchmarks.py +242 -0
- sensor_modeling/online/pipeline.py +485 -0
- sensor_modeling/simulation/__init__.py +54 -0
- sensor_modeling/simulation/faults.py +191 -0
- sensor_modeling/simulation/household.py +862 -0
- sensor_modeling/states/__init__.py +23 -0
- sensor_modeling/states/markov.py +105 -0
- sensor_modeling/states/ontology.py +238 -0
- sensor_modeling/utils/__init__.py +41 -0
- sensor_modeling/utils/data_io.py +199 -0
- sensor_modeling/utils/logging_config.py +10 -0
- sensor_modeling/utils/missing.py +188 -0
- sensor_modeling/utils/plotting.py +98 -0
- sensor_modeling/utils/validation.py +117 -0
- sensor_modeling/visualization/__init__.py +3 -0
- sensor_modeling/visualization/clinical.py +67 -0
- sensor_modeling/visualization/interactive.py +208 -0
- sensor_modeling/visualization/research.py +60 -0
- sensor_modeling/visualization/web_app.py +137 -0
- sensor_modeling-0.2.0.dist-info/METADATA +683 -0
- sensor_modeling-0.2.0.dist-info/RECORD +114 -0
- sensor_modeling-0.2.0.dist-info/WHEEL +5 -0
- sensor_modeling-0.2.0.dist-info/entry_points.txt +18 -0
- sensor_modeling-0.2.0.dist-info/licenses/LICENSE +21 -0
- sensor_modeling-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
"""Measuring whether the online pipeline is genuinely edge-capable.
|
|
2
|
+
|
|
3
|
+
Claiming bounded memory is easy; demonstrating it is the point of this module.
|
|
4
|
+
The measurement that matters is not peak allocation on one run -- that varies
|
|
5
|
+
with the interpreter and tells you little -- but whether the *retained* state
|
|
6
|
+
grows with how long the pipeline has been running.
|
|
7
|
+
|
|
8
|
+
So the central measurement here compares the serialised snapshot after a short
|
|
9
|
+
run against the snapshot after a much longer one. If the pipeline is bounded,
|
|
10
|
+
those are close to the same size no matter how many observations went through.
|
|
11
|
+
If it is quietly accumulating, the second is larger, and no amount of docstring
|
|
12
|
+
prose about bounded deques will hide it.
|
|
13
|
+
|
|
14
|
+
Correctness came first. Nothing in the pipeline has been optimised, and these
|
|
15
|
+
numbers exist to establish a baseline and catch regressions, not to advertise
|
|
16
|
+
performance.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import json
|
|
22
|
+
import logging
|
|
23
|
+
import time
|
|
24
|
+
import tracemalloc
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from datetime import timedelta
|
|
27
|
+
|
|
28
|
+
import numpy as np
|
|
29
|
+
|
|
30
|
+
from ..baseline.adaptive import BaselineConfig
|
|
31
|
+
from ..simulation.household import HouseholdConfig, simulate
|
|
32
|
+
from .pipeline import BehaviouralSensingPipeline, PipelineConfig
|
|
33
|
+
|
|
34
|
+
logger = logging.getLogger(__name__)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass(frozen=True)
|
|
38
|
+
class PipelineBenchmark:
|
|
39
|
+
"""Throughput, latency and retained-state measurements for one run.
|
|
40
|
+
|
|
41
|
+
Attributes
|
|
42
|
+
----------
|
|
43
|
+
days, observations, steps
|
|
44
|
+
Size of the workload.
|
|
45
|
+
wall_seconds
|
|
46
|
+
Total processing time.
|
|
47
|
+
observations_per_second
|
|
48
|
+
Ingestion throughput.
|
|
49
|
+
microseconds_per_observation
|
|
50
|
+
Mean cost of admitting one record.
|
|
51
|
+
step_latency_ms
|
|
52
|
+
Mean, median, 95th percentile and maximum cost of one pipeline step,
|
|
53
|
+
which is where health, context, fusion and baselines all run.
|
|
54
|
+
snapshot_bytes
|
|
55
|
+
Size of the serialised restartable state. This is the number that
|
|
56
|
+
decides whether the pipeline fits on a constrained device.
|
|
57
|
+
peak_memory_mb
|
|
58
|
+
Peak traced allocation during the run, including the simulated
|
|
59
|
+
record itself, so it is an upper bound rather than the pipeline's
|
|
60
|
+
own footprint.
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
days: int
|
|
64
|
+
observations: int
|
|
65
|
+
steps: int
|
|
66
|
+
wall_seconds: float
|
|
67
|
+
observations_per_second: float
|
|
68
|
+
microseconds_per_observation: float
|
|
69
|
+
step_latency_ms: dict[str, float]
|
|
70
|
+
snapshot_bytes: int
|
|
71
|
+
peak_memory_mb: float
|
|
72
|
+
|
|
73
|
+
def to_dict(self) -> dict[str, object]:
|
|
74
|
+
"""Return a serialisable form of the benchmark."""
|
|
75
|
+
return {
|
|
76
|
+
"days": self.days,
|
|
77
|
+
"observations": self.observations,
|
|
78
|
+
"steps": self.steps,
|
|
79
|
+
"wall_seconds": self.wall_seconds,
|
|
80
|
+
"observations_per_second": self.observations_per_second,
|
|
81
|
+
"microseconds_per_observation": self.microseconds_per_observation,
|
|
82
|
+
"step_latency_ms": self.step_latency_ms,
|
|
83
|
+
"snapshot_bytes": self.snapshot_bytes,
|
|
84
|
+
"peak_memory_mb": self.peak_memory_mb,
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def benchmark_pipeline(
|
|
89
|
+
days: int = 7,
|
|
90
|
+
*,
|
|
91
|
+
seed: int = 4242,
|
|
92
|
+
step: timedelta = timedelta(minutes=15),
|
|
93
|
+
trace_memory: bool = True,
|
|
94
|
+
) -> PipelineBenchmark:
|
|
95
|
+
"""Run the pipeline over a simulated household and measure it.
|
|
96
|
+
|
|
97
|
+
Latency is measured per *step* rather than per observation, because a
|
|
98
|
+
step is where the work happens: observations between steps only accumulate
|
|
99
|
+
in a buffer, while a step runs health, context, fusion and -- at a day
|
|
100
|
+
boundary -- the baselines and alerting.
|
|
101
|
+
"""
|
|
102
|
+
result = simulate(HouseholdConfig(days=days, seed=seed))
|
|
103
|
+
pipeline = BehaviouralSensingPipeline(
|
|
104
|
+
result.registry, config=PipelineConfig(tz=result.config.tz, step=step)
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
if trace_memory:
|
|
108
|
+
tracemalloc.start()
|
|
109
|
+
|
|
110
|
+
latencies: list[float] = []
|
|
111
|
+
started = time.perf_counter()
|
|
112
|
+
latest = None
|
|
113
|
+
for observation in result.observations:
|
|
114
|
+
arrival = observation.received_at or observation.timestamp
|
|
115
|
+
pipeline.push(observation)
|
|
116
|
+
if latest is None or arrival > latest:
|
|
117
|
+
latest = arrival
|
|
118
|
+
before = time.perf_counter()
|
|
119
|
+
produced = pipeline.advance(arrival)
|
|
120
|
+
elapsed = time.perf_counter() - before
|
|
121
|
+
if produced:
|
|
122
|
+
latencies.extend([elapsed / len(produced)] * len(produced))
|
|
123
|
+
pipeline.close(result.end)
|
|
124
|
+
wall = time.perf_counter() - started
|
|
125
|
+
|
|
126
|
+
peak_mb = 0.0
|
|
127
|
+
if trace_memory:
|
|
128
|
+
peak_mb = tracemalloc.get_traced_memory()[1] / 1_000_000
|
|
129
|
+
tracemalloc.stop()
|
|
130
|
+
|
|
131
|
+
snapshot = json.dumps(pipeline.snapshot(), default=str)
|
|
132
|
+
count = len(result.observations)
|
|
133
|
+
samples = np.array(latencies) * 1000.0 if latencies else np.zeros(1)
|
|
134
|
+
|
|
135
|
+
return PipelineBenchmark(
|
|
136
|
+
days=days,
|
|
137
|
+
observations=count,
|
|
138
|
+
steps=len(latencies),
|
|
139
|
+
wall_seconds=wall,
|
|
140
|
+
observations_per_second=count / wall if wall > 0 else 0.0,
|
|
141
|
+
microseconds_per_observation=(wall / count * 1e6) if count else 0.0,
|
|
142
|
+
step_latency_ms={
|
|
143
|
+
"mean": float(samples.mean()),
|
|
144
|
+
"median": float(np.median(samples)),
|
|
145
|
+
"p95": float(np.percentile(samples, 95)),
|
|
146
|
+
"max": float(samples.max()),
|
|
147
|
+
},
|
|
148
|
+
snapshot_bytes=len(snapshot.encode("utf-8")),
|
|
149
|
+
peak_memory_mb=peak_mb,
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
@dataclass(frozen=True)
|
|
154
|
+
class BoundedStateResult:
|
|
155
|
+
"""Evidence for or against the bounded-memory claim."""
|
|
156
|
+
|
|
157
|
+
short_days: int
|
|
158
|
+
long_days: int
|
|
159
|
+
short_snapshot_bytes: int
|
|
160
|
+
long_snapshot_bytes: int
|
|
161
|
+
|
|
162
|
+
@property
|
|
163
|
+
def growth_ratio(self) -> float:
|
|
164
|
+
"""How much retained state grew for a much longer run.
|
|
165
|
+
|
|
166
|
+
A bounded pipeline stays near one. A value tracking the ratio of the
|
|
167
|
+
run lengths would mean state is accumulating with the stream.
|
|
168
|
+
"""
|
|
169
|
+
if self.short_snapshot_bytes == 0:
|
|
170
|
+
return float("inf")
|
|
171
|
+
return self.long_snapshot_bytes / self.short_snapshot_bytes
|
|
172
|
+
|
|
173
|
+
@property
|
|
174
|
+
def workload_ratio(self) -> float:
|
|
175
|
+
"""How much longer the long run was."""
|
|
176
|
+
return self.long_days / self.short_days if self.short_days else float("inf")
|
|
177
|
+
|
|
178
|
+
def to_dict(self) -> dict[str, object]:
|
|
179
|
+
"""Return a serialisable form of the result."""
|
|
180
|
+
return {
|
|
181
|
+
"short_days": self.short_days,
|
|
182
|
+
"long_days": self.long_days,
|
|
183
|
+
"short_snapshot_bytes": self.short_snapshot_bytes,
|
|
184
|
+
"long_snapshot_bytes": self.long_snapshot_bytes,
|
|
185
|
+
"growth_ratio": self.growth_ratio,
|
|
186
|
+
"workload_ratio": self.workload_ratio,
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def measure_bounded_state(
|
|
191
|
+
short_days: int = 3,
|
|
192
|
+
long_days: int = 21,
|
|
193
|
+
*,
|
|
194
|
+
seed: int = 4242,
|
|
195
|
+
step: timedelta = timedelta(minutes=30),
|
|
196
|
+
baseline_config: BaselineConfig | None = None,
|
|
197
|
+
) -> BoundedStateResult:
|
|
198
|
+
"""Compare retained state after a short run against a much longer one.
|
|
199
|
+
|
|
200
|
+
This is the direct test of the bounded-memory claim. Every stage but the
|
|
201
|
+
baselines keeps a fixed-size belief, and the baselines keep a history
|
|
202
|
+
capped at ``history_days``.
|
|
203
|
+
|
|
204
|
+
That cap is why the growth ratio is not exactly one by default: over a few
|
|
205
|
+
weeks the history is still filling, so a longer run legitimately retains
|
|
206
|
+
more. Pass a *baseline_config* whose ``history_days`` both runs exceed to
|
|
207
|
+
observe the asymptotic behaviour, where retained state stops growing
|
|
208
|
+
however long the pipeline runs.
|
|
209
|
+
"""
|
|
210
|
+
if short_days >= long_days:
|
|
211
|
+
raise ValueError("long_days must exceed short_days")
|
|
212
|
+
|
|
213
|
+
sizes: dict[int, int] = {}
|
|
214
|
+
for days in (short_days, long_days):
|
|
215
|
+
result = simulate(HouseholdConfig(days=days, seed=seed))
|
|
216
|
+
pipeline = BehaviouralSensingPipeline(
|
|
217
|
+
result.registry,
|
|
218
|
+
config=PipelineConfig(tz=result.config.tz, step=step),
|
|
219
|
+
baseline_config=baseline_config,
|
|
220
|
+
)
|
|
221
|
+
pipeline.run(result.observations)
|
|
222
|
+
pipeline.close(result.end)
|
|
223
|
+
sizes[days] = len(json.dumps(pipeline.snapshot(), default=str).encode("utf-8"))
|
|
224
|
+
|
|
225
|
+
outcome = BoundedStateResult(
|
|
226
|
+
short_days=short_days,
|
|
227
|
+
long_days=long_days,
|
|
228
|
+
short_snapshot_bytes=sizes[short_days],
|
|
229
|
+
long_snapshot_bytes=sizes[long_days],
|
|
230
|
+
)
|
|
231
|
+
logger.info(
|
|
232
|
+
"State grew %.2fx for a %.1fx longer run",
|
|
233
|
+
outcome.growth_ratio,
|
|
234
|
+
outcome.workload_ratio,
|
|
235
|
+
)
|
|
236
|
+
return outcome
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
if __name__ == "__main__": # pragma: no cover - manual invocation
|
|
240
|
+
logging.basicConfig(level=logging.INFO)
|
|
241
|
+
print(json.dumps(benchmark_pipeline().to_dict(), indent=2))
|
|
242
|
+
print(json.dumps(measure_bounded_state().to_dict(), indent=2))
|