sensor-modeling 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. sensor_modeling/__init__.py +45 -0
  2. sensor_modeling/alerts/__init__.py +26 -0
  3. sensor_modeling/alerts/alert.py +532 -0
  4. sensor_modeling/analysis/__init__.py +43 -0
  5. sensor_modeling/analysis/_frame.py +19 -0
  6. sensor_modeling/analysis/behavioral_analysis.py +57 -0
  7. sensor_modeling/analysis/behavioral_metrics.py +66 -0
  8. sensor_modeling/analysis/comparison.py +164 -0
  9. sensor_modeling/analysis/dependency_network.py +408 -0
  10. sensor_modeling/analysis/granger_causality.py +314 -0
  11. sensor_modeling/analysis/pipeline.py +168 -0
  12. sensor_modeling/analysis/reporting.py +109 -0
  13. sensor_modeling/baseline/__init__.py +30 -0
  14. sensor_modeling/baseline/adaptive.py +520 -0
  15. sensor_modeling/baseline/features.py +224 -0
  16. sensor_modeling/change_point/__init__.py +13 -0
  17. sensor_modeling/change_point/_validation.py +31 -0
  18. sensor_modeling/change_point/adaptive_normalization.py +55 -0
  19. sensor_modeling/change_point/embedding_cpd.py +60 -0
  20. sensor_modeling/change_point/energy_efficient.py +57 -0
  21. sensor_modeling/change_point/genetic_optimization.py +65 -0
  22. sensor_modeling/cli.py +416 -0
  23. sensor_modeling/context/__init__.py +33 -0
  24. sensor_modeling/context/occupancy.py +529 -0
  25. sensor_modeling/data/__init__.py +5 -0
  26. sensor_modeling/data/loaders.py +146 -0
  27. sensor_modeling/data/preprocessing.py +83 -0
  28. sensor_modeling/data/synthetic.py +121 -0
  29. sensor_modeling/data/validation.py +81 -0
  30. sensor_modeling/evaluation/__init__.py +92 -0
  31. sensor_modeling/evaluation/ablation.py +303 -0
  32. sensor_modeling/evaluation/attribution.py +474 -0
  33. sensor_modeling/evaluation/detection.py +297 -0
  34. sensor_modeling/evaluation/metrics.py +541 -0
  35. sensor_modeling/evaluation/provenance.py +309 -0
  36. sensor_modeling/examples/__init__.py +1 -0
  37. sensor_modeling/examples/demos/__init__.py +1 -0
  38. sensor_modeling/examples/demos/ambient_pipeline_demo.py +418 -0
  39. sensor_modeling/examples/demos/bernoulli_ar_demo.py +356 -0
  40. sensor_modeling/examples/demos/cpd_ar_demo.py +25 -0
  41. sensor_modeling/examples/demos/cpd_benchmark.py +42 -0
  42. sensor_modeling/examples/demos/hmm_granger_demo.py +30 -0
  43. sensor_modeling/examples/demos/nhpp_pelt_demo.py +80 -0
  44. sensor_modeling/examples/tutorials/__init__.py +1 -0
  45. sensor_modeling/fusion/__init__.py +46 -0
  46. sensor_modeling/fusion/defaults.py +296 -0
  47. sensor_modeling/fusion/emissions.py +339 -0
  48. sensor_modeling/fusion/estimate.py +375 -0
  49. sensor_modeling/fusion/filter.py +323 -0
  50. sensor_modeling/health/__init__.py +31 -0
  51. sensor_modeling/health/monitor.py +590 -0
  52. sensor_modeling/health/status.py +74 -0
  53. sensor_modeling/hmm/__init__.py +15 -0
  54. sensor_modeling/hmm/adaptive_hmm.py +22 -0
  55. sensor_modeling/hmm/base.py +134 -0
  56. sensor_modeling/hmm/circadian_hmm.py +22 -0
  57. sensor_modeling/hmm/heterogeneous_hmm.py +22 -0
  58. sensor_modeling/hmm/hierarchical_hmm.py +35 -0
  59. sensor_modeling/hmm/scaled_dirichlet_hmm.py +23 -0
  60. sensor_modeling/interop/__init__.py +57 -0
  61. sensor_modeling/interop/fhir.py +418 -0
  62. sensor_modeling/interop/privacy.py +308 -0
  63. sensor_modeling/models/__init__.py +12 -0
  64. sensor_modeling/models/bernoulli_ar/__init__.py +6 -0
  65. sensor_modeling/models/bernoulli_ar/base_model.py +569 -0
  66. sensor_modeling/models/bernoulli_ar/multivariate_model.py +411 -0
  67. sensor_modeling/models/change_point_detection/__init__.py +10 -0
  68. sensor_modeling/models/change_point_detection/deep.py +65 -0
  69. sensor_modeling/models/change_point_detection/pelt.py +159 -0
  70. sensor_modeling/models/nhpp_pelt/__init__.py +5 -0
  71. sensor_modeling/models/nhpp_pelt/bspline.py +96 -0
  72. sensor_modeling/models/nhpp_pelt/cli.py +243 -0
  73. sensor_modeling/models/nhpp_pelt/diagnostics.py +234 -0
  74. sensor_modeling/models/nhpp_pelt/io.py +58 -0
  75. sensor_modeling/models/nhpp_pelt/model.py +408 -0
  76. sensor_modeling/models/nhpp_pelt/optimizer.py +142 -0
  77. sensor_modeling/models/nhpp_pelt/plotting.py +218 -0
  78. sensor_modeling/models/nhpp_pelt/quad.py +72 -0
  79. sensor_modeling/models/nhpp_pelt/regularization.py +121 -0
  80. sensor_modeling/models/nhpp_pelt/utils.py +174 -0
  81. sensor_modeling/observations/__init__.py +59 -0
  82. sensor_modeling/observations/adapters.py +195 -0
  83. sensor_modeling/observations/ingest.py +269 -0
  84. sensor_modeling/observations/observation.py +270 -0
  85. sensor_modeling/observations/registry.py +262 -0
  86. sensor_modeling/observations/stream.py +342 -0
  87. sensor_modeling/observations/types.py +107 -0
  88. sensor_modeling/observations/units.py +117 -0
  89. sensor_modeling/online/__init__.py +36 -0
  90. sensor_modeling/online/benchmarks.py +242 -0
  91. sensor_modeling/online/pipeline.py +485 -0
  92. sensor_modeling/simulation/__init__.py +54 -0
  93. sensor_modeling/simulation/faults.py +191 -0
  94. sensor_modeling/simulation/household.py +862 -0
  95. sensor_modeling/states/__init__.py +23 -0
  96. sensor_modeling/states/markov.py +105 -0
  97. sensor_modeling/states/ontology.py +238 -0
  98. sensor_modeling/utils/__init__.py +41 -0
  99. sensor_modeling/utils/data_io.py +199 -0
  100. sensor_modeling/utils/logging_config.py +10 -0
  101. sensor_modeling/utils/missing.py +188 -0
  102. sensor_modeling/utils/plotting.py +98 -0
  103. sensor_modeling/utils/validation.py +117 -0
  104. sensor_modeling/visualization/__init__.py +3 -0
  105. sensor_modeling/visualization/clinical.py +67 -0
  106. sensor_modeling/visualization/interactive.py +208 -0
  107. sensor_modeling/visualization/research.py +60 -0
  108. sensor_modeling/visualization/web_app.py +137 -0
  109. sensor_modeling-0.2.0.dist-info/METADATA +683 -0
  110. sensor_modeling-0.2.0.dist-info/RECORD +114 -0
  111. sensor_modeling-0.2.0.dist-info/WHEEL +5 -0
  112. sensor_modeling-0.2.0.dist-info/entry_points.txt +18 -0
  113. sensor_modeling-0.2.0.dist-info/licenses/LICENSE +21 -0
  114. sensor_modeling-0.2.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,242 @@
1
+ """Measuring whether the online pipeline is genuinely edge-capable.
2
+
3
+ Claiming bounded memory is easy; demonstrating it is the point of this module.
4
+ The measurement that matters is not peak allocation on one run -- that varies
5
+ with the interpreter and tells you little -- but whether the *retained* state
6
+ grows with how long the pipeline has been running.
7
+
8
+ So the central measurement here compares the serialised snapshot after a short
9
+ run against the snapshot after a much longer one. If the pipeline is bounded,
10
+ those are close to the same size no matter how many observations went through.
11
+ If it is quietly accumulating, the second is larger, and no amount of docstring
12
+ prose about bounded deques will hide it.
13
+
14
+ Correctness came first. Nothing in the pipeline has been optimised, and these
15
+ numbers exist to establish a baseline and catch regressions, not to advertise
16
+ performance.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import json
22
+ import logging
23
+ import time
24
+ import tracemalloc
25
+ from dataclasses import dataclass
26
+ from datetime import timedelta
27
+
28
+ import numpy as np
29
+
30
+ from ..baseline.adaptive import BaselineConfig
31
+ from ..simulation.household import HouseholdConfig, simulate
32
+ from .pipeline import BehaviouralSensingPipeline, PipelineConfig
33
+
34
+ logger = logging.getLogger(__name__)
35
+
36
+
37
+ @dataclass(frozen=True)
38
+ class PipelineBenchmark:
39
+ """Throughput, latency and retained-state measurements for one run.
40
+
41
+ Attributes
42
+ ----------
43
+ days, observations, steps
44
+ Size of the workload.
45
+ wall_seconds
46
+ Total processing time.
47
+ observations_per_second
48
+ Ingestion throughput.
49
+ microseconds_per_observation
50
+ Mean cost of admitting one record.
51
+ step_latency_ms
52
+ Mean, median, 95th percentile and maximum cost of one pipeline step,
53
+ which is where health, context, fusion and baselines all run.
54
+ snapshot_bytes
55
+ Size of the serialised restartable state. This is the number that
56
+ decides whether the pipeline fits on a constrained device.
57
+ peak_memory_mb
58
+ Peak traced allocation during the run, including the simulated
59
+ record itself, so it is an upper bound rather than the pipeline's
60
+ own footprint.
61
+ """
62
+
63
+ days: int
64
+ observations: int
65
+ steps: int
66
+ wall_seconds: float
67
+ observations_per_second: float
68
+ microseconds_per_observation: float
69
+ step_latency_ms: dict[str, float]
70
+ snapshot_bytes: int
71
+ peak_memory_mb: float
72
+
73
+ def to_dict(self) -> dict[str, object]:
74
+ """Return a serialisable form of the benchmark."""
75
+ return {
76
+ "days": self.days,
77
+ "observations": self.observations,
78
+ "steps": self.steps,
79
+ "wall_seconds": self.wall_seconds,
80
+ "observations_per_second": self.observations_per_second,
81
+ "microseconds_per_observation": self.microseconds_per_observation,
82
+ "step_latency_ms": self.step_latency_ms,
83
+ "snapshot_bytes": self.snapshot_bytes,
84
+ "peak_memory_mb": self.peak_memory_mb,
85
+ }
86
+
87
+
88
+ def benchmark_pipeline(
89
+ days: int = 7,
90
+ *,
91
+ seed: int = 4242,
92
+ step: timedelta = timedelta(minutes=15),
93
+ trace_memory: bool = True,
94
+ ) -> PipelineBenchmark:
95
+ """Run the pipeline over a simulated household and measure it.
96
+
97
+ Latency is measured per *step* rather than per observation, because a
98
+ step is where the work happens: observations between steps only accumulate
99
+ in a buffer, while a step runs health, context, fusion and -- at a day
100
+ boundary -- the baselines and alerting.
101
+ """
102
+ result = simulate(HouseholdConfig(days=days, seed=seed))
103
+ pipeline = BehaviouralSensingPipeline(
104
+ result.registry, config=PipelineConfig(tz=result.config.tz, step=step)
105
+ )
106
+
107
+ if trace_memory:
108
+ tracemalloc.start()
109
+
110
+ latencies: list[float] = []
111
+ started = time.perf_counter()
112
+ latest = None
113
+ for observation in result.observations:
114
+ arrival = observation.received_at or observation.timestamp
115
+ pipeline.push(observation)
116
+ if latest is None or arrival > latest:
117
+ latest = arrival
118
+ before = time.perf_counter()
119
+ produced = pipeline.advance(arrival)
120
+ elapsed = time.perf_counter() - before
121
+ if produced:
122
+ latencies.extend([elapsed / len(produced)] * len(produced))
123
+ pipeline.close(result.end)
124
+ wall = time.perf_counter() - started
125
+
126
+ peak_mb = 0.0
127
+ if trace_memory:
128
+ peak_mb = tracemalloc.get_traced_memory()[1] / 1_000_000
129
+ tracemalloc.stop()
130
+
131
+ snapshot = json.dumps(pipeline.snapshot(), default=str)
132
+ count = len(result.observations)
133
+ samples = np.array(latencies) * 1000.0 if latencies else np.zeros(1)
134
+
135
+ return PipelineBenchmark(
136
+ days=days,
137
+ observations=count,
138
+ steps=len(latencies),
139
+ wall_seconds=wall,
140
+ observations_per_second=count / wall if wall > 0 else 0.0,
141
+ microseconds_per_observation=(wall / count * 1e6) if count else 0.0,
142
+ step_latency_ms={
143
+ "mean": float(samples.mean()),
144
+ "median": float(np.median(samples)),
145
+ "p95": float(np.percentile(samples, 95)),
146
+ "max": float(samples.max()),
147
+ },
148
+ snapshot_bytes=len(snapshot.encode("utf-8")),
149
+ peak_memory_mb=peak_mb,
150
+ )
151
+
152
+
153
+ @dataclass(frozen=True)
154
+ class BoundedStateResult:
155
+ """Evidence for or against the bounded-memory claim."""
156
+
157
+ short_days: int
158
+ long_days: int
159
+ short_snapshot_bytes: int
160
+ long_snapshot_bytes: int
161
+
162
+ @property
163
+ def growth_ratio(self) -> float:
164
+ """How much retained state grew for a much longer run.
165
+
166
+ A bounded pipeline stays near one. A value tracking the ratio of the
167
+ run lengths would mean state is accumulating with the stream.
168
+ """
169
+ if self.short_snapshot_bytes == 0:
170
+ return float("inf")
171
+ return self.long_snapshot_bytes / self.short_snapshot_bytes
172
+
173
+ @property
174
+ def workload_ratio(self) -> float:
175
+ """How much longer the long run was."""
176
+ return self.long_days / self.short_days if self.short_days else float("inf")
177
+
178
+ def to_dict(self) -> dict[str, object]:
179
+ """Return a serialisable form of the result."""
180
+ return {
181
+ "short_days": self.short_days,
182
+ "long_days": self.long_days,
183
+ "short_snapshot_bytes": self.short_snapshot_bytes,
184
+ "long_snapshot_bytes": self.long_snapshot_bytes,
185
+ "growth_ratio": self.growth_ratio,
186
+ "workload_ratio": self.workload_ratio,
187
+ }
188
+
189
+
190
+ def measure_bounded_state(
191
+ short_days: int = 3,
192
+ long_days: int = 21,
193
+ *,
194
+ seed: int = 4242,
195
+ step: timedelta = timedelta(minutes=30),
196
+ baseline_config: BaselineConfig | None = None,
197
+ ) -> BoundedStateResult:
198
+ """Compare retained state after a short run against a much longer one.
199
+
200
+ This is the direct test of the bounded-memory claim. Every stage but the
201
+ baselines keeps a fixed-size belief, and the baselines keep a history
202
+ capped at ``history_days``.
203
+
204
+ That cap is why the growth ratio is not exactly one by default: over a few
205
+ weeks the history is still filling, so a longer run legitimately retains
206
+ more. Pass a *baseline_config* whose ``history_days`` both runs exceed to
207
+ observe the asymptotic behaviour, where retained state stops growing
208
+ however long the pipeline runs.
209
+ """
210
+ if short_days >= long_days:
211
+ raise ValueError("long_days must exceed short_days")
212
+
213
+ sizes: dict[int, int] = {}
214
+ for days in (short_days, long_days):
215
+ result = simulate(HouseholdConfig(days=days, seed=seed))
216
+ pipeline = BehaviouralSensingPipeline(
217
+ result.registry,
218
+ config=PipelineConfig(tz=result.config.tz, step=step),
219
+ baseline_config=baseline_config,
220
+ )
221
+ pipeline.run(result.observations)
222
+ pipeline.close(result.end)
223
+ sizes[days] = len(json.dumps(pipeline.snapshot(), default=str).encode("utf-8"))
224
+
225
+ outcome = BoundedStateResult(
226
+ short_days=short_days,
227
+ long_days=long_days,
228
+ short_snapshot_bytes=sizes[short_days],
229
+ long_snapshot_bytes=sizes[long_days],
230
+ )
231
+ logger.info(
232
+ "State grew %.2fx for a %.1fx longer run",
233
+ outcome.growth_ratio,
234
+ outcome.workload_ratio,
235
+ )
236
+ return outcome
237
+
238
+
239
+ if __name__ == "__main__": # pragma: no cover - manual invocation
240
+ logging.basicConfig(level=logging.INFO)
241
+ print(json.dumps(benchmark_pipeline().to_dict(), indent=2))
242
+ print(json.dumps(measure_bounded_state().to_dict(), indent=2))