sensor-modeling 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. sensor_modeling/__init__.py +45 -0
  2. sensor_modeling/alerts/__init__.py +26 -0
  3. sensor_modeling/alerts/alert.py +532 -0
  4. sensor_modeling/analysis/__init__.py +43 -0
  5. sensor_modeling/analysis/_frame.py +19 -0
  6. sensor_modeling/analysis/behavioral_analysis.py +57 -0
  7. sensor_modeling/analysis/behavioral_metrics.py +66 -0
  8. sensor_modeling/analysis/comparison.py +164 -0
  9. sensor_modeling/analysis/dependency_network.py +408 -0
  10. sensor_modeling/analysis/granger_causality.py +314 -0
  11. sensor_modeling/analysis/pipeline.py +168 -0
  12. sensor_modeling/analysis/reporting.py +109 -0
  13. sensor_modeling/baseline/__init__.py +30 -0
  14. sensor_modeling/baseline/adaptive.py +520 -0
  15. sensor_modeling/baseline/features.py +224 -0
  16. sensor_modeling/change_point/__init__.py +13 -0
  17. sensor_modeling/change_point/_validation.py +31 -0
  18. sensor_modeling/change_point/adaptive_normalization.py +55 -0
  19. sensor_modeling/change_point/embedding_cpd.py +60 -0
  20. sensor_modeling/change_point/energy_efficient.py +57 -0
  21. sensor_modeling/change_point/genetic_optimization.py +65 -0
  22. sensor_modeling/cli.py +416 -0
  23. sensor_modeling/context/__init__.py +33 -0
  24. sensor_modeling/context/occupancy.py +529 -0
  25. sensor_modeling/data/__init__.py +5 -0
  26. sensor_modeling/data/loaders.py +146 -0
  27. sensor_modeling/data/preprocessing.py +83 -0
  28. sensor_modeling/data/synthetic.py +121 -0
  29. sensor_modeling/data/validation.py +81 -0
  30. sensor_modeling/evaluation/__init__.py +92 -0
  31. sensor_modeling/evaluation/ablation.py +303 -0
  32. sensor_modeling/evaluation/attribution.py +474 -0
  33. sensor_modeling/evaluation/detection.py +297 -0
  34. sensor_modeling/evaluation/metrics.py +541 -0
  35. sensor_modeling/evaluation/provenance.py +309 -0
  36. sensor_modeling/examples/__init__.py +1 -0
  37. sensor_modeling/examples/demos/__init__.py +1 -0
  38. sensor_modeling/examples/demos/ambient_pipeline_demo.py +418 -0
  39. sensor_modeling/examples/demos/bernoulli_ar_demo.py +356 -0
  40. sensor_modeling/examples/demos/cpd_ar_demo.py +25 -0
  41. sensor_modeling/examples/demos/cpd_benchmark.py +42 -0
  42. sensor_modeling/examples/demos/hmm_granger_demo.py +30 -0
  43. sensor_modeling/examples/demos/nhpp_pelt_demo.py +80 -0
  44. sensor_modeling/examples/tutorials/__init__.py +1 -0
  45. sensor_modeling/fusion/__init__.py +46 -0
  46. sensor_modeling/fusion/defaults.py +296 -0
  47. sensor_modeling/fusion/emissions.py +339 -0
  48. sensor_modeling/fusion/estimate.py +375 -0
  49. sensor_modeling/fusion/filter.py +323 -0
  50. sensor_modeling/health/__init__.py +31 -0
  51. sensor_modeling/health/monitor.py +590 -0
  52. sensor_modeling/health/status.py +74 -0
  53. sensor_modeling/hmm/__init__.py +15 -0
  54. sensor_modeling/hmm/adaptive_hmm.py +22 -0
  55. sensor_modeling/hmm/base.py +134 -0
  56. sensor_modeling/hmm/circadian_hmm.py +22 -0
  57. sensor_modeling/hmm/heterogeneous_hmm.py +22 -0
  58. sensor_modeling/hmm/hierarchical_hmm.py +35 -0
  59. sensor_modeling/hmm/scaled_dirichlet_hmm.py +23 -0
  60. sensor_modeling/interop/__init__.py +57 -0
  61. sensor_modeling/interop/fhir.py +418 -0
  62. sensor_modeling/interop/privacy.py +308 -0
  63. sensor_modeling/models/__init__.py +12 -0
  64. sensor_modeling/models/bernoulli_ar/__init__.py +6 -0
  65. sensor_modeling/models/bernoulli_ar/base_model.py +569 -0
  66. sensor_modeling/models/bernoulli_ar/multivariate_model.py +411 -0
  67. sensor_modeling/models/change_point_detection/__init__.py +10 -0
  68. sensor_modeling/models/change_point_detection/deep.py +65 -0
  69. sensor_modeling/models/change_point_detection/pelt.py +159 -0
  70. sensor_modeling/models/nhpp_pelt/__init__.py +5 -0
  71. sensor_modeling/models/nhpp_pelt/bspline.py +96 -0
  72. sensor_modeling/models/nhpp_pelt/cli.py +243 -0
  73. sensor_modeling/models/nhpp_pelt/diagnostics.py +234 -0
  74. sensor_modeling/models/nhpp_pelt/io.py +58 -0
  75. sensor_modeling/models/nhpp_pelt/model.py +408 -0
  76. sensor_modeling/models/nhpp_pelt/optimizer.py +142 -0
  77. sensor_modeling/models/nhpp_pelt/plotting.py +218 -0
  78. sensor_modeling/models/nhpp_pelt/quad.py +72 -0
  79. sensor_modeling/models/nhpp_pelt/regularization.py +121 -0
  80. sensor_modeling/models/nhpp_pelt/utils.py +174 -0
  81. sensor_modeling/observations/__init__.py +59 -0
  82. sensor_modeling/observations/adapters.py +195 -0
  83. sensor_modeling/observations/ingest.py +269 -0
  84. sensor_modeling/observations/observation.py +270 -0
  85. sensor_modeling/observations/registry.py +262 -0
  86. sensor_modeling/observations/stream.py +342 -0
  87. sensor_modeling/observations/types.py +107 -0
  88. sensor_modeling/observations/units.py +117 -0
  89. sensor_modeling/online/__init__.py +36 -0
  90. sensor_modeling/online/benchmarks.py +242 -0
  91. sensor_modeling/online/pipeline.py +485 -0
  92. sensor_modeling/simulation/__init__.py +54 -0
  93. sensor_modeling/simulation/faults.py +191 -0
  94. sensor_modeling/simulation/household.py +862 -0
  95. sensor_modeling/states/__init__.py +23 -0
  96. sensor_modeling/states/markov.py +105 -0
  97. sensor_modeling/states/ontology.py +238 -0
  98. sensor_modeling/utils/__init__.py +41 -0
  99. sensor_modeling/utils/data_io.py +199 -0
  100. sensor_modeling/utils/logging_config.py +10 -0
  101. sensor_modeling/utils/missing.py +188 -0
  102. sensor_modeling/utils/plotting.py +98 -0
  103. sensor_modeling/utils/validation.py +117 -0
  104. sensor_modeling/visualization/__init__.py +3 -0
  105. sensor_modeling/visualization/clinical.py +67 -0
  106. sensor_modeling/visualization/interactive.py +208 -0
  107. sensor_modeling/visualization/research.py +60 -0
  108. sensor_modeling/visualization/web_app.py +137 -0
  109. sensor_modeling-0.2.0.dist-info/METADATA +683 -0
  110. sensor_modeling-0.2.0.dist-info/RECORD +114 -0
  111. sensor_modeling-0.2.0.dist-info/WHEEL +5 -0
  112. sensor_modeling-0.2.0.dist-info/entry_points.txt +18 -0
  113. sensor_modeling-0.2.0.dist-info/licenses/LICENSE +21 -0
  114. sensor_modeling-0.2.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,529 @@
1
+ """Who is in the home, and who probably caused what the sensors saw.
2
+
3
+ Ambient sensing has no notion of identity. A door opening, a hallway motion
4
+ trip, a kettle switching on -- none of these say *whose* activity they are. A
5
+ monitoring system that assumes every event belongs to the monitored resident
6
+ will read a daughter's Sunday visit as a sudden improvement in mobility, and a
7
+ carer's morning round as the resident getting up early.
8
+
9
+ This module estimates the household occupancy context probabilistically and
10
+ converts it into an attribution weight per sensor:
11
+
12
+ .. code-block:: text
13
+
14
+ P(resident_home | O)
15
+ P(visitor_present | O)
16
+ P(multiple_people_present | O)
17
+ P(event from sensor s was generated by the resident | O)
18
+
19
+ The last of these is what the fusion layer consumes. The goal is explicitly
20
+ *not* biometric identification: the platform stays compatible with
21
+ privacy-preserving sensing, so there are no cameras, no microphones, and no
22
+ face or voice recognition anywhere in this design. The goal is honest
23
+ uncertainty about attribution, which is achievable from anonymous evidence:
24
+ whether a personal device is in range, how many tracks a radar reports, and
25
+ whether activity is happening in two places at once.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import logging
31
+ from collections.abc import Iterable, Mapping, Sequence
32
+ from dataclasses import dataclass, field
33
+ from datetime import datetime, timedelta
34
+ from enum import Enum
35
+
36
+ import numpy as np
37
+ from scipy.special import logsumexp
38
+
39
+ from ..observations.observation import Observation, require_aware
40
+ from ..observations.registry import SensorRegistry
41
+ from ..observations.types import Modality
42
+ from ..states.markov import build_generator, stationary_distribution, transition_matrix
43
+
44
+ logger = logging.getLogger(__name__)
45
+
46
+ EPSILON = 1e-9
47
+
48
+
49
+ class OccupancyContext(str, Enum):
50
+ """Who is present in the home."""
51
+
52
+ EMPTY = "empty"
53
+ """Nobody present. Ambient activity here indicates a sensor problem."""
54
+
55
+ RESIDENT_ALONE = "resident_alone"
56
+ """Only the monitored resident. Ambient events are theirs."""
57
+
58
+ RESIDENT_WITH_VISITOR = "resident_with_visitor"
59
+ """Resident plus at least one other person; ambient events are shared."""
60
+
61
+ VISITOR_ONLY = "visitor_only"
62
+ """Someone else present while the resident is out, such as a carer."""
63
+
64
+
65
+ #: Canonical vector order for occupancy contexts.
66
+ CONTEXTS: tuple[OccupancyContext, ...] = (
67
+ OccupancyContext.EMPTY,
68
+ OccupancyContext.RESIDENT_ALONE,
69
+ OccupancyContext.RESIDENT_WITH_VISITOR,
70
+ OccupancyContext.VISITOR_ONLY,
71
+ )
72
+
73
+ #: Typical persistence of each context.
74
+ DEFAULT_CONTEXT_DWELL: dict[OccupancyContext, timedelta] = {
75
+ OccupancyContext.EMPTY: timedelta(hours=3),
76
+ OccupancyContext.RESIDENT_ALONE: timedelta(hours=8),
77
+ OccupancyContext.RESIDENT_WITH_VISITOR: timedelta(hours=1),
78
+ OccupancyContext.VISITOR_ONLY: timedelta(minutes=45),
79
+ }
80
+
81
+ #: Expected share of ambient activity generated by the resident in each
82
+ #: context. In a shared household the split is uncertain rather than known,
83
+ #: which is precisely why attribution is a probability and not a flag.
84
+ DEFAULT_RESIDENT_SHARE: dict[OccupancyContext, float] = {
85
+ OccupancyContext.EMPTY: 0.0,
86
+ OccupancyContext.RESIDENT_ALONE: 1.0,
87
+ OccupancyContext.RESIDENT_WITH_VISITOR: 0.5,
88
+ OccupancyContext.VISITOR_ONLY: 0.0,
89
+ }
90
+
91
+ #: Expected number of simultaneously tracked people in each context, for
92
+ #: radar or room-occupancy devices that report a count.
93
+ DEFAULT_TRACK_COUNTS: dict[OccupancyContext, float] = {
94
+ OccupancyContext.EMPTY: 0.02,
95
+ OccupancyContext.RESIDENT_ALONE: 1.0,
96
+ OccupancyContext.RESIDENT_WITH_VISITOR: 2.0,
97
+ OccupancyContext.VISITOR_ONLY: 1.0,
98
+ }
99
+
100
+ #: Probability a personal presence beacon reports the resident in range.
101
+ #: Never one, because adherence is never perfect: a wearable left on the
102
+ #: dresser is not a resident who left the house.
103
+ DEFAULT_BEACON_PRESENCE: dict[OccupancyContext, float] = {
104
+ OccupancyContext.EMPTY: 0.02,
105
+ OccupancyContext.RESIDENT_ALONE: 0.9,
106
+ OccupancyContext.RESIDENT_WITH_VISITOR: 0.9,
107
+ OccupancyContext.VISITOR_ONLY: 0.05,
108
+ }
109
+
110
+
111
+ @dataclass(frozen=True)
112
+ class ContextEstimate:
113
+ """The household occupancy posterior and what it implies for attribution."""
114
+
115
+ at: datetime
116
+ belief: np.ndarray
117
+ concurrency: int
118
+ door_events: int
119
+
120
+ def __post_init__(self) -> None:
121
+ """Validate and normalise the context belief."""
122
+ belief = np.asarray(self.belief, dtype=float)
123
+ if belief.shape != (len(CONTEXTS),):
124
+ raise ValueError("belief must have one entry per occupancy context")
125
+ total = belief.sum()
126
+ if not np.all(np.isfinite(belief)) or belief.min() < 0.0 or total <= 0.0:
127
+ raise ValueError("belief must be finite, non-negative, and non-zero")
128
+ object.__setattr__(self, "belief", belief / total)
129
+
130
+ @property
131
+ def probabilities(self) -> dict[OccupancyContext, float]:
132
+ """Posterior probability of each occupancy context."""
133
+ return dict(zip(CONTEXTS, (float(p) for p in self.belief)))
134
+
135
+ @property
136
+ def resident_home(self) -> float:
137
+ """``P(resident is in the home)``."""
138
+ return float(
139
+ self.belief[CONTEXTS.index(OccupancyContext.RESIDENT_ALONE)]
140
+ + self.belief[CONTEXTS.index(OccupancyContext.RESIDENT_WITH_VISITOR)]
141
+ )
142
+
143
+ @property
144
+ def visitor_present(self) -> float:
145
+ """``P(at least one other person is in the home)``."""
146
+ return float(
147
+ self.belief[CONTEXTS.index(OccupancyContext.RESIDENT_WITH_VISITOR)]
148
+ + self.belief[CONTEXTS.index(OccupancyContext.VISITOR_ONLY)]
149
+ )
150
+
151
+ @property
152
+ def multiple_people(self) -> float:
153
+ """``P(more than one person is in the home)``."""
154
+ return float(
155
+ self.belief[CONTEXTS.index(OccupancyContext.RESIDENT_WITH_VISITOR)]
156
+ )
157
+
158
+ @property
159
+ def most_likely(self) -> OccupancyContext:
160
+ """The highest-posterior occupancy context."""
161
+ return CONTEXTS[int(np.argmax(self.belief))]
162
+
163
+ def ambient_attribution(
164
+ self, shares: Mapping[OccupancyContext, float] | None = None
165
+ ) -> float:
166
+ """``P(an ambient event was generated by the resident)``.
167
+
168
+ Marginalises the per-context resident share over the occupancy
169
+ posterior. With the resident certainly alone this is one; with a
170
+ visitor certainly present it falls toward the shared-household split;
171
+ with the home empty it is zero.
172
+ """
173
+ weights = shares if shares is not None else DEFAULT_RESIDENT_SHARE
174
+ return float(
175
+ sum(
176
+ self.belief[index] * float(weights.get(context, 0.5))
177
+ for index, context in enumerate(CONTEXTS)
178
+ )
179
+ )
180
+
181
+ def to_dict(self) -> dict[str, object]:
182
+ """Return a serialisable form of the estimate."""
183
+ return {
184
+ "at": self.at.isoformat(),
185
+ "most_likely": self.most_likely.value,
186
+ "resident_home": self.resident_home,
187
+ "visitor_present": self.visitor_present,
188
+ "multiple_people": self.multiple_people,
189
+ "ambient_attribution": self.ambient_attribution(),
190
+ "concurrency": self.concurrency,
191
+ "door_events": self.door_events,
192
+ "probabilities": {
193
+ context.value: probability
194
+ for context, probability in self.probabilities.items()
195
+ },
196
+ }
197
+
198
+
199
+ @dataclass
200
+ class ContextConfig:
201
+ """Configuration for occupancy context estimation.
202
+
203
+ Parameters
204
+ ----------
205
+ dwell
206
+ Mean persistence of each occupancy context.
207
+ resident_share
208
+ Expected share of ambient activity generated by the resident.
209
+ track_counts
210
+ Expected simultaneous track count reported by radar-style devices.
211
+ beacon_presence
212
+ Probability a personal presence beacon reports in-range.
213
+ door_mixing
214
+ How much a door event relaxes the context belief toward a transition.
215
+ A door crossing is the moment occupancy is most likely to change, so
216
+ it loosens the belief rather than supplying evidence for one context.
217
+ concurrency_window
218
+ Window used to decide that activity in two rooms was simultaneous.
219
+ concurrency_odds
220
+ Likelihood ratio favouring multi-person contexts for each additional
221
+ room active at once. Concurrent activity in separate rooms is the
222
+ strongest anonymous evidence of more than one person.
223
+ sample_weight
224
+ Discount applied to each individual presence sample. Successive
225
+ readings from the same radar or beacon are strongly correlated --
226
+ they mostly re-observe the same unchanged situation -- so counting
227
+ them as independent measurements would drive the posterior to
228
+ certainty within minutes. This is a deliberate, inspectable
229
+ correction rather than a claim that the samples are independent.
230
+ """
231
+
232
+ dwell: Mapping[OccupancyContext, timedelta] = field(
233
+ default_factory=lambda: dict(DEFAULT_CONTEXT_DWELL)
234
+ )
235
+ resident_share: Mapping[OccupancyContext, float] = field(
236
+ default_factory=lambda: dict(DEFAULT_RESIDENT_SHARE)
237
+ )
238
+ track_counts: Mapping[OccupancyContext, float] = field(
239
+ default_factory=lambda: dict(DEFAULT_TRACK_COUNTS)
240
+ )
241
+ beacon_presence: Mapping[OccupancyContext, float] = field(
242
+ default_factory=lambda: dict(DEFAULT_BEACON_PRESENCE)
243
+ )
244
+ door_mixing: float = 0.35
245
+ concurrency_window: timedelta = timedelta(seconds=60)
246
+ concurrency_odds: float = 4.0
247
+ sample_weight: float = 0.2
248
+
249
+ def __post_init__(self) -> None:
250
+ """Validate the configuration."""
251
+ for context in CONTEXTS:
252
+ duration = self.dwell.get(context)
253
+ if duration is None or duration <= timedelta(0):
254
+ raise ValueError(f"dwell for {context.value} must be positive")
255
+ for name in ("resident_share", "beacon_presence"):
256
+ values = getattr(self, name)
257
+ if any(not 0.0 <= float(v) <= 1.0 for v in values.values()):
258
+ raise ValueError(f"{name} values must lie in [0, 1]")
259
+ if any(float(v) < 0.0 for v in self.track_counts.values()):
260
+ raise ValueError("track_counts must be non-negative")
261
+ if not 0.0 <= self.door_mixing <= 1.0:
262
+ raise ValueError("door_mixing must lie in [0, 1]")
263
+ if self.concurrency_window <= timedelta(0):
264
+ raise ValueError("concurrency_window must be positive")
265
+ if self.concurrency_odds < 1.0:
266
+ raise ValueError("concurrency_odds must be at least 1")
267
+ if not 0.0 < self.sample_weight <= 1.0:
268
+ raise ValueError("sample_weight must lie in (0, 1]")
269
+
270
+
271
+ class ResidentContextEstimator:
272
+ """Estimate household occupancy and per-sensor attribution online.
273
+
274
+ Parameters
275
+ ----------
276
+ registry
277
+ Sensor declarations. The ``attributable`` flag decides which sensors
278
+ identify the person behind an observation and which do not.
279
+ config
280
+ Model configuration.
281
+ prior
282
+ Initial context belief; defaults to the chain's stationary
283
+ distribution.
284
+ """
285
+
286
+ def __init__(
287
+ self,
288
+ registry: SensorRegistry,
289
+ config: ContextConfig | None = None,
290
+ prior: np.ndarray | None = None,
291
+ ) -> None:
292
+ self.registry = registry
293
+ self.config = config or ContextConfig()
294
+ rates = np.array(
295
+ [1.0 / self.config.dwell[context].total_seconds() for context in CONTEXTS],
296
+ dtype=float,
297
+ )
298
+ self._generator = build_generator(
299
+ rates, np.ones((len(CONTEXTS), len(CONTEXTS)))
300
+ )
301
+ self._prior: np.ndarray = (
302
+ stationary_distribution(self._generator)
303
+ if prior is None
304
+ else self._validated(prior)
305
+ )
306
+ self._belief: np.ndarray = self._prior.copy()
307
+ self._at: datetime | None = None
308
+
309
+ @staticmethod
310
+ def _validated(prior: np.ndarray) -> np.ndarray:
311
+ """Return a normalised context prior."""
312
+ vector = np.asarray(prior, dtype=float)
313
+ if vector.shape != (len(CONTEXTS),):
314
+ raise ValueError("prior must have one entry per occupancy context")
315
+ total = vector.sum()
316
+ if not np.all(np.isfinite(vector)) or vector.min() < 0.0 or total <= 0.0:
317
+ raise ValueError("prior must be finite, non-negative, and non-zero")
318
+ normalised: np.ndarray = vector / total
319
+ return normalised
320
+
321
+ # ------------------------------------------------------------------
322
+ @property
323
+ def belief(self) -> np.ndarray:
324
+ """A copy of the current occupancy posterior."""
325
+ current: np.ndarray = self._belief.copy()
326
+ return current
327
+
328
+ def reset(self) -> None:
329
+ """Return the estimator to its prior and clear its clock."""
330
+ self._belief = self._prior.copy()
331
+ self._at = None
332
+
333
+ def _per_context(
334
+ self, values: Mapping[OccupancyContext, float], default: float
335
+ ) -> np.ndarray:
336
+ """Expand a per-context mapping into a vector in canonical order."""
337
+ return np.array(
338
+ [float(values.get(context, default)) for context in CONTEXTS], dtype=float
339
+ )
340
+
341
+ def _concurrency(self, observations: Sequence[Observation]) -> int:
342
+ """Count distinct rooms with simultaneous activity events.
343
+
344
+ Two rooms lighting up within the concurrency window is the clearest
345
+ anonymous signal that more than one person is moving about: a single
346
+ resident cannot be in the kitchen and the bathroom at the same time.
347
+
348
+ Only discrete activation *events* count. A radar or occupancy device
349
+ reporting a continuous track count is not evidence of a second room
350
+ being used -- its count is already the stronger evidence, handled by
351
+ the track-count likelihood, and counting it here as well would both
352
+ double-count it and turn one tracked person into two.
353
+ """
354
+ located = [
355
+ (obs.timestamp, room)
356
+ for obs in observations
357
+ if obs.is_event
358
+ and obs.value != 0.0
359
+ and (spec := self.registry.get(obs.sensor_id)) is not None
360
+ and not spec.attributable
361
+ and (room := spec.room) is not None
362
+ ]
363
+ if len(located) < 2:
364
+ return len({room for _, room in located})
365
+
366
+ located.sort(key=lambda item: item[0])
367
+ window = self.config.concurrency_window
368
+ best = 1
369
+ for index, (start, _) in enumerate(located):
370
+ rooms = {
371
+ room for moment, room in located[index:] if moment - start <= window
372
+ }
373
+ best = max(best, len(rooms))
374
+ return best
375
+
376
+ def _evidence(
377
+ self,
378
+ observations: Sequence[Observation],
379
+ reliabilities: Mapping[str, float] | None,
380
+ ) -> tuple[np.ndarray, int]:
381
+ """Accumulate the log-likelihood of the interval over contexts."""
382
+ log_likelihood = np.zeros(len(CONTEXTS))
383
+ doors = 0
384
+
385
+ for observation in observations:
386
+ spec = self.registry.get(observation.sensor_id)
387
+ if spec is None:
388
+ continue
389
+ weight = (
390
+ 1.0
391
+ if reliabilities is None
392
+ else float(reliabilities.get(observation.sensor_id, 1.0))
393
+ )
394
+ if weight <= 0.0:
395
+ continue
396
+
397
+ if spec.modality is Modality.DOOR and observation.value != 0.0:
398
+ doors += 1
399
+ continue
400
+
401
+ if spec.modality is Modality.PROXIMITY and spec.attributable:
402
+ presence = np.clip(
403
+ self._per_context(self.config.beacon_presence, 0.5),
404
+ EPSILON,
405
+ 1.0 - EPSILON,
406
+ )
407
+ in_range = observation.value != 0.0
408
+ log_likelihood += (
409
+ weight
410
+ * self.config.sample_weight
411
+ * (np.log(presence) if in_range else np.log1p(-presence))
412
+ )
413
+ continue
414
+
415
+ if spec.modality in (Modality.RADAR, Modality.ROOM_OCCUPANCY):
416
+ counts = np.maximum(
417
+ self._per_context(self.config.track_counts, 1.0), EPSILON
418
+ )
419
+ observed = max(observation.value, 0.0)
420
+ log_likelihood += (
421
+ weight
422
+ * self.config.sample_weight
423
+ * (observed * np.log(counts) - counts)
424
+ )
425
+ continue
426
+
427
+ return log_likelihood, doors
428
+
429
+ def _concurrency_evidence(self, rooms_active: int) -> np.ndarray:
430
+ """Return the log-likelihood contributed by simultaneous room activity."""
431
+ if rooms_active < 2:
432
+ return np.zeros(len(CONTEXTS))
433
+ odds = np.log(self.config.concurrency_odds) * (rooms_active - 1)
434
+ multi = np.array(
435
+ [
436
+ 1.0 if context is OccupancyContext.RESIDENT_WITH_VISITOR else 0.0
437
+ for context in CONTEXTS
438
+ ]
439
+ )
440
+ evidence: np.ndarray = odds * multi
441
+ return evidence
442
+
443
+ def update(
444
+ self,
445
+ now: datetime,
446
+ observations: Sequence[Observation] = (),
447
+ *,
448
+ reliabilities: Mapping[str, float] | None = None,
449
+ ) -> ContextEstimate:
450
+ """Advance the occupancy estimate to *now* over the interval's evidence."""
451
+ moment = require_aware(now, "now")
452
+ if self._at is not None and moment < self._at:
453
+ raise ValueError("context updates must be non-decreasing in time")
454
+
455
+ elapsed = (moment - self._at).total_seconds() if self._at is not None else 0.0
456
+ predicted = self._belief @ transition_matrix(self._generator, elapsed)
457
+
458
+ log_likelihood, doors = self._evidence(observations, reliabilities)
459
+ rooms_active = self._concurrency(observations)
460
+ log_likelihood = log_likelihood + self._concurrency_evidence(rooms_active)
461
+
462
+ log_belief = np.log(np.maximum(predicted, 1e-300)) + log_likelihood
463
+ belief = np.exp(log_belief - logsumexp(log_belief))
464
+
465
+ # A door crossing is the moment occupancy is most likely to change,
466
+ # so it relaxes the belief toward the transition rather than voting
467
+ # for any particular context. Without this the model would be far too
468
+ # confident that whoever was home an hour ago is still home.
469
+ if doors:
470
+ mixing = 1.0 - (1.0 - self.config.door_mixing) ** doors
471
+ belief = (1.0 - mixing) * belief + mixing * self._prior
472
+
473
+ self._belief = belief / belief.sum()
474
+ self._at = moment
475
+ return ContextEstimate(
476
+ at=moment,
477
+ belief=self._belief.copy(),
478
+ concurrency=rooms_active,
479
+ door_events=doors,
480
+ )
481
+
482
+ # ------------------------------------------------------------------
483
+ def attribution(self, estimate: ContextEstimate) -> dict[str, float]:
484
+ """Return the per-sensor attribution weights the fusion layer needs.
485
+
486
+ Sensors declared ``attributable`` -- a worn device, a personal beacon
487
+ -- are bound to the resident by construction and get a weight of one.
488
+ Every ambient sensor gets the marginal probability that the resident,
489
+ rather than someone else in the home, generated what it saw.
490
+ """
491
+ ambient = estimate.ambient_attribution(self.config.resident_share)
492
+ return {
493
+ spec.sensor_id: 1.0 if spec.attributable else ambient
494
+ for spec in self.registry
495
+ }
496
+
497
+ def snapshot(self) -> dict[str, object]:
498
+ """Return restartable estimator state."""
499
+ return {
500
+ "belief": self._belief.tolist(),
501
+ "at": self._at.isoformat() if self._at else None,
502
+ "contexts": [context.value for context in CONTEXTS],
503
+ }
504
+
505
+ def restore(self, state: Mapping[str, object]) -> None:
506
+ """Restore estimator state produced by :meth:`snapshot`."""
507
+ contexts = state.get("contexts")
508
+ if contexts is not None and list(contexts) != [c.value for c in CONTEXTS]: # type: ignore[call-overload]
509
+ raise ValueError("snapshot was taken under different occupancy contexts")
510
+ self._belief = self._validated(np.asarray(state["belief"], dtype=float))
511
+ moment = state.get("at")
512
+ self._at = datetime.fromisoformat(str(moment)) if moment else None
513
+
514
+
515
+ def rooms_active_at(
516
+ registry: SensorRegistry, observations: Iterable[Observation]
517
+ ) -> set[str]:
518
+ """Return the distinct rooms with ambient activations among *observations*."""
519
+ rooms: set[str] = set()
520
+ for observation in observations:
521
+ spec = registry.get(observation.sensor_id)
522
+ if (
523
+ spec is not None
524
+ and spec.room
525
+ and not spec.attributable
526
+ and observation.value
527
+ ):
528
+ rooms.add(spec.room)
529
+ return rooms
@@ -0,0 +1,5 @@
1
+ """Data loading, preprocessing, validation, and simulation utilities."""
2
+
3
+ from . import loaders, preprocessing, synthetic, validation
4
+
5
+ __all__ = ["loaders", "preprocessing", "validation", "synthetic"]
@@ -0,0 +1,146 @@
1
+ """Flexible data loaders for multiple sensor data formats."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import logging
7
+ from collections.abc import Generator, Iterable, Mapping
8
+ from os import PathLike
9
+
10
+ import pandas as pd
11
+
12
+ from sensor_modeling.utils.data_io import SensorDataset, read_sensor_csv
13
+
14
+ logger = logging.getLogger(__name__)
15
+
16
+
17
+ def _parse_timestamps(values: object, field_name: str) -> pd.Series:
18
+ """Parse timestamp values and reject missing or invalid entries."""
19
+ timestamps = pd.to_datetime(values, errors="coerce")
20
+ if pd.isna(timestamps).any():
21
+ raise ValueError(f"Timestamp field '{field_name}' contains invalid timestamps")
22
+ return timestamps
23
+
24
+
25
+ def load_csv(
26
+ path: str | PathLike[str], timestamp_col: str = "timestamp", **kwargs
27
+ ) -> SensorDataset:
28
+ """Load sensor readings from a CSV file.
29
+
30
+ Parameters
31
+ ----------
32
+ path : str
33
+ Path to the CSV file.
34
+ timestamp_col : str, default="timestamp"
35
+ Preferred timestamp column. If absent, an unnamed saved index is parsed
36
+ as datetimes when possible; otherwise the CSV is loaded as a plain
37
+ tabular sensor matrix.
38
+
39
+ Returns
40
+ -------
41
+ SensorDataset
42
+ Dataset containing sensor readings indexed by timestamps.
43
+ """
44
+ try:
45
+ df = read_sensor_csv(path, timestamp_col=timestamp_col, **kwargs)
46
+ except (
47
+ OSError,
48
+ TypeError,
49
+ UnicodeError,
50
+ ValueError,
51
+ pd.errors.ParserError,
52
+ ) as exc:
53
+ logger.error("Failed to read CSV %s: %s", path, exc)
54
+ raise ValueError(f"Unable to read CSV file: {path}") from exc
55
+ logger.info("Loaded CSV with shape %s from %s", df.shape, path)
56
+ return SensorDataset(df)
57
+
58
+
59
+ def load_json(
60
+ path: str | PathLike[str], timestamp_field: str = "timestamp"
61
+ ) -> SensorDataset:
62
+ """Load sensor event log data from a JSON file."""
63
+ try:
64
+ with open(path, encoding="utf-8") as f:
65
+ records = json.load(f)
66
+ except (OSError, UnicodeError, json.JSONDecodeError) as exc:
67
+ logger.error("Failed to read JSON %s: %s", path, exc)
68
+ raise ValueError(f"Unable to read JSON file: {path}") from exc
69
+
70
+ df = _records_to_frame(records, timestamp_field=timestamp_field)
71
+ logger.info("Loaded JSON with shape %s from %s", df.shape, path)
72
+ return SensorDataset(df)
73
+
74
+
75
+ def _records_to_frame(records: object, timestamp_field: str) -> pd.DataFrame:
76
+ """Convert JSON records into a timestamp-indexed DataFrame."""
77
+ if not isinstance(records, list):
78
+ raise ValueError("JSON file must contain a list of records")
79
+ if any(not isinstance(record, Mapping) for record in records):
80
+ raise ValueError("Invalid JSON structure for tabular data")
81
+
82
+ try:
83
+ df = pd.DataFrame(records)
84
+ except (TypeError, ValueError) as exc:
85
+ logger.error("JSON structure invalid: %s", exc)
86
+ raise ValueError("Invalid JSON structure for tabular data") from exc
87
+
88
+ if timestamp_field not in df.columns:
89
+ raise ValueError(
90
+ f"Timestamp field '{timestamp_field}' missing from JSON records"
91
+ )
92
+
93
+ df[timestamp_field] = _parse_timestamps(df[timestamp_field], timestamp_field)
94
+ return df.set_index(timestamp_field).sort_index()
95
+
96
+
97
+ def load_hdf5(path: str | PathLike[str], key: str = "data") -> SensorDataset:
98
+ """Load sensor data from an HDF5 file."""
99
+ try:
100
+ import h5py
101
+ except ImportError as exc: # pragma: no cover - dependency is installed in CI
102
+ raise ImportError("h5py is required for HDF5 support") from exc
103
+
104
+ try:
105
+ with h5py.File(path, "r") as h5:
106
+ if key not in h5:
107
+ raise ValueError(f"Dataset '{key}' not found in HDF5 file")
108
+ data = pd.DataFrame(h5[key][:])
109
+ if "timestamp" in h5[key].attrs:
110
+ ts = pd.to_datetime(h5[key].attrs["timestamp"])
111
+ data.index = ts
112
+ except OSError as exc:
113
+ logger.error("Failed to read HDF5 %s: %s", path, exc)
114
+ raise ValueError(f"Unable to read HDF5 file: {path}") from exc
115
+ logger.info("Loaded HDF5 dataset '%s' with shape %s from %s", key, data.shape, path)
116
+ return SensorDataset(data)
117
+
118
+
119
+ def stream_data(source: Iterable[object]) -> Generator[SensorDataset, None, None]:
120
+ """Yield datasets from a real-time streaming source.
121
+
122
+ Parameters
123
+ ----------
124
+ source : Iterable[Dict]
125
+ Iterable producing dictionaries with sensor readings and timestamps.
126
+ """
127
+ for item in source:
128
+ try:
129
+ yield _stream_item_to_dataset(item)
130
+ except (TypeError, ValueError, KeyError) as exc:
131
+ logger.warning("Skipping malformed streaming item %s: %s", item, exc)
132
+ continue
133
+
134
+
135
+ def _stream_item_to_dataset(item: object) -> SensorDataset:
136
+ """Convert one streaming record into a single-row dataset."""
137
+ if not isinstance(item, Mapping):
138
+ raise TypeError("Streaming item must be a mapping")
139
+
140
+ timestamp = item.get("timestamp")
141
+ if timestamp is None:
142
+ raise ValueError("Streaming item missing 'timestamp' field")
143
+
144
+ timestamp_index = _parse_timestamps([timestamp], "timestamp")
145
+ df = pd.DataFrame([item]).set_index(timestamp_index)
146
+ return SensorDataset(df.drop(columns=["timestamp"]))