sensor-modeling 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sensor_modeling/__init__.py +45 -0
- sensor_modeling/alerts/__init__.py +26 -0
- sensor_modeling/alerts/alert.py +532 -0
- sensor_modeling/analysis/__init__.py +43 -0
- sensor_modeling/analysis/_frame.py +19 -0
- sensor_modeling/analysis/behavioral_analysis.py +57 -0
- sensor_modeling/analysis/behavioral_metrics.py +66 -0
- sensor_modeling/analysis/comparison.py +164 -0
- sensor_modeling/analysis/dependency_network.py +408 -0
- sensor_modeling/analysis/granger_causality.py +314 -0
- sensor_modeling/analysis/pipeline.py +168 -0
- sensor_modeling/analysis/reporting.py +109 -0
- sensor_modeling/baseline/__init__.py +30 -0
- sensor_modeling/baseline/adaptive.py +520 -0
- sensor_modeling/baseline/features.py +224 -0
- sensor_modeling/change_point/__init__.py +13 -0
- sensor_modeling/change_point/_validation.py +31 -0
- sensor_modeling/change_point/adaptive_normalization.py +55 -0
- sensor_modeling/change_point/embedding_cpd.py +60 -0
- sensor_modeling/change_point/energy_efficient.py +57 -0
- sensor_modeling/change_point/genetic_optimization.py +65 -0
- sensor_modeling/cli.py +416 -0
- sensor_modeling/context/__init__.py +33 -0
- sensor_modeling/context/occupancy.py +529 -0
- sensor_modeling/data/__init__.py +5 -0
- sensor_modeling/data/loaders.py +146 -0
- sensor_modeling/data/preprocessing.py +83 -0
- sensor_modeling/data/synthetic.py +121 -0
- sensor_modeling/data/validation.py +81 -0
- sensor_modeling/evaluation/__init__.py +92 -0
- sensor_modeling/evaluation/ablation.py +303 -0
- sensor_modeling/evaluation/attribution.py +474 -0
- sensor_modeling/evaluation/detection.py +297 -0
- sensor_modeling/evaluation/metrics.py +541 -0
- sensor_modeling/evaluation/provenance.py +309 -0
- sensor_modeling/examples/__init__.py +1 -0
- sensor_modeling/examples/demos/__init__.py +1 -0
- sensor_modeling/examples/demos/ambient_pipeline_demo.py +418 -0
- sensor_modeling/examples/demos/bernoulli_ar_demo.py +356 -0
- sensor_modeling/examples/demos/cpd_ar_demo.py +25 -0
- sensor_modeling/examples/demos/cpd_benchmark.py +42 -0
- sensor_modeling/examples/demos/hmm_granger_demo.py +30 -0
- sensor_modeling/examples/demos/nhpp_pelt_demo.py +80 -0
- sensor_modeling/examples/tutorials/__init__.py +1 -0
- sensor_modeling/fusion/__init__.py +46 -0
- sensor_modeling/fusion/defaults.py +296 -0
- sensor_modeling/fusion/emissions.py +339 -0
- sensor_modeling/fusion/estimate.py +375 -0
- sensor_modeling/fusion/filter.py +323 -0
- sensor_modeling/health/__init__.py +31 -0
- sensor_modeling/health/monitor.py +590 -0
- sensor_modeling/health/status.py +74 -0
- sensor_modeling/hmm/__init__.py +15 -0
- sensor_modeling/hmm/adaptive_hmm.py +22 -0
- sensor_modeling/hmm/base.py +134 -0
- sensor_modeling/hmm/circadian_hmm.py +22 -0
- sensor_modeling/hmm/heterogeneous_hmm.py +22 -0
- sensor_modeling/hmm/hierarchical_hmm.py +35 -0
- sensor_modeling/hmm/scaled_dirichlet_hmm.py +23 -0
- sensor_modeling/interop/__init__.py +57 -0
- sensor_modeling/interop/fhir.py +418 -0
- sensor_modeling/interop/privacy.py +308 -0
- sensor_modeling/models/__init__.py +12 -0
- sensor_modeling/models/bernoulli_ar/__init__.py +6 -0
- sensor_modeling/models/bernoulli_ar/base_model.py +569 -0
- sensor_modeling/models/bernoulli_ar/multivariate_model.py +411 -0
- sensor_modeling/models/change_point_detection/__init__.py +10 -0
- sensor_modeling/models/change_point_detection/deep.py +65 -0
- sensor_modeling/models/change_point_detection/pelt.py +159 -0
- sensor_modeling/models/nhpp_pelt/__init__.py +5 -0
- sensor_modeling/models/nhpp_pelt/bspline.py +96 -0
- sensor_modeling/models/nhpp_pelt/cli.py +243 -0
- sensor_modeling/models/nhpp_pelt/diagnostics.py +234 -0
- sensor_modeling/models/nhpp_pelt/io.py +58 -0
- sensor_modeling/models/nhpp_pelt/model.py +408 -0
- sensor_modeling/models/nhpp_pelt/optimizer.py +142 -0
- sensor_modeling/models/nhpp_pelt/plotting.py +218 -0
- sensor_modeling/models/nhpp_pelt/quad.py +72 -0
- sensor_modeling/models/nhpp_pelt/regularization.py +121 -0
- sensor_modeling/models/nhpp_pelt/utils.py +174 -0
- sensor_modeling/observations/__init__.py +59 -0
- sensor_modeling/observations/adapters.py +195 -0
- sensor_modeling/observations/ingest.py +269 -0
- sensor_modeling/observations/observation.py +270 -0
- sensor_modeling/observations/registry.py +262 -0
- sensor_modeling/observations/stream.py +342 -0
- sensor_modeling/observations/types.py +107 -0
- sensor_modeling/observations/units.py +117 -0
- sensor_modeling/online/__init__.py +36 -0
- sensor_modeling/online/benchmarks.py +242 -0
- sensor_modeling/online/pipeline.py +485 -0
- sensor_modeling/simulation/__init__.py +54 -0
- sensor_modeling/simulation/faults.py +191 -0
- sensor_modeling/simulation/household.py +862 -0
- sensor_modeling/states/__init__.py +23 -0
- sensor_modeling/states/markov.py +105 -0
- sensor_modeling/states/ontology.py +238 -0
- sensor_modeling/utils/__init__.py +41 -0
- sensor_modeling/utils/data_io.py +199 -0
- sensor_modeling/utils/logging_config.py +10 -0
- sensor_modeling/utils/missing.py +188 -0
- sensor_modeling/utils/plotting.py +98 -0
- sensor_modeling/utils/validation.py +117 -0
- sensor_modeling/visualization/__init__.py +3 -0
- sensor_modeling/visualization/clinical.py +67 -0
- sensor_modeling/visualization/interactive.py +208 -0
- sensor_modeling/visualization/research.py +60 -0
- sensor_modeling/visualization/web_app.py +137 -0
- sensor_modeling-0.2.0.dist-info/METADATA +683 -0
- sensor_modeling-0.2.0.dist-info/RECORD +114 -0
- sensor_modeling-0.2.0.dist-info/WHEEL +5 -0
- sensor_modeling-0.2.0.dist-info/entry_points.txt +18 -0
- sensor_modeling-0.2.0.dist-info/licenses/LICENSE +21 -0
- sensor_modeling-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,529 @@
|
|
|
1
|
+
"""Who is in the home, and who probably caused what the sensors saw.
|
|
2
|
+
|
|
3
|
+
Ambient sensing has no notion of identity. A door opening, a hallway motion
|
|
4
|
+
trip, a kettle switching on -- none of these say *whose* activity they are. A
|
|
5
|
+
monitoring system that assumes every event belongs to the monitored resident
|
|
6
|
+
will read a daughter's Sunday visit as a sudden improvement in mobility, and a
|
|
7
|
+
carer's morning round as the resident getting up early.
|
|
8
|
+
|
|
9
|
+
This module estimates the household occupancy context probabilistically and
|
|
10
|
+
converts it into an attribution weight per sensor:
|
|
11
|
+
|
|
12
|
+
.. code-block:: text
|
|
13
|
+
|
|
14
|
+
P(resident_home | O)
|
|
15
|
+
P(visitor_present | O)
|
|
16
|
+
P(multiple_people_present | O)
|
|
17
|
+
P(event from sensor s was generated by the resident | O)
|
|
18
|
+
|
|
19
|
+
The last of these is what the fusion layer consumes. The goal is explicitly
|
|
20
|
+
*not* biometric identification: the platform stays compatible with
|
|
21
|
+
privacy-preserving sensing, so there are no cameras, no microphones, and no
|
|
22
|
+
face or voice recognition anywhere in this design. The goal is honest
|
|
23
|
+
uncertainty about attribution, which is achievable from anonymous evidence:
|
|
24
|
+
whether a personal device is in range, how many tracks a radar reports, and
|
|
25
|
+
whether activity is happening in two places at once.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import logging
|
|
31
|
+
from collections.abc import Iterable, Mapping, Sequence
|
|
32
|
+
from dataclasses import dataclass, field
|
|
33
|
+
from datetime import datetime, timedelta
|
|
34
|
+
from enum import Enum
|
|
35
|
+
|
|
36
|
+
import numpy as np
|
|
37
|
+
from scipy.special import logsumexp
|
|
38
|
+
|
|
39
|
+
from ..observations.observation import Observation, require_aware
|
|
40
|
+
from ..observations.registry import SensorRegistry
|
|
41
|
+
from ..observations.types import Modality
|
|
42
|
+
from ..states.markov import build_generator, stationary_distribution, transition_matrix
|
|
43
|
+
|
|
44
|
+
logger = logging.getLogger(__name__)
|
|
45
|
+
|
|
46
|
+
EPSILON = 1e-9
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class OccupancyContext(str, Enum):
|
|
50
|
+
"""Who is present in the home."""
|
|
51
|
+
|
|
52
|
+
EMPTY = "empty"
|
|
53
|
+
"""Nobody present. Ambient activity here indicates a sensor problem."""
|
|
54
|
+
|
|
55
|
+
RESIDENT_ALONE = "resident_alone"
|
|
56
|
+
"""Only the monitored resident. Ambient events are theirs."""
|
|
57
|
+
|
|
58
|
+
RESIDENT_WITH_VISITOR = "resident_with_visitor"
|
|
59
|
+
"""Resident plus at least one other person; ambient events are shared."""
|
|
60
|
+
|
|
61
|
+
VISITOR_ONLY = "visitor_only"
|
|
62
|
+
"""Someone else present while the resident is out, such as a carer."""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
#: Canonical vector order for occupancy contexts.
|
|
66
|
+
CONTEXTS: tuple[OccupancyContext, ...] = (
|
|
67
|
+
OccupancyContext.EMPTY,
|
|
68
|
+
OccupancyContext.RESIDENT_ALONE,
|
|
69
|
+
OccupancyContext.RESIDENT_WITH_VISITOR,
|
|
70
|
+
OccupancyContext.VISITOR_ONLY,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
#: Typical persistence of each context.
|
|
74
|
+
DEFAULT_CONTEXT_DWELL: dict[OccupancyContext, timedelta] = {
|
|
75
|
+
OccupancyContext.EMPTY: timedelta(hours=3),
|
|
76
|
+
OccupancyContext.RESIDENT_ALONE: timedelta(hours=8),
|
|
77
|
+
OccupancyContext.RESIDENT_WITH_VISITOR: timedelta(hours=1),
|
|
78
|
+
OccupancyContext.VISITOR_ONLY: timedelta(minutes=45),
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
#: Expected share of ambient activity generated by the resident in each
|
|
82
|
+
#: context. In a shared household the split is uncertain rather than known,
|
|
83
|
+
#: which is precisely why attribution is a probability and not a flag.
|
|
84
|
+
DEFAULT_RESIDENT_SHARE: dict[OccupancyContext, float] = {
|
|
85
|
+
OccupancyContext.EMPTY: 0.0,
|
|
86
|
+
OccupancyContext.RESIDENT_ALONE: 1.0,
|
|
87
|
+
OccupancyContext.RESIDENT_WITH_VISITOR: 0.5,
|
|
88
|
+
OccupancyContext.VISITOR_ONLY: 0.0,
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
#: Expected number of simultaneously tracked people in each context, for
|
|
92
|
+
#: radar or room-occupancy devices that report a count.
|
|
93
|
+
DEFAULT_TRACK_COUNTS: dict[OccupancyContext, float] = {
|
|
94
|
+
OccupancyContext.EMPTY: 0.02,
|
|
95
|
+
OccupancyContext.RESIDENT_ALONE: 1.0,
|
|
96
|
+
OccupancyContext.RESIDENT_WITH_VISITOR: 2.0,
|
|
97
|
+
OccupancyContext.VISITOR_ONLY: 1.0,
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
#: Probability a personal presence beacon reports the resident in range.
|
|
101
|
+
#: Never one, because adherence is never perfect: a wearable left on the
|
|
102
|
+
#: dresser is not a resident who left the house.
|
|
103
|
+
DEFAULT_BEACON_PRESENCE: dict[OccupancyContext, float] = {
|
|
104
|
+
OccupancyContext.EMPTY: 0.02,
|
|
105
|
+
OccupancyContext.RESIDENT_ALONE: 0.9,
|
|
106
|
+
OccupancyContext.RESIDENT_WITH_VISITOR: 0.9,
|
|
107
|
+
OccupancyContext.VISITOR_ONLY: 0.05,
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
@dataclass(frozen=True)
|
|
112
|
+
class ContextEstimate:
|
|
113
|
+
"""The household occupancy posterior and what it implies for attribution."""
|
|
114
|
+
|
|
115
|
+
at: datetime
|
|
116
|
+
belief: np.ndarray
|
|
117
|
+
concurrency: int
|
|
118
|
+
door_events: int
|
|
119
|
+
|
|
120
|
+
def __post_init__(self) -> None:
|
|
121
|
+
"""Validate and normalise the context belief."""
|
|
122
|
+
belief = np.asarray(self.belief, dtype=float)
|
|
123
|
+
if belief.shape != (len(CONTEXTS),):
|
|
124
|
+
raise ValueError("belief must have one entry per occupancy context")
|
|
125
|
+
total = belief.sum()
|
|
126
|
+
if not np.all(np.isfinite(belief)) or belief.min() < 0.0 or total <= 0.0:
|
|
127
|
+
raise ValueError("belief must be finite, non-negative, and non-zero")
|
|
128
|
+
object.__setattr__(self, "belief", belief / total)
|
|
129
|
+
|
|
130
|
+
@property
|
|
131
|
+
def probabilities(self) -> dict[OccupancyContext, float]:
|
|
132
|
+
"""Posterior probability of each occupancy context."""
|
|
133
|
+
return dict(zip(CONTEXTS, (float(p) for p in self.belief)))
|
|
134
|
+
|
|
135
|
+
@property
|
|
136
|
+
def resident_home(self) -> float:
|
|
137
|
+
"""``P(resident is in the home)``."""
|
|
138
|
+
return float(
|
|
139
|
+
self.belief[CONTEXTS.index(OccupancyContext.RESIDENT_ALONE)]
|
|
140
|
+
+ self.belief[CONTEXTS.index(OccupancyContext.RESIDENT_WITH_VISITOR)]
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
@property
|
|
144
|
+
def visitor_present(self) -> float:
|
|
145
|
+
"""``P(at least one other person is in the home)``."""
|
|
146
|
+
return float(
|
|
147
|
+
self.belief[CONTEXTS.index(OccupancyContext.RESIDENT_WITH_VISITOR)]
|
|
148
|
+
+ self.belief[CONTEXTS.index(OccupancyContext.VISITOR_ONLY)]
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
@property
|
|
152
|
+
def multiple_people(self) -> float:
|
|
153
|
+
"""``P(more than one person is in the home)``."""
|
|
154
|
+
return float(
|
|
155
|
+
self.belief[CONTEXTS.index(OccupancyContext.RESIDENT_WITH_VISITOR)]
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
@property
|
|
159
|
+
def most_likely(self) -> OccupancyContext:
|
|
160
|
+
"""The highest-posterior occupancy context."""
|
|
161
|
+
return CONTEXTS[int(np.argmax(self.belief))]
|
|
162
|
+
|
|
163
|
+
def ambient_attribution(
|
|
164
|
+
self, shares: Mapping[OccupancyContext, float] | None = None
|
|
165
|
+
) -> float:
|
|
166
|
+
"""``P(an ambient event was generated by the resident)``.
|
|
167
|
+
|
|
168
|
+
Marginalises the per-context resident share over the occupancy
|
|
169
|
+
posterior. With the resident certainly alone this is one; with a
|
|
170
|
+
visitor certainly present it falls toward the shared-household split;
|
|
171
|
+
with the home empty it is zero.
|
|
172
|
+
"""
|
|
173
|
+
weights = shares if shares is not None else DEFAULT_RESIDENT_SHARE
|
|
174
|
+
return float(
|
|
175
|
+
sum(
|
|
176
|
+
self.belief[index] * float(weights.get(context, 0.5))
|
|
177
|
+
for index, context in enumerate(CONTEXTS)
|
|
178
|
+
)
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
def to_dict(self) -> dict[str, object]:
|
|
182
|
+
"""Return a serialisable form of the estimate."""
|
|
183
|
+
return {
|
|
184
|
+
"at": self.at.isoformat(),
|
|
185
|
+
"most_likely": self.most_likely.value,
|
|
186
|
+
"resident_home": self.resident_home,
|
|
187
|
+
"visitor_present": self.visitor_present,
|
|
188
|
+
"multiple_people": self.multiple_people,
|
|
189
|
+
"ambient_attribution": self.ambient_attribution(),
|
|
190
|
+
"concurrency": self.concurrency,
|
|
191
|
+
"door_events": self.door_events,
|
|
192
|
+
"probabilities": {
|
|
193
|
+
context.value: probability
|
|
194
|
+
for context, probability in self.probabilities.items()
|
|
195
|
+
},
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
@dataclass
|
|
200
|
+
class ContextConfig:
|
|
201
|
+
"""Configuration for occupancy context estimation.
|
|
202
|
+
|
|
203
|
+
Parameters
|
|
204
|
+
----------
|
|
205
|
+
dwell
|
|
206
|
+
Mean persistence of each occupancy context.
|
|
207
|
+
resident_share
|
|
208
|
+
Expected share of ambient activity generated by the resident.
|
|
209
|
+
track_counts
|
|
210
|
+
Expected simultaneous track count reported by radar-style devices.
|
|
211
|
+
beacon_presence
|
|
212
|
+
Probability a personal presence beacon reports in-range.
|
|
213
|
+
door_mixing
|
|
214
|
+
How much a door event relaxes the context belief toward a transition.
|
|
215
|
+
A door crossing is the moment occupancy is most likely to change, so
|
|
216
|
+
it loosens the belief rather than supplying evidence for one context.
|
|
217
|
+
concurrency_window
|
|
218
|
+
Window used to decide that activity in two rooms was simultaneous.
|
|
219
|
+
concurrency_odds
|
|
220
|
+
Likelihood ratio favouring multi-person contexts for each additional
|
|
221
|
+
room active at once. Concurrent activity in separate rooms is the
|
|
222
|
+
strongest anonymous evidence of more than one person.
|
|
223
|
+
sample_weight
|
|
224
|
+
Discount applied to each individual presence sample. Successive
|
|
225
|
+
readings from the same radar or beacon are strongly correlated --
|
|
226
|
+
they mostly re-observe the same unchanged situation -- so counting
|
|
227
|
+
them as independent measurements would drive the posterior to
|
|
228
|
+
certainty within minutes. This is a deliberate, inspectable
|
|
229
|
+
correction rather than a claim that the samples are independent.
|
|
230
|
+
"""
|
|
231
|
+
|
|
232
|
+
dwell: Mapping[OccupancyContext, timedelta] = field(
|
|
233
|
+
default_factory=lambda: dict(DEFAULT_CONTEXT_DWELL)
|
|
234
|
+
)
|
|
235
|
+
resident_share: Mapping[OccupancyContext, float] = field(
|
|
236
|
+
default_factory=lambda: dict(DEFAULT_RESIDENT_SHARE)
|
|
237
|
+
)
|
|
238
|
+
track_counts: Mapping[OccupancyContext, float] = field(
|
|
239
|
+
default_factory=lambda: dict(DEFAULT_TRACK_COUNTS)
|
|
240
|
+
)
|
|
241
|
+
beacon_presence: Mapping[OccupancyContext, float] = field(
|
|
242
|
+
default_factory=lambda: dict(DEFAULT_BEACON_PRESENCE)
|
|
243
|
+
)
|
|
244
|
+
door_mixing: float = 0.35
|
|
245
|
+
concurrency_window: timedelta = timedelta(seconds=60)
|
|
246
|
+
concurrency_odds: float = 4.0
|
|
247
|
+
sample_weight: float = 0.2
|
|
248
|
+
|
|
249
|
+
def __post_init__(self) -> None:
|
|
250
|
+
"""Validate the configuration."""
|
|
251
|
+
for context in CONTEXTS:
|
|
252
|
+
duration = self.dwell.get(context)
|
|
253
|
+
if duration is None or duration <= timedelta(0):
|
|
254
|
+
raise ValueError(f"dwell for {context.value} must be positive")
|
|
255
|
+
for name in ("resident_share", "beacon_presence"):
|
|
256
|
+
values = getattr(self, name)
|
|
257
|
+
if any(not 0.0 <= float(v) <= 1.0 for v in values.values()):
|
|
258
|
+
raise ValueError(f"{name} values must lie in [0, 1]")
|
|
259
|
+
if any(float(v) < 0.0 for v in self.track_counts.values()):
|
|
260
|
+
raise ValueError("track_counts must be non-negative")
|
|
261
|
+
if not 0.0 <= self.door_mixing <= 1.0:
|
|
262
|
+
raise ValueError("door_mixing must lie in [0, 1]")
|
|
263
|
+
if self.concurrency_window <= timedelta(0):
|
|
264
|
+
raise ValueError("concurrency_window must be positive")
|
|
265
|
+
if self.concurrency_odds < 1.0:
|
|
266
|
+
raise ValueError("concurrency_odds must be at least 1")
|
|
267
|
+
if not 0.0 < self.sample_weight <= 1.0:
|
|
268
|
+
raise ValueError("sample_weight must lie in (0, 1]")
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
class ResidentContextEstimator:
|
|
272
|
+
"""Estimate household occupancy and per-sensor attribution online.
|
|
273
|
+
|
|
274
|
+
Parameters
|
|
275
|
+
----------
|
|
276
|
+
registry
|
|
277
|
+
Sensor declarations. The ``attributable`` flag decides which sensors
|
|
278
|
+
identify the person behind an observation and which do not.
|
|
279
|
+
config
|
|
280
|
+
Model configuration.
|
|
281
|
+
prior
|
|
282
|
+
Initial context belief; defaults to the chain's stationary
|
|
283
|
+
distribution.
|
|
284
|
+
"""
|
|
285
|
+
|
|
286
|
+
def __init__(
|
|
287
|
+
self,
|
|
288
|
+
registry: SensorRegistry,
|
|
289
|
+
config: ContextConfig | None = None,
|
|
290
|
+
prior: np.ndarray | None = None,
|
|
291
|
+
) -> None:
|
|
292
|
+
self.registry = registry
|
|
293
|
+
self.config = config or ContextConfig()
|
|
294
|
+
rates = np.array(
|
|
295
|
+
[1.0 / self.config.dwell[context].total_seconds() for context in CONTEXTS],
|
|
296
|
+
dtype=float,
|
|
297
|
+
)
|
|
298
|
+
self._generator = build_generator(
|
|
299
|
+
rates, np.ones((len(CONTEXTS), len(CONTEXTS)))
|
|
300
|
+
)
|
|
301
|
+
self._prior: np.ndarray = (
|
|
302
|
+
stationary_distribution(self._generator)
|
|
303
|
+
if prior is None
|
|
304
|
+
else self._validated(prior)
|
|
305
|
+
)
|
|
306
|
+
self._belief: np.ndarray = self._prior.copy()
|
|
307
|
+
self._at: datetime | None = None
|
|
308
|
+
|
|
309
|
+
@staticmethod
|
|
310
|
+
def _validated(prior: np.ndarray) -> np.ndarray:
|
|
311
|
+
"""Return a normalised context prior."""
|
|
312
|
+
vector = np.asarray(prior, dtype=float)
|
|
313
|
+
if vector.shape != (len(CONTEXTS),):
|
|
314
|
+
raise ValueError("prior must have one entry per occupancy context")
|
|
315
|
+
total = vector.sum()
|
|
316
|
+
if not np.all(np.isfinite(vector)) or vector.min() < 0.0 or total <= 0.0:
|
|
317
|
+
raise ValueError("prior must be finite, non-negative, and non-zero")
|
|
318
|
+
normalised: np.ndarray = vector / total
|
|
319
|
+
return normalised
|
|
320
|
+
|
|
321
|
+
# ------------------------------------------------------------------
|
|
322
|
+
@property
|
|
323
|
+
def belief(self) -> np.ndarray:
|
|
324
|
+
"""A copy of the current occupancy posterior."""
|
|
325
|
+
current: np.ndarray = self._belief.copy()
|
|
326
|
+
return current
|
|
327
|
+
|
|
328
|
+
def reset(self) -> None:
|
|
329
|
+
"""Return the estimator to its prior and clear its clock."""
|
|
330
|
+
self._belief = self._prior.copy()
|
|
331
|
+
self._at = None
|
|
332
|
+
|
|
333
|
+
def _per_context(
|
|
334
|
+
self, values: Mapping[OccupancyContext, float], default: float
|
|
335
|
+
) -> np.ndarray:
|
|
336
|
+
"""Expand a per-context mapping into a vector in canonical order."""
|
|
337
|
+
return np.array(
|
|
338
|
+
[float(values.get(context, default)) for context in CONTEXTS], dtype=float
|
|
339
|
+
)
|
|
340
|
+
|
|
341
|
+
def _concurrency(self, observations: Sequence[Observation]) -> int:
|
|
342
|
+
"""Count distinct rooms with simultaneous activity events.
|
|
343
|
+
|
|
344
|
+
Two rooms lighting up within the concurrency window is the clearest
|
|
345
|
+
anonymous signal that more than one person is moving about: a single
|
|
346
|
+
resident cannot be in the kitchen and the bathroom at the same time.
|
|
347
|
+
|
|
348
|
+
Only discrete activation *events* count. A radar or occupancy device
|
|
349
|
+
reporting a continuous track count is not evidence of a second room
|
|
350
|
+
being used -- its count is already the stronger evidence, handled by
|
|
351
|
+
the track-count likelihood, and counting it here as well would both
|
|
352
|
+
double-count it and turn one tracked person into two.
|
|
353
|
+
"""
|
|
354
|
+
located = [
|
|
355
|
+
(obs.timestamp, room)
|
|
356
|
+
for obs in observations
|
|
357
|
+
if obs.is_event
|
|
358
|
+
and obs.value != 0.0
|
|
359
|
+
and (spec := self.registry.get(obs.sensor_id)) is not None
|
|
360
|
+
and not spec.attributable
|
|
361
|
+
and (room := spec.room) is not None
|
|
362
|
+
]
|
|
363
|
+
if len(located) < 2:
|
|
364
|
+
return len({room for _, room in located})
|
|
365
|
+
|
|
366
|
+
located.sort(key=lambda item: item[0])
|
|
367
|
+
window = self.config.concurrency_window
|
|
368
|
+
best = 1
|
|
369
|
+
for index, (start, _) in enumerate(located):
|
|
370
|
+
rooms = {
|
|
371
|
+
room for moment, room in located[index:] if moment - start <= window
|
|
372
|
+
}
|
|
373
|
+
best = max(best, len(rooms))
|
|
374
|
+
return best
|
|
375
|
+
|
|
376
|
+
def _evidence(
|
|
377
|
+
self,
|
|
378
|
+
observations: Sequence[Observation],
|
|
379
|
+
reliabilities: Mapping[str, float] | None,
|
|
380
|
+
) -> tuple[np.ndarray, int]:
|
|
381
|
+
"""Accumulate the log-likelihood of the interval over contexts."""
|
|
382
|
+
log_likelihood = np.zeros(len(CONTEXTS))
|
|
383
|
+
doors = 0
|
|
384
|
+
|
|
385
|
+
for observation in observations:
|
|
386
|
+
spec = self.registry.get(observation.sensor_id)
|
|
387
|
+
if spec is None:
|
|
388
|
+
continue
|
|
389
|
+
weight = (
|
|
390
|
+
1.0
|
|
391
|
+
if reliabilities is None
|
|
392
|
+
else float(reliabilities.get(observation.sensor_id, 1.0))
|
|
393
|
+
)
|
|
394
|
+
if weight <= 0.0:
|
|
395
|
+
continue
|
|
396
|
+
|
|
397
|
+
if spec.modality is Modality.DOOR and observation.value != 0.0:
|
|
398
|
+
doors += 1
|
|
399
|
+
continue
|
|
400
|
+
|
|
401
|
+
if spec.modality is Modality.PROXIMITY and spec.attributable:
|
|
402
|
+
presence = np.clip(
|
|
403
|
+
self._per_context(self.config.beacon_presence, 0.5),
|
|
404
|
+
EPSILON,
|
|
405
|
+
1.0 - EPSILON,
|
|
406
|
+
)
|
|
407
|
+
in_range = observation.value != 0.0
|
|
408
|
+
log_likelihood += (
|
|
409
|
+
weight
|
|
410
|
+
* self.config.sample_weight
|
|
411
|
+
* (np.log(presence) if in_range else np.log1p(-presence))
|
|
412
|
+
)
|
|
413
|
+
continue
|
|
414
|
+
|
|
415
|
+
if spec.modality in (Modality.RADAR, Modality.ROOM_OCCUPANCY):
|
|
416
|
+
counts = np.maximum(
|
|
417
|
+
self._per_context(self.config.track_counts, 1.0), EPSILON
|
|
418
|
+
)
|
|
419
|
+
observed = max(observation.value, 0.0)
|
|
420
|
+
log_likelihood += (
|
|
421
|
+
weight
|
|
422
|
+
* self.config.sample_weight
|
|
423
|
+
* (observed * np.log(counts) - counts)
|
|
424
|
+
)
|
|
425
|
+
continue
|
|
426
|
+
|
|
427
|
+
return log_likelihood, doors
|
|
428
|
+
|
|
429
|
+
def _concurrency_evidence(self, rooms_active: int) -> np.ndarray:
|
|
430
|
+
"""Return the log-likelihood contributed by simultaneous room activity."""
|
|
431
|
+
if rooms_active < 2:
|
|
432
|
+
return np.zeros(len(CONTEXTS))
|
|
433
|
+
odds = np.log(self.config.concurrency_odds) * (rooms_active - 1)
|
|
434
|
+
multi = np.array(
|
|
435
|
+
[
|
|
436
|
+
1.0 if context is OccupancyContext.RESIDENT_WITH_VISITOR else 0.0
|
|
437
|
+
for context in CONTEXTS
|
|
438
|
+
]
|
|
439
|
+
)
|
|
440
|
+
evidence: np.ndarray = odds * multi
|
|
441
|
+
return evidence
|
|
442
|
+
|
|
443
|
+
def update(
|
|
444
|
+
self,
|
|
445
|
+
now: datetime,
|
|
446
|
+
observations: Sequence[Observation] = (),
|
|
447
|
+
*,
|
|
448
|
+
reliabilities: Mapping[str, float] | None = None,
|
|
449
|
+
) -> ContextEstimate:
|
|
450
|
+
"""Advance the occupancy estimate to *now* over the interval's evidence."""
|
|
451
|
+
moment = require_aware(now, "now")
|
|
452
|
+
if self._at is not None and moment < self._at:
|
|
453
|
+
raise ValueError("context updates must be non-decreasing in time")
|
|
454
|
+
|
|
455
|
+
elapsed = (moment - self._at).total_seconds() if self._at is not None else 0.0
|
|
456
|
+
predicted = self._belief @ transition_matrix(self._generator, elapsed)
|
|
457
|
+
|
|
458
|
+
log_likelihood, doors = self._evidence(observations, reliabilities)
|
|
459
|
+
rooms_active = self._concurrency(observations)
|
|
460
|
+
log_likelihood = log_likelihood + self._concurrency_evidence(rooms_active)
|
|
461
|
+
|
|
462
|
+
log_belief = np.log(np.maximum(predicted, 1e-300)) + log_likelihood
|
|
463
|
+
belief = np.exp(log_belief - logsumexp(log_belief))
|
|
464
|
+
|
|
465
|
+
# A door crossing is the moment occupancy is most likely to change,
|
|
466
|
+
# so it relaxes the belief toward the transition rather than voting
|
|
467
|
+
# for any particular context. Without this the model would be far too
|
|
468
|
+
# confident that whoever was home an hour ago is still home.
|
|
469
|
+
if doors:
|
|
470
|
+
mixing = 1.0 - (1.0 - self.config.door_mixing) ** doors
|
|
471
|
+
belief = (1.0 - mixing) * belief + mixing * self._prior
|
|
472
|
+
|
|
473
|
+
self._belief = belief / belief.sum()
|
|
474
|
+
self._at = moment
|
|
475
|
+
return ContextEstimate(
|
|
476
|
+
at=moment,
|
|
477
|
+
belief=self._belief.copy(),
|
|
478
|
+
concurrency=rooms_active,
|
|
479
|
+
door_events=doors,
|
|
480
|
+
)
|
|
481
|
+
|
|
482
|
+
# ------------------------------------------------------------------
|
|
483
|
+
def attribution(self, estimate: ContextEstimate) -> dict[str, float]:
|
|
484
|
+
"""Return the per-sensor attribution weights the fusion layer needs.
|
|
485
|
+
|
|
486
|
+
Sensors declared ``attributable`` -- a worn device, a personal beacon
|
|
487
|
+
-- are bound to the resident by construction and get a weight of one.
|
|
488
|
+
Every ambient sensor gets the marginal probability that the resident,
|
|
489
|
+
rather than someone else in the home, generated what it saw.
|
|
490
|
+
"""
|
|
491
|
+
ambient = estimate.ambient_attribution(self.config.resident_share)
|
|
492
|
+
return {
|
|
493
|
+
spec.sensor_id: 1.0 if spec.attributable else ambient
|
|
494
|
+
for spec in self.registry
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
def snapshot(self) -> dict[str, object]:
|
|
498
|
+
"""Return restartable estimator state."""
|
|
499
|
+
return {
|
|
500
|
+
"belief": self._belief.tolist(),
|
|
501
|
+
"at": self._at.isoformat() if self._at else None,
|
|
502
|
+
"contexts": [context.value for context in CONTEXTS],
|
|
503
|
+
}
|
|
504
|
+
|
|
505
|
+
def restore(self, state: Mapping[str, object]) -> None:
|
|
506
|
+
"""Restore estimator state produced by :meth:`snapshot`."""
|
|
507
|
+
contexts = state.get("contexts")
|
|
508
|
+
if contexts is not None and list(contexts) != [c.value for c in CONTEXTS]: # type: ignore[call-overload]
|
|
509
|
+
raise ValueError("snapshot was taken under different occupancy contexts")
|
|
510
|
+
self._belief = self._validated(np.asarray(state["belief"], dtype=float))
|
|
511
|
+
moment = state.get("at")
|
|
512
|
+
self._at = datetime.fromisoformat(str(moment)) if moment else None
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def rooms_active_at(
|
|
516
|
+
registry: SensorRegistry, observations: Iterable[Observation]
|
|
517
|
+
) -> set[str]:
|
|
518
|
+
"""Return the distinct rooms with ambient activations among *observations*."""
|
|
519
|
+
rooms: set[str] = set()
|
|
520
|
+
for observation in observations:
|
|
521
|
+
spec = registry.get(observation.sensor_id)
|
|
522
|
+
if (
|
|
523
|
+
spec is not None
|
|
524
|
+
and spec.room
|
|
525
|
+
and not spec.attributable
|
|
526
|
+
and observation.value
|
|
527
|
+
):
|
|
528
|
+
rooms.add(spec.room)
|
|
529
|
+
return rooms
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""Flexible data loaders for multiple sensor data formats."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
from collections.abc import Generator, Iterable, Mapping
|
|
8
|
+
from os import PathLike
|
|
9
|
+
|
|
10
|
+
import pandas as pd
|
|
11
|
+
|
|
12
|
+
from sensor_modeling.utils.data_io import SensorDataset, read_sensor_csv
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger(__name__)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _parse_timestamps(values: object, field_name: str) -> pd.Series:
|
|
18
|
+
"""Parse timestamp values and reject missing or invalid entries."""
|
|
19
|
+
timestamps = pd.to_datetime(values, errors="coerce")
|
|
20
|
+
if pd.isna(timestamps).any():
|
|
21
|
+
raise ValueError(f"Timestamp field '{field_name}' contains invalid timestamps")
|
|
22
|
+
return timestamps
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def load_csv(
|
|
26
|
+
path: str | PathLike[str], timestamp_col: str = "timestamp", **kwargs
|
|
27
|
+
) -> SensorDataset:
|
|
28
|
+
"""Load sensor readings from a CSV file.
|
|
29
|
+
|
|
30
|
+
Parameters
|
|
31
|
+
----------
|
|
32
|
+
path : str
|
|
33
|
+
Path to the CSV file.
|
|
34
|
+
timestamp_col : str, default="timestamp"
|
|
35
|
+
Preferred timestamp column. If absent, an unnamed saved index is parsed
|
|
36
|
+
as datetimes when possible; otherwise the CSV is loaded as a plain
|
|
37
|
+
tabular sensor matrix.
|
|
38
|
+
|
|
39
|
+
Returns
|
|
40
|
+
-------
|
|
41
|
+
SensorDataset
|
|
42
|
+
Dataset containing sensor readings indexed by timestamps.
|
|
43
|
+
"""
|
|
44
|
+
try:
|
|
45
|
+
df = read_sensor_csv(path, timestamp_col=timestamp_col, **kwargs)
|
|
46
|
+
except (
|
|
47
|
+
OSError,
|
|
48
|
+
TypeError,
|
|
49
|
+
UnicodeError,
|
|
50
|
+
ValueError,
|
|
51
|
+
pd.errors.ParserError,
|
|
52
|
+
) as exc:
|
|
53
|
+
logger.error("Failed to read CSV %s: %s", path, exc)
|
|
54
|
+
raise ValueError(f"Unable to read CSV file: {path}") from exc
|
|
55
|
+
logger.info("Loaded CSV with shape %s from %s", df.shape, path)
|
|
56
|
+
return SensorDataset(df)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def load_json(
|
|
60
|
+
path: str | PathLike[str], timestamp_field: str = "timestamp"
|
|
61
|
+
) -> SensorDataset:
|
|
62
|
+
"""Load sensor event log data from a JSON file."""
|
|
63
|
+
try:
|
|
64
|
+
with open(path, encoding="utf-8") as f:
|
|
65
|
+
records = json.load(f)
|
|
66
|
+
except (OSError, UnicodeError, json.JSONDecodeError) as exc:
|
|
67
|
+
logger.error("Failed to read JSON %s: %s", path, exc)
|
|
68
|
+
raise ValueError(f"Unable to read JSON file: {path}") from exc
|
|
69
|
+
|
|
70
|
+
df = _records_to_frame(records, timestamp_field=timestamp_field)
|
|
71
|
+
logger.info("Loaded JSON with shape %s from %s", df.shape, path)
|
|
72
|
+
return SensorDataset(df)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _records_to_frame(records: object, timestamp_field: str) -> pd.DataFrame:
|
|
76
|
+
"""Convert JSON records into a timestamp-indexed DataFrame."""
|
|
77
|
+
if not isinstance(records, list):
|
|
78
|
+
raise ValueError("JSON file must contain a list of records")
|
|
79
|
+
if any(not isinstance(record, Mapping) for record in records):
|
|
80
|
+
raise ValueError("Invalid JSON structure for tabular data")
|
|
81
|
+
|
|
82
|
+
try:
|
|
83
|
+
df = pd.DataFrame(records)
|
|
84
|
+
except (TypeError, ValueError) as exc:
|
|
85
|
+
logger.error("JSON structure invalid: %s", exc)
|
|
86
|
+
raise ValueError("Invalid JSON structure for tabular data") from exc
|
|
87
|
+
|
|
88
|
+
if timestamp_field not in df.columns:
|
|
89
|
+
raise ValueError(
|
|
90
|
+
f"Timestamp field '{timestamp_field}' missing from JSON records"
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
df[timestamp_field] = _parse_timestamps(df[timestamp_field], timestamp_field)
|
|
94
|
+
return df.set_index(timestamp_field).sort_index()
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def load_hdf5(path: str | PathLike[str], key: str = "data") -> SensorDataset:
|
|
98
|
+
"""Load sensor data from an HDF5 file."""
|
|
99
|
+
try:
|
|
100
|
+
import h5py
|
|
101
|
+
except ImportError as exc: # pragma: no cover - dependency is installed in CI
|
|
102
|
+
raise ImportError("h5py is required for HDF5 support") from exc
|
|
103
|
+
|
|
104
|
+
try:
|
|
105
|
+
with h5py.File(path, "r") as h5:
|
|
106
|
+
if key not in h5:
|
|
107
|
+
raise ValueError(f"Dataset '{key}' not found in HDF5 file")
|
|
108
|
+
data = pd.DataFrame(h5[key][:])
|
|
109
|
+
if "timestamp" in h5[key].attrs:
|
|
110
|
+
ts = pd.to_datetime(h5[key].attrs["timestamp"])
|
|
111
|
+
data.index = ts
|
|
112
|
+
except OSError as exc:
|
|
113
|
+
logger.error("Failed to read HDF5 %s: %s", path, exc)
|
|
114
|
+
raise ValueError(f"Unable to read HDF5 file: {path}") from exc
|
|
115
|
+
logger.info("Loaded HDF5 dataset '%s' with shape %s from %s", key, data.shape, path)
|
|
116
|
+
return SensorDataset(data)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def stream_data(source: Iterable[object]) -> Generator[SensorDataset, None, None]:
|
|
120
|
+
"""Yield datasets from a real-time streaming source.
|
|
121
|
+
|
|
122
|
+
Parameters
|
|
123
|
+
----------
|
|
124
|
+
source : Iterable[Dict]
|
|
125
|
+
Iterable producing dictionaries with sensor readings and timestamps.
|
|
126
|
+
"""
|
|
127
|
+
for item in source:
|
|
128
|
+
try:
|
|
129
|
+
yield _stream_item_to_dataset(item)
|
|
130
|
+
except (TypeError, ValueError, KeyError) as exc:
|
|
131
|
+
logger.warning("Skipping malformed streaming item %s: %s", item, exc)
|
|
132
|
+
continue
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _stream_item_to_dataset(item: object) -> SensorDataset:
|
|
136
|
+
"""Convert one streaming record into a single-row dataset."""
|
|
137
|
+
if not isinstance(item, Mapping):
|
|
138
|
+
raise TypeError("Streaming item must be a mapping")
|
|
139
|
+
|
|
140
|
+
timestamp = item.get("timestamp")
|
|
141
|
+
if timestamp is None:
|
|
142
|
+
raise ValueError("Streaming item missing 'timestamp' field")
|
|
143
|
+
|
|
144
|
+
timestamp_index = _parse_timestamps([timestamp], "timestamp")
|
|
145
|
+
df = pd.DataFrame([item]).set_index(timestamp_index)
|
|
146
|
+
return SensorDataset(df.drop(columns=["timestamp"]))
|