sensor-modeling 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sensor_modeling/__init__.py +45 -0
- sensor_modeling/alerts/__init__.py +26 -0
- sensor_modeling/alerts/alert.py +532 -0
- sensor_modeling/analysis/__init__.py +43 -0
- sensor_modeling/analysis/_frame.py +19 -0
- sensor_modeling/analysis/behavioral_analysis.py +57 -0
- sensor_modeling/analysis/behavioral_metrics.py +66 -0
- sensor_modeling/analysis/comparison.py +164 -0
- sensor_modeling/analysis/dependency_network.py +408 -0
- sensor_modeling/analysis/granger_causality.py +314 -0
- sensor_modeling/analysis/pipeline.py +168 -0
- sensor_modeling/analysis/reporting.py +109 -0
- sensor_modeling/baseline/__init__.py +30 -0
- sensor_modeling/baseline/adaptive.py +520 -0
- sensor_modeling/baseline/features.py +224 -0
- sensor_modeling/change_point/__init__.py +13 -0
- sensor_modeling/change_point/_validation.py +31 -0
- sensor_modeling/change_point/adaptive_normalization.py +55 -0
- sensor_modeling/change_point/embedding_cpd.py +60 -0
- sensor_modeling/change_point/energy_efficient.py +57 -0
- sensor_modeling/change_point/genetic_optimization.py +65 -0
- sensor_modeling/cli.py +416 -0
- sensor_modeling/context/__init__.py +33 -0
- sensor_modeling/context/occupancy.py +529 -0
- sensor_modeling/data/__init__.py +5 -0
- sensor_modeling/data/loaders.py +146 -0
- sensor_modeling/data/preprocessing.py +83 -0
- sensor_modeling/data/synthetic.py +121 -0
- sensor_modeling/data/validation.py +81 -0
- sensor_modeling/evaluation/__init__.py +92 -0
- sensor_modeling/evaluation/ablation.py +303 -0
- sensor_modeling/evaluation/attribution.py +474 -0
- sensor_modeling/evaluation/detection.py +297 -0
- sensor_modeling/evaluation/metrics.py +541 -0
- sensor_modeling/evaluation/provenance.py +309 -0
- sensor_modeling/examples/__init__.py +1 -0
- sensor_modeling/examples/demos/__init__.py +1 -0
- sensor_modeling/examples/demos/ambient_pipeline_demo.py +418 -0
- sensor_modeling/examples/demos/bernoulli_ar_demo.py +356 -0
- sensor_modeling/examples/demos/cpd_ar_demo.py +25 -0
- sensor_modeling/examples/demos/cpd_benchmark.py +42 -0
- sensor_modeling/examples/demos/hmm_granger_demo.py +30 -0
- sensor_modeling/examples/demos/nhpp_pelt_demo.py +80 -0
- sensor_modeling/examples/tutorials/__init__.py +1 -0
- sensor_modeling/fusion/__init__.py +46 -0
- sensor_modeling/fusion/defaults.py +296 -0
- sensor_modeling/fusion/emissions.py +339 -0
- sensor_modeling/fusion/estimate.py +375 -0
- sensor_modeling/fusion/filter.py +323 -0
- sensor_modeling/health/__init__.py +31 -0
- sensor_modeling/health/monitor.py +590 -0
- sensor_modeling/health/status.py +74 -0
- sensor_modeling/hmm/__init__.py +15 -0
- sensor_modeling/hmm/adaptive_hmm.py +22 -0
- sensor_modeling/hmm/base.py +134 -0
- sensor_modeling/hmm/circadian_hmm.py +22 -0
- sensor_modeling/hmm/heterogeneous_hmm.py +22 -0
- sensor_modeling/hmm/hierarchical_hmm.py +35 -0
- sensor_modeling/hmm/scaled_dirichlet_hmm.py +23 -0
- sensor_modeling/interop/__init__.py +57 -0
- sensor_modeling/interop/fhir.py +418 -0
- sensor_modeling/interop/privacy.py +308 -0
- sensor_modeling/models/__init__.py +12 -0
- sensor_modeling/models/bernoulli_ar/__init__.py +6 -0
- sensor_modeling/models/bernoulli_ar/base_model.py +569 -0
- sensor_modeling/models/bernoulli_ar/multivariate_model.py +411 -0
- sensor_modeling/models/change_point_detection/__init__.py +10 -0
- sensor_modeling/models/change_point_detection/deep.py +65 -0
- sensor_modeling/models/change_point_detection/pelt.py +159 -0
- sensor_modeling/models/nhpp_pelt/__init__.py +5 -0
- sensor_modeling/models/nhpp_pelt/bspline.py +96 -0
- sensor_modeling/models/nhpp_pelt/cli.py +243 -0
- sensor_modeling/models/nhpp_pelt/diagnostics.py +234 -0
- sensor_modeling/models/nhpp_pelt/io.py +58 -0
- sensor_modeling/models/nhpp_pelt/model.py +408 -0
- sensor_modeling/models/nhpp_pelt/optimizer.py +142 -0
- sensor_modeling/models/nhpp_pelt/plotting.py +218 -0
- sensor_modeling/models/nhpp_pelt/quad.py +72 -0
- sensor_modeling/models/nhpp_pelt/regularization.py +121 -0
- sensor_modeling/models/nhpp_pelt/utils.py +174 -0
- sensor_modeling/observations/__init__.py +59 -0
- sensor_modeling/observations/adapters.py +195 -0
- sensor_modeling/observations/ingest.py +269 -0
- sensor_modeling/observations/observation.py +270 -0
- sensor_modeling/observations/registry.py +262 -0
- sensor_modeling/observations/stream.py +342 -0
- sensor_modeling/observations/types.py +107 -0
- sensor_modeling/observations/units.py +117 -0
- sensor_modeling/online/__init__.py +36 -0
- sensor_modeling/online/benchmarks.py +242 -0
- sensor_modeling/online/pipeline.py +485 -0
- sensor_modeling/simulation/__init__.py +54 -0
- sensor_modeling/simulation/faults.py +191 -0
- sensor_modeling/simulation/household.py +862 -0
- sensor_modeling/states/__init__.py +23 -0
- sensor_modeling/states/markov.py +105 -0
- sensor_modeling/states/ontology.py +238 -0
- sensor_modeling/utils/__init__.py +41 -0
- sensor_modeling/utils/data_io.py +199 -0
- sensor_modeling/utils/logging_config.py +10 -0
- sensor_modeling/utils/missing.py +188 -0
- sensor_modeling/utils/plotting.py +98 -0
- sensor_modeling/utils/validation.py +117 -0
- sensor_modeling/visualization/__init__.py +3 -0
- sensor_modeling/visualization/clinical.py +67 -0
- sensor_modeling/visualization/interactive.py +208 -0
- sensor_modeling/visualization/research.py +60 -0
- sensor_modeling/visualization/web_app.py +137 -0
- sensor_modeling-0.2.0.dist-info/METADATA +683 -0
- sensor_modeling-0.2.0.dist-info/RECORD +114 -0
- sensor_modeling-0.2.0.dist-info/WHEEL +5 -0
- sensor_modeling-0.2.0.dist-info/entry_points.txt +18 -0
- sensor_modeling-0.2.0.dist-info/licenses/LICENSE +21 -0
- sensor_modeling-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,342 @@
|
|
|
1
|
+
"""Ordered storage and tabular views of canonical observations.
|
|
2
|
+
|
|
3
|
+
An :class:`ObservationStream` keeps observations in timestamp order regardless
|
|
4
|
+
of the order in which they arrive, collapses exact duplicates, and can report
|
|
5
|
+
where a sensor went quiet. It also converts observations into the tabular
|
|
6
|
+
form the existing modelling code expects.
|
|
7
|
+
|
|
8
|
+
The conversion deliberately offers three different framings, because the three
|
|
9
|
+
:class:`~sensor_modeling.observations.types.ObservationKind` values cannot be
|
|
10
|
+
tabulated the same way. Event streams are counted, never forward-filled: an
|
|
11
|
+
empty bin means "no event was recorded", which is evidence about the sensor as
|
|
12
|
+
much as about the resident, and turning it into a zero would silently convert a
|
|
13
|
+
dead sensor into observed inactivity.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import bisect
|
|
19
|
+
import logging
|
|
20
|
+
from collections.abc import Iterable, Iterator, Sequence
|
|
21
|
+
from dataclasses import dataclass, field
|
|
22
|
+
from datetime import datetime, timedelta, timezone, tzinfo
|
|
23
|
+
|
|
24
|
+
import numpy as np
|
|
25
|
+
import pandas as pd
|
|
26
|
+
|
|
27
|
+
from .observation import Observation, require_aware
|
|
28
|
+
from .types import ObservationKind
|
|
29
|
+
|
|
30
|
+
logger = logging.getLogger(__name__)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _utc_key(observation: Observation) -> datetime:
|
|
34
|
+
"""Return the UTC instant used to order an observation."""
|
|
35
|
+
return observation.timestamp.astimezone(timezone.utc)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class Gap:
|
|
40
|
+
"""A period during which a sensor reported nothing.
|
|
41
|
+
|
|
42
|
+
A gap is a statement about the *record*, not about the resident. Whether
|
|
43
|
+
it reflects a broken sensor or a genuinely quiet period is decided by the
|
|
44
|
+
health monitor, not here.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
sensor_id: str
|
|
48
|
+
start: datetime
|
|
49
|
+
end: datetime
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def duration(self) -> timedelta:
|
|
53
|
+
"""Length of the silent period."""
|
|
54
|
+
return self.end - self.start
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass
|
|
58
|
+
class ObservationStream:
|
|
59
|
+
"""A timestamp-ordered, duplicate-free collection of observations."""
|
|
60
|
+
|
|
61
|
+
_items: list[Observation] = field(default_factory=list, repr=False)
|
|
62
|
+
_identities: set[tuple[datetime, str, float]] = field(
|
|
63
|
+
default_factory=set, repr=False
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
@classmethod
|
|
67
|
+
def from_observations(
|
|
68
|
+
cls, observations: Iterable[Observation]
|
|
69
|
+
) -> ObservationStream:
|
|
70
|
+
"""Build a stream from *observations*, in any order."""
|
|
71
|
+
stream = cls()
|
|
72
|
+
stream.extend(observations)
|
|
73
|
+
return stream
|
|
74
|
+
|
|
75
|
+
# ------------------------------------------------------------------
|
|
76
|
+
def add(self, observation: Observation) -> bool:
|
|
77
|
+
"""Insert *observation* in timestamp order.
|
|
78
|
+
|
|
79
|
+
Returns
|
|
80
|
+
-------
|
|
81
|
+
bool
|
|
82
|
+
``True`` when the observation was stored, ``False`` when it was
|
|
83
|
+
an exact duplicate of a record already present and was dropped.
|
|
84
|
+
"""
|
|
85
|
+
identity = observation.identity()
|
|
86
|
+
if identity in self._identities:
|
|
87
|
+
logger.debug("Dropping duplicate observation %s", identity)
|
|
88
|
+
return False
|
|
89
|
+
bisect.insort(self._items, observation, key=_utc_key)
|
|
90
|
+
self._identities.add(identity)
|
|
91
|
+
return True
|
|
92
|
+
|
|
93
|
+
def extend(self, observations: Iterable[Observation]) -> int:
|
|
94
|
+
"""Insert many observations and return how many were stored."""
|
|
95
|
+
return sum(1 for obs in observations if self.add(obs))
|
|
96
|
+
|
|
97
|
+
def would_be_out_of_order(self, observation: Observation) -> bool:
|
|
98
|
+
"""Whether *observation* predates the newest record already stored."""
|
|
99
|
+
if not self._items:
|
|
100
|
+
return False
|
|
101
|
+
return _utc_key(observation) < _utc_key(self._items[-1])
|
|
102
|
+
|
|
103
|
+
# ------------------------------------------------------------------
|
|
104
|
+
def __len__(self) -> int:
|
|
105
|
+
return len(self._items)
|
|
106
|
+
|
|
107
|
+
def __iter__(self) -> Iterator[Observation]:
|
|
108
|
+
return iter(self._items)
|
|
109
|
+
|
|
110
|
+
def __getitem__(self, index: int) -> Observation:
|
|
111
|
+
return self._items[index]
|
|
112
|
+
|
|
113
|
+
@property
|
|
114
|
+
def start(self) -> datetime | None:
|
|
115
|
+
"""Timestamp of the earliest observation, if any."""
|
|
116
|
+
return self._items[0].timestamp if self._items else None
|
|
117
|
+
|
|
118
|
+
@property
|
|
119
|
+
def end(self) -> datetime | None:
|
|
120
|
+
"""Timestamp of the latest observation, if any."""
|
|
121
|
+
return self._items[-1].timestamp if self._items else None
|
|
122
|
+
|
|
123
|
+
def sensor_ids(self) -> list[str]:
|
|
124
|
+
"""Return the sorted set of sensors that appear in the stream."""
|
|
125
|
+
return sorted({obs.sensor_id for obs in self._items})
|
|
126
|
+
|
|
127
|
+
def by_sensor(self, sensor_id: str) -> list[Observation]:
|
|
128
|
+
"""Return every observation from *sensor_id*, in timestamp order."""
|
|
129
|
+
return [obs for obs in self._items if obs.sensor_id == sensor_id]
|
|
130
|
+
|
|
131
|
+
def between(self, start: datetime, end: datetime) -> list[Observation]:
|
|
132
|
+
"""Return observations in the half-open interval ``[start, end)``."""
|
|
133
|
+
start_utc = require_aware(start, "start").astimezone(timezone.utc)
|
|
134
|
+
end_utc = require_aware(end, "end").astimezone(timezone.utc)
|
|
135
|
+
if end_utc < start_utc:
|
|
136
|
+
raise ValueError("end must not precede start")
|
|
137
|
+
keys = [_utc_key(obs) for obs in self._items]
|
|
138
|
+
left = bisect.bisect_left(keys, start_utc)
|
|
139
|
+
right = bisect.bisect_left(keys, end_utc)
|
|
140
|
+
return self._items[left:right]
|
|
141
|
+
|
|
142
|
+
# ------------------------------------------------------------------
|
|
143
|
+
def gaps(self, sensor_id: str, max_interval: timedelta) -> list[Gap]:
|
|
144
|
+
"""Return periods where *sensor_id* was silent for longer than allowed.
|
|
145
|
+
|
|
146
|
+
Only the interior of the record is examined. Silence before the first
|
|
147
|
+
or after the last observation is not reported, because the stream
|
|
148
|
+
cannot tell a missing sensor from one that has not started yet.
|
|
149
|
+
"""
|
|
150
|
+
if max_interval <= timedelta(0):
|
|
151
|
+
raise ValueError("max_interval must be positive")
|
|
152
|
+
observations = self.by_sensor(sensor_id)
|
|
153
|
+
found: list[Gap] = []
|
|
154
|
+
for previous, current in zip(observations, observations[1:]):
|
|
155
|
+
if current.timestamp - previous.timestamp > max_interval:
|
|
156
|
+
found.append(Gap(sensor_id, previous.timestamp, current.timestamp))
|
|
157
|
+
return found
|
|
158
|
+
|
|
159
|
+
# ------------------------------------------------------------------
|
|
160
|
+
def _grid(self, freq: str, tz: tzinfo | None) -> pd.DatetimeIndex:
|
|
161
|
+
"""Return the regular time grid the framing helpers bin onto.
|
|
162
|
+
|
|
163
|
+
Bin edges are aligned to the local clock, because behavioural rhythms
|
|
164
|
+
follow local time, but the grid itself is generated from a fixed
|
|
165
|
+
offset so that its instants stay contiguous across DST transitions.
|
|
166
|
+
"""
|
|
167
|
+
if not self._items:
|
|
168
|
+
return pd.DatetimeIndex([], tz=tz or timezone.utc)
|
|
169
|
+
zone = tz if tz is not None else self._items[0].timestamp.tzinfo
|
|
170
|
+
step = pd.Timedelta(freq)
|
|
171
|
+
if step <= pd.Timedelta(0):
|
|
172
|
+
raise ValueError("freq must be a positive fixed frequency")
|
|
173
|
+
first = pd.Timestamp(self._items[0].timestamp).tz_convert(zone)
|
|
174
|
+
last = pd.Timestamp(self._items[-1].timestamp).tz_convert(zone)
|
|
175
|
+
start = first.floor(freq, ambiguous=True, nonexistent="shift_backward")
|
|
176
|
+
end = last.floor(freq, ambiguous=True, nonexistent="shift_backward")
|
|
177
|
+
return pd.date_range(start=start, end=end, freq=step, tz=zone)
|
|
178
|
+
|
|
179
|
+
@staticmethod
|
|
180
|
+
def _positions(
|
|
181
|
+
index: pd.DatetimeIndex, observations: Sequence[Observation], freq: str
|
|
182
|
+
) -> np.ndarray:
|
|
183
|
+
"""Map observations onto grid positions, or ``-1`` when outside it.
|
|
184
|
+
|
|
185
|
+
The lookup is done on absolute UTC instants rather than local wall
|
|
186
|
+
times. Flooring a wall time that falls inside a spring-forward gap
|
|
187
|
+
produces a local instant that does not exist on the grid, which would
|
|
188
|
+
drop every observation recorded during the transition.
|
|
189
|
+
"""
|
|
190
|
+
if len(index) == 0 or not observations:
|
|
191
|
+
return np.full(len(observations), -1, dtype=int)
|
|
192
|
+
edges = index.tz_convert("UTC").to_numpy(dtype="datetime64[ns]").astype("int64")
|
|
193
|
+
span = int(pd.Timedelta(freq).value)
|
|
194
|
+
stamps = np.array(
|
|
195
|
+
[
|
|
196
|
+
pd.Timestamp(obs.timestamp).tz_convert("UTC").value
|
|
197
|
+
for obs in observations
|
|
198
|
+
],
|
|
199
|
+
dtype="int64",
|
|
200
|
+
)
|
|
201
|
+
positions = np.searchsorted(edges, stamps, side="right") - 1
|
|
202
|
+
positions[stamps >= edges[-1] + span] = -1
|
|
203
|
+
return positions
|
|
204
|
+
|
|
205
|
+
def _selected(self, sensor_ids: Sequence[str] | None) -> list[str]:
|
|
206
|
+
"""Return the sensor columns to build, defaulting to all present."""
|
|
207
|
+
return list(sensor_ids) if sensor_ids is not None else self.sensor_ids()
|
|
208
|
+
|
|
209
|
+
def _placed(
|
|
210
|
+
self,
|
|
211
|
+
kind: ObservationKind,
|
|
212
|
+
freq: str,
|
|
213
|
+
index: pd.DatetimeIndex,
|
|
214
|
+
columns: list[str],
|
|
215
|
+
) -> list[tuple[int, int, float]]:
|
|
216
|
+
"""Return ``(row, column, value)`` triples for one observation kind.
|
|
217
|
+
|
|
218
|
+
Triples are produced in timestamp order, so a consumer that only
|
|
219
|
+
wants the most recent value per cell can simply overwrite as it goes.
|
|
220
|
+
"""
|
|
221
|
+
column_of = {name: position for position, name in enumerate(columns)}
|
|
222
|
+
selected = [
|
|
223
|
+
obs
|
|
224
|
+
for obs in self._items
|
|
225
|
+
if obs.kind is kind and obs.sensor_id in column_of
|
|
226
|
+
]
|
|
227
|
+
rows = self._positions(index, selected, freq)
|
|
228
|
+
placed = [
|
|
229
|
+
(int(row), column_of[obs.sensor_id], obs.value)
|
|
230
|
+
for obs, row in zip(selected, rows)
|
|
231
|
+
if row >= 0
|
|
232
|
+
]
|
|
233
|
+
dropped = len(selected) - len(placed)
|
|
234
|
+
if dropped:
|
|
235
|
+
logger.warning(
|
|
236
|
+
"%d %s observations fell outside the framing grid", dropped, kind.value
|
|
237
|
+
)
|
|
238
|
+
return placed
|
|
239
|
+
|
|
240
|
+
def event_counts(
|
|
241
|
+
self,
|
|
242
|
+
freq: str = "15min",
|
|
243
|
+
*,
|
|
244
|
+
sensor_ids: Sequence[str] | None = None,
|
|
245
|
+
tz: tzinfo | None = None,
|
|
246
|
+
) -> pd.DataFrame:
|
|
247
|
+
"""Return the number of recorded activations per sensor per time bin.
|
|
248
|
+
|
|
249
|
+
A zero means no event was *recorded* in that bin. It does not mean the
|
|
250
|
+
sensor was working and observed nothing; pair this frame with
|
|
251
|
+
:meth:`observed_mask` and sensor health output before treating zeros
|
|
252
|
+
as evidence of inactivity.
|
|
253
|
+
"""
|
|
254
|
+
index = self._grid(freq, tz)
|
|
255
|
+
columns = self._selected(sensor_ids)
|
|
256
|
+
counts = np.zeros((len(index), len(columns)))
|
|
257
|
+
for row, column, value in self._placed(
|
|
258
|
+
ObservationKind.EVENT, freq, index, columns
|
|
259
|
+
):
|
|
260
|
+
counts[row, column] += 1.0 if value != 0.0 else 0.0
|
|
261
|
+
return pd.DataFrame(counts, index=index, columns=columns)
|
|
262
|
+
|
|
263
|
+
def sample_frame(
|
|
264
|
+
self,
|
|
265
|
+
freq: str = "15min",
|
|
266
|
+
*,
|
|
267
|
+
sensor_ids: Sequence[str] | None = None,
|
|
268
|
+
tz: tzinfo | None = None,
|
|
269
|
+
) -> pd.DataFrame:
|
|
270
|
+
"""Return the mean sampled value per sensor per bin.
|
|
271
|
+
|
|
272
|
+
Bins with no sample are ``NaN`` and are left that way. Filling them
|
|
273
|
+
would fabricate measurements of a continuously existing quantity.
|
|
274
|
+
"""
|
|
275
|
+
index = self._grid(freq, tz)
|
|
276
|
+
columns = self._selected(sensor_ids)
|
|
277
|
+
totals = np.zeros((len(index), len(columns)))
|
|
278
|
+
counts = np.zeros((len(index), len(columns)))
|
|
279
|
+
for row, column, value in self._placed(
|
|
280
|
+
ObservationKind.SAMPLE, freq, index, columns
|
|
281
|
+
):
|
|
282
|
+
totals[row, column] += value
|
|
283
|
+
counts[row, column] += 1.0
|
|
284
|
+
means = np.divide(
|
|
285
|
+
totals, counts, out=np.full_like(totals, np.nan), where=counts > 0
|
|
286
|
+
)
|
|
287
|
+
return pd.DataFrame(means, index=index, columns=columns)
|
|
288
|
+
|
|
289
|
+
def state_frame(
|
|
290
|
+
self,
|
|
291
|
+
freq: str = "15min",
|
|
292
|
+
*,
|
|
293
|
+
sensor_ids: Sequence[str] | None = None,
|
|
294
|
+
tz: tzinfo | None = None,
|
|
295
|
+
max_hold: timedelta | None = None,
|
|
296
|
+
) -> pd.DataFrame:
|
|
297
|
+
"""Return the last reported state per sensor per bin.
|
|
298
|
+
|
|
299
|
+
State observations persist until the next reported change, so carrying
|
|
300
|
+
the last value forward is meaningful here -- but only for *max_hold*.
|
|
301
|
+
Beyond that the state is unknown rather than unchanged, and the cell
|
|
302
|
+
becomes ``NaN``.
|
|
303
|
+
"""
|
|
304
|
+
index = self._grid(freq, tz)
|
|
305
|
+
columns = self._selected(sensor_ids)
|
|
306
|
+
values = np.full((len(index), len(columns)), np.nan)
|
|
307
|
+
for row, column, value in self._placed(
|
|
308
|
+
ObservationKind.STATE, freq, index, columns
|
|
309
|
+
):
|
|
310
|
+
values[row, column] = value
|
|
311
|
+
|
|
312
|
+
frame = pd.DataFrame(values, index=index, columns=columns)
|
|
313
|
+
if len(index) == 0 or max_hold is None:
|
|
314
|
+
return frame.ffill() if len(index) else frame
|
|
315
|
+
limit = max(int(max_hold / pd.Timedelta(freq)), 1)
|
|
316
|
+
return frame.ffill(limit=limit)
|
|
317
|
+
|
|
318
|
+
def observed_mask(
|
|
319
|
+
self,
|
|
320
|
+
freq: str = "15min",
|
|
321
|
+
*,
|
|
322
|
+
sensor_ids: Sequence[str] | None = None,
|
|
323
|
+
tz: tzinfo | None = None,
|
|
324
|
+
) -> pd.DataFrame:
|
|
325
|
+
"""Return which bins contain at least one record per sensor.
|
|
326
|
+
|
|
327
|
+
This is the companion to :meth:`event_counts`: it separates "the
|
|
328
|
+
sensor reported nothing" from "the sensor reported no activity".
|
|
329
|
+
"""
|
|
330
|
+
index = self._grid(freq, tz)
|
|
331
|
+
columns = self._selected(sensor_ids)
|
|
332
|
+
mask = np.zeros((len(index), len(columns)), dtype=bool)
|
|
333
|
+
column_of = {name: position for position, name in enumerate(columns)}
|
|
334
|
+
relevant = [obs for obs in self._items if obs.sensor_id in column_of]
|
|
335
|
+
for obs, row in zip(relevant, self._positions(index, relevant, freq)):
|
|
336
|
+
if row >= 0:
|
|
337
|
+
mask[int(row), column_of[obs.sensor_id]] = True
|
|
338
|
+
return pd.DataFrame(mask, index=index, columns=columns)
|
|
339
|
+
|
|
340
|
+
def to_dicts(self) -> list[dict[str, object]]:
|
|
341
|
+
"""Return every observation as a JSON-serialisable mapping."""
|
|
342
|
+
return [obs.to_dict() for obs in self._items]
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""Core enumerations for the canonical sensor observation model.
|
|
2
|
+
|
|
3
|
+
The types in this module are deliberately hardware-neutral. A concrete
|
|
4
|
+
device is mapped onto a :class:`Modality` and an :class:`ObservationKind`
|
|
5
|
+
by an adapter, so that downstream inference never depends on a specific
|
|
6
|
+
manufacturer, protocol, or product line.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from enum import Enum
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class Modality(str, Enum):
|
|
15
|
+
"""Sensing modality of an observation.
|
|
16
|
+
|
|
17
|
+
The modality describes *what kind of physical evidence* a sensor
|
|
18
|
+
produces, not what behaviour it implies. A ``CONTACT`` observation on a
|
|
19
|
+
fridge door records that the door moved; it does not record eating.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
CONTACT = "contact"
|
|
23
|
+
"""Binary open/close contact on an object (cupboard, fridge, drawer)."""
|
|
24
|
+
|
|
25
|
+
DOOR = "door"
|
|
26
|
+
"""Contact sensor on an entrance or room door used for transitions."""
|
|
27
|
+
|
|
28
|
+
MOTION = "motion"
|
|
29
|
+
"""Passive infrared or equivalent binary movement detection."""
|
|
30
|
+
|
|
31
|
+
VIBRATION = "vibration"
|
|
32
|
+
"""Accelerometer-derived vibration on furniture or appliances."""
|
|
33
|
+
|
|
34
|
+
ENVIRONMENTAL = "environmental"
|
|
35
|
+
"""Ambient scalar measurement (temperature, humidity, light, CO2)."""
|
|
36
|
+
|
|
37
|
+
BED_PRESSURE = "bed_pressure"
|
|
38
|
+
"""Bed or chair occupancy from pressure or load-cell sensing."""
|
|
39
|
+
|
|
40
|
+
WEARABLE_MOTION = "wearable_motion"
|
|
41
|
+
"""Accelerometer-derived activity counts or magnitude from a wearable."""
|
|
42
|
+
|
|
43
|
+
WEARABLE_PHYSIOLOGY = "wearable_physiology"
|
|
44
|
+
"""Physiological signal from a wearable (heart rate, skin temperature)."""
|
|
45
|
+
|
|
46
|
+
ROOM_OCCUPANCY = "room_occupancy"
|
|
47
|
+
"""Room-level occupancy estimate produced by an upstream device."""
|
|
48
|
+
|
|
49
|
+
RADAR = "radar"
|
|
50
|
+
"""Derived feature from mmWave/radar sensing; never raw radar cubes."""
|
|
51
|
+
|
|
52
|
+
PROXIMITY = "proximity"
|
|
53
|
+
"""Short-range presence beacon (BLE-style) associated with an identity."""
|
|
54
|
+
|
|
55
|
+
OTHER = "other"
|
|
56
|
+
"""Modality not covered above; adapters should document the semantics."""
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class ObservationKind(str, Enum):
|
|
60
|
+
"""Temporal semantics of an observation.
|
|
61
|
+
|
|
62
|
+
This distinction controls what may legitimately be done with gaps in the
|
|
63
|
+
record, and is the reason event streams are never forward-filled.
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
EVENT = "event"
|
|
67
|
+
"""Instantaneous occurrence. Absence of an event is *not* a zero value."""
|
|
68
|
+
|
|
69
|
+
STATE = "state"
|
|
70
|
+
"""A level that persists until the next reported change (e.g. door open)."""
|
|
71
|
+
|
|
72
|
+
SAMPLE = "sample"
|
|
73
|
+
"""A measurement of a continuously existing quantity at a point in time."""
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class ObservationFlag(str, Enum):
|
|
77
|
+
"""Provenance and integrity annotations attached to an observation.
|
|
78
|
+
|
|
79
|
+
Flags are set by ingestion and validation code, never by inference. They
|
|
80
|
+
let downstream stages distinguish a genuinely measured value from one
|
|
81
|
+
that was repaired, reordered, or reconstructed.
|
|
82
|
+
"""
|
|
83
|
+
|
|
84
|
+
LATE_ARRIVAL = "late_arrival"
|
|
85
|
+
"""Observation reached the system materially after its own timestamp."""
|
|
86
|
+
|
|
87
|
+
OUT_OF_ORDER = "out_of_order"
|
|
88
|
+
"""Observation was inserted before an already-ingested later observation."""
|
|
89
|
+
|
|
90
|
+
DUPLICATE_VALUE = "duplicate_value"
|
|
91
|
+
"""An identical observation was already present and was collapsed."""
|
|
92
|
+
|
|
93
|
+
UNIT_CONVERTED = "unit_converted"
|
|
94
|
+
"""The value was converted from the reported unit to the canonical unit."""
|
|
95
|
+
|
|
96
|
+
CLOCK_ADJUSTED = "clock_adjusted"
|
|
97
|
+
"""The timestamp was corrected using an estimated per-source clock offset."""
|
|
98
|
+
|
|
99
|
+
IMPUTED = "imputed"
|
|
100
|
+
"""The value was reconstructed rather than measured; treat as weak evidence."""
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
#: Modalities whose default temporal semantics are event-like. Adapters may
|
|
104
|
+
#: override the kind explicitly, but these defaults keep the common case safe.
|
|
105
|
+
EVENT_LIKE_MODALITIES = frozenset(
|
|
106
|
+
{Modality.CONTACT, Modality.DOOR, Modality.MOTION, Modality.VIBRATION}
|
|
107
|
+
)
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""Unit handling for canonical sensor observations.
|
|
2
|
+
|
|
3
|
+
Ambient deployments routinely mix units: one gateway reports Celsius, another
|
|
4
|
+
Fahrenheit; one radar reports metres, another centimetres. Silently mixing
|
|
5
|
+
them corrupts every downstream statistic, so units are explicit on every
|
|
6
|
+
observation and converted to a canonical unit at the ingestion boundary.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from enum import Enum
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class Unit(str, Enum):
|
|
15
|
+
"""Units supported by the canonical observation model."""
|
|
16
|
+
|
|
17
|
+
NONE = "none"
|
|
18
|
+
"""Dimensionless value, including binary 0/1 activations."""
|
|
19
|
+
|
|
20
|
+
PROBABILITY = "probability"
|
|
21
|
+
"""Value constrained to ``[0, 1]``, used for derived presence features."""
|
|
22
|
+
|
|
23
|
+
COUNT = "count"
|
|
24
|
+
"""Non-negative integer count (tracks, steps, activation counts)."""
|
|
25
|
+
|
|
26
|
+
CELSIUS = "degC"
|
|
27
|
+
FAHRENHEIT = "degF"
|
|
28
|
+
PERCENT = "percent"
|
|
29
|
+
LUX = "lux"
|
|
30
|
+
PPM = "ppm"
|
|
31
|
+
METRE = "m"
|
|
32
|
+
CENTIMETRE = "cm"
|
|
33
|
+
METRE_PER_SECOND = "m/s"
|
|
34
|
+
CENTIMETRE_PER_SECOND = "cm/s"
|
|
35
|
+
G = "g"
|
|
36
|
+
"""Acceleration in multiples of standard gravity."""
|
|
37
|
+
|
|
38
|
+
MILLI_G = "mg"
|
|
39
|
+
BPM = "bpm"
|
|
40
|
+
"""Beats or breaths per minute."""
|
|
41
|
+
|
|
42
|
+
SECOND = "s"
|
|
43
|
+
MINUTE = "min"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
#: Multiplicative conversions to the canonical unit of each dimension.
|
|
47
|
+
#: Affine conversions (temperature) are handled separately in :func:`convert`.
|
|
48
|
+
_SCALE_TO_CANONICAL: dict[Unit, tuple[Unit, float]] = {
|
|
49
|
+
Unit.CENTIMETRE: (Unit.METRE, 0.01),
|
|
50
|
+
Unit.CENTIMETRE_PER_SECOND: (Unit.METRE_PER_SECOND, 0.01),
|
|
51
|
+
Unit.MILLI_G: (Unit.G, 0.001),
|
|
52
|
+
Unit.MINUTE: (Unit.SECOND, 60.0),
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
#: Units that are already canonical for their dimension.
|
|
56
|
+
_CANONICAL_UNITS = frozenset(
|
|
57
|
+
{
|
|
58
|
+
Unit.NONE,
|
|
59
|
+
Unit.PROBABILITY,
|
|
60
|
+
Unit.COUNT,
|
|
61
|
+
Unit.CELSIUS,
|
|
62
|
+
Unit.PERCENT,
|
|
63
|
+
Unit.LUX,
|
|
64
|
+
Unit.PPM,
|
|
65
|
+
Unit.METRE,
|
|
66
|
+
Unit.METRE_PER_SECOND,
|
|
67
|
+
Unit.G,
|
|
68
|
+
Unit.BPM,
|
|
69
|
+
Unit.SECOND,
|
|
70
|
+
}
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def canonical_unit(unit: Unit) -> Unit:
|
|
75
|
+
"""Return the canonical unit for the dimension of *unit*."""
|
|
76
|
+
if unit is Unit.FAHRENHEIT:
|
|
77
|
+
return Unit.CELSIUS
|
|
78
|
+
scaled = _SCALE_TO_CANONICAL.get(unit)
|
|
79
|
+
return scaled[0] if scaled is not None else unit
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def convert(value: float, from_unit: Unit, to_unit: Unit) -> float:
|
|
83
|
+
"""Convert *value* between two units of the same dimension.
|
|
84
|
+
|
|
85
|
+
Raises
|
|
86
|
+
------
|
|
87
|
+
ValueError
|
|
88
|
+
If the units belong to different dimensions and no conversion is
|
|
89
|
+
defined. Mismatched dimensions are a data-integrity problem, not
|
|
90
|
+
something to paper over with a pass-through.
|
|
91
|
+
"""
|
|
92
|
+
if from_unit is to_unit:
|
|
93
|
+
return float(value)
|
|
94
|
+
|
|
95
|
+
if from_unit is Unit.FAHRENHEIT and to_unit is Unit.CELSIUS:
|
|
96
|
+
return (float(value) - 32.0) * 5.0 / 9.0
|
|
97
|
+
if from_unit is Unit.CELSIUS and to_unit is Unit.FAHRENHEIT:
|
|
98
|
+
return float(value) * 9.0 / 5.0 + 32.0
|
|
99
|
+
|
|
100
|
+
from_scaled = _SCALE_TO_CANONICAL.get(from_unit)
|
|
101
|
+
to_scaled = _SCALE_TO_CANONICAL.get(to_unit)
|
|
102
|
+
from_base, from_factor = from_scaled if from_scaled else (from_unit, 1.0)
|
|
103
|
+
to_base, to_factor = to_scaled if to_scaled else (to_unit, 1.0)
|
|
104
|
+
if from_base is to_base:
|
|
105
|
+
return float(value) * from_factor / to_factor
|
|
106
|
+
|
|
107
|
+
raise ValueError(
|
|
108
|
+
f"cannot convert {from_unit.value} to {to_unit.value}: different dimensions"
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def to_canonical(value: float, unit: Unit) -> tuple[float, Unit]:
|
|
113
|
+
"""Return *value* expressed in the canonical unit for its dimension."""
|
|
114
|
+
target = canonical_unit(unit)
|
|
115
|
+
if target is unit:
|
|
116
|
+
return float(value), unit
|
|
117
|
+
return convert(value, unit, target), target
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Incremental, edge-capable orchestration of the inference chain.
|
|
2
|
+
|
|
3
|
+
This package owns orchestration only. It performs no inference of its own and
|
|
4
|
+
knows nothing about files, HTTP, dashboards, or storage, which keeps the
|
|
5
|
+
scientific layers usable without it.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from .benchmarks import (
|
|
9
|
+
BoundedStateResult,
|
|
10
|
+
PipelineBenchmark,
|
|
11
|
+
benchmark_pipeline,
|
|
12
|
+
measure_bounded_state,
|
|
13
|
+
)
|
|
14
|
+
from .pipeline import (
|
|
15
|
+
BehaviouralSensingPipeline,
|
|
16
|
+
PipelineConfig,
|
|
17
|
+
PipelineStep,
|
|
18
|
+
collect_alerts,
|
|
19
|
+
collect_changes,
|
|
20
|
+
daily_summaries,
|
|
21
|
+
local_midnight,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
__all__ = [
|
|
25
|
+
"BehaviouralSensingPipeline",
|
|
26
|
+
"BoundedStateResult",
|
|
27
|
+
"PipelineBenchmark",
|
|
28
|
+
"PipelineConfig",
|
|
29
|
+
"PipelineStep",
|
|
30
|
+
"benchmark_pipeline",
|
|
31
|
+
"collect_alerts",
|
|
32
|
+
"collect_changes",
|
|
33
|
+
"daily_summaries",
|
|
34
|
+
"local_midnight",
|
|
35
|
+
"measure_bounded_state",
|
|
36
|
+
]
|