sensor-modeling 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sensor_modeling/__init__.py +45 -0
- sensor_modeling/alerts/__init__.py +26 -0
- sensor_modeling/alerts/alert.py +532 -0
- sensor_modeling/analysis/__init__.py +43 -0
- sensor_modeling/analysis/_frame.py +19 -0
- sensor_modeling/analysis/behavioral_analysis.py +57 -0
- sensor_modeling/analysis/behavioral_metrics.py +66 -0
- sensor_modeling/analysis/comparison.py +164 -0
- sensor_modeling/analysis/dependency_network.py +408 -0
- sensor_modeling/analysis/granger_causality.py +314 -0
- sensor_modeling/analysis/pipeline.py +168 -0
- sensor_modeling/analysis/reporting.py +109 -0
- sensor_modeling/baseline/__init__.py +30 -0
- sensor_modeling/baseline/adaptive.py +520 -0
- sensor_modeling/baseline/features.py +224 -0
- sensor_modeling/change_point/__init__.py +13 -0
- sensor_modeling/change_point/_validation.py +31 -0
- sensor_modeling/change_point/adaptive_normalization.py +55 -0
- sensor_modeling/change_point/embedding_cpd.py +60 -0
- sensor_modeling/change_point/energy_efficient.py +57 -0
- sensor_modeling/change_point/genetic_optimization.py +65 -0
- sensor_modeling/cli.py +416 -0
- sensor_modeling/context/__init__.py +33 -0
- sensor_modeling/context/occupancy.py +529 -0
- sensor_modeling/data/__init__.py +5 -0
- sensor_modeling/data/loaders.py +146 -0
- sensor_modeling/data/preprocessing.py +83 -0
- sensor_modeling/data/synthetic.py +121 -0
- sensor_modeling/data/validation.py +81 -0
- sensor_modeling/evaluation/__init__.py +92 -0
- sensor_modeling/evaluation/ablation.py +303 -0
- sensor_modeling/evaluation/attribution.py +474 -0
- sensor_modeling/evaluation/detection.py +297 -0
- sensor_modeling/evaluation/metrics.py +541 -0
- sensor_modeling/evaluation/provenance.py +309 -0
- sensor_modeling/examples/__init__.py +1 -0
- sensor_modeling/examples/demos/__init__.py +1 -0
- sensor_modeling/examples/demos/ambient_pipeline_demo.py +418 -0
- sensor_modeling/examples/demos/bernoulli_ar_demo.py +356 -0
- sensor_modeling/examples/demos/cpd_ar_demo.py +25 -0
- sensor_modeling/examples/demos/cpd_benchmark.py +42 -0
- sensor_modeling/examples/demos/hmm_granger_demo.py +30 -0
- sensor_modeling/examples/demos/nhpp_pelt_demo.py +80 -0
- sensor_modeling/examples/tutorials/__init__.py +1 -0
- sensor_modeling/fusion/__init__.py +46 -0
- sensor_modeling/fusion/defaults.py +296 -0
- sensor_modeling/fusion/emissions.py +339 -0
- sensor_modeling/fusion/estimate.py +375 -0
- sensor_modeling/fusion/filter.py +323 -0
- sensor_modeling/health/__init__.py +31 -0
- sensor_modeling/health/monitor.py +590 -0
- sensor_modeling/health/status.py +74 -0
- sensor_modeling/hmm/__init__.py +15 -0
- sensor_modeling/hmm/adaptive_hmm.py +22 -0
- sensor_modeling/hmm/base.py +134 -0
- sensor_modeling/hmm/circadian_hmm.py +22 -0
- sensor_modeling/hmm/heterogeneous_hmm.py +22 -0
- sensor_modeling/hmm/hierarchical_hmm.py +35 -0
- sensor_modeling/hmm/scaled_dirichlet_hmm.py +23 -0
- sensor_modeling/interop/__init__.py +57 -0
- sensor_modeling/interop/fhir.py +418 -0
- sensor_modeling/interop/privacy.py +308 -0
- sensor_modeling/models/__init__.py +12 -0
- sensor_modeling/models/bernoulli_ar/__init__.py +6 -0
- sensor_modeling/models/bernoulli_ar/base_model.py +569 -0
- sensor_modeling/models/bernoulli_ar/multivariate_model.py +411 -0
- sensor_modeling/models/change_point_detection/__init__.py +10 -0
- sensor_modeling/models/change_point_detection/deep.py +65 -0
- sensor_modeling/models/change_point_detection/pelt.py +159 -0
- sensor_modeling/models/nhpp_pelt/__init__.py +5 -0
- sensor_modeling/models/nhpp_pelt/bspline.py +96 -0
- sensor_modeling/models/nhpp_pelt/cli.py +243 -0
- sensor_modeling/models/nhpp_pelt/diagnostics.py +234 -0
- sensor_modeling/models/nhpp_pelt/io.py +58 -0
- sensor_modeling/models/nhpp_pelt/model.py +408 -0
- sensor_modeling/models/nhpp_pelt/optimizer.py +142 -0
- sensor_modeling/models/nhpp_pelt/plotting.py +218 -0
- sensor_modeling/models/nhpp_pelt/quad.py +72 -0
- sensor_modeling/models/nhpp_pelt/regularization.py +121 -0
- sensor_modeling/models/nhpp_pelt/utils.py +174 -0
- sensor_modeling/observations/__init__.py +59 -0
- sensor_modeling/observations/adapters.py +195 -0
- sensor_modeling/observations/ingest.py +269 -0
- sensor_modeling/observations/observation.py +270 -0
- sensor_modeling/observations/registry.py +262 -0
- sensor_modeling/observations/stream.py +342 -0
- sensor_modeling/observations/types.py +107 -0
- sensor_modeling/observations/units.py +117 -0
- sensor_modeling/online/__init__.py +36 -0
- sensor_modeling/online/benchmarks.py +242 -0
- sensor_modeling/online/pipeline.py +485 -0
- sensor_modeling/simulation/__init__.py +54 -0
- sensor_modeling/simulation/faults.py +191 -0
- sensor_modeling/simulation/household.py +862 -0
- sensor_modeling/states/__init__.py +23 -0
- sensor_modeling/states/markov.py +105 -0
- sensor_modeling/states/ontology.py +238 -0
- sensor_modeling/utils/__init__.py +41 -0
- sensor_modeling/utils/data_io.py +199 -0
- sensor_modeling/utils/logging_config.py +10 -0
- sensor_modeling/utils/missing.py +188 -0
- sensor_modeling/utils/plotting.py +98 -0
- sensor_modeling/utils/validation.py +117 -0
- sensor_modeling/visualization/__init__.py +3 -0
- sensor_modeling/visualization/clinical.py +67 -0
- sensor_modeling/visualization/interactive.py +208 -0
- sensor_modeling/visualization/research.py +60 -0
- sensor_modeling/visualization/web_app.py +137 -0
- sensor_modeling-0.2.0.dist-info/METADATA +683 -0
- sensor_modeling-0.2.0.dist-info/RECORD +114 -0
- sensor_modeling-0.2.0.dist-info/WHEEL +5 -0
- sensor_modeling-0.2.0.dist-info/entry_points.txt +18 -0
- sensor_modeling-0.2.0.dist-info/licenses/LICENSE +21 -0
- sensor_modeling-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""Sensor health and reliability as part of the analytical model.
|
|
2
|
+
|
|
3
|
+
Sensor failures are treated as a first-class inference problem rather than an
|
|
4
|
+
operations concern. The output of this package is an evidence weight per
|
|
5
|
+
sensor that the fusion layer applies, which is what prevents a broken sensor
|
|
6
|
+
from being read as a resident who has stopped moving.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from .monitor import (
|
|
10
|
+
HealthConfig,
|
|
11
|
+
SensorHealthMonitor,
|
|
12
|
+
SensorHealthReport,
|
|
13
|
+
SystemHealthReport,
|
|
14
|
+
)
|
|
15
|
+
from .status import (
|
|
16
|
+
FAULTY_STATUSES,
|
|
17
|
+
STATUS_RELIABILITY,
|
|
18
|
+
SensorStatus,
|
|
19
|
+
status_reliability,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"FAULTY_STATUSES",
|
|
24
|
+
"STATUS_RELIABILITY",
|
|
25
|
+
"HealthConfig",
|
|
26
|
+
"SensorHealthMonitor",
|
|
27
|
+
"SensorHealthReport",
|
|
28
|
+
"SensorStatus",
|
|
29
|
+
"SystemHealthReport",
|
|
30
|
+
"status_reliability",
|
|
31
|
+
]
|
|
@@ -0,0 +1,590 @@
|
|
|
1
|
+
"""Online estimation of sensor health from the observation stream.
|
|
2
|
+
|
|
3
|
+
:class:`SensorHealthMonitor` watches ingested observations and maintains, per
|
|
4
|
+
sensor, an interpretable verdict about whether that sensor can currently be
|
|
5
|
+
trusted. It holds bounded state -- a few short deques per sensor -- so it runs
|
|
6
|
+
unchanged on an edge device for months.
|
|
7
|
+
|
|
8
|
+
Three design choices matter scientifically:
|
|
9
|
+
|
|
10
|
+
*Silence is only evidence of failure when the sensor promised to speak.* A
|
|
11
|
+
sensor that declares an ``expected_interval`` is expected to report on that
|
|
12
|
+
cadence, so prolonged silence is diagnostic. A purely event-driven contact
|
|
13
|
+
sensor makes no such promise, and its silence is genuinely ambiguous between
|
|
14
|
+
"broken" and "nobody opened the cupboard". For those sensors the monitor
|
|
15
|
+
declines to call a failure and leaves the status at ``UNKNOWN``.
|
|
16
|
+
|
|
17
|
+
*Verdicts are separated from behaviour.* The monitor reads values but never
|
|
18
|
+
interprets them as activity. Its output is consumed by fusion as an evidence
|
|
19
|
+
weight, which is what stops a dead sensor from being read as a quiet resident.
|
|
20
|
+
|
|
21
|
+
*Drift is reported, not corrected.* The monitor cannot distinguish sensor
|
|
22
|
+
drift from a genuine environmental change without redundant sensing, so it
|
|
23
|
+
flags the shift and leaves the judgement to the analyst.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import logging
|
|
29
|
+
import statistics
|
|
30
|
+
from collections import deque
|
|
31
|
+
from collections.abc import Iterable, Mapping
|
|
32
|
+
from dataclasses import dataclass, field, replace
|
|
33
|
+
from datetime import datetime, timedelta
|
|
34
|
+
|
|
35
|
+
from ..observations.observation import Observation
|
|
36
|
+
from ..observations.registry import SensorRegistry, SensorSpec
|
|
37
|
+
from ..observations.types import ObservationKind
|
|
38
|
+
from .status import FAULTY_STATUSES, SensorStatus, status_reliability
|
|
39
|
+
|
|
40
|
+
logger = logging.getLogger(__name__)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass(frozen=True)
|
|
44
|
+
class SensorHealthReport:
|
|
45
|
+
"""The health verdict for one sensor at one moment.
|
|
46
|
+
|
|
47
|
+
Attributes
|
|
48
|
+
----------
|
|
49
|
+
sensor_id
|
|
50
|
+
Sensor the verdict refers to.
|
|
51
|
+
status
|
|
52
|
+
Assessed operating condition.
|
|
53
|
+
reliability
|
|
54
|
+
Evidence weight in ``[0, 1]`` that fusion should apply to this
|
|
55
|
+
sensor's observations.
|
|
56
|
+
last_seen
|
|
57
|
+
Timestamp of the most recent observation, if any.
|
|
58
|
+
silence
|
|
59
|
+
How long the sensor has been quiet at the time of the report.
|
|
60
|
+
observations
|
|
61
|
+
Number of observations seen since the monitor was started or reset.
|
|
62
|
+
detail
|
|
63
|
+
Short human-readable explanation of the verdict.
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
sensor_id: str
|
|
67
|
+
status: SensorStatus
|
|
68
|
+
reliability: float
|
|
69
|
+
last_seen: datetime | None
|
|
70
|
+
silence: timedelta | None
|
|
71
|
+
observations: int
|
|
72
|
+
detail: str
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def is_faulty(self) -> bool:
|
|
76
|
+
"""Whether the sensor currently cannot be trusted as evidence."""
|
|
77
|
+
return self.status in FAULTY_STATUSES
|
|
78
|
+
|
|
79
|
+
def to_dict(self) -> dict[str, object]:
|
|
80
|
+
"""Return a serialisable form of the report."""
|
|
81
|
+
return {
|
|
82
|
+
"sensor_id": self.sensor_id,
|
|
83
|
+
"status": self.status.value,
|
|
84
|
+
"reliability": self.reliability,
|
|
85
|
+
"last_seen": self.last_seen.isoformat() if self.last_seen else None,
|
|
86
|
+
"silence_seconds": (
|
|
87
|
+
self.silence.total_seconds() if self.silence is not None else None
|
|
88
|
+
),
|
|
89
|
+
"observations": self.observations,
|
|
90
|
+
"detail": self.detail,
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@dataclass(frozen=True)
|
|
95
|
+
class SystemHealthReport:
|
|
96
|
+
"""Deployment-wide health, observable independently of behaviour."""
|
|
97
|
+
|
|
98
|
+
at: datetime
|
|
99
|
+
sensors: dict[str, SensorHealthReport]
|
|
100
|
+
|
|
101
|
+
@property
|
|
102
|
+
def faulty(self) -> list[str]:
|
|
103
|
+
"""Sensors that currently cannot be trusted, sorted by identifier."""
|
|
104
|
+
return sorted(sid for sid, r in self.sensors.items() if r.is_faulty)
|
|
105
|
+
|
|
106
|
+
@property
|
|
107
|
+
def coverage(self) -> float:
|
|
108
|
+
"""Mean reliability across registered sensors, in ``[0, 1]``.
|
|
109
|
+
|
|
110
|
+
This is a system-integrity measure. It says how much of the sensing
|
|
111
|
+
apparatus is working, and says nothing at all about the resident.
|
|
112
|
+
"""
|
|
113
|
+
if not self.sensors:
|
|
114
|
+
return 0.0
|
|
115
|
+
return sum(r.reliability for r in self.sensors.values()) / len(self.sensors)
|
|
116
|
+
|
|
117
|
+
def reliabilities(self) -> dict[str, float]:
|
|
118
|
+
"""Return the per-sensor evidence weights fusion should apply."""
|
|
119
|
+
return {sid: report.reliability for sid, report in self.sensors.items()}
|
|
120
|
+
|
|
121
|
+
def to_dict(self) -> dict[str, object]:
|
|
122
|
+
"""Return a serialisable form of the system report."""
|
|
123
|
+
return {
|
|
124
|
+
"at": self.at.isoformat(),
|
|
125
|
+
"coverage": self.coverage,
|
|
126
|
+
"faulty": self.faulty,
|
|
127
|
+
"sensors": {sid: r.to_dict() for sid, r in self.sensors.items()},
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@dataclass
|
|
132
|
+
class HealthConfig:
|
|
133
|
+
"""Thresholds governing health verdicts.
|
|
134
|
+
|
|
135
|
+
Parameters
|
|
136
|
+
----------
|
|
137
|
+
degraded_after
|
|
138
|
+
Multiple of the declared reporting interval after which a sensor is
|
|
139
|
+
called degraded.
|
|
140
|
+
dropout_after
|
|
141
|
+
Multiple after which a fault is suspected.
|
|
142
|
+
missing_after
|
|
143
|
+
Multiple after which the sensor is treated as supplying no evidence.
|
|
144
|
+
stuck_samples
|
|
145
|
+
Consecutive identical readings that indicate a stuck *sampled*
|
|
146
|
+
sensor. Not applied to state sensors: reporting an unchanged level is
|
|
147
|
+
what a state sensor is for, and a bed that reads occupied for eight
|
|
148
|
+
hours is a sleeping resident, not a fault.
|
|
149
|
+
stuck_state_hours
|
|
150
|
+
How long a *state* sensor may report one unchanged level before it is
|
|
151
|
+
suspected of being stuck. Set well beyond any plausible real
|
|
152
|
+
duration of the states being reported.
|
|
153
|
+
quality_floor
|
|
154
|
+
Smoothed device-reported quality below which a sensor is degraded.
|
|
155
|
+
quality_smoothing
|
|
156
|
+
Weight of each new quality reading in the exponential average.
|
|
157
|
+
drift_reference
|
|
158
|
+
Readings retained as the calibration reference for a sampled sensor.
|
|
159
|
+
drift_recent
|
|
160
|
+
Readings compared against that reference.
|
|
161
|
+
drift_sigma
|
|
162
|
+
Robust standard deviations of shift that count as drift.
|
|
163
|
+
drift_floor
|
|
164
|
+
Absolute shift below which drift is never reported, in sensor units.
|
|
165
|
+
Prevents a very stable sensor from being flagged for trivial moves.
|
|
166
|
+
out_of_range_tolerance
|
|
167
|
+
Consecutive implausible readings tolerated before flagging.
|
|
168
|
+
outage_canary_fraction
|
|
169
|
+
Fraction of the sensors that *can* be checked for silence which must
|
|
170
|
+
be failing before the silence of the sensors that cannot be checked
|
|
171
|
+
is also distrusted. See :meth:`SensorHealthMonitor.report`.
|
|
172
|
+
delivery_floor
|
|
173
|
+
Fraction of its promised reporting rate a sensor must sustain before
|
|
174
|
+
it is called degraded. Catches a sensor that keeps reporting but
|
|
175
|
+
drops a share of its records -- no single gap is long enough to look
|
|
176
|
+
like a dropout, yet a large part of the evidence never arrives.
|
|
177
|
+
delivery_window
|
|
178
|
+
Inter-arrival samples retained for the delivery-rate estimate.
|
|
179
|
+
"""
|
|
180
|
+
|
|
181
|
+
degraded_after: float = 2.0
|
|
182
|
+
dropout_after: float = 5.0
|
|
183
|
+
missing_after: float = 20.0
|
|
184
|
+
stuck_samples: int = 12
|
|
185
|
+
stuck_state_hours: float = 36.0
|
|
186
|
+
quality_floor: float = 0.5
|
|
187
|
+
quality_smoothing: float = 0.2
|
|
188
|
+
drift_reference: int = 64
|
|
189
|
+
drift_recent: int = 32
|
|
190
|
+
drift_sigma: float = 6.0
|
|
191
|
+
drift_floor: float = 0.0
|
|
192
|
+
out_of_range_tolerance: int = 2
|
|
193
|
+
outage_canary_fraction: float = 0.6
|
|
194
|
+
delivery_floor: float = 0.6
|
|
195
|
+
delivery_window: int = 32
|
|
196
|
+
|
|
197
|
+
def __post_init__(self) -> None:
|
|
198
|
+
"""Validate threshold configuration."""
|
|
199
|
+
if not 0.0 < self.degraded_after < self.dropout_after < self.missing_after:
|
|
200
|
+
raise ValueError(
|
|
201
|
+
"silence thresholds must be increasing and positive: "
|
|
202
|
+
"degraded_after < dropout_after < missing_after"
|
|
203
|
+
)
|
|
204
|
+
if self.stuck_samples < 2:
|
|
205
|
+
raise ValueError("stuck_samples must be at least 2")
|
|
206
|
+
if self.stuck_state_hours <= 0:
|
|
207
|
+
raise ValueError("stuck_state_hours must be positive")
|
|
208
|
+
if not 0.0 <= self.quality_floor <= 1.0:
|
|
209
|
+
raise ValueError("quality_floor must lie in [0, 1]")
|
|
210
|
+
if not 0.0 < self.quality_smoothing <= 1.0:
|
|
211
|
+
raise ValueError("quality_smoothing must lie in (0, 1]")
|
|
212
|
+
if self.drift_reference < 2 or self.drift_recent < 2:
|
|
213
|
+
raise ValueError("drift windows must contain at least 2 readings")
|
|
214
|
+
if self.drift_sigma <= 0:
|
|
215
|
+
raise ValueError("drift_sigma must be positive")
|
|
216
|
+
if self.drift_floor < 0:
|
|
217
|
+
raise ValueError("drift_floor must be non-negative")
|
|
218
|
+
if self.out_of_range_tolerance < 1:
|
|
219
|
+
raise ValueError("out_of_range_tolerance must be at least 1")
|
|
220
|
+
if not 0.0 < self.outage_canary_fraction <= 1.0:
|
|
221
|
+
raise ValueError("outage_canary_fraction must lie in (0, 1]")
|
|
222
|
+
if not 0.0 < self.delivery_floor <= 1.0:
|
|
223
|
+
raise ValueError("delivery_floor must lie in (0, 1]")
|
|
224
|
+
if self.delivery_window < 2:
|
|
225
|
+
raise ValueError("delivery_window must be at least 2")
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
@dataclass
|
|
229
|
+
class _SensorState:
|
|
230
|
+
"""Bounded per-sensor state maintained by the monitor."""
|
|
231
|
+
|
|
232
|
+
spec: SensorSpec
|
|
233
|
+
quality: float
|
|
234
|
+
observations: int = 0
|
|
235
|
+
last_seen: datetime | None = None
|
|
236
|
+
last_value: float | None = None
|
|
237
|
+
repeats: int = 0
|
|
238
|
+
repeat_since: datetime | None = None
|
|
239
|
+
out_of_range_run: int = 0
|
|
240
|
+
intervals: deque[float] = field(default_factory=deque)
|
|
241
|
+
reference: deque[float] = field(default_factory=deque)
|
|
242
|
+
recent: deque[float] = field(default_factory=deque)
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
class SensorHealthMonitor:
|
|
246
|
+
"""Track the operating condition of every registered sensor.
|
|
247
|
+
|
|
248
|
+
Parameters
|
|
249
|
+
----------
|
|
250
|
+
registry
|
|
251
|
+
Declared sensors. Observations from unregistered sensors are ignored,
|
|
252
|
+
because their expected behaviour is unknown.
|
|
253
|
+
config
|
|
254
|
+
Thresholds governing the verdicts.
|
|
255
|
+
"""
|
|
256
|
+
|
|
257
|
+
def __init__(
|
|
258
|
+
self, registry: SensorRegistry, config: HealthConfig | None = None
|
|
259
|
+
) -> None:
|
|
260
|
+
self.registry = registry
|
|
261
|
+
self.config = config or HealthConfig()
|
|
262
|
+
self._states: dict[str, _SensorState] = {
|
|
263
|
+
spec.sensor_id: _SensorState(
|
|
264
|
+
spec=spec,
|
|
265
|
+
quality=spec.prior_reliability,
|
|
266
|
+
reference=deque(maxlen=self.config.drift_reference),
|
|
267
|
+
recent=deque(maxlen=self.config.drift_recent),
|
|
268
|
+
intervals=deque(maxlen=self.config.delivery_window),
|
|
269
|
+
)
|
|
270
|
+
for spec in registry
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
# ------------------------------------------------------------------
|
|
274
|
+
def observe(self, observation: Observation) -> None:
|
|
275
|
+
"""Update health state from a single ingested observation."""
|
|
276
|
+
state = self._states.get(observation.sensor_id)
|
|
277
|
+
if state is None:
|
|
278
|
+
logger.debug(
|
|
279
|
+
"Ignoring health update for unregistered sensor '%s'",
|
|
280
|
+
observation.sensor_id,
|
|
281
|
+
)
|
|
282
|
+
return
|
|
283
|
+
|
|
284
|
+
state.observations += 1
|
|
285
|
+
if state.last_seen is None or observation.timestamp > state.last_seen:
|
|
286
|
+
if state.last_seen is not None:
|
|
287
|
+
gap = (observation.timestamp - state.last_seen).total_seconds()
|
|
288
|
+
if gap > 0:
|
|
289
|
+
state.intervals.append(gap)
|
|
290
|
+
state.last_seen = observation.timestamp
|
|
291
|
+
|
|
292
|
+
smoothing = self.config.quality_smoothing
|
|
293
|
+
state.quality = (
|
|
294
|
+
1.0 - smoothing
|
|
295
|
+
) * state.quality + smoothing * observation.quality
|
|
296
|
+
|
|
297
|
+
if state.last_value is not None and observation.value == state.last_value:
|
|
298
|
+
state.repeats += 1
|
|
299
|
+
else:
|
|
300
|
+
state.repeats = 1
|
|
301
|
+
state.repeat_since = observation.timestamp
|
|
302
|
+
state.last_value = observation.value
|
|
303
|
+
|
|
304
|
+
if state.spec.contains(observation.value):
|
|
305
|
+
state.out_of_range_run = 0
|
|
306
|
+
else:
|
|
307
|
+
state.out_of_range_run += 1
|
|
308
|
+
|
|
309
|
+
if observation.kind is ObservationKind.SAMPLE:
|
|
310
|
+
if len(state.reference) < state.reference.maxlen: # type: ignore[operator]
|
|
311
|
+
state.reference.append(observation.value)
|
|
312
|
+
state.recent.append(observation.value)
|
|
313
|
+
|
|
314
|
+
def observe_many(self, observations: Iterable[Observation]) -> None:
|
|
315
|
+
"""Update health state from many observations."""
|
|
316
|
+
for observation in observations:
|
|
317
|
+
self.observe(observation)
|
|
318
|
+
|
|
319
|
+
# ------------------------------------------------------------------
|
|
320
|
+
def _silence(self, state: _SensorState, now: datetime) -> timedelta | None:
|
|
321
|
+
"""Return how long a sensor has been quiet, if it has ever reported."""
|
|
322
|
+
if state.last_seen is None:
|
|
323
|
+
return None
|
|
324
|
+
return max(now - state.last_seen, timedelta(0))
|
|
325
|
+
|
|
326
|
+
def _silence_status(
|
|
327
|
+
self, state: _SensorState, silence: timedelta | None
|
|
328
|
+
) -> tuple[SensorStatus, str] | None:
|
|
329
|
+
"""Return a silence-based verdict, or ``None`` when silence is mute.
|
|
330
|
+
|
|
331
|
+
A sensor that never declared a reporting interval made no promise to
|
|
332
|
+
speak, so its silence cannot be turned into a failure claim without
|
|
333
|
+
also turning a quiet resident into a broken sensor.
|
|
334
|
+
"""
|
|
335
|
+
interval = state.spec.expected_interval
|
|
336
|
+
if interval is None or silence is None:
|
|
337
|
+
return None
|
|
338
|
+
elapsed = silence / interval
|
|
339
|
+
if elapsed >= self.config.missing_after:
|
|
340
|
+
return SensorStatus.MISSING, f"silent for {elapsed:.1f} expected intervals"
|
|
341
|
+
if elapsed >= self.config.dropout_after:
|
|
342
|
+
return SensorStatus.DROPOUT, f"silent for {elapsed:.1f} expected intervals"
|
|
343
|
+
if elapsed >= self.config.degraded_after:
|
|
344
|
+
return SensorStatus.DEGRADED, f"silent for {elapsed:.1f} expected intervals"
|
|
345
|
+
return None
|
|
346
|
+
|
|
347
|
+
def _under_delivering(self, state: _SensorState) -> tuple[SensorStatus, str] | None:
|
|
348
|
+
"""Detect a sensor that keeps reporting but drops part of its record.
|
|
349
|
+
|
|
350
|
+
Silence-based checks look at the gap since the last observation, so a
|
|
351
|
+
sensor delivering only a fraction of its promised samples slips
|
|
352
|
+
through: every individual gap is short, yet most of the evidence
|
|
353
|
+
never arrives.
|
|
354
|
+
|
|
355
|
+
This matters more than it sounds. Losing records at random costs
|
|
356
|
+
accuracy roughly in proportion; losing them *when the resident is
|
|
357
|
+
active* -- a radio contended by movement, a battery sagging under
|
|
358
|
+
load -- biases inference toward inactivity, which is the exact
|
|
359
|
+
conclusion this platform must never reach by accident. Detecting the
|
|
360
|
+
shortfall does not repair the bias, but it stops the sensor being
|
|
361
|
+
weighted as though it were healthy.
|
|
362
|
+
"""
|
|
363
|
+
interval = state.spec.expected_interval
|
|
364
|
+
if (
|
|
365
|
+
interval is None
|
|
366
|
+
or len(state.intervals) < max(state.intervals.maxlen or 2, 2) // 2
|
|
367
|
+
):
|
|
368
|
+
return None
|
|
369
|
+
typical = statistics.median(state.intervals)
|
|
370
|
+
if typical <= 0:
|
|
371
|
+
return None
|
|
372
|
+
delivered = interval.total_seconds() / typical
|
|
373
|
+
if delivered >= self.config.delivery_floor:
|
|
374
|
+
return None
|
|
375
|
+
return (
|
|
376
|
+
SensorStatus.DEGRADED,
|
|
377
|
+
f"delivering about {delivered:.0%} of its promised reporting rate",
|
|
378
|
+
)
|
|
379
|
+
|
|
380
|
+
def _drift(self, state: _SensorState) -> tuple[SensorStatus, str] | None:
|
|
381
|
+
"""Compare the recent calibration of a sampled sensor to its own past."""
|
|
382
|
+
reference, recent = state.reference, state.recent
|
|
383
|
+
if len(reference) < 2 or len(recent) < max(2, recent.maxlen or 2):
|
|
384
|
+
return None
|
|
385
|
+
reference_median = statistics.median(reference)
|
|
386
|
+
recent_median = statistics.median(recent)
|
|
387
|
+
shift = abs(recent_median - reference_median)
|
|
388
|
+
deviations = [abs(value - reference_median) for value in reference]
|
|
389
|
+
scale = 1.4826 * statistics.median(deviations)
|
|
390
|
+
threshold = max(self.config.drift_sigma * scale, self.config.drift_floor)
|
|
391
|
+
if threshold <= 0.0 or shift <= threshold:
|
|
392
|
+
return None
|
|
393
|
+
return (
|
|
394
|
+
SensorStatus.DRIFTING,
|
|
395
|
+
f"median shifted by {shift:.3g} against a reference scale of {scale:.3g}",
|
|
396
|
+
)
|
|
397
|
+
|
|
398
|
+
def _value_status(self, state: _SensorState) -> tuple[SensorStatus, str] | None:
|
|
399
|
+
"""Return a verdict based on the values themselves."""
|
|
400
|
+
if state.out_of_range_run >= self.config.out_of_range_tolerance:
|
|
401
|
+
return (
|
|
402
|
+
SensorStatus.OUT_OF_RANGE,
|
|
403
|
+
f"{state.out_of_range_run} consecutive implausible readings",
|
|
404
|
+
)
|
|
405
|
+
under = self._under_delivering(state)
|
|
406
|
+
if under is not None:
|
|
407
|
+
return under
|
|
408
|
+
# Repeated identical readings mean different things per kind. An
|
|
409
|
+
# event sensor reports the same value on every activation, and a
|
|
410
|
+
# state sensor reports an unchanged level for as long as the level is
|
|
411
|
+
# unchanged; neither is a fault. Only a sampled sensor, which is
|
|
412
|
+
# supposed to track a varying quantity, is suspicious when it stops
|
|
413
|
+
# moving.
|
|
414
|
+
if (
|
|
415
|
+
state.spec.kind is ObservationKind.SAMPLE
|
|
416
|
+
and state.repeats >= self.config.stuck_samples
|
|
417
|
+
):
|
|
418
|
+
return (
|
|
419
|
+
SensorStatus.STUCK,
|
|
420
|
+
f"{state.repeats} identical readings of {state.last_value}",
|
|
421
|
+
)
|
|
422
|
+
# A state sensor is only stuck once its level has persisted beyond
|
|
423
|
+
# any plausible real duration.
|
|
424
|
+
if (
|
|
425
|
+
state.spec.kind is ObservationKind.STATE
|
|
426
|
+
and state.repeat_since is not None
|
|
427
|
+
and state.last_seen is not None
|
|
428
|
+
):
|
|
429
|
+
held = state.last_seen - state.repeat_since
|
|
430
|
+
if held >= timedelta(hours=self.config.stuck_state_hours):
|
|
431
|
+
return (
|
|
432
|
+
SensorStatus.STUCK,
|
|
433
|
+
f"held {state.last_value} for {held.total_seconds() / 3600:.0f} h",
|
|
434
|
+
)
|
|
435
|
+
return self._drift(state)
|
|
436
|
+
|
|
437
|
+
def report_for(self, sensor_id: str, now: datetime) -> SensorHealthReport:
|
|
438
|
+
"""Return the health verdict for one sensor as of *now*."""
|
|
439
|
+
state = self._states[sensor_id]
|
|
440
|
+
silence = self._silence(state, now)
|
|
441
|
+
|
|
442
|
+
verdict = self._silence_status(state, silence) or self._value_status(state)
|
|
443
|
+
if verdict is None:
|
|
444
|
+
if state.observations == 0:
|
|
445
|
+
verdict = (SensorStatus.UNKNOWN, "no observations received")
|
|
446
|
+
elif state.quality < self.config.quality_floor:
|
|
447
|
+
verdict = (
|
|
448
|
+
SensorStatus.DEGRADED,
|
|
449
|
+
f"reported quality averaging {state.quality:.2f}",
|
|
450
|
+
)
|
|
451
|
+
else:
|
|
452
|
+
verdict = (SensorStatus.HEALTHY, "reporting as declared")
|
|
453
|
+
|
|
454
|
+
status, detail = verdict
|
|
455
|
+
reliability = (
|
|
456
|
+
state.spec.prior_reliability * state.quality * status_reliability(status)
|
|
457
|
+
)
|
|
458
|
+
return SensorHealthReport(
|
|
459
|
+
sensor_id=sensor_id,
|
|
460
|
+
status=status,
|
|
461
|
+
reliability=min(max(reliability, 0.0), 1.0),
|
|
462
|
+
last_seen=state.last_seen,
|
|
463
|
+
silence=silence,
|
|
464
|
+
observations=state.observations,
|
|
465
|
+
detail=detail,
|
|
466
|
+
)
|
|
467
|
+
|
|
468
|
+
def report(self, now: datetime) -> SystemHealthReport:
|
|
469
|
+
"""Return the deployment-wide health verdict as of *now*.
|
|
470
|
+
|
|
471
|
+
Individual verdicts are then corrected for one failure mode no single
|
|
472
|
+
sensor can detect on its own. A purely event-driven sensor makes no
|
|
473
|
+
promise to report, so its silence is normally uninformative about its
|
|
474
|
+
health -- but if every sensor that *did* promise has simultaneously
|
|
475
|
+
gone missing, the most likely explanation is that the pathway
|
|
476
|
+
carrying all of them has failed, not that the resident stopped using
|
|
477
|
+
every room at once.
|
|
478
|
+
|
|
479
|
+
The sensors with a declared cadence therefore act as canaries for the
|
|
480
|
+
whole delivery path. When enough of them fail, event sensors that
|
|
481
|
+
have been silent for the same period are downgraded to ``DROPOUT``,
|
|
482
|
+
so their silence stops being read as observed inactivity.
|
|
483
|
+
"""
|
|
484
|
+
reports = {sid: self.report_for(sid, now) for sid in self._states}
|
|
485
|
+
|
|
486
|
+
verifiable = [
|
|
487
|
+
sid
|
|
488
|
+
for sid, state in self._states.items()
|
|
489
|
+
if state.spec.expected_interval is not None
|
|
490
|
+
]
|
|
491
|
+
if not verifiable:
|
|
492
|
+
return SystemHealthReport(at=now, sensors=reports)
|
|
493
|
+
|
|
494
|
+
# Only *silence* indicates a broken delivery path. A stuck or
|
|
495
|
+
# out-of-range sensor is still delivering records, so it says nothing
|
|
496
|
+
# about whether other sensors' records are getting through.
|
|
497
|
+
failing = [
|
|
498
|
+
sid
|
|
499
|
+
for sid in verifiable
|
|
500
|
+
if reports[sid].status in (SensorStatus.DROPOUT, SensorStatus.MISSING)
|
|
501
|
+
]
|
|
502
|
+
# One dead canary is more likely a dead canary than a dead mine.
|
|
503
|
+
if len(failing) < 2 or len(failing) < self.config.outage_canary_fraction * len(
|
|
504
|
+
verifiable
|
|
505
|
+
):
|
|
506
|
+
return SystemHealthReport(at=now, sensors=reports)
|
|
507
|
+
|
|
508
|
+
# The outage can only have begun after the last canary still speaking.
|
|
509
|
+
last_heard: list[datetime] = [
|
|
510
|
+
seen for sid in failing if (seen := reports[sid].last_seen) is not None
|
|
511
|
+
]
|
|
512
|
+
if not last_heard:
|
|
513
|
+
return SystemHealthReport(at=now, sensors=reports)
|
|
514
|
+
outage_since = max(last_heard)
|
|
515
|
+
|
|
516
|
+
for sid, state in self._states.items():
|
|
517
|
+
if state.spec.expected_interval is not None:
|
|
518
|
+
continue
|
|
519
|
+
if state.last_seen is not None and state.last_seen > outage_since:
|
|
520
|
+
continue
|
|
521
|
+
reports[sid] = replace(
|
|
522
|
+
reports[sid],
|
|
523
|
+
status=SensorStatus.DROPOUT,
|
|
524
|
+
reliability=state.spec.prior_reliability
|
|
525
|
+
* state.quality
|
|
526
|
+
* status_reliability(SensorStatus.DROPOUT),
|
|
527
|
+
detail=(
|
|
528
|
+
"silent throughout a deployment-wide outage; its silence "
|
|
529
|
+
"cannot be read as an absence of activity"
|
|
530
|
+
),
|
|
531
|
+
)
|
|
532
|
+
return SystemHealthReport(at=now, sensors=reports)
|
|
533
|
+
|
|
534
|
+
def reliabilities(self, now: datetime) -> dict[str, float]:
|
|
535
|
+
"""Return per-sensor evidence weights, the fusion layer's input."""
|
|
536
|
+
return self.report(now).reliabilities()
|
|
537
|
+
|
|
538
|
+
# ------------------------------------------------------------------
|
|
539
|
+
def snapshot(self) -> dict[str, object]:
|
|
540
|
+
"""Return restartable monitor state."""
|
|
541
|
+
return {
|
|
542
|
+
sensor_id: {
|
|
543
|
+
"quality": state.quality,
|
|
544
|
+
"observations": state.observations,
|
|
545
|
+
"last_seen": (state.last_seen.isoformat() if state.last_seen else None),
|
|
546
|
+
"last_value": state.last_value,
|
|
547
|
+
"repeats": state.repeats,
|
|
548
|
+
"repeat_since": (
|
|
549
|
+
state.repeat_since.isoformat() if state.repeat_since else None
|
|
550
|
+
),
|
|
551
|
+
"out_of_range_run": state.out_of_range_run,
|
|
552
|
+
"intervals": list(state.intervals),
|
|
553
|
+
"reference": list(state.reference),
|
|
554
|
+
"recent": list(state.recent),
|
|
555
|
+
}
|
|
556
|
+
for sensor_id, state in self._states.items()
|
|
557
|
+
}
|
|
558
|
+
|
|
559
|
+
def restore(self, snapshot: Mapping[str, object]) -> None:
|
|
560
|
+
"""Restore monitor state produced by :meth:`snapshot`."""
|
|
561
|
+
for sensor_id, payload in snapshot.items():
|
|
562
|
+
state = self._states.get(sensor_id)
|
|
563
|
+
if state is None or not isinstance(payload, Mapping):
|
|
564
|
+
continue
|
|
565
|
+
last_seen = payload.get("last_seen")
|
|
566
|
+
state.quality = float(payload.get("quality", state.quality))
|
|
567
|
+
state.observations = int(payload.get("observations", 0))
|
|
568
|
+
state.last_seen = (
|
|
569
|
+
datetime.fromisoformat(str(last_seen)) if last_seen else None
|
|
570
|
+
)
|
|
571
|
+
last_value = payload.get("last_value")
|
|
572
|
+
state.last_value = None if last_value is None else float(last_value)
|
|
573
|
+
state.repeats = int(payload.get("repeats", 0))
|
|
574
|
+
repeat_since = payload.get("repeat_since")
|
|
575
|
+
state.repeat_since = (
|
|
576
|
+
datetime.fromisoformat(str(repeat_since)) if repeat_since else None
|
|
577
|
+
)
|
|
578
|
+
state.out_of_range_run = int(payload.get("out_of_range_run", 0))
|
|
579
|
+
state.intervals = deque(
|
|
580
|
+
(float(v) for v in payload.get("intervals", ()) or ()),
|
|
581
|
+
maxlen=self.config.delivery_window,
|
|
582
|
+
)
|
|
583
|
+
state.reference = deque(
|
|
584
|
+
(float(v) for v in payload.get("reference", ()) or ()),
|
|
585
|
+
maxlen=self.config.drift_reference,
|
|
586
|
+
)
|
|
587
|
+
state.recent = deque(
|
|
588
|
+
(float(v) for v in payload.get("recent", ()) or ()),
|
|
589
|
+
maxlen=self.config.drift_recent,
|
|
590
|
+
)
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Sensor health states and the reliability weights they imply.
|
|
2
|
+
|
|
3
|
+
Sensor reliability is part of the analytical model, not an operational
|
|
4
|
+
afterthought. The central rule this module exists to enforce is that a failed
|
|
5
|
+
sensor must never be read as reduced human activity: when a sensor stops
|
|
6
|
+
reporting, the correct conclusion is that evidence is missing, not that
|
|
7
|
+
nothing happened.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from enum import Enum
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class SensorStatus(str, Enum):
|
|
16
|
+
"""Assessed operating condition of a sensor."""
|
|
17
|
+
|
|
18
|
+
HEALTHY = "healthy"
|
|
19
|
+
"""Reporting as declared, with plausible values."""
|
|
20
|
+
|
|
21
|
+
DEGRADED = "degraded"
|
|
22
|
+
"""Still reporting, but with reduced measurement quality."""
|
|
23
|
+
|
|
24
|
+
MISSING = "missing"
|
|
25
|
+
"""Silent for far longer than its declared reporting interval."""
|
|
26
|
+
|
|
27
|
+
DROPOUT = "dropout"
|
|
28
|
+
"""Silent long enough to suspect a fault, but not yet written off."""
|
|
29
|
+
|
|
30
|
+
STUCK = "stuck"
|
|
31
|
+
"""Returning an unchanging value where variation is expected."""
|
|
32
|
+
|
|
33
|
+
DRIFTING = "drifting"
|
|
34
|
+
"""Calibration appears to have shifted relative to its own history."""
|
|
35
|
+
|
|
36
|
+
OUT_OF_RANGE = "out_of_range"
|
|
37
|
+
"""Reporting values outside the physically plausible declared range."""
|
|
38
|
+
|
|
39
|
+
UNKNOWN = "unknown"
|
|
40
|
+
"""Not enough evidence to judge. The honest default, not a failure."""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
#: Multiplicative reliability attached to each status.
|
|
44
|
+
#:
|
|
45
|
+
#: ``MISSING`` is deliberately zero: a sensor that is not reporting supplies no
|
|
46
|
+
#: evidence at all, and the fusion layer must fall back on other modalities
|
|
47
|
+
#: rather than treating its silence as an observation of inactivity.
|
|
48
|
+
#: ``UNKNOWN`` sits at an intermediate value because absence of a verdict is
|
|
49
|
+
#: not the same as a verdict of failure.
|
|
50
|
+
STATUS_RELIABILITY: dict[SensorStatus, float] = {
|
|
51
|
+
SensorStatus.HEALTHY: 1.0,
|
|
52
|
+
SensorStatus.DEGRADED: 0.6,
|
|
53
|
+
SensorStatus.DRIFTING: 0.5,
|
|
54
|
+
SensorStatus.UNKNOWN: 0.5,
|
|
55
|
+
SensorStatus.OUT_OF_RANGE: 0.2,
|
|
56
|
+
SensorStatus.STUCK: 0.1,
|
|
57
|
+
SensorStatus.DROPOUT: 0.05,
|
|
58
|
+
SensorStatus.MISSING: 0.0,
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
#: Statuses that indicate the sensor cannot currently be trusted as evidence.
|
|
62
|
+
FAULTY_STATUSES = frozenset(
|
|
63
|
+
{
|
|
64
|
+
SensorStatus.MISSING,
|
|
65
|
+
SensorStatus.DROPOUT,
|
|
66
|
+
SensorStatus.STUCK,
|
|
67
|
+
SensorStatus.OUT_OF_RANGE,
|
|
68
|
+
}
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def status_reliability(status: SensorStatus) -> float:
|
|
73
|
+
"""Return the multiplicative reliability weight implied by *status*."""
|
|
74
|
+
return STATUS_RELIABILITY[SensorStatus(status)]
|